mirror of
https://github.com/GoogleCloudPlatform/vertex-ai-samples.git
synced 2026-09-26 14:42:04 +00:00
365 KiB
365 KiB
In [ ]:
# Copyright 2021 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# https://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.In [ ]:
! pip3 install -U google-cloud-automl --userIn [ ]:
! pip3 install google-cloud-storageIn [ ]:
import os
if not os.getenv("AUTORUN"):
# Automatically restart kernel after installs
import IPython
app = IPython.Application.instance()
app.kernel.do_shutdown(True)In [ ]:
PROJECT_ID = "[your-project-id]" # @param {type:"string"}In [ ]:
if PROJECT_ID == "" or PROJECT_ID is None or PROJECT_ID == "[your-project-id]":
# Get your GCP project id from gcloud
shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null
PROJECT_ID = shell_output[0]
print("Project ID:", PROJECT_ID)In [ ]:
! gcloud config set project $PROJECT_IDIn [ ]:
REGION = "us-central1" # @param {type: "string"}In [ ]:
from datetime import datetime
TIMESTAMP = datetime.now().strftime("%Y%m%d%H%M%S")In [ ]:
import os
import sys
# If you are running this notebook in Colab, run this cell and follow the
# instructions to authenticate your Google Cloud account. This provides access
# to your Cloud Storage bucket and lets you submit training jobs and prediction
# requests.
# If on Vertex, then don't execute this code
if not os.path.exists("/opt/deeplearning/metadata/env_version"):
if "google.colab" in sys.modules:
from google.colab import auth as google_auth
google_auth.authenticate_user()
# If you are running this tutorial in a notebook locally, replace the string
# below with the path to your service account key and run this cell to
# authenticate your Google Cloud account.
else:
%env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json
# Log in to your account on Google Cloud
! gcloud auth loginIn [ ]:
BUCKET_NAME = "[your-bucket-name]" # @param {type:"string"}In [ ]:
if BUCKET_NAME == "" or BUCKET_NAME is None or BUCKET_NAME == "[your-bucket-name]":
BUCKET_NAME = PROJECT_ID + "aip-" + TIMESTAMPIn [ ]:
! gsutil mb -l $REGION gs://$BUCKET_NAMEIn [ ]:
! gsutil ls -al gs://$BUCKET_NAMEIn [ ]:
import os
import sys
import time
from google.cloud import automl
from google.protobuf.json_format import MessageToJsonIn [ ]:
# AutoM location root path for your dataset, model and endpoint resources
PARENT = "projects/" + PROJECT_ID + "/locations/" + REGIONIn [ ]:
def automl_client():
return automl.AutoMlClient()
def prediction_client():
return automl.PredictionServiceClient()
def operations_client():
return automl.AutoMlClient()._transport.operations_client
clients = {}
clients["automl"] = automl_client()
clients["prediction"] = prediction_client()
clients["operations"] = operations_client()
for client in clients.items():
print(client)In [ ]:
IMPORT_FILE = "gs://cloud-ml-data/img/openimage/csv/salads_ml_use.csv"In [ ]:
%%capture
! gsutil cp -r gs://cloud-ml-data/img/openimage/ gs://$BUCKET_NAMEIn [ ]:
! gsutil ls gs://$BUCKET_NAMEIn [ ]:
import tensorflow as tf
all_files_csv = ! gsutil cat $IMPORT_FILE
all_files_csv = [l.replace("cloud-ml-data/img", BUCKET_NAME) for l in all_files_csv]
IMPORT_FILE = "gs://" + BUCKET_NAME + "/openimage/salads_ml_use.csv"
with tf.io.gfile.GFile(IMPORT_FILE, "w") as f:
for l in all_files_csv:
f.write(l + "\n")In [ ]:
! gsutil cat $IMPORT_FILE | head -n 10In [ ]:
dataset = {
"display_name": "salads_20210301091741",
"image_object_detection_dataset_metadata": {},
}
print(
MessageToJson(
automl.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__["_pb"]
)
)In [ ]:
request = clients["automl"].create_dataset(parent=PARENT, dataset=dataset)In [ ]:
result = request.result()
print(MessageToJson(result.__dict__["_pb"]))In [ ]:
# The full unique ID for the dataset
dataset_id = result.name
# The short numeric ID for the dataset
dataset_short_id = dataset_id.split("/")[-1]
print(dataset_id)In [ ]:
input_config = {"gcs_source": {"input_uris": [IMPORT_FILE]}}
print(
MessageToJson(
automl.ImportDataRequest(name=dataset_id, input_config=input_config).__dict__[
"_pb"
]
)
)In [ ]:
request = clients["automl"].import_data(name=dataset_id, input_config=input_config)In [ ]:
result = request.result()
print(MessageToJson(result))In [ ]:
model = {
"display_name": "salads_" + TIMESTAMP,
"dataset_id": dataset_short_id,
"image_object_detection_model_metadata": {"train_budget_milli_node_hours": 20000},
}
print(
MessageToJson(automl.CreateModelRequest(parent=PARENT, model=model).__dict__["_pb"])
)In [ ]:
request = clients["automl"].create_model(parent=PARENT, model=model)In [ ]:
result = request.result()
print(MessageToJson(result.__dict__["_pb"]))In [ ]:
# The full unique ID for the training pipeline
model_id = result.name
# The short numeric ID for the training pipeline
model_short_id = model_id.split("/")[-1]
print(model_id)In [ ]:
request = clients["automl"].list_model_evaluations(parent=model_id, filter="")In [ ]:
import json
model_evaluations = [json.loads(MessageToJson(me.__dict__["_pb"])) for me in request]
# The evaluation slice
evaluation_slice = request.model_evaluation[0].name
print(json.dumps(model_evaluations, indent=2))In [ ]:
request = clients["automl"].get_model_evaluation(name=evaluation_slice)In [ ]:
print(MessageToJson(request.__dict__["_pb"]))In [ ]:
import json
import tensorflow as tf
test_items = ! gsutil cat $IMPORT_FILE | head -n 10
gcs_input_uri = "gs://" + BUCKET_NAME + "/test.csv"
with tf.io.gfile.GFile(gcs_input_uri, "w") as f:
for item in test_items:
f.write(item.split(",")[1] + "\n")
! gsutil cat $gcs_input_uriIn [ ]:
input_config = {"gcs_source": {"input_uris": [gcs_input_uri]}}
output_config = {
"gcs_destination": {"output_uri_prefix": "gs://" + f"{BUCKET_NAME}/batch_output/"}
}
print(
MessageToJson(
automl.BatchPredictRequest(
name=model_id, input_config=input_config, output_config=output_config
).__dict__["_pb"]
)
)In [ ]:
request = clients["prediction"].batch_predict(
name=model_id, input_config=input_config, output_config=output_config
)In [ ]:
result = request.result()
print(MessageToJson(result.__dict__["_pb"]))In [ ]:
destination_uri = output_config["gcs_destination"]["output_uri_prefix"][:-1]
! gsutil ls $destination_uri/*
! gsutil cat $destination_uri/prediction*/*.jsonlWarning:
Output truncated. This notebook contains too many cells to display efficiently.