mirror of
https://github.com/GoogleCloudPlatform/vertex-ai-samples.git
synced 2026-09-26 14:42:04 +00:00
* Migrate gsutil usage to gcloud storage
* update
* linter changes for 4298
* Revert "linter changes for 4298"
This reverts commit c7a00a9710.
* Linter fixes
* update
* update
* Update distributed_hyperparameter_tuning.ipynb
* removed changes in model_garden folder
---------
Co-authored-by: gurusai-voleti <gvoleti@google.com>
26 KiB
26 KiB
In [ ]:
# Copyright 2022 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# https://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.In [ ]:
! pip3 install --upgrade tensorflow==2.8 \
protobuf==3.20.3 \
google-cloud-aiplatform \
matplotlib \
pandas \
'numpy<2' -q --no-warn-conflictsIn [ ]:
import sys
if "google.colab" in sys.modules:
import IPython
app = IPython.Application.instance()
app.kernel.do_shutdown(True)In [ ]:
import sys
if "google.colab" in sys.modules:
from google.colab import auth
auth.authenticate_user()In [ ]:
PROJECT_ID = "[your-project-id]" # @param {type:"string"}
LOCATION = "us-central1" # @param {type:"string"}In [ ]:
BUCKET_URI = f"gs://your-bucket-name-{PROJECT_ID}-unique" # @param {type:"string"}In [ ]:
! gcloud storage buckets create --location $LOCATION --project $PROJECT_ID $BUCKET_URIIn [ ]:
import uuid
import matplotlib.pyplot as plt
import numpy as np
import pandas as pd
from google.cloud import aiplatform as vertex_ai
from tensorflow.python.keras import Sequential, layers
from tensorflow.python.keras.utils import data_utilsIn [ ]:
EXPERIMENT_NAME = "[your-experiment-name]" # @param {type:"string"}In [ ]:
if EXPERIMENT_NAME == "[your-experiment-name]" or EXPERIMENT_NAME is None:
EXPERIMENT_NAME = f"my-experiment-{uuid.uuid1()}"In [ ]:
vertex_ai.init(project=PROJECT_ID, location=LOCATION, staging_bucket=BUCKET_URI)In [ ]:
vertex_ai.init(experiment=EXPERIMENT_NAME)In [ ]:
# Helpers ----------------------------------------------------------------------
def read_data(uri):
"""
Read data
Args:
uri: path to data
Returns:
pandas dataframe
"""
dataset_path = data_utils.get_file("auto-mpg.data", uri)
column_names = [
"MPG",
"Cylinders",
"Displacement",
"Horsepower",
"Weight",
"Acceleration",
"Model Year",
"Origin",
]
raw_dataset = pd.read_csv(
dataset_path,
names=column_names,
na_values="?",
comment="\t",
sep=" ",
skipinitialspace=True,
)
dataset = raw_dataset.dropna()
dataset["Origin"] = dataset["Origin"].map(
lambda x: {1: "USA", 2: "Europe", 3: "Japan"}.get(x)
)
dataset = pd.get_dummies(dataset, prefix="", prefix_sep="", dtype=float)
return dataset
def train_test_split(dataset, split_frac=0.8, random_state=0):
"""
Split data into train and test
Args:
dataset: pandas dataframe
split_frac: fraction of data to use for training
random_state: random seed
Returns:
train and test dataframes
"""
train_dataset = dataset.sample(frac=split_frac, random_state=random_state)
test_dataset = dataset.drop(train_dataset.index)
train_labels = train_dataset.pop("MPG")
test_labels = test_dataset.pop("MPG")
return train_dataset, test_dataset, train_labels, test_labels
def normalize_dataset(train_dataset, test_dataset):
"""
Normalize data
Args:
train_dataset: pandas dataframe
test_dataset: pandas dataframe
Returns:
"""
train_stats = train_dataset.describe()
train_stats = train_stats.transpose()
def norm(x):
return (x - train_stats["mean"]) / train_stats["std"]
normed_train_data = norm(train_dataset)
normed_test_data = norm(test_dataset)
return normed_train_data, normed_test_data
def build_model(num_units, dropout_rate):
"""
Build model
Args:
num_units: number of units in hidden layer
dropout_rate: dropout rate
Returns:
compiled model
"""
model = Sequential(
[
layers.Dense(
num_units,
activation="relu",
input_shape=[9],
),
layers.Dropout(rate=dropout_rate),
layers.Dense(num_units, activation="relu"),
layers.Dense(1),
]
)
model.compile(loss="mse", optimizer="adam", metrics=["mae", "mse"])
return model
def train(
model,
train_data,
train_labels,
validation_split=0.2,
epochs=10,
):
"""
Train model
Args:
train_data: pandas dataframe
train_labels: pandas dataframe
model: compiled model
validation_split: fraction of data to use for validation
epochs: number of epochs to train for
Returns:
history
"""
history = model.fit(
train_data, train_labels, epochs=epochs, validation_split=validation_split
)
return historyIn [ ]:
# Define experiment parameters
parameters = [
{"num_units": 16, "dropout_rate": 0.1, "epochs": 3},
{"num_units": 16, "dropout_rate": 0.1, "epochs": 10},
{"num_units": 16, "dropout_rate": 0.2, "epochs": 10},
{"num_units": 32, "dropout_rate": 0.1, "epochs": 10},
{"num_units": 32, "dropout_rate": 0.2, "epochs": 10},
]
# Read data
dataset = read_data(
"http://archive.ics.uci.edu/ml/machine-learning-databases/auto-mpg/auto-mpg.data"
)
# Split data
train_dataset, test_dataset, train_labels, test_labels = train_test_split(dataset)
# Normalize data
normed_train_data, normed_test_data = normalize_dataset(train_dataset, test_dataset)
# Run experiments
for i, params in enumerate(parameters):
# Initialize Vertex AI Experiment run
vertex_ai.start_run(run=f"auto-mpg-local-run-{i}")
# Log training parameters
vertex_ai.log_params(params)
# Build model
model = build_model(
num_units=params["num_units"], dropout_rate=params["dropout_rate"]
)
# Train model
history = train(
model,
normed_train_data,
train_labels,
epochs=params["epochs"],
)
# Log additional parameters
vertex_ai.log_params(history.params)
# Log metrics per epochs
for idx in range(0, history.params["epochs"]):
vertex_ai.log_time_series_metrics(
{
"train_mae": history.history["mae"][idx],
"train_mse": history.history["mse"][idx],
}
)
# Log final metrics
loss, mae, mse = model.evaluate(normed_test_data, test_labels, verbose=2)
if np.isnan(loss):
loss = 0
if np.isnan(mae):
mae = 0
if np.isnan(mse):
mse = 0
vertex_ai.log_metrics({"eval_loss": loss, "eval_mae": mae, "eval_mse": mse})
vertex_ai.end_run()In [ ]:
experiment_df = vertex_ai.get_experiment_df()
experiment_df.TIn [ ]:
plt.rcParams["figure.figsize"] = [15, 5]
ax = pd.plotting.parallel_coordinates(
experiment_df.reset_index(level=0),
"run_name",
cols=[
"param.num_units",
"param.dropout_rate",
"param.epochs",
"metric.eval_loss",
"metric.eval_mse",
"metric.eval_mae",
],
color=["blue", "green", "pink", "red"],
)
ax.set_yscale("symlog")
ax.legend(bbox_to_anchor=(1.0, 0.5))In [ ]:
print("Vertex AI Experiments:")
print(
f"https://console.cloud.google.com/ai/platform/experiments/experiments?folder=&organizationId=&project={PROJECT_ID}"
)In [ ]:
# Delete experiment
exp = vertex_ai.Experiment(EXPERIMENT_NAME)
backing_tensorboard = exp.get_backing_tensorboard_resource()
exp.delete(delete_backing_tensorboard_runs=True)
# Delete Tensorboard
delete_tensorboard = False # Set True for deletion
if delete_tensorboard:
backing_tensorboard.delete()
# Delete Cloud Storage objects that were created
delete_bucket = False # Set True for deletion
if delete_bucket:
! gcloud storage rm --recursive --continue-on-error {BUCKET_URI}