From 1ff7fc70edbfd4ffc775f3ce44f02a1864768032 Mon Sep 17 00:00:00 2001 From: Ivan Cheung Date: Wed, 23 Nov 2022 11:16:42 -0500 Subject: [PATCH] custom-tabular-bq-managed-dataset.ipynb: Remove preprocessing (#1278) - [x] Removed pre-processing (normalization) This simplifies the code but reduces accuracy. --- .../custom-tabular-bq-managed-dataset.ipynb | 529 +++++------------- 1 file changed, 126 insertions(+), 403 deletions(-) diff --git a/notebooks/official/custom/custom-tabular-bq-managed-dataset.ipynb b/notebooks/official/custom/custom-tabular-bq-managed-dataset.ipynb index 7e05ac12f..3fb088626 100644 --- a/notebooks/official/custom/custom-tabular-bq-managed-dataset.ipynb +++ b/notebooks/official/custom/custom-tabular-bq-managed-dataset.ipynb @@ -346,26 +346,6 @@ "! gsutil mb -l $REGION -p $PROJECT_ID $BUCKET_URI" ] }, - { - "cell_type": "markdown", - "metadata": { - "id": "import_aip" - }, - "source": [ - "### Import libraries" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "id": "cNEiwLd0lugu" - }, - "outputs": [], - "source": [ - "from google.cloud import aiplatform, bigquery" - ] - }, { "cell_type": "markdown", "metadata": { @@ -385,6 +365,8 @@ }, "outputs": [], "source": [ + "from google.cloud import aiplatform\n", + "\n", "# Initialize the Vertex AI SDK\n", "aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=BUCKET_URI)" ] @@ -397,7 +379,9 @@ "source": [ "### Initialize BigQuery Client\n", "\n", - "Initialize the BigQuery Python client for your project." + "Initialize the BigQuery Python client for your project.\n", + "\n", + "To use BigQuery, make sure your account has the \"BigQuery User\" role." ] }, { @@ -408,132 +392,129 @@ }, "outputs": [], "source": [ + "from google.cloud import bigquery\n", + "\n", "# Set up BigQuery client\n", - "bqclient = bigquery.Client(project=PROJECT_ID)" + "bq_client = bigquery.Client(project=PROJECT_ID)" ] }, { "cell_type": "markdown", "metadata": { - "id": "container:training,prediction" + "id": "0a2c41bc91a6" }, "source": [ - "### Set pre-built containers\n", + "## Preprocess data and split data\n", + "First you should download and preprocess your data for training and testing.\n", "\n", - "Vertex AI provides pre-built containers to run training and prediction.\n", - "\n", - "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers) and [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + "- Convert categorical features to numeric\n", + "- Remove unused columns\n", + "- Remove unusable rows\n", + "- Split train and test data" ] }, { "cell_type": "code", "execution_count": null, "metadata": { - "id": "1u1mr18jlugv" + "id": "b5dfe6890137" }, "outputs": [], "source": [ - "TRAIN_VERSION = \"tf-cpu.2-8\"\n", - "DEPLOY_VERSION = \"tf2-cpu.2-8\"\n", - "\n", - "TRAIN_IMAGE = \"us-docker.pkg.dev/vertex-ai/training/{}:latest\".format(TRAIN_VERSION)\n", - "DEPLOY_IMAGE = \"us-docker.pkg.dev/vertex-ai/prediction/{}:latest\".format(DEPLOY_VERSION)" - ] - }, - { - "cell_type": "markdown", - "metadata": { - "id": "59f24e7d2269" - }, - "source": [ - "## Prepare the data\n", - "\n", - "To improve the convergence of the custom deep learning model, normalize the data. To prepare for this, calculate the mean and standard deviation for each numeric column.\n", - "\n", - "Pass these summary statistics to the training script to normalize the data before training. Later, during prediction, use these summary statistics again to normalize the testing data." + "# Define the BigQuery source dataset\n", + "BQ_SOURCE = \"bigquery-public-data.ml_datasets.penguins\"" ] }, { "cell_type": "code", "execution_count": null, "metadata": { - "id": "8e52b7832cd3" + "id": "e3a2449cfcf1" }, "outputs": [], "source": [ - "import json\n", - "\n", "import numpy as np\n", + "import pandas as pd\n", "\n", - "# Calculate mean and std across all rows\n", + "LABEL_COLUMN = \"species\"\n", "\n", "# Define the BigQuery source dataset\n", - "BQ_SOURCE = \"bq://bigquery-public-data.ml_datasets.penguins\"\n", + "BQ_SOURCE = \"bigquery-public-data.ml_datasets.penguins\"\n", "\n", "# Define NA values\n", "NA_VALUES = [\"NA\", \".\"]\n", "\n", "# Download a table\n", + "table = bq_client.get_table(BQ_SOURCE)\n", + "df = bq_client.list_rows(table).to_dataframe()\n", "\n", + "# Drop unusable rows\n", + "df = df.replace(to_replace=NA_VALUES, value=np.NaN).dropna()\n", "\n", - "def download_table(bq_table_uri: str):\n", - " # Remove bq:// prefix if present\n", - " prefix = \"bq://\"\n", - " if bq_table_uri.startswith(prefix):\n", - " bq_table_uri = bq_table_uri[len(prefix) :]\n", + "df_numeric = df.select_dtypes(include=\"number\").astype(\"float32\")\n", + "df_numeric = (df_numeric - df_numeric.mean()) / df_numeric.std()\n", + "df[df_numeric.columns] = df_numeric\n", "\n", - " table = bigquery.TableReference.from_string(bq_table_uri)\n", - " rows = bqclient.list_rows(\n", - " table,\n", - " )\n", - " return rows.to_dataframe()\n", + "# Convert categorical columns to numeric\n", + "df[\"island\"], _ = pd.factorize(df[\"island\"])\n", + "df[\"species\"], _ = pd.factorize(df[\"species\"])\n", + "df[\"sex\"], _ = pd.factorize(df[\"sex\"])\n", "\n", - "\n", - "# Remove NA values\n", - "def clean_dataframe(df):\n", - " return df.replace(to_replace=NA_VALUES, value=np.NaN).dropna()\n", - "\n", - "\n", - "def calculate_mean_and_std(df):\n", - " # Calculate mean and std for each applicable column\n", - " mean_and_std = {}\n", - " dtypes = list(zip(df.dtypes.index, map(str, df.dtypes)))\n", - " # Normalize numeric columns.\n", - " for column, dtype in dtypes:\n", - " if dtype == \"float32\" or dtype == \"float64\":\n", - " mean_and_std[column] = {\n", - " \"mean\": df[column].mean(),\n", - " \"std\": df[column].std(),\n", - " }\n", - "\n", - " return mean_and_std\n", - "\n", - "\n", - "dataframe = download_table(BQ_SOURCE)\n", - "dataframe = clean_dataframe(dataframe)\n", - "mean_and_std = calculate_mean_and_std(dataframe)\n", - "\n", - "print(\"The mean and stds for each column are: \" + str(mean_and_std))\n", - "\n", - "# Write to a file\n", - "MEAN_AND_STD_JSON_FILE = \"mean_and_std.json\"\n", - "\n", - "with open(MEAN_AND_STD_JSON_FILE, \"w\") as outfile:\n", - " json.dump(mean_and_std, outfile)\n", - "\n", - "# Save to the staging bucket\n", - "! gsutil cp {MEAN_AND_STD_JSON_FILE} {BUCKET_URI}" + "# Split into a training and holdout dataset\n", + "df_train = df.sample(frac=0.8, random_state=100)\n", + "df_holdout = df[~df.index.isin(df_train.index)]" ] }, { "cell_type": "markdown", "metadata": { - "id": "5c7732822757" + "id": "4e6a4fd28bab" + }, + "source": [ + "### Write the training dataset to BigQuery\n", + "Use the BigQuery SDK to create a dataset and write your training dataframe to it." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8b41bbd2380a" + }, + "outputs": [], + "source": [ + "# Write training dataset to BigQuery\n", + "\n", + "# Create BigQuery dataset\n", + "dataset_id = \"dataset_id_unique\"\n", + "bq_dataset = bigquery.Dataset(f\"{PROJECT_ID}.{dataset_id}\")\n", + "bq_dataset = bq_client.create_dataset(bq_dataset, exists_ok=True)\n", + "\n", + "# Reference: https://cloud.google.com/bigquery/docs/samples/bigquery-load-table-dataframe\n", + "table_id = \"table_id_unique\"\n", + "job = bq_client.load_table_from_dataframe(\n", + " dataframe=df_train,\n", + " destination=f\"{PROJECT_ID}.{dataset_id}.{table_id}\",\n", + ")\n", + "\n", + "job.result()\n", + "\n", + "BQ_TRAIN_URI = str(job.destination)\n", + "\n", + "BQ_TRAIN_URI" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "a39df4692a70" }, "source": [ "## Create a Vertex AI Tabular Dataset from BigQuery dataset\n", "\n", - "Your first step in training the model is to create a Vertex AI tabular dataset resource." + "Create a Vertex AI tabular dataset resource from your BigQuery training data.\n", + "\n", + "See more info here: https://cloud.google.com/vertex-ai/docs/training/using-managed-datasets" ] }, { @@ -545,7 +526,7 @@ "outputs": [], "source": [ "dataset = aiplatform.TabularDataset.create(\n", - " display_name=\"sample-penguins\", bq_source=BQ_SOURCE\n", + " display_name=\"sample-penguins\", bq_source=f\"bq://{BQ_TRAIN_URI}\"\n", ")" ] }, @@ -574,13 +555,9 @@ "\n", "Prepare the command-line arguments to pass to your training script.\n", "- `args`: The command line arguments to pass to the corresponding Python module. In this example, they are:\n", - " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", - " - `\"--batch_size=\" + BATCH_SIZE`: The number of batch size for training.\n", - " - `\"--distribute=\" + TRAIN_STRATEGY` : The training distribution strategy to use for single or distributed training.\n", - " - `\"single\"`: single device.\n", - " - `\"mirror\"`: all GPU devices on a single compute instance.\n", - " - `\"multi\"`: all GPU devices on all compute instances.\n", - " - `\"--mean_and_std_json_file=\" + FILE_PATH`: The file on Cloud Storage with pre-calculated means and standard deviations." + " - `label_column`: The label column in your data to predict.\n", + " - `epochs`: The number of epochs for training.\n", + " - `batch_size`: The number of batch size for training." ] }, { @@ -593,16 +570,13 @@ "source": [ "JOB_NAME = \"custom_job_unique\"\n", "\n", - "TRAIN_STRATEGY = \"single\"\n", - "\n", "EPOCHS = 20\n", "BATCH_SIZE = 10\n", "\n", "CMDARGS = [\n", + " \"--label_column=\" + LABEL_COLUMN,\n", " \"--epochs=\" + str(EPOCHS),\n", " \"--batch_size=\" + str(BATCH_SIZE),\n", - " \"--distribute=\" + TRAIN_STRATEGY,\n", - " \"--mean_and_std_json_file=\" + f\"{BUCKET_URI}/{MEAN_AND_STD_JSON_FILE}\",\n", "]" ] }, @@ -623,7 +597,11 @@ "- Sets a training distribution strategy according to the argument `args.distribute`.\n", "- Trains the model (`fit()`) with epochs and batch size according to the arguments `args.epochs` and `args.batch_size`\n", "- Gets the directory where to save the model artifacts from the environment variable `AIP_MODEL_DIR`. This variable is [set by the training service](https://cloud.google.com/vertex-ai/docs/training/code-requirements#environment-variables).\n", - "- Saves the trained model to the model directory." + "- Saves the trained model to the model directory.\n", + "\n", + "> **_NOTE:_** To improve model performance, it's recommended to normalize your inputs to the model before training. See the TensorFlow tutorial at https://www.tensorflow.org/tutorials/structured_data/preprocessing_layers#numerical_columns for details.\n", + "\n", + "> **_NOTE:_** The following training code requires you to grant the training account the \"BigQuery Read Session User\" role. See \"https://cloud.google.com/vertex-ai/docs/general/access-control#service-agents for details on how to find this account." ] }, { @@ -637,7 +615,6 @@ "%%writefile task.py\n", "\n", "import argparse\n", - "import tensorflow as tf\n", "import numpy as np\n", "import os\n", "\n", @@ -654,229 +631,67 @@ "\n", "# Read args\n", "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--label_column', dest='label_column',\n", + " required=True, type=str,\n", + " help='Label column.')\n", "parser.add_argument('--epochs', dest='epochs',\n", " default=10, type=int,\n", " help='Number of epochs.')\n", "parser.add_argument('--batch_size', dest='batch_size',\n", " default=10, type=int,\n", " help='Batch size.')\n", - "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", - " help='Distributed training strategy.')\n", - "parser.add_argument('--mean_and_std_json_file', dest='mean_and_std_json_file', type=str,\n", - " help='GCS URI to the JSON file with pre-calculated column means and standard deviations.')\n", "args = parser.parse_args()\n", "\n", - "def download_blob(bucket_name, source_blob_name, destination_file_name):\n", - " \"\"\"Downloads a blob from the bucket.\"\"\"\n", - " # bucket_name = \"your-bucket-name\"\n", - " # source_blob_name = \"storage-object-name\"\n", - " # destination_file_name = \"local/path/to/file\"\n", - "\n", - " storage_client = storage.Client()\n", - "\n", - " bucket = storage_client.bucket(bucket_name)\n", - "\n", - " # Construct a client side representation of a blob.\n", - " # Note `Bucket.blob` differs from `Bucket.get_blob` as it doesn't retrieve\n", - " # any content from Cloud Storage. As we don't need additional data,\n", - " # using `Bucket.blob` is preferred here.\n", - " blob = bucket.blob(source_blob_name)\n", - " blob.download_to_filename(destination_file_name)\n", - "\n", - " print(\n", - " \"Blob {} downloaded to {}.\".format(\n", - " source_blob_name, destination_file_name\n", - " )\n", - " )\n", - "\n", - "def extract_bucket_and_prefix_from_gcs_path(gcs_path: str):\n", - " \"\"\"Given a complete GCS path, return the bucket name and prefix as a tuple.\n", - "\n", - " Example Usage:\n", - "\n", - " bucket, prefix = extract_bucket_and_prefix_from_gcs_path(\n", - " \"gs://example-bucket/path/to/folder\"\n", - " )\n", - "\n", - " # bucket = \"example-bucket\"\n", - " # prefix = \"path/to/folder\"\n", - "\n", - " Args:\n", - " gcs_path (str):\n", - " Required. A full path to a Cloud Storage folder or resource.\n", - " Can optionally include \"gs://\" prefix or end in a trailing slash \"/\".\n", - "\n", - " Returns:\n", - " Tuple[str, Optional[str]]\n", - " A (bucket, prefix) pair from provided GCS path. If a prefix is not\n", - " present, None is returned in its place.\n", - " \"\"\"\n", - " if gcs_path.startswith(\"gs://\"):\n", - " gcs_path = gcs_path[5:]\n", - " if gcs_path.endswith(\"/\"):\n", - " gcs_path = gcs_path[:-1]\n", - "\n", - " gcs_parts = gcs_path.split(\"/\", 1)\n", - " gcs_bucket = gcs_parts[0]\n", - " gcs_blob_prefix = None if len(gcs_parts) == 1 else gcs_parts[1]\n", - "\n", - " return (gcs_bucket, gcs_blob_prefix)\n", - "\n", - "# Download means and std\n", - "def download_mean_and_std(mean_and_std_json_file):\n", - " \"\"\"Download mean and std for each column\"\"\"\n", - " import json\n", - " \n", - " bucket, file_path = extract_bucket_and_prefix_from_gcs_path(mean_and_std_json_file)\n", - " download_blob(bucket_name=bucket, source_blob_name=file_path, destination_file_name=file_path)\n", - " \n", - " with open(file_path, 'r') as file:\n", - " return json.loads(file.read())\n", - " \n", - "mean_and_std = download_mean_and_std(args.mean_and_std_json_file)\n", - "\n", - "# Single Machine, single compute device\n", - "if args.distribute == 'single':\n", - " if tf.test.is_gpu_available():\n", - " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", - " else:\n", - " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", - "# Single Machine, multiple compute device\n", - "elif args.distribute == 'mirror':\n", - " strategy = tf.distribute.MirroredStrategy()\n", - "# Multiple Machine, multiple compute device\n", - "elif args.distribute == 'multi':\n", - " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", - "\n", "# Set up training variables\n", - "LABEL_COLUMN = \"species\"\n", - "UNUSED_COLUMNS = []\n", - "NA_VALUES = [\"NA\", \".\"]\n", + "LABEL_COLUMN = args.label_column\n", "\n", - "# Possible categorical values\n", - "SPECIES = ['Adelie Penguin (Pygoscelis adeliae)',\n", - " 'Chinstrap penguin (Pygoscelis antarctica)',\n", - " 'Gentoo penguin (Pygoscelis papua)']\n", - "ISLANDS = ['Dream', 'Biscoe', 'Torgersen']\n", - "SEXES = ['FEMALE', 'MALE']\n", + "# See https://cloud.google.com/vertex-ai/docs/workbench/managed/executor#explicit-project-selection for issues regarding permissions.\n", + "PROJECT_NUMBER = os.environ[\"CLOUD_ML_PROJECT_ID\"]\n", + "bq_client = bigquery.Client(project=PROJECT_NUMBER)\n", "\n", - "# Set up BigQuery clients\n", - "bqclient = bigquery.Client()\n", "\n", "# Download a table\n", "def download_table(bq_table_uri: str):\n", " # Remove bq:// prefix if present\n", " prefix = \"bq://\"\n", " if bq_table_uri.startswith(prefix):\n", - " bq_table_uri = bq_table_uri[len(prefix):]\n", - "\n", - " table = bigquery.TableReference.from_string(bq_table_uri)\n", - " rows = bqclient.list_rows(\n", - " table,\n", - " )\n", - " return rows.to_dataframe(create_bqstorage_client=False)\n", - "\n", + " bq_table_uri = bq_table_uri[len(prefix) :]\n", + " \n", + " # Download the BigQuery table as a dataframe\n", + " # This requires the \"BigQuery Read Session User\" role on the custom training service account.\n", + " table = bq_client.get_table(bq_table_uri)\n", + " return bq_client.list_rows(table).to_dataframe()\n", "\n", + "# Download dataset splits\n", "df_train = download_table(training_data_uri)\n", "df_validation = download_table(validation_data_uri)\n", "df_test = download_table(test_data_uri)\n", "\n", - "# Remove NA values\n", - "def clean_dataframe(df):\n", - " return df.replace(to_replace=NA_VALUES, value=np.NaN).dropna()\n", - "\n", - "\n", - "df_train = clean_dataframe(df_train)\n", - "df_validation = clean_dataframe(df_validation)\n", - "\n", - "_CATEGORICAL_TYPES = {\n", - " \"island\": pd.api.types.CategoricalDtype(categories=ISLANDS),\n", - " \"species\": pd.api.types.CategoricalDtype(categories=SPECIES),\n", - " \"sex\": pd.api.types.CategoricalDtype(categories=SEXES),\n", - "}\n", - "\n", - "\n", - "def standardize(df, mean_and_std):\n", - " \"\"\"Scales numerical columns using their means and standard deviation to get\n", - " z-scores: the mean of each numerical column becomes 0, and the standard\n", - " deviation becomes 1. This can help the model converge during training.\n", - "\n", - " Args:\n", - " df: Pandas df\n", - "\n", - " Returns:\n", - " Input df with the numerical columns scaled to z-scores\n", - " \"\"\"\n", - " dtypes = list(zip(df.dtypes.index, map(str, df.dtypes)))\n", - " # Normalize numeric columns.\n", - " for column, dtype in dtypes:\n", - " if dtype == \"float32\":\n", - " df[column] -= mean_and_std[column][\"mean\"]\n", - " df[column] /= mean_and_std[column][\"std\"]\n", - " return df\n", - "\n", - "def preprocess(df):\n", - " \"\"\"Converts categorical features to numeric. Removes unused columns.\n", - "\n", - " Args:\n", - " df: Pandas df with raw data\n", - "\n", - " Returns:\n", - " df with preprocessed data\n", - " \"\"\"\n", - " df = df.drop(columns=UNUSED_COLUMNS)\n", - "\n", - " # Drop rows with NaN's\n", - " df = df.dropna()\n", - "\n", - " # Convert integer valued (numeric) columns to floating point\n", - " numeric_columns = df.select_dtypes([\"int32\", \"float32\", \"float64\"]).columns\n", - " df[numeric_columns] = df[numeric_columns].astype(\"float32\")\n", - "\n", - " # Convert categorical columns to numeric\n", - " cat_columns = df.select_dtypes([\"object\"]).columns\n", - "\n", - " df[cat_columns] = df[cat_columns].apply(\n", - " lambda x: x.astype(_CATEGORICAL_TYPES[x.name])\n", - " )\n", - " df[cat_columns] = df[cat_columns].apply(lambda x: x.cat.codes)\n", - " return df\n", - "\n", - "\n", "def convert_dataframe_to_dataset(\n", - " df_train,\n", - " df_validation,\n", - " mean_and_std\n", + " df_train: pd.DataFrame,\n", + " df_validation: pd.DataFrame,\n", "):\n", - " df_train = preprocess(df_train)\n", - " df_validation = preprocess(df_validation)\n", - "\n", " df_train_x, df_train_y = df_train, df_train.pop(LABEL_COLUMN)\n", " df_validation_x, df_validation_y = df_validation, df_validation.pop(LABEL_COLUMN)\n", "\n", - " # Join train_x and eval_x to normalize on overall means and standard\n", - " # deviations. Then separate them again.\n", - " all_x = pd.concat([df_train_x, df_validation_x], keys=[\"train\", \"eval\"])\n", - " all_x = standardize(all_x, mean_and_std)\n", - " df_train_x, df_validation_x = all_x.xs(\"train\"), all_x.xs(\"eval\")\n", - "\n", " y_train = np.asarray(df_train_y).astype(\"float32\")\n", " y_validation = np.asarray(df_validation_y).astype(\"float32\")\n", "\n", " # Convert to numpy representation\n", - " x_train = np.asarray(df_train_x)\n", + " x_train = np.asarray(df_train_x) \n", " x_test = np.asarray(df_validation_x)\n", "\n", " # Convert to one-hot representation\n", - " y_train = tf.keras.utils.to_categorical(y_train, num_classes=len(SPECIES))\n", - " y_validation = tf.keras.utils.to_categorical(y_validation, num_classes=len(SPECIES))\n", + " num_species = len(df_train_y.unique())\n", + " y_train = tf.keras.utils.to_categorical(y_train, num_classes=num_species)\n", + " y_validation = tf.keras.utils.to_categorical(y_validation, num_classes=num_species)\n", "\n", " dataset_train = tf.data.Dataset.from_tensor_slices((x_train, y_train))\n", " dataset_validation = tf.data.Dataset.from_tensor_slices((x_test, y_validation))\n", " return (dataset_train, dataset_validation)\n", "\n", "# Create datasets\n", - "dataset_train, dataset_validation = convert_dataframe_to_dataset(df_train, df_validation, mean_and_std)\n", + "dataset_train, dataset_validation = convert_dataframe_to_dataset(df_train, df_validation)\n", "\n", "# Shuffle train set\n", "dataset_train = dataset_train.shuffle(len(df_train))\n", @@ -893,7 +708,7 @@ " input_dim=num_features,\n", " ),\n", " Dense(75, activation=tf.nn.relu),\n", - " Dense(50, activation=tf.nn.relu),\n", + " Dense(50, activation=tf.nn.relu), \n", " Dense(25, activation=tf.nn.relu),\n", " Dense(3, activation=tf.nn.softmax),\n", " ]\n", @@ -908,23 +723,16 @@ " return model\n", "\n", "# Create the model\n", - "with strategy.scope():\n", - " model = create_model(num_features=dataset_train._flat_shapes[0].dims[0].value)\n", + "model = create_model(num_features=dataset_train._flat_shapes[0].dims[0].value)\n", "\n", "# Set up datasets\n", - "NUM_WORKERS = strategy.num_replicas_in_sync\n", - "# Here the batch size scales up by number of workers since\n", - "# `tf.data.Dataset.batch` expects the global batch size.\n", - "GLOBAL_BATCH_SIZE = args.batch_size * NUM_WORKERS\n", - "dataset_train = dataset_train.batch(GLOBAL_BATCH_SIZE)\n", - "dataset_validation = dataset_validation.batch(GLOBAL_BATCH_SIZE)\n", + "dataset_train = dataset_train.batch(args.batch_size)\n", + "dataset_validation = dataset_validation.batch(args.batch_size)\n", "\n", "# Train the model\n", "model.fit(dataset_train, epochs=args.epochs, validation_data=dataset_validation)\n", "\n", - "tf.saved_model.save(model, os.getenv(\"AIP_MODEL_DIR\"))\n", - "\n", - "df_test.head()" + "tf.saved_model.save(model, os.getenv(\"AIP_MODEL_DIR\"))" ] }, { @@ -968,9 +776,9 @@ "job = aiplatform.CustomTrainingJob(\n", " display_name=JOB_NAME,\n", " script_path=\"task.py\",\n", - " container_uri=TRAIN_IMAGE,\n", + " container_uri=\"us-docker.pkg.dev/vertex-ai/training/tf-cpu.2-8:latest\",\n", " requirements=[\"google-cloud-bigquery>=2.20.0\", \"db-dtypes\"],\n", - " model_serving_container_image_uri=DEPLOY_IMAGE,\n", + " model_serving_container_image_uri=\"us-docker.pkg.dev/vertex-ai/prediction/tf2-cpu.2-8:latest\",\n", ")\n", "\n", "MODEL_DISPLAY_NAME = \"penguins_model_unique\"\n", @@ -1059,108 +867,23 @@ "source": [ "### Prepare test data\n", "\n", - "Prepare test data by normalizing it and converting categorical values to numeric values.\n", - "You must normalize these values in the same way that your normalized training data.\n", - "\n", - "In this example, perform testing with the same dataset that you used for training. In practice, you generally want to use a separate test dataset to verify your results." + "Prepare test data by convert it to a Python list" ] }, { "cell_type": "code", "execution_count": null, "metadata": { - "id": "e3a2449cfcf1" + "id": "67aeea91384a" }, "outputs": [], "source": [ - "import pandas as pd\n", - "from google.cloud import bigquery\n", + "df_holdout_y = df_holdout.pop(LABEL_COLUMN)\n", + "df_holdout_x = df_holdout\n", "\n", - "UNUSED_COLUMNS = []\n", - "LABEL_COLUMN = \"species\"\n", - "\n", - "# Possible categorical values\n", - "SPECIES = [\n", - " \"Adelie Penguin (Pygoscelis adeliae)\",\n", - " \"Chinstrap penguin (Pygoscelis antarctica)\",\n", - " \"Gentoo penguin (Pygoscelis papua)\",\n", - "]\n", - "ISLANDS = [\"Dream\", \"Biscoe\", \"Torgersen\"]\n", - "SEXES = [\"FEMALE\", \"MALE\"]\n", - "\n", - "_CATEGORICAL_TYPES = {\n", - " \"island\": pd.api.types.CategoricalDtype(categories=ISLANDS),\n", - " \"species\": pd.api.types.CategoricalDtype(categories=SPECIES),\n", - " \"sex\": pd.api.types.CategoricalDtype(categories=SEXES),\n", - "}\n", - "\n", - "\n", - "def standardize(df, mean_and_std):\n", - " \"\"\"Scales numerical columns using their means and standard deviation to get\n", - " z-scores: the mean of each numerical column becomes 0, and the standard\n", - " deviation becomes 1. This can help the model converge during training.\n", - "\n", - " Args:\n", - " df: Pandas df\n", - "\n", - " Returns:\n", - " Input df with the numerical columns scaled to z-scores\n", - " \"\"\"\n", - " dtypes = list(zip(df.dtypes.index, map(str, df.dtypes)))\n", - " # Normalize numeric columns.\n", - " for column, dtype in dtypes:\n", - " if dtype == \"float32\":\n", - " df[column] -= mean_and_std[column][\"mean\"]\n", - " df[column] /= mean_and_std[column][\"std\"]\n", - " return df\n", - "\n", - "\n", - "def preprocess(df, mean_and_std):\n", - " \"\"\"Converts categorical features to numeric. Removes unused columns.\n", - "\n", - " Args:\n", - " df: Pandas df with raw data\n", - "\n", - " Returns:\n", - " df with preprocessed data\n", - " \"\"\"\n", - " df = df.drop(columns=UNUSED_COLUMNS)\n", - "\n", - " # Drop rows with NaN's\n", - " df = df.dropna()\n", - "\n", - " # Convert integer valued (numeric) columns to floating point\n", - " numeric_columns = df.select_dtypes([\"int32\", \"float32\", \"float64\"]).columns\n", - " df[numeric_columns] = df[numeric_columns].astype(\"float32\")\n", - "\n", - " # Convert categorical columns to numeric\n", - " cat_columns = df.select_dtypes([\"object\"]).columns\n", - "\n", - " df[cat_columns] = df[cat_columns].apply(\n", - " lambda x: x.astype(_CATEGORICAL_TYPES[x.name])\n", - " )\n", - " df[cat_columns] = df[cat_columns].apply(lambda x: x.cat.codes)\n", - " return df\n", - "\n", - "\n", - "def convert_dataframe_to_list(df, mean_and_std):\n", - " df = preprocess(df, mean_and_std)\n", - "\n", - " df_x, df_y = df, df.pop(LABEL_COLUMN)\n", - "\n", - " # Normalize on overall means and standard deviations.\n", - " df = standardize(df, mean_and_std)\n", - "\n", - " y = np.asarray(df_y).astype(\"float32\")\n", - "\n", - " # Convert to numpy representation\n", - " x = np.asarray(df_x)\n", - "\n", - " # Convert to one-hot representation\n", - " return x.tolist(), y.tolist()\n", - "\n", - "\n", - "x_test, y_test = convert_dataframe_to_list(dataframe, mean_and_std)" + "# Convert to list representation\n", + "holdout_x = np.array(df_holdout_x).tolist()\n", + "holdout_y = np.array(df_holdout_y).astype(\"float32\").tolist()" ] }, { @@ -1193,10 +916,10 @@ }, "outputs": [], "source": [ - "predictions = endpoint.predict(instances=x_test)\n", + "predictions = endpoint.predict(instances=holdout_x)\n", "y_predicted = np.argmax(predictions.predictions, axis=1)\n", "\n", - "correct = sum(y_predicted == np.array(y_test))\n", + "correct = sum(y_predicted == np.array(holdout_y))\n", "accuracy = len(y_predicted)\n", "print(\n", " f\"Correct predictions = {correct}, Total predictions = {accuracy}, Accuracy = {correct/accuracy}\"\n",