mirror of
https://github.com/GoogleCloudPlatform/vertex-ai-samples.git
synced 2026-09-29 00:21:56 +00:00
Compare commits
47
Commits
@@ -1,13 +1,14 @@
|
||||
"""Common util functions for notebook."""
|
||||
|
||||
import base64
|
||||
from collections.abc import Sequence
|
||||
import datetime
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
from typing import Any, Dict, Sequence
|
||||
from typing import Any
|
||||
|
||||
from google import auth
|
||||
from google.cloud import storage
|
||||
@@ -283,7 +284,7 @@ def decode_image(
|
||||
return image
|
||||
|
||||
|
||||
def get_label_map(label_map_yaml_filepath: str) -> Dict[int, str]:
|
||||
def get_label_map(label_map_yaml_filepath: str) -> dict[int, str]:
|
||||
"""Returns class id to label mapping given a filepath to the label map.
|
||||
|
||||
Args:
|
||||
@@ -509,6 +510,17 @@ def get_quota(project_id: str, region: str, resource_id: str) -> int:
|
||||
):
|
||||
return -1
|
||||
all_regions_data = quota_data[0]["consumerQuotaLimits"][0]["quotaBuckets"]
|
||||
|
||||
# If the quota data does not have dimensions, it is global quota. However,
|
||||
# global quota may be overridden by regional quota. So we need to check the
|
||||
# global quota first.
|
||||
global_quota = -1
|
||||
if (
|
||||
all_regions_data
|
||||
and "dimensions" not in all_regions_data[0]
|
||||
and "effectiveLimit" in all_regions_data[0]
|
||||
):
|
||||
global_quota = int(all_regions_data[0]["effectiveLimit"])
|
||||
for region_data in all_regions_data:
|
||||
if (
|
||||
region_data.get("dimensions")
|
||||
@@ -518,12 +530,13 @@ def get_quota(project_id: str, region: str, resource_id: str) -> int:
|
||||
return int(region_data["effectiveLimit"])
|
||||
else:
|
||||
return 0
|
||||
return -1
|
||||
return global_quota
|
||||
|
||||
|
||||
def get_resource_id(
|
||||
accelerator_type: str,
|
||||
is_for_training: bool,
|
||||
is_spot: bool = False,
|
||||
is_restricted_image: bool = False,
|
||||
is_dynamic_workload_scheduler: bool = False,
|
||||
) -> str:
|
||||
@@ -533,6 +546,7 @@ def get_resource_id(
|
||||
accelerator_type: The accelerator type.
|
||||
is_for_training: Whether the resource is used for training. Set false for
|
||||
serving use case.
|
||||
is_spot: Whether the resource is used with Spot.
|
||||
is_restricted_image: Whether the image is hosted in `vertex-ai-restricted`.
|
||||
is_dynamic_workload_scheduler: Whether the resource is used with Dynamic
|
||||
Workload Scheduler.
|
||||
@@ -548,7 +562,9 @@ def get_resource_id(
|
||||
"NVIDIA_A100_80GB": "nvidia_a100_80gb_gpus",
|
||||
"NVIDIA_H100_80GB": "nvidia_h100_gpus",
|
||||
"NVIDIA_H100_MEGA_80GB": "nvidia_h100_mega_gpus",
|
||||
"NVIDIA_H200_141GB": "nvidia_h200_gpus",
|
||||
"NVIDIA_TESLA_T4": "nvidia_t4_gpus",
|
||||
"TPU_V6e": "tpu_v6e",
|
||||
"TPU_V5e": "tpu_v5e",
|
||||
"TPU_V3": "tpu_v3",
|
||||
}
|
||||
@@ -563,6 +579,10 @@ def get_resource_id(
|
||||
restricted_image_training_accelerator_map = {
|
||||
"NVIDIA_A100_80GB": "restricted_image_training_nvidia_a100_80gb_gpus",
|
||||
}
|
||||
spot_serving_accelerator_map = {
|
||||
key: f"custom_model_serving_preemptible_{accelerator_suffix_map[key]}"
|
||||
for key in accelerator_suffix_map
|
||||
}
|
||||
serving_accelerator_map = {
|
||||
key: f"custom_model_serving_{accelerator_suffix_map[key]}"
|
||||
for key in accelerator_suffix_map
|
||||
@@ -591,8 +611,11 @@ def get_resource_id(
|
||||
else:
|
||||
if is_dynamic_workload_scheduler:
|
||||
raise ValueError("Dynamic Workload Scheduler does not work for serving.")
|
||||
if accelerator_type in serving_accelerator_map:
|
||||
return serving_accelerator_map[accelerator_type]
|
||||
accelerator_map = (
|
||||
spot_serving_accelerator_map if is_spot else serving_accelerator_map
|
||||
)
|
||||
if accelerator_type in accelerator_map:
|
||||
return accelerator_map[accelerator_type]
|
||||
else:
|
||||
raise ValueError(
|
||||
f"Could not find accelerator type: {accelerator_type} for serving."
|
||||
@@ -605,13 +628,28 @@ def check_quota(
|
||||
accelerator_type: str,
|
||||
accelerator_count: int,
|
||||
is_for_training: bool,
|
||||
is_spot: bool = False,
|
||||
is_restricted_image: bool = False,
|
||||
is_dynamic_workload_scheduler: bool = False,
|
||||
):
|
||||
"""Checks if the project and the region has the required quota."""
|
||||
) -> None:
|
||||
"""Checks if the project and the region has the required quota.
|
||||
|
||||
Args:
|
||||
project_id: The project id.
|
||||
region: The region.
|
||||
accelerator_type: The accelerator type.
|
||||
accelerator_count: The number of accelerators to check quota for.
|
||||
is_for_training: Whether the resource is used for training. Set false for
|
||||
serving use case.
|
||||
is_spot: Whether the resource is used with Spot.
|
||||
is_restricted_image: Whether the image is hosted in `vertex-ai-restricted`.
|
||||
is_dynamic_workload_scheduler: Whether the resource is used with Dynamic
|
||||
Workload Scheduler.
|
||||
"""
|
||||
resource_id = get_resource_id(
|
||||
accelerator_type,
|
||||
is_for_training=is_for_training,
|
||||
is_spot=is_spot,
|
||||
is_restricted_image=is_restricted_image,
|
||||
is_dynamic_workload_scheduler=is_dynamic_workload_scheduler,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,746 @@
|
||||
"""Common util functions for notebook."""
|
||||
|
||||
import base64
|
||||
from collections.abc import Sequence
|
||||
import datetime
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
from google import auth
|
||||
from google.cloud import storage
|
||||
import matplotlib.pyplot as plt
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
import requests
|
||||
import tensorflow as tf
|
||||
import yaml
|
||||
|
||||
|
||||
GCS_URI_PREFIX = "gs://"
|
||||
CHECKPOINT_BUCKET = "gs://model_garden_checkpoints"
|
||||
|
||||
|
||||
def convert_numpy_array_to_byte_string_via_tf_tensor(
|
||||
np_array: np.ndarray,
|
||||
) -> str:
|
||||
"""Serializes a numpy array to tensor bytes.
|
||||
|
||||
Args:
|
||||
np_array: A numpy array.
|
||||
|
||||
Returns:
|
||||
A tensor bytes.
|
||||
"""
|
||||
tensor_array = tf.convert_to_tensor(np_array)
|
||||
tensor_byte_string = tf.io.serialize_tensor(tensor_array)
|
||||
return tensor_byte_string.numpy()
|
||||
|
||||
|
||||
def get_jpeg_bytes(local_image_path: str, new_width: int = -1) -> bytes:
|
||||
"""Returns jpeg bytes given an image path and resizes if required.
|
||||
|
||||
Args:
|
||||
local_image_path: A string of local image path.
|
||||
new_width: An integer of new image width.
|
||||
|
||||
Returns:
|
||||
A jpeg bytes.
|
||||
"""
|
||||
image = Image.open(local_image_path)
|
||||
if new_width <= 0:
|
||||
new_image = image
|
||||
else:
|
||||
width, height = image.size
|
||||
print("original input image size: ", width, " , ", height)
|
||||
new_height = int(height * new_width / width)
|
||||
print("new input image size: ", new_width, " , ", new_height)
|
||||
new_image = image.resize((new_width, new_height))
|
||||
buffered = io.BytesIO()
|
||||
new_image.save(buffered, format="JPEG")
|
||||
return buffered.getvalue()
|
||||
|
||||
|
||||
def gcs_fuse_path(path: str) -> str:
|
||||
"""Try to convert path to gcsfuse path if it starts with gs:// else do not modify it.
|
||||
|
||||
Args:
|
||||
path: A string of path.
|
||||
|
||||
Returns:
|
||||
A gcsfuse path.
|
||||
"""
|
||||
path = path.strip()
|
||||
if path.startswith("gs://"):
|
||||
return "/gcs/" + path[5:]
|
||||
return path
|
||||
|
||||
|
||||
def get_job_name_with_datetime(prefix: str) -> str:
|
||||
"""Gets a job name by adding current time to prefix.
|
||||
|
||||
Args:
|
||||
prefix: A string of job name prefix.
|
||||
|
||||
Returns:
|
||||
A job name.
|
||||
"""
|
||||
now = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
job_name = f"{prefix}-{now}".replace("_", "-")
|
||||
return job_name
|
||||
|
||||
|
||||
def create_job_name(prefix: str) -> str:
|
||||
"""Creates a job name.
|
||||
|
||||
Args:
|
||||
prefix: A string of job name prefix.
|
||||
|
||||
Returns:
|
||||
A job name.
|
||||
"""
|
||||
user = os.environ.get("USER")
|
||||
now = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
job_name = f"{prefix}-{user}-{now}".replace("_", "-")
|
||||
return job_name
|
||||
|
||||
|
||||
def save_subset_annotation(
|
||||
input_annotation_path: str, output_annotation_path: str
|
||||
):
|
||||
"""Saves a subset of COCO annotation json file with CCA 4.0 license.
|
||||
|
||||
Args:
|
||||
input_annotation_path: A string of input annotation path.
|
||||
output_annotation_path: A string of output annotation path.
|
||||
"""
|
||||
|
||||
with open(input_annotation_path) as f:
|
||||
coco_json = json.load(f)
|
||||
|
||||
img_ids = set()
|
||||
images = []
|
||||
annotations = []
|
||||
|
||||
for img in coco_json["images"]:
|
||||
if img["license"] in [4, 5]: # CCA 4.0 license.
|
||||
img_ids.add(img["id"])
|
||||
images.append(img)
|
||||
|
||||
for ann in coco_json["annotations"]:
|
||||
if ann["image_id"] in img_ids:
|
||||
annotations.append(ann)
|
||||
|
||||
new_json = {
|
||||
"info": coco_json["info"],
|
||||
"licenses": coco_json["licenses"],
|
||||
"images": images,
|
||||
"annotations": annotations,
|
||||
"categories": coco_json["categories"],
|
||||
}
|
||||
|
||||
with open(output_annotation_path, "w") as f:
|
||||
json.dump(new_json, f)
|
||||
|
||||
|
||||
def image_to_base64(image: Any, image_format: str = "JPEG") -> str:
|
||||
"""Converts an image to base64.
|
||||
|
||||
Args:
|
||||
image: A PIL.Image instance.
|
||||
image_format: A string of image format.
|
||||
|
||||
Returns:
|
||||
A base64 string.
|
||||
"""
|
||||
buffer = io.BytesIO()
|
||||
image.save(buffer, format=image_format)
|
||||
image_str = base64.b64encode(buffer.getvalue()).decode("utf-8")
|
||||
return image_str
|
||||
|
||||
|
||||
def base64_to_image(image_str: str) -> Any:
|
||||
"""Convert base64 encoded string to an image.
|
||||
|
||||
Args:
|
||||
image_str: A string of base64 encoded image.
|
||||
|
||||
Returns:
|
||||
A PIL.Image instance.
|
||||
"""
|
||||
image = Image.open(io.BytesIO(base64.b64decode(image_str)))
|
||||
return image
|
||||
|
||||
|
||||
def image_grid(imgs: Sequence[Any], rows: int = 2, cols: int = 2) -> Any:
|
||||
"""Creates an image grid.
|
||||
|
||||
Args:
|
||||
imgs: A list of PIL.Image instances.
|
||||
rows: An integer of number of rows.
|
||||
cols: An integer of number of columns.
|
||||
|
||||
Returns:
|
||||
A PIL.Image instance.
|
||||
"""
|
||||
w, h = imgs[0].size
|
||||
grid = Image.new(
|
||||
mode="RGB", size=(cols * w + 10 * cols, rows * h), color=(255, 255, 255)
|
||||
)
|
||||
for i, img in enumerate(imgs):
|
||||
grid.paste(img, box=(i % cols * w + 10 * i, i // cols * h))
|
||||
return grid
|
||||
|
||||
|
||||
def display_image(image: Any):
|
||||
"""Displays an image.
|
||||
|
||||
Args:
|
||||
image: A PIL.Image instance.
|
||||
"""
|
||||
_ = plt.figure(figsize=(20, 15))
|
||||
plt.grid(False)
|
||||
plt.imshow(image)
|
||||
|
||||
|
||||
def download_gcs_file_to_local(gcs_uri: str, local_path: str):
|
||||
"""Download a gcs file to a local path.
|
||||
|
||||
Args:
|
||||
gcs_uri: A string of file path on GCS.
|
||||
local_path: A string of local file path.
|
||||
"""
|
||||
if not gcs_uri.startswith(GCS_URI_PREFIX):
|
||||
raise ValueError(
|
||||
f"{gcs_uri} is not a GCS path starting with {GCS_URI_PREFIX}."
|
||||
)
|
||||
client = storage.Client()
|
||||
os.makedirs(os.path.dirname(local_path), exist_ok=True)
|
||||
with open(local_path, "wb") as f:
|
||||
client.download_blob_to_file(gcs_uri, f)
|
||||
|
||||
|
||||
def download_image(url: str) -> str:
|
||||
"""Downloads an image from the given URL.
|
||||
|
||||
Args:
|
||||
url: The URL of the image to download.
|
||||
|
||||
Returns:
|
||||
base64 encoded image.
|
||||
"""
|
||||
response = requests.get(url)
|
||||
return Image.open(io.BytesIO(response.content)) # pytype: disable=bad-return-type # pillow-102-upgrade
|
||||
|
||||
|
||||
def resize_image(image: Any, new_width: int = 1000) -> Any:
|
||||
"""Resizes an image to a certain width.
|
||||
|
||||
Args:
|
||||
image: The image which has to be resized.
|
||||
new_width: New width of the image.
|
||||
|
||||
Returns:
|
||||
New resized image.
|
||||
"""
|
||||
width, height = image.size
|
||||
new_height = int(height * new_width / width)
|
||||
new_img = image.resize((new_width, new_height))
|
||||
return new_img
|
||||
|
||||
|
||||
def load_img(path: str) -> Any:
|
||||
"""Reads image from path and return PIL.Image instance.
|
||||
|
||||
Args:
|
||||
path: A string of image path.
|
||||
|
||||
Returns:
|
||||
A PIL.Image instance.
|
||||
"""
|
||||
img = tf.io.read_file(path)
|
||||
img = tf.image.decode_jpeg(img, channels=3)
|
||||
return Image.fromarray(np.uint8(img)).convert("RGB")
|
||||
|
||||
|
||||
def decode_image(
|
||||
image_str_tensor: tf.string, new_height: int, new_width: int
|
||||
) -> tf.float32:
|
||||
"""Converts and resizes image bytes to image tensor.
|
||||
|
||||
Args:
|
||||
image_str_tensor: A string of image bytes.
|
||||
new_height: An integer of new image height.
|
||||
new_width: An integer of new image width.
|
||||
|
||||
Returns:
|
||||
An image tensor.
|
||||
"""
|
||||
image = tf.io.decode_image(image_str_tensor, 3, expand_animations=False)
|
||||
image = tf.image.resize(image, (new_height, new_width))
|
||||
return image
|
||||
|
||||
|
||||
def get_label_map(label_map_yaml_filepath: str) -> dict[int, str]:
|
||||
"""Returns class id to label mapping given a filepath to the label map.
|
||||
|
||||
Args:
|
||||
label_map_yaml_filepath: A string of label map yaml file path.
|
||||
|
||||
Returns:
|
||||
A dictionary of class id to label mapping.
|
||||
"""
|
||||
with tf.io.gfile.GFile(label_map_yaml_filepath, "rb") as input_file:
|
||||
label_map = yaml.safe_load(input_file.read())["label_map"]
|
||||
return label_map
|
||||
|
||||
|
||||
def get_prediction_instances(test_filepath: str, new_width: int = -1) -> Any:
|
||||
"""Generate instance from image path to pass to Vertex AI Endpoint for prediction.
|
||||
|
||||
Args:
|
||||
test_filepath: A string of test image path.
|
||||
new_width: An integer of new image width.
|
||||
|
||||
Returns:
|
||||
A list of instances.
|
||||
"""
|
||||
if new_width <= 0:
|
||||
with tf.io.gfile.GFile(test_filepath, "rb") as input_file:
|
||||
encoded_string = base64.b64encode(input_file.read()).decode("utf-8")
|
||||
else:
|
||||
img = load_img(test_filepath)
|
||||
width, height = img.size
|
||||
print("original input image size: ", width, " , ", height)
|
||||
new_height = int(height * new_width / width)
|
||||
new_img = img.resize((new_width, new_height))
|
||||
print("resized input image size: ", new_width, " , ", new_height)
|
||||
buffered = io.BytesIO()
|
||||
new_img.save(buffered, format="JPEG")
|
||||
encoded_string = base64.b64encode(buffered.getvalue()).decode("utf-8")
|
||||
|
||||
instances = [{
|
||||
"encoded_image": {"b64": encoded_string},
|
||||
}]
|
||||
return instances
|
||||
|
||||
|
||||
def vqa_predict(
|
||||
endpoint: Any,
|
||||
question_prompts: Sequence[str],
|
||||
image: Any,
|
||||
language_code: str = "en",
|
||||
new_width: int = 1000,
|
||||
use_dedicated_endpoint: bool = False,
|
||||
) -> Sequence[str]:
|
||||
"""Predicts the answer to a question about an image using an Endpoint."""
|
||||
# Resize and convert image to base64 string.
|
||||
resized_image = resize_image(image, new_width)
|
||||
resized_image_base64 = image_to_base64(resized_image)
|
||||
|
||||
instances = []
|
||||
if question_prompts:
|
||||
# Format question prompt
|
||||
question_prompt_format = "answer {} {}\n"
|
||||
for question_prompt in question_prompts:
|
||||
if question_prompt:
|
||||
instances.append({
|
||||
"prompt": question_prompt_format.format(
|
||||
language_code, question_prompt
|
||||
),
|
||||
"image": resized_image_base64,
|
||||
})
|
||||
else:
|
||||
instances.append({
|
||||
"image": resized_image_base64,
|
||||
})
|
||||
|
||||
response = endpoint.predict(
|
||||
instances=instances, use_dedicated_endpoint=use_dedicated_endpoint
|
||||
)
|
||||
return [pred.get("response") for pred in response.predictions]
|
||||
|
||||
|
||||
def caption_predict(
|
||||
endpoint: Any,
|
||||
language_code: str,
|
||||
image: Any,
|
||||
caption_prompt: bool = False,
|
||||
new_width: int = 1000,
|
||||
use_dedicated_endpoint: bool = False,
|
||||
) -> str:
|
||||
"""Predicts a caption for a given image using an Endpoint."""
|
||||
# Resize and convert image to base64 string.
|
||||
resized_image = resize_image(image, new_width)
|
||||
resized_image_base64 = image_to_base64(resized_image)
|
||||
|
||||
instance = {"image": resized_image_base64}
|
||||
|
||||
if caption_prompt:
|
||||
# Format caption prompt
|
||||
caption_prompt_format = "caption {}\n"
|
||||
instance["prompt"] = caption_prompt_format.format(language_code)
|
||||
|
||||
instances = [instance]
|
||||
response = endpoint.predict(
|
||||
instances=instances, use_dedicated_endpoint=use_dedicated_endpoint
|
||||
)
|
||||
return response.predictions[0].get("response")
|
||||
|
||||
|
||||
def ocr_predict(
|
||||
endpoint: Any,
|
||||
ocr_prompt: str,
|
||||
image: Any,
|
||||
new_width: int = 1000,
|
||||
use_dedicated_endpoint: bool = False,
|
||||
) -> str:
|
||||
"""Extracts text from a given image using an Endpoint."""
|
||||
# Resize and convert image to base64 string.
|
||||
resized_image = resize_image(image, new_width)
|
||||
resized_image_base64 = image_to_base64(resized_image)
|
||||
|
||||
instance = {"image": resized_image_base64}
|
||||
if ocr_prompt:
|
||||
instance["prompt"] = ocr_prompt
|
||||
instances = [instance]
|
||||
|
||||
response = endpoint.predict(
|
||||
instances=instances, use_dedicated_endpoint=use_dedicated_endpoint
|
||||
)
|
||||
return response.predictions[0].get("response")
|
||||
|
||||
|
||||
def detect_predict(
|
||||
endpoint: Any,
|
||||
detect_prompt: str,
|
||||
image: Any,
|
||||
new_width: int = 1000,
|
||||
use_dedicated_endpoint: bool = False,
|
||||
) -> str:
|
||||
"""Predicts the answer to a question about an image using an Endpoint."""
|
||||
# Resize and convert image to base64 string.
|
||||
resized_image = resize_image(image, new_width)
|
||||
resized_image_base64 = image_to_base64(resized_image)
|
||||
|
||||
instance = {"image": resized_image_base64}
|
||||
if detect_prompt:
|
||||
instance["prompt"] = detect_prompt
|
||||
instances = [instance]
|
||||
|
||||
response = endpoint.predict(
|
||||
instances=instances, use_dedicated_endpoint=use_dedicated_endpoint
|
||||
)
|
||||
return response.predictions[0].get("response")
|
||||
|
||||
|
||||
def copy_model_artifacts(
|
||||
model_id: str,
|
||||
model_source: str,
|
||||
model_destination: str,
|
||||
) -> None:
|
||||
"""Copies model artifacts from model_source to model_destination.
|
||||
|
||||
model_source and model_destination should be GCS path.
|
||||
|
||||
Args:
|
||||
model_id: The model id.
|
||||
model_source: The source of the model artifact.
|
||||
model_destination: The destination of the model artifact.
|
||||
"""
|
||||
if not model_source.startswith(GCS_URI_PREFIX):
|
||||
raise ValueError(
|
||||
f"{model_source} is not a GCS path starting with {GCS_URI_PREFIX}."
|
||||
)
|
||||
if not model_destination.startswith(GCS_URI_PREFIX):
|
||||
raise ValueError(
|
||||
f"{model_destination} is not a GCS path starting with {GCS_URI_PREFIX}."
|
||||
)
|
||||
model_source = f"{model_source}/{model_id}"
|
||||
model_destination = f"{model_destination}/{model_id}"
|
||||
print("Copying model artifact from ", model_source, " to ", model_destination)
|
||||
subprocess.check_output([
|
||||
"gcloud",
|
||||
"storage",
|
||||
"cp",
|
||||
"-r",
|
||||
model_source,
|
||||
model_destination,
|
||||
])
|
||||
|
||||
|
||||
def get_quota(project_id: str, region: str, resource_id: str) -> int:
|
||||
"""Returns the quota for a resource in a region.
|
||||
|
||||
Args:
|
||||
project_id: The project id.
|
||||
region: The region.
|
||||
resource_id: The resource id.
|
||||
|
||||
Returns:
|
||||
The quota for the resource in the region. Returns -1 if can not figure out
|
||||
the quota.
|
||||
|
||||
Raises:
|
||||
RuntimeError: If the command to get quota fails.
|
||||
"""
|
||||
service_endpoint = "aiplatform.googleapis.com"
|
||||
|
||||
command = (
|
||||
"gcloud alpha services quota list"
|
||||
f" --service={service_endpoint} --consumer=projects/{project_id}"
|
||||
f" --filter='{service_endpoint}/{resource_id}' --format=json"
|
||||
)
|
||||
process = subprocess.run(
|
||||
command, shell=True, capture_output=True, text=True, check=True
|
||||
)
|
||||
if process.returncode == 0:
|
||||
quota_data = json.loads(process.stdout)
|
||||
else:
|
||||
raise RuntimeError(f"Error fetching quota data: {process.stderr}")
|
||||
|
||||
if not quota_data or "consumerQuotaLimits" not in quota_data[0]:
|
||||
return -1
|
||||
if (
|
||||
not quota_data[0]["consumerQuotaLimits"]
|
||||
or "quotaBuckets" not in quota_data[0]["consumerQuotaLimits"][0]
|
||||
):
|
||||
return -1
|
||||
all_regions_data = quota_data[0]["consumerQuotaLimits"][0]["quotaBuckets"]
|
||||
|
||||
# If the quota data does not have dimensions, it is global quota. However,
|
||||
# global quota may be overridden by regional quota. So we need to check the
|
||||
# global quota first.
|
||||
global_quota = -1
|
||||
if (
|
||||
all_regions_data
|
||||
and "dimensions" not in all_regions_data[0]
|
||||
and "effectiveLimit" in all_regions_data[0]
|
||||
):
|
||||
global_quota = int(all_regions_data[0]["effectiveLimit"])
|
||||
for region_data in all_regions_data:
|
||||
if (
|
||||
region_data.get("dimensions")
|
||||
and region_data["dimensions"]["region"] == region
|
||||
):
|
||||
if "effectiveLimit" in region_data:
|
||||
return int(region_data["effectiveLimit"])
|
||||
else:
|
||||
return 0
|
||||
return global_quota
|
||||
|
||||
|
||||
def get_resource_id(
|
||||
accelerator_type: str,
|
||||
is_for_training: bool,
|
||||
is_spot: bool = False,
|
||||
is_restricted_image: bool = False,
|
||||
is_dynamic_workload_scheduler: bool = False,
|
||||
) -> str:
|
||||
"""Returns the resource id for a given accelerator type and the use case.
|
||||
|
||||
Args:
|
||||
accelerator_type: The accelerator type.
|
||||
is_for_training: Whether the resource is used for training. Set false for
|
||||
serving use case.
|
||||
is_spot: Whether the resource is used with Spot.
|
||||
is_restricted_image: Whether the image is hosted in `vertex-ai-restricted`.
|
||||
is_dynamic_workload_scheduler: Whether the resource is used with Dynamic
|
||||
Workload Scheduler.
|
||||
|
||||
Returns:
|
||||
The resource id.
|
||||
"""
|
||||
accelerator_suffix_map = {
|
||||
"NVIDIA_TESLA_V100": "nvidia_v100_gpus",
|
||||
"NVIDIA_TESLA_P100": "nvidia_p100_gpus",
|
||||
"NVIDIA_L4": "nvidia_l4_gpus",
|
||||
"NVIDIA_TESLA_A100": "nvidia_a100_gpus",
|
||||
"NVIDIA_A100_80GB": "nvidia_a100_80gb_gpus",
|
||||
"NVIDIA_H100_80GB": "nvidia_h100_gpus",
|
||||
"NVIDIA_H100_MEGA_80GB": "nvidia_h100_mega_gpus",
|
||||
"NVIDIA_H200_141GB": "nvidia_h200_gpus",
|
||||
"NVIDIA_TESLA_T4": "nvidia_t4_gpus",
|
||||
"TPU_V6e": "tpu_v6e",
|
||||
"TPU_V5e": "tpu_v5e",
|
||||
"TPU_V3": "tpu_v3",
|
||||
}
|
||||
default_training_accelerator_map = {
|
||||
key: f"custom_model_training_{accelerator_suffix_map[key]}"
|
||||
for key in accelerator_suffix_map
|
||||
}
|
||||
dws_training_accelerator_map = {
|
||||
key: f"custom_model_training_preemptible_{accelerator_suffix_map[key]}"
|
||||
for key in accelerator_suffix_map
|
||||
}
|
||||
restricted_image_training_accelerator_map = {
|
||||
"NVIDIA_A100_80GB": "restricted_image_training_nvidia_a100_80gb_gpus",
|
||||
}
|
||||
spot_serving_accelerator_map = {
|
||||
key: f"custom_model_serving_preemptible_{accelerator_suffix_map[key]}"
|
||||
for key in accelerator_suffix_map
|
||||
}
|
||||
serving_accelerator_map = {
|
||||
key: f"custom_model_serving_{accelerator_suffix_map[key]}"
|
||||
for key in accelerator_suffix_map
|
||||
}
|
||||
|
||||
if is_for_training:
|
||||
if is_restricted_image and is_dynamic_workload_scheduler:
|
||||
raise ValueError(
|
||||
"Dynamic Workload Scheduler does not work for restricted image"
|
||||
" training."
|
||||
)
|
||||
training_accelerator_map = (
|
||||
restricted_image_training_accelerator_map
|
||||
if is_restricted_image
|
||||
else default_training_accelerator_map
|
||||
)
|
||||
if accelerator_type in training_accelerator_map:
|
||||
if is_dynamic_workload_scheduler:
|
||||
return dws_training_accelerator_map[accelerator_type]
|
||||
else:
|
||||
return training_accelerator_map[accelerator_type]
|
||||
else:
|
||||
raise ValueError(
|
||||
f"Could not find accelerator type: {accelerator_type} for training."
|
||||
)
|
||||
else:
|
||||
if is_dynamic_workload_scheduler:
|
||||
raise ValueError("Dynamic Workload Scheduler does not work for serving.")
|
||||
accelerator_map = (
|
||||
spot_serving_accelerator_map if is_spot else serving_accelerator_map
|
||||
)
|
||||
if accelerator_type in accelerator_map:
|
||||
return accelerator_map[accelerator_type]
|
||||
else:
|
||||
raise ValueError(
|
||||
f"Could not find accelerator type: {accelerator_type} for serving."
|
||||
)
|
||||
|
||||
|
||||
def check_quota(
|
||||
project_id: str,
|
||||
region: str,
|
||||
accelerator_type: str,
|
||||
accelerator_count: int,
|
||||
is_for_training: bool,
|
||||
is_spot: bool = False,
|
||||
is_restricted_image: bool = False,
|
||||
is_dynamic_workload_scheduler: bool = False,
|
||||
) -> None:
|
||||
"""Checks if the project and the region has the required quota.
|
||||
|
||||
Args:
|
||||
project_id: The project id.
|
||||
region: The region.
|
||||
accelerator_type: The accelerator type.
|
||||
accelerator_count: The number of accelerators to check quota for.
|
||||
is_for_training: Whether the resource is used for training. Set false for
|
||||
serving use case.
|
||||
is_spot: Whether the resource is used with Spot.
|
||||
is_restricted_image: Whether the image is hosted in `vertex-ai-restricted`.
|
||||
is_dynamic_workload_scheduler: Whether the resource is used with Dynamic
|
||||
Workload Scheduler.
|
||||
"""
|
||||
resource_id = get_resource_id(
|
||||
accelerator_type,
|
||||
is_for_training=is_for_training,
|
||||
is_spot=is_spot,
|
||||
is_restricted_image=is_restricted_image,
|
||||
is_dynamic_workload_scheduler=is_dynamic_workload_scheduler,
|
||||
)
|
||||
quota = get_quota(project_id, region, resource_id)
|
||||
quota_request_instruction = (
|
||||
"Either use "
|
||||
"a different region or request additional quota. Follow "
|
||||
"instructions here "
|
||||
"https://cloud.google.com/docs/quotas/view-manage#requesting_higher_quota"
|
||||
" to check quota in a region or request additional quota for "
|
||||
"your project."
|
||||
)
|
||||
if quota == -1:
|
||||
raise ValueError(
|
||||
f"Quota not found for: {resource_id} in {region}."
|
||||
f" {quota_request_instruction}"
|
||||
)
|
||||
if quota < accelerator_count:
|
||||
raise ValueError(
|
||||
f"Quota not enough for {resource_id} in {region}: {quota} <"
|
||||
f" {accelerator_count}. {quota_request_instruction}"
|
||||
)
|
||||
|
||||
|
||||
def get_deploy_source() -> str:
|
||||
"""Gets deploy_source string based on running environment."""
|
||||
vertex_product = os.environ.get("VERTEX_PRODUCT", "")
|
||||
match vertex_product:
|
||||
case "COLAB_ENTERPRISE":
|
||||
return "notebook_colab_enterprise"
|
||||
case "WORKBENCH_INSTANCE":
|
||||
return "notebook_workbench"
|
||||
case _:
|
||||
# Legacy workbench, legacy colab, or other custom environments.
|
||||
return "notebook_environment_unspecified"
|
||||
|
||||
|
||||
def _is_operation_done(op_name: str, region: str) -> bool:
|
||||
"""Checks if the operation is done.
|
||||
|
||||
Args:
|
||||
op_name: The name of the operation to poll.
|
||||
region: The region of the operation.
|
||||
|
||||
Returns:
|
||||
True if the operation is done, False otherwise.
|
||||
|
||||
Raises:
|
||||
ValueError: If the operation failed.
|
||||
"""
|
||||
creds, _ = auth.default()
|
||||
auth_req = auth.transport.requests.Request()
|
||||
creds.refresh(auth_req)
|
||||
headers = {
|
||||
"Authorization": f"Bearer {creds.token}",
|
||||
}
|
||||
url = f"https://{region}-aiplatform.googleapis.com/ui/{op_name}"
|
||||
response = requests.get(url, headers=headers)
|
||||
operation_data = response.json()
|
||||
if "error" in operation_data:
|
||||
raise ValueError(f"Operation failed: {operation_data['error']}")
|
||||
return operation_data.get("done", False)
|
||||
|
||||
|
||||
def poll_and_wait(
|
||||
op_name: str, region: str, total_wait: int, interval: int = 60
|
||||
) -> None:
|
||||
"""Polls the operation and waits for it to complete.
|
||||
|
||||
Args:
|
||||
op_name: The name of the operation to poll.
|
||||
region: The region of the operation.
|
||||
total_wait: The total wait time in seconds.
|
||||
interval: The interval between each poll in seconds.
|
||||
|
||||
Raises:
|
||||
TimeoutError: If the operation times out.
|
||||
"""
|
||||
start_time = time.time()
|
||||
while True:
|
||||
if _is_operation_done(op_name, region):
|
||||
break
|
||||
time_elapsed = time.time() - start_time
|
||||
if time_elapsed > total_wait:
|
||||
raise TimeoutError(
|
||||
f"Operation timed out after {int(time_elapsed)} seconds."
|
||||
)
|
||||
print(
|
||||
"\rStill waiting for operation... Elapsed time in seconds:"
|
||||
f" {int(time_elapsed):<6}",
|
||||
end="",
|
||||
flush=True,
|
||||
)
|
||||
time.sleep(interval)
|
||||
+592
@@ -0,0 +1,592 @@
|
||||
"""Functions for dataset validation.
|
||||
|
||||
This tool is used to validate the dataset against the given template.
|
||||
"""
|
||||
|
||||
from collections.abc import Callable
|
||||
import json
|
||||
import multiprocessing
|
||||
import os
|
||||
import subprocess
|
||||
from typing import Any, Union
|
||||
from absl import logging
|
||||
import accelerate
|
||||
import datasets
|
||||
import transformers
|
||||
|
||||
GCS_URI_PREFIX = "gs://"
|
||||
GCSFUSE_URI_PREFIX = "/gcs/"
|
||||
LOCAL_BASE_MODEL_DIR = "/tmp/base_model_dir"
|
||||
LOCAL_TEMPLATE_DIR = "/tmp/template_dir"
|
||||
_TEMPLATE_DIRNAME = "templates"
|
||||
_VERTEX_AI_SAMPLES_GITHUB_REPO_NAME = "vertex-ai-samples"
|
||||
_VERTEX_AI_SAMPLES_GITHUB_TEMPLATE_DIR = (
|
||||
"community-content/vertex_model_garden/model_oss/peft/train/vmg/templates"
|
||||
)
|
||||
_MODELS_REQUIRING_PAD_TOKEN = ("llama", "falcon", "mistral", "mixtral")
|
||||
_MODELS_REQUIRING_EOS_TOEKN = ("gemma-2b", "gemma-7b")
|
||||
_DESCRIPTION_KEY = "description"
|
||||
_SOURCE_KEY = "source"
|
||||
_PROMPT_INPUT_KEY = "prompt_input"
|
||||
_PROMPT_NO_INPUT_KEY = "prompt_no_input"
|
||||
_RESPONSE_SEPARATOR = "response_separator"
|
||||
_INSTRUCTION_SEPARATOR = "instruction_separator"
|
||||
_CHAT_TEMPLATE_KEY = "chat_template"
|
||||
_KNOWN_KEYS = (
|
||||
_DESCRIPTION_KEY,
|
||||
_SOURCE_KEY,
|
||||
_PROMPT_INPUT_KEY,
|
||||
_PROMPT_NO_INPUT_KEY,
|
||||
_RESPONSE_SEPARATOR,
|
||||
_INSTRUCTION_SEPARATOR,
|
||||
_CHAT_TEMPLATE_KEY,
|
||||
)
|
||||
|
||||
|
||||
def is_gcs_path(input_path: str) -> bool:
|
||||
"""Checks if the input path is a Google Cloud Storage (GCS) path.
|
||||
|
||||
Args:
|
||||
input_path: The input path to be checked.
|
||||
|
||||
Returns:
|
||||
True if the input path is a GCS path, False otherwise.
|
||||
"""
|
||||
return input_path is not None and input_path.startswith(GCS_URI_PREFIX)
|
||||
|
||||
|
||||
def force_gcs_fuse_path(gcs_uri: str) -> str:
|
||||
"""Converts gs:// uris to their /gcs/ equivalents. No-op for other uris.
|
||||
|
||||
Args:
|
||||
gcs_uri: The GCS URI to convert.
|
||||
|
||||
Returns:
|
||||
The converted GCS URI.
|
||||
"""
|
||||
if is_gcs_path(gcs_uri):
|
||||
return GCSFUSE_URI_PREFIX + gcs_uri[len(GCS_URI_PREFIX) :]
|
||||
else:
|
||||
return gcs_uri
|
||||
|
||||
|
||||
def download_gcs_uri_to_local(
|
||||
gcs_uri: str,
|
||||
destination_dir: str = LOCAL_BASE_MODEL_DIR,
|
||||
check_path_exists: bool = True,
|
||||
) -> str:
|
||||
"""Downloads GCS URI to local.
|
||||
|
||||
If GCS URI is a directory, gs://some/folder is downloaded to
|
||||
/destination_dir/folder. If GCS URI is a file, gs://some/file is downloaded to
|
||||
/destination_dir/file.
|
||||
|
||||
Args:
|
||||
gcs_uri: GCS URI to download.
|
||||
destination_dir: Local directory directory.
|
||||
check_path_exists: Whether to check if the path exists.
|
||||
|
||||
Returns:
|
||||
Local path to target folder/file.
|
||||
"""
|
||||
target = os.path.join(
|
||||
destination_dir,
|
||||
os.path.basename(os.path.normpath(gcs_uri)),
|
||||
)
|
||||
if check_path_exists and os.path.exists(target):
|
||||
logging.info("File %s already exists.", target)
|
||||
return target
|
||||
if accelerate.PartialState().is_local_main_process:
|
||||
logging.info(
|
||||
"Downloading file(s) from %s to %s...", gcs_uri, destination_dir
|
||||
)
|
||||
if not os.path.exists(destination_dir):
|
||||
os.mkdir(destination_dir)
|
||||
subprocess.check_output([
|
||||
"gsutil",
|
||||
"-m",
|
||||
"cp",
|
||||
"-r",
|
||||
gcs_uri,
|
||||
destination_dir,
|
||||
])
|
||||
logging.info("Downloaded file(s) from %s to %s.", gcs_uri, destination_dir)
|
||||
# Make sure ALL processes process to next step after data downloading is done.
|
||||
# It matters for the main process to wait for other processes as well.
|
||||
accelerate.PartialState().wait_for_everyone()
|
||||
return target
|
||||
|
||||
|
||||
def get_template(template_path: str) -> dict[str, str]:
|
||||
"""Gets the template dictionary given the file path.
|
||||
|
||||
Args:
|
||||
template_path: Path to the template file.
|
||||
|
||||
Returns:
|
||||
A dictionary of the template.
|
||||
|
||||
Raises:
|
||||
ValueError: If the template file does not exist or contains unknown keys.
|
||||
"""
|
||||
if is_gcs_path(template_path):
|
||||
template_path = force_gcs_fuse_path(template_path)
|
||||
elif not os.path.isfile(template_path):
|
||||
template_path = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
_TEMPLATE_DIRNAME,
|
||||
template_path + ".json",
|
||||
)
|
||||
if not os.path.isfile(template_path):
|
||||
raise ValueError(f"Template file {template_path} does not exist.")
|
||||
with open(template_path, "r") as f:
|
||||
template_json: dict[str, str] = json.load(f)
|
||||
for key in template_json:
|
||||
if key not in _KNOWN_KEYS:
|
||||
raise ValueError(f"Unknown key {key} in template {template_path}.")
|
||||
return template_json
|
||||
|
||||
|
||||
def get_response_separator(template_json: dict[str, str]) -> Union[str, None]:
|
||||
return template_json.get(_RESPONSE_SEPARATOR, None)
|
||||
|
||||
|
||||
def get_instruction_separator(
|
||||
template_json: dict[str, str],
|
||||
) -> Union[str, None]:
|
||||
return template_json.get(_INSTRUCTION_SEPARATOR, None)
|
||||
|
||||
|
||||
def _format_template_fn(
|
||||
template: str,
|
||||
input_column: str,
|
||||
tokenizer: transformers.PreTrainedTokenizer | None = None,
|
||||
) -> Callable[[dict[str, str]], dict[str, str]]:
|
||||
"""Formats a dataset example according to a template.
|
||||
|
||||
Args:
|
||||
template: Name of the JSON template file under `templates/` or GCS path to
|
||||
the template file.
|
||||
input_column: The input column in the dataset to be used or updated by the
|
||||
template. If it does not exist, the template's `prompt_no_input` will be
|
||||
used, and the input_column will be created.
|
||||
tokenizer: The tokenizer to use for chat_template templates.
|
||||
|
||||
Returns:
|
||||
A function that formats data according to the template.
|
||||
"""
|
||||
template_json = get_template(template)
|
||||
|
||||
if _CHAT_TEMPLATE_KEY not in template_json:
|
||||
|
||||
def format_fn(example: dict[str, str]) -> dict[str, str]:
|
||||
format_dict = {key: value for key, value in example.items()}
|
||||
if format_dict.get(input_column):
|
||||
format_str = template_json[_PROMPT_INPUT_KEY]
|
||||
elif _PROMPT_NO_INPUT_KEY in template_json:
|
||||
format_str = template_json[_PROMPT_NO_INPUT_KEY]
|
||||
else:
|
||||
raise KeyError(
|
||||
f"The template {os.path.basename(template)} does not contain"
|
||||
f" {_PROMPT_INPUT_KEY} or {_PROMPT_NO_INPUT_KEY} key."
|
||||
)
|
||||
try:
|
||||
return {input_column: format_str.format(**format_dict)}
|
||||
except KeyError as e:
|
||||
raise KeyError(
|
||||
f"The template {os.path.basename(template)} contains a key {e} in"
|
||||
f" {_PROMPT_INPUT_KEY} or {_PROMPT_NO_INPUT_KEY} that does not"
|
||||
" exist in the dataset example. The dataset example looks like"
|
||||
f" {format_dict}."
|
||||
) from e
|
||||
|
||||
return format_fn
|
||||
elif (
|
||||
_PROMPT_INPUT_KEY in template_json
|
||||
or _PROMPT_NO_INPUT_KEY in template_json
|
||||
):
|
||||
raise ValueError(
|
||||
f"chat_template templates do not support {_PROMPT_INPUT_KEY} or"
|
||||
f" {_PROMPT_NO_INPUT_KEY} templates."
|
||||
)
|
||||
else:
|
||||
if tokenizer is None:
|
||||
raise ValueError("A tokenizer is required for chat_template templates.")
|
||||
# Assign HuggingFace jinja template.
|
||||
tokenizer.chat_template = template_json[_CHAT_TEMPLATE_KEY]
|
||||
|
||||
def format_fn(example: dict[str, str]) -> dict[str, str]:
|
||||
try:
|
||||
return {
|
||||
input_column: tokenizer.apply_chat_template(
|
||||
example[input_column],
|
||||
tokenize=False,
|
||||
add_generation_prompt=False,
|
||||
)
|
||||
}
|
||||
except KeyError as e:
|
||||
raise KeyError(
|
||||
f"The template {os.path.basename(template)} contains a key {e} in"
|
||||
f" {_CHAT_TEMPLATE_KEY} that does not exist in the dataset example."
|
||||
) from e
|
||||
|
||||
return format_fn
|
||||
|
||||
|
||||
def _get_split_string(
|
||||
split: str,
|
||||
dataset_percent: int | None = None,
|
||||
dataset_k_rows: int | None = None,
|
||||
) -> str:
|
||||
"""Gets the formatted split string for the dataset.
|
||||
|
||||
This is used to format the split string as per
|
||||
https://huggingface.co/docs/datasets/v2.21.0/loading#slice-splits. Also, this
|
||||
function will only be used to load the partial dataset for validating the
|
||||
dataset against the template.
|
||||
|
||||
Args:
|
||||
split: Split of the dataset.
|
||||
dataset_percent: The percentage of the dataset to load.
|
||||
dataset_k_rows: The top k sequences to load from the dataset.
|
||||
|
||||
Returns:
|
||||
A formatted split string.
|
||||
"""
|
||||
# Validate the dataset_percent and dataset_k_rows values.
|
||||
if dataset_percent and dataset_k_rows:
|
||||
raise ValueError(
|
||||
"You can set either validate_percentage_of_dataset or"
|
||||
" validate_k_rows_of_dataset, but not both."
|
||||
)
|
||||
|
||||
if dataset_percent:
|
||||
logging.info("Loading %d percent of the dataset...", dataset_percent)
|
||||
return f"{split}[:{dataset_percent}%]"
|
||||
|
||||
if dataset_k_rows:
|
||||
logging.info("Loading top %d rows of the dataset...", dataset_k_rows)
|
||||
return f"{split}[:{dataset_k_rows}]"
|
||||
|
||||
return split
|
||||
|
||||
|
||||
def _github_template_path(template: str) -> str:
|
||||
"""Generates the path to the template in the Vertex AI Samples GitHub repo.
|
||||
|
||||
Args:
|
||||
template: Name of the template.
|
||||
|
||||
Returns:
|
||||
The path to the template in the Vertex AI Samples GitHub repo.
|
||||
"""
|
||||
# vertex-ai-samples directory may lie under separate directory depending on
|
||||
# the scratch_dir parameter in the notebook execution environment.
|
||||
vertex_ai_samples_abs_path = os.getcwd().split(
|
||||
_VERTEX_AI_SAMPLES_GITHUB_REPO_NAME
|
||||
)[0]
|
||||
return os.path.join(
|
||||
vertex_ai_samples_abs_path,
|
||||
_VERTEX_AI_SAMPLES_GITHUB_REPO_NAME,
|
||||
_VERTEX_AI_SAMPLES_GITHUB_TEMPLATE_DIR,
|
||||
template + ".json",
|
||||
)
|
||||
|
||||
|
||||
def _get_dataset(
|
||||
dataset_name: str,
|
||||
split: str,
|
||||
num_proc: int | None = None,
|
||||
) -> datasets.DatasetDict:
|
||||
"""Gets a dataset.
|
||||
|
||||
Args:
|
||||
dataset_name: Name of the dataset or path to a custom dataset.
|
||||
split: Split of the dataset.
|
||||
num_proc: Number of processors to use.
|
||||
|
||||
Returns:
|
||||
A dataset.
|
||||
"""
|
||||
dataset_name = force_gcs_fuse_path(dataset_name)
|
||||
if os.path.isfile(dataset_name):
|
||||
# Custom dataset.
|
||||
return datasets.load_dataset(
|
||||
"json",
|
||||
data_files=[dataset_name],
|
||||
split=split,
|
||||
num_proc=num_proc,
|
||||
)
|
||||
# HF dataset.
|
||||
return datasets.load_dataset(dataset_name, split=split, num_proc=num_proc)
|
||||
|
||||
|
||||
def should_add_pad_token(model_id: str) -> bool:
|
||||
"""Returns whether the model requires adding a special pad token.
|
||||
|
||||
Args:
|
||||
model_id: The name of the model.
|
||||
|
||||
Returns:
|
||||
True if the model requires adding a special pad token, False otherwise.
|
||||
"""
|
||||
return any(s.lower() in model_id.lower() for s in _MODELS_REQUIRING_PAD_TOKEN)
|
||||
|
||||
|
||||
def should_add_eos_token(model_id: str) -> bool:
|
||||
"""Returns whether the model requires adding a special eos token.
|
||||
|
||||
Args:
|
||||
model_id: The name of the model.
|
||||
|
||||
Returns:
|
||||
True if the model requires adding a special eos token, False otherwise.
|
||||
"""
|
||||
return any(m in model_id for m in _MODELS_REQUIRING_EOS_TOEKN)
|
||||
|
||||
|
||||
def load_tokenizer(
|
||||
pretrained_model_id: str,
|
||||
padding_side: str | None = None,
|
||||
access_token: str | None = None,
|
||||
) -> transformers.AutoTokenizer:
|
||||
"""Loads tokenizer based on `pretrained_model_id`.
|
||||
|
||||
Args:
|
||||
pretrained_model_id: The name of the pretrained model.
|
||||
padding_side: The side to pad the input on.
|
||||
access_token: The access token to use for the tokenizer.
|
||||
|
||||
Returns:
|
||||
The tokenizer.
|
||||
"""
|
||||
tokenizer_kwargs = {}
|
||||
if should_add_eos_token(pretrained_model_id):
|
||||
tokenizer_kwargs["add_eos_token"] = True
|
||||
if padding_side:
|
||||
tokenizer_kwargs["padding_side"] = padding_side
|
||||
|
||||
with accelerate.PartialState().local_main_process_first():
|
||||
tokenizer = transformers.AutoTokenizer.from_pretrained(
|
||||
pretrained_model_id,
|
||||
trust_remote_code=False,
|
||||
use_fast=True,
|
||||
token=access_token,
|
||||
**tokenizer_kwargs,
|
||||
)
|
||||
|
||||
if should_add_pad_token(pretrained_model_id):
|
||||
tokenizer.add_special_tokens({"pad_token": "[PAD]"})
|
||||
|
||||
return tokenizer
|
||||
|
||||
|
||||
def get_filtered_dataset(
|
||||
dataset: Any,
|
||||
input_column: str,
|
||||
max_seq_length: int,
|
||||
tokenizer: transformers.PreTrainedTokenizer,
|
||||
example_removed_threshold: float = 50.0,
|
||||
) -> Any:
|
||||
"""Returns the dataset by removing examples that are longer than max_seq_length.
|
||||
|
||||
Args:
|
||||
dataset: The dataset to filter.
|
||||
input_column: The input column in the dataset to be used.
|
||||
max_seq_length: The maximum sequence length.
|
||||
tokenizer: The tokenizer.
|
||||
example_removed_threshold: The percent threshold for the number of examples
|
||||
removed from the dataset. It should be in the range of [0, 100].
|
||||
|
||||
Returns:
|
||||
The filtered dataset.
|
||||
|
||||
Raises:
|
||||
ValueError: If more than `example_removed_threshold` of the dataset is
|
||||
filtered out.
|
||||
"""
|
||||
actual_dataset_length = len(dataset)
|
||||
filtered_dataset = dataset.filter(
|
||||
lambda x: len(tokenizer(x[input_column])["input_ids"]) <= max_seq_length
|
||||
)
|
||||
filtered_dataset_length = len(filtered_dataset)
|
||||
if actual_dataset_length != filtered_dataset_length:
|
||||
examples_removed_percent = (
|
||||
(actual_dataset_length - filtered_dataset_length)
|
||||
* 100
|
||||
/ actual_dataset_length
|
||||
)
|
||||
logging.info(
|
||||
"(%.2f%%) of examples token length is <= max-seq-length(%d); (%.2f%%) >"
|
||||
" max-seq-length. Filtering out %d example(s) which are longer than"
|
||||
" max-seq-length.",
|
||||
100 - examples_removed_percent,
|
||||
max_seq_length,
|
||||
examples_removed_percent,
|
||||
actual_dataset_length - filtered_dataset_length,
|
||||
)
|
||||
if examples_removed_percent > example_removed_threshold:
|
||||
raise ValueError(
|
||||
"More than %.2f%% of the dataset is filtered out. This may be due to"
|
||||
" small value of max-seq-length(%d) or incorrect template. Please"
|
||||
" increase the max-seq-length or check the template."
|
||||
% (examples_removed_percent, max_seq_length)
|
||||
)
|
||||
print(f"Some formatted examples from the dataset are: {filtered_dataset[:5]}")
|
||||
return filtered_dataset
|
||||
|
||||
|
||||
def format_dataset(
|
||||
dataset: datasets.Dataset,
|
||||
input_column: str,
|
||||
template: str = None,
|
||||
tokenizer: transformers.PreTrainedTokenizer | None = None,
|
||||
) -> datasets.Dataset:
|
||||
"""Takes a raw dataset and formats it using a template and tokenizer.
|
||||
|
||||
Args:
|
||||
dataset: The raw (unprocessed) dataset to format.
|
||||
input_column: The input column in the dataset to be used or updaded by the
|
||||
template. If it does not exist, the template's `prompt_no_input` will be
|
||||
used, and the input_column will be created.
|
||||
template: Name of the JSON template file under `templates/` or GCS path to
|
||||
the template file.
|
||||
tokenizer: The tokenizer to use for chat_template templates.
|
||||
|
||||
Returns:
|
||||
A dataset compatible with the template.
|
||||
"""
|
||||
return dataset.map(
|
||||
_format_template_fn(
|
||||
template,
|
||||
input_column=input_column,
|
||||
tokenizer=tokenizer,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def load_dataset_with_template(
|
||||
dataset_name: str,
|
||||
split: str,
|
||||
input_column: str,
|
||||
template: str = None,
|
||||
tokenizer: transformers.PreTrainedTokenizer | None = None,
|
||||
) -> tuple[Any, Any]:
|
||||
"""Loads dataset with templates.
|
||||
|
||||
Args:
|
||||
dataset_name: Name of the dataset or path to a custom dataset.
|
||||
split: Split of the dataset.
|
||||
input_column: The input column in the dataset to be used or updaded by the
|
||||
template. If it does not exist, the template's `prompt_no_input` will be
|
||||
used, and the input_column will be created.
|
||||
template: Name of the JSON template file under `templates/` or GCS path to
|
||||
the template file.
|
||||
tokenizer: The tokenizer to use for chat_template templates.
|
||||
|
||||
Returns:
|
||||
The raw dataset and the dataset compatible with the template.
|
||||
"""
|
||||
raw = _get_dataset(dataset_name, split=split)
|
||||
if template:
|
||||
templated = format_dataset(raw, input_column, template, tokenizer)
|
||||
else:
|
||||
templated = None
|
||||
|
||||
return raw, templated
|
||||
|
||||
|
||||
def validate_dataset_with_template(
|
||||
dataset_name: str,
|
||||
split: str,
|
||||
input_column: str,
|
||||
template: str,
|
||||
tokenizer: transformers.PreTrainedTokenizer | None = None,
|
||||
max_seq_length: int | None = None,
|
||||
use_multiprocessing: bool = False,
|
||||
validate_percentage_of_dataset: int | None = None,
|
||||
validate_k_rows_of_dataset: int | None = None,
|
||||
example_removed_threshold: float = 50.0,
|
||||
) -> Any:
|
||||
"""Validates dataset with templates.
|
||||
|
||||
This function will be used to load the dataset and validate it against the
|
||||
template. In case of validation, we also allow the users to load the dataset
|
||||
partially by allowing them to read x% or top k rows of the dataset. To
|
||||
validate the dataset, the template file must be available in the GCS bucket
|
||||
and the dataset must be available either in the GCS bucket or Hugging Face.
|
||||
|
||||
Args:
|
||||
dataset_name: Name of the dataset or path to a custom dataset.
|
||||
split: Split of the dataset.
|
||||
input_column: The input column in the dataset to be used or updaded by the
|
||||
template. If it does not exist, the template's `prompt_no_input` will be
|
||||
used, and the input_column will be created.
|
||||
template: Name of the JSON template file under `templates/` or GCS path to
|
||||
the template file.
|
||||
tokenizer: The tokenizer to use for chat_template templates.
|
||||
max_seq_length: The maximum sequence length.
|
||||
use_multiprocessing: If True, it will use multiprocessing to load the
|
||||
dataset.
|
||||
validate_percentage_of_dataset: The percentage of the dataset to load.
|
||||
validate_k_rows_of_dataset: The top k sequences to load from the dataset.
|
||||
example_removed_threshold: The threshold for the number of examples removed
|
||||
from the dataset.
|
||||
|
||||
Returns:
|
||||
None if the validation is successful, otherwise returns the error message.
|
||||
"""
|
||||
if not template:
|
||||
raise ValueError("template is required for validate_dataset.")
|
||||
|
||||
if not dataset_name:
|
||||
raise ValueError("dataset_name is empty.")
|
||||
|
||||
if not split:
|
||||
raise ValueError("split is empty.")
|
||||
|
||||
split = _get_split_string(
|
||||
split,
|
||||
validate_percentage_of_dataset,
|
||||
validate_k_rows_of_dataset,
|
||||
)
|
||||
|
||||
num_proc = multiprocessing.cpu_count() if use_multiprocessing else 1
|
||||
|
||||
# gcsfuse cannot be used from the notebook runtime env. Hence, we have
|
||||
# to download dataset and template from gcs to local.
|
||||
if is_gcs_path(dataset_name):
|
||||
dataset_name = download_gcs_uri_to_local(dataset_name, LOCAL_BASE_MODEL_DIR)
|
||||
|
||||
if is_gcs_path(template):
|
||||
template_path = download_gcs_uri_to_local(template, LOCAL_TEMPLATE_DIR)
|
||||
elif os.path.isfile(_github_template_path(template)):
|
||||
template_path = _github_template_path(template)
|
||||
else:
|
||||
raise ValueError(
|
||||
f"Template file {template} does not exist. To validate the"
|
||||
" dataset, please provide a valid GCS path for the template or a valid"
|
||||
" template name from"
|
||||
f" https://github.com/GoogleCloudPlatform/{_VERTEX_AI_SAMPLES_GITHUB_REPO_NAME}/tree/main/{_VERTEX_AI_SAMPLES_GITHUB_TEMPLATE_DIR}."
|
||||
)
|
||||
|
||||
dataset = format_dataset(
|
||||
_get_dataset(dataset_name, split, num_proc),
|
||||
input_column,
|
||||
template_path,
|
||||
tokenizer,
|
||||
)
|
||||
|
||||
if tokenizer is not None:
|
||||
get_filtered_dataset(
|
||||
dataset=dataset,
|
||||
input_column=input_column,
|
||||
max_seq_length=max_seq_length,
|
||||
tokenizer=tokenizer,
|
||||
example_removed_threshold=example_removed_threshold,
|
||||
)
|
||||
print(
|
||||
"Dataset {} is compatible with the {} template.".format(
|
||||
os.path.basename(dataset_name), os.path.basename(template)
|
||||
)
|
||||
)
|
||||
@@ -0,0 +1,275 @@
|
||||
"""Utility functions for interacting with Google Cloud Platform."""
|
||||
|
||||
import datetime
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import uuid
|
||||
|
||||
from google.cloud import aiplatform
|
||||
import requests
|
||||
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def get_project_id() -> str:
|
||||
"""Read cloud project id from metadata service."""
|
||||
project_request = requests.get(
|
||||
"http://metadata.google.internal/computeMetadata/v1/project/project-id",
|
||||
headers={"Metadata-Flavor": "Google"},
|
||||
)
|
||||
return project_request.text
|
||||
|
||||
|
||||
def get_region() -> str:
|
||||
"""Read region from metadata service."""
|
||||
region_request = requests.get(
|
||||
"http://metadata.google.internal/computeMetadata/v1/instance/region",
|
||||
headers={"Metadata-Flavor": "Google"},
|
||||
)
|
||||
return region_request.text.split("/")[-1]
|
||||
|
||||
|
||||
# Get the default cloud project id and region
|
||||
PROJECT_ID = get_project_id()
|
||||
REGION = get_region()
|
||||
|
||||
|
||||
def init_aiplatform(project: str = None, location: str = None) -> None:
|
||||
"""Initialize the Vertex AI SDK.
|
||||
|
||||
Args:
|
||||
project: The Google Cloud project ID.
|
||||
location: The Google Cloud location.
|
||||
"""
|
||||
project = PROJECT_ID if project is None else project
|
||||
location = REGION if location is None else location
|
||||
aiplatform.init(project=project, location=location)
|
||||
subprocess.call([
|
||||
"gcloud",
|
||||
"services",
|
||||
"enable",
|
||||
"aiplatform.googleapis.com",
|
||||
"compute.googleapis.com",
|
||||
])
|
||||
|
||||
|
||||
def run_command(command: list[str]) -> str:
|
||||
"""Runs a shell command and returns the output.
|
||||
|
||||
Args:
|
||||
command: The shell command to run as a list.
|
||||
|
||||
Returns:
|
||||
The output of the command.
|
||||
"""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
command,
|
||||
check=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
)
|
||||
return result.stdout
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.error("Error: %s", e.stderr)
|
||||
raise e
|
||||
|
||||
|
||||
def enable_apis() -> None:
|
||||
"""Enable the Vertex AI API and Compute Engine API."""
|
||||
logger.info("Enabling Vertex AI API and Compute Engine API.")
|
||||
run_command([
|
||||
"gcloud",
|
||||
"services",
|
||||
"enable",
|
||||
"aiplatform.googleapis.com",
|
||||
"compute.googleapis.com",
|
||||
])
|
||||
|
||||
|
||||
def setup_buckets(bucket_uri: str, model_bucket_name: str) -> tuple[str, str]:
|
||||
"""Set up Cloud Storage buckets for storing experiment artifacts.
|
||||
|
||||
Args:
|
||||
bucket_uri: The bucket URI provided by the user.
|
||||
model_bucket_name: The name of the model bucket.
|
||||
|
||||
Returns:
|
||||
A tuple containing the bucket name and model bucket path.
|
||||
"""
|
||||
if not bucket_uri.strip():
|
||||
# Generate a default bucket URI if none provided
|
||||
now = datetime.datetime.now().strftime("%Y%m%d%H%M%S")
|
||||
bucket_uri = f"gs://{PROJECT_ID}-tmp-{now}-{str(uuid.uuid4())[:4]}"
|
||||
logger.info("No bucket URI provided. Using default bucket: %s", bucket_uri)
|
||||
else:
|
||||
if not bucket_uri.startswith("gs://"):
|
||||
raise ValueError("Bucket URI must start with 'gs://'.")
|
||||
# Remove any trailing slashes
|
||||
bucket_uri = bucket_uri.rstrip("/")
|
||||
|
||||
bucket_name = "/".join(bucket_uri.split("/")[:3])
|
||||
|
||||
# Check if bucket exists
|
||||
try:
|
||||
run_command(["gsutil", "ls", "-b", bucket_uri])
|
||||
logger.info("Bucket %s already exists.", bucket_uri)
|
||||
except subprocess.CalledProcessError:
|
||||
logger.info("Creating bucket %s.", bucket_uri)
|
||||
# Create the bucket in the same region as the project
|
||||
run_command(["gsutil", "mb", "-l", REGION, bucket_uri])
|
||||
|
||||
# Construct the model bucket path
|
||||
model_bucket = os.path.join(bucket_uri, model_bucket_name)
|
||||
|
||||
# Check if the model bucket exists (as a folder within the main bucket)
|
||||
try:
|
||||
run_command(["gsutil", "ls", model_bucket])
|
||||
logger.info("Model bucket %s already exists.", model_bucket)
|
||||
except subprocess.CalledProcessError:
|
||||
logger.info("Creating model bucket %s.", model_bucket)
|
||||
# Create the model bucket folder
|
||||
run_command(["gsutil", "cp", "/dev/null", model_bucket + "/"])
|
||||
|
||||
return bucket_name, model_bucket
|
||||
|
||||
|
||||
def get_service_account() -> str:
|
||||
"""Get the default service account."""
|
||||
shell_output = run_command(["gcloud", "projects", "describe", PROJECT_ID])
|
||||
project_number_line = next(
|
||||
(line for line in shell_output.splitlines() if "projectNumber" in line),
|
||||
None,
|
||||
)
|
||||
if project_number_line:
|
||||
project_number = project_number_line.split(":")[1].strip().replace("'", "")
|
||||
service_account = f"{project_number}-compute@developer.gserviceaccount.com"
|
||||
logger.info("Using default Service Account: %s", service_account)
|
||||
return service_account
|
||||
else:
|
||||
raise ValueError("Could not find project number in gcloud output.")
|
||||
|
||||
|
||||
def get_project_number() -> str:
|
||||
"""Get the default project number."""
|
||||
shell_output = run_command(["gcloud", "projects", "describe", PROJECT_ID])
|
||||
project_number_line = next(
|
||||
(line for line in shell_output.splitlines() if "projectNumber" in line),
|
||||
None,
|
||||
)
|
||||
if project_number_line:
|
||||
project_number = project_number_line.split(":")[1].strip().replace("'", "")
|
||||
logger.info("Using default Project Number: %s", project_number)
|
||||
return project_number
|
||||
else:
|
||||
raise ValueError("Could not find project number in gcloud output.")
|
||||
|
||||
|
||||
def provision_permissions(service_account: str, bucket_name: str) -> None:
|
||||
"""Provision permissions to the service account with the GCS bucket."""
|
||||
if bucket_name:
|
||||
run_command([
|
||||
"gsutil",
|
||||
"iam",
|
||||
"ch",
|
||||
f"serviceAccount:{service_account}:roles/storage.admin",
|
||||
bucket_name,
|
||||
])
|
||||
|
||||
|
||||
def set_gcloud_project() -> None:
|
||||
"""Set gcloud config project."""
|
||||
run_command(["gcloud", "config", "set", "project", PROJECT_ID])
|
||||
|
||||
|
||||
def initialize(
|
||||
bucket_uri: str, model_bucket_name: str, create_bucket: bool
|
||||
) -> tuple[str, str]:
|
||||
"""Initialize the environment.
|
||||
|
||||
Args:
|
||||
bucket_uri: The bucket URI provided by the user.
|
||||
model_bucket_name: The name of the model bucket.
|
||||
create_bucket: Whether to create the bucket or not.
|
||||
|
||||
Returns:
|
||||
A tuple containing the model bucket path and service account.
|
||||
"""
|
||||
enable_apis()
|
||||
bucket_name = None
|
||||
if create_bucket:
|
||||
bucket_name, model_bucket = setup_buckets(bucket_uri, model_bucket_name)
|
||||
else:
|
||||
model_bucket = None
|
||||
service_account = get_service_account()
|
||||
provision_permissions(service_account, bucket_name)
|
||||
set_gcloud_project()
|
||||
return model_bucket, service_account
|
||||
|
||||
|
||||
def clean_resources_ui(
|
||||
project_id: str,
|
||||
region: str,
|
||||
endpoint_name: str,
|
||||
delete_bucket: bool,
|
||||
bucket_name: str = None,
|
||||
) -> str:
|
||||
"""UI function for cleaning a specific Vertex AI endpoint and its model."""
|
||||
if delete_bucket and not bucket_name:
|
||||
raise ValueError("Bucket name is required when 'Delete Bucket' is checked.")
|
||||
|
||||
try:
|
||||
delete_endpoint_and_model(project_id, region, endpoint_name)
|
||||
bucket_status_message = ""
|
||||
if delete_bucket:
|
||||
bucket_status_message = delete_gcs_bucket(bucket_name)
|
||||
if endpoint_name:
|
||||
return (
|
||||
f"Endpoint {endpoint_name} and associated model deleted successfully!"
|
||||
f" {bucket_status_message}"
|
||||
)
|
||||
else:
|
||||
return (
|
||||
"There are currently no endpoints available for deletion."
|
||||
f" {bucket_status_message}"
|
||||
)
|
||||
except Exception as e: # pylint: disable=broad-exception-caught
|
||||
return f"Error cleaning up resources: {e}"
|
||||
|
||||
|
||||
def delete_endpoint_and_model(
|
||||
project_id: str, region: str, endpoint_name: str
|
||||
) -> None:
|
||||
"""Deletes a specific Vertex AI endpoint and its associated model."""
|
||||
if endpoint_name:
|
||||
endpoint_id = endpoint_name.split(" - ")[0]
|
||||
endpoint_resource_name = (
|
||||
f"projects/{project_id}/locations/{region}/endpoints/{endpoint_id}"
|
||||
)
|
||||
endpoint = aiplatform.Endpoint(
|
||||
endpoint_resource_name, project=project_id, location=region
|
||||
)
|
||||
deployed_models = endpoint.list_models()
|
||||
for deployed_model in deployed_models:
|
||||
endpoint.undeploy(deployed_model_id=deployed_model.id)
|
||||
model = aiplatform.Model(deployed_model.model)
|
||||
model.delete()
|
||||
endpoint.delete()
|
||||
|
||||
|
||||
def delete_gcs_bucket(bucket_name: str) -> str:
|
||||
"""Deletes a GCS bucket using gsutil."""
|
||||
try:
|
||||
run_command(["gsutil", "-m", "rm", "-r", bucket_name])
|
||||
logger.info("Bucket %s deleted using gsutil.", bucket_name)
|
||||
return f"Bucket {bucket_name} deleted successfully!"
|
||||
except subprocess.CalledProcessError as e:
|
||||
logger.error(
|
||||
"Error deleting bucket %s using gsutil: %s", bucket_name, str(e)
|
||||
)
|
||||
return f"Bucket {bucket_name} could not be found or deleted. "
|
||||
@@ -42,7 +42,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/gke_model_ui_deployment_notebook.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -42,7 +42,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/gke_model_ui_deployment_notebook_auto.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -162,8 +162,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title # Chat completion for text-only models { vertical-output: true}\n",
|
||||
"# @markdown You may send prompts to the model server for prediction.\n",
|
||||
"# @title # Chat completion for text-only models {vertical-output: true}\n",
|
||||
"# @markdown Run cell to prompt the model server for prediction.\n",
|
||||
"# @markdown\n",
|
||||
"# @markdown * **user_prompt (string):** This is the text prompt you provide to the language model. It's the question or instruction e (e.g., \"Explain neural networks\").\n",
|
||||
"# @markdown * **temperature (number):** This parameter controls the randomness of the model's output. It influences how the model selects the next token in the sequence it generates. Typical values range from 0.2 to 1.0.\n",
|
||||
@@ -180,38 +180,68 @@
|
||||
"REGION = \"\" # @param {type:\"string\", isTemplate:true}\n",
|
||||
"NAMESPACE = \"\" # @param {type:\"string\", isTemplate:true}\n",
|
||||
"DEPLOYMENT = \"\" # @param {type:\"string\", isTemplate:true}\n",
|
||||
"DEPLOYMENT_APP_LABEL = \"\" # @param {type:\"string\", isTemplate:true}\n",
|
||||
"\n",
|
||||
"SERVICE = f\"{DEPLOYMENT}-service\"\n",
|
||||
"POD_PORT = \"\" # @param {type:\"string\", isTemplate:true}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def _run_kubectl(cmd):\n",
|
||||
" \"\"\"Executes a kubectl command and returns its stdout.\"\"\"\n",
|
||||
" result = subprocess.run(cmd, capture_output=True, text=True, check=True, timeout=60)\n",
|
||||
" return result.stdout.strip()\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def fetch_cluster_credential(cluster, region, project_id):\n",
|
||||
"def _run_kubectl(cmd, timeout=60):\n",
|
||||
" \"\"\"Executes a kubectl command.\"\"\"\n",
|
||||
" try:\n",
|
||||
" # Ensure credentials for the target cluster\n",
|
||||
" cred_cmd = [\n",
|
||||
" \"gcloud\",\n",
|
||||
" \"container\",\n",
|
||||
" \"clusters\",\n",
|
||||
" \"get-credentials\",\n",
|
||||
" cluster,\n",
|
||||
" f\"--location={region}\",\n",
|
||||
" f\"--project={project_id}\",\n",
|
||||
" ]\n",
|
||||
" _run_kubectl(cred_cmd)\n",
|
||||
" except Exception as e:\n",
|
||||
" # Original code prints error and returns empty dict\n",
|
||||
" print(f\"Error fetching cluster credentials: {e}\")\n",
|
||||
" return {}\n",
|
||||
" result = subprocess.run(\n",
|
||||
" cmd, capture_output=True, text=True, check=True, timeout=timeout\n",
|
||||
" )\n",
|
||||
" return result.stdout.strip()\n",
|
||||
" except subprocess.CalledProcessError as e:\n",
|
||||
" raise RuntimeError(\n",
|
||||
" f\"Kubectl command failed: {' '.join(e.cmd)}\\nStderr: {e.stderr}\"\n",
|
||||
" ) from e\n",
|
||||
" except subprocess.TimeoutExpired as e:\n",
|
||||
" raise RuntimeError(f\"Kubectl command timed out: {' '.join(e.cmd)}\") from e\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_deployment_pod_name(deployment, namespace, deployment_app_label):\n",
|
||||
" \"\"\"Finds the running pod name for a given deployment and namespace.\"\"\"\n",
|
||||
"def fetch_cluster_credentials(cluster, region, project_id):\n",
|
||||
" \"\"\"Ensures credentials for the target GKE cluster.\"\"\"\n",
|
||||
" cred_cmd = [\n",
|
||||
" \"gcloud\",\n",
|
||||
" \"container\",\n",
|
||||
" \"clusters\",\n",
|
||||
" \"get-credentials\",\n",
|
||||
" cluster,\n",
|
||||
" f\"--location={region}\",\n",
|
||||
" f\"--project={project_id}\",\n",
|
||||
" ]\n",
|
||||
" _run_kubectl(cred_cmd)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_deployment_selector_labels(deployment_name, namespace):\n",
|
||||
" \"\"\"Retrieves the selector labels for a given Kubernetes deployment.\"\"\"\n",
|
||||
" cmd = [\n",
|
||||
" \"kubectl\",\n",
|
||||
" \"get\",\n",
|
||||
" \"deployment\",\n",
|
||||
" deployment_name,\n",
|
||||
" \"-n\",\n",
|
||||
" namespace,\n",
|
||||
" \"-o\",\n",
|
||||
" \"json\",\n",
|
||||
" ]\n",
|
||||
" deployment_json = _run_kubectl(cmd)\n",
|
||||
" deployment_data = json.loads(deployment_json)\n",
|
||||
"\n",
|
||||
" selector_labels = (\n",
|
||||
" deployment_data.get(\"spec\", {}).get(\"selector\", {}).get(\"matchLabels\")\n",
|
||||
" )\n",
|
||||
" if not selector_labels:\n",
|
||||
" raise RuntimeError(\n",
|
||||
" f\"No selector labels found for deployment '{deployment_name}' in\"\n",
|
||||
" f\" namespace '{namespace}'.\"\n",
|
||||
" )\n",
|
||||
" return selector_labels\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_running_pod_name(deployment_name, namespace):\n",
|
||||
" \"\"\"Retrieves the name of a running pod associated with a deployment.\"\"\"\n",
|
||||
" selector_labels = get_deployment_selector_labels(deployment_name, namespace)\n",
|
||||
" label_selector_str = \",\".join(f\"{k}={v}\" for k, v in selector_labels.items())\n",
|
||||
"\n",
|
||||
" cmd = [\n",
|
||||
" \"kubectl\",\n",
|
||||
@@ -222,91 +252,99 @@
|
||||
" \"-o\",\n",
|
||||
" \"json\",\n",
|
||||
" \"-l\",\n",
|
||||
" f\"app={deployment_app_label}\",\n",
|
||||
" label_selector_str,\n",
|
||||
" \"--field-selector=status.phase=Running\",\n",
|
||||
" ]\n",
|
||||
" try:\n",
|
||||
" pods_json = _run_kubectl(cmd)\n",
|
||||
" pods = json.loads(pods_json)\n",
|
||||
" if pods.get(\"items\"):\n",
|
||||
" return pods[\"items\"][0][\"metadata\"][\"name\"]\n",
|
||||
" print(f\"No running pods found for {deployment} in {namespace}.\")\n",
|
||||
" return None\n",
|
||||
" except (\n",
|
||||
" subprocess.CalledProcessError,\n",
|
||||
" json.JSONDecodeError,\n",
|
||||
" IndexError,\n",
|
||||
" KeyError,\n",
|
||||
" ) as e:\n",
|
||||
" print(f\"Error getting pod name for {deployment} in {namespace}: {e}\")\n",
|
||||
" return None\n",
|
||||
" pods_json = _run_kubectl(cmd)\n",
|
||||
" pods_data = json.loads(pods_json)\n",
|
||||
"\n",
|
||||
" if not pods_data.get(\"items\"):\n",
|
||||
" raise RuntimeError(\n",
|
||||
" f\"No running pods found for deployment '{deployment_name}' in namespace\"\n",
|
||||
" f\" '{namespace}' with selector '{label_selector_str}'.\"\n",
|
||||
" )\n",
|
||||
" return pods_data[\"items\"][0][\"metadata\"][\"name\"]\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def check_inference_label(pod_name, namespace):\n",
|
||||
"def check_vllm_inference_label(pod_name, namespace):\n",
|
||||
" \"\"\"Checks if the specified pod has the vLLM inference server label.\"\"\"\n",
|
||||
"\n",
|
||||
" cmd = [\"kubectl\", \"get\", \"pod\", pod_name, \"-n\", namespace, \"-o\", \"json\"]\n",
|
||||
" try:\n",
|
||||
" pod_json = _run_kubectl(cmd)\n",
|
||||
" labels = json.loads(pod_json).get(\"metadata\", {}).get(\"labels\", {})\n",
|
||||
" return labels.get(\"ai.gke.io/inference-server\") == \"vllm\"\n",
|
||||
" except (subprocess.CalledProcessError, json.JSONDecodeError, KeyError) as e:\n",
|
||||
" print(f\"Error checking labels for pod {pod_name} in {namespace}: {e}\")\n",
|
||||
" return False\n",
|
||||
" pod_json = _run_kubectl(cmd)\n",
|
||||
" labels = json.loads(pod_json).get(\"metadata\", {}).get(\"labels\", {})\n",
|
||||
" return labels.get(\"ai.gke.io/inference-server\") == \"vllm\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_service_endpoint(service, namespace):\n",
|
||||
" \"\"\"Retrieve the service endpoint of the deployment\"\"\"\n",
|
||||
" endpoint_cmd = [\n",
|
||||
" \"kubectl\",\n",
|
||||
" \"get\",\n",
|
||||
" \"endpoints\",\n",
|
||||
" service,\n",
|
||||
" \"-n\",\n",
|
||||
" namespace,\n",
|
||||
" ]\n",
|
||||
" try:\n",
|
||||
" endpoint_output = _run_kubectl(endpoint_cmd).splitlines()\n",
|
||||
" if len(endpoint_output) < 2 or len(endpoint_output[1].split()) < 2:\n",
|
||||
" print(f\"Endpoint data incomplete for {service}.\")\n",
|
||||
" return None\n",
|
||||
" endpoint = endpoint_output[1].split()[\n",
|
||||
" 1\n",
|
||||
" ] # Assumes format: NAME ENDPOINTS AGE -> service ip:port,... age\n",
|
||||
" return endpoint\n",
|
||||
" except subprocess.CalledProcessError as e:\n",
|
||||
" print(f\"Error getting endpoints for {service}: {e}\")\n",
|
||||
" return None\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def process_response(request, pod_name, pod_endpoint, is_vllm_inference, namespace):\n",
|
||||
" \"\"\"Sends a request to the pod and processes the response.\"\"\"\n",
|
||||
"\n",
|
||||
" json_data_escaped = json.dumps(request).replace(\"'\", \"'\\\\''\")\n",
|
||||
"def send_inference_request(\n",
|
||||
" request_payload, pod_name, pod_port, is_vllm_inference, namespace\n",
|
||||
"):\n",
|
||||
" \"\"\"Sends an inference request to the specified pod and returns the model's response.\"\"\"\n",
|
||||
" json_data_escaped = json.dumps(request_payload).replace(\"'\", \"'\\\\''\")\n",
|
||||
" curl_cmd = (\n",
|
||||
" f\"kubectl exec -n {namespace} -t {pod_name} -- curl -s -X POST\"\n",
|
||||
" f' http://{pod_endpoint}/generate -H \"Content-Type: application/json\"'\n",
|
||||
" f' http://localhost:{pod_port}/generate -H \"Content-Type:'\n",
|
||||
" ' application/json\"'\n",
|
||||
" f\" -d '{json_data_escaped}' 2> /dev/null\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" response_raw = _run_kubectl([\"bash\", \"-c\", curl_cmd])\n",
|
||||
"\n",
|
||||
" if not response_raw:\n",
|
||||
" raise RuntimeError(f\"Empty response received from pod '{pod_name}'.\")\n",
|
||||
"\n",
|
||||
" try:\n",
|
||||
" response_raw = _run_kubectl([\"bash\", \"-c\", curl_cmd])\n",
|
||||
" if not response_raw:\n",
|
||||
" return f\"Error: Empty response from pod {pod_name}.\"\n",
|
||||
" first_line = response_raw.splitlines()[0]\n",
|
||||
" data = json.loads(first_line)\n",
|
||||
" except json.JSONDecodeError as e:\n",
|
||||
" raise RuntimeError(\n",
|
||||
" f\"Failed to decode JSON response from pod: {e}. Raw: {response_raw}\"\n",
|
||||
" ) from e\n",
|
||||
" except IndexError:\n",
|
||||
" raise RuntimeError(\n",
|
||||
" f\"Unexpected empty response line from pod. Raw: {response_raw}\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" if is_vllm_inference: # vLLM format\n",
|
||||
" predictions = data.get(\"predictions\")\n",
|
||||
" if isinstance(predictions, (list, tuple)) and predictions:\n",
|
||||
" return predictions[0]\n",
|
||||
" return f\"Error: Unexpected vLLM format. Raw: {first_line}\"\n",
|
||||
" else: # TGI format\n",
|
||||
" generated_text = data.get(\"generated_text\")\n",
|
||||
" if generated_text is not None:\n",
|
||||
" return generated_text\n",
|
||||
" return f\"Error: Unexpected TGI format. Raw: {first_line}\"\n",
|
||||
" except Exception as e:\n",
|
||||
" return f\"Unexpected error during response processing: {e}\"\n",
|
||||
" if is_vllm_inference:\n",
|
||||
" predictions = data.get(\"predictions\")\n",
|
||||
" if isinstance(predictions, list) and predictions:\n",
|
||||
" return predictions[0]\n",
|
||||
" raise RuntimeError(f\"Unexpected vLLM response format. Raw data: {data}\")\n",
|
||||
" else: # TGI format\n",
|
||||
" generated_text = data.get(\"generated_text\")\n",
|
||||
" if generated_text is not None:\n",
|
||||
" return generated_text\n",
|
||||
" raise RuntimeError(f\"Unexpected TGI response format. Raw data: {data}\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# --- Main Execution Logic ---\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def execute_chat_completion(\n",
|
||||
" deployment_name, namespace, pod_port, user_prompt, temperature, max_tokens\n",
|
||||
"):\n",
|
||||
" \"\"\"Executes the full chat completion process: fetches credentials, finds a pod,\n",
|
||||
"\n",
|
||||
" determines inference type, sends a request, and returns the response.\n",
|
||||
" \"\"\"\n",
|
||||
" display(Markdown(\"Establishing cluster credentials...\"))\n",
|
||||
" fetch_cluster_credentials(CLUSTER, REGION, PROJECT_ID)\n",
|
||||
"\n",
|
||||
" display(Markdown(\"Retrieving pod information...\"))\n",
|
||||
" pod_name = get_running_pod_name(deployment_name, namespace)\n",
|
||||
" display(Markdown(f\"Successfully identified pod: `{pod_name}`\"))\n",
|
||||
"\n",
|
||||
" is_vllm = check_vllm_inference_label(pod_name, namespace)\n",
|
||||
"\n",
|
||||
" request_payload = {\n",
|
||||
" \"max_tokens\": max_tokens,\n",
|
||||
" \"temperature\": temperature,\n",
|
||||
" \"prompt\" if is_vllm else \"inputs\": user_prompt,\n",
|
||||
" }\n",
|
||||
" display(Markdown(\"Sending inference request...\"))\n",
|
||||
" response = send_inference_request(\n",
|
||||
" request_payload, pod_name, pod_port, is_vllm, namespace\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" return response\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# --- Widgets Setup ---\n",
|
||||
@@ -330,40 +368,24 @@
|
||||
"\n",
|
||||
"# --- Submit Button Logic ---\n",
|
||||
"def on_submit_clicked(b):\n",
|
||||
" \"\"\"Handles the submit button click event.\"\"\"\n",
|
||||
" with output_area_response:\n",
|
||||
" clear_output()\n",
|
||||
" display(Markdown(\"Loading...\"))\n",
|
||||
"\n",
|
||||
" fetch_cluster_credential(CLUSTER, REGION, PROJECT_ID)\n",
|
||||
"\n",
|
||||
" # retrieve deployment pod\n",
|
||||
" pod_name = get_deployment_pod_name(DEPLOYMENT, NAMESPACE, DEPLOYMENT_APP_LABEL)\n",
|
||||
" if not pod_name:\n",
|
||||
" display(\n",
|
||||
" Markdown(f\"**Error:** Could not find running pod for `{DEPLOYMENT}`.\")\n",
|
||||
" )\n",
|
||||
" return\n",
|
||||
"\n",
|
||||
" # build the request message\n",
|
||||
" is_vllm = check_inference_label(pod_name, NAMESPACE)\n",
|
||||
" request = {\n",
|
||||
" \"max_tokens\": max_tokens_widget.value,\n",
|
||||
" \"temperature\": temperature_widget.value,\n",
|
||||
" \"prompt\" if is_vllm else \"inputs\": user_prompt_widget.value,\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
" # retrieve service endpoint for the deployment\n",
|
||||
" endpoint = get_service_endpoint(SERVICE, NAMESPACE)\n",
|
||||
" if not endpoint:\n",
|
||||
" display(Markdown(f\"**Error getting endpoints for `{SERVICE}`:**\\n\"))\n",
|
||||
" return\n",
|
||||
"\n",
|
||||
" # prompt test the deployment endpoint\n",
|
||||
" try:\n",
|
||||
" response = process_response(request, pod_name, endpoint, is_vllm, NAMESPACE)\n",
|
||||
" display(Markdown(f\"**Response:**\\n\\n{response}\"))\n",
|
||||
" model_response = execute_chat_completion(\n",
|
||||
" DEPLOYMENT,\n",
|
||||
" NAMESPACE,\n",
|
||||
" POD_PORT,\n",
|
||||
" user_prompt_widget.value,\n",
|
||||
" temperature_widget.value,\n",
|
||||
" max_tokens_widget.value,\n",
|
||||
" )\n",
|
||||
" clear_output()\n",
|
||||
" display(Markdown(f\"**Response:**\\n\\n{model_response}\"))\n",
|
||||
" except Exception as e:\n",
|
||||
" display(Markdown(f\"**Unexpected Error:**\\n```\\n{e}\\n```\"))\n",
|
||||
" clear_output()\n",
|
||||
" display(Markdown(f\"**An error occurred:**\\n```\\n{e}\\n```\"))\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# --- Display Widgets ---\n",
|
||||
|
||||
@@ -29,7 +29,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "99c1c3fc2ca5",
|
||||
"metadata": {
|
||||
"id": "778cc1227be8"
|
||||
"id": "71a642b5575a"
|
||||
},
|
||||
"source": [
|
||||
"# Vertex AI Model Garden - Advanced Features\n",
|
||||
@@ -42,7 +42,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_advanced_features.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -120,7 +120,7 @@
|
||||
"id": "L3dqbxovo5t6",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "2b585189a670"
|
||||
"id": "86a3d4d4d3f5"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -148,7 +148,7 @@
|
||||
"# Install and import the necessary packages\n",
|
||||
"! pip install -q openai google-auth requests\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.93.1'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.97.0'\n",
|
||||
"\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
|
||||
@@ -50,7 +50,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_agent_engine_llama3_1.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\"><br> View on GitHub\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_axolotl_finetuning.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -231,7 +231,7 @@
|
||||
"import yaml\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.93.1'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.97.0'\n",
|
||||
"\n",
|
||||
"common_util = importlib.import_module(\n",
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
@@ -367,34 +367,42 @@
|
||||
"# @markdown | falcon | tiiuae/falcon-7b | examples/falcon/config-7b-lora.yml |\n",
|
||||
"# @markdown | falcon | tiiuae/falcon-7b | examples/falcon/config-7b.yml |\n",
|
||||
"# @markdown | gemma | google/gemma-7b | examples/gemma/qlora.yml |\n",
|
||||
"# @markdown | gemma2 | google/gemma-7b | examples/gemma2/qlora.yml |\n",
|
||||
"# @markdown | gemma3 | google/gemma-3-1b-it | examples/gemma3/gemma-3-1b-qlora.yml |\n",
|
||||
"# @markdown | gemma3 | google/gemma-3-4b-it | examples/gemma3/gemma-3-4b-qlora.yml |\n",
|
||||
"# @markdown | llama-2 | NousResearch/Llama-2-7b-hf | examples/llama-2/fft_optimized.yml |\n",
|
||||
"# @markdown | llama-2 | NousResearch/Llama-2-7b-hf | examples/llama-2/loftq.yml |\n",
|
||||
"# @markdown | llama-2 | NousResearch/Llama-2-7b-hf | examples/llama-2/lora.yml |\n",
|
||||
"# @markdown | llama-2 | NousResearch/Llama-2-7b-hf | examples/llama-2/qlora-fsdp.yml |\n",
|
||||
"# @markdown | llama-2 | NousResearch/Llama-2-7b-hf | examples/llama-2/qlora.yml |\n",
|
||||
"# @markdown | llama-3 | hugging-quants/Meta-Llama-3.1-405B-BNB-NF4-BF16 | examples/llama-3/qlora-fsdp-405b.yaml |\n",
|
||||
"# @markdown | llama-3 | NousResearch/Llama-3.2-1B | examples/llama-3/lora-1b-kernels.yml |\n",
|
||||
"# @markdown | llama-3 | NousResearch/Meta-Llama-3.1-8B | examples/llama-3/fft-8b.yaml |\n",
|
||||
"# @markdown | llama-3 | NousResearch/Meta-Llama-3-8B-Instruct | examples/llama-3/instruct-lora-8b.yml |\n",
|
||||
"# @markdown | llama-3 | NousResearch/Meta-Llama-3-8B | examples/llama-3/lora-8b.yml |\n",
|
||||
"# @markdown | llama-3 | NousResearch/Llama-3.2-1B | examples/llama-3/qlora-1b.yml |\n",
|
||||
"# @markdown | llama-3 | NousResearch/Llama-3.2-1B | examples/llama-3/lora-1b.yml |\n",
|
||||
"# @markdown | llama-3 | casperhansen/llama-3-70b-fp16 | examples/llama-3/qlora-fsdp-70b.yaml |\n",
|
||||
"# @markdown | llama-3 | meta-llama/Llama-3.2-1B | examples/llama-3/lora-1b-deduplicate-dpo.yml |\n",
|
||||
"# @markdown | llama-3 | meta-llama/Llama-3.2-1B | examples/llama-3/lora-1b-deduplicate-sft.yml |\n",
|
||||
"# @markdown | llama-3 | meta-llama/Llama-3.2-1B | examples/llama-3/lora-1b-sample-packing-sequentially.yml |\n",
|
||||
"# @markdown | mistral | mistralai/Mistral-7B-v0.1 | examples/mistral/config.yml |\n",
|
||||
"# @markdown | mistral | mistralai/Mistral-7B-v0.1 | examples/mistral/lora-mps.yml |\n",
|
||||
"# @markdown | mistral | mistralai/Mistral-7B-v0.1 | examples/mistral/lora.yml |\n",
|
||||
"# @markdown | mistral | mistralai/Mistral-7B-v0.1 | examples/mistral/mistral-qlora-orpo.yml |\n",
|
||||
"# @markdown | mistral | mistralai/Mistral-7B-Instruct-v0.2 | examples/mistral/mistral-dpo-qlora.yml |\n",
|
||||
"# @markdown | mistral | mistral-community/Mixtral-8x22B-v0.1 | examples/mistral/mixtral-8x22b-qlora-fsdp.yml |\n",
|
||||
"# @markdown | mistral | mistralai/Mixtral-8x7B-v0.1 | examples/mistral/mixtral.yml |\n",
|
||||
"# @markdown | mistral | mistralai/Mixtral-8x7B-v0.1 | examples/mistral/mixtral-qlora-fsdp.yml |\n",
|
||||
"# @markdown | mistral | mistralai/Mistral-7B-v0.1 | examples/mistral/qlora.yml |\n",
|
||||
"# @markdown | openllama-3b | openlm-research/open_llama_3b_v2 | examples/openllama-3b/config.yml |\n",
|
||||
"# @markdown | openllama-3b | openlm-research/open_llama_3b_v2 | examples/openllama-3b/lora.yml |\n",
|
||||
"# @markdown | openllama-3b | openlm-research/open_llama_3b_v2 | examples/openllama-3b/qlora.yml |\n",
|
||||
"# @markdown | phi | microsoft/Phi-3.5-mini-instruct | examples/phi/lora-3.5.yaml |\n",
|
||||
"# @markdown | phi | microsoft/phi-1_5 | examples/phi/phi-ft.yml |\n",
|
||||
"# @markdown | phi | microsoft/phi-1_5 | examples/phi/phi-qlora.yml |\n",
|
||||
"# @markdown | phi | microsoft/phi-2 | examples/phi/phi2-ft.yml |\n",
|
||||
"# @markdown | phi | microsoft/Phi-3-mini-4k-instruct | examples/phi/phi3-ft.yml |\n",
|
||||
"# @markdown | qwen | Qwen/Qwen1.5-MoE-A2.7B | examples/qwen/qwen2-moe-lora.yaml |\n",
|
||||
"# @markdown | qwen | Qwen/Qwen1.5-MoE-A2.7B | examples/qwen/qwen2-moe-qlora.yaml |\n",
|
||||
"# @markdown | qwen2 | Qwen/Qwen2.5-0.5B | examples/qwen2/dpo.yaml |\n",
|
||||
"# @markdown | qwen2 | Qwen/Qwen2.5-3B | examples/qwen2/prm.yaml |\n",
|
||||
"# @markdown | qwen2 | Qwen/Qwen2-7B | examples/qwen2/qlora-fsdp.yaml |\n",
|
||||
"# @markdown | qwen3 | Qwen/Qwen3-8B | examples/qwen3/qlora-fsdp.yaml |\n",
|
||||
"# @markdown | qwen3 | Qwen/Qwen3-32B | examples/qwen3/32b-qlora.yaml |\n",
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_axolotl_qwen3_finetuning.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -226,7 +226,7 @@
|
||||
"# @title Import utility packages for fine-tuning\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.93.1'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"# Import the necessary packages.\n",
|
||||
"! rm -rf vertex-ai-samples && git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
@@ -250,9 +250,38 @@
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def run_cmd_and_check_output(\n",
|
||||
" cmd: list[str], env: dict[str, str] = None, input: str = \"\", cwd: str = None\n",
|
||||
"):\n",
|
||||
" \"\"\"Runs the given command and raises exception if the command fails.\"\"\"\n",
|
||||
" with subprocess.Popen(\n",
|
||||
" cmd,\n",
|
||||
" stdin=subprocess.PIPE,\n",
|
||||
" stdout=subprocess.PIPE,\n",
|
||||
" stderr=subprocess.STDOUT,\n",
|
||||
" text=True,\n",
|
||||
" bufsize=1,\n",
|
||||
" env=env,\n",
|
||||
" cwd=cwd,\n",
|
||||
" ) as p:\n",
|
||||
" if input:\n",
|
||||
" p.stdin.write(input)\n",
|
||||
" p.stdin.flush()\n",
|
||||
" p.stdin.close()\n",
|
||||
" for line in p.stdout:\n",
|
||||
" print(line, end=\"\", flush=True)\n",
|
||||
" if p.returncode:\n",
|
||||
" raise ValueError(\n",
|
||||
" f\"Command '{' '.join(cmd)}' execution failed with return code {p.returncode}\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"train_job = None\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"HF_TOKEN = \"\""
|
||||
"HF_TOKEN = \"\"\n",
|
||||
"WORKING_DIR = os.getcwd()\n",
|
||||
"print(f\"Current working directory for notebook: {WORKING_DIR}\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -369,7 +398,7 @@
|
||||
"source": [
|
||||
"# @title Set model to fine-tune\n",
|
||||
"# @markdown Note: This overrides Axolotl's `base_model` flag.\n",
|
||||
"HF_MODEL_ID = \"Qwen/Qwen3-32B\" # @param [\"Qwen/Qwen3-32B\", \"Qwen/Qwen3-14B\", \"Qwen/Qwen3-8B\", \"Qwen/Qwen3-4B\", \"Qwen/Qwen3-1.7B\"]"
|
||||
"HF_MODEL_ID = \"Qwen/Qwen3-32B\" # @param [\"Qwen/Qwen3-32B\", \"Qwen/Qwen3-14B\", \"Qwen/Qwen3-8B\", \"Qwen/Qwen3-4B\", \"Qwen/Qwen3-1.7B\", \"deepseek-ai/DeepSeek-R1-0528-Qwen3-8B\"]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -392,6 +421,7 @@
|
||||
"# @markdown | Qwen/Qwen3-1.7B | examples/qwen3/qlora-fsdp.yaml |\n",
|
||||
"# @markdown | Qwen/Qwen3-4B | examples/qwen3/qlora-fsdp.yaml |\n",
|
||||
"# @markdown | Qwen/Qwen3-8B | examples/qwen3/qlora-fsdp.yaml |\n",
|
||||
"# @markdown | deepseek-ai/DeepSeek-R1-0528-Qwen3-8B | examples/qwen3/qlora-fsdp.yaml |\n",
|
||||
"# @markdown | Qwen/Qwen3-14B | examples/qwen3/qlora-fsdp.yaml |\n",
|
||||
"# @markdown | Qwen/Qwen3-32B | examples/qwen3/32b-qlora.yaml |\n",
|
||||
"\n",
|
||||
@@ -399,7 +429,7 @@
|
||||
"# @markdown Alternatively, you can specify github axolotl config and override flags using `Setup Axolotl Flags` section below.\n",
|
||||
"\n",
|
||||
"# @markdown 1. Set Axolotl config source.<br>\n",
|
||||
"# @markdown For **GITHUB** as source, you can explore different Axolotl configurations in the [examples directory](https://github.com/axolotl-ai-cloud/axolotl/tree/6ba5c0ed2c42a0e069b28c83646ee5a2a6904430/examples). For `GITHUB` source, `AXOLOTL_CONFIG_PATH` should start with `examples/`. e.g. `examples/qwen3/32b-qlora.yaml`.<br>\n",
|
||||
"# @markdown For **GITHUB** as source, you can explore different Axolotl configurations in the [examples directory](https://github.com/axolotl-ai-cloud/axolotl/tree/6ba5c0ed2c42a0e069b28c83646ee5a2a6904430/examples). For `GITHUB` source, `AXOLOTL_CONFIG_PATH` should start with `examples/`. e.g. \"examples/qwen3/qlora-fsdp.yaml\".<br>\n",
|
||||
"# @markdown For **LOCAL** as source, create Axolotl config yaml file and specify correct path below. Note that, the local file will be copied to GCS bucket before running Vertex AI training job. For `LOCAL` source, `AXOLOTL_CONFIG_PATH` should be a absolute path of the config file, e.g. /content/lora.yml.<br>\n",
|
||||
"# @markdown For **GCS** as source, specify the GCS URI to the Axolotl config file. Make sure the file is accessible to service account used in the notebook. For `GCS` source, `AXOLOTL_CONFIG_PATH` should be a complete GCS URI of the config file, e.g. gs://bucket/path/to/config/file.yml.\n",
|
||||
"\n",
|
||||
@@ -424,7 +454,7 @@
|
||||
" file_content = config_path.read_text()\n",
|
||||
" axolotl_config = yaml.safe_load(file_content)\n",
|
||||
"elif AXOLOTL_SOURCE == \"GCS\":\n",
|
||||
" local_path = pathlib.Path(\"/content/tmp/axolotl_config.yml\")\n",
|
||||
" local_path = pathlib.Path(f\"{WORKING_DIR}/tmp/axolotl_config.yml\")\n",
|
||||
" common_util.download_gcs_file_to_local(AXOLOTL_CONFIG_PATH, local_path.absolute())\n",
|
||||
" file_content = local_path.read_text()\n",
|
||||
" axolotl_config = yaml.safe_load(file_content)\n",
|
||||
@@ -473,7 +503,9 @@
|
||||
"source": [
|
||||
"# @title **[Optional]** Setup dataset\n",
|
||||
"\n",
|
||||
"# @markdown This section configures the dataset used for fine-tuning. **Note: If you don't fill any of the dataset options given below, then the dataset used will be the one defined in the Axolotl config file.** You have two options to configure the dataset:\n",
|
||||
"# @markdown This section configures the dataset used for fine-tuning.\n",
|
||||
"\n",
|
||||
"# @markdown **Note: If you don't fill any of the dataset options given below, then the dataset used will be the one defined in the Axolotl config file.** You have two options to configure the dataset:\n",
|
||||
"\n",
|
||||
"# @markdown **1. Use a Hugging Face Dataset**\n",
|
||||
"# @markdown - Requires specifying the dataset name and type.\n",
|
||||
@@ -571,19 +603,28 @@
|
||||
"\n",
|
||||
"# @markdown **Training can take a long time (20+ hours) to complete depending on the model, dataset and axololt config.** You can reduce the training time by reducing the max training steps. This can be done by setting `max_steps` flag to some smaller value. Note that, this might also reduce the fine-tuned model's quality.\n",
|
||||
"\n",
|
||||
"# @markdown For example, if you want to run only single step of training, then you can set [\"--use-tensorboard=True\", \"--max_steps=1\"] in the `axolotl_flag_overrides` to achieve that.\n",
|
||||
"# @markdown For example, if you want to run only single step of training, then you can set `[\"--use-tensorboard=True\", \"--max_steps=1\"]` in the `axolotl_flag_overrides` to achieve that.\n",
|
||||
"\n",
|
||||
"axolotl_flag_overrides = [\"--use-tensorboard=True\"] # @param {type:\"raw\"}\n",
|
||||
"assert type(axolotl_flag_overrides) is list, \"axolotl_flag_overrides must be a list.\"\n",
|
||||
"\n",
|
||||
"for flag in axolotl_flag_overrides:\n",
|
||||
" if flag.startswith(\"--base_model=\"):\n",
|
||||
" raise ValueError(\"Do not override base_model flag here.\")\n",
|
||||
"axolotl_flag_overrides.append(f\"--base_model={HF_MODEL_ID}\")\n",
|
||||
"\n",
|
||||
"# Set model_id and publisher. This is required for Vertex AI fine-tuning job and Vertex AI model deployment.\n",
|
||||
"publisher = HF_MODEL_ID.split(\"/\")[0]\n",
|
||||
"model_id = HF_MODEL_ID.split(\"/\")[1]\n",
|
||||
"\n",
|
||||
"# Check if duplicate flags are passed.\n",
|
||||
"flags_seen = set()\n",
|
||||
"for flag in axolotl_flag_overrides:\n",
|
||||
" if flag in flags_seen:\n",
|
||||
" raise ValueError(f\"Duplicate flag: {flag}\")\n",
|
||||
" flags_seen.add(flag)\n",
|
||||
"\n",
|
||||
"base_model = axolotl_config[\"base_model\"]\n",
|
||||
"for overrides in axolotl_flag_overrides:\n",
|
||||
" if overrides.startswith(\"--base_model=\"):\n",
|
||||
" base_model = overrides.split(\"=\")[1]\n",
|
||||
" break\n",
|
||||
"publisher = base_model.split(\"/\")[0]\n",
|
||||
"model_id = base_model.split(\"/\")[1]\n",
|
||||
"model_id = model_id.replace(\".\", \"-\")"
|
||||
]
|
||||
},
|
||||
@@ -605,7 +646,7 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Install Axolotl And GCSFUSE\n",
|
||||
"# @title Install Axolotl And gscfuse\n",
|
||||
"# @markdown 1. Check machine type\n",
|
||||
"import subprocess\n",
|
||||
"\n",
|
||||
@@ -630,7 +671,7 @@
|
||||
"! cd axolotl && python scripts/unsloth_install.py | sh\n",
|
||||
"! cd axolotl && python scripts/cutcrossentropy_install.py | sh\n",
|
||||
"\n",
|
||||
"# @markdown 4. Install GCSFUSE\n",
|
||||
"# @markdown 4. Install gscfuse\n",
|
||||
"! apt-get install gcsfuse -y"
|
||||
]
|
||||
},
|
||||
@@ -646,21 +687,21 @@
|
||||
"# @title Run Local fine-tuning\n",
|
||||
"# @markdown This section runs the Axolotl training locally (i.e. colab runtime).\n",
|
||||
"# @markdown **Note: This section can take a long time to run. You can reduce the training time by reducing the max training steps as mentioned in `Setup Axolotl Flags` section.**\n",
|
||||
"# @markdown Model trained using Axolotl will be saved in the GCS bucket with the help of GCSFUSE.\n",
|
||||
"# @markdown Model trained using Axolotl will be saved in the GCS bucket with the help of gscfuse.\n",
|
||||
"\n",
|
||||
"# @markdown 1. Run GCSFUSE so that Axolotl can store the training output in the GCS bucket.\n",
|
||||
"# @markdown 1. Run gscfuse so that Axolotl can store the training output in the GCS bucket.\n",
|
||||
"! mkdir -p /gcs/\n",
|
||||
"! gcsfuse /gcs\n",
|
||||
"\n",
|
||||
"# @markdown 2. Set up huggingface cache dir and access token.\n",
|
||||
"os.environ[\"HF_HOME\"] = \"/content/hf\"\n",
|
||||
"os.environ[\"HF_HOME\"] = f\"{WORKING_DIR}/hf\"\n",
|
||||
"os.environ[\"HF_TOKEN\"] = HF_TOKEN\n",
|
||||
"\n",
|
||||
"# @markdown 3. Run Axolotl training.\n",
|
||||
"\n",
|
||||
"local_config_path = AXOLOTL_CONFIG_PATH\n",
|
||||
"if AXOLOTL_SOURCE == \"GITHUB\":\n",
|
||||
" local_config_path = f\"/content/axolotl/{AXOLOTL_CONFIG_PATH}\"\n",
|
||||
" local_config_path = f\"{WORKING_DIR}/axolotl/{AXOLOTL_CONFIG_PATH}\"\n",
|
||||
"finetuning_time = time.time_ns()\n",
|
||||
"AXOLOTL_OUTPUT_GCS_URI = (\n",
|
||||
" f\"{BASE_AXOLOTL_OUTPUT_GCS_URI}/local/time_ns_{finetuning_time}\"\n",
|
||||
@@ -691,11 +732,34 @@
|
||||
"source": [
|
||||
"# @title Run Local inference\n",
|
||||
"# @markdown This section performs inference using the finetuned model.\n",
|
||||
"# @markdown There are two options for inference:\n",
|
||||
"# @markdown 1. Gradio: This option provides a URL for the playground to test the model.\n",
|
||||
"# @markdown 2. CLI: This is option outputs the inference results in the console.\n",
|
||||
"\n",
|
||||
"# @markdown 1. Run Axolotl inference using gradio on local finetuned model.\n",
|
||||
"! cd axolotl && export CUDA_VISIBLE_DEVICES=0 && axolotl inference $local_config_path --output-dir=$AXOLOTL_OUTPUT_DIR --gradio\n",
|
||||
"INFERENCE_METHOD = \"gradio\" # @param [\"gradio\", \"cli\"]\n",
|
||||
"\n",
|
||||
"# @markdown 2. After running the cell, a public URL ([\"https://*.gradio.live\"](#)) will appear in the cell output. The playground is available in a separate browser tab when you click the URL."
|
||||
"# @markdown **Note: `CLI_PROMPT` will be only used if `INFERENCE_METHOD` is `cli`.**\n",
|
||||
"CLI_PROMPT = \"What is car?\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"if INFERENCE_METHOD == \"gradio\":\n",
|
||||
" ! cd axolotl && export CUDA_VISIBLE_DEVICES=0 && axolotl inference --base-model=$base_model $local_config_path --lora-model-dir=$AXOLOTL_OUTPUT_DIR --gradio\n",
|
||||
"elif INFERENCE_METHOD == \"cli\":\n",
|
||||
" assert CLI_PROMPT, \"CLI_PROMPT must be set if INFERENCE_METHOD is 'cli'.\"\n",
|
||||
" env = os.environ.copy()\n",
|
||||
" env[\"CUDA_VISIBLE_DEVICES\"] = \"0\"\n",
|
||||
" cmd = [\n",
|
||||
" \"axolotl\",\n",
|
||||
" \"inference\",\n",
|
||||
" local_config_path,\n",
|
||||
" f\"--base-model={base_model}\",\n",
|
||||
" f\"--lora-model-dir={AXOLOTL_OUTPUT_DIR}\",\n",
|
||||
" ]\n",
|
||||
" run_cmd_and_check_output(cmd, env, f\"{CLI_PROMPT}\\x04\", f\"{WORKING_DIR}/axolotl/\")\n",
|
||||
"else:\n",
|
||||
" raise ValueError(f\"Unsupported inference method: {INFERENCE_METHOD}\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# @markdown For Gradio, after running the cell, a public URL ([\"https://*.gradio.live\"](#)) will appear in the cell output. The playground is available in a separate browser tab when you click the URL."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -718,7 +782,15 @@
|
||||
" raise ValueError(\"This cell is only needed for lora and qlora.\")\n",
|
||||
"\n",
|
||||
"# @markdown 1. Run Axolotl merge. **Note: Based on model size, this step can take 5-20 minutes to complete.**\n",
|
||||
"! cd axolotl && python3 -m axolotl.cli.merge_lora $axolotl_args $AXOLOTL_CONFIG_PATH --output-dir=$AXOLOTL_OUTPUT_DIR"
|
||||
"cmd = [\n",
|
||||
" \"python3\",\n",
|
||||
" \"-m\",\n",
|
||||
" \"axolotl.cli.merge_lora\",\n",
|
||||
" f\"--base-model={base_model}\",\n",
|
||||
" f\"--output-dir={AXOLOTL_OUTPUT_DIR}\",\n",
|
||||
" local_config_path,\n",
|
||||
"]\n",
|
||||
"run_cmd_and_check_output(cmd, None, None, f\"{WORKING_DIR}/axolotl/\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -940,7 +1012,7 @@
|
||||
" elif model_id == \"Qwen3-14B\":\n",
|
||||
" machine_type = \"g2-standard-24\"\n",
|
||||
" accelerator_count = 2\n",
|
||||
" elif model_id == \"Qwen3-8B\":\n",
|
||||
" elif model_id == \"Qwen3-8B\" or model_id == \"DeepSeek-R1-0528-Qwen3-8B\":\n",
|
||||
" machine_type = \"g2-standard-12\"\n",
|
||||
" accelerator_count = 1\n",
|
||||
" elif model_id in [\"Qwen3-4B\", \"Qwen3-1-7B\"]:\n",
|
||||
@@ -1007,6 +1079,8 @@
|
||||
" dtype: str | None = None,\n",
|
||||
" enable_trust_remote_code: bool = False,\n",
|
||||
" enable_torch_compile: bool = False,\n",
|
||||
" torch_compile_max_bs: int | None = None,\n",
|
||||
" attention_backend: str = \"\",\n",
|
||||
" enable_flashinfer_mla: bool = False,\n",
|
||||
" disable_cuda_graph: bool = False,\n",
|
||||
" speculative_algorithm: str | None = None,\n",
|
||||
@@ -1054,6 +1128,11 @@
|
||||
"\n",
|
||||
" if enable_torch_compile:\n",
|
||||
" sglang_args.append(\"--enable-torch-compile\")\n",
|
||||
" if torch_compile_max_bs:\n",
|
||||
" sglang_args.append(f\"--torch-compile-max-bs={torch_compile_max_bs}\")\n",
|
||||
"\n",
|
||||
" if attention_backend:\n",
|
||||
" sglang_args.append(f\"--attention-backend={attention_backend}\")\n",
|
||||
"\n",
|
||||
" if enable_flashinfer_mla:\n",
|
||||
" sglang_args.append(\"--enable-flashinfer-mla\")\n",
|
||||
@@ -1083,6 +1162,13 @@
|
||||
" if enable_jit_deepgemm:\n",
|
||||
" env_vars[\"SGL_ENABLE_JIT_DEEPGEMM\"] = \"1\"\n",
|
||||
"\n",
|
||||
" # HF_TOKEN is not a compulsory field and may not be defined.\n",
|
||||
" try:\n",
|
||||
" if HF_TOKEN:\n",
|
||||
" env_vars[\"HF_TOKEN\"] = HF_TOKEN\n",
|
||||
" except NameError:\n",
|
||||
" pass\n",
|
||||
"\n",
|
||||
" model = aiplatform.Model.upload(\n",
|
||||
" display_name=model_name,\n",
|
||||
" serving_container_image_uri=SGLANG_DOCKER_URI,\n",
|
||||
@@ -1125,7 +1211,7 @@
|
||||
" \"maxReplicaCount\": 1,\n",
|
||||
" },\n",
|
||||
" \"system_labels\": {\n",
|
||||
" \"NOTEBOOK_NAME\": \"model_garden_axolotl_finetuning.ipynb\",\n",
|
||||
" \"NOTEBOOK_NAME\": \"model_garden_axolotl_qwen3_finetuning.ipynb\",\n",
|
||||
" \"NOTEBOOK_ENVIRONMENT\": common_util.get_deploy_source(),\n",
|
||||
" },\n",
|
||||
" },\n",
|
||||
@@ -1207,11 +1293,11 @@
|
||||
"prompt = \"<|im_start|>user What is the best way to diagnose and fix a flickering light in my house?<|im_end|><|im_start|>assistant\" # @param {type: \"string\"}\n",
|
||||
"\n",
|
||||
"# @markdown By default, Qwen3 has thinking capabilities enabled, similar to QwQ-32B. This means the model will use its reasoning abilities to enhance the quality of generated responses.\n",
|
||||
"# @markdown The model will generate think content wrapped in a \\...\\ block, followed by the final response.\n",
|
||||
"# @markdown The model will generate think content wrapped in a \\<think>...\\</think> block, followed by the final response.\n",
|
||||
"# @markdown `max_new_tokens` may need to be increased to accommodate the additional think content.\n",
|
||||
"enable_thinking = True # @param {type:\"boolean\"}\n",
|
||||
"if not enable_thinking:\n",
|
||||
" prompt += \"\"\n",
|
||||
" prompt += \"<think></think>\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"max_new_tokens = 1024 # @param {type:\"integer\"}\n",
|
||||
@@ -1249,9 +1335,7 @@
|
||||
")\n",
|
||||
"\n",
|
||||
"for prediction in response.predictions:\n",
|
||||
" print(prediction)\n",
|
||||
"\n",
|
||||
"# @markdown Click \"Show Code\" to see more details."
|
||||
" print(prediction)"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -38,7 +38,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_camp_zipnerf.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -38,7 +38,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_camp_zipnerf_gradio.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_codegemma_deployment_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_deployment_tutorial.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_deployment_tutorial.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -107,7 +107,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"import os\n",
|
||||
@@ -166,7 +166,7 @@
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Choose the model to deploy\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"# @markdown List all deployable models and then get the ID of the model to deploy.\n",
|
||||
"\n",
|
||||
@@ -222,14 +222,16 @@
|
||||
"# @title Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(MODEL_ID)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
" hugging_face_access_token=HF_TOKEN,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -4,11 +4,12 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "7d9bbf86da5e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2024 Google LLC\n",
|
||||
"# Copyright 2025 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -33,13 +34,18 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_e5.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/colab/import/https:%2F%2Fraw.githubusercontent.com%2FGoogleCloudPlatform%2Fvertex-ai-samples%2Fmain%2Fnotebooks%2Fcommunity%2Fmodel_garden%2Fmodel_garden_e5.ipynb\">\n",
|
||||
" <img alt=\"Google Cloud Colab Enterprise logo\" src=\"https://lh3.googleusercontent.com/JmcxdQi-qOpctIvWKgPtrzZdJJK-J3sWE1RsfjZNwshCFgE_9fULcNpuXYTilIR2hjwN\" width=\"32px\"><br> Run in Colab Enterprise\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_e5.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -67,6 +73,10 @@
|
||||
"- Run inference on the deployed Vertex AI Endpoint\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"### File a bug\n",
|
||||
"\n",
|
||||
"File a bug on [GitHub](https://github.com/GoogleCloudPlatform/vertex-ai-samples/issues/new) if you encounter any issue with the notebook.\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
@@ -74,7 +84,7 @@
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing), [Cloud NL API pricing](https://cloud.google.com/natural-language/pricing) and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -99,133 +109,61 @@
|
||||
"\n",
|
||||
"# @markdown 1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"# @markdown 2. [Optional] [Create a Cloud Storage bucket](https://cloud.google.com/storage/docs/creating-buckets) for storing experiment outputs. Set the BUCKET_URI for the experiment environment. The specified Cloud Storage bucket (`BUCKET_URI`) should be located in the same region as where the notebook was launched. Note that a multi-region bucket (eg. \"us\") is not considered a match for a single region covered by the multi-region range (eg. \"us-central1\"). If not set, a unique GCS bucket will be created instead.\n",
|
||||
"# @markdown 2. **[Optional]** Set region. If not set, the region will be set automatically according to Colab Enterprise environment.\n",
|
||||
"\n",
|
||||
"REGION = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# @markdown 3. If you want to run predictions with A100 80GB or H100 GPUs, we recommend using the regions listed below. **NOTE:** Make sure you have associated quota in selected regions. Click the links to see your current quota for each GPU type: [Nvidia A100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_a100_80gb_gpus), [Nvidia H100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_h100_gpus). You can request for quota following the instructions at [\"Request a higher quota\"](https://cloud.google.com/docs/quota/view-manage#requesting_higher_quota).\n",
|
||||
"\n",
|
||||
"# @markdown > | Machine Type | Accelerator Type | Recommended Regions |\n",
|
||||
"# @markdown | ----------- | ----------- | ----------- |\n",
|
||||
"# @markdown | a2-ultragpu-1g | 1 NVIDIA_A100_80GB | us-central1, us-east4, europe-west4, asia-southeast1, us-east4 |\n",
|
||||
"# @markdown | a3-highgpu-2g | 2 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-4g | 4 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"! pip3 install --quiet torchvision\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
"import uuid\n",
|
||||
"from datetime import datetime\n",
|
||||
"from typing import Tuple\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"from torch import Tensor\n",
|
||||
"\n",
|
||||
"if os.environ.get(\"VERTEX_PRODUCT\") != \"COLAB_ENTERPRISE\":\n",
|
||||
" ! pip install --upgrade tensorflow\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"common_util = importlib.import_module(\n",
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
"PROJECT_ID = os.environ[\"GOOGLE_CLOUD_PROJECT\"]\n",
|
||||
"\n",
|
||||
"# Get the default region for launching jobs.\n",
|
||||
"REGION = os.environ[\"GOOGLE_CLOUD_REGION\"]\n",
|
||||
"\n",
|
||||
"# Cloud Storage bucket for storing the experiment artifacts.\n",
|
||||
"# A unique GCS bucket will be created for the purpose of this notebook. If you\n",
|
||||
"# prefer using your own GCS bucket, change the value yourself below.\n",
|
||||
"now = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n",
|
||||
"BUCKET_URI = \"gs://\" # @param {type: \"string\"}\n",
|
||||
"assert BUCKET_URI.startswith(\"gs://\"), \"BUCKET_URI must start with `gs://`.\"\n",
|
||||
"\n",
|
||||
"# @markdown Click \"Show code\" to see more details.\n",
|
||||
"\n",
|
||||
"# Create a unique GCS bucket for this notebook, if not specified by the user.\n",
|
||||
"assert BUCKET_URI.startswith(\"gs://\"), \"BUCKET_URI must start with `gs://`.\"\n",
|
||||
"if BUCKET_URI is None or BUCKET_URI.strip() == \"\" or BUCKET_URI == \"gs://\":\n",
|
||||
" BUCKET_URI = f\"gs://{PROJECT_ID}-tmp-{now}-{str(uuid.uuid4())[:4]}\"\n",
|
||||
" ! gsutil mb -l {REGION} {BUCKET_URI}\n",
|
||||
" BUCKET_NAME = \"/\".join(BUCKET_URI.split(\"/\")[:3])\n",
|
||||
"else:\n",
|
||||
" BUCKET_NAME = \"/\".join(BUCKET_URI.split(\"/\")[:3])\n",
|
||||
" shell_output = ! gsutil ls -Lb {BUCKET_NAME} | grep \"Location constraint:\" | sed \"s/Location constraint://\"\n",
|
||||
" bucket_region = shell_output[0].strip().lower()\n",
|
||||
" if bucket_region != REGION:\n",
|
||||
" raise ValueError(\n",
|
||||
" \"Bucket region %s is different from notebook region %s\"\n",
|
||||
" % (bucket_region, REGION)\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"print(f\"Using this GCS Bucket: {BUCKET_URI}\")\n",
|
||||
"if not REGION:\n",
|
||||
" REGION = os.environ[\"GOOGLE_CLOUD_REGION\"]\n",
|
||||
"\n",
|
||||
"# Initialize Vertex AI API.\n",
|
||||
"print(\"Initializing Vertex AI API.\")\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)\n",
|
||||
"\n",
|
||||
"! gcloud config set project $PROJECT_ID\n",
|
||||
"! gcloud services enable language.googleapis.com\n",
|
||||
"import vertexai\n",
|
||||
"\n",
|
||||
"STAGING_BUCKET = os.path.join(BUCKET_URI, \"temporal\")\n",
|
||||
"\n",
|
||||
"# Gets the default BUCKET_URI and SERVICE_ACCOUNT if they were not specified by the user.\n",
|
||||
"\n",
|
||||
"SERVICE_ACCOUNT = None\n",
|
||||
"shell_output = ! gcloud projects describe $PROJECT_ID\n",
|
||||
"project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
|
||||
"SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
|
||||
"\n",
|
||||
"print(\"Using this default Service Account:\", SERVICE_ACCOUNT)\n",
|
||||
"\n",
|
||||
"# Provision permissions to the SERVICE_ACCOUNT with the GCS bucket\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.admin $BUCKET_NAME\n",
|
||||
"\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=STAGING_BUCKET)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def create_name_with_datetime(prefix: str) -> str:\n",
|
||||
" \"\"\"Creates a name with date time when triggering training or deployment\n",
|
||||
" jobs in Vertex AI.\n",
|
||||
" \"\"\"\n",
|
||||
" return prefix + datetime.now().strftime(\"_%Y%m%d_%H%M%S\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def deploy_model_tei(\n",
|
||||
" model_name: str,\n",
|
||||
" model_id: str,\n",
|
||||
" service_account: str,\n",
|
||||
" docker_uri: str,\n",
|
||||
" machine_type: str = \"g2-standard-8\",\n",
|
||||
" accelerator_type: str = \"NVIDIA_L4\",\n",
|
||||
" accelerator_count: int = 1,\n",
|
||||
" max_model_len: int = 512,\n",
|
||||
" gpu_memory_utilization: float = 0.9,\n",
|
||||
") -> Tuple[aiplatform.Model, aiplatform.Endpoint]:\n",
|
||||
" \"\"\"Deploys E5 models with TEI on Vertex AI.\n",
|
||||
"\n",
|
||||
" Args:\n",
|
||||
" model_name: Display name of the model.\n",
|
||||
" model_id: Model ID or path to model weights.\n",
|
||||
" service_account: Service account for model uploading and deployment.\n",
|
||||
" machine_type: Deployment machine type.\n",
|
||||
" accelerator_type: Deployment accelerator type.\n",
|
||||
" accelerator_count: Number of accelerators to use.\n",
|
||||
" max_model_len: Maximum model length.\n",
|
||||
" gpu_memory_utilization: Fraction of GPU memory to be used for the model\n",
|
||||
" executor.\n",
|
||||
"\n",
|
||||
" Returns:\n",
|
||||
" Model instance and endpoint instance.\n",
|
||||
" \"\"\"\n",
|
||||
" endpoint = aiplatform.Endpoint.create(display_name=f\"{model_name}-endpoint\")\n",
|
||||
"\n",
|
||||
" tei_args = [\n",
|
||||
" f\"--model-id={model_id}\",\n",
|
||||
" ]\n",
|
||||
" serving_env = {\n",
|
||||
" \"MODEL_ID\": model_id,\n",
|
||||
" \"DEPLOY_SOURCE\": \"notebook\",\n",
|
||||
" }\n",
|
||||
" model = aiplatform.Model.upload(\n",
|
||||
" display_name=model_name,\n",
|
||||
" serving_container_image_uri=docker_uri,\n",
|
||||
" serving_container_args=tei_args,\n",
|
||||
" serving_container_ports=[7080],\n",
|
||||
" serving_container_environment_variables=serving_env,\n",
|
||||
" model_garden_source_model_name=\"publishers/intfloat/models/e5\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" model.deploy(\n",
|
||||
" endpoint=endpoint,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" deploy_request_timeout=1800,\n",
|
||||
" service_account=service_account,\n",
|
||||
" system_labels={\n",
|
||||
" \"NOTEBOOK_NAME\": \"model_garden_e5.ipynb\"\n",
|
||||
" },\n",
|
||||
" )\n",
|
||||
" return model, endpoint"
|
||||
"vertexai.init(\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -241,16 +179,19 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "I1u2FLa9XgVD"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Deploy\n",
|
||||
"# @title Select the model variants\n",
|
||||
"# @markdown This section uploads a prebuilt model to Model Registry and deploys it on the Endpoint. The model deployment step will take ~15 minutes to complete.\n",
|
||||
"\n",
|
||||
"prebuilt_model_id = \"intfloat/e5-small-v2\" # @param [\"intfloat/multilingual-e5-large-instruct\", \"intfloat/multilingual-e5-large\", \"intfloat/e5-large-v2\", \"intfloat/multilingual-e5-small\", \"intfloat/e5-base-v2\", \"intfloat/e5-small-v2\"]\n",
|
||||
"\n",
|
||||
"# @markdown Specify a processor for the TEI docker image. E5 models can be run on either GPU or CPU.\n",
|
||||
"# Find Vertex AI prediction supported accelerators and regions in\n",
|
||||
"# https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
|
||||
"processor = \"NVIDIA_L4\" # @param[\"NVIDIA_TESLA_T4\", \"NVIDIA_TESLA_V100\", \"NVIDIA_L4\", \"NVIDIA_TESLA_A100\", \"CPU\"]\n",
|
||||
"\n",
|
||||
"if processor == \"NVIDIA_TESLA_T4\":\n",
|
||||
@@ -283,24 +224,108 @@
|
||||
"else:\n",
|
||||
" TEI_DOCKER_URI = \"us-docker.pkg.dev/deeplearning-platform-release/gcr.io/huggingface-text-embeddings-inference-cu122.1-2.ubuntu2204\"\n",
|
||||
"\n",
|
||||
"# @markdown Click \"Show code\" to see more details.\n",
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"# Finds Vertex AI prediction supported accelerators and regions in\n",
|
||||
"# https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
|
||||
"# @markdown Click \"Show code\" to see more details."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "6dY6_ppyObQy"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Deploy model using custom configuration\n",
|
||||
"# @markdown This section uploads prebuilt E5 models to Model Registry and deploys it to a Vertex AI Endpoint. It might take ~15 minutes to 1 hour to finish depending on the size of the model.\n",
|
||||
"\n",
|
||||
"model, endpoint = deploy_model_tei(\n",
|
||||
" model_name=create_name_with_datetime(prefix=\"e5-serve-tei\"),\n",
|
||||
"\n",
|
||||
"def deploy_model_tei(\n",
|
||||
" model_name: str,\n",
|
||||
" model_id: str,\n",
|
||||
" docker_uri: str,\n",
|
||||
" machine_type: str = \"g2-standard-8\",\n",
|
||||
" accelerator_type: str = \"NVIDIA_L4\",\n",
|
||||
" accelerator_count: int = 1,\n",
|
||||
" max_model_len: int = 512,\n",
|
||||
" gpu_memory_utilization: float = 0.9,\n",
|
||||
" use_dedicated_endpoint: bool = True,\n",
|
||||
") -> Tuple[aiplatform.Model, aiplatform.Endpoint]:\n",
|
||||
" \"\"\"Deploys E5 models with TEI on Vertex AI.\n",
|
||||
"\n",
|
||||
" Args:\n",
|
||||
" model_name: Display name of the model.\n",
|
||||
" model_id: Model ID or path to model weights.\n",
|
||||
" machine_type: Deployment machine type.\n",
|
||||
" accelerator_type: Deployment accelerator type.\n",
|
||||
" accelerator_count: Number of accelerators to use.\n",
|
||||
" max_model_len: Maximum model length.\n",
|
||||
" gpu_memory_utilization: Fraction of GPU memory to be used for the model\n",
|
||||
" executor.\n",
|
||||
" use_dedicated_endpoint: A dedicated endpoint is an endpoint for online\n",
|
||||
" prediction,provide a secure connection for private communication\n",
|
||||
" between on-premises and Google Cloud.\n",
|
||||
"\n",
|
||||
" Returns:\n",
|
||||
" Model instance and endpoint instance.\n",
|
||||
" \"\"\"\n",
|
||||
" endpoint = aiplatform.Endpoint.create(\n",
|
||||
" display_name=f\"{model_name}-endpoint\",\n",
|
||||
" dedicated_endpoint_enabled=use_dedicated_endpoint,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" tei_args = [\n",
|
||||
" f\"--model-id={model_id}\",\n",
|
||||
" ]\n",
|
||||
" serving_env = {\n",
|
||||
" \"MODEL_ID\": model_id,\n",
|
||||
" \"DEPLOY_SOURCE\": \"notebook\",\n",
|
||||
" }\n",
|
||||
" model = aiplatform.Model.upload(\n",
|
||||
" display_name=model_name,\n",
|
||||
" serving_container_image_uri=docker_uri,\n",
|
||||
" serving_container_args=tei_args,\n",
|
||||
" serving_container_ports=[7080],\n",
|
||||
" serving_container_environment_variables=serving_env,\n",
|
||||
" model_garden_source_model_name=\"publishers/intfloat/models/e5\",\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" model.deploy(\n",
|
||||
" endpoint=endpoint,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" deploy_request_timeout=1800,\n",
|
||||
" system_labels={\"NOTEBOOK_NAME\": \"model_garden_e5.ipynb\"},\n",
|
||||
" )\n",
|
||||
" return model, endpoint\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"common_util.check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
" region=REGION,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" is_for_training=False,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"LABEL = \"tei\"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model_tei(\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=\"e5-serve-tei\"),\n",
|
||||
" model_id=prebuilt_model_id,\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
" docker_uri=TEI_DOCKER_URI,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(\"endpoint_name:\", endpoint.name)\n",
|
||||
"print(\"model_name:\", model.display_name)\n",
|
||||
"print(\"model_id:\", model.resource_name)"
|
||||
"model = models[LABEL]\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -358,7 +383,6 @@
|
||||
"# )\n",
|
||||
"# endpoint = aiplatform.Endpoint(aip_endpoint_name)\n",
|
||||
"\n",
|
||||
"from torch import Tensor\n",
|
||||
"\n",
|
||||
"# Each input text should start with \"query: \" or \"passage: \".\n",
|
||||
"# For tasks other than retrieval, you can simply use the \"query: \" prefix.\n",
|
||||
@@ -372,7 +396,9 @@
|
||||
" ],\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"response = endpoint.predict(instances=instances)\n",
|
||||
"response = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"for prediction in response.predictions:\n",
|
||||
" embeddings = Tensor(prediction)\n",
|
||||
@@ -400,9 +426,6 @@
|
||||
"# @markdown Instruct: Given a web search query, retrieve relevant passages that answer the query\n",
|
||||
"# @markdown Query: how much protein should a female eat\n",
|
||||
"# @markdown Instruct: Given a web search query, retrieve relevant passages that answer the query\n",
|
||||
"# @markdown Query: 南瓜的家常做法\n",
|
||||
"# @markdown As a general guideline, the CDC's average requirement of protein for women ages 19 to 70 is 46 grams per day. But, as you can see from this chart, you'll need to increase that if you're expecting or training for a marathon. Check out the chart below to see how much protein you should be eating each day.\n",
|
||||
"# @markdown 1.清炒南瓜丝 原料:嫩南瓜半个 调料:葱、盐、白糖、鸡精 做法: 1、南瓜用刀薄薄的削去表面一层皮,用勺子刮去瓤 2、擦成细丝(没有擦菜板就用刀慢慢切成细丝) 3、锅烧热放油,入葱花煸出香味 4、入南瓜丝快速翻炒一分钟左右,放盐、一点白糖和鸡精调味出锅 2.香葱炒南瓜 原料:南瓜1只 调料:香葱、蒜末、橄榄油、盐 做法: 1、将南瓜去皮,切成片 2、油锅8成热后,将蒜末放入爆香 3、爆香后,将南瓜片放入,翻炒 4、在翻炒的同时,可以不时地往锅里加水,但不要太多 5、放入盐,炒匀 6、南瓜差不多软和绵了之后,就可以关火 7、撒入香葱,即可出锅\n",
|
||||
"# @markdown ```\n",
|
||||
"\n",
|
||||
"# @markdown API reference link to HuggingFace : [Text Embeddings Inference API](https://huggingface.github.io/text-embeddings-inference/#/).\n",
|
||||
@@ -431,8 +454,6 @@
|
||||
"# )\n",
|
||||
"# endpoint = aiplatform.Endpoint(aip_endpoint_name)\n",
|
||||
"\n",
|
||||
"from torch import Tensor\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_detailed_instruct(task_description: str, query: str) -> str:\n",
|
||||
" return f\"Instruct: {task_description}\\nQuery: {query}\"\n",
|
||||
@@ -451,7 +472,9 @@
|
||||
"\n",
|
||||
"instances = [{\"inputs\": queries + documents}]\n",
|
||||
"\n",
|
||||
"response = endpoint.predict(instances=instances)\n",
|
||||
"response = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"for prediction in response.predictions:\n",
|
||||
" embeddings = Tensor(prediction)\n",
|
||||
@@ -482,12 +505,13 @@
|
||||
"# @markdown Delete the experiment models and endpoints to recycle the resources\n",
|
||||
"# @markdown and avoid unnecessary continuous charges that may incur.\n",
|
||||
"\n",
|
||||
"endpoint.delete(force=True)\n",
|
||||
"model.delete()\n",
|
||||
"# Undeploy model and delete endpoint.\n",
|
||||
"for endpoint in endpoints.values():\n",
|
||||
" endpoint.delete(force=True)\n",
|
||||
"\n",
|
||||
"delete_bucket = False # @param {type:\"boolean\"}\n",
|
||||
"if delete_bucket:\n",
|
||||
" ! gsutil -m rm -r $BUCKET_URI"
|
||||
"# Delete models.\n",
|
||||
"for model in models.values():\n",
|
||||
" model.delete()"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" \u003c/td\u003e\n",
|
||||
" \u003ctd style=\"text-align: center\"\u003e\n",
|
||||
" \u003ca href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_finetuning_tutorial.ipynb\"\u003e\n",
|
||||
" \u003cimg alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"\u003e\u003cbr\u003e View on GitHub\n",
|
||||
" \u003cimg alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"\u003e\u003cbr\u003e View on GitHub\n",
|
||||
" \u003c/a\u003e\n",
|
||||
" \u003c/td\u003e\n",
|
||||
"\u003c/tr\u003e\u003c/tbody\u003e\u003c/table\u003e"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_gemma2_deployment_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gemma2_deployment_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -133,7 +133,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
@@ -554,7 +554,7 @@
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -563,7 +563,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -36,7 +36,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_gemma2_finetuning_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -47,7 +47,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gemma2_finetuning_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_gemma3_deployment_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gemma3_deployment_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -116,7 +116,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"import importlib\n",
|
||||
@@ -134,8 +134,9 @@
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Initialize models and endpoints as a dict\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"LABEL = \"vllm_gpu\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
"PROJECT_ID = os.environ[\"GOOGLE_CLOUD_PROJECT\"]\n",
|
||||
@@ -223,8 +224,9 @@
|
||||
"source": [
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"LABEL = \"sdk-deploy-1b\"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -233,7 +235,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -378,7 +382,9 @@
|
||||
" return model, endpoint\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"models[\"vllm_gpu\"], endpoints[\"vllm_gpu\"] = deploy_model_vllm(\n",
|
||||
"LABEL = \"custom-deploy-1b\"\n",
|
||||
"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model_vllm(\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=\"gemma3-serve\"),\n",
|
||||
" model_id=model_id,\n",
|
||||
" publisher=\"google\",\n",
|
||||
@@ -390,7 +396,10 @@
|
||||
" gpu_memory_utilization=gpu_memory_utilization,\n",
|
||||
" max_model_len=max_model_len,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"model = models[LABEL]\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -448,7 +457,7 @@
|
||||
" \"raw_response\": raw_response,\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"response = endpoints[\"vllm_gpu\"].predict(\n",
|
||||
"response = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"\n",
|
||||
@@ -470,8 +479,8 @@
|
||||
"# @title Chat completion\n",
|
||||
"\n",
|
||||
"if use_dedicated_endpoint:\n",
|
||||
" DEDICATED_ENDPOINT_DNS = endpoints[\"vllm_gpu\"].gca_resource.dedicated_endpoint_dns\n",
|
||||
"ENDPOINT_RESOURCE_NAME = endpoints[\"vllm_gpu\"].resource_name\n",
|
||||
" DEDICATED_ENDPOINT_DNS = endpoint.gca_resource.dedicated_endpoint_dns\n",
|
||||
"ENDPOINT_RESOURCE_NAME = endpoint.resource_name\n",
|
||||
"\n",
|
||||
"# @title Chat Completions Inference\n",
|
||||
"\n",
|
||||
@@ -598,8 +607,9 @@
|
||||
"source": [
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"LABEL = \"sdk-deploy\"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -608,7 +618,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -627,6 +639,8 @@
|
||||
"gpu_memory_utilization = 0.95\n",
|
||||
"max_model_len = 131072\n",
|
||||
"\n",
|
||||
"LABEL = \"multimodal-deploy\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def deploy_model_vllm(\n",
|
||||
" model_name: str,\n",
|
||||
@@ -753,7 +767,7 @@
|
||||
" return model, endpoint\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"models[\"vllm_gpu\"], endpoints[\"vllm_gpu\"] = deploy_model_vllm(\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model_vllm(\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=\"gemma3-serve\"),\n",
|
||||
" model_id=model_id,\n",
|
||||
" publisher=\"google\",\n",
|
||||
@@ -765,7 +779,10 @@
|
||||
" gpu_memory_utilization=gpu_memory_utilization,\n",
|
||||
" max_model_len=max_model_len,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"model = models[LABEL]\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -780,8 +797,8 @@
|
||||
"# @title Chat completion with text-only requests\n",
|
||||
"\n",
|
||||
"if use_dedicated_endpoint:\n",
|
||||
" DEDICATED_ENDPOINT_DNS = endpoints[\"vllm_gpu\"].gca_resource.dedicated_endpoint_dns\n",
|
||||
"ENDPOINT_RESOURCE_NAME = endpoints[\"vllm_gpu\"].resource_name\n",
|
||||
" DEDICATED_ENDPOINT_DNS = endpoint.gca_resource.dedicated_endpoint_dns\n",
|
||||
"ENDPOINT_RESOURCE_NAME = endpoint.resource_name\n",
|
||||
"\n",
|
||||
"# @title Chat Completions Inference\n",
|
||||
"\n",
|
||||
@@ -855,8 +872,8 @@
|
||||
"# @title Chat completion with multimodal requests\n",
|
||||
"\n",
|
||||
"if use_dedicated_endpoint:\n",
|
||||
" DEDICATED_ENDPOINT_DNS = endpoints[\"vllm_gpu\"].gca_resource.dedicated_endpoint_dns\n",
|
||||
"ENDPOINT_RESOURCE_NAME = endpoints[\"vllm_gpu\"].resource_name\n",
|
||||
" DEDICATED_ENDPOINT_DNS = endpoint.gca_resource.dedicated_endpoint_dns\n",
|
||||
"ENDPOINT_RESOURCE_NAME = endpoint.resource_name\n",
|
||||
"\n",
|
||||
"# @title Chat Completions Inference\n",
|
||||
"\n",
|
||||
|
||||
@@ -3,7 +3,6 @@
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"language": "python",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "7d9bbf86da5e"
|
||||
@@ -27,7 +26,6 @@
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"language": "markdown",
|
||||
"metadata": {
|
||||
"id": "99c1c3fc2ca5"
|
||||
},
|
||||
@@ -36,13 +34,18 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_gemma3_finetuning_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/colab/import/https:%2F%2Fraw.githubusercontent.com%2FGoogleCloudPlatform%2Fvertex-ai-samples%2Fmain%2Fnotebooks%2Fcommunity%2Fmodel_garden%2Fmodel_garden_gemma3_finetuning_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"Google Cloud Colab Enterprise logo\" src=\"https://lh3.googleusercontent.com/JmcxdQi-qOpctIvWKgPtrzZdJJK-J3sWE1RsfjZNwshCFgE_9fULcNpuXYTilIR2hjwN\" width=\"32px\"><br> Run in Colab Enterprise\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gemma3_finetuning_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -112,9 +115,8 @@
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"language": "python",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"cellView": "code",
|
||||
"id": "855d6b96f291"
|
||||
},
|
||||
"outputs": [],
|
||||
@@ -156,12 +158,19 @@
|
||||
"from google.cloud.aiplatform.compat.types import \\\n",
|
||||
" custom_job as gca_custom_job_compat\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"if os.environ.get(\"VERTEX_PRODUCT\") != \"COLAB_ENTERPRISE\":\n",
|
||||
" ! pip install --upgrade tensorflow\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"common_util = importlib.import_module(\n",
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Initialize models and endpoints as a dict\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
"PROJECT_ID = os.environ[\"GOOGLE_CLOUD_PROJECT\"]\n",
|
||||
"\n",
|
||||
@@ -390,7 +399,6 @@
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"language": "python",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "ivVGS9dHXPOz"
|
||||
@@ -717,7 +725,6 @@
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"language": "python",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "qmHW6m8xG_4U"
|
||||
@@ -905,6 +912,9 @@
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model = models[\"vllm_gpu\"]\n",
|
||||
"endpoint = endpoints[\"vllm_gpu\"]\n",
|
||||
"\n",
|
||||
"# @markdown Click \"Show code\" to see more details."
|
||||
]
|
||||
},
|
||||
@@ -963,7 +973,7 @@
|
||||
" \"raw_response\": raw_response,\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"response = endpoints[\"vllm_gpu\"].predict(\n",
|
||||
"response = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gemma_deployment_on_gke.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gemma_deployment_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -41,7 +41,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gemma_evaluation.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -42,7 +42,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gemma_finetuning_on_vertex.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_gradio_streaming_chat_completions.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_hexllm_deep_dive_tutorial.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_hf_paligemma2_deployment.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_hf_paligemma2_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -118,7 +118,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"# Import the necessary packages\n",
|
||||
@@ -228,7 +228,7 @@
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -237,7 +237,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_huggingface_local_inference.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
+1
-1
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_huggingface_pytorch_inference_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_huggingface_tei_deployment.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_huggingface_tei_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -106,7 +106,7 @@
|
||||
"\n",
|
||||
"REGION = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
@@ -193,7 +193,7 @@
|
||||
"accelerator_count = 1\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(HUGGING_FACE_MODEL_ID)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -203,7 +203,9 @@
|
||||
" hugging_face_access_token=HF_TOKEN,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_huggingface_tgi_deployment.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_huggingface_tgi_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -110,7 +110,7 @@
|
||||
"# @markdown | a3-highgpu-2g | 2 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-4g | 4 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"# @markdown 5. You must agree to the license on the [model card](https://huggingface.co/google/gemma-2-2b-it) before accessing the Gemma 2 models.\n",
|
||||
"\n",
|
||||
@@ -202,7 +202,7 @@
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(HUGGING_FACE_MODEL_ID)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -213,6 +213,8 @@
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]\n",
|
||||
"# @markdown Click \"Show Code\" to see more details."
|
||||
]
|
||||
},
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_huggingface_vllm_deployment.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_huggingface_vllm_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -113,7 +113,7 @@
|
||||
"\n",
|
||||
"# @markdown 4. Follow the [Hugging Face documentation](https://huggingface.co/docs/hub/en/security-tokens) to create a **read** access token and put it in the `HF_TOKEN` field below.\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
@@ -222,7 +222,7 @@
|
||||
" is_for_training=False,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(HUGGING_FACE_MODEL_ID)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -233,6 +233,8 @@
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]\n",
|
||||
"# @markdown Click \"Show Code\" to see more details."
|
||||
]
|
||||
},
|
||||
|
||||
@@ -1,588 +0,0 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "20qcPG1PmFUM"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2025 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
"# You may obtain a copy of the License at\n",
|
||||
"#\n",
|
||||
"# https://www.apache.org/licenses/LICENSE-2.0\n",
|
||||
"#\n",
|
||||
"# Unless required by applicable law or agreed to in writing, software\n",
|
||||
"# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
|
||||
"# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
|
||||
"# See the License for the specific language governing permissions and\n",
|
||||
"# limitations under the License."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "QXYOa1odnikj"
|
||||
},
|
||||
"source": [
|
||||
"# Vertex AI Model Garden Integration With ADK\n",
|
||||
"\n",
|
||||
"\u003ctable\u003e\u003ctbody\u003e\u003ctr\u003e\n",
|
||||
" \u003ctd style=\"text-align: center\"\u003e\n",
|
||||
" \u003ca href=\"https://console.cloud.google.com/vertex-ai/colab/import/https:%2F%2Fraw.githubusercontent.com%2FGoogleCloudPlatform%2Fvertex-ai-samples%2Fmain%2Fnotebooks%2Fcommunity%2Fmodel_garden%2Fmodel_garden_integration_with_adk.ipynb\"\u003e\n",
|
||||
" \u003cimg alt=\"Google Cloud Colab Enterprise logo\" src=\"https://lh3.googleusercontent.com/JmcxdQi-qOpctIvWKgPtrzZdJJK-J3sWE1RsfjZNwshCFgE_9fULcNpuXYTilIR2hjwN\" width=\"32px\"\u003e\u003cbr\u003e Run in Colab Enterprise\n",
|
||||
" \u003c/a\u003e\n",
|
||||
" \u003c/td\u003e\n",
|
||||
" \u003ctd style=\"text-align: center\"\u003e\n",
|
||||
" \u003ca href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_integration_with_adk.ipynb\"\u003e\n",
|
||||
" \u003cimg alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"\u003e\u003cbr\u003e View on GitHub\n",
|
||||
" \u003c/a\u003e\n",
|
||||
" \u003c/td\u003e\n",
|
||||
"\u003c/tr\u003e\u003c/tbody\u003e\u003c/table\u003e"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "cbDI9ag4oR4C"
|
||||
},
|
||||
"source": [
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"This notebook demonstrates how to build an agent with a Model Garden open model deployed through Vertex endpoint and [Google Agent Development Kit](https://google.github.io/adk-docs/) (ADK). The agent can automatically call function tools like `get_weather` and `get_current_time`.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"- Deploy Vertex AI Model Garden OSS LLMs properly for ADK integration\n",
|
||||
"- Test deployed endpoints\n",
|
||||
"- Build agent web apps with deployed endpoints and ADK\n",
|
||||
"- Deploy agent web apps to Cloud Run\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
"\n",
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "hQJWRopioSKT"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "J_jmxcIZoSxU",
|
||||
"cellView": "form"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Setup Google Cloud project\n",
|
||||
"\n",
|
||||
"# @markdown 1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"# @markdown 2. **[Optional]** Set region. If not set, the region will be set automatically according to Colab Enterprise environment.\n",
|
||||
"\n",
|
||||
"REGION = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# @markdown 3. If you want to run predictions with A100 80GB or H100 GPUs, we recommend using the regions listed below. **NOTE:** Make sure you have associated quota in selected regions. Click the links to see your current quota for each GPU type: [Nvidia A100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_a100_80gb_gpus), [Nvidia H100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_h100_gpus). You can request for quota following the instructions at [\"Request a higher quota\"](https://cloud.google.com/docs/quota/view-manage#requesting_higher_quota).\n",
|
||||
"\n",
|
||||
"# @markdown \u003e | Machine Type | Accelerator Type | Recommended Regions |\n",
|
||||
"# @markdown | ----------- | ----------- | ----------- |\n",
|
||||
"# @markdown | a2-ultragpu-1g | 1 NVIDIA_A100_80GB | us-central1, us-east4, europe-west4, asia-southeast1, us-east4 |\n",
|
||||
"# @markdown | a3-highgpu-2g | 2 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-4g | 4 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform\u003e=1.84.0'\n",
|
||||
"! pip install -qU openai google-auth requests\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"import datetime\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
"import uuid\n",
|
||||
"from typing import Tuple\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"\n",
|
||||
"if not os.path.exists(\"./vertex-ai-samples\"):\n",
|
||||
" ! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"\n",
|
||||
"common_util = importlib.import_module(\n",
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
"PROJECT_ID = os.environ[\"GOOGLE_CLOUD_PROJECT\"]\n",
|
||||
"\n",
|
||||
"# Get the default region for launching jobs.\n",
|
||||
"if not REGION:\n",
|
||||
" if not os.environ.get(\"GOOGLE_CLOUD_REGION\"):\n",
|
||||
" raise ValueError(\n",
|
||||
" \"REGION must be set. See\"\n",
|
||||
" \" https://cloud.google.com/vertex-ai/docs/general/locations for\"\n",
|
||||
" \" available cloud locations.\"\n",
|
||||
" )\n",
|
||||
" REGION = os.environ[\"GOOGLE_CLOUD_REGION\"]\n",
|
||||
"\n",
|
||||
"# Enable the Vertex AI API and Compute Engine API, if not already.\n",
|
||||
"print(\"Enabling Vertex AI API and Compute Engine API.\")\n",
|
||||
"! gcloud services enable aiplatform.googleapis.com compute.googleapis.com\n",
|
||||
"\n",
|
||||
"# Initialize Vertex AI API.\n",
|
||||
"print(\"Initializing Vertex AI API.\")\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)\n",
|
||||
"\n",
|
||||
"# Gets the default SERVICE_ACCOUNT.\n",
|
||||
"shell_output = ! gcloud projects describe $PROJECT_ID\n",
|
||||
"project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
|
||||
"SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
|
||||
"print(\"Using this default Service Account:\", SERVICE_ACCOUNT)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Provision permissions to the SERVICE_ACCOUNT with the GCS bucket\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.admin $BUCKET_NAME\n",
|
||||
"\n",
|
||||
"! gcloud config set project $PROJECT_ID\n",
|
||||
"! gcloud projects add-iam-policy-binding --no-user-output-enabled {PROJECT_ID} --member=serviceAccount:{SERVICE_ACCOUNT} --role=\"roles/storage.admin\"\n",
|
||||
"! gcloud projects add-iam-policy-binding --no-user-output-enabled {PROJECT_ID} --member=serviceAccount:{SERVICE_ACCOUNT} --role=\"roles/aiplatform.user\"\n",
|
||||
"\n",
|
||||
"from etils import epath\n",
|
||||
"TUTORIAL_DIR = epath.Path(\"vmg_adk_agent_tutorial\")\n",
|
||||
"BUILD_DIR = TUTORIAL_DIR / \"build\"\n",
|
||||
"BUILD_DIR.mkdir(exist_ok=True, parents=True)\n",
|
||||
"\n",
|
||||
"REPOSITORY_NAME = \"vertex-vision-model-garden-dockers\"\n",
|
||||
"\n",
|
||||
"!gcloud artifacts repositories create $REPOSITORY_NAME \\\n",
|
||||
" --repository-format=docker \\\n",
|
||||
" --location=$REGION \\\n",
|
||||
" --project=$PROJECT_ID\n",
|
||||
"\n",
|
||||
"! gcloud auth configure-docker $REGION-docker.pkg.dev --quiet"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"source": [
|
||||
"## Integrate OSS LLMs With ADK"
|
||||
],
|
||||
"metadata": {
|
||||
"id": "T9UiFwbiowOV"
|
||||
}
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"source": [
|
||||
"# @title Deploy OSS LLMs For Agents\n",
|
||||
"# @markdown In order to use OSS LLM endpoints smoothly with ADK,\n",
|
||||
"# @markdown these models should be deployed with:\n",
|
||||
"\n",
|
||||
"# @markdown - **enabling tool calls**. e.g.:\n",
|
||||
"# @markdown If the models are deployed with *vllm*,\n",
|
||||
"# @markdown the deployment should specify settings like `--enable-auto-tool-choice`\n",
|
||||
"# @markdown and `--tool-call-parser=hermes`.\n",
|
||||
"# @markdown If the models are deployed with *sglang*, the deployment should specify\n",
|
||||
"# @markdown setting like `--tool-call-parser=qwen25`.\n",
|
||||
"# @markdown Refer to tool calls in\n",
|
||||
"# @markdown [vllm](https://docs.vllm.ai/en/stable/features/tool_calling.html)\n",
|
||||
"# @markdown and [sglang](https://docs.sglang.ai/backend/function_calling.html) for more details.\n",
|
||||
"\n",
|
||||
"# @markdown - **disable dedicated endpoints**. The dedicated endpoints are not\n",
|
||||
"# @markdown supported in ADK yet.\n",
|
||||
"\n",
|
||||
"# @markdown You can deploy models below with proper deployment settings for ADK integration.\n",
|
||||
"\n",
|
||||
"MODEL_ID = \"Qwen3-32B\" # @param [\"Qwen3-32B\"] {isTemplate: true}\n",
|
||||
"accelerator_type = \"NVIDIA_H100_80GB\" # @param [\"NVIDIA_L4\", \"NVIDIA_A100_80GB\", \"NVIDIA_H100_80GB\"] {isTemplate: true}\n",
|
||||
"\n",
|
||||
"if accelerator_type == \"NVIDIA_L4\":\n",
|
||||
" accelerator_count = 4\n",
|
||||
" # Sets machine type to g2-standard-48 for 4 L4's\n",
|
||||
" machine_type = \"g2-standard-48\"\n",
|
||||
"elif accelerator_type == \"NVIDIA_A100_80GB\":\n",
|
||||
" accelerator_count = 1\n",
|
||||
" # Sets machine type to a2-ultragpu-1g for 1 Nvidia A100 80 GB.\n",
|
||||
" machine_type = \"a2-ultragpu-1g\"\n",
|
||||
"elif accelerator_type == \"NVIDIA_H100_80GB\":\n",
|
||||
" accelerator_count = 2\n",
|
||||
" machine_type = \"a3-highgpu-2g\"\n",
|
||||
"\n",
|
||||
"else:\n",
|
||||
" raise ValueError(\n",
|
||||
" \"Recommended machine settings not found for accelerator type: %s\"\n",
|
||||
" % accelerator_type\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"deploy_request_timeout = 1800 # 30 minutes\n",
|
||||
"\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"publisher_model_name = f\"publishers/qwen/models/qwen3@{MODEL_ID.lower()}\"\n",
|
||||
"model = model_garden.OpenModel(publisher_model_name)\n",
|
||||
"\n",
|
||||
"container_spec = model.list_deploy_options()[0].container_spec\n",
|
||||
"updated_args = container_spec.args[:-2] + [f\"--tp={accelerator_count}\", \"--tool-call-parser=qwen25\"]\n",
|
||||
"container_spec.args = updated_args\n",
|
||||
"\n",
|
||||
"print(\"The container spec are:\")\n",
|
||||
"print(container_spec)\n",
|
||||
"\n",
|
||||
"print(\"Start to check quota for the deployment.\")\n",
|
||||
"common_util.check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
" region=REGION,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" is_for_training=False,\n",
|
||||
")\n",
|
||||
"print(\"Finished to check quota for the deployment.\")\n",
|
||||
"\n",
|
||||
"print(\"Start to deploy models to endpoints.\")\n",
|
||||
"endpoint = model.deploy(\n",
|
||||
" serving_container_spec=container_spec,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=False,\n",
|
||||
" spot=False,\n",
|
||||
" deploy_request_timeout=deploy_request_timeout,\n",
|
||||
" accept_eula=False,\n",
|
||||
")\n",
|
||||
"print(\"Finished to deploy models to endpoints.\")\n",
|
||||
"# @markdown After endpoints are deployed successfully, you get the endpoint\n",
|
||||
"# @markdown resource name with the format as\n",
|
||||
"# @markdown `projects/{PROJECT_ID}/locations/{REGION}/endpoints/{ENDPOINT_ID}`.\n",
|
||||
"# @markdown The endpoint resource name will be used in local predictions and\n",
|
||||
"# @markdown integration with ADK below.\n",
|
||||
"endpoint_resource_name = endpoint.resource_name\n",
|
||||
"print(\"The deployed endpoint resource name is:\")\n",
|
||||
"print(endpoint_resource_name)\n",
|
||||
"# @markdown Click \"Show Code\" to see more details.\n"
|
||||
],
|
||||
"metadata": {
|
||||
"id": "gbfFZLT5KkUV",
|
||||
"cellView": "form"
|
||||
},
|
||||
"execution_count": null,
|
||||
"outputs": []
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"source": [
|
||||
"# @title Test The Endpoint\n",
|
||||
"# endpoint_resource_name = (\n",
|
||||
"# f\"projects/{PROJECT_ID}/locations/{REGION}/endpoints/{ENDPOINT_NAME}\"\n",
|
||||
"# )\n",
|
||||
"endpoint = aiplatform.Endpoint(endpoint_resource_name)\n",
|
||||
"\n",
|
||||
"location = endpoint_resource_name.split('/')[3]\n",
|
||||
"base_url = f\"https://{location}-aiplatform.googleapis.com/v1beta1/{endpoint.resource_name}\"\n",
|
||||
"\n",
|
||||
"# @markdown Predict locally with some requests.\n",
|
||||
"\n",
|
||||
"user_message = \"How is your day going?\" # @param {type: \"string\"}\n",
|
||||
"# @markdown If you encounter the issue like `ServiceUnavailable: 503 Took too long to respond when processing`, you can reduce the maximum number of output tokens, such as set `max_tokens` as 20.\n",
|
||||
"max_tokens = 100 # @param {type: \"integer\"}\n",
|
||||
"temperature = 1.0 # @param {type: \"number\"}\n",
|
||||
"stream = True # @param {type: \"boolean\"}\n",
|
||||
"\n",
|
||||
"import google.auth\n",
|
||||
"import openai\n",
|
||||
"\n",
|
||||
"creds, project = google.auth.default()\n",
|
||||
"auth_req = google.auth.transport.requests.Request()\n",
|
||||
"creds.refresh(auth_req)\n",
|
||||
"\n",
|
||||
"client = openai.OpenAI(base_url=base_url, api_key=creds.token)\n",
|
||||
"\n",
|
||||
"model_response = client.chat.completions.create(\n",
|
||||
" model=\"\",\n",
|
||||
" messages=[{\"role\": \"user\", \"content\": user_message}],\n",
|
||||
" temperature=temperature,\n",
|
||||
" max_tokens=max_tokens,\n",
|
||||
" stream=stream,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"if stream:\n",
|
||||
" usage = None\n",
|
||||
" contents = []\n",
|
||||
" for chunk in model_response:\n",
|
||||
" if chunk.usage is not None:\n",
|
||||
" usage = chunk.usage\n",
|
||||
" continue\n",
|
||||
" print(chunk.choices[0].delta.content, end=\"\")\n",
|
||||
" contents.append(chunk.choices[0].delta.content)\n",
|
||||
" print(f\"\\n\\n{usage}\")\n",
|
||||
"else:\n",
|
||||
" print(model_response.choices[0].message.content)"
|
||||
],
|
||||
"metadata": {
|
||||
"id": "_MNVrfJomCd0",
|
||||
"cellView": "form"
|
||||
},
|
||||
"execution_count": null,
|
||||
"outputs": []
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "Aa4e1-6FvRAP",
|
||||
"cellView": "form"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Build Agent Web App Dockers With VMG Endpoints\n",
|
||||
"# @markdown The section will create required python and docker files first, and\n",
|
||||
"# @markdown then build the dockers with cloud build.\n",
|
||||
"\n",
|
||||
"# @markdown 1. Create `agent.py` by loading VMG endpoints and example tool functions.\n",
|
||||
"agent_app = '''\n",
|
||||
"\"\"\"This is a sample agent for model garden agents.\"\"\"\n",
|
||||
"\n",
|
||||
"import datetime\n",
|
||||
"import os\n",
|
||||
"import re\n",
|
||||
"import zoneinfo\n",
|
||||
"\n",
|
||||
"from google.adk.agents import LlmAgent\n",
|
||||
"from google.adk.models.lite_llm import LiteLlm\n",
|
||||
"import google.auth\n",
|
||||
"\n",
|
||||
"_MODEL_GARDEN_ENDPOINT_REGEX = r\"projects\\/.+\\/locations\\/.+\\/endpoints\\/.+\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_weather(city: str) -\u003e str:\n",
|
||||
" \"\"\"Simulates a web search. Use it get information on weather.\n",
|
||||
"\n",
|
||||
" Args:\n",
|
||||
" city: A string containing the location to get weather information for.\n",
|
||||
"\n",
|
||||
" Returns:\n",
|
||||
" A string with the simulated weather information for the queried city.\n",
|
||||
" \"\"\"\n",
|
||||
" if \"sf\" in city.lower() or \"san francisco\" in city.lower():\n",
|
||||
" return \"It's 70 degrees and foggy.\"\n",
|
||||
" return \"It's 80 degrees and sunny.\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_current_time(city: str) -\u003e str:\n",
|
||||
" \"\"\"Simulates getting the current time for a city.\n",
|
||||
"\n",
|
||||
" Args:\n",
|
||||
" city: The name of the city to get the current time for.\n",
|
||||
"\n",
|
||||
" Returns:\n",
|
||||
" A string with the current time information.\n",
|
||||
" \"\"\"\n",
|
||||
" if \"sf\" in city.lower() or \"san francisco\" in city.lower():\n",
|
||||
" tz_identifier = \"America/Los_Angeles\"\n",
|
||||
" else:\n",
|
||||
" return f\"Sorry, I don't have timezone information for city: {city}.\"\n",
|
||||
"\n",
|
||||
" tz = zoneinfo.ZoneInfo(tz_identifier)\n",
|
||||
" now = datetime.datetime.now(tz)\n",
|
||||
" return (\n",
|
||||
" f\"The current time for city {city} is\"\n",
|
||||
" f\" {now.strftime('%Y-%m-%d %H:%M:%S %Z%z')}\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def _get_auth_headers() -\u003e dict[str, str]:\n",
|
||||
" \"\"\"Gets the auth headers for the model garden endpoint.\"\"\"\n",
|
||||
" creds, _ = google.auth.default(\n",
|
||||
" scopes=[\"https://www.googleapis.com/auth/cloud-platform\"]\n",
|
||||
" )\n",
|
||||
" auth_req = google.auth.transport.requests.Request()\n",
|
||||
" creds.refresh(auth_req)\n",
|
||||
" return {\n",
|
||||
" \"Content-Type\": \"application/json\",\n",
|
||||
" \"Authorization\": f\"Bearer {creds.token}\",\n",
|
||||
" }\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def _setup_model_garden_endpoint():\n",
|
||||
" \"\"\"Sets up the model garden endpoint.\"\"\"\n",
|
||||
" endpoint = os.environ.get(\"GOOGLE_MODEL_GARDEN_ENDPOINT\", \"\")\n",
|
||||
"\n",
|
||||
" if not re.compile(_MODEL_GARDEN_ENDPOINT_REGEX).fullmatch(endpoint):\n",
|
||||
" raise ValueError(\n",
|
||||
" f\"Invalid model garden endpoint: {endpoint}. Please use the format\"\n",
|
||||
" \" projects/{project}/locations/{location}/endpoints/{endpoint}.\"\n",
|
||||
" )\n",
|
||||
" endpoint_parts = endpoint.split(\"/\")\n",
|
||||
" os.environ.setdefault(\"GOOGLE_GENAI_USE_VERTEXAI\", \"True\")\n",
|
||||
" os.environ[\"VERTEXAI_PROJECT\"] = endpoint_parts[1]\n",
|
||||
" os.environ[\"VERTEXAI_LOCATION\"] = endpoint_parts[3]\n",
|
||||
" os.environ[\"LITELLM_LOG\"] = \"DEBUG\"\n",
|
||||
" return f\"vertex_ai/openai/{endpoint_parts[5]}\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"auth_headers = _get_auth_headers()\n",
|
||||
"model = _setup_model_garden_endpoint()\n",
|
||||
"\n",
|
||||
"print(\"The current model is: {model}\")\n",
|
||||
"\n",
|
||||
"root_agent = LlmAgent(\n",
|
||||
" name=\"root_agent\",\n",
|
||||
" model=LiteLlm(\n",
|
||||
" model=model,\n",
|
||||
" extra_headers=auth_headers,\n",
|
||||
" ),\n",
|
||||
" instruction=(\n",
|
||||
" \"You are a helpful AI assistant designed to provide accurate and useful\"\n",
|
||||
" \" information. Please output the tool callings with json format if\"\n",
|
||||
" \" exists.\"\n",
|
||||
" ),\n",
|
||||
" description=\"Retrieves the weather and current time using specific tools.\",\n",
|
||||
" tools=[get_weather, get_current_time],\n",
|
||||
")\n",
|
||||
"'''\n",
|
||||
"with BUILD_DIR.joinpath(\"agent.py\").open(\"w\") as f:\n",
|
||||
" f.write(agent_app)\n",
|
||||
"\n",
|
||||
"# @markdown 2. Create `__init__.py` to load agent.py for ADK apps.\n",
|
||||
"initialize = '''\n",
|
||||
"from . import agent\n",
|
||||
"'''\n",
|
||||
"with BUILD_DIR.joinpath(\"__init__.py\").open(\"w\") as f:\n",
|
||||
" f.write(initialize)\n",
|
||||
"\n",
|
||||
"# @markdown 3. Create `Dockerfile` to build agent app dockers.\n",
|
||||
"dockerfile_content = '''\n",
|
||||
"FROM python:3.11-slim\n",
|
||||
"WORKDIR /app\n",
|
||||
"RUN adduser --disabled-password --gecos \"\" myuser\n",
|
||||
"RUN chown -R myuser:myuser /app\n",
|
||||
"USER myuser\n",
|
||||
"ENV PATH=\"/home/myuser/.local/bin:$PATH\"\n",
|
||||
"RUN pip install \\\n",
|
||||
" google-adk~=0.4.0 \\\n",
|
||||
" google-cloud-logging~=3.11.4 \\\n",
|
||||
" opentelemetry-exporter-gcp-trace~=1.9.0 \\\n",
|
||||
" google-cloud-aiplatform[evaluation,agent-engines]~=1.88.0 \\\n",
|
||||
" litellm~=1.66.2\n",
|
||||
"\n",
|
||||
"COPY agent.py \"/app/agents/model_garden_agents/\"\n",
|
||||
"COPY __init__.py \"/app/agents/model_garden_agents/\"\n",
|
||||
"ENV GOOGLE_MODEL_GARDEN_ENDPOINT YOUR_ENDPOINT\n",
|
||||
"EXPOSE 8000\n",
|
||||
"CMD adk web --port=8000 --trace_to_cloud \"/app/agents\"\n",
|
||||
"'''\n",
|
||||
"with BUILD_DIR.joinpath(\"Dockerfile\").open(\"w\") as f:\n",
|
||||
" f.write(dockerfile_content)\n",
|
||||
"\n",
|
||||
"# @markdown 4. Build agent web app dockers.\n",
|
||||
"VMG_AGENT_UI_CONTAINER_IMAGE_URI = (\n",
|
||||
" f\"{REGION}-docker.pkg.dev/{PROJECT_ID}/{REPOSITORY_NAME}/vmg-adk-ui\"\n",
|
||||
")\n",
|
||||
"! gcloud builds submit --tag $VMG_AGENT_UI_CONTAINER_IMAGE_URI --project $PROJECT_ID --machine-type e2-highcpu-32 $BUILD_DIR\n",
|
||||
"print(\"The agent UI docker is :\")\n",
|
||||
"print(VMG_AGENT_UI_CONTAINER_IMAGE_URI)\n",
|
||||
"# @markdown Click \"Show Code\" to see more details."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"source": [
|
||||
"# @title Deploy Agent Web App Dockers To Cloud Run\n",
|
||||
"# @markdown After the deployment, there will be a service URL. You can click\n",
|
||||
"# @markdown the service URL and interact with the agent web app. The agent web\n",
|
||||
"# @markdown is a built-in development UI in [ADK](https://github.com/google/adk-python?tab=readme-ov-file)\n",
|
||||
"# @markdown to help you test, evaluate, debug, and showcase your agent(s).\n",
|
||||
"\n",
|
||||
"# @markdown \n",
|
||||
"\n",
|
||||
"! gcloud run deploy vmg-agent-ui-1 \\\n",
|
||||
" --port 8000 \\\n",
|
||||
" --image=\"{VMG_AGENT_UI_CONTAINER_IMAGE_URI}\" \\\n",
|
||||
" --region=\"{REGION}\" \\\n",
|
||||
" --platform=managed \\\n",
|
||||
" --allow-unauthenticated \\\n",
|
||||
" --memory=1024Mi \\\n",
|
||||
" --set-env-vars=\"GOOGLE_MODEL_GARDEN_ENDPOINT={endpoint_resource_name}\"\n",
|
||||
"# @markdown Click \"Show Code\" to see more details."
|
||||
],
|
||||
"metadata": {
|
||||
"id": "L2AGmQZmuVam",
|
||||
"cellView": "form"
|
||||
},
|
||||
"execution_count": null,
|
||||
"outputs": []
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "tAelDidov5AW"
|
||||
},
|
||||
"source": [
|
||||
"## Clean up resources"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "8SeZCFo5v7z-"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @markdown Delete the experiment resources to avoid unnecessary continuous\n",
|
||||
"# @markdown charges that may incur.\n",
|
||||
"\n",
|
||||
"delete_endpoint = False # @param {type:\"boolean\"}\n",
|
||||
"delete_artifact_registry = False # @param {type:\"boolean\"}\n",
|
||||
"delete_tutorial_folder = False # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"if delete_endpoint:\n",
|
||||
" # Undeploy model and delete endpoint.\n",
|
||||
" endpoint.delete(force=True)\n",
|
||||
"\n",
|
||||
"if delete_artifact_registry:\n",
|
||||
" ! gcloud artifacts repositories delete $REPOSITORY_NAME \\\n",
|
||||
" --repository-format=docker \\\n",
|
||||
" --location=$REGION \\\n",
|
||||
" --project=$PROJECT_ID\n",
|
||||
"\n",
|
||||
"if delete_tutorial_folder:\n",
|
||||
" import shutil\n",
|
||||
" shutil.rmtree(TUTORIAL_DIR)\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"colab": {
|
||||
"name": "model_garden_integration_with_adk.ipynb",
|
||||
"toc_visible": true,
|
||||
"provenance": [],
|
||||
"collapsed_sections": [
|
||||
"tAelDidov5AW"
|
||||
]
|
||||
},
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"name": "python3"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 0
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_jax_dito.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_jax_fvlm.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_jax_owl_vit_v2.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_jax_paligemma_deployment.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_jax_paligemma_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -121,7 +121,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"! pip install -q gradio==4.21.0\n",
|
||||
@@ -299,7 +299,7 @@
|
||||
"# @markdown Kindly note that the deployment using custom_paligemma_model_uri is not supported.\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -308,7 +308,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -43,7 +43,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_jax_paligemma_finetuning.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_jax_stable_diffusion_xl.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_jax_vision_transformer.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_keras_stable_diffusion.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_keras_yolov8.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
+2
-2
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_llama3_1_finetuning_with_workbench.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_llama3_1_finetuning_with_workbench.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_llama3_2_deployment_on_gke.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_llama3_2_evaluation.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_llama_guard_deployment.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_llama_guard_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -136,7 +136,7 @@
|
||||
"# Import the necessary packages\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
@@ -268,7 +268,7 @@
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -277,7 +277,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_mammut.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_mediapipe_face_stylizer.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_mediapipe_gesture_recognition.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_mediapipe_image_classification.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -41,7 +41,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_mediapipe_image_generation.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_mediapipe_object_detection.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_mediapipe_text_classification.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_movinet_action_recognition.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_movinet_clip_classification.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_nvidia_cosmos_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -226,7 +226,7 @@
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Also offload the text encoder model for 14B models, to avoid CUDA OOM issue.\n",
|
||||
" if model_id.lower().includes(\"14b\"):\n",
|
||||
" if \"14b\" in model_id.lower():\n",
|
||||
" serving_env[\"OFFLOAD_TEXT_ENCODER_MODEL\"] = \"true\"\n",
|
||||
"\n",
|
||||
" model = aiplatform.Model.upload(\n",
|
||||
@@ -251,7 +251,7 @@
|
||||
" return model, endpoint\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"models[\"model\"], endpoints[\"endpoint\"] = deploy_model(\n",
|
||||
"models[\"model\"], endpoints[\"tw-endpoint\"] = deploy_model(\n",
|
||||
" model_id=model_id,\n",
|
||||
" task=task,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
@@ -259,7 +259,7 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(\"endpoint_name:\", endpoints[\"endpoint\"].name)"
|
||||
"print(\"endpoint_name:\", endpoints[\"tw-endpoint\"].name)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -322,7 +322,7 @@
|
||||
"}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"response = endpoints[\"endpoint\"].predict(\n",
|
||||
"response = endpoints[\"tw-endpoint\"].predict(\n",
|
||||
" instances=instances, parameters=parameters, use_dedicated_endpoint=True\n",
|
||||
")\n",
|
||||
"\n",
|
||||
@@ -355,6 +355,7 @@
|
||||
"# @markdown A valid HF_TOKEN is required for model deployment.\n",
|
||||
"# @markdown Follow the instructions at [Hugging Face Token Guide](https://huggingface.co/docs/hub/en/security-tokens) to obtain your HF_TOKEN.\n",
|
||||
"# @markdown Additionally, ensure you have access to the model by following the instructions on its Hugging Face model card page.\n",
|
||||
"# @markdown You may also need access to the Pixtral models such as mistralai/Pixtral-12B-2409.\n",
|
||||
"\n",
|
||||
"HF_TOKEN = \"\" # @param {type:\"string\", isTemplate: true}\n",
|
||||
"if not HF_TOKEN:\n",
|
||||
@@ -415,7 +416,7 @@
|
||||
" }\n",
|
||||
"\n",
|
||||
" # Also offload the text encoder model for 14B models, to avoid CUDA OOM issue.\n",
|
||||
" if model_id.lower().includes(\"14b\"):\n",
|
||||
" if \"14b\" in model_id.lower():\n",
|
||||
" serving_env[\"OFFLOAD_TEXT_ENCODER_MODEL\"] = \"true\"\n",
|
||||
"\n",
|
||||
" model = aiplatform.Model.upload(\n",
|
||||
@@ -545,7 +546,7 @@
|
||||
"\n",
|
||||
"# @markdown For inference tasks exceeding 10 minutes, we recommend using CURL for predictions.\n",
|
||||
"\n",
|
||||
"os.environ[\"ENDPOINT_ID\"] = endpoints[\"endpoint\"].name\n",
|
||||
"os.environ[\"ENDPOINT_ID\"] = endpoints[\"tw-endpoint\"].name\n",
|
||||
"os.environ[\"PROJECT_ID\"] = project_number\n",
|
||||
"os.environ[\"REGION\"] = REGION"
|
||||
]
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_ollama_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -50,7 +50,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_openai_api_llama3_1.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\"><br> View on GitHub\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>"
|
||||
|
||||
@@ -50,7 +50,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_openai_api_llama3_2.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\"><br> View on GitHub\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>"
|
||||
|
||||
@@ -50,7 +50,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_openai_api_llama4.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\"><br> View on GitHub\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_phi3_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_phi4_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
+1
-1
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_proprietary_image_classification.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
+1
-1
@@ -40,7 +40,7 @@
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_proprietary_image_object_detection.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_autogluon.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_bart_large_cnn.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_biogpt_serve.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_biomedclip.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_biomedclip.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -121,7 +121,7 @@
|
||||
"# Import the necessary packages\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
@@ -139,7 +139,6 @@
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"LABEL = \"biomedclip_serve\"\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
@@ -216,8 +215,9 @@
|
||||
"source": [
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"LABEL = \"sdk-deploy\"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -226,7 +226,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -256,9 +258,13 @@
|
||||
" machine_type: str,\n",
|
||||
" accelerator_type: str,\n",
|
||||
" accelerator_count: int,\n",
|
||||
" use_dedicated_endpoint: bool = False,\n",
|
||||
") -> Tuple[aiplatform.Model, aiplatform.Endpoint]:\n",
|
||||
" \"\"\"Deploys trained models into Vertex AI.\"\"\"\n",
|
||||
" endpoint = aiplatform.Endpoint.create(display_name=f\"{model_name}-endpoint\")\n",
|
||||
" endpoint = aiplatform.Endpoint.create(\n",
|
||||
" display_name=f\"{model_name}-endpoint\",\n",
|
||||
" dedicated_endpoint_enabled=use_dedicated_endpoint,\n",
|
||||
" )\n",
|
||||
" serving_env = {\n",
|
||||
" \"MODEL\": model_id,\n",
|
||||
" \"TASK\": task,\n",
|
||||
@@ -291,6 +297,8 @@
|
||||
" return model, endpoint\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"LABEL = \"open-clip-deploy\"\n",
|
||||
"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model(\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=\"biomedclip-serve\"),\n",
|
||||
" model_id=model_id,\n",
|
||||
@@ -299,7 +307,11 @@
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=1,\n",
|
||||
")"
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model = models[LABEL]\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -362,7 +374,9 @@
|
||||
" {\"text\": \"This is a photo of hematoxylin and eosin histopathology\"},\n",
|
||||
" {\"text\": \"This is a photo of pie chart\"},\n",
|
||||
"]\n",
|
||||
"response = endpoints[LABEL].predict(instances=instances)\n",
|
||||
"response = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(response.predictions)\n",
|
||||
"\n",
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_blip2.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_blip_image_captioning.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_blip_image_captioning.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -119,7 +119,7 @@
|
||||
"# Import the necessary packages\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"# Import the necessary packages.\n",
|
||||
"import importlib\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_blip_vqa.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_clip.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_clip.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -92,21 +92,24 @@
|
||||
"source": [
|
||||
"# @title Setup Google Cloud project\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"\n",
|
||||
"# @markdown 1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"# @markdown 2. **[Optional]** [Create a Cloud Storage bucket](https://cloud.google.com/storage/docs/creating-buckets) for storing experiment outputs. Set the BUCKET_URI for the experiment environment. The specified Cloud Storage bucket (`BUCKET_URI`) should be located in the same region as where the notebook was launched. Note that a multi-region bucket (eg. \"us\") is not considered a match for a single region covered by the multi-region range (eg. \"us-central1\"). If not set, a unique GCS bucket will be created instead.\n",
|
||||
"\n",
|
||||
"BUCKET_URI = \"gs://\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# @markdown 3. **[Optional]** Set region. If not set, the region will be set automatically according to Colab Enterprise environment.\n",
|
||||
"# @markdown 2. **[Optional]** Set region. If not set, the region will be set automatically according to Colab Enterprise environment.\n",
|
||||
"\n",
|
||||
"REGION = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# @markdown 3. If you want to run predictions with A100 80GB or H100 GPUs, we recommend using the regions listed below. **NOTE:** Make sure you have associated quota in selected regions. Click the links to see your current quota for each GPU type: [Nvidia A100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_a100_80gb_gpus), [Nvidia H100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_h100_gpus). You can request for quota following the instructions at [\"Request a higher quota\"](https://cloud.google.com/docs/quota/view-manage#requesting_higher_quota).\n",
|
||||
"\n",
|
||||
"# @markdown > | Machine Type | Accelerator Type | Recommended Regions |\n",
|
||||
"# @markdown | ----------- | ----------- | ----------- |\n",
|
||||
"# @markdown | a2-ultragpu-1g | 1 NVIDIA_A100_80GB | us-central1, us-east4, europe-west4, asia-southeast1, us-east4 |\n",
|
||||
"# @markdown | a3-highgpu-2g | 2 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-4g | 4 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
@@ -133,15 +136,6 @@
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)\n",
|
||||
"\n",
|
||||
"! gcloud config set project $PROJECT_ID\n",
|
||||
"\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)\n",
|
||||
"\n",
|
||||
"# The pre-built serving docker image. It contains serving scripts and models.\n",
|
||||
"SERVE_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-transformers-serve\"\n",
|
||||
"\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"LABEL = \"endpoint\"\n",
|
||||
"\n",
|
||||
"import vertexai\n",
|
||||
"\n",
|
||||
"vertexai.init(\n",
|
||||
@@ -149,17 +143,102 @@
|
||||
" location=REGION,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"version_id = \"clip-vit-base-patch32\"\n",
|
||||
"PUBLISHER_MODEL_NAME = f\"publishers/openai/models/clip-vit-base-patch32@{version_id}\"\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"\n",
|
||||
"import vertexai\n",
|
||||
"\n",
|
||||
"vertexai.init(\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "aG99vbcAqsuZ"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Set the parameters\n",
|
||||
"\n",
|
||||
"MODEL_NAME = \"clip-vit-base-patch32\"\n",
|
||||
"PUBLISHER_MODEL_NAME = f\"publishers/openai/models/clip-vit-base-patch32@{MODEL_NAME}\"\n",
|
||||
"\n",
|
||||
"# @markdown Find Vertex AI prediction supported accelerators and regions at https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
|
||||
"accelerator_type = \"NVIDIA_TESLA_T4\" # @param [\"NVIDIA_TESLA_T4\"]\n",
|
||||
"machine_type = \"n1-standard-8\"\n",
|
||||
"accelerator_count = 1\n",
|
||||
"\n",
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "LQir_tczhdEP"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"LABEL = \"sdk-deploy\"\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "vBvfVTWOhdEP"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Deploy Zero-shot image classification with customized configs\n",
|
||||
"\n",
|
||||
"# @markdown This section uploads the pre-trained model to Model Registry and deploys it on the Endpoint with 1 T4 GPU.\n",
|
||||
"# @markdown The model deployment step will take ~15 minutes to complete.\n",
|
||||
"\n",
|
||||
"# The pre-built serving docker image. It contains serving scripts and models.\n",
|
||||
"SERVE_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-transformers-serve\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def deploy_model(model_id, task):\n",
|
||||
"def deploy_model(\n",
|
||||
" model_id,\n",
|
||||
" task,\n",
|
||||
" accelerator_type,\n",
|
||||
" machine_type,\n",
|
||||
" accelerator_count,\n",
|
||||
" use_dedicated_endpoint,\n",
|
||||
"):\n",
|
||||
" model_name = \"clip\"\n",
|
||||
" endpoint = aiplatform.Endpoint.create(display_name=f\"{model_name}-endpoint\")\n",
|
||||
" endpoint = aiplatform.Endpoint.create(\n",
|
||||
" display_name=f\"{model_name}-endpoint\",\n",
|
||||
" dedicated_endpoint_enabled=use_dedicated_endpoint,\n",
|
||||
" )\n",
|
||||
" serving_env = {\n",
|
||||
" \"MODEL_ID\": model_id,\n",
|
||||
" \"TASK\": task,\n",
|
||||
" \"DEPLOY_SOURCE\": \"notebook\",\n",
|
||||
" \"DEPLOY_SOURCE\": common_util.get_deploy_source(),\n",
|
||||
" }\n",
|
||||
" # If the model_id is a GCS path, use artifact_uri to pass it to serving docker.\n",
|
||||
" artifact_uri = model_id if model_id.startswith(\"gs://\") else None\n",
|
||||
@@ -184,34 +263,29 @@
|
||||
" \"NOTEBOOK_ENVIRONMENT\": common_util.get_deploy_source(),\n",
|
||||
" },\n",
|
||||
" )\n",
|
||||
" return model, endpoint"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "LQir_tczhdEP"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"accelerator_type = \"NVIDIA_TESLA_T4\"\n",
|
||||
"machine_type = \"n1-standard-8\"\n",
|
||||
"accelerator_count = 1\n",
|
||||
" return model, endpoint\n",
|
||||
"\n",
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
" machine_type=machine_type,\n",
|
||||
"common_util.check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
" region=REGION,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
" is_for_training=False,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"LABEL = \"zero-shot-icn-deploy\"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model(\n",
|
||||
" model_id=\"openai/clip-vit-base-patch32\",\n",
|
||||
" task=\"zero-shot-image-classification\",\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]\n",
|
||||
"model = models[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -219,25 +293,17 @@
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "vBvfVTWOhdEP"
|
||||
"id": "hHvzCfAsvt1J"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title [Option 2] Deploy with customized configs\n",
|
||||
"# @title Prediction\n",
|
||||
"\n",
|
||||
"# @markdown This section uploads the pre-trained model to Model Registry and deploys it on the Endpoint with 1 T4 GPU.\n",
|
||||
"# @markdown The model deployment step will take ~15 minutes to complete.\n",
|
||||
"USER_IMAGE1 = \"http://images.cocodataset.org/val2017/000000039769.jpg\" # @param\n",
|
||||
"USER_IMAGE2 = \"http://images.cocodataset.org/val2017/000000000285.jpg\" # @param\n",
|
||||
"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model(\n",
|
||||
" model_id=\"openai/clip-vit-base-patch32\", task=\"zero-shot-image-classification\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"image1 = common_util.download_image(\n",
|
||||
" \"http://images.cocodataset.org/val2017/000000039769.jpg\"\n",
|
||||
")\n",
|
||||
"image2 = common_util.download_image(\n",
|
||||
" \"http://images.cocodataset.org/val2017/000000000285.jpg\"\n",
|
||||
")\n",
|
||||
"image1 = common_util.download_image(USER_IMAGE1)\n",
|
||||
"image2 = common_util.download_image(USER_IMAGE2)\n",
|
||||
"grid = common_util.image_grid([image1, image2], 1, 2)\n",
|
||||
"display(grid)\n",
|
||||
"\n",
|
||||
@@ -245,13 +311,49 @@
|
||||
" {\"image\": common_util.image_to_base64(image1), \"text\": \"two cats\"},\n",
|
||||
" {\"image\": common_util.image_to_base64(image2), \"text\": \"a bear\"},\n",
|
||||
"]\n",
|
||||
"preds = endpoints[LABEL].predict(instances=instances).predictions\n",
|
||||
"print(preds)\n",
|
||||
"preds = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"print(preds)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "W7JNfp8FyHaM"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Deploy Image/text feature embedding with customized configs\n",
|
||||
"\n",
|
||||
"LABEL = \"feature-embedding-deploy\"\n",
|
||||
"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model(\n",
|
||||
" model_id=\"openai/clip-vit-base-patch32\", task=\"feature-embedding\"\n",
|
||||
" model_id=\"openai/clip-vit-base-patch32\",\n",
|
||||
" task=\"feature-embedding\",\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]\n",
|
||||
"model = models[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "NJz3X2D2yZVT"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Prediction\n",
|
||||
"\n",
|
||||
"import numpy as np\n",
|
||||
"\n",
|
||||
"# Extract feature embedding of images.\n",
|
||||
@@ -262,7 +364,9 @@
|
||||
"instances = [\n",
|
||||
" {\"image\": common_util.image_to_base64(image)},\n",
|
||||
"]\n",
|
||||
"preds = endpoints[LABEL].predict(instances=instances).predictions\n",
|
||||
"preds = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
").predictions\n",
|
||||
"image_features = np.array(preds[0][\"image_features\"])\n",
|
||||
"print(image_features.shape)\n",
|
||||
"\n",
|
||||
@@ -271,7 +375,7 @@
|
||||
" {\"text\": \"two cats\"},\n",
|
||||
" {\"text\": \"hello world\"},\n",
|
||||
"]\n",
|
||||
"preds = endpoints[LABEL].predict(instances=instances).predictions\n",
|
||||
"preds = endpoint.predict(instances=instances).predictions\n",
|
||||
"text_features = np.array(preds[0][\"text_features\"])\n",
|
||||
"print(text_features.shape)"
|
||||
]
|
||||
@@ -285,6 +389,8 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Delete the models and endpoints\n",
|
||||
"\n",
|
||||
"# @markdown Delete the experiment models and endpoints to recycle the resources\n",
|
||||
"# @markdown and avoid unnecessary continuous charges that may incur.\n",
|
||||
"\n",
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_codellama.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_codellama.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -117,7 +117,7 @@
|
||||
"# Import the necessary packages\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"# ! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
@@ -443,7 +443,7 @@
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -452,7 +452,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_codellama_evaluation.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_controlnet.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\"><br>\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\"><br>\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_csm_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_deepseek_deployment.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_deepseek_deployment.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -64,7 +64,7 @@
|
||||
"\n",
|
||||
"### Objective\n",
|
||||
"\n",
|
||||
"- Deploy DeepSeek-V3 and DeepSeek-R1 with vLLM, SGLang, or TensorRT-LLM on GPU using single-host and multi-host serving, and [Spot VMs](https://cloud.google.com/compute/docs/instances/spot) (Optional). Multi-host GPU serving is a preview feature.\n",
|
||||
"- Deploy DeepSeek-V3 and DeepSeek-R1 largest variants with vLLM, SGLang, or TensorRT-LLM on GPU using single-host and multi-host serving, and [Spot VMs](https://cloud.google.com/compute/docs/instances/spot) (Optional). Multi-host GPU serving is a preview feature.\n",
|
||||
"\n",
|
||||
"### File a bug\n",
|
||||
"\n",
|
||||
@@ -132,7 +132,7 @@
|
||||
"# @markdown | a3-highgpu-8g (regular VM) | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"import importlib\n",
|
||||
@@ -233,7 +233,7 @@
|
||||
"\n",
|
||||
"# @markdown Set the model to deploy.\n",
|
||||
"\n",
|
||||
"base_model_name = \"DeepSeek-R1\" # @param [\"DeepSeek-V3\", \"DeepSeek-V3-Base\", \"DeepSeek-V3-0324\", \"DeepSeek-R1\"] {isTemplate:true}\n",
|
||||
"base_model_name = \"DeepSeek-R1\" # @param [\"DeepSeek-V3\", \"DeepSeek-V3-Base\", \"DeepSeek-V3-0324\", \"DeepSeek-R1\", \"DeepSeek-R1-0528\"] {isTemplate:true}\n",
|
||||
"model_id = \"deepseek-ai/\" + base_model_name\n",
|
||||
"hf_model_id = model_id\n",
|
||||
"if \"R1\" in model_id:\n",
|
||||
@@ -766,7 +766,7 @@
|
||||
"\n",
|
||||
"# @markdown Set the model to deploy.\n",
|
||||
"\n",
|
||||
"base_model_name = \"DeepSeek-R1\" # @param [\"DeepSeek-V3\", \"DeepSeek-V3-Base\", \"DeepSeek-V3-0324\", \"DeepSeek-R1\"] {isTemplate:true}\n",
|
||||
"base_model_name = \"DeepSeek-R1\" # @param [\"DeepSeek-V3\", \"DeepSeek-V3-Base\", \"DeepSeek-V3-0324\", \"DeepSeek-R1\", \"DeepSeek-R1-0528\"] {isTemplate:true}\n",
|
||||
"model_id = \"deepseek-ai/\" + base_model_name\n",
|
||||
"hf_model_id = model_id\n",
|
||||
"\n",
|
||||
@@ -1232,7 +1232,7 @@
|
||||
"\n",
|
||||
"# @markdown Set the model to deploy.\n",
|
||||
"\n",
|
||||
"base_model_name = \"DeepSeek-R1\" # @param [\"DeepSeek-V3\", \"DeepSeek-V3-Base\", \"DeepSeek-V3-0324\", \"DeepSeek-R1\"] {isTemplate:true}\n",
|
||||
"base_model_name = \"DeepSeek-R1\" # @param [\"DeepSeek-V3\", \"DeepSeek-V3-Base\", \"DeepSeek-V3-0324\", \"DeepSeek-R1\", \"DeepSeek-R1-0528\"] {isTemplate:true}\n",
|
||||
"model_id = \"deepseek-ai/\" + base_model_name\n",
|
||||
"hf_model_id = model_id\n",
|
||||
"\n",
|
||||
@@ -1248,6 +1248,7 @@
|
||||
"# @markdown Find Vertex AI prediction supported accelerators and regions at https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
|
||||
"trtllm_accelerator_type = \"NVIDIA_H200_141GB\" # @param [\"NVIDIA_H200_141GB\"] {isTemplate:true}\n",
|
||||
"accelerator_count = 8\n",
|
||||
"trtllm_region = \"us-east4\" # @param [\"us-east4\"] {isTemplate:true}\n",
|
||||
"if trtllm_accelerator_type == \"NVIDIA_H200_141GB\":\n",
|
||||
" machine_type = \"a3-ultragpu-8g\"\n",
|
||||
" multihost_gpu_node_count = 1\n",
|
||||
@@ -1257,7 +1258,7 @@
|
||||
"\n",
|
||||
"check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
" region=REGION,\n",
|
||||
" region=trtllm_region,\n",
|
||||
" resource_id=resource_id,\n",
|
||||
" accelerator_count=int(accelerator_count * multihost_gpu_node_count),\n",
|
||||
")\n",
|
||||
@@ -1269,7 +1270,7 @@
|
||||
"GPU_MEMORY_UTILIZATION = 0.55\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def poll_operation(op_name: str) -> bool: # noqa: F811\n",
|
||||
"def poll_operation(op_name: str, trtllm_region: str) -> bool: # noqa: F811\n",
|
||||
" creds, _ = auth.default()\n",
|
||||
" auth_req = auth.transport.requests.Request()\n",
|
||||
" creds.refresh(auth_req)\n",
|
||||
@@ -1277,7 +1278,7 @@
|
||||
" \"Authorization\": f\"Bearer {creds.token}\",\n",
|
||||
" }\n",
|
||||
" get_resp = requests.get(\n",
|
||||
" f\"https://{REGION}-aiplatform.googleapis.com/ui/{op_name}\",\n",
|
||||
" f\"https://{trtllm_region}-aiplatform.googleapis.com/ui/{op_name}\",\n",
|
||||
" headers=headers,\n",
|
||||
" )\n",
|
||||
" opjs = get_resp.json()\n",
|
||||
@@ -1286,9 +1287,11 @@
|
||||
" return opjs.get(\"done\", False)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def poll_and_wait(op_name: str, total_wait: int, interval: int = 60): # noqa: F811\n",
|
||||
"def poll_and_wait_trtllm(\n",
|
||||
" op_name: str, total_wait: int, trtllm_region: str, interval: int = 60\n",
|
||||
"): # noqa: F811\n",
|
||||
" waited = 0\n",
|
||||
" while not poll_operation(op_name):\n",
|
||||
" while not poll_operation(op_name, trtllm_region):\n",
|
||||
" if waited > total_wait:\n",
|
||||
" raise TimeoutError(\"Operation timed out\")\n",
|
||||
" print(\n",
|
||||
@@ -1319,10 +1322,12 @@
|
||||
" enable_chunked_prefill: bool = False,\n",
|
||||
" use_dedicated_endpoint: bool = False,\n",
|
||||
" is_spot: bool = True,\n",
|
||||
" trtllm_region: str = REGION,\n",
|
||||
") -> Tuple[aiplatform.Model, aiplatform.Endpoint]:\n",
|
||||
" \"\"\"Deploys trained models with TensorRT-LLM on Vertex AI.\"\"\"\n",
|
||||
" endpoint = aiplatform.Endpoint.create(\n",
|
||||
" display_name=f\"{model_name}-endpoint\",\n",
|
||||
" location=trtllm_region,\n",
|
||||
" dedicated_endpoint_enabled=use_dedicated_endpoint,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
@@ -1370,6 +1375,7 @@
|
||||
"\n",
|
||||
" model = aiplatform.Model.upload(\n",
|
||||
" display_name=model_name,\n",
|
||||
" location=trtllm_region,\n",
|
||||
" serving_container_image_uri=TRTLLM_DOCKER_URI,\n",
|
||||
" serving_container_args=container_args,\n",
|
||||
" serving_container_ports=[7080],\n",
|
||||
@@ -1390,7 +1396,7 @@
|
||||
" auth_req = auth.transport.requests.Request()\n",
|
||||
" creds.refresh(auth_req)\n",
|
||||
"\n",
|
||||
" url = f\"https://{REGION}-aiplatform.googleapis.com/ui/projects/{PROJECT_ID}/locations/{REGION}/endpoints/{endpoint.name}:deployModel\"\n",
|
||||
" url = f\"https://{trtllm_region}-aiplatform.googleapis.com/ui/projects/{PROJECT_ID}/locations/{trtllm_region}/endpoints/{endpoint.name}:deployModel\"\n",
|
||||
" headers = {\n",
|
||||
" \"Content-Type\": \"application/json\",\n",
|
||||
" \"Authorization\": f\"Bearer {creds.token}\",\n",
|
||||
@@ -1423,7 +1429,7 @@
|
||||
" print(f\"Deploy Model response: {response.json()}\")\n",
|
||||
" if response.status_code != 200 or \"name\" not in response.json():\n",
|
||||
" raise ValueError(f\"Failed to deploy model: {response.text}\")\n",
|
||||
" poll_and_wait(response.json()[\"name\"], 7200)\n",
|
||||
" poll_and_wait_trtllm(response.json()[\"name\"], 7200, trtllm_region)\n",
|
||||
" print(\"endpoint_name:\", endpoint.name)\n",
|
||||
"\n",
|
||||
" return model, endpoint\n",
|
||||
@@ -1446,6 +1452,7 @@
|
||||
" enable_trust_remote_code=True,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" is_spot=is_spot,\n",
|
||||
" trtllm_region=trtllm_region,\n",
|
||||
")\n",
|
||||
"# @markdown Click \"Show Code\" to see more details."
|
||||
]
|
||||
|
||||
+1
-1
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_deployed_model_agent_engine.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
+1
-1
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_deployed_model_reasoning_engine.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -4,11 +4,12 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "ur8xi4C7S06n"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2023 Google LLC\n",
|
||||
"# Copyright 2025 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -31,26 +32,23 @@
|
||||
"source": [
|
||||
"# Vertex AI Model Garden - Detectron2\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_detectron2.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_detectron2.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_detectron2.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
"Open in Vertex AI Workbench\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>"
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/colab/import/https:%2F%2Fraw.githubusercontent.com%2FGoogleCloudPlatform%2Fvertex-ai-samples%2Fmain%2Fnotebooks%2Fcommunity%2Fmodel_garden%2Fmodel_garden_pytorch_detectron2.ipynb\">\n",
|
||||
" <img alt=\"Google Cloud Colab Enterprise logo\" src=\"https://lh3.googleusercontent.com/JmcxdQi-qOpctIvWKgPtrzZdJJK-J3sWE1RsfjZNwshCFgE_9fULcNpuXYTilIR2hjwN\" width=\"32px\"><br> Run in Colab Enterprise\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_detectron2.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -72,6 +70,10 @@
|
||||
"- Deploy the model on [Endpoint](https://cloud.google.com/vertex-ai/docs/predictions/using-private-endpoints).\n",
|
||||
"- Run online predictions for image object detection and segmentation.\n",
|
||||
"\n",
|
||||
"### File a bug\n",
|
||||
"\n",
|
||||
"File a bug on [GitHub](https://github.com/GoogleCloudPlatform/vertex-ai-samples/issues/new) if you encounter any issue with the notebook.\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
@@ -79,7 +81,7 @@
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -88,68 +90,23 @@
|
||||
"id": "264c07757582"
|
||||
},
|
||||
"source": [
|
||||
"## Setup environment\n",
|
||||
"\n",
|
||||
"**NOTE**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "d73ffa0c0b83"
|
||||
},
|
||||
"source": [
|
||||
"### Colab only"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "2707b02ef5df"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"!pip3 install --upgrade google-cloud-aiplatform"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b60a4d7100bf"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google.colab import auth as google_auth\n",
|
||||
"\n",
|
||||
"google_auth.authenticate_user()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "fb671e75ca7b"
|
||||
},
|
||||
"source": [
|
||||
"### Install dependencies"
|
||||
"## Before you begin"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "dc8ee367fb42"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Install gdown for downloading example training images.\n",
|
||||
"!pip install gdown\n",
|
||||
"# Install gsutil for downloading/uploading data from/to Cloud Storage buckets.\n",
|
||||
"!pip install gsutil\n",
|
||||
"# @title Import the necessary packages\n",
|
||||
"\n",
|
||||
"# Install libraries for COCO format conversion of datasets.\n",
|
||||
"!pip install pycocotools==2.0.6\n",
|
||||
"!pip install opencv-python==4.7.0.72"
|
||||
"!pip install --upgrade --quiet pycocotools==2.0.6\n",
|
||||
"!pip install --upgrade --quiet opencv-python==4.7.0.72"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -165,115 +122,147 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "567212ff53a6"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import IPython\n",
|
||||
"# @title Setup Google Cloud project\n",
|
||||
"\n",
|
||||
"app = IPython.Application.instance()\n",
|
||||
"app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bb7adab99e41"
|
||||
},
|
||||
"source": [
|
||||
"### Setup Google Cloud project\n",
|
||||
"# @markdown 1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"# @markdown 2. For finetuning, **[click here](https://console.cloud.google.com/iam-admin/quotas?location=us-central1&metric=aiplatform.googleapis.com%2Frestricted_image_training_nvidia_a100_80gb_gpus)** to check if your project already has the required 8 Nvidia A100 80 GB GPUs in the us-central1 region. If yes, then run this notebook in the us-central1 region. If you do not have 8 Nvidia A100 80 GPUs or have more GPU requirements than this, then schedule your job with Nvidia H100 GPUs via Dynamic Workload Scheduler using [these instructions](https://cloud.google.com/vertex-ai/docs/training/schedule-jobs-dws). For Dynamic Workload Scheduler, check the [us-central1](https://console.cloud.google.com/iam-admin/quotas?location=us-central1&metric=aiplatform.googleapis.com%2Fcustom_model_training_preemptible_nvidia_h100_gpus) or [europe-west4](https://console.cloud.google.com/iam-admin/quotas?location=europe-west4&metric=aiplatform.googleapis.com%2Fcustom_model_training_preemptible_nvidia_h100_gpus) quota for Nvidia H100 GPUs. If you do not have enough GPUs, then you can follow [these instructions](https://cloud.google.com/docs/quotas/view-manage#viewing_your_quota_console) to request quota.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"# @markdown 3. For serving, **[click here](https://console.cloud.google.com/iam-admin/quotas?location=us-central1&metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_l4_gpus)** to check if your project already has the required 1 L4 GPU in the us-central1 region. If yes, then run this notebook in the us-central1 region. If you need more L4 GPUs for your project, then you can follow [these instructions](https://cloud.google.com/docs/quotas/view-manage#viewing_your_quota_console) to request more. Alternatively, if you want to run predictions with A100 80GB or H100 GPUs, we recommend using the regions listed below. **NOTE:** Make sure you have associated quota in selected regions. Click the links to see your current quota for each GPU type: [Nvidia A100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_a100_80gb_gpus), [Nvidia H100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_h100_gpus).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component).\n",
|
||||
"# @markdown > | Machine Type | Accelerator Type | Recommended Regions |\n",
|
||||
"# @markdown | ----------- | ----------- | ----------- |\n",
|
||||
"# @markdown | a2-ultragpu-1g | 1 NVIDIA_A100_80GB | us-central1, us-east4, europe-west4, asia-southeast1, us-east4 |\n",
|
||||
"# @markdown | a3-highgpu-2g | 2 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-4g | 4 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"1. [Create a Cloud Storage bucket](https://cloud.google.com/storage/docs/creating-buckets) for storing experiment outputs.\n",
|
||||
"# @markdown 4. **[Optional]** [Create a Cloud Storage bucket](https://cloud.google.com/storage/docs/creating-buckets) for storing experiment outputs. Set the BUCKET_URI for the experiment environment. The specified Cloud Storage bucket (`BUCKET_URI`) should be located in the same region as where the notebook was launched. Note that a multi-region bucket (eg. \"us\") is not considered a match for a single region covered by the multi-region range (eg. \"us-central1\"). If not set, a unique GCS bucket will be created instead.\n",
|
||||
"\n",
|
||||
"1. [Create a service account](https://cloud.google.com/iam/docs/service-accounts-create#iam-service-accounts-create-console) with `Vertex AI User`, `Storage Object Admin`, and `GCS Storage Bucket Owner` roles for deploying fine tuned model to Vertex AI endpoint."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "6c460088b873"
|
||||
},
|
||||
"source": [
|
||||
"Fill following variables for experiments environment:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "855d6b96f291"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Cloud project id.\n",
|
||||
"PROJECT_ID = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# The region you want to launch jobs in.\n",
|
||||
"REGION = \"us-central1\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# The Cloud Storage bucket for storing experiments output. For example 'gs://my_bucket'.\n",
|
||||
"BUCKET_URI = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# The service account for deploying fine tuned model.\n",
|
||||
"# The service account looks like:\n",
|
||||
"# '<account_name>@<project>.iam.gserviceaccount.com'\n",
|
||||
"# Follow step 5 above to create this account.\n",
|
||||
"SERVICE_ACCOUNT = \"\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2cc825514deb"
|
||||
},
|
||||
"source": [
|
||||
"### Define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b42bd4fa2b2d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# The pre-built training docker image. It contains training scripts and models.\n",
|
||||
"TRAIN_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-detectron2-train\"\n",
|
||||
"# The pre-built serving docker image. It contains serving scripts and models.\n",
|
||||
"SERVE_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-detectron2-serve\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "0c250872074f"
|
||||
},
|
||||
"source": [
|
||||
"### Define common functions"
|
||||
"BUCKET_URI = \"gs://\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# @markdown 5. **[Optional]** Set region. If not set, the region will be set automatically according to Colab Enterprise environment.\n",
|
||||
"\n",
|
||||
"REGION = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"import base64\n",
|
||||
"import datetime\n",
|
||||
"import importlib\n",
|
||||
"# Import the necessary packages\n",
|
||||
"import json\n",
|
||||
"import os\n",
|
||||
"import uuid\n",
|
||||
"from typing import Dict, List, Union\n",
|
||||
"\n",
|
||||
"import cv2\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"from google.protobuf import json_format\n",
|
||||
"from google.protobuf.struct_pb2 import Value\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"if os.environ.get(\"VERTEX_PRODUCT\") != \"COLAB_ENTERPRISE\":\n",
|
||||
" ! pip install --upgrade tensorflow\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"common_util = importlib.import_module(\n",
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
"PROJECT_ID = os.environ[\"GOOGLE_CLOUD_PROJECT\"]\n",
|
||||
"\n",
|
||||
"# Get the default region for launching jobs.\n",
|
||||
"if not REGION:\n",
|
||||
" if not os.environ.get(\"GOOGLE_CLOUD_REGION\"):\n",
|
||||
" raise ValueError(\n",
|
||||
" \"REGION must be set. See\"\n",
|
||||
" \" https://cloud.google.com/vertex-ai/docs/general/locations for\"\n",
|
||||
" \" available cloud locations.\"\n",
|
||||
" )\n",
|
||||
" REGION = os.environ[\"GOOGLE_CLOUD_REGION\"]\n",
|
||||
"\n",
|
||||
"# Enable the Vertex AI API and Compute Engine API, if not already.\n",
|
||||
"print(\"Enabling Vertex AI API and Compute Engine API.\")\n",
|
||||
"! gcloud services enable aiplatform.googleapis.com compute.googleapis.com\n",
|
||||
"\n",
|
||||
"# Cloud Storage bucket for storing the experiment artifacts.\n",
|
||||
"# A unique GCS bucket will be created for the purpose of this notebook. If you\n",
|
||||
"# prefer using your own GCS bucket, change the value yourself below.\n",
|
||||
"now = datetime.datetime.now().strftime(\"%Y%m%d%H%M%S\")\n",
|
||||
"BUCKET_NAME = \"/\".join(BUCKET_URI.split(\"/\")[:3])\n",
|
||||
"\n",
|
||||
"if BUCKET_URI is None or BUCKET_URI.strip() == \"\" or BUCKET_URI == \"gs://\":\n",
|
||||
" BUCKET_URI = f\"gs://{PROJECT_ID}-tmp-{now}-{str(uuid.uuid4())[:4]}\"\n",
|
||||
" BUCKET_NAME = \"/\".join(BUCKET_URI.split(\"/\")[:3])\n",
|
||||
" ! gsutil mb -l {REGION} {BUCKET_URI}\n",
|
||||
"else:\n",
|
||||
" assert BUCKET_URI.startswith(\"gs://\"), \"BUCKET_URI must start with `gs://`.\"\n",
|
||||
" shell_output = ! gsutil ls -Lb {BUCKET_NAME} | grep \"Location constraint:\" | sed \"s/Location constraint://\"\n",
|
||||
" bucket_region = shell_output[0].strip().lower()\n",
|
||||
" if bucket_region != REGION:\n",
|
||||
" raise ValueError(\n",
|
||||
" \"Bucket region %s is different from notebook region %s\"\n",
|
||||
" % (bucket_region, REGION)\n",
|
||||
" )\n",
|
||||
"print(f\"Using this GCS Bucket: {BUCKET_URI}\")\n",
|
||||
"\n",
|
||||
"STAGING_BUCKET = os.path.join(BUCKET_URI, \"temporal\")\n",
|
||||
"MODEL_BUCKET = os.path.join(BUCKET_URI, \"detectron2\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Initialize Vertex AI API.\n",
|
||||
"print(\"Initializing Vertex AI API.\")\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=STAGING_BUCKET)\n",
|
||||
"\n",
|
||||
"# Gets the default SERVICE_ACCOUNT.\n",
|
||||
"shell_output = ! gcloud projects describe $PROJECT_ID\n",
|
||||
"project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
|
||||
"SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
|
||||
"print(\"Using this default Service Account:\", SERVICE_ACCOUNT)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Provision permissions to the SERVICE_ACCOUNT with the GCS bucket\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.admin $BUCKET_NAME\n",
|
||||
"\n",
|
||||
"! gcloud config set project $PROJECT_ID\n",
|
||||
"! gcloud projects add-iam-policy-binding --no-user-output-enabled {PROJECT_ID} --member=serviceAccount:{SERVICE_ACCOUNT} --role=\"roles/storage.admin\"\n",
|
||||
"! gcloud projects add-iam-policy-binding --no-user-output-enabled {PROJECT_ID} --member=serviceAccount:{SERVICE_ACCOUNT} --role=\"roles/aiplatform.user\"\n",
|
||||
"import vertexai\n",
|
||||
"\n",
|
||||
"vertexai.init(\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "XcYUGwr-AJGY"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import json\n",
|
||||
"import os\n",
|
||||
"from typing import Dict, List, Union\n",
|
||||
"# @title Define helper functions and constants\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"from google.protobuf import json_format\n",
|
||||
"from google.protobuf.struct_pb2 import Value\n",
|
||||
"# The pre-built training docker image. It contains training scripts and models.\n",
|
||||
"TRAIN_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-detectron2-train\"\n",
|
||||
"# The pre-built serving docker image. It contains serving scripts and models.\n",
|
||||
"SERVE_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-detectron2-serve\"\n",
|
||||
"\n",
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def gcs_fuse_path(path: str) -> str:\n",
|
||||
@@ -284,7 +273,6 @@
|
||||
" return path\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Training\n",
|
||||
"def upload_model(\n",
|
||||
" project: str,\n",
|
||||
" location: str,\n",
|
||||
@@ -348,7 +336,6 @@
|
||||
"\n",
|
||||
"\n",
|
||||
"# Prediction\n",
|
||||
"import base64\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_prediction_instances(local_test_filepath):\n",
|
||||
@@ -803,36 +790,31 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "RB_xY9ipr7ZU"
|
||||
"id": "3E5rmCWMiSFG"
|
||||
},
|
||||
"source": [
|
||||
"### (Optional) Download and prepare Balloon dataset\n",
|
||||
"\n",
|
||||
"You only need this step if you do not have your own dataset and want to use the Balloon dataset as a demo. If using your own dataset, convert it into [COCO format](https://opencv.org/introduction-to-the-coco-dataset/)."
|
||||
"## Optional - Download a sample dataset"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "RcoOH0yEpJQ7"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Download Balloon data.\n",
|
||||
"# @title Download the balloon dataset\n",
|
||||
"\n",
|
||||
"# @markdown This step is only necessary if you don't have your own dataset and wish to use the Balloon dataset for demonstration purposes.\n",
|
||||
"\n",
|
||||
"# @markdown In case you are using your own dataset, kindly convert it to [COCO format](https://opencv.org/introduction-to-the-coco-dataset/).\n",
|
||||
"\n",
|
||||
"!wget https://github.com/matterport/Mask_RCNN/releases/download/v2.1/balloon_dataset.zip\n",
|
||||
"!unzip balloon_dataset.zip > /dev/null"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "E5StJ42NpJQ8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"local_balloon_data_directory = \"balloon\" # @param {type:\"string\"}\n",
|
||||
"!unzip balloon_dataset.zip > /dev/null\n",
|
||||
"\n",
|
||||
"local_balloon_data_directory = \"balloon\"\n",
|
||||
"BALLOON_DATA_GCS_PATH = os.path.join(BUCKET_URI, \"balloon_dataset\")"
|
||||
]
|
||||
},
|
||||
@@ -840,12 +822,12 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "ah0lXyvXpJQ8"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Convert Balloon data to COCO format\n",
|
||||
"import cv2\n",
|
||||
"# @title Convert Balloon data to COCO format\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def save_coco_format_json(img_dir, output_coco_format_json_filename):\n",
|
||||
@@ -921,15 +903,12 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "0OykIZen9gCC"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Move Balloon data from local directory to Cloud Storage.\n",
|
||||
"\n",
|
||||
"import glob\n",
|
||||
"\n",
|
||||
"from google.cloud import storage\n",
|
||||
"# @title Upload the Balloon data Cloud Storage.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_bucket_and_blob_name(filepath):\n",
|
||||
@@ -940,18 +919,7 @@
|
||||
"\n",
|
||||
"def upload_local_dir_to_gcs(local_dir_path, gcs_dir_path):\n",
|
||||
" \"\"\"Uploads files in a local directory to a GCS directory.\"\"\"\n",
|
||||
" client = storage.Client()\n",
|
||||
" bucket_name = gcs_dir_path.split(\"/\")[2]\n",
|
||||
" bucket = client.get_bucket(bucket_name)\n",
|
||||
" for local_file in glob.glob(local_dir_path + \"/**\"):\n",
|
||||
" if not os.path.isfile(local_file):\n",
|
||||
" continue\n",
|
||||
" filename = local_file[1 + len(local_dir_path) :]\n",
|
||||
" gcs_file_path = os.path.join(gcs_dir_path, filename)\n",
|
||||
" _, blob_name = get_bucket_and_blob_name(gcs_file_path)\n",
|
||||
" blob = bucket.blob(blob_name)\n",
|
||||
" blob.upload_from_filename(local_file)\n",
|
||||
" print(\"Copied {} to {}.\".format(local_file, gcs_file_path))\n",
|
||||
" ! gcloud storage cp -R $local_dir_path $gcs_dir_path\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"upload_local_dir_to_gcs(\n",
|
||||
@@ -967,30 +935,31 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "aCpLmWPMpJQ8"
|
||||
"id": "NBv28PhtifaM"
|
||||
},
|
||||
"source": [
|
||||
"## Finetune with Detectron2\n",
|
||||
"\n",
|
||||
"You will use the Vertex AI SDK to create and run the training job with the model-garden detectron2 training docker. You can choose one of the Faster R-CNN, RetinaNet, or Mask R-CNN models to finetune by uncommenting the corresponding code sections below. The training uses one V100 GPU and runs for around 3 mins once the training job begins."
|
||||
"## Finetune"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "aec22792ee84"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from datetime import datetime\n",
|
||||
"\n",
|
||||
"TIMESTAMP = datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n",
|
||||
"# @title Finetune with Detectron2\n",
|
||||
"# @markdown You will use the Vertex AI SDK to create and run the training job with the model-garden detectron2 training docker. You can choose one of the Faster R-CNN, RetinaNet, or Mask R-CNN models to finetune by uncommenting the corresponding code sections below. The training uses one V100 GPU and runs for around 3 mins once the training job begins.\n",
|
||||
"TIMESTAMP = datetime.datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n",
|
||||
"JOB_NAME = \"detectron2_balloon_\" + TIMESTAMP\n",
|
||||
"\n",
|
||||
"NUM_GPU = 1\n",
|
||||
"container_uri = TRAIN_DOCKER_URI\n",
|
||||
"staging_bucket = os.path.join(BUCKET_URI, \"training/temporal\")\n",
|
||||
"TRAINING_ACCELERATOR_TYPE = \"NVIDIA_TESLA_V100\"\n",
|
||||
"TRAINING_MACHINE_TYPE = \"n1-standard-4\"\n",
|
||||
"TRAINING_AACCELERATOR_COUNT = 1\n",
|
||||
"\n",
|
||||
"# Dataset and output directory related parameters.\n",
|
||||
"train_dataset_name = \"balloon_train\" # @param {type:\"string\"}\n",
|
||||
@@ -1082,7 +1051,7 @@
|
||||
" \"--lr\",\n",
|
||||
" f\"{lr}\",\n",
|
||||
" \"--num-gpus\",\n",
|
||||
" f\"{NUM_GPU}\",\n",
|
||||
" f\"{TRAINING_AACCELERATOR_COUNT}\",\n",
|
||||
" \"--output_dir\",\n",
|
||||
" f\"{gcs_fuse_path(output_dir)}\",\n",
|
||||
" \"--config-file\",\n",
|
||||
@@ -1095,58 +1064,69 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "2ELphfgj1f3Q"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Create and run training job.\n",
|
||||
"# @title Create and run training job.\n",
|
||||
"# Click on the generated link in the output under \"View backing custom job:\" to see your run in the Cloud Console.\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=staging_bucket)\n",
|
||||
"job = aiplatform.CustomContainerTrainingJob(\n",
|
||||
" display_name=JOB_NAME,\n",
|
||||
" container_uri=container_uri,\n",
|
||||
")\n",
|
||||
"model = job.run(\n",
|
||||
"\n",
|
||||
"common_util.check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
" region=REGION,\n",
|
||||
" accelerator_type=TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" accelerator_count=TRAINING_AACCELERATOR_COUNT,\n",
|
||||
" is_for_training=True,\n",
|
||||
")\n",
|
||||
"LABEL = \"detectron2\"\n",
|
||||
"\n",
|
||||
"models[LABEL] = job.run(\n",
|
||||
" args=docker_args_list,\n",
|
||||
" base_output_dir=f\"{output_dir}\",\n",
|
||||
" replica_count=1,\n",
|
||||
" machine_type=\"n1-standard-4\",\n",
|
||||
" accelerator_type=\"NVIDIA_TESLA_V100\",\n",
|
||||
" accelerator_count=NUM_GPU,\n",
|
||||
" machine_type=TRAINING_MACHINE_TYPE,\n",
|
||||
" accelerator_type=TRAINING_ACCELERATOR_TYPE,\n",
|
||||
" accelerator_count=TRAINING_AACCELERATOR_COUNT,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "g0BGaofgsMsy"
|
||||
"id": "1rXvIgegikCd"
|
||||
},
|
||||
"source": [
|
||||
"## Upload and deploy Models\n",
|
||||
"\n",
|
||||
"This section uploads the model to Model Registry and deploys it on an Endpoint resource. The model deployment step will take ~15 minutes to complete. You need to set the model path below from the training output Cloud Storage directory."
|
||||
"## Upload and Deploy model"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "3R_3DgCRD3U-"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Upload models to model registry\n",
|
||||
"from datetime import datetime\n",
|
||||
"# @title Upload models to model registry\n",
|
||||
"\n",
|
||||
"# @markdown This section uploads the model to Model Registry and deploys it on an Endpoint resource. The model deployment step will take ~15 minutes to complete. You need to set the model path below from the training output Cloud Storage directory.\n",
|
||||
"\n",
|
||||
"PRETRAINED_MODEL_PTH_FILE = os.path.join(output_dir, \"model_final.pth\")\n",
|
||||
"PRETRAINED_MODEL_CFG_YAML_FILE = os.path.join(output_dir, \"config.yaml\")\n",
|
||||
"TEST_THRESHOLD = 0.7\n",
|
||||
"PREDICTION_CONTAINER_URI = SERVE_DOCKER_URI\n",
|
||||
"PREDICTION_DISPLAY_NAME = \"upload_detectron2_\" + datetime.now().strftime(\n",
|
||||
"PREDICTION_DISPLAY_NAME = \"upload_detectron2_\" + datetime.datetime.now().strftime(\n",
|
||||
" \"%Y%m%d_%H%M%S\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model = upload_model(\n",
|
||||
"models[LABEL] = upload_model(\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
" display_name=PREDICTION_DISPLAY_NAME,\n",
|
||||
@@ -1163,14 +1143,14 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "NYuQowyZEtxK"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Deploy uploaded models\n",
|
||||
"from datetime import datetime\n",
|
||||
"# @title Deploy uploaded models\n",
|
||||
"\n",
|
||||
"DEPLOYED_NAME = \"deploy_iod_\" + datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n",
|
||||
"DEPLOYED_NAME = \"deploy_iod_\" + datetime.datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n",
|
||||
"MACHINE_TYPE = \"n1-highmem-16\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"TRAFFIC_SPLIT = {\"0\": 100}\n",
|
||||
@@ -1178,34 +1158,34 @@
|
||||
"MIN_NODES = 1\n",
|
||||
"MAX_NODES = 1\n",
|
||||
"\n",
|
||||
"endpoint = model.deploy(\n",
|
||||
"endpoints[LABEL] = models[LABEL].deploy(\n",
|
||||
" deployed_model_display_name=DEPLOYED_NAME,\n",
|
||||
" traffic_split=TRAFFIC_SPLIT,\n",
|
||||
" machine_type=MACHINE_TYPE,\n",
|
||||
" min_replica_count=MIN_NODES,\n",
|
||||
" max_replica_count=MAX_NODES,\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
" system_labels={\n",
|
||||
" \"NOTEBOOK_NAME\": \"model_garden_pytorch_detectron2.ipynb\"\n",
|
||||
" },\n",
|
||||
" system_labels={\"NOTEBOOK_NAME\": \"model_garden_pytorch_detectron2.ipynb\"},\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(\"endpoint id is: \", endpoint.name)"
|
||||
"model = models[LABEL]\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "vbIW9me1F2RY"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Run predictions\n",
|
||||
"# Fill the \"endpoint_id\" from previous step below.\n",
|
||||
"# For example 'endpoint_id = \"8211918096324100096\"'.\n",
|
||||
"# @title Run predictions\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"endpoint_id = endpoint.name\n",
|
||||
"\n",
|
||||
"endpoint_id = \"\" # @param {type:\"string\"}\n",
|
||||
"local_test_filepath = os.path.join(\n",
|
||||
" local_balloon_data_directory, \"val/410488422_5f8991f26e_b.jpg\"\n",
|
||||
")\n",
|
||||
@@ -1255,28 +1235,32 @@
|
||||
"id": "aP9ZYHaeYCaf"
|
||||
},
|
||||
"source": [
|
||||
"## Clean up"
|
||||
"## Clean up resources"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "aP9ZYHaeYCas"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Cleans up endpoint from previous step.\n",
|
||||
"# Or you can overwrite \"endpoint_id\" for a different one.\n",
|
||||
"# @markdown Delete the experiment models and endpoints to recycle the resources\n",
|
||||
"# @markdown and avoid unnecessary continuous charges that may incur.\n",
|
||||
"\n",
|
||||
"aip_endpoint_name = f\"projects/{PROJECT_ID}/locations/{REGION}/endpoints/{endpoint_id}\"\n",
|
||||
"endpoint = aiplatform.Endpoint(aip_endpoint_name)\n",
|
||||
"# Undeploy model and delete endpoint.\n",
|
||||
"for endpoint in endpoints.values():\n",
|
||||
" endpoint.delete(force=True)\n",
|
||||
"\n",
|
||||
"# Undeploy model and delete endpoint\n",
|
||||
"endpoint.delete(force=True)\n",
|
||||
"# Delete models.\n",
|
||||
"for model in models.values():\n",
|
||||
" model.delete()\n",
|
||||
"\n",
|
||||
"# Delete the model resource\n",
|
||||
"model.delete()"
|
||||
"delete_bucket = False # @param {type:\"boolean\"}\n",
|
||||
"if delete_bucket:\n",
|
||||
" ! gsutil -m rm -r $BUCKET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_dia_1_6b.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_dia_1_6b.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -117,7 +117,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
@@ -190,7 +190,7 @@
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -199,7 +199,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" \u003c/td\u003e\n",
|
||||
" \u003ctd\u003e\n",
|
||||
" \u003ca href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_dolly_v2.ipynb\"\u003e\n",
|
||||
" \u003cimg src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\"\u003e\n",
|
||||
" \u003cimg src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\"\u003e\n",
|
||||
" View on GitHub\n",
|
||||
" \u003c/a\u003e\n",
|
||||
" \u003c/td\u003e\n",
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_falcon_evaluation.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
+213
-159
@@ -4,11 +4,12 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "7d9bbf86da5e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2023 Google LLC\n",
|
||||
"# Copyright 2025 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -31,19 +32,23 @@
|
||||
"source": [
|
||||
"# Vertex AI Model Garden - Falcon Instruct Deployment\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_falcon_instruct_deployment.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/colab/import/https:%2F%2Fraw.githubusercontent.com%2FGoogleCloudPlatform%2Fvertex-ai-samples%2Fmain%2Fnotebooks%2Fcommunity%2Fmodel_garden%2Fmodel_garden_pytorch_falcon_instruct_deployment.ipynb\">\n",
|
||||
" <img alt=\"Google Cloud Colab Enterprise logo\" src=\"https://lh3.googleusercontent.com/JmcxdQi-qOpctIvWKgPtrzZdJJK-J3sWE1RsfjZNwshCFgE_9fULcNpuXYTilIR2hjwN\" width=\"32px\"><br> Run in Colab Enterprise\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_falcon_instruct_deployment.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>"
|
||||
"</tr></tbody></table>"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -67,6 +72,10 @@
|
||||
"| [tiiuae/falcon-7b-instruct](https://huggingface.co/tiiuae/falcon-7b-instruct) |\n",
|
||||
"| [tiiuae/falcon-40b-instruct](https://huggingface.co/tiiuae/falcon-40b-instruct) |\n",
|
||||
"\n",
|
||||
"### File a bug\n",
|
||||
"\n",
|
||||
"File a bug on [GitHub](https://github.com/GoogleCloudPlatform/vertex-ai-samples/issues/new) if you encounter any issue with the notebook.\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
@@ -74,7 +83,7 @@
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -83,136 +92,7 @@
|
||||
"id": "jjgxHUBnG5ym"
|
||||
},
|
||||
"source": [
|
||||
"## Run the notebook"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "855d6b96f291"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Setup Google Cloud project\n",
|
||||
"\n",
|
||||
"# @markdown 1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"# @markdown 2. [Optional] [Create a Cloud Storage bucket](https://cloud.google.com/storage/docs/creating-buckets) for storing experiment outputs. Set the BUCKET_URI for the experiment environment. The specified Cloud Storage bucket (`BUCKET_URI`) should be located in the same region as where the notebook was launched. Note that a multi-region bucket (eg. \"us\") is not considered a match for a single region covered by the multi-region range (eg. \"us-central1\"). If not set, a unique GCS bucket will be created instead.\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"import os\n",
|
||||
"from datetime import datetime\n",
|
||||
"from typing import Tuple\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
"PROJECT_ID = os.environ[\"GOOGLE_CLOUD_PROJECT\"]\n",
|
||||
"\n",
|
||||
"# Get the default region for launching jobs.\n",
|
||||
"REGION = os.environ[\"GOOGLE_CLOUD_REGION\"]\n",
|
||||
"\n",
|
||||
"# Enable the Vertex AI API and Compute Engine API, if not already.\n",
|
||||
"print(\"Enabling Vertex AI API and Compute Engine API.\")\n",
|
||||
"! gcloud services enable aiplatform.googleapis.com compute.googleapis.com\n",
|
||||
"\n",
|
||||
"# Cloud Storage bucket for storing the experiment artifacts.\n",
|
||||
"# A unique GCS bucket will be created for the purpose of this notebook. If you\n",
|
||||
"# prefer using your own GCS bucket, please change the value yourself below.\n",
|
||||
"now = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n",
|
||||
"BUCKET_URI = \"gs://\" # @param {type:\"string\"}\n",
|
||||
"if BUCKET_URI is None or BUCKET_URI.strip() == \"\" or BUCKET_URI == \"gs://\":\n",
|
||||
" # Create a unique GCS bucket for this notebook if not specified\n",
|
||||
" BUCKET_URI = f\"gs://{PROJECT_ID}-tmp-{now}\"\n",
|
||||
" ! gsutil mb -l {REGION} {BUCKET_URI}\n",
|
||||
"else:\n",
|
||||
" assert BUCKET_URI.startswith(\"gs://\"), \"BUCKET_URI must start with `gs://`.\"\n",
|
||||
" BUCKET_NAME = \"/\".join(BUCKET_URI.split(\"/\")[:3])\n",
|
||||
" shell_output = ! gsutil ls -Lb {BUCKET_NAME} | grep \"Location constraint:\" | sed \"s/Location constraint://\"\n",
|
||||
" bucket_region = shell_output[0].strip().lower()\n",
|
||||
" if bucket_region != REGION:\n",
|
||||
" raise ValueError(\n",
|
||||
" \"Bucket region %s is different from notebook region %s\"\n",
|
||||
" % (bucket_region, REGION)\n",
|
||||
" )\n",
|
||||
"print(f\"Using this GCS Bucket: {BUCKET_URI}\")\n",
|
||||
"\n",
|
||||
"STAGING_BUCKET = os.path.join(BUCKET_URI, \"temporal\")\n",
|
||||
"\n",
|
||||
"# Set up default SERVICE_ACCOUNT\n",
|
||||
"SERVICE_ACCOUNT = None\n",
|
||||
"shell_output = ! gcloud projects describe $PROJECT_ID\n",
|
||||
"project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
|
||||
"SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
|
||||
"print(\"Using this default Service Account:\", SERVICE_ACCOUNT)\n",
|
||||
"\n",
|
||||
"# Provision permissions to the SERVICE_ACCOUNT with the GCS bucket\n",
|
||||
"BUCKET_NAME = \"/\".join(BUCKET_URI.split(\"/\")[:3])\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.admin $BUCKET_NAME\n",
|
||||
"\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=STAGING_BUCKET)\n",
|
||||
"\n",
|
||||
"# The pre-built training and serving docker images.\n",
|
||||
"TRAIN_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-peft-train:20240409_0936_RC00\"\n",
|
||||
"VLLM_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-vllm-serve:20240410_0916_RC00\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_job_name_with_datetime(prefix: str) -> str:\n",
|
||||
" \"\"\"Gets the job name with date time when triggering training or deployment\n",
|
||||
" jobs in Vertex AI.\n",
|
||||
" \"\"\"\n",
|
||||
" return prefix + datetime.now().strftime(\"_%Y%m%d_%H%M%S\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def deploy_model(\n",
|
||||
" model_name: str,\n",
|
||||
" model_id: str,\n",
|
||||
" service_account: str,\n",
|
||||
" machine_type: str = \"n1-standard-8\",\n",
|
||||
" accelerator_type: str = \"NVIDIA_TESLA_V100\",\n",
|
||||
" accelerator_count: int = 1,\n",
|
||||
") -> Tuple[aiplatform.Model, aiplatform.Endpoint]:\n",
|
||||
" \"\"\"Deploys trained models with vLLM into Vertex AI.\"\"\"\n",
|
||||
" endpoint = aiplatform.Endpoint.create(display_name=f\"{model_name}-endpoint\")\n",
|
||||
"\n",
|
||||
" vllm_args = [\n",
|
||||
" \"--host=0.0.0.0\",\n",
|
||||
" \"--port=7080\",\n",
|
||||
" f\"--model={model_id}\",\n",
|
||||
" f\"--tensor-parallel-size={accelerator_count}\",\n",
|
||||
" \"--swap-space=16\",\n",
|
||||
" \"--gpu-memory-utilization=0.9\",\n",
|
||||
" \"--disable-log-stats\",\n",
|
||||
" \"--dtype=float16\",\n",
|
||||
" \"--trust-remote-code\",\n",
|
||||
" ]\n",
|
||||
" serving_env = {\"MODEL_ID\": model_id, \"DEPLOY_SOURCE\": \"notebook\"}\n",
|
||||
"\n",
|
||||
" model = aiplatform.Model.upload(\n",
|
||||
" display_name=model_name,\n",
|
||||
" serving_container_image_uri=VLLM_DOCKER_URI,\n",
|
||||
" serving_container_command=[\"python\", \"-m\", \"vllm.entrypoints.api_server\"],\n",
|
||||
" serving_container_args=vllm_args,\n",
|
||||
" serving_container_ports=[7080],\n",
|
||||
" serving_container_predict_route=\"/generate\",\n",
|
||||
" serving_container_health_route=\"/ping\",\n",
|
||||
" serving_container_environment_variables=serving_env,\n",
|
||||
" model_garden_source_model_name=\"publishers/tiiuae/models/falcon-instruct-7b-peft\"\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" model.deploy(\n",
|
||||
" endpoint=endpoint,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" deploy_request_timeout=1800,\n",
|
||||
" service_account=service_account,\n",
|
||||
" system_labels={\n",
|
||||
" \"NOTEBOOK_NAME\": \"model_garden_pytorch_falcon_instruct_deployment.ipynb\"\n",
|
||||
" },\n",
|
||||
" )\n",
|
||||
" return model, endpoint"
|
||||
"## Before you begin"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -220,22 +100,78 @@
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "Uak1pyEeExYM"
|
||||
"id": "855d6b96f291"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Deploy prebuilt Falcon Instruct models\n",
|
||||
"# @title Setup Google Cloud project\n",
|
||||
"# @markdown 1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"# @markdown This section deploys prebuilt Falcon Instruct models on the Endpoint. The model deployment step will take 15 minutes to 40 minutes to complete.\n",
|
||||
"# @markdown 2. **[Optional]** Set region. If not set, the region will be set automatically according to Colab Enterprise environment.\n",
|
||||
"\n",
|
||||
"# @markdown The peak GPU memory usages for [tiiuae/falcon-7b-instruct](https://huggingface.co/tiiuae/falcon-7b-instruct), and [tiiuae/falcon-40b-instruct](https://huggingface.co/tiiuae/falcon-40b-instruct) are ~15.5G and ~84G separately with the default settings. Please adjust the machine type, accelerator type and accelerator count accordingly. We use V100 in deployments as an example. Note that V100 serving generally offers better throughput and latency performance than L4 serving, while L4 serving is generally more cost efficient than V100 serving. The serving efficiency of V100 and L4 GPUs is inferior to that of A100 GPUs, but V100 and L4 GPUs are nevertheless good serving solutions if you do not have A100 quota.\n",
|
||||
"REGION = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# @markdown We use the PEFT serving images to deploy prebuilt Falcon Instruct models, by setting finetuning LoRA model paths as empty.\n",
|
||||
"# @markdown 3. If you want to run predictions with A100 80GB or H100 GPUs, we recommend using the regions listed below. **NOTE:** Make sure you have associated quota in selected regions. Click the links to see your current quota for each GPU type: [Nvidia A100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_a100_80gb_gpus), [Nvidia H100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_h100_gpus). You can request for quota following the instructions at [\"Request a higher quota\"](https://cloud.google.com/docs/quota/view-manage#requesting_higher_quota).\n",
|
||||
"\n",
|
||||
"# @markdown > | Machine Type | Accelerator Type | Recommended Regions |\n",
|
||||
"# @markdown | ----------- | ----------- | ----------- |\n",
|
||||
"# @markdown | a2-ultragpu-1g | 1 NVIDIA_A100_80GB | us-central1, us-east4, europe-west4, asia-southeast1, us-east4 |\n",
|
||||
"# @markdown | a3-highgpu-2g | 2 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-4g | 4 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
"from typing import Tuple\n",
|
||||
"\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"\n",
|
||||
"if os.environ.get(\"VERTEX_PRODUCT\") != \"COLAB_ENTERPRISE\":\n",
|
||||
" ! pip install --upgrade tensorflow\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"common_util = importlib.import_module(\n",
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Find Vertex AI supported accelerators and regions in:\n",
|
||||
"# https://cloud.google.com/vertex-ai/docs/predictions/configure-compute\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
"PROJECT_ID = os.environ[\"GOOGLE_CLOUD_PROJECT\"]\n",
|
||||
"\n",
|
||||
"# Get the default region for launching jobs.\n",
|
||||
"if not REGION:\n",
|
||||
" REGION = os.environ[\"GOOGLE_CLOUD_REGION\"]\n",
|
||||
"\n",
|
||||
"# Initialize Vertex AI API.\n",
|
||||
"print(\"Initializing Vertex AI API.\")\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION)\n",
|
||||
"\n",
|
||||
"! gcloud config set project $PROJECT_ID\n",
|
||||
"import vertexai\n",
|
||||
"\n",
|
||||
"vertexai.init(\n",
|
||||
" project=PROJECT_ID,\n",
|
||||
" location=REGION,\n",
|
||||
")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "NA6_KOw8Scu7"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Select the model parameters\n",
|
||||
"\n",
|
||||
"# @markdown Set the prebuilt model id.\n",
|
||||
"prebuilt_model_id = \"tiiuae/falcon-7b-instruct\" # @param [\"tiiuae/falcon-7b-instruct\", \"tiiuae/falcon-40b-instruct\"]\n",
|
||||
@@ -245,16 +181,19 @@
|
||||
"\n",
|
||||
"# Sets V100 (16G) to deploy tiiuae/falcon-7b-instruct or tiiuae/falcon-40b-instruct.\n",
|
||||
"# If A100 is not available, you may deploy tiiuae/falcon-40b-instruct with\n",
|
||||
"# multiple V100s. Please keep in mind that the efficiency of serving with\n",
|
||||
"# multiple V100s. Kindly keep in mind that the efficiency of serving with\n",
|
||||
"# multiple V100s is inferior to that of serving with A100s.\n",
|
||||
"# Compared with L4, V100 serving can have better throughput and latency.\n",
|
||||
"\n",
|
||||
"# Sets L4 (24G) to deploy tiiuae/falcon-7b-instruct or tiiuae/falcon-40b-instruct.\n",
|
||||
"# If A100 is not available, you may deploy tiiuae/falcon-40b-instruct with\n",
|
||||
"# multiple L4s. Please keep in mind that the efficiency of serving with\n",
|
||||
"# multiple L4s. Kindly keep in mind that the efficiency of serving with\n",
|
||||
"# multiple L4s is inferior to that of serving with A100s.\n",
|
||||
"# Compared with V100, L4 serving can be more cost efficient.\n",
|
||||
"\n",
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"if \"7b\" in prebuilt_model_id:\n",
|
||||
" if accelerator_type == \"NVIDIA_TESLA_A100\":\n",
|
||||
" machine_type = \"a2-highgpu-1g\"\n",
|
||||
@@ -282,17 +221,128 @@
|
||||
" else:\n",
|
||||
" raise ValueError(\n",
|
||||
" f\"Recommended GPU setting not found for: {accelerator_type} and {prebuilt_model_id}.\"\n",
|
||||
" )\n",
|
||||
" )"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "QuQBDE49UgUq"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title [Option 1] Deploy with SDK\n",
|
||||
"\n",
|
||||
"model, endpoint = deploy_model(\n",
|
||||
" model_name=get_job_name_with_datetime(prefix=\"falcon-instruct-serve\"),\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"LABEL = \"sdk-deploy\"\n",
|
||||
"model = model_garden.OpenModel(prebuilt_model_id)\n",
|
||||
"endpoints[LABEL] = model.deploy(use_dedicated_endpoint=use_dedicated_endpoint)\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "Uak1pyEeExYM"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title [Option 2] Deploy prebuilt Falcon Instruct models\n",
|
||||
"\n",
|
||||
"# @markdown This section deploys prebuilt Falcon Instruct models on the Endpoint. The model deployment step will take 15 minutes to 40 minutes to complete.\n",
|
||||
"\n",
|
||||
"# @markdown The peak GPU memory usages for [tiiuae/falcon-7b-instruct](https://huggingface.co/tiiuae/falcon-7b-instruct), and [tiiuae/falcon-40b-instruct](https://huggingface.co/tiiuae/falcon-40b-instruct) are ~15.5G and ~84G separately with the default settings. Kindly adjust the machine type, accelerator type and accelerator count accordingly. We use V100 in deployments as an example. Note that V100 serving generally offers better throughput and latency performance than L4 serving, while L4 serving is generally more cost efficient than V100 serving. The serving efficiency of V100 and L4 GPUs is inferior to that of A100 GPUs, but V100 and L4 GPUs are nevertheless good serving solutions if you do not have A100 quota.\n",
|
||||
"\n",
|
||||
"# @markdown We use the PEFT serving images to deploy prebuilt Falcon Instruct models, by setting finetuning LoRA model paths as empty.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Find Vertex AI supported accelerators and regions in:\n",
|
||||
"# https://cloud.google.com/vertex-ai/docs/predictions/configure-compute\n",
|
||||
"\n",
|
||||
"# The pre-built training and serving docker images.\n",
|
||||
"VLLM_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-vllm-serve:20240410_0916_RC00\"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def deploy_model(\n",
|
||||
" model_name: str,\n",
|
||||
" model_id: str,\n",
|
||||
" machine_type: str = \"n1-standard-8\",\n",
|
||||
" accelerator_type: str = \"NVIDIA_TESLA_V100\",\n",
|
||||
" accelerator_count: int = 1,\n",
|
||||
" use_dedicated_endpoint: bool = False,\n",
|
||||
") -> Tuple[aiplatform.Model, aiplatform.Endpoint]:\n",
|
||||
" \"\"\"Deploys trained models with vLLM into Vertex AI.\"\"\"\n",
|
||||
" endpoint = aiplatform.Endpoint.create(\n",
|
||||
" display_name=f\"{model_name}-endpoint\",\n",
|
||||
" dedicated_endpoint_enabled=use_dedicated_endpoint,\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" vllm_args = [\n",
|
||||
" \"--host=0.0.0.0\",\n",
|
||||
" \"--port=7080\",\n",
|
||||
" f\"--model={model_id}\",\n",
|
||||
" f\"--tensor-parallel-size={accelerator_count}\",\n",
|
||||
" \"--swap-space=16\",\n",
|
||||
" \"--gpu-memory-utilization=0.9\",\n",
|
||||
" \"--disable-log-stats\",\n",
|
||||
" \"--dtype=float16\",\n",
|
||||
" \"--trust-remote-code\",\n",
|
||||
" ]\n",
|
||||
" serving_env = {\"MODEL_ID\": model_id, \"DEPLOY_SOURCE\": \"notebook\"}\n",
|
||||
"\n",
|
||||
" model = aiplatform.Model.upload(\n",
|
||||
" display_name=model_name,\n",
|
||||
" serving_container_image_uri=VLLM_DOCKER_URI,\n",
|
||||
" serving_container_command=[\"python\", \"-m\", \"vllm.entrypoints.api_server\"],\n",
|
||||
" serving_container_args=vllm_args,\n",
|
||||
" serving_container_ports=[7080],\n",
|
||||
" serving_container_predict_route=\"/generate\",\n",
|
||||
" serving_container_health_route=\"/ping\",\n",
|
||||
" serving_container_environment_variables=serving_env,\n",
|
||||
" model_garden_source_model_name=\"publishers/tiiuae/models/falcon-instruct-7b-peft\",\n",
|
||||
" )\n",
|
||||
"\n",
|
||||
" model.deploy(\n",
|
||||
" endpoint=endpoint,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" deploy_request_timeout=1800,\n",
|
||||
" system_labels={\n",
|
||||
" \"NOTEBOOK_NAME\": \"model_garden_pytorch_falcon_instruct_deployment.ipynb\"\n",
|
||||
" },\n",
|
||||
" )\n",
|
||||
" return model, endpoint\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"common_util.check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
" region=REGION,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" is_for_training=False,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"LABEL = \"falcon_instruct\"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model(\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=\"falcon-instruct-serve\"),\n",
|
||||
" model_id=prebuilt_model_id,\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")\n",
|
||||
"print(\"endpoint_name:\", endpoint.name)"
|
||||
"print(\"endpoint_name:\", endpoint.name)\n",
|
||||
"\n",
|
||||
"model = models[LABEL]\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -343,7 +393,10 @@
|
||||
" \"top_k\": top_k,\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"response = endpoint.predict(instances=instances)\n",
|
||||
"response = endpoints[LABEL].predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"for prediction in response.predictions:\n",
|
||||
" print(prediction)"
|
||||
@@ -358,17 +411,18 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Clean up resources\n",
|
||||
"# @title Delete the models and endpoints\n",
|
||||
"\n",
|
||||
"delete_bucket = False # @param {type:\"boolean\"}\n",
|
||||
"if delete_bucket:\n",
|
||||
" ! gsutil -m rm -r $BUCKET_URI\n",
|
||||
"# @markdown Delete the experiment models and endpoints to recycle the resources\n",
|
||||
"# @markdown and avoid unnecessary continuous charges that may incur.\n",
|
||||
"\n",
|
||||
"# Undeploy models and delete endpoints.\n",
|
||||
"endpoint.delete(force=True)\n",
|
||||
"# Undeploy model and delete endpoint.\n",
|
||||
"for endpoint in endpoints.values():\n",
|
||||
" endpoint.delete(force=True)\n",
|
||||
"\n",
|
||||
"# Delete models.\n",
|
||||
"model.delete()"
|
||||
"for model in models.values():\n",
|
||||
" model.delete()"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
+1
-1
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_falcon_instruct_finetuning.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_falcon_instruct_peft.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
+1
-1
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_falcon_instruct_quantization.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_fill_mask.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_flux.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_flux.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -117,7 +117,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
@@ -202,7 +202,7 @@
|
||||
"# @markdown Set `use_dedicated_endpoint` to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint).\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -211,7 +211,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_flux_gradio.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
+1
-1
@@ -40,7 +40,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_gemma_peft_finetuning_hf.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_hidream_i1.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_hidream_i1.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -117,7 +117,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
|
||||
@@ -4,11 +4,12 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "7d9bbf86da5e"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Copyright 2023 Google LLC\n",
|
||||
"# Copyright 2025 Google LLC\n",
|
||||
"#\n",
|
||||
"# Licensed under the Apache License, Version 2.0 (the \"License\");\n",
|
||||
"# you may not use this file except in compliance with the License.\n",
|
||||
@@ -31,25 +32,23 @@
|
||||
"source": [
|
||||
"# Vertex AI Model Garden - ImageBind\n",
|
||||
"\n",
|
||||
"<table align=\"left\">\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://colab.research.google.com/github/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_imagebind.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/colab-logo-32px.png\" alt=\"Colab logo\"> Run in Colab\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_imagebind.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_imagebind.ipynb\">\n",
|
||||
" <img src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" alt=\"Vertex AI logo\">\n",
|
||||
"Open in Vertex AI Workbench\n",
|
||||
" </a> (A Python-3 CPU notebook is recommended)\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</table>"
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/colab/import/https:%2F%2Fraw.githubusercontent.com%2FGoogleCloudPlatform%2Fvertex-ai-samples%2Fmain%2Fnotebooks%2Fcommunity%2Fmodel_garden%2Fmodel_garden_pytorch_imagebind.ipynb\">\n",
|
||||
" <img alt=\"Google Cloud Colab Enterprise logo\" src=\"https://lh3.googleusercontent.com/JmcxdQi-qOpctIvWKgPtrzZdJJK-J3sWE1RsfjZNwshCFgE_9fULcNpuXYTilIR2hjwN\" width=\"32px\"><br> Run in Colab Enterprise\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_imagebind.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -68,6 +67,10 @@
|
||||
"- Deploy the ImageBind to a [Vertex AI Endpoint resource](https://cloud.google.com/vertex-ai/docs/predictions/using-private-endpoints).\n",
|
||||
"- Run online prediction for feature embedding generation and zero-shot classification.\n",
|
||||
"\n",
|
||||
"### File a bug\n",
|
||||
"\n",
|
||||
"File a bug on [GitHub](https://github.com/GoogleCloudPlatform/vertex-ai-samples/issues/new) if you encounter any issue with the notebook.\n",
|
||||
"\n",
|
||||
"### Costs\n",
|
||||
"\n",
|
||||
"This tutorial uses billable components of Google Cloud:\n",
|
||||
@@ -75,7 +78,7 @@
|
||||
"* Vertex AI\n",
|
||||
"* Cloud Storage\n",
|
||||
"\n",
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
"Learn about [Vertex AI pricing](https://cloud.google.com/vertex-ai/pricing), [Cloud Storage pricing](https://cloud.google.com/storage/pricing), and use the [Pricing Calculator](https://cloud.google.com/products/calculator/) to generate a cost estimate based on your projected usage."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -84,9 +87,7 @@
|
||||
"id": "264c07757582"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin\n",
|
||||
"\n",
|
||||
"**NOTE**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands."
|
||||
"## Before you begin"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -103,167 +104,162 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "2707b02ef5df"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import sys\n",
|
||||
"# @title Setup Google Cloud project\n",
|
||||
"\n",
|
||||
"if \"google.colab\" in sys.modules:\n",
|
||||
" ! pip3 install --upgrade google-cloud-aiplatform\n",
|
||||
" from google.colab import auth as google_auth\n",
|
||||
"# @markdown 1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
" google_auth.authenticate_user()\n",
|
||||
"# @markdown 2. **[Optional]** [Create a Cloud Storage bucket](https://cloud.google.com/storage/docs/creating-buckets) for storing experiment outputs. Set the BUCKET_URI for the experiment environment. The specified Cloud Storage bucket (`BUCKET_URI`) should be located in the same region as where the notebook was launched. Note that a multi-region bucket (eg. \"us\") is not considered a match for a single region covered by the multi-region range (eg. \"us-central1\"). If not set, a unique GCS bucket will be created instead.\n",
|
||||
"\n",
|
||||
" # Restart the notebook kernel after installs.\n",
|
||||
" import IPython\n",
|
||||
"\n",
|
||||
" app = IPython.Application.instance()\n",
|
||||
" app.kernel.do_shutdown(True)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "bb7adab99e41"
|
||||
},
|
||||
"source": [
|
||||
"### Setup Google Cloud project\n",
|
||||
"\n",
|
||||
"1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n",
|
||||
"\n",
|
||||
"1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
"1. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component.googleapis.com).\n",
|
||||
"\n",
|
||||
"1. [Create a Cloud Storage bucket](https://cloud.google.com/storage/docs/creating-buckets) for storing experiment outputs.\n",
|
||||
"\n",
|
||||
"1. [Create a service account](https://cloud.google.com/iam/docs/service-accounts-create#iam-service-accounts-create-console) with `Vertex AI User` and `Storage Object Admin` roles for deploying fine tuned model to Vertex AI endpoint."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "6c460088b873"
|
||||
},
|
||||
"source": [
|
||||
"Set the following variables for the experiment environment. The specified Cloud Storage bucket (BUCKET_URI) should be located in the specified region (REGION). Note that a multi-region bucket (eg. \"us\") is not considered a match for a single region covered by the multi-region range (eg. \"us-central1\")."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "855d6b96f291"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Cloud project id.\n",
|
||||
"PROJECT_ID = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# The region you want to launch jobs in.\n",
|
||||
"REGION = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# The Cloud Storage bucket for storing experiments output.\n",
|
||||
"# Start with gs:// prefix, e.g. gs://foo_bucket.\n",
|
||||
"BUCKET_URI = \"gs://\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"! gcloud config set project $PROJECT_ID\n",
|
||||
"# @markdown 3. **[Optional]** Set region. If not set, the region will be set automatically according to Colab Enterprise environment.\n",
|
||||
"\n",
|
||||
"REGION = \"\" # @param {type:\"string\"}\n",
|
||||
"\n",
|
||||
"# @markdown 4. If you want to run predictions with A100 80GB or H100 GPUs, we recommend using the regions listed below. **NOTE:** Make sure you have associated quota in selected regions. Click the links to see your current quota for each GPU type: [Nvidia A100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_a100_80gb_gpus), [Nvidia H100 80GB](https://console.cloud.google.com/iam-admin/quotas?metric=aiplatform.googleapis.com%2Fcustom_model_serving_nvidia_h100_gpus). You can request for quota following the instructions at [\"Request a higher quota\"](https://cloud.google.com/docs/quota/view-manage#requesting_higher_quota).\n",
|
||||
"\n",
|
||||
"# @markdown > | Machine Type | Accelerator Type | Recommended Regions |\n",
|
||||
"# @markdown | ----------- | ----------- | ----------- |\n",
|
||||
"# @markdown | a2-ultragpu-1g | 1 NVIDIA_A100_80GB | us-central1, us-east4, europe-west4, asia-southeast1, us-east4 |\n",
|
||||
"# @markdown | a3-highgpu-2g | 2 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-4g | 4 NVIDIA_H100_80GB | us-west1, asia-southeast1, europe-west4 |\n",
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"import datetime\n",
|
||||
"import importlib\n",
|
||||
"import os\n",
|
||||
"\n",
|
||||
"STAGING_BUCKET = os.path.join(BUCKET_URI, \"temporal\")\n",
|
||||
"DATA_BUCKET = os.path.join(BUCKET_URI, \"data\")\n",
|
||||
"\n",
|
||||
"# The service account looks like:\n",
|
||||
"# '@.iam.gserviceaccount.com'\n",
|
||||
"# Please go to https://cloud.google.com/iam/docs/service-accounts-create#iam-service-accounts-create-console\n",
|
||||
"# and create service account with `Vertex AI User` and `Storage Object Admin` roles.\n",
|
||||
"# The service account for deploying fine tuned model.\n",
|
||||
"SERVICE_ACCOUNT = \"\" # @param {type:\"string\"}"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "e828eb320337"
|
||||
},
|
||||
"source": [
|
||||
"### Initialize Vertex AI API"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "12cd25839741"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"from google.cloud import aiplatform\n",
|
||||
"\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=STAGING_BUCKET)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "2cc825514deb"
|
||||
},
|
||||
"source": [
|
||||
"### Define constants"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "b42bd4fa2b2d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# The pre-built serving docker image.\n",
|
||||
"PREDICTION_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-imagebind-serve\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "0c250872074f"
|
||||
},
|
||||
"source": [
|
||||
"### Define common functions"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "354da31189dc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import os\n",
|
||||
"from datetime import datetime\n",
|
||||
"import uuid\n",
|
||||
"\n",
|
||||
"import numpy as np\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"\n",
|
||||
"if os.environ.get(\"VERTEX_PRODUCT\") != \"COLAB_ENTERPRISE\":\n",
|
||||
" ! pip install --upgrade tensorflow\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"common_util = importlib.import_module(\n",
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def get_job_name_with_datetime(prefix: str) -> str:\n",
|
||||
" \"\"\"Gets the job name with date time when triggering deployment jobs.\"\"\"\n",
|
||||
" return prefix + datetime.now().strftime(\"_%Y%m%d_%H%M%S\")\n",
|
||||
"# Get the default cloud project id.\n",
|
||||
"PROJECT_ID = os.environ[\"GOOGLE_CLOUD_PROJECT\"]\n",
|
||||
"\n",
|
||||
"# Get the default region for launching jobs.\n",
|
||||
"if not REGION:\n",
|
||||
" if not os.environ.get(\"GOOGLE_CLOUD_REGION\"):\n",
|
||||
" raise ValueError(\n",
|
||||
" \"REGION must be set. See\"\n",
|
||||
" \" https://cloud.google.com/vertex-ai/docs/general/locations for\"\n",
|
||||
" \" available cloud locations.\"\n",
|
||||
" )\n",
|
||||
" REGION = os.environ[\"GOOGLE_CLOUD_REGION\"]\n",
|
||||
"\n",
|
||||
"# Enable the Vertex AI API and Compute Engine API, if not already.\n",
|
||||
"print(\"Enabling Vertex AI API and Compute Engine API.\")\n",
|
||||
"! gcloud services enable aiplatform.googleapis.com compute.googleapis.com\n",
|
||||
"\n",
|
||||
"# Cloud Storage bucket for storing the experiment artifacts.\n",
|
||||
"# A unique GCS bucket will be created for the purpose of this notebook. If you\n",
|
||||
"# prefer using your own GCS bucket, change the value yourself below.\n",
|
||||
"now = datetime.datetime.now().strftime(\"%Y%m%d%H%M%S\")\n",
|
||||
"BUCKET_NAME = \"/\".join(BUCKET_URI.split(\"/\")[:3])\n",
|
||||
"\n",
|
||||
"if BUCKET_URI is None or BUCKET_URI.strip() == \"\" or BUCKET_URI == \"gs://\":\n",
|
||||
" BUCKET_URI = f\"gs://{PROJECT_ID}-tmp-{now}-{str(uuid.uuid4())[:4]}\"\n",
|
||||
" BUCKET_NAME = \"/\".join(BUCKET_URI.split(\"/\")[:3])\n",
|
||||
" ! gsutil mb -l {REGION} {BUCKET_URI}\n",
|
||||
"else:\n",
|
||||
" assert BUCKET_URI.startswith(\"gs://\"), \"BUCKET_URI must start with `gs://`.\"\n",
|
||||
" shell_output = ! gsutil ls -Lb {BUCKET_NAME} | grep \"Location constraint:\" | sed \"s/Location constraint://\"\n",
|
||||
" bucket_region = shell_output[0].strip().lower()\n",
|
||||
" if bucket_region != REGION:\n",
|
||||
" raise ValueError(\n",
|
||||
" \"Bucket region %s is different from notebook region %s\"\n",
|
||||
" % (bucket_region, REGION)\n",
|
||||
" )\n",
|
||||
"print(f\"Using this GCS Bucket: {BUCKET_URI}\")\n",
|
||||
"\n",
|
||||
"STAGING_BUCKET = os.path.join(BUCKET_URI, \"temporal\")\n",
|
||||
"MODEL_BUCKET = os.path.join(BUCKET_URI, \"imagebind\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Initialize Vertex AI API.\n",
|
||||
"print(\"Initializing Vertex AI API.\")\n",
|
||||
"aiplatform.init(project=PROJECT_ID, location=REGION, staging_bucket=STAGING_BUCKET)\n",
|
||||
"\n",
|
||||
"# Gets the default SERVICE_ACCOUNT.\n",
|
||||
"shell_output = ! gcloud projects describe $PROJECT_ID\n",
|
||||
"project_number = shell_output[-1].split(\":\")[1].strip().replace(\"'\", \"\")\n",
|
||||
"SERVICE_ACCOUNT = f\"{project_number}-compute@developer.gserviceaccount.com\"\n",
|
||||
"print(\"Using this default Service Account:\", SERVICE_ACCOUNT)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Provision permissions to the SERVICE_ACCOUNT with the GCS bucket\n",
|
||||
"! gsutil iam ch serviceAccount:{SERVICE_ACCOUNT}:roles/storage.admin $BUCKET_NAME\n",
|
||||
"\n",
|
||||
"! gcloud config set project $PROJECT_ID\n",
|
||||
"! gcloud projects add-iam-policy-binding --no-user-output-enabled {PROJECT_ID} --member=serviceAccount:{SERVICE_ACCOUNT} --role=\"roles/storage.admin\"\n",
|
||||
"! gcloud projects add-iam-policy-binding --no-user-output-enabled {PROJECT_ID} --member=serviceAccount:{SERVICE_ACCOUNT} --role=\"roles/aiplatform.user\"\n",
|
||||
"\n",
|
||||
"models, endpoints = {}, {}\n",
|
||||
"DATA_BUCKET = os.path.join(MODEL_BUCKET, \"data\")\n",
|
||||
"\n",
|
||||
"# @markdown Click \"Show Code\" to see more details."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "0olaYZtKMV0f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Set the model variants\n",
|
||||
"\n",
|
||||
"# @markdown Select the accelerator type.\n",
|
||||
"accelerator_type = \"NVIDIA_L4\" # @param [\"NVIDIA_L4\", \"NVIDIA_TESLA_V100\", \"NVIDIA_TESLA_T4\"]\n",
|
||||
"\n",
|
||||
"if accelerator_type == \"NVIDIA_L4\":\n",
|
||||
" machine_type = \"g2-standard-8\"\n",
|
||||
" accelerator_count = 1\n",
|
||||
"elif accelerator_type == \"NVIDIA_TESLA_V100\":\n",
|
||||
" machine_type = \"n1-standard-4\"\n",
|
||||
" accelerator_count = 1\n",
|
||||
"elif accelerator_type == \"NVIDIA_TESLA_T4\":\n",
|
||||
" machine_type = \"n1-standard-4\"\n",
|
||||
" accelerator_count = 1\n",
|
||||
"else:\n",
|
||||
" raise ValueError(\"Unknown accelerator type\")\n",
|
||||
"\n",
|
||||
"# The pre-built serving docker image.\n",
|
||||
"PREDICTION_DOCKER_URI = \"us-docker.pkg.dev/vertex-ai/vertex-vision-model-garden-dockers/pytorch-imagebind-serve\"\n",
|
||||
"\n",
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def deploy_model(\n",
|
||||
" model_name: str,\n",
|
||||
" service_account: str,\n",
|
||||
" task: str,\n",
|
||||
" service_account: str,\n",
|
||||
" machine_type: str = \"g2-standard-8\",\n",
|
||||
" accelerator_type: str = \"NVIDIA_L4\",\n",
|
||||
" accelerator_count: str = 1,\n",
|
||||
" use_dedicated_endpoint: bool = False,\n",
|
||||
") -> tuple[aiplatform.Model, aiplatform.Endpoint]:\n",
|
||||
" \"\"\"Deploys prebuilt model in Vertex AI.\"\"\"\n",
|
||||
" endpoint = aiplatform.Endpoint.create(display_name=f\"{model_name}-{task}-endpoint\")\n",
|
||||
" endpoint = aiplatform.Endpoint.create(\n",
|
||||
" display_name=f\"{model_name}-{task}-endpoint\",\n",
|
||||
" dedicated_endpoint_enabled=use_dedicated_endpoint,\n",
|
||||
" )\n",
|
||||
" serving_env = {\n",
|
||||
" \"MODEL_ID\": \"ImageBind-feature-embedding-generation-001\",\n",
|
||||
" \"TASK\": task,\n",
|
||||
@@ -276,42 +272,30 @@
|
||||
" serving_container_predict_route=\"/predictions/imagebind_serving\",\n",
|
||||
" serving_container_health_route=\"/ping\",\n",
|
||||
" serving_container_environment_variables=serving_env,\n",
|
||||
" model_garden_source_model_name=\"publishers/meta/models/imagebind\"\n",
|
||||
" model_garden_source_model_name=\"publishers/meta/models/imagebind\",\n",
|
||||
" )\n",
|
||||
" model.deploy(\n",
|
||||
" endpoint=endpoint,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" deploy_request_timeout=1800,\n",
|
||||
" service_account=service_account,\n",
|
||||
" system_labels={\n",
|
||||
" \"NOTEBOOK_NAME\": \"model_garden_pytorch_imagebind.ipynb\"\n",
|
||||
" },\n",
|
||||
" deploy_request_timeout=1800,\n",
|
||||
" system_labels={\"NOTEBOOK_NAME\": \"model_garden_pytorch_imagebind.ipynb\"},\n",
|
||||
" )\n",
|
||||
" return model, endpoint"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "8neJc8CnDDpu"
|
||||
},
|
||||
"source": [
|
||||
"## Deploy prebuilt ImageBind model\n",
|
||||
"\n",
|
||||
"This section deploys the prebuilt ImageBind model on Vertex AI endpoints for the tasks of feature embedding generation and zero-shot classification. The model deployment step will take ~15 minutes to complete."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "e7edb830212d"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Prepares example input data.\n",
|
||||
"# @title Prepares example input data.\n",
|
||||
"! git clone https://github.com/facebookresearch/ImageBind.git\n",
|
||||
"%cd ImageBind/.assets\n",
|
||||
"! git reset --hard 95d27c7fd5a8362f3527e176c3a80ae5a4d880c0\n",
|
||||
@@ -321,59 +305,49 @@
|
||||
"%cd ../.."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "52c0776ea427"
|
||||
},
|
||||
"source": [
|
||||
"### Deploy prebuilt ImageBind model for feature embedding generation\n",
|
||||
"\n",
|
||||
"In this section, we deploys an ImageBind that generates feature embeddings for different data modalities.\n",
|
||||
"\n",
|
||||
"The peak GPU memory usage for the ImageBind model is ~8G. Please adjust the machine type, accelerator type and accelerator count accordingly. We use one L4 (24G) in deployments as an example."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "641375dce6a1"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"task = \"feature-embedding-generation\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "Uak1pyEeExYM"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Finds Vertex AI prediction supported accelerators and regions in\n",
|
||||
"# https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
|
||||
"# @title Deploy prebuilt ImageBind model for feature embedding generation\n",
|
||||
"\n",
|
||||
"# Sets L4 to deploy ImageBind.\n",
|
||||
"machine_type = \"g2-standard-8\"\n",
|
||||
"accelerator_type = \"NVIDIA_L4\"\n",
|
||||
"accelerator_count = 1\n",
|
||||
"# @markdown In this section, an ImageBind model is deployed that generates feature embeddings for different data modalities.\n",
|
||||
"# @markdown The peak GPU memory usage for the ImageBind model is ~8G. This step takes around 10 minutes to deploy.\n",
|
||||
"\n",
|
||||
"# Sets V100 to deploy ImageBind.\n",
|
||||
"# machine_type = \"n1-standard-8\"\n",
|
||||
"# accelerator_type = \"NVIDIA_TESLA_V100\"\n",
|
||||
"# accelerator_count = 1\n",
|
||||
"task = \"feature-embedding-generation\"\n",
|
||||
"\n",
|
||||
"model, endpoint = deploy_model(\n",
|
||||
" model_name=get_job_name_with_datetime(prefix=\"ImageBind-serve\"),\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"common_util.check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
" region=REGION,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" is_for_training=False,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"LABEL = \"feature-embedding-deploy\"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model(\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=\"ImageBind-serve\"),\n",
|
||||
" task=task,\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model = models[LABEL]\n",
|
||||
"endpoint = endpoints[LABEL]\n",
|
||||
"\n",
|
||||
"print(f\"Endpoint name: {endpoint.name}\")"
|
||||
]
|
||||
},
|
||||
@@ -403,10 +377,13 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "rDHsCOqvFYBi"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Predict\n",
|
||||
"\n",
|
||||
"# Loads an existing endpoint instance using the endpoint name:\n",
|
||||
"# - Using `endpoint_name = endpoint.name` allows us to get the endpoint name of\n",
|
||||
"# the endpoint `endpoint` created in the cell above.\n",
|
||||
@@ -439,96 +416,68 @@
|
||||
" ],\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"response = endpoint.predict(instances=instances)\n",
|
||||
"response = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"for modality, embedding in response.predictions[0].items():\n",
|
||||
" print(f\"Modality {modality}: embedding shape {np.array(embedding).shape}\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "af21a3cff1e0"
|
||||
},
|
||||
"source": [
|
||||
"#### Clean up resources"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "911406c1561e"
|
||||
"cellView": "form",
|
||||
"id": "0a2DFTrwLhME"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Undeploy model and delete endpoint.\n",
|
||||
"endpoint.delete(force=True)\n",
|
||||
"# @title Deploy prebuilt ImageBind model for zero-shot classification\n",
|
||||
"\n",
|
||||
"# Delete model.\n",
|
||||
"model.delete()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "f10b8d5bb80a"
|
||||
},
|
||||
"source": [
|
||||
"### Deploy prebuilt ImageBind model for zero-shot classification\n",
|
||||
"# @markdown In this section, we deploys an ImageBind that performs zero-shot classification between pairs of data modalities.\n",
|
||||
"\n",
|
||||
"In this section, we deploys an ImageBind that performs zero-shot classification between pairs of data modalities.\n",
|
||||
"# @markdown The peak GPU memory usage for the ImageBind model is ~8G. This step takes around 10 minutes to deploy.\n",
|
||||
"\n",
|
||||
"The peak GPU memory usage for the ImageBind model is ~8G. Please adjust the machine type, accelerator type and accelerator count accordingly. We use one L4 (24G) in deployments as an example."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "3bf7295919fc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"task = \"zero-shot-classification\""
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "Uak1pyEeExYM"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Finds Vertex AI prediction supported accelerators and regions in\n",
|
||||
"# https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
|
||||
"# https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
|
||||
"\n",
|
||||
"# Sets L4 to deploy ImageBind.\n",
|
||||
"machine_type = \"g2-standard-8\"\n",
|
||||
"accelerator_type = \"NVIDIA_L4\"\n",
|
||||
"accelerator_count = 1\n",
|
||||
"task = \"zero-shot-classification\"\n",
|
||||
"\n",
|
||||
"# Sets V100 to deploy ImageBind.\n",
|
||||
"# machine_type = \"n1-standard-8\"\n",
|
||||
"# accelerator_type = \"NVIDIA_TESLA_V100\"\n",
|
||||
"# accelerator_count = 1\n",
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"model, endpoint = deploy_model(\n",
|
||||
" model_name=get_job_name_with_datetime(prefix=\"ImageBind-serve\"),\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
"\n",
|
||||
"common_util.check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
" region=REGION,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" is_for_training=False,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"LABEL = \"zero-shot-deploy\"\n",
|
||||
"models[LABEL], endpoints[LABEL] = deploy_model(\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=task),\n",
|
||||
" task=task,\n",
|
||||
" service_account=SERVICE_ACCOUNT,\n",
|
||||
" machine_type=machine_type,\n",
|
||||
" accelerator_type=accelerator_type,\n",
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"model = models[LABEL]\n",
|
||||
"endpoint = endpoints[LABEL]\n",
|
||||
"\n",
|
||||
"print(f\"Endpoint name: {endpoint.name}\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "sGKIjgmDFRW2"
|
||||
"id": "LbEkAvNWLhMF"
|
||||
},
|
||||
"source": [
|
||||
"NOTE: The prebuilt model weights will be downloaded on the fly after deployment succeeds. Thus, an additional 5 minutes of waiting time is needed **after** the above model deployment step succeeds and before you can run the next step below. Otherwise you might see a `ServiceUnavailable: 503 502:Bad Gateway` error when you send requests to the endpoint.\n",
|
||||
@@ -551,10 +500,13 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "rDHsCOqvFYBi"
|
||||
"cellView": "form",
|
||||
"id": "pq3dKm-DLhMF"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# @title Predict\n",
|
||||
"\n",
|
||||
"# Loads an existing endpoint instance using the endpoint name:\n",
|
||||
"# - Using `endpoint_name = endpoint.name` allows us to get the endpoint name of\n",
|
||||
"# the endpoint `endpoint` created in the cell above.\n",
|
||||
@@ -587,34 +539,39 @@
|
||||
" ],\n",
|
||||
" },\n",
|
||||
"]\n",
|
||||
"response = endpoint.predict(instances=instances)\n",
|
||||
"response = endpoint.predict(\n",
|
||||
" instances=instances, use_dedicated_endpoint=use_dedicated_endpoint\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"for modality_pair, probs in response.predictions[0].items():\n",
|
||||
" print(f\"{modality_pair}:\\n{np.array(probs)}\\n\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {
|
||||
"id": "af21a3cff1e0"
|
||||
},
|
||||
"source": [
|
||||
"#### Clean up resources"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"id": "911406c1561e"
|
||||
"cellView": "form",
|
||||
"id": "iM7nfGtoLhMG"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Undeploy model and delete endpoint.\n",
|
||||
"endpoint.delete(force=True)\n",
|
||||
"# @title Delete the models and endpoints\n",
|
||||
"\n",
|
||||
"# Delete model.\n",
|
||||
"model.delete()"
|
||||
"# @markdown Delete the experiment models and endpoints to recycle the resources\n",
|
||||
"# @markdown and avoid unnecessary continuous charges that may incur.\n",
|
||||
"\n",
|
||||
"# Undeploy model and delete endpoint.\n",
|
||||
"for endpoint in endpoints.values():\n",
|
||||
" endpoint.delete(force=True)\n",
|
||||
"\n",
|
||||
"# Delete models.\n",
|
||||
"for model in models.values():\n",
|
||||
" model.delete()\n",
|
||||
"\n",
|
||||
"delete_bucket = False # @param {type:\"boolean\"}\n",
|
||||
"if delete_bucket:\n",
|
||||
" ! gsutil -m rm -r $BUCKET_NAME"
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_instant_id.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_instant_id.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -101,7 +101,7 @@
|
||||
"# @title Setup Google Cloud project\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"\n",
|
||||
"# @markdown 1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n",
|
||||
"\n",
|
||||
@@ -249,7 +249,7 @@
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -260,6 +260,8 @@
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]\n",
|
||||
"\n",
|
||||
"# @title [Option 2] Deploy with customized configs\n",
|
||||
"\n",
|
||||
"# @markdown This section uploads the model to Model Registry and deploys it on the Endpoint. It takes ~15 minutes to finish.\n",
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_instant_id_gradio.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
"\n",
|
||||
"<table><tbody><tr>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/workbench/instances\">\n",
|
||||
" <a href=\"https://console.cloud.google.com/vertex-ai/notebooks/deploy-notebook?download_url=https://raw.githubusercontent.com/GoogleCloudPlatform/vertex-ai-samples/main/notebooks/community/model_garden/model_garden_pytorch_instructpix2pix.ipynb\">\n",
|
||||
" <img alt=\"Workbench logo\" src=\"https://lh3.googleusercontent.com/UiNooY4LUgW_oTvpsNhPpQzsstV5W8F7rYgxgGBD85cWJoLmrOzhVs_ksK_vgx40SHs7jCqkTkCk=e14-rj-sc0xffffff-h130-w32\" width=\"32px\"><br> Run in Workbench\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
@@ -45,7 +45,7 @@
|
||||
" </td>\n",
|
||||
" <td style=\"text-align: center\">\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_instructpix2pix.ipynb\">\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" <img alt=\"GitHub logo\" src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" width=\"32px\"><br> View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
"</tr></tbody></table>"
|
||||
@@ -117,7 +117,7 @@
|
||||
"# @markdown | a3-highgpu-8g | 8 NVIDIA_H100_80GB | us-central1, europe-west4, us-west1, asia-southeast1 |\n",
|
||||
"\n",
|
||||
"# Upgrade Vertex AI SDK.\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.84.0'\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform==1.97.0'\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"# Import the necessary packages\n",
|
||||
@@ -196,7 +196,7 @@
|
||||
"# @title [Option 1] Deploy with Model Garden SDK\n",
|
||||
"\n",
|
||||
"# @markdown Deploy with Gen AI model-centric SDK. This section uploads the prebuilt model to Model Registry and deploys it to a Vertex AI Endpoint. It takes 15 minutes to 1 hour to finish depending on the size of the model. See [use open models with Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/docs/open-models/use-open-models) for documentation on other use cases.\n",
|
||||
"from vertexai.preview import model_garden\n",
|
||||
"from vertexai import model_garden\n",
|
||||
"\n",
|
||||
"model = model_garden.OpenModel(PUBLISHER_MODEL_NAME)\n",
|
||||
"endpoints[LABEL] = model.deploy(\n",
|
||||
@@ -205,7 +205,9 @@
|
||||
" accelerator_count=accelerator_count,\n",
|
||||
" use_dedicated_endpoint=use_dedicated_endpoint,\n",
|
||||
" accept_eula=True, # Accept the End User License Agreement (EULA) on the model card before deploy. Otherwise, the deployment will be forbidden.\n",
|
||||
")"
|
||||
")\n",
|
||||
"\n",
|
||||
"endpoint = endpoints[LABEL]"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
@@ -39,7 +39,7 @@
|
||||
" </td>\n",
|
||||
" <td>\n",
|
||||
" <a href=\"https://github.com/GoogleCloudPlatform/vertex-ai-samples/blob/main/notebooks/community/model_garden/model_garden_pytorch_lama.ipynb\">\n",
|
||||
" <img src=\"https://cloud.google.com/ml-engine/images/github-logo-32px.png\" alt=\"GitHub logo\">\n",
|
||||
" <img src=\"https://github.githubassets.com/assets/GitHub-Mark-ea2971cee799.png\" alt=\"GitHub logo\">\n",
|
||||
" View on GitHub\n",
|
||||
" </a>\n",
|
||||
" </td>\n",
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user