Fix llama notebooks

PiperOrigin-RevId: 761112151
This commit is contained in:
Rayan Dasoriya
2025-05-20 09:22:11 -07:00
committed by Copybara-Service
parent dbb226a7a4
commit 48ead2727f
6 changed files with 41 additions and 42 deletions
@@ -6,7 +6,7 @@
"id": "DZ1j6RRg-Td6",
"metadata": {
"cellView": "form",
"id": "DZ1j6RRg-Td6"
"id": "f705f4be70e9"
},
"outputs": [],
"source": [
@@ -29,7 +29,7 @@
"cell_type": "markdown",
"id": "99c1c3fc2ca5",
"metadata": {
"id": "99c1c3fc2ca5"
"id": "778cc1227be8"
},
"source": [
"# Vertex AI Model Garden - Advanced Features\n",
@@ -52,7 +52,7 @@
"cell_type": "markdown",
"id": "f9-tJ6RfDLIs",
"metadata": {
"id": "f9-tJ6RfDLIs"
"id": "0779b48f654e"
},
"source": [
"## Overview\n",
@@ -90,7 +90,7 @@
"cell_type": "markdown",
"id": "47GcOrZjosOx",
"metadata": {
"id": "47GcOrZjosOx"
"id": "69453bf7230e"
},
"source": [
"## Before you begin"
@@ -100,7 +100,7 @@
"cell_type": "markdown",
"id": "1D_pWejJPHP3",
"metadata": {
"id": "1D_pWejJPHP3"
"id": "bf3706e69f61"
},
"source": [
"### Request for quota\n",
@@ -118,10 +118,9 @@
"cell_type": "code",
"execution_count": null,
"id": "L3dqbxovo5t6",
"language": "python",
"metadata": {
"cellView": "form",
"id": "L3dqbxovo5t6"
"id": "2b585189a670"
},
"outputs": [],
"source": [
@@ -149,6 +148,8 @@
"# Install and import the necessary packages\n",
"! pip install -q openai google-auth requests\n",
"\n",
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.93.1'\n",
"\n",
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
"\n",
"import datetime\n",
@@ -224,7 +225,7 @@
"cell_type": "markdown",
"id": "SeGqxuMfRBS5",
"metadata": {
"id": "SeGqxuMfRBS5"
"id": "4782dd003acb"
},
"source": [
"### Access Llama 3.1, 3.2, and 3.3 models on Vertex AI for serving"
@@ -236,7 +237,7 @@
"id": "BxlzWU2KQqmw",
"metadata": {
"cellView": "form",
"id": "BxlzWU2KQqmw"
"id": "798068fc0355"
},
"outputs": [],
"source": [
@@ -275,7 +276,7 @@
"cell_type": "markdown",
"id": "JpNBJJgjWL7j",
"metadata": {
"id": "JpNBJJgjWL7j"
"id": "10ed490e28e5"
},
"source": [
"## Prefix Caching <a name=\"prefix-caching\"></a>\n",
@@ -303,7 +304,7 @@
"cell_type": "markdown",
"id": "9gZJ8cB27e1m",
"metadata": {
"id": "9gZJ8cB27e1m"
"id": "30ddb93fdd7b"
},
"source": [
"### Try out Prefix Caching with Hex-LLM\n",
@@ -319,7 +320,7 @@
"id": "RpmoA2nXjdCd",
"metadata": {
"cellView": "form",
"id": "RpmoA2nXjdCd"
"id": "b56d82c1aa6f"
},
"outputs": [],
"source": [
@@ -508,7 +509,7 @@
"id": "5QoK8c0R9U3B",
"metadata": {
"cellView": "form",
"id": "5QoK8c0R9U3B"
"id": "96c5afed49b4"
},
"outputs": [],
"source": [
@@ -574,7 +575,7 @@
"id": "29rn5ATmB2YC",
"metadata": {
"cellView": "form",
"id": "29rn5ATmB2YC"
"id": "9a95c9f90358"
},
"outputs": [],
"source": [
@@ -647,7 +648,7 @@
"cell_type": "markdown",
"id": "KjbM8E9DGuuR",
"metadata": {
"id": "KjbM8E9DGuuR"
"id": "12ad6d1ff725"
},
"source": [
"#### Delete the models and endpoints"
@@ -659,7 +660,7 @@
"id": "JpLU7GRQGuuR",
"metadata": {
"cellView": "form",
"id": "JpLU7GRQGuuR"
"id": "1ab4e3bb74b4"
},
"outputs": [],
"source": [
@@ -685,7 +686,7 @@
"cell_type": "markdown",
"id": "XZ33HhYmOxCS",
"metadata": {
"id": "XZ33HhYmOxCS"
"id": "7a8a9a1b2ddf"
},
"source": [
"### Try out Prefix Caching with vLLM\n",
@@ -708,7 +709,7 @@
"id": "E8OiHHNNE_wj",
"metadata": {
"cellView": "form",
"id": "E8OiHHNNE_wj"
"id": "4425cc0bdedc"
},
"outputs": [],
"source": [
@@ -912,7 +913,7 @@
"id": "zex1oXl36A70",
"metadata": {
"cellView": "form",
"id": "zex1oXl36A70"
"id": "bcbafec839cd"
},
"outputs": [],
"source": [
@@ -975,7 +976,7 @@
"id": "gDOC_nfsJeUR",
"metadata": {
"cellView": "form",
"id": "gDOC_nfsJeUR"
"id": "e984f43422d5"
},
"outputs": [],
"source": [
@@ -1048,7 +1049,7 @@
"cell_type": "markdown",
"id": "GdGxaTirJeUR",
"metadata": {
"id": "GdGxaTirJeUR"
"id": "dff0d10dcc20"
},
"source": [
"#### Delete the models and endpoints"
@@ -1060,7 +1061,7 @@
"id": "OgoqXE-VJeUR",
"metadata": {
"cellView": "form",
"id": "OgoqXE-VJeUR"
"id": "5b8751773e7f"
},
"outputs": [],
"source": [
@@ -1086,7 +1087,7 @@
"cell_type": "markdown",
"id": "w4Guijaw_NEs",
"metadata": {
"id": "w4Guijaw_NEs"
"id": "863775857a46"
},
"source": [
"### Best practices\n",
@@ -1101,7 +1102,7 @@
"cell_type": "markdown",
"id": "ml8fgoIQWSbY",
"metadata": {
"id": "ml8fgoIQWSbY"
"id": "565cbdc3a06b"
},
"source": [
"## Speculative Decoding <a name=\"spec-decoding\"></a>\n",
@@ -1143,7 +1144,7 @@
"cell_type": "markdown",
"id": "NmWRro8Q-Td6",
"metadata": {
"id": "NmWRro8Q-Td6"
"id": "94eaa9050abb"
},
"source": [
"### Try out Speculative Decoding with vLLM"
@@ -1155,7 +1156,7 @@
"id": "72d1GlrYifKU",
"metadata": {
"cellView": "form",
"id": "72d1GlrYifKU"
"id": "5f358cc230a6"
},
"outputs": [],
"source": [
@@ -1472,7 +1473,7 @@
"id": "CNiItf5hdVFU",
"metadata": {
"cellView": "form",
"id": "CNiItf5hdVFU"
"id": "be3170e0e05a"
},
"outputs": [],
"source": [
@@ -1547,7 +1548,7 @@
"cell_type": "markdown",
"id": "WahYGAZyq6Gl",
"metadata": {
"id": "WahYGAZyq6Gl"
"id": "30c5d2535df3"
},
"source": [
"## Clean up resources"
@@ -1557,7 +1558,7 @@
"cell_type": "markdown",
"id": "bV5Yjkgav9BZ",
"metadata": {
"id": "bV5Yjkgav9BZ"
"id": "63c10917ff95"
},
"source": [
"### Delete the models and endpoints"
@@ -1569,7 +1570,7 @@
"id": "qsks36cOH9rb",
"metadata": {
"cellView": "form",
"id": "qsks36cOH9rb"
"id": "92892e1b1730"
},
"outputs": [],
"source": [
@@ -231,6 +231,8 @@
"import yaml\n",
"from google.cloud import aiplatform\n",
"\n",
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.93.1'\n",
"\n",
"common_util = importlib.import_module(\n",
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
")\n",
@@ -1072,7 +1072,7 @@
" vllm_args.append(\"--tool-call-parser=vertex-llama-3\")\n",
" else:\n",
" vllm_args.append(\"--enable-auto-tool-choice\")\n",
" vllm_args.append(\"--llama3_json\")\n",
" vllm_args.append(\"--tool-call-parser=llama3_json\")\n",
"\n",
" env_vars = {\n",
" \"MODEL_ID\": base_model_id,\n",
@@ -1052,7 +1052,7 @@
" vllm_args.append(\"--tool-call-parser=vertex-llama-3\")\n",
" else:\n",
" vllm_args.append(\"--enable-auto-tool-choice\")\n",
" vllm_args.append(\"--llama3_json\")\n",
" vllm_args.append(\"--tool-call-parser=llama3_json\")\n",
"\n",
" env_vars = {\n",
" \"MODEL_ID\": base_model_id,\n",
@@ -406,7 +406,7 @@
" vllm_args.append(\"--tool-call-parser=vertex-llama-3\")\n",
" else:\n",
" vllm_args.append(\"--enable-auto-tool-choice\")\n",
" vllm_args.append(\"--llama3_json\")\n",
" vllm_args.append(\"--tool-call-parser=llama3_json\")\n",
"\n",
" env_vars = {\n",
" \"MODEL_ID\": base_model_id,\n",
@@ -697,14 +697,10 @@
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
"\n",
"# @markdown Find Vertex AI prediction supported accelerators and regions at https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
"accelerator_type = \"NVIDIA_H100_80GB\" # @param [\"NVIDIA_H100_80GB\"] {isTemplate:true}\n",
"accelerator_type = \"NVIDIA_H100_80GB\"\n",
"accelerator_count = 8\n",
"if accelerator_type == \"NVIDIA_H100_80GB\":\n",
" machine_type = \"a3-highgpu-8g\"\n",
" multihost_gpu_node_count = 1\n",
"else:\n",
" raise ValueError(\"Only NVIDIA_H100_80GB is supported.\")\n",
"machine_type = \"a3-highgpu-8g\"\n",
"multihost_gpu_node_count = 1\n",
"\n",
"common_util.check_quota(\n",
" project_id=PROJECT_ID,\n",
@@ -380,7 +380,7 @@
" vllm_args.append(\"--tool-call-parser=vertex-llama-3\")\n",
" else:\n",
" vllm_args.append(\"--enable-auto-tool-choice\")\n",
" vllm_args.append(\"--llama3_json\")\n",
" vllm_args.append(\"--tool-call-parser=llama3_json\")\n",
"\n",
" env_vars = {\n",
" \"MODEL_ID\": base_model_id,\n",
@@ -891,7 +891,7 @@
"\n",
"\n",
"models[\"hexllm_tpu\"], endpoints[\"hexllm_tpu\"] = deploy_model_hexllm(\n",
" model_name=common_util.get_job_name_with_datetime(prefix=MODEL_ID),\n",
" model_name=common_util.get_job_name_with_datetime(prefix=\"llama4-serve-hexllm\"),\n",
" model_id=model_id,\n",
" publisher=\"meta\",\n",
" publisher_model_id=\"llama4\",\n",