mirror of
https://github.com/GoogleCloudPlatform/vertex-ai-samples.git
synced 2026-09-26 14:42:04 +00:00
Fix llama notebooks
PiperOrigin-RevId: 761112151
This commit is contained in:
committed by
Copybara-Service
parent
dbb226a7a4
commit
48ead2727f
@@ -6,7 +6,7 @@
|
||||
"id": "DZ1j6RRg-Td6",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "DZ1j6RRg-Td6"
|
||||
"id": "f705f4be70e9"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -29,7 +29,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "99c1c3fc2ca5",
|
||||
"metadata": {
|
||||
"id": "99c1c3fc2ca5"
|
||||
"id": "778cc1227be8"
|
||||
},
|
||||
"source": [
|
||||
"# Vertex AI Model Garden - Advanced Features\n",
|
||||
@@ -52,7 +52,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "f9-tJ6RfDLIs",
|
||||
"metadata": {
|
||||
"id": "f9-tJ6RfDLIs"
|
||||
"id": "0779b48f654e"
|
||||
},
|
||||
"source": [
|
||||
"## Overview\n",
|
||||
@@ -90,7 +90,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "47GcOrZjosOx",
|
||||
"metadata": {
|
||||
"id": "47GcOrZjosOx"
|
||||
"id": "69453bf7230e"
|
||||
},
|
||||
"source": [
|
||||
"## Before you begin"
|
||||
@@ -100,7 +100,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "1D_pWejJPHP3",
|
||||
"metadata": {
|
||||
"id": "1D_pWejJPHP3"
|
||||
"id": "bf3706e69f61"
|
||||
},
|
||||
"source": [
|
||||
"### Request for quota\n",
|
||||
@@ -118,10 +118,9 @@
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"id": "L3dqbxovo5t6",
|
||||
"language": "python",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "L3dqbxovo5t6"
|
||||
"id": "2b585189a670"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -149,6 +148,8 @@
|
||||
"# Install and import the necessary packages\n",
|
||||
"! pip install -q openai google-auth requests\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.93.1'\n",
|
||||
"\n",
|
||||
"! git clone https://github.com/GoogleCloudPlatform/vertex-ai-samples.git\n",
|
||||
"\n",
|
||||
"import datetime\n",
|
||||
@@ -224,7 +225,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "SeGqxuMfRBS5",
|
||||
"metadata": {
|
||||
"id": "SeGqxuMfRBS5"
|
||||
"id": "4782dd003acb"
|
||||
},
|
||||
"source": [
|
||||
"### Access Llama 3.1, 3.2, and 3.3 models on Vertex AI for serving"
|
||||
@@ -236,7 +237,7 @@
|
||||
"id": "BxlzWU2KQqmw",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "BxlzWU2KQqmw"
|
||||
"id": "798068fc0355"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -275,7 +276,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "JpNBJJgjWL7j",
|
||||
"metadata": {
|
||||
"id": "JpNBJJgjWL7j"
|
||||
"id": "10ed490e28e5"
|
||||
},
|
||||
"source": [
|
||||
"## Prefix Caching <a name=\"prefix-caching\"></a>\n",
|
||||
@@ -303,7 +304,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "9gZJ8cB27e1m",
|
||||
"metadata": {
|
||||
"id": "9gZJ8cB27e1m"
|
||||
"id": "30ddb93fdd7b"
|
||||
},
|
||||
"source": [
|
||||
"### Try out Prefix Caching with Hex-LLM\n",
|
||||
@@ -319,7 +320,7 @@
|
||||
"id": "RpmoA2nXjdCd",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "RpmoA2nXjdCd"
|
||||
"id": "b56d82c1aa6f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -508,7 +509,7 @@
|
||||
"id": "5QoK8c0R9U3B",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "5QoK8c0R9U3B"
|
||||
"id": "96c5afed49b4"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -574,7 +575,7 @@
|
||||
"id": "29rn5ATmB2YC",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "29rn5ATmB2YC"
|
||||
"id": "9a95c9f90358"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -647,7 +648,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "KjbM8E9DGuuR",
|
||||
"metadata": {
|
||||
"id": "KjbM8E9DGuuR"
|
||||
"id": "12ad6d1ff725"
|
||||
},
|
||||
"source": [
|
||||
"#### Delete the models and endpoints"
|
||||
@@ -659,7 +660,7 @@
|
||||
"id": "JpLU7GRQGuuR",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "JpLU7GRQGuuR"
|
||||
"id": "1ab4e3bb74b4"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -685,7 +686,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "XZ33HhYmOxCS",
|
||||
"metadata": {
|
||||
"id": "XZ33HhYmOxCS"
|
||||
"id": "7a8a9a1b2ddf"
|
||||
},
|
||||
"source": [
|
||||
"### Try out Prefix Caching with vLLM\n",
|
||||
@@ -708,7 +709,7 @@
|
||||
"id": "E8OiHHNNE_wj",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "E8OiHHNNE_wj"
|
||||
"id": "4425cc0bdedc"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -912,7 +913,7 @@
|
||||
"id": "zex1oXl36A70",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "zex1oXl36A70"
|
||||
"id": "bcbafec839cd"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -975,7 +976,7 @@
|
||||
"id": "gDOC_nfsJeUR",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "gDOC_nfsJeUR"
|
||||
"id": "e984f43422d5"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1048,7 +1049,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "GdGxaTirJeUR",
|
||||
"metadata": {
|
||||
"id": "GdGxaTirJeUR"
|
||||
"id": "dff0d10dcc20"
|
||||
},
|
||||
"source": [
|
||||
"#### Delete the models and endpoints"
|
||||
@@ -1060,7 +1061,7 @@
|
||||
"id": "OgoqXE-VJeUR",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "OgoqXE-VJeUR"
|
||||
"id": "5b8751773e7f"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1086,7 +1087,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "w4Guijaw_NEs",
|
||||
"metadata": {
|
||||
"id": "w4Guijaw_NEs"
|
||||
"id": "863775857a46"
|
||||
},
|
||||
"source": [
|
||||
"### Best practices\n",
|
||||
@@ -1101,7 +1102,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "ml8fgoIQWSbY",
|
||||
"metadata": {
|
||||
"id": "ml8fgoIQWSbY"
|
||||
"id": "565cbdc3a06b"
|
||||
},
|
||||
"source": [
|
||||
"## Speculative Decoding <a name=\"spec-decoding\"></a>\n",
|
||||
@@ -1143,7 +1144,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "NmWRro8Q-Td6",
|
||||
"metadata": {
|
||||
"id": "NmWRro8Q-Td6"
|
||||
"id": "94eaa9050abb"
|
||||
},
|
||||
"source": [
|
||||
"### Try out Speculative Decoding with vLLM"
|
||||
@@ -1155,7 +1156,7 @@
|
||||
"id": "72d1GlrYifKU",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "72d1GlrYifKU"
|
||||
"id": "5f358cc230a6"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1472,7 +1473,7 @@
|
||||
"id": "CNiItf5hdVFU",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "CNiItf5hdVFU"
|
||||
"id": "be3170e0e05a"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
@@ -1547,7 +1548,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "WahYGAZyq6Gl",
|
||||
"metadata": {
|
||||
"id": "WahYGAZyq6Gl"
|
||||
"id": "30c5d2535df3"
|
||||
},
|
||||
"source": [
|
||||
"## Clean up resources"
|
||||
@@ -1557,7 +1558,7 @@
|
||||
"cell_type": "markdown",
|
||||
"id": "bV5Yjkgav9BZ",
|
||||
"metadata": {
|
||||
"id": "bV5Yjkgav9BZ"
|
||||
"id": "63c10917ff95"
|
||||
},
|
||||
"source": [
|
||||
"### Delete the models and endpoints"
|
||||
@@ -1569,7 +1570,7 @@
|
||||
"id": "qsks36cOH9rb",
|
||||
"metadata": {
|
||||
"cellView": "form",
|
||||
"id": "qsks36cOH9rb"
|
||||
"id": "92892e1b1730"
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
|
||||
@@ -231,6 +231,8 @@
|
||||
"import yaml\n",
|
||||
"from google.cloud import aiplatform\n",
|
||||
"\n",
|
||||
"! pip3 install --upgrade --quiet 'google-cloud-aiplatform>=1.93.1'\n",
|
||||
"\n",
|
||||
"common_util = importlib.import_module(\n",
|
||||
" \"vertex-ai-samples.community-content.vertex_model_garden.model_oss.notebook_util.common_util\"\n",
|
||||
")\n",
|
||||
|
||||
@@ -1072,7 +1072,7 @@
|
||||
" vllm_args.append(\"--tool-call-parser=vertex-llama-3\")\n",
|
||||
" else:\n",
|
||||
" vllm_args.append(\"--enable-auto-tool-choice\")\n",
|
||||
" vllm_args.append(\"--llama3_json\")\n",
|
||||
" vllm_args.append(\"--tool-call-parser=llama3_json\")\n",
|
||||
"\n",
|
||||
" env_vars = {\n",
|
||||
" \"MODEL_ID\": base_model_id,\n",
|
||||
|
||||
@@ -1052,7 +1052,7 @@
|
||||
" vllm_args.append(\"--tool-call-parser=vertex-llama-3\")\n",
|
||||
" else:\n",
|
||||
" vllm_args.append(\"--enable-auto-tool-choice\")\n",
|
||||
" vllm_args.append(\"--llama3_json\")\n",
|
||||
" vllm_args.append(\"--tool-call-parser=llama3_json\")\n",
|
||||
"\n",
|
||||
" env_vars = {\n",
|
||||
" \"MODEL_ID\": base_model_id,\n",
|
||||
|
||||
@@ -406,7 +406,7 @@
|
||||
" vllm_args.append(\"--tool-call-parser=vertex-llama-3\")\n",
|
||||
" else:\n",
|
||||
" vllm_args.append(\"--enable-auto-tool-choice\")\n",
|
||||
" vllm_args.append(\"--llama3_json\")\n",
|
||||
" vllm_args.append(\"--tool-call-parser=llama3_json\")\n",
|
||||
"\n",
|
||||
" env_vars = {\n",
|
||||
" \"MODEL_ID\": base_model_id,\n",
|
||||
@@ -697,14 +697,10 @@
|
||||
"# @markdown Set use_dedicated_endpoint to False if you don't want to use [dedicated endpoint](https://cloud.google.com/vertex-ai/docs/general/deployment#create-dedicated-endpoint). Note that [dedicated endpoint does not support VPC Service Controls](https://cloud.google.com/vertex-ai/docs/predictions/choose-endpoint-type), uncheck the box if you are using VPC-SC.\n",
|
||||
"use_dedicated_endpoint = True # @param {type:\"boolean\"}\n",
|
||||
"\n",
|
||||
"# @markdown Find Vertex AI prediction supported accelerators and regions at https://cloud.google.com/vertex-ai/docs/predictions/configure-compute.\n",
|
||||
"accelerator_type = \"NVIDIA_H100_80GB\" # @param [\"NVIDIA_H100_80GB\"] {isTemplate:true}\n",
|
||||
"accelerator_type = \"NVIDIA_H100_80GB\"\n",
|
||||
"accelerator_count = 8\n",
|
||||
"if accelerator_type == \"NVIDIA_H100_80GB\":\n",
|
||||
" machine_type = \"a3-highgpu-8g\"\n",
|
||||
" multihost_gpu_node_count = 1\n",
|
||||
"else:\n",
|
||||
" raise ValueError(\"Only NVIDIA_H100_80GB is supported.\")\n",
|
||||
"machine_type = \"a3-highgpu-8g\"\n",
|
||||
"multihost_gpu_node_count = 1\n",
|
||||
"\n",
|
||||
"common_util.check_quota(\n",
|
||||
" project_id=PROJECT_ID,\n",
|
||||
|
||||
@@ -380,7 +380,7 @@
|
||||
" vllm_args.append(\"--tool-call-parser=vertex-llama-3\")\n",
|
||||
" else:\n",
|
||||
" vllm_args.append(\"--enable-auto-tool-choice\")\n",
|
||||
" vllm_args.append(\"--llama3_json\")\n",
|
||||
" vllm_args.append(\"--tool-call-parser=llama3_json\")\n",
|
||||
"\n",
|
||||
" env_vars = {\n",
|
||||
" \"MODEL_ID\": base_model_id,\n",
|
||||
@@ -891,7 +891,7 @@
|
||||
"\n",
|
||||
"\n",
|
||||
"models[\"hexllm_tpu\"], endpoints[\"hexllm_tpu\"] = deploy_model_hexllm(\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=MODEL_ID),\n",
|
||||
" model_name=common_util.get_job_name_with_datetime(prefix=\"llama4-serve-hexllm\"),\n",
|
||||
" model_id=model_id,\n",
|
||||
" publisher=\"meta\",\n",
|
||||
" publisher_model_id=\"llama4\",\n",
|
||||
|
||||
Reference in New Issue
Block a user