mirror of
https://github.com/GoogleCloudPlatform/vertex-ai-samples.git
synced 2026-09-28 08:01:55 +00:00
Compare commits
4
Commits
uj10
...
triton_ensemble
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2fe4f26bec | ||
|
|
e1248758d7 | ||
|
|
a160e6d3e2 | ||
|
|
0ecdc39417 |
+2045
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,56 @@
|
||||
import numpy as np
|
||||
import sys
|
||||
import json
|
||||
|
||||
import triton_python_backend_utils as pb_utils
|
||||
import transformers
|
||||
|
||||
class TritonPythonModel:
|
||||
|
||||
def initialize(self, args):
|
||||
self.log = open("/tmp/combine.loq", "w")
|
||||
self.log.write("DEBUG: ------------------------ hello world init combine/model.py------------------------------------\n")
|
||||
|
||||
self.model_config = model_config = json.loads(args['model_config'])
|
||||
output_config = pb_utils.get_output_config_by_name(model_config, "OUTPUT0")
|
||||
self.output_dtype = pb_utils.triton_string_to_numpy(output_config['data_type'])
|
||||
|
||||
def execute(self, requests):
|
||||
self.log.write("DEBUG: ------------------------hello world execute combine/model.py\n")
|
||||
|
||||
output_dtype = self.output_dtype
|
||||
responses = []
|
||||
out_tensor = []
|
||||
self.log.write("DEBUG: ------------------------requests: combine/model.py " + str(requests) + "\n")
|
||||
for request in requests:
|
||||
xgb_class = pb_utils.get_input_tensor_by_name(request, "xgb_class")
|
||||
tf_class = pb_utils.get_input_tensor_by_name(request, "tf_class")
|
||||
sci_1_class = pb_utils.get_input_tensor_by_name(request, "sci_1_class")
|
||||
sci_2_class = pb_utils.get_input_tensor_by_name(request, "sci_2_class")
|
||||
|
||||
self.log.write("DEBUG: ------------------------ xgb_class tf_class sci_1_class sci_2_class \n"
|
||||
+ str(xgb_class.as_numpy()) + '\n'
|
||||
+ str(tf_class.as_numpy()) + '\n'
|
||||
+ str(sci_1_class.as_numpy()) + '\n'
|
||||
+ str(sci_2_class.as_numpy()) + '\n' )
|
||||
|
||||
out_tensor.append(pb_utils.Tensor("OUTPUT0",
|
||||
(xgb_class.as_numpy()
|
||||
+ tf_class.as_numpy()
|
||||
+ sci_1_class.as_numpy()
|
||||
+ sci_2_class.as_numpy()) / 4.0))
|
||||
|
||||
inference_response = pb_utils.InferenceResponse(output_tensors = out_tensor)
|
||||
responses.append(inference_response)
|
||||
|
||||
self.log.flush()
|
||||
return responses
|
||||
|
||||
def finalize(self):
|
||||
self.log.write("DEBUG: ------------------------ hello world finalize combine/model.py------------------------------------\n")
|
||||
self.log.write('Cleaning up - custom model combine')
|
||||
self.log.close()
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
name: "combine"
|
||||
backend: "python"
|
||||
max_batch_size: 0
|
||||
input [
|
||||
{
|
||||
name: "xgb_class"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ 1 ]
|
||||
},
|
||||
{
|
||||
name: "tf_class"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 1 ]
|
||||
},
|
||||
{
|
||||
name: "sci_1_class"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ 1 ]
|
||||
},
|
||||
{
|
||||
name: "sci_2_class"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ 1 ]
|
||||
}
|
||||
]
|
||||
output [
|
||||
{
|
||||
name: "OUTPUT0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 1 ]
|
||||
}
|
||||
]
|
||||
parameters [
|
||||
{
|
||||
key: "output_class"
|
||||
value: { string_value: "true" }
|
||||
},
|
||||
{
|
||||
key: "threshold"
|
||||
value: { string_value: "0.5" }
|
||||
}
|
||||
]
|
||||
|
||||
instance_group[ { kind: KIND_CPU } ]
|
||||
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
platform: "ensemble"
|
||||
max_batch_size: 0
|
||||
input [
|
||||
{
|
||||
name: "INPUT0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
}
|
||||
]
|
||||
output [
|
||||
{
|
||||
name: "OUTPUT0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 1 ]
|
||||
}
|
||||
]
|
||||
ensemble_scheduling {
|
||||
step [
|
||||
{
|
||||
model_name: "mux"
|
||||
model_version: -1
|
||||
input_map {
|
||||
key: "mux_in"
|
||||
value: "INPUT0"
|
||||
}
|
||||
output_map {
|
||||
key: "mux_xgb_out"
|
||||
value: "mux_xgb_out"
|
||||
}
|
||||
output_map {
|
||||
key: "mux_tf_out"
|
||||
value: "mux_tf_out"
|
||||
}
|
||||
output_map {
|
||||
key: "mux_sci_1_out"
|
||||
value: "mux_sci_1_out"
|
||||
}
|
||||
output_map {
|
||||
key: "mux_sci_2_out"
|
||||
value: "mux_sci_2_out"
|
||||
}
|
||||
},
|
||||
{
|
||||
model_name: "xgb"
|
||||
model_version: -1
|
||||
input_map {
|
||||
key: "input__0"
|
||||
value: "mux_xgb_out"
|
||||
}
|
||||
output_map {
|
||||
key: "output__0"
|
||||
value: "xgb_class"
|
||||
}
|
||||
},
|
||||
{
|
||||
model_name: "tf"
|
||||
model_version: -1
|
||||
input_map {
|
||||
key: "dense_input"
|
||||
value: "mux_tf_out"
|
||||
}
|
||||
output_map {
|
||||
key: "round"
|
||||
value: "tf_class"
|
||||
}
|
||||
},
|
||||
{
|
||||
model_name: "sci_1"
|
||||
model_version: -1
|
||||
input_map {
|
||||
key: "input__0"
|
||||
value: "mux_sci_1_out"
|
||||
}
|
||||
output_map {
|
||||
key: "output__0"
|
||||
value: "sci_1_class"
|
||||
}
|
||||
},
|
||||
{
|
||||
model_name: "sci_2"
|
||||
model_version: -1
|
||||
input_map {
|
||||
key: "input__0"
|
||||
value: "mux_sci_2_out"
|
||||
}
|
||||
output_map {
|
||||
key: "output__0"
|
||||
value: "sci_2_class"
|
||||
}
|
||||
},
|
||||
{
|
||||
model_name: "combine"
|
||||
model_version: -1
|
||||
input_map {
|
||||
key: "xgb_class"
|
||||
value: "xgb_class"
|
||||
}
|
||||
input_map {
|
||||
key: "tf_class"
|
||||
value: "tf_class"
|
||||
}
|
||||
input_map {
|
||||
key: "sci_1_class"
|
||||
value: "sci_1_class"
|
||||
}
|
||||
input_map {
|
||||
key: "sci_2_class"
|
||||
value: "sci_2_class"
|
||||
}
|
||||
output_map {
|
||||
key: "OUTPUT0"
|
||||
value: "OUTPUT0"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
parameters: [
|
||||
{
|
||||
key: "predict_proba"
|
||||
value: { string_value: "false" }
|
||||
},
|
||||
{
|
||||
key: "output_class"
|
||||
value: { string_value: "false" }
|
||||
},
|
||||
{
|
||||
key: "threshold"
|
||||
value: { string_value: "0.5" }
|
||||
},
|
||||
{
|
||||
key: "algo"
|
||||
value: { string_value: "ALGO_AUTO" }
|
||||
},
|
||||
{
|
||||
key: "storage_type"
|
||||
value: { string_value: "AUTO" }
|
||||
},
|
||||
{
|
||||
key: "blocks_per_sm"
|
||||
value: { string_value: "0" }
|
||||
}
|
||||
]
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
import numpy as np
|
||||
import sys
|
||||
import json
|
||||
|
||||
import triton_python_backend_utils as pb_utils
|
||||
import transformers
|
||||
|
||||
class TritonPythonModel:
|
||||
|
||||
def initialize(self, args):
|
||||
self.log = open("/tmp/mux.loq", "w")
|
||||
|
||||
self.log.write("DEBUG: ------------------------ hello world init mux/model.py ------------------------------------\n")
|
||||
self.out_dtypes = {}
|
||||
self.model_config = model_config = json.loads(args['model_config'])
|
||||
|
||||
mux_xgb_out_config = pb_utils.get_output_config_by_name(model_config, "mux_xgb_out")
|
||||
self.out_dtypes["mux_xgb_out"] = pb_utils.triton_string_to_numpy(mux_xgb_out_config["data_type"])
|
||||
|
||||
mux_tf_out_config = pb_utils.get_output_config_by_name(model_config, "mux_tf_out")
|
||||
self.out_dtypes["mux_tf_out"] = pb_utils.triton_string_to_numpy(mux_tf_out_config["data_type"])
|
||||
|
||||
mux_sci_1_out_config = pb_utils.get_output_config_by_name(model_config, "mux_sci_1_out")
|
||||
self.out_dtypes["mux_sci_1_out"] = pb_utils.triton_string_to_numpy(mux_sci_1_out_config["data_type"])
|
||||
|
||||
mux_sci_2_out_config = pb_utils.get_output_config_by_name(model_config, "mux_sci_2_out")
|
||||
self.out_dtypes["mux_sci_2_out"] = pb_utils.triton_string_to_numpy(mux_sci_1_out_config["data_type"])
|
||||
|
||||
|
||||
def execute(self, requests):
|
||||
|
||||
self.log.write("DEBUG: ------------------------requests: mux/model.py \n" + str(requests) + '\n')
|
||||
|
||||
responses = []
|
||||
for request in requests:
|
||||
|
||||
mux_in = pb_utils.get_input_tensor_by_name(request, "mux_in")
|
||||
out_tensors = []
|
||||
for model in ["mux_xgb_out", "mux_tf_out", "mux_sci_1_out", "mux_sci_2_out"]:
|
||||
self.log.write("DEBUG: ------------------------ model dtype out_tensor tensor.astype" + model + '\n'
|
||||
+ str(self.out_dtypes[model]) + " "
|
||||
+ str(mux_in.as_numpy()) + " "
|
||||
+ str(mux_in.as_numpy().astype(self.out_dtypes[model])) + '\n')
|
||||
|
||||
out_tensors.append(pb_utils.Tensor(model, mux_in.as_numpy().astype(self.out_dtypes[model])))
|
||||
|
||||
inference_response = pb_utils.InferenceResponse(output_tensors = out_tensors)
|
||||
responses.append(inference_response)
|
||||
|
||||
self.log.flush()
|
||||
return responses
|
||||
|
||||
def finalize(self):
|
||||
self.log.write("DEBUG: ------------------------ hello world finalize mux/model.py ------------------------------------ \n")
|
||||
|
||||
self.log.write('Cleaning up - custom model combine \n')
|
||||
self.log.close()
|
||||
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
name: "mux"
|
||||
backend: "python"
|
||||
max_batch_size: 0
|
||||
input [
|
||||
{
|
||||
name: "mux_in"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
}
|
||||
]
|
||||
output [
|
||||
{
|
||||
name: "mux_xgb_out"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
},
|
||||
{
|
||||
name: "mux_tf_out"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
},
|
||||
{
|
||||
name: "mux_sci_1_out"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
},
|
||||
{
|
||||
name: "mux_sci_2_out"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
}
|
||||
]
|
||||
|
||||
parameters [
|
||||
{
|
||||
key: "output_class"
|
||||
value: { string_value: "false" }
|
||||
},
|
||||
{
|
||||
key: "threshold"
|
||||
value: { string_value: "0.5" }
|
||||
}
|
||||
]
|
||||
|
||||
instance_group[ { kind: KIND_CPU } ]
|
||||
|
||||
|
||||
|
||||
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,50 @@
|
||||
backend: "fil"
|
||||
max_batch_size: 0
|
||||
input [
|
||||
{
|
||||
name: "input__0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
}
|
||||
]
|
||||
output [
|
||||
{
|
||||
name: "output__0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ 1 ]
|
||||
}
|
||||
]
|
||||
instance_group [{ kind: KIND_GPU }]
|
||||
parameters [
|
||||
{
|
||||
key: "model_type"
|
||||
value: { string_value: "treelite_checkpoint" }
|
||||
},
|
||||
{
|
||||
key: "predict_proba"
|
||||
value: { string_value: "false" }
|
||||
},
|
||||
{
|
||||
key: "output_class"
|
||||
value: { string_value: "true" }
|
||||
},
|
||||
{
|
||||
key: "threshold"
|
||||
value: { string_value: "0.5" }
|
||||
},
|
||||
{
|
||||
key: "algo"
|
||||
value: { string_value: "ALGO_AUTO" }
|
||||
},
|
||||
{
|
||||
key: "storage_type"
|
||||
value: { string_value: "AUTO" }
|
||||
},
|
||||
{
|
||||
key: "blocks_per_sm"
|
||||
value: { string_value: "0" }
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,50 @@
|
||||
backend: "fil"
|
||||
max_batch_size: 0
|
||||
input [
|
||||
{
|
||||
name: "input__0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
}
|
||||
]
|
||||
output [
|
||||
{
|
||||
name: "output__0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ 1 ]
|
||||
}
|
||||
]
|
||||
instance_group [{ kind: KIND_GPU }]
|
||||
parameters [
|
||||
{
|
||||
key: "model_type"
|
||||
value: { string_value: "treelite_checkpoint" }
|
||||
},
|
||||
{
|
||||
key: "predict_proba"
|
||||
value: { string_value: "false" }
|
||||
},
|
||||
{
|
||||
key: "output_class"
|
||||
value: { string_value: "true" }
|
||||
},
|
||||
{
|
||||
key: "threshold"
|
||||
value: { string_value: "0.5" }
|
||||
},
|
||||
{
|
||||
key: "algo"
|
||||
value: { string_value: "ALGO_AUTO" }
|
||||
},
|
||||
{
|
||||
key: "storage_type"
|
||||
value: { string_value: "AUTO" }
|
||||
},
|
||||
{
|
||||
key: "blocks_per_sm"
|
||||
value: { string_value: "0" }
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
|
||||
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
@@ -0,0 +1,51 @@
|
||||
backend: "tensorflow"
|
||||
platform: "tensorflow_savedmodel"
|
||||
max_batch_size: 0
|
||||
input [
|
||||
{
|
||||
name: "dense_input"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
}
|
||||
]
|
||||
output [
|
||||
{
|
||||
name: "round"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 1 ]
|
||||
}
|
||||
]
|
||||
instance_group [{ kind: KIND_GPU }]
|
||||
parameters [
|
||||
{
|
||||
key: "model_type"
|
||||
value: { string_value: "tensorflow_savedmodel" }
|
||||
},
|
||||
{
|
||||
key: "predict_proba"
|
||||
value: { string_value: "false" }
|
||||
},
|
||||
{
|
||||
key: "output_class"
|
||||
value: { string_value: "true" }
|
||||
},
|
||||
{
|
||||
key: "threshold"
|
||||
value: { string_value: "0.5" }
|
||||
},
|
||||
{
|
||||
key: "algo"
|
||||
value: { string_value: "ALGO_AUTO" }
|
||||
},
|
||||
{
|
||||
key: "storage_type"
|
||||
value: { string_value: "AUTO" }
|
||||
},
|
||||
{
|
||||
key: "blocks_per_sm"
|
||||
value: { string_value: "0" }
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,49 @@
|
||||
backend: "fil"
|
||||
max_batch_size: 0
|
||||
input [
|
||||
{
|
||||
name: "input__0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ -1, 4 ]
|
||||
}
|
||||
]
|
||||
output [
|
||||
{
|
||||
name: "output__0"
|
||||
data_type: TYPE_FP32
|
||||
dims: [ 1 ]
|
||||
}
|
||||
]
|
||||
instance_group [{ kind: KIND_GPU }]
|
||||
parameters [
|
||||
{
|
||||
key: "model_type"
|
||||
value: { string_value: "xgboost_json" }
|
||||
},
|
||||
{
|
||||
key: "predict_proba"
|
||||
value: { string_value: "false" }
|
||||
},
|
||||
{
|
||||
key: "output_class"
|
||||
value: { string_value: "true" }
|
||||
},
|
||||
{
|
||||
key: "threshold"
|
||||
value: { string_value: "0.5" }
|
||||
},
|
||||
{
|
||||
key: "algo"
|
||||
value: { string_value: "ALGO_AUTO" }
|
||||
},
|
||||
{
|
||||
key: "storage_type"
|
||||
value: { string_value: "AUTO" }
|
||||
},
|
||||
{
|
||||
key: "blocks_per_sm"
|
||||
value: { string_value: "0" }
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
Reference in New Issue
Block a user