Compare commits

...
4 Commits
Author SHA1 Message Date
Andrew Ferlitsch 2fe4f26bec feat: triton ensemble 2023-02-17 22:02:49 +00:00
Andrew Ferlitsch e1248758d7 feat: triton ensemble 2023-02-17 21:04:58 +00:00
Andrew Ferlitsch a160e6d3e2 feat: triton ensemble 2023-02-17 21:04:27 +00:00
Andrew Ferlitsch 0ecdc39417 feat: triton ensemble 2023-02-17 21:01:54 +00:00
20 changed files with 4143 additions and 0 deletions
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,56 @@
import numpy as np
import sys
import json
import triton_python_backend_utils as pb_utils
import transformers
class TritonPythonModel:
def initialize(self, args):
self.log = open("/tmp/combine.loq", "w")
self.log.write("DEBUG: ------------------------ hello world init combine/model.py------------------------------------\n")
self.model_config = model_config = json.loads(args['model_config'])
output_config = pb_utils.get_output_config_by_name(model_config, "OUTPUT0")
self.output_dtype = pb_utils.triton_string_to_numpy(output_config['data_type'])
def execute(self, requests):
self.log.write("DEBUG: ------------------------hello world execute combine/model.py\n")
output_dtype = self.output_dtype
responses = []
out_tensor = []
self.log.write("DEBUG: ------------------------requests: combine/model.py " + str(requests) + "\n")
for request in requests:
xgb_class = pb_utils.get_input_tensor_by_name(request, "xgb_class")
tf_class = pb_utils.get_input_tensor_by_name(request, "tf_class")
sci_1_class = pb_utils.get_input_tensor_by_name(request, "sci_1_class")
sci_2_class = pb_utils.get_input_tensor_by_name(request, "sci_2_class")
self.log.write("DEBUG: ------------------------ xgb_class tf_class sci_1_class sci_2_class \n"
+ str(xgb_class.as_numpy()) + '\n'
+ str(tf_class.as_numpy()) + '\n'
+ str(sci_1_class.as_numpy()) + '\n'
+ str(sci_2_class.as_numpy()) + '\n' )
out_tensor.append(pb_utils.Tensor("OUTPUT0",
(xgb_class.as_numpy()
+ tf_class.as_numpy()
+ sci_1_class.as_numpy()
+ sci_2_class.as_numpy()) / 4.0))
inference_response = pb_utils.InferenceResponse(output_tensors = out_tensor)
responses.append(inference_response)
self.log.flush()
return responses
def finalize(self):
self.log.write("DEBUG: ------------------------ hello world finalize combine/model.py------------------------------------\n")
self.log.write('Cleaning up - custom model combine')
self.log.close()
@@ -0,0 +1,46 @@
name: "combine"
backend: "python"
max_batch_size: 0
input [
{
name: "xgb_class"
data_type: TYPE_FP32
dims: [ 1 ]
},
{
name: "tf_class"
data_type: TYPE_FP32
dims: [ -1, 1 ]
},
{
name: "sci_1_class"
data_type: TYPE_FP32
dims: [ 1 ]
},
{
name: "sci_2_class"
data_type: TYPE_FP32
dims: [ 1 ]
}
]
output [
{
name: "OUTPUT0"
data_type: TYPE_FP32
dims: [ -1, 1 ]
}
]
parameters [
{
key: "output_class"
value: { string_value: "true" }
},
{
key: "threshold"
value: { string_value: "0.5" }
}
]
instance_group[ { kind: KIND_CPU } ]
@@ -0,0 +1,143 @@
platform: "ensemble"
max_batch_size: 0
input [
{
name: "INPUT0"
data_type: TYPE_FP32
dims: [ -1, 4 ]
}
]
output [
{
name: "OUTPUT0"
data_type: TYPE_FP32
dims: [ -1, 1 ]
}
]
ensemble_scheduling {
step [
{
model_name: "mux"
model_version: -1
input_map {
key: "mux_in"
value: "INPUT0"
}
output_map {
key: "mux_xgb_out"
value: "mux_xgb_out"
}
output_map {
key: "mux_tf_out"
value: "mux_tf_out"
}
output_map {
key: "mux_sci_1_out"
value: "mux_sci_1_out"
}
output_map {
key: "mux_sci_2_out"
value: "mux_sci_2_out"
}
},
{
model_name: "xgb"
model_version: -1
input_map {
key: "input__0"
value: "mux_xgb_out"
}
output_map {
key: "output__0"
value: "xgb_class"
}
},
{
model_name: "tf"
model_version: -1
input_map {
key: "dense_input"
value: "mux_tf_out"
}
output_map {
key: "round"
value: "tf_class"
}
},
{
model_name: "sci_1"
model_version: -1
input_map {
key: "input__0"
value: "mux_sci_1_out"
}
output_map {
key: "output__0"
value: "sci_1_class"
}
},
{
model_name: "sci_2"
model_version: -1
input_map {
key: "input__0"
value: "mux_sci_2_out"
}
output_map {
key: "output__0"
value: "sci_2_class"
}
},
{
model_name: "combine"
model_version: -1
input_map {
key: "xgb_class"
value: "xgb_class"
}
input_map {
key: "tf_class"
value: "tf_class"
}
input_map {
key: "sci_1_class"
value: "sci_1_class"
}
input_map {
key: "sci_2_class"
value: "sci_2_class"
}
output_map {
key: "OUTPUT0"
value: "OUTPUT0"
}
}
]
}
parameters: [
{
key: "predict_proba"
value: { string_value: "false" }
},
{
key: "output_class"
value: { string_value: "false" }
},
{
key: "threshold"
value: { string_value: "0.5" }
},
{
key: "algo"
value: { string_value: "ALGO_AUTO" }
},
{
key: "storage_type"
value: { string_value: "AUTO" }
},
{
key: "blocks_per_sm"
value: { string_value: "0" }
}
]
@@ -0,0 +1,59 @@
import numpy as np
import sys
import json
import triton_python_backend_utils as pb_utils
import transformers
class TritonPythonModel:
def initialize(self, args):
self.log = open("/tmp/mux.loq", "w")
self.log.write("DEBUG: ------------------------ hello world init mux/model.py ------------------------------------\n")
self.out_dtypes = {}
self.model_config = model_config = json.loads(args['model_config'])
mux_xgb_out_config = pb_utils.get_output_config_by_name(model_config, "mux_xgb_out")
self.out_dtypes["mux_xgb_out"] = pb_utils.triton_string_to_numpy(mux_xgb_out_config["data_type"])
mux_tf_out_config = pb_utils.get_output_config_by_name(model_config, "mux_tf_out")
self.out_dtypes["mux_tf_out"] = pb_utils.triton_string_to_numpy(mux_tf_out_config["data_type"])
mux_sci_1_out_config = pb_utils.get_output_config_by_name(model_config, "mux_sci_1_out")
self.out_dtypes["mux_sci_1_out"] = pb_utils.triton_string_to_numpy(mux_sci_1_out_config["data_type"])
mux_sci_2_out_config = pb_utils.get_output_config_by_name(model_config, "mux_sci_2_out")
self.out_dtypes["mux_sci_2_out"] = pb_utils.triton_string_to_numpy(mux_sci_1_out_config["data_type"])
def execute(self, requests):
self.log.write("DEBUG: ------------------------requests: mux/model.py \n" + str(requests) + '\n')
responses = []
for request in requests:
mux_in = pb_utils.get_input_tensor_by_name(request, "mux_in")
out_tensors = []
for model in ["mux_xgb_out", "mux_tf_out", "mux_sci_1_out", "mux_sci_2_out"]:
self.log.write("DEBUG: ------------------------ model dtype out_tensor tensor.astype" + model + '\n'
+ str(self.out_dtypes[model]) + " "
+ str(mux_in.as_numpy()) + " "
+ str(mux_in.as_numpy().astype(self.out_dtypes[model])) + '\n')
out_tensors.append(pb_utils.Tensor(model, mux_in.as_numpy().astype(self.out_dtypes[model])))
inference_response = pb_utils.InferenceResponse(output_tensors = out_tensors)
responses.append(inference_response)
self.log.flush()
return responses
def finalize(self):
self.log.write("DEBUG: ------------------------ hello world finalize mux/model.py ------------------------------------ \n")
self.log.write('Cleaning up - custom model combine \n')
self.log.close()
@@ -0,0 +1,49 @@
name: "mux"
backend: "python"
max_batch_size: 0
input [
{
name: "mux_in"
data_type: TYPE_FP32
dims: [ -1, 4 ]
}
]
output [
{
name: "mux_xgb_out"
data_type: TYPE_FP32
dims: [ -1, 4 ]
},
{
name: "mux_tf_out"
data_type: TYPE_FP32
dims: [ -1, 4 ]
},
{
name: "mux_sci_1_out"
data_type: TYPE_FP32
dims: [ -1, 4 ]
},
{
name: "mux_sci_2_out"
data_type: TYPE_FP32
dims: [ -1, 4 ]
}
]
parameters [
{
key: "output_class"
value: { string_value: "false" }
},
{
key: "threshold"
value: { string_value: "0.5" }
}
]
instance_group[ { kind: KIND_CPU } ]
@@ -0,0 +1,50 @@
backend: "fil"
max_batch_size: 0
input [
{
name: "input__0"
data_type: TYPE_FP32
dims: [ -1, 4 ]
}
]
output [
{
name: "output__0"
data_type: TYPE_FP32
dims: [ 1 ]
}
]
instance_group [{ kind: KIND_GPU }]
parameters [
{
key: "model_type"
value: { string_value: "treelite_checkpoint" }
},
{
key: "predict_proba"
value: { string_value: "false" }
},
{
key: "output_class"
value: { string_value: "true" }
},
{
key: "threshold"
value: { string_value: "0.5" }
},
{
key: "algo"
value: { string_value: "ALGO_AUTO" }
},
{
key: "storage_type"
value: { string_value: "AUTO" }
},
{
key: "blocks_per_sm"
value: { string_value: "0" }
}
]
@@ -0,0 +1,50 @@
backend: "fil"
max_batch_size: 0
input [
{
name: "input__0"
data_type: TYPE_FP32
dims: [ -1, 4 ]
}
]
output [
{
name: "output__0"
data_type: TYPE_FP32
dims: [ 1 ]
}
]
instance_group [{ kind: KIND_GPU }]
parameters [
{
key: "model_type"
value: { string_value: "treelite_checkpoint" }
},
{
key: "predict_proba"
value: { string_value: "false" }
},
{
key: "output_class"
value: { string_value: "true" }
},
{
key: "threshold"
value: { string_value: "0.5" }
},
{
key: "algo"
value: { string_value: "ALGO_AUTO" }
},
{
key: "storage_type"
value: { string_value: "AUTO" }
},
{
key: "blocks_per_sm"
value: { string_value: "0" }
}
]
@@ -0,0 +1,51 @@
backend: "tensorflow"
platform: "tensorflow_savedmodel"
max_batch_size: 0
input [
{
name: "dense_input"
data_type: TYPE_FP32
dims: [ -1, 4 ]
}
]
output [
{
name: "round"
data_type: TYPE_FP32
dims: [ -1, 1 ]
}
]
instance_group [{ kind: KIND_GPU }]
parameters [
{
key: "model_type"
value: { string_value: "tensorflow_savedmodel" }
},
{
key: "predict_proba"
value: { string_value: "false" }
},
{
key: "output_class"
value: { string_value: "true" }
},
{
key: "threshold"
value: { string_value: "0.5" }
},
{
key: "algo"
value: { string_value: "ALGO_AUTO" }
},
{
key: "storage_type"
value: { string_value: "AUTO" }
},
{
key: "blocks_per_sm"
value: { string_value: "0" }
}
]
File diff suppressed because one or more lines are too long
@@ -0,0 +1,49 @@
backend: "fil"
max_batch_size: 0
input [
{
name: "input__0"
data_type: TYPE_FP32
dims: [ -1, 4 ]
}
]
output [
{
name: "output__0"
data_type: TYPE_FP32
dims: [ 1 ]
}
]
instance_group [{ kind: KIND_GPU }]
parameters [
{
key: "model_type"
value: { string_value: "xgboost_json" }
},
{
key: "predict_proba"
value: { string_value: "false" }
},
{
key: "output_class"
value: { string_value: "true" }
},
{
key: "threshold"
value: { string_value: "0.5" }
},
{
key: "algo"
value: { string_value: "ALGO_AUTO" }
},
{
key: "storage_type"
value: { string_value: "AUTO" }
},
{
key: "blocks_per_sm"
value: { string_value: "0" }
}
]