adding models

This commit is contained in:
2026-07-08 10:57:49 -04:00
commit f18db3f45b
18 changed files with 661 additions and 0 deletions
+127
View File
@@ -0,0 +1,127 @@
import tensorrt as trt
from cuda.bindings import driver as cuda, runtime as cudart
import numpy as np
def cuda_call(call):
err, res = call[0], call[1:]
check_cuda_err(err)
if len(res) == 1:
res = res[0]
return res
def check_cuda_err(err):
if isinstance(err, cuda.CUresult):
if err != cuda.CUresult.CUDA_SUCCESS:
raise RuntimeError("Cuda Error: {}".format(err))
if isinstance(err, cudart.cudaError_t):
if err != cudart.cudaError_t.cudaSuccess:
raise RuntimeError("Cuda Runtime Error: {}".format(err))
else:
raise RuntimeError("Unknown error type: {}".format(err))
def memcpy_device_to_host(host_arr: np.ndarray, device_ptr: int):
nbytes = host_arr.size * host_arr.itemsize
cuda_call(
cudart.cudaMemcpy(
host_arr, device_ptr, nbytes, cudart.cudaMemcpyKind.cudaMemcpyDeviceToHost
)
)
def memcpy_host_to_device(device_ptr: int, host_arr: np.ndarray):
nbytes = host_arr.size * host_arr.itemsize
cuda_call(
cudart.cudaMemcpy(
device_ptr, host_arr, nbytes, cudart.cudaMemcpyKind.cudaMemcpyHostToDevice
)
)
class TensorRTModel:
def __init__(self, engine_path):
logger = trt.Logger(trt.Logger.ERROR)
trt.init_libnvinfer_plugins(logger, namespace="")
with open(engine_path, "rb") as f, trt.Runtime(logger) as runtime:
assert runtime
engine = runtime.deserialize_cuda_engine(f.read())
assert engine
context = engine.create_execution_context()
assert context
self.engine = engine
self.context = context
self.logger = logger
def prepare(self):
engine = self.engine
inputs = []
outputs = []
allocations = []
for i in range(engine.num_io_tensors):
name = engine.get_tensor_name(i)
is_input = False
if engine.get_tensor_mode(name) == trt.TensorIOMode.INPUT:
is_input = True
dtype = engine.get_tensor_dtype(name)
shape = engine.get_tensor_shape(name)
if is_input:
batch_size = shape[0]
size = np.dtype(trt.nptype(dtype)).itemsize
for s in shape:
size *= s
allocation = cuda_call(cudart.cudaMalloc(size))
binding = {
"index": i,
"name": name,
"dtype": np.dtype(trt.nptype(dtype)),
"shape": list(shape),
"allocation": allocation,
"size": size,
}
allocations.append(allocation)
if is_input:
inputs.append(binding)
else:
outputs.append(binding)
assert batch_size > 0
assert len(inputs) > 0
assert len(outputs) > 0
assert len(allocations) > 0
self.batch_size = batch_size
self.inputs = inputs
self.outputs = outputs
self.allocations = allocations
assert(len(self.inputs) == 1)
def allocate_outputs(self):
outputs = self.outputs
cpu_allocs = dict()
for output in outputs:
dtype = output['dtype']
shape = output['shape']
name = output['name']
output.update({'allocated': np.empty(shape, dtype=dtype)})
self.cpu_allocs = cpu_allocs
def score(self, data_ptr):
curr_alloc = self.allocations.copy()
curr_alloc[0] = data_ptr
self.context.execute_v2(curr_alloc)
results = dict()
for c_out in self.outputs:
c_name = c_out['name']
c_alloc_idx = c_out['index']
c_alloc_array = c_out['allocated']
memcpy_device_to_host(c_alloc_array, curr_alloc[c_alloc_idx])
results[c_name] = c_alloc_array
return results