CenterNet TensorRT works. TensorRT serialization not yet implemented
Signed-oof-by: Davide Sapienza <sapienza.dav@gmail.com>
This commit is contained in:
@@ -32,7 +32,7 @@ enum layerType_t {
|
|||||||
class Layer {
|
class Layer {
|
||||||
|
|
||||||
public:
|
public:
|
||||||
Layer(Network *net);
|
Layer(Network *net, bool final = false);
|
||||||
virtual ~Layer();
|
virtual ~Layer();
|
||||||
virtual layerType_t getLayerType() = 0;
|
virtual layerType_t getLayerType() = 0;
|
||||||
|
|
||||||
@@ -43,6 +43,7 @@ public:
|
|||||||
|
|
||||||
dataDim_t input_dim, output_dim;
|
dataDim_t input_dim, output_dim;
|
||||||
dnnType *dstData; //where results will be putted
|
dnnType *dstData; //where results will be putted
|
||||||
|
bool final; //if the layer is the final one
|
||||||
|
|
||||||
std::string getLayerName() {
|
std::string getLayerName() {
|
||||||
layerType_t type = getLayerType();
|
layerType_t type = getLayerType();
|
||||||
@@ -80,7 +81,7 @@ class LayerWgs : public Layer {
|
|||||||
|
|
||||||
public:
|
public:
|
||||||
LayerWgs(Network *net, int inputs, int outputs, int kh, int kw, int kt,
|
LayerWgs(Network *net, int inputs, int outputs, int kh, int kw, int kt,
|
||||||
std::string fname_weights, bool batchnorm = false, bool additional_bias = false);
|
std::string fname_weights, bool batchnorm = false, bool additional_bias = false, bool final = false);
|
||||||
virtual ~LayerWgs();
|
virtual ~LayerWgs();
|
||||||
|
|
||||||
int inputs, outputs;
|
int inputs, outputs;
|
||||||
@@ -160,7 +161,7 @@ class Conv2d : public LayerWgs {
|
|||||||
public:
|
public:
|
||||||
Conv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
Conv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
||||||
int strideH, int strideW, int paddingH, int paddingW,
|
int strideH, int strideW, int paddingH, int paddingW,
|
||||||
std::string fname_weights, bool batchnorm = false, bool deConv = false);
|
std::string fname_weights, bool batchnorm = false, bool deConv = false, bool final = false);
|
||||||
virtual ~Conv2d();
|
virtual ~Conv2d();
|
||||||
virtual layerType_t getLayerType() { return LAYER_CONV2D; };
|
virtual layerType_t getLayerType() { return LAYER_CONV2D; };
|
||||||
|
|
||||||
@@ -217,15 +218,15 @@ public:
|
|||||||
int out_ch;
|
int out_ch;
|
||||||
int deformableGroup;
|
int deformableGroup;
|
||||||
int kernelH, kernelW, strideH, strideW, paddingH, paddingW;
|
int kernelH, kernelW, strideH, strideW, paddingH, paddingW;
|
||||||
protected:
|
|
||||||
|
|
||||||
dnnType *ones_d1;
|
dnnType *ones_d1;
|
||||||
dnnType *ones_d2;
|
dnnType *ones_d2;
|
||||||
cudnnTensorDescriptor_t biasTensorDesc;
|
|
||||||
int chunk_dim;
|
int chunk_dim;
|
||||||
dnnType *offset, *mask;
|
dnnType *offset, *mask;
|
||||||
dnnType *output_conv;
|
dnnType *output_conv;
|
||||||
|
|
||||||
|
protected:
|
||||||
|
|
||||||
|
cudnnTensorDescriptor_t biasTensorDesc;
|
||||||
void initCUDNN();
|
void initCUDNN();
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -31,6 +31,7 @@ using namespace nvinfer1;
|
|||||||
#include "pluginsRT/YoloRT.h"
|
#include "pluginsRT/YoloRT.h"
|
||||||
#include "pluginsRT/UpsampleRT.h"
|
#include "pluginsRT/UpsampleRT.h"
|
||||||
//#include "pluginsRT/Int8Calibrator.h"
|
//#include "pluginsRT/Int8Calibrator.h"
|
||||||
|
#include "pluginsRT/DeformableConvRT.h"
|
||||||
|
|
||||||
class PluginFactory : IPluginFactory
|
class PluginFactory : IPluginFactory
|
||||||
{
|
{
|
||||||
@@ -85,6 +86,7 @@ public:
|
|||||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Shortcut *l);
|
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Shortcut *l);
|
||||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Yolo *l);
|
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Yolo *l);
|
||||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Upsample *l);
|
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Upsample *l);
|
||||||
|
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, DeformConv2d *l);
|
||||||
|
|
||||||
bool serialize(const char *filename);
|
bool serialize(const char *filename);
|
||||||
bool deserialize(const char *filename);
|
bool deserialize(const char *filename);
|
||||||
|
|||||||
@@ -0,0 +1,80 @@
|
|||||||
|
#include<cassert>
|
||||||
|
#include "../kernels.h"
|
||||||
|
|
||||||
|
|
||||||
|
class DeformableConvRT : public IPlugin {
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
public:
|
||||||
|
DeformableConvRT(tk::dnn::DeformConv2d *deformable) {
|
||||||
|
this->defRT = deformable;
|
||||||
|
}
|
||||||
|
|
||||||
|
~DeformableConvRT(){
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
int getNbOutputs() const override {
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||||
|
return DimsCHW{defRT->output_dim.c, defRT->output_dim.h, defRT->output_dim.w};
|
||||||
|
}
|
||||||
|
|
||||||
|
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||||
|
}
|
||||||
|
|
||||||
|
int initialize() override {
|
||||||
|
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
virtual void terminate() override {
|
||||||
|
}
|
||||||
|
|
||||||
|
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||||
|
|
||||||
|
dnnType *srcData = (dnnType*)reinterpret_cast<const dnnType*>(inputs[0]);
|
||||||
|
dnnType *output_conv = (dnnType*)reinterpret_cast<const dnnType*>(inputs[1]);
|
||||||
|
|
||||||
|
// split conv2d outputs into offset to mask
|
||||||
|
checkCuda(cudaMemcpy(defRT->offset, defRT->output_conv, 2*defRT->chunk_dim*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||||
|
checkCuda(cudaMemcpy(defRT->mask, defRT->output_conv + 2*defRT->chunk_dim, defRT->chunk_dim*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||||
|
// kernel sigmoide
|
||||||
|
activationSIGMOIDForward(defRT->mask, defRT->mask, defRT->chunk_dim);
|
||||||
|
|
||||||
|
// deformable convolution
|
||||||
|
dcn_v2_cuda_forward(srcData, defRT->data_d,
|
||||||
|
defRT->bias2_d, defRT->ones_d1,
|
||||||
|
defRT->offset, defRT->mask,
|
||||||
|
reinterpret_cast<dnnType*>(outputs[0]), defRT->ones_d2,
|
||||||
|
defRT->kernelH, defRT->kernelW,
|
||||||
|
defRT->strideH, defRT->strideW,
|
||||||
|
defRT->paddingH, defRT->paddingW,
|
||||||
|
1, 1,
|
||||||
|
defRT->deformableGroup,
|
||||||
|
defRT->preconv->input_dim.n, defRT->preconv->input_dim.c, defRT->preconv->input_dim.h, defRT->preconv->input_dim.w,
|
||||||
|
defRT->output_dim.n, defRT->output_dim.c, defRT->output_dim.h, defRT->output_dim.w,
|
||||||
|
defRT->chunk_dim);
|
||||||
|
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
virtual size_t getSerializationSize() override {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
virtual void serialize(void* buffer) override {
|
||||||
|
char *buf = reinterpret_cast<char*>(buffer);
|
||||||
|
}
|
||||||
|
|
||||||
|
int size;
|
||||||
|
tk::dnn::DeformConv2d *defRT;
|
||||||
|
};
|
||||||
+2
-2
@@ -119,10 +119,10 @@ void Conv2d::inferCUDNN(dnnType* srcData, bool back) {
|
|||||||
|
|
||||||
Conv2d::Conv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
Conv2d::Conv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
||||||
int strideH, int strideW, int paddingH, int paddingW,
|
int strideH, int strideW, int paddingH, int paddingW,
|
||||||
std::string fname_weights, bool batchnorm, bool deConv) :
|
std::string fname_weights, bool batchnorm, bool deConv, bool final) :
|
||||||
|
|
||||||
LayerWgs(net, net->getOutputDim().c, out_ch, kernelH, kernelW, 1,
|
LayerWgs(net, net->getOutputDim().c, out_ch, kernelH, kernelW, 1,
|
||||||
fname_weights, batchnorm) {
|
fname_weights, batchnorm, false, final) {
|
||||||
|
|
||||||
this->kernelH = kernelH;
|
this->kernelH = kernelH;
|
||||||
this->kernelW = kernelW;
|
this->kernelW = kernelW;
|
||||||
|
|||||||
@@ -25,25 +25,23 @@ void DeformConv2d::initCUDNN() {
|
|||||||
if (dst_dim % 3 != 0 )
|
if (dst_dim % 3 != 0 )
|
||||||
std::cout<<"take attention\n\n";
|
std::cout<<"take attention\n\n";
|
||||||
chunk_dim = dst_dim/3;
|
chunk_dim = dst_dim/3;
|
||||||
checkCuda(cudaMalloc(&offset, 2*chunk_dim*sizeof(dnnType)));
|
checkCuda( cudaMalloc(&offset, 2*chunk_dim*sizeof(dnnType)));
|
||||||
checkCuda(cudaMalloc(&mask, chunk_dim*sizeof(dnnType)));
|
checkCuda( cudaMalloc(&mask, chunk_dim*sizeof(dnnType)));
|
||||||
|
|
||||||
// kernel ones
|
// kernel ones
|
||||||
|
|
||||||
cudaMallocHost(&ones_d1, (height_ones*width_ones)*sizeof(dnnType));
|
checkCuda( cudaMalloc(&ones_d1, (height_ones*width_ones)*sizeof(dnnType)) );
|
||||||
float aus1[height_ones*width_ones];
|
float aus1[height_ones*width_ones];
|
||||||
for(int i=0; i<height_ones*width_ones; i++)
|
for(int i=0; i<height_ones*width_ones; i++)
|
||||||
aus1[i]=1.0f;
|
aus1[i]=1.0f;
|
||||||
cudaMemcpy(ones_d1, aus1, (height_ones*width_ones)*sizeof(dnnType), cudaMemcpyHostToDevice);
|
checkCuda( cudaMemcpy(ones_d1, aus1, (height_ones*width_ones)*sizeof(dnnType), cudaMemcpyHostToDevice) );
|
||||||
cudaDeviceSynchronize();
|
|
||||||
|
|
||||||
cudaMallocHost(&ones_d2, dim_ones*sizeof(dnnType));
|
checkCuda( cudaMalloc(&ones_d2, dim_ones*sizeof(dnnType)) );
|
||||||
float aus2[dim_ones];
|
float aus2[dim_ones];
|
||||||
for(int i=0; i<dim_ones; i++)
|
for(int i=0; i<dim_ones; i++)
|
||||||
aus2[i]=1.0f;
|
aus2[i]=1.0f;
|
||||||
cudaMemcpy(ones_d2, aus2, (dim_ones)*sizeof(dnnType), cudaMemcpyHostToDevice);
|
checkCuda( cudaMemcpy(ones_d2, aus2, (dim_ones)*sizeof(dnnType), cudaMemcpyHostToDevice) );
|
||||||
cudaDeviceSynchronize();
|
checkCuda( cudaDeviceSynchronize() );
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
DeformConv2d::DeformConv2d( Network *net, int out_ch, int deformable_group, int kernelH, int kernelW,
|
DeformConv2d::DeformConv2d( Network *net, int out_ch, int deformable_group, int kernelH, int kernelW,
|
||||||
|
|||||||
+2
-2
@@ -4,10 +4,10 @@
|
|||||||
|
|
||||||
namespace tk { namespace dnn {
|
namespace tk { namespace dnn {
|
||||||
|
|
||||||
Layer::Layer(Network *net) {
|
Layer::Layer(Network *net, bool final) {
|
||||||
|
|
||||||
this->net = net;
|
this->net = net;
|
||||||
|
this->final = final;
|
||||||
if(net != nullptr) {
|
if(net != nullptr) {
|
||||||
this->input_dim = net->getOutputDim();
|
this->input_dim = net->getOutputDim();
|
||||||
this->output_dim = input_dim;
|
this->output_dim = input_dim;
|
||||||
|
|||||||
+2
-2
@@ -8,12 +8,12 @@ namespace tk { namespace dnn {
|
|||||||
|
|
||||||
LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
||||||
int kh, int kw, int kl,
|
int kh, int kw, int kl,
|
||||||
std::string fname_weights, bool batchnorm, bool additional_bias) : Layer(net) {
|
std::string fname_weights, bool batchnorm, bool additional_bias, bool final) : Layer(net, final) {
|
||||||
|
|
||||||
this->inputs = inputs;
|
this->inputs = inputs;
|
||||||
this->outputs = outputs;
|
this->outputs = outputs;
|
||||||
this->weights_path = std::string(fname_weights);
|
this->weights_path = std::string(fname_weights);
|
||||||
|
|
||||||
std::cout<<"Reading weights: I="<<inputs<<" O="<<outputs<<" KERNEL="<<kh<<"x"<<kw<<"x"<<kl<<"\n";
|
std::cout<<"Reading weights: I="<<inputs<<" O="<<outputs<<" KERNEL="<<kh<<"x"<<kw<<"x"<<kl<<"\n";
|
||||||
int seek = 0;
|
int seek = 0;
|
||||||
readBinaryFile(weights_path.c_str(), inputs*outputs*kh*kw*kl, &data_h, &data_d, seek, net->dontLoadWeights);
|
readBinaryFile(weights_path.c_str(), inputs*outputs*kh*kw*kl, &data_h, &data_d, seek, net->dontLoadWeights);
|
||||||
|
|||||||
+52
-1
@@ -75,7 +75,7 @@ NetworkRT::NetworkRT(Network *net, const char *name) {
|
|||||||
input = Ilay->getOutput(0);
|
input = Ilay->getOutput(0);
|
||||||
input->setName( (l->getLayerName() + std::to_string(i) + "_out").c_str() );
|
input->setName( (l->getLayerName() + std::to_string(i) + "_out").c_str() );
|
||||||
|
|
||||||
if(l->getLayerType() == LAYER_YOLO)
|
if(l->getLayerType() == LAYER_YOLO || l->final)
|
||||||
networkRT->markOutput(*input);
|
networkRT->markOutput(*input);
|
||||||
tensors[l] = input;
|
tensors[l] = input;
|
||||||
}
|
}
|
||||||
@@ -182,6 +182,8 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Layer *l) {
|
|||||||
return convert_layer(input, (Yolo*) l);
|
return convert_layer(input, (Yolo*) l);
|
||||||
if(type == LAYER_UPSAMPLE)
|
if(type == LAYER_UPSAMPLE)
|
||||||
return convert_layer(input, (Upsample*) l);
|
return convert_layer(input, (Upsample*) l);
|
||||||
|
if(type == LAYER_DEFORMCONV2D)
|
||||||
|
return convert_layer(input, (DeformConv2d*) l);
|
||||||
|
|
||||||
std::cout<<l->getLayerName()<<"\n";
|
std::cout<<l->getLayerName()<<"\n";
|
||||||
FatalError("Layer not implemented in tensorRT");
|
FatalError("Layer not implemented in tensorRT");
|
||||||
@@ -254,6 +256,8 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Conv2d *l) {
|
|||||||
lRTconv->setPadding(DimsHW{l->paddingH, l->paddingW});
|
lRTconv->setPadding(DimsHW{l->paddingH, l->paddingW});
|
||||||
lRT = (ILayer*) lRTconv;
|
lRT = (ILayer*) lRTconv;
|
||||||
|
|
||||||
|
Dims d = lRTconv->getOutput(0)->getDimensions();
|
||||||
|
std::cout<<"DECONV: "<<d.d[0]<<" "<<d.d[1]<<" "<<d.d[2]<<" "<<d.d[3]<<"\n";
|
||||||
}
|
}
|
||||||
|
|
||||||
checkNULL(lRT);
|
checkNULL(lRT);
|
||||||
@@ -423,6 +427,53 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Upsample *l) {
|
|||||||
return lRT;
|
return lRT;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
ILayer* NetworkRT::convert_layer(ITensor *input, DeformConv2d *l) {
|
||||||
|
//std::cout<<"convert DEFORMABLE\n";
|
||||||
|
ILayer *preconv = convert_layer(input, l->preconv);
|
||||||
|
|
||||||
|
ITensor **inputs = new ITensor*[2];
|
||||||
|
inputs[0] = input;
|
||||||
|
inputs[1] = preconv->getOutput(0);
|
||||||
|
|
||||||
|
//std::cout<<"New plugin DEFORMABLE\n";
|
||||||
|
IPlugin *plugin = new DeformableConvRT(l);
|
||||||
|
IPluginLayer *lRT = networkRT->addPlugin(&input, 1, *plugin);
|
||||||
|
checkNULL(lRT);
|
||||||
|
|
||||||
|
// batchnorm
|
||||||
|
void *bias_b, *power_b, *mean_b, *variance_b, *scales_b;
|
||||||
|
if(dtRT == DataType::kHALF) {
|
||||||
|
bias_b = l->bias16_h;
|
||||||
|
power_b = l->power16_h;
|
||||||
|
mean_b = l->mean16_h;
|
||||||
|
variance_b = l->variance16_h;
|
||||||
|
scales_b = l->scales16_h;
|
||||||
|
} else {
|
||||||
|
bias_b = l->bias_h;
|
||||||
|
power_b = l->power_h;
|
||||||
|
mean_b = l->mean_h;
|
||||||
|
variance_b = l->variance_h;
|
||||||
|
scales_b = l->scales_h;
|
||||||
|
}
|
||||||
|
|
||||||
|
Weights power{dtRT, power_b, l->outputs};
|
||||||
|
Weights shift{dtRT, mean_b, l->outputs};
|
||||||
|
Weights scale{dtRT, variance_b, l->outputs};
|
||||||
|
std::cout<<lRT->getNbOutputs()<<std::endl;
|
||||||
|
IScaleLayer *lRT2 = networkRT->addScale(*lRT->getOutput(0), ScaleMode::kCHANNEL,
|
||||||
|
shift, scale, power);
|
||||||
|
|
||||||
|
checkNULL(lRT2);
|
||||||
|
|
||||||
|
Weights shift2{dtRT, bias_b, l->outputs};
|
||||||
|
Weights scale2{dtRT, scales_b, l->outputs};
|
||||||
|
IScaleLayer *lRT3 = networkRT->addScale(*lRT2->getOutput(0), ScaleMode::kCHANNEL,
|
||||||
|
shift2, scale2, power);
|
||||||
|
checkNULL(lRT3);
|
||||||
|
|
||||||
|
return lRT3;
|
||||||
|
}
|
||||||
|
|
||||||
bool NetworkRT::serialize(const char *filename) {
|
bool NetworkRT::serialize(const char *filename) {
|
||||||
|
|
||||||
std::ofstream p(filename);
|
std::ofstream p(filename);
|
||||||
|
|||||||
@@ -172,9 +172,9 @@ const char *reg_conv2_bin = "../tests/resnet101_cnet/layers/reg-2.bin";
|
|||||||
const char *fc_bin = "../tests/resnet101_cnet/layers/fc.bin";
|
const char *fc_bin = "../tests/resnet101_cnet/layers/fc.bin";
|
||||||
|
|
||||||
const char *output_bin[]={
|
const char *output_bin[]={
|
||||||
"../tests/resnet101_cnet/debug/hm.bin",
|
"../tests/resnet101_cnet/debug/hm.bin",
|
||||||
"../tests/resnet101_cnet/debug/wh.bin",
|
"../tests/resnet101_cnet/debug/wh.bin",
|
||||||
"../tests/resnet101_cnet/debug/reg.bin"};
|
"../tests/resnet101_cnet/debug/reg.bin"};
|
||||||
|
|
||||||
int main()
|
int main()
|
||||||
{
|
{
|
||||||
@@ -317,17 +317,17 @@ int main()
|
|||||||
tk::dnn::Layer *route_1_0_layers[1] = { layer2_deconv1_relu };
|
tk::dnn::Layer *route_1_0_layers[1] = { layer2_deconv1_relu };
|
||||||
tk::dnn::Conv2d *hm_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, hm_conv1_bin, false);
|
tk::dnn::Conv2d *hm_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, hm_conv1_bin, false);
|
||||||
tk::dnn::Activation *hm_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU);
|
tk::dnn::Activation *hm_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU);
|
||||||
tk::dnn::Conv2d *hm = new tk::dnn::Conv2d(&net, 80, 1, 1, 1, 1, 0, 0, hm_conv2_bin, false);
|
tk::dnn::Conv2d *hm = new tk::dnn::Conv2d(&net, 80, 1, 1, 1, 1, 0, 0, hm_conv2_bin, false, false, true);
|
||||||
|
|
||||||
tk::dnn::Route *route_1_0 = new tk::dnn::Route(&net, route_1_0_layers, 1);
|
tk::dnn::Route *route_1_0 = new tk::dnn::Route(&net, route_1_0_layers, 1);
|
||||||
tk::dnn::Conv2d *wh_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, wh_conv1_bin, false);
|
tk::dnn::Conv2d *wh_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, wh_conv1_bin, false);
|
||||||
tk::dnn::Activation *wh_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU);
|
tk::dnn::Activation *wh_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU);
|
||||||
tk::dnn::Conv2d *wh = new tk::dnn::Conv2d(&net, 2, 1, 1, 1, 1, 0, 0, wh_conv2_bin, false);
|
tk::dnn::Conv2d *wh = new tk::dnn::Conv2d(&net, 2, 1, 1, 1, 1, 0, 0, wh_conv2_bin, false, false, true);
|
||||||
|
|
||||||
tk::dnn::Route *route_2_0 = new tk::dnn::Route(&net, route_1_0_layers, 1);
|
tk::dnn::Route *route_2_0 = new tk::dnn::Route(&net, route_1_0_layers, 1);
|
||||||
tk::dnn::Conv2d *reg_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, reg_conv1_bin, false);
|
tk::dnn::Conv2d *reg_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, reg_conv1_bin, false);
|
||||||
tk::dnn::Activation *reg_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU);
|
tk::dnn::Activation *reg_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU);
|
||||||
tk::dnn::Conv2d *reg = new tk::dnn::Conv2d(&net, 2, 1, 1, 1, 1, 0, 0, reg_conv2_bin, false);
|
tk::dnn::Conv2d *reg = new tk::dnn::Conv2d(&net, 2, 1, 1, 1, 1, 0, 0, reg_conv2_bin, false, false, true);
|
||||||
|
|
||||||
// Load input
|
// Load input
|
||||||
dnnType *data;
|
dnnType *data;
|
||||||
@@ -339,7 +339,7 @@ int main()
|
|||||||
net.print();
|
net.print();
|
||||||
|
|
||||||
//convert network to tensorRT
|
//convert network to tensorRT
|
||||||
// tk::dnn::NetworkRT netRT(&net, "resnet101_cnet.rt");
|
tk::dnn::NetworkRT netRT(&net, "resnet101_cnet.rt");
|
||||||
|
|
||||||
|
|
||||||
tk::dnn::dataDim_t dim1 = dim; //input dim
|
tk::dnn::dataDim_t dim1 = dim; //input dim
|
||||||
@@ -354,7 +354,7 @@ int main()
|
|||||||
|
|
||||||
// printDeviceVector(64, cudnn_out, true);
|
// printDeviceVector(64, cudnn_out, true);
|
||||||
|
|
||||||
/* tk::dnn::dataDim_t dim2 = dim;
|
tk::dnn::dataDim_t dim2 = dim;
|
||||||
printCenteredTitle(" TENSORRT inference ", '=', 30);
|
printCenteredTitle(" TENSORRT inference ", '=', 30);
|
||||||
{
|
{
|
||||||
dim2.print();
|
dim2.print();
|
||||||
@@ -363,10 +363,9 @@ int main()
|
|||||||
TIMER_STOP
|
TIMER_STOP
|
||||||
dim2.print();
|
dim2.print();
|
||||||
}
|
}
|
||||||
rt_out = (dnnType *)netRT.buffersRT[1];
|
|
||||||
*/
|
|
||||||
|
|
||||||
tk::dnn::Conv2d *outs[3] = { hm, wh, reg };
|
tk::dnn::Layer *outs[3] = { hm, wh, reg };
|
||||||
|
|
||||||
for(int i=0; i<3; i++) {
|
for(int i=0; i<3; i++) {
|
||||||
printCenteredTitle((std::string(" RESNET CHECK RESULTS ") + std::to_string(i) + " ").c_str(), '=', 30);
|
printCenteredTitle((std::string(" RESNET CHECK RESULTS ") + std::to_string(i) + " ").c_str(), '=', 30);
|
||||||
|
|
||||||
@@ -382,14 +381,15 @@ int main()
|
|||||||
|
|
||||||
dnnType *cudnn_out, *rt_out;
|
dnnType *cudnn_out, *rt_out;
|
||||||
cudnn_out = outs[i]->dstData;
|
cudnn_out = outs[i]->dstData;
|
||||||
|
rt_out = (dnnType *)netRT.buffersRT[i+1];
|
||||||
|
|
||||||
std::cout << "CUDNN vs correct";
|
std::cout << "CUDNN vs correct";
|
||||||
checkResult(odim, cudnn_out, out);
|
checkResult(odim, cudnn_out, out);
|
||||||
|
|
||||||
/* std::cout << "TRT vs correct";
|
std::cout << "TRT vs correct";
|
||||||
checkResult(odim, rt_out, out);
|
checkResult(odim, rt_out, out);
|
||||||
std::cout << "CUDNN vs TRT ";
|
std::cout << "CUDNN vs TRT ";
|
||||||
checkResult(odim, cudnn_out, rt_out);*/
|
checkResult(odim, cudnn_out, rt_out);
|
||||||
}
|
}
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user