csresnext50-panet-spp works with TensorRT

Signed-off-by: Francesco Gatti <gattifrancesco@hotmail.it>
This commit is contained in:
Francesco Gatti
2020-03-12 18:24:41 +01:00
parent 5f4508a5a6
commit 4845edbc88
8 changed files with 1225 additions and 102 deletions
+2 -2
View File
@@ -321,7 +321,7 @@ public:
int winH, winW; int winH, winW;
int strideH, strideW; int strideH, strideW;
int paddingH, paddingW; int paddingH, paddingW;
bool test; bool maxpoolfixedsize;
tkdnnPoolingMode_t pool_mode; tkdnnPoolingMode_t pool_mode;
Pooling(Network *net, int winH, int winW, Pooling(Network *net, int winH, int winW,
@@ -474,7 +474,7 @@ public:
dnnType *predictions; dnnType *predictions;
static const int MAX_DETECTIONS = 4096; static const int MAX_DETECTIONS = 1024;
static Yolo::detection *allocateDetections(int nboxes, int classes); static Yolo::detection *allocateDetections(int nboxes, int classes);
static void mergeDetections(Yolo::detection *dets, int ndets, int classes); static void mergeDetections(Yolo::detection *dets, int ndets, int classes);
}; };
+1
View File
@@ -36,6 +36,7 @@ using namespace nvinfer1;
#include "pluginsRT/DeformableConvRT.h" #include "pluginsRT/DeformableConvRT.h"
#include "pluginsRT/FlattenConcatRT.h" #include "pluginsRT/FlattenConcatRT.h"
#include "pluginsRT/ReshapeRT.h" #include "pluginsRT/ReshapeRT.h"
#include "pluginsRT/MaxPoolingFixedSizeRT.h"
class PluginFactory : IPluginFactory class PluginFactory : IPluginFactory
{ {
@@ -0,0 +1,74 @@
#include<cassert>
#include "../kernels.h"
class MaxPoolFixedSizeRT : public IPlugin {
public:
MaxPoolFixedSizeRT(int c, int h, int w, int n, int strideH, int strideW, int winSize, int padding) {
this->c = c;
this->h = h;
this->w = w;
this->n = n;
this->stride_H = strideH;
this->stride_W = strideW;
this->winSize = winSize;
this->padding = padding;
}
~MaxPoolFixedSizeRT(){
}
int getNbOutputs() const override {
return 1;
}
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
return DimsCHW{this->c, this->h, this->w};
}
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
}
int initialize() override {
return 0;
}
virtual void terminate() override {
}
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
return 0;
}
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
std::cout<<this->n<<" "<<this->c<<" "<<this->h<<" "<<this->w<<" "<<this->stride_H<<" "<<this->stride_W<<" "<<this->winSize<<" "<<this->padding<<std::endl;
dnnType *srcData = (dnnType*)reinterpret_cast<const dnnType*>(inputs[0]);
dnnType *dstData = reinterpret_cast<dnnType*>(outputs[0]);
MaxPoolingForward(srcData, dstData, this->n, this->c, this->h, this->w, this->stride_H, this->stride_W, this->winSize, this->padding);
return 0;
}
virtual size_t getSerializationSize() override {
return 8*sizeof(int);
}
virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer);
tk::dnn::writeBUF(buf, this->c);
tk::dnn::writeBUF(buf, this->h);
tk::dnn::writeBUF(buf, this->w);
tk::dnn::writeBUF(buf, this->n);
tk::dnn::writeBUF(buf, this->stride_H);
tk::dnn::writeBUF(buf, this->stride_W);
tk::dnn::writeBUF(buf, this->winSize);
tk::dnn::writeBUF(buf, this->padding);
}
int n, c, h, w;
int stride_H, stride_W;
int winSize;
int padding;
};
+11 -3
View File
@@ -4,7 +4,10 @@
class ShortcutRT : public IPlugin { class ShortcutRT : public IPlugin {
public: public:
ShortcutRT() { ShortcutRT(tk::dnn::dataDim_t bdim) {
this->bc = bdim.c;
this->bh = bdim.h;
this->bw = bdim.w;
} }
~ShortcutRT(){ ~ShortcutRT(){
@@ -44,22 +47,27 @@ public:
dnnType *dstData = reinterpret_cast<dnnType*>(outputs[0]); dnnType *dstData = reinterpret_cast<dnnType*>(outputs[0]);
checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream)); checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream));
shortcutForward(srcDataBack, dstData, batchSize, c, h, w, 1, batchSize, c, h, w, 1, stream); shortcutForward(srcDataBack, dstData, batchSize, c, h, w, 1, batchSize, bc, bh, bw, 1, stream);
return 0; return 0;
} }
virtual size_t getSerializationSize() override { virtual size_t getSerializationSize() override {
return 3*sizeof(int); return 6*sizeof(int);
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer);
tk::dnn::writeBUF(buf, bc);
tk::dnn::writeBUF(buf, bh);
tk::dnn::writeBUF(buf, bw);
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c);
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h);
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w);
} }
int c, h, w; int c, h, w;
int bc, bh, bw;
}; };
+49 -18
View File
@@ -304,23 +304,36 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Pooling *l) {
if(l->pool_mode == tkdnnPoolingMode_t::POOLING_AVERAGE) ptype = PoolingType::kAVERAGE; if(l->pool_mode == tkdnnPoolingMode_t::POOLING_AVERAGE) ptype = PoolingType::kAVERAGE;
if(l->pool_mode == tkdnnPoolingMode_t::POOLING_AVERAGE_EXCLUDE_PADDING) ptype = PoolingType::kMAX_AVERAGE_BLEND; if(l->pool_mode == tkdnnPoolingMode_t::POOLING_AVERAGE_EXCLUDE_PADDING) ptype = PoolingType::kMAX_AVERAGE_BLEND;
if(l->maxpoolfixedsize)
if(l->paddingH == 0 && l->paddingW == 0 && l->input_dim.h == l->output_dim.h && l->input_dim.w == l->output_dim.w)
{ {
IPlugin *plugin = new ResizeLayerRT( l->output_dim.c,l->output_dim.h+1,l->output_dim.w+1 ); IPlugin *plugin = new MaxPoolFixedSizeRT(l->output_dim.c, l->output_dim.h, l->output_dim.w, l->output_dim.n, l->strideH, l->strideW, l->winH, l->winH-1);
IPluginLayer *lRT = networkRT->addPlugin(&input, 1, *plugin);
checkNULL(lRT);
lRT->setName( "Resize" );
input = lRT->getOutput(0); IPluginLayer *lRT = networkRT->addPlugin(&input, 1, *plugin);
checkNULL(lRT);
lRT->setName( "MaxPoolingFixedSize" );
return lRT;
} }
else
{
if(l->paddingH == 0 && l->paddingW == 0 && l->input_dim.h == l->output_dim.h && l->input_dim.w == l->output_dim.w)
{
IPlugin *plugin = new ResizeLayerRT( l->output_dim.c,l->output_dim.h+1,l->output_dim.w+1 );
IPluginLayer *lRT = networkRT->addPlugin(&input, 1, *plugin);
checkNULL(lRT);
lRT->setName( "Resize" );
IPoolingLayer *lRT = networkRT->addPooling(*input, ptype, DimsHW{l->winH, l->winW}); input = lRT->getOutput(0);
checkNULL(lRT); }
lRT->setPadding(DimsHW{l->paddingH, l->paddingW}); IPoolingLayer *lRT = networkRT->addPooling(*input, ptype, DimsHW{l->winH, l->winW});
lRT->setStride(DimsHW{l->strideH, l->strideW}); checkNULL(lRT);
return lRT;
lRT->setPadding(DimsHW{l->paddingH, l->paddingW});
lRT->setStride(DimsHW{l->strideH, l->strideW});
return lRT;
}
} }
ILayer* NetworkRT::convert_layer(ITensor *input, Activation *l) { ILayer* NetworkRT::convert_layer(ITensor *input, Activation *l) {
@@ -437,18 +450,18 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Shortcut *l) {
//std::cout<<"New plugin Shortcut\n"; //std::cout<<"New plugin Shortcut\n";
ITensor *back_tens = tensors[l->backLayer]; ITensor *back_tens = tensors[l->backLayer];
/*
// plugin version // plugin version
IPlugin *plugin = new ShortcutRT(); IPlugin *plugin = new ShortcutRT(l->backLayer->output_dim);
ITensor **inputs = new ITensor*[2]; ITensor **inputs = new ITensor*[2];
inputs[0] = input; inputs[0] = input;
inputs[1] = back_tens; inputs[1] = back_tens;
IPluginLayer *lRT = networkRT->addPlugin(inputs, 2, *plugin); IPluginLayer *lRT = networkRT->addPlugin(inputs, 2, *plugin);
checkNULL(lRT); checkNULL(lRT);
*/
IElementWiseLayer *lRT = networkRT->addElementWise(*input, *back_tens, ElementWiseOperation::kSUM);
checkNULL(lRT); // IElementWiseLayer *lRT = networkRT->addElementWise(*input, *back_tens, ElementWiseOperation::kSUM);
// checkNULL(lRT);
return lRT; return lRT;
} }
@@ -602,13 +615,31 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
} }
if(name.find("Shortcut") == 0) { if(name.find("Shortcut") == 0) {
ShortcutRT *r = new ShortcutRT(); tk::dnn::dataDim_t bdim;
bdim.c = readBUF<int>(buf);
bdim.h = readBUF<int>(buf);
bdim.w = readBUF<int>(buf);
bdim.l = 1;
ShortcutRT *r = new ShortcutRT(bdim);
r->c = readBUF<int>(buf); r->c = readBUF<int>(buf);
r->h = readBUF<int>(buf); r->h = readBUF<int>(buf);
r->w = readBUF<int>(buf); r->w = readBUF<int>(buf);
return r; return r;
} }
if(name.find("Pooling") == 0) {
MaxPoolFixedSizeRT *r = new MaxPoolFixedSizeRT( readBUF<int>(buf), //c
readBUF<int>(buf), //h
readBUF<int>(buf), //w
readBUF<int>(buf), //n
readBUF<int>(buf), //strideH
readBUF<int>(buf), //strideW
readBUF<int>(buf), //winSize
readBUF<int>(buf)); //padding
return r;
}
if(name.find("Resize") == 0) { if(name.find("Resize") == 0) {
ResizeLayerRT *r = new ResizeLayerRT(readBUF<int>(buf), //o_c ResizeLayerRT *r = new ResizeLayerRT(readBUF<int>(buf), //o_c
readBUF<int>(buf), //o_h readBUF<int>(buf), //o_h
+3 -3
View File
@@ -7,7 +7,7 @@ namespace tk { namespace dnn {
Pooling::Pooling( Network *net, int winH, int winW, int strideH, int strideW, Pooling::Pooling( Network *net, int winH, int winW, int strideH, int strideW,
int paddingH, int paddingW, int paddingH, int paddingW,
tkdnnPoolingMode_t pool_mode, bool final, bool test) : tkdnnPoolingMode_t pool_mode, bool final, bool maxpoolfixedsize) :
Layer(net, final) { Layer(net, final) {
this->winH = winH; this->winH = winH;
@@ -17,7 +17,7 @@ Pooling::Pooling( Network *net, int winH, int winW, int strideH, int strideW,
this->pool_mode = pool_mode; this->pool_mode = pool_mode;
this->paddingH = paddingH; this->paddingH = paddingH;
this->paddingW = paddingW; this->paddingW = paddingW;
this->test = test; this->maxpoolfixedsize = maxpoolfixedsize;
checkCUDNN( cudnnCreatePoolingDescriptor(&poolingDesc) ); checkCUDNN( cudnnCreatePoolingDescriptor(&poolingDesc) );
@@ -114,7 +114,7 @@ dnnType* Pooling::infer(dataDim_t &dim, dnnType* srcData) {
if(this->test) if(this->maxpoolfixedsize)
{ {
MaxPoolingForward(poolSrc, poolDst, dim.n, dim.c, dim.h, dim.w, this->strideH, this->strideW, this->winH, this->winH-1); MaxPoolingForward(poolSrc, poolDst, dim.n, dim.c, dim.h, dim.w, this->strideH, this->strideW, this->winH, this->winH-1);
} }
File diff suppressed because it is too large Load Diff
@@ -339,7 +339,7 @@ int main()
tk::dnn::Conv2d c83(&net, 512, 1, 1, 1, 1, 0, 0, c83_bin, true); tk::dnn::Conv2d c83(&net, 512, 1, 1, 1, 1, 0, 0, c83_bin, true);
tk::dnn::Activation a83(&net, tk::dnn::ACTIVATION_LEAKY); tk::dnn::Activation a83(&net, tk::dnn::ACTIVATION_LEAKY);
// //SPP //SPP
tk::dnn::Pooling p84(&net, 5, 5, 1, 1,0,0, tk::dnn::POOLING_MAX, false, true); tk::dnn::Pooling p84(&net, 5, 5, 1, 1,0,0, tk::dnn::POOLING_MAX, false, true);
tk::dnn::Layer *r85_layers[1] = {&a83}; tk::dnn::Layer *r85_layers[1] = {&a83};
tk::dnn::Route r85(&net, r85_layers, 1); tk::dnn::Route r85(&net, r85_layers, 1);
@@ -351,7 +351,7 @@ int main()
tk::dnn::Pooling p88(&net, 13, 13, 1, 1, 12, 12, tk::dnn::POOLING_MAX, false, true); tk::dnn::Pooling p88(&net, 13, 13, 1, 1, 12, 12, tk::dnn::POOLING_MAX, false, true);
tk::dnn::Layer *r89_layers[4] = {&p88, &p86, &p84, &a83}; tk::dnn::Layer *r89_layers[4] = {&p88, &p86, &p84, &a83};
tk::dnn::Route r89(&net, r89_layers, 4); tk::dnn::Route r89(&net, r89_layers, 4);
// //END SPP //END SPP
tk::dnn::Conv2d c90(&net, 512, 1, 1, 1, 1, 0, 0, c90_bin, true); tk::dnn::Conv2d c90(&net, 512, 1, 1, 1, 1, 0, 0, c90_bin, true);
tk::dnn::Activation a90(&net, tk::dnn::ACTIVATION_LEAKY); tk::dnn::Activation a90(&net, tk::dnn::ACTIVATION_LEAKY);
@@ -454,17 +454,15 @@ int main()
tk::dnn::Conv2d c136(&net, 255, 1, 1, 1, 1, 0, 0, c136_bin, false); tk::dnn::Conv2d c136(&net, 255, 1, 1, 1, 1, 0, 0, c136_bin, false);
tk::dnn::Yolo yolo137(&net, classes, 3, g137_bin); tk::dnn::Yolo yolo137(&net, classes, 3, g137_bin);
yolo[0] = &yolo115;
yolo[1] = &yolo126;
yolo[2] = &yolo137;
// fill classes names
// yolo[0] = &yolo115; for (int i = 0; i < 3; i++)
// yolo[1] = &yolo126; {
// yolo[2] = &yolo137; yolo[i]->classesNames = {"person", "bicycle", "car", "motorbike", "aeroplane", "bus", "train", "truck", "boat", "traffic light", "fire hydrant", "stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse", "sheep", "cow", "elephant", "bear", "zebra", "giraffe", "backpack", "umbrella", "handbag", "tie", "suitcase", "frisbee", "skis", "snowboard", "sports ball", "kite", "baseball bat", "baseball glove", "skateboard", "surfboard", "tennis racket", "bottle", "wine glass", "cup", "fork", "knife", "spoon", "bowl", "banana", "apple", "sandwich", "orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake", "chair", "sofa", "pottedplant", "bed", "diningtable", "toilet", "tvmonitor", "laptop", "mouse", "remote", "keyboard", "cell phone", "microwave", "oven", "toaster", "sink", "refrigerator", "book", "clock", "vase", "scissors", "teddy bear", "hair drier", "toothbrush"};
}
// // fill classes names
// for (int i = 0; i < 3; i++)
// {
// yolo[i]->classesNames = {"person", "bicycle", "car", "motorbike", "aeroplane", "bus", "train", "truck", "boat", "traffic light", "fire hydrant", "stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse", "sheep", "cow", "elephant", "bear", "zebra", "giraffe", "backpack", "umbrella", "handbag", "tie", "suitcase", "frisbee", "skis", "snowboard", "sports ball", "kite", "baseball bat", "baseball glove", "skateboard", "surfboard", "tennis racket", "bottle", "wine glass", "cup", "fork", "knife", "spoon", "bowl", "banana", "apple", "sandwich", "orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake", "chair", "sofa", "pottedplant", "bed", "diningtable", "toilet", "tvmonitor", "laptop", "mouse", "remote", "keyboard", "cell phone", "microwave", "oven", "toaster", "sink", "refrigerator", "book", "clock", "vase", "scissors", "teddy bear", "hair drier", "toothbrush"};
// }
// Load input // Load input
dnnType *data; dnnType *data;
@@ -475,13 +473,13 @@ int main()
net.print(); net.print();
// //convert network to tensorRT // //convert network to tensorRT
// tk::dnn::NetworkRT netRT(&net, "csresnext50-panet-spp.rt"); tk::dnn::NetworkRT netRT(&net, "csresnext50-panet-spp.rt");
// the network have 3 outputs // the network have 3 outputs
// tk::dnn::dataDim_t out_dim[3]; tk::dnn::dataDim_t out_dim[3];
// for (int i = 0; i < 3; i++) for (int i = 0; i < 3; i++)
// out_dim[i] = yolo[i]->output_dim; out_dim[i] = yolo[i]->output_dim;
// dnnType *cudnn_out[3], *rt_out[3]; dnnType *cudnn_out[3], *rt_out[3];
tk::dnn::dataDim_t dim1 = dim; //input dim tk::dnn::dataDim_t dim1 = dim; //input dim
printCenteredTitle(" CUDNN inference ", '=', 30); printCenteredTitle(" CUDNN inference ", '=', 30);
@@ -492,69 +490,62 @@ int main()
TIMER_STOP TIMER_STOP
dim1.print(); dim1.print();
} }
dnnType *cudnn_out = net.layers[net.num_layers-1]->dstData;
tk::dnn::dataDim_t out_dim = net.layers[net.num_layers-1]->output_dim;
dnnType *out1, *out1_h;
int odim1 = out_dim.tot();
readBinaryFile(output_bin, odim1, &out1_h, &out1);
std::cout << "CUDNN vs correct" << std::endl;
// printDeviceVector(odim1, cudnn_out);
checkResult(odim1, cudnn_out, out1);
// for (int i = 0; i < 3; i++) for (int i = 0; i < 3; i++)
// cudnn_out[i] = yolo[i]->dstData; cudnn_out[i] = yolo[i]->dstData;
// printCenteredTitle(" compute detections ", '=', 30); printCenteredTitle(" compute detections ", '=', 30);
// TIMER_START TIMER_START
// int ndets = 0; int ndets = 0;
// tk::dnn::Yolo::detection *dets = tk::dnn::Yolo::allocateDetections(tk::dnn::Yolo::MAX_DETECTIONS, classes); tk::dnn::Yolo::detection *dets = tk::dnn::Yolo::allocateDetections(tk::dnn::Yolo::MAX_DETECTIONS, classes);
// for (int i = 0; i < 3; i++) for (int i = 0; i < 3; i++)
// yolo[i]->computeDetections(dets, ndets, net.input_dim.w, net.input_dim.h, 0.5); yolo[i]->computeDetections(dets, ndets, net.input_dim.w, net.input_dim.h, 0.5);
// tk::dnn::Yolo::mergeDetections(dets, ndets, classes); tk::dnn::Yolo::mergeDetections(dets, ndets, classes);
// for (int j = 0; j < ndets; j++) for (int j = 0; j < ndets; j++)
// { {
// tk::dnn::Yolo::box b = dets[j].bbox; tk::dnn::Yolo::box b = dets[j].bbox;
// int x0 = (b.x - b.w / 2.); int x0 = (b.x - b.w / 2.);
// int x1 = (b.x + b.w / 2.); int x1 = (b.x + b.w / 2.);
// int y0 = (b.y - b.h / 2.); int y0 = (b.y - b.h / 2.);
// int y1 = (b.y + b.h / 2.); int y1 = (b.y + b.h / 2.);
// int cl = 0; int cl = 0;
// for (int c = 0; c < classes; ++c) for (int c = 0; c < classes; ++c)
// { {
// float prob = dets[j].prob[c]; float prob = dets[j].prob[c];
// if (prob > 0) if (prob > 0)
// cl = c; cl = c;
// } }
// std::cout << cl << ": " << x0 << " " << y0 << " " << x1 << " " << y1 << "\n"; std::cout << cl << ": " << x0 << " " << y0 << " " << x1 << " " << y1 << "\n";
// } }
// TIMER_STOP TIMER_STOP
// tk::dnn::dataDim_t dim2 = dim; tk::dnn::dataDim_t dim2 = dim;
// printCenteredTitle(" TENSORRT inference ", '=', 30); printCenteredTitle(" TENSORRT inference ", '=', 30);
// { {
// dim2.print(); dim2.print();
// TIMER_START TIMER_START
// netRT.infer(dim2, data); netRT.infer(dim2, data);
// TIMER_STOP TIMER_STOP
// dim2.print(); dim2.print();
// } }
// for (int i = 0; i < 3; i++)
// rt_out[i] = (dnnType *)netRT.buffersRT[i + 1];
// for (int i = 0; i < 3; i++) for (int i = 0; i < 3; i++)
// { rt_out[i] = (dnnType *)netRT.buffersRT[i + 1];
// printCenteredTitle((std::string(" YOLO ") + std::to_string(i) + " CHECK RESULTS ").c_str(), '=', 30);
// dnnType *out, *out_h; for (int i = 0; i < 3; i++)
// int odim = out_dim[i].tot(); {
// readBinaryFile(output_bins[i], odim, &out_h, &out); printCenteredTitle((std::string(" YOLO ") + std::to_string(i) + " CHECK RESULTS ").c_str(), '=', 30);
// std::cout << "CUDNN vs correct"; dnnType *out, *out_h;
// checkResult(odim, cudnn_out[i], out); int odim = out_dim[i].tot();
// std::cout << "TRT vs correct"; readBinaryFile(output_bins[i], odim, &out_h, &out);
// checkResult(odim, rt_out[i], out); std::cout << "CUDNN vs correct";
// std::cout << "CUDNN vs TRT "; checkResult(odim, cudnn_out[i], out);
// checkResult(odim, cudnn_out[i], rt_out[i]); std::cout << "TRT vs correct";
// } checkResult(odim, rt_out[i], out);
std::cout << "CUDNN vs TRT ";
checkResult(odim, cudnn_out[i], rt_out[i]);
}
return 0; return 0;
} }