Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| fe2b06d607 |
@@ -1,9 +1,9 @@
|
||||
# tkDNN
|
||||
tkDNN is a Deep Neural Network library built with cuDNN and tensorRT primitives, specifically thought to work on NVIDIA Jetson Boards. It has been tested on TK1(branch cudnn2), TX1, TX2, AGX Xavier, Nano and several discrete GPUs.
|
||||
tkDNN is a Deep Neural Network library built with cuDNN and tensorRT primitives, specifically thought to work on NVIDIA Jetson Boards. It has been tested on TK1(branch cudnn2), TX1, TX2, AGX Xavier and several discrete GPU.
|
||||
The main goal of this project is to exploit NVIDIA boards as much as possible to obtain the best inference performance. It does not allow training.
|
||||
|
||||
|
||||
If you use tkDNN in your research, please cite one of the following papers. For use in commercial solutions, write at gattifrancesco@hotmail.it and micaela.verucchi@unimore.it or refer to https://hipert.unimore.it/ .
|
||||
If you use tkDNN in your research, please cite one of the following papers. For use in commercial solutions, write at gattifrancesco@hotmail.it or refer to https://hipert.unimore.it/ .
|
||||
|
||||
```
|
||||
Accepted paper @ IRC 2020, will soon be published.
|
||||
@@ -175,25 +175,15 @@ All models from darknet are now parsed directly from cfg, you still need to expo
|
||||
mish
|
||||
</details>
|
||||
|
||||
## Run the demo
|
||||
This is an example using yolov4.
|
||||
## Run the demo
|
||||
|
||||
To run the an object detection first create the .rt file by running:
|
||||
To run the an object detection demo follow these steps (example with yolov3):
|
||||
```
|
||||
rm yolo4_fp32.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_yolo4 # run the yolo test (is slow)
|
||||
rm yolo3_fp32.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_yolo3 # run the yolo test (is slow)
|
||||
./demo yolo3_fp32.rt ../demo/yolo_test.mp4 y
|
||||
```
|
||||
If you get problems in the creation, try to check the error activating the debug of TensorRT in this way:
|
||||
```
|
||||
cmake .. -DDEBUG=True
|
||||
make
|
||||
```
|
||||
|
||||
Once you have succesfully created your rt file, run the demo:
|
||||
```
|
||||
./demo yolo4_fp32.rt ../demo/yolo_test.mp4 y
|
||||
```
|
||||
In general the demo program takes 6 parameters:
|
||||
In general the demo program takes 4 parameters:
|
||||
```
|
||||
./demo <network-rt-file> <path-to-video> <kind-of-network> <number-of-classes> <n-batches> <show-flag>
|
||||
```
|
||||
@@ -207,7 +197,6 @@ where
|
||||
|
||||
N.b. By default it is used FP32 inference
|
||||
|
||||
|
||||

|
||||
|
||||
### FP16 inference
|
||||
|
||||
@@ -13,7 +13,6 @@ private:
|
||||
int num = 0;
|
||||
int nMasks = 0;
|
||||
int nDets = 0;
|
||||
bool letterbox = false;
|
||||
tk::dnn::Yolo::detection *dets = nullptr;
|
||||
tk::dnn::Yolo* yolo[3];
|
||||
|
||||
@@ -22,7 +21,7 @@ private:
|
||||
cv::Mat bgr_h;
|
||||
|
||||
public:
|
||||
Yolo3Detection(const bool letter_box=false) :letterbox(letter_box){}
|
||||
Yolo3Detection() {};
|
||||
~Yolo3Detection() {};
|
||||
|
||||
bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1);
|
||||
|
||||
+45
-122
@@ -52,79 +52,27 @@ bool Yolo3Detection::init(const std::string& tensor_path, const int n_classes, c
|
||||
return true;
|
||||
}
|
||||
|
||||
cv::Mat resize_image(cv::Mat im, int w, int h)
|
||||
{
|
||||
cv::Mat resized = cv::Mat(cv::Size(w,h), CV_32FC3, cv::Scalar(0) );
|
||||
cv::Mat part = cv::Mat(cv::Size(w,im.rows), CV_32FC3, cv::Scalar(0) );
|
||||
int r, c, k;
|
||||
float w_scale = (float)(im.cols - 1) / (w - 1);
|
||||
float h_scale = (float)(im.rows - 1) / (h - 1);
|
||||
|
||||
for(k = 0; k < im.channels(); ++k){
|
||||
for(r = 0; r < im.rows; ++r){
|
||||
for(c = 0; c < w; ++c){
|
||||
float val = 0;
|
||||
if(c == w-1 || im.cols == 1){
|
||||
val = im.at<cv::Vec3f>(r, im.cols-1)[k];
|
||||
} else {
|
||||
float sx = c*w_scale;
|
||||
int ix = (int) sx;
|
||||
float dx = sx - ix;
|
||||
val = (1 - dx) * im.at<cv::Vec3f>(r, ix)[k] + dx * im.at<cv::Vec3f>(r,ix+1)[k];
|
||||
}
|
||||
part.at<cv::Vec3f>(r,c)[k] = val;
|
||||
}
|
||||
}
|
||||
}
|
||||
for(k = 0; k < im.channels(); ++k){
|
||||
for(r = 0; r < h; ++r){
|
||||
float sy = r*h_scale;
|
||||
int iy = (int) sy;
|
||||
float dy = sy - iy;
|
||||
for(c = 0; c < w; ++c){
|
||||
float val = (1-dy) * part.at<cv::Vec3f>(iy, c)[k];
|
||||
resized.at<cv::Vec3f>(r, c)[k] = val;
|
||||
}
|
||||
if(r == h-1 || im.rows == 1) continue;
|
||||
for(c = 0; c < w; ++c){
|
||||
float val = dy * part.at<cv::Vec3f>(iy+1, c)[k];
|
||||
resized.at<cv::Vec3f>(r,c)[k] += val;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return resized;
|
||||
}
|
||||
|
||||
void Yolo3Detection::preprocess(cv::Mat &frame, const int bi){
|
||||
frame.convertTo(imagePreproc, CV_32FC3, 1/255.0);
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
cv::cuda::GpuMat orig_img, img_resized;
|
||||
orig_img = cv::cuda::GpuMat(frame);
|
||||
cv::cuda::resize(orig_img, img_resized, cv::Size(netRT->input_dim.w, netRT->input_dim.h));
|
||||
|
||||
if(letterbox){
|
||||
int im_w = frame.cols;
|
||||
int im_h = frame.rows;
|
||||
int net_w = netRT->input_dim.w;
|
||||
int net_h = netRT->input_dim.h;
|
||||
if(net_w == net_h && letterbox){
|
||||
float ratio = ( im_w > im_h ) ? float(im_w)/float(net_w) : float(im_h)/float(net_h);
|
||||
img_resized.convertTo(imagePreproc, CV_32FC3, 1/255.0);
|
||||
|
||||
int new_h = im_h/ratio;
|
||||
int new_w = im_w/ratio;
|
||||
//split channels
|
||||
cv::cuda::split(imagePreproc,bgr);//split source
|
||||
|
||||
imagePreproc = resize_image(imagePreproc, new_w, new_h);
|
||||
|
||||
cv::Mat borders;
|
||||
int top = (net_h - new_h)/2;
|
||||
int bottom = (net_h - new_h) - top;
|
||||
int left = (net_w - new_w)/2;
|
||||
int right = (net_w - new_w) - left;
|
||||
|
||||
cv::copyMakeBorder(imagePreproc,imagePreproc, top, bottom, left, right, cv::BORDER_CONSTANT, cv::Scalar(0.5,0.5,0.5));
|
||||
}
|
||||
else
|
||||
FatalError("letterbox not spported with h!=w");
|
||||
//write channels
|
||||
for(int i=0; i<netRT->input_dim.c; i++) {
|
||||
int size = imagePreproc.rows * imagePreproc.cols;
|
||||
int ch = netRT->input_dim.c-1 -i;
|
||||
bgr[ch].download(bgr_h); //TODO: don't copy back on CPU
|
||||
checkCuda( cudaMemcpy(input_d + i*size + netRT->input_dim.tot()*bi, (float*)bgr_h.data, size*sizeof(dnnType), cudaMemcpyHostToDevice));
|
||||
}
|
||||
else
|
||||
imagePreproc = resize_image(imagePreproc, netRT->input_dim.w, netRT->input_dim.h);
|
||||
#else
|
||||
cv::resize(frame, frame, cv::Size(netRT->input_dim.w, netRT->input_dim.h));
|
||||
frame.convertTo(imagePreproc, CV_32FC3, 1/255.0);
|
||||
|
||||
//split channels
|
||||
cv::split(imagePreproc,bgr);//split source
|
||||
@@ -136,6 +84,7 @@ void Yolo3Detection::preprocess(cv::Mat &frame, const int bi){
|
||||
memcpy((void*)&input[idx + netRT->input_dim.tot()*bi], (void*)bgr[ch].data, imagePreproc.rows*imagePreproc.cols*sizeof(dnnType));
|
||||
}
|
||||
checkCuda(cudaMemcpyAsync(input_d + netRT->input_dim.tot()*bi, input + netRT->input_dim.tot()*bi, netRT->input_dim.tot()*sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream));
|
||||
#endif
|
||||
}
|
||||
|
||||
void Yolo3Detection::postprocess(const int bi, const bool mAP){
|
||||
@@ -156,68 +105,42 @@ void Yolo3Detection::postprocess(const int bi, const bool mAP){
|
||||
}
|
||||
tk::dnn::Yolo::mergeDetections(dets, nDets, classes);
|
||||
|
||||
int im_w = originalSize[bi].width;
|
||||
int im_h = originalSize[bi].height;
|
||||
int net_w = netRT->input_dim.w;
|
||||
int net_h = netRT->input_dim.h;
|
||||
int new_h, new_w;
|
||||
|
||||
int top = 0, left = 0;
|
||||
|
||||
if(letterbox){
|
||||
float ratio = ( im_w > im_h ) ? float(im_w)/float(net_w) : float(im_h)/float(net_h);
|
||||
x_ratio = ratio;
|
||||
y_ratio = ratio;
|
||||
std::cout<<ratio<<std::endl;
|
||||
|
||||
int new_h = im_h/ratio;
|
||||
int new_w = im_w/ratio;
|
||||
|
||||
top = (net_h - new_h)/2;
|
||||
left = (net_w - new_w)/2;
|
||||
}
|
||||
else{
|
||||
new_h = net_h;
|
||||
new_w = net_w;
|
||||
}
|
||||
|
||||
float deltaw = net_w - new_w;
|
||||
float deltah = net_h - new_h;
|
||||
float ratiow = (float)new_w / net_w;
|
||||
float ratioh = (float)new_h / net_h;
|
||||
|
||||
// fill detected
|
||||
detected.clear();
|
||||
for(int j=0; j<nDets; j++) {
|
||||
tk::dnn::Yolo::box b = dets[j].bbox;
|
||||
|
||||
float x0 = (b.x - left - b.w/2.);
|
||||
float x1 = (b.x - left + b.w/2.);
|
||||
float y0 = (b.y - top - b.h/2.);
|
||||
float y1 = (b.y - top + b.h/2.);
|
||||
|
||||
// convert to image coords
|
||||
x0 = x_ratio*x0;
|
||||
x1 = x_ratio*x1;
|
||||
y0 = y_ratio*y0;
|
||||
y1 = y_ratio*y1;
|
||||
|
||||
int x0 = (b.x-b.w/2.);
|
||||
int x1 = (b.x+b.w/2.);
|
||||
int y0 = (b.y-b.h/2.);
|
||||
int y1 = (b.y+b.h/2.);
|
||||
int obj_class = -1;
|
||||
float prob = 0;
|
||||
for(int c=0; c<classes; c++) {
|
||||
if(dets[j].prob[c] >= confThreshold) {
|
||||
int obj_class = c;
|
||||
float prob = dets[j].prob[c];
|
||||
|
||||
tk::dnn::box res;
|
||||
res.cl = obj_class;
|
||||
res.prob = prob;
|
||||
res.x = x0;
|
||||
res.y = y0;
|
||||
res.w = x1 - x0;
|
||||
res.h = y1 - y0;
|
||||
|
||||
detected.push_back(res);
|
||||
obj_class = c;
|
||||
prob = dets[j].prob[c];
|
||||
}
|
||||
}
|
||||
|
||||
if(obj_class >= 0) {
|
||||
// convert to image coords
|
||||
x0 = x_ratio*x0;
|
||||
x1 = x_ratio*x1;
|
||||
y0 = y_ratio*y0;
|
||||
y1 = y_ratio*y1;
|
||||
|
||||
tk::dnn::box res;
|
||||
res.cl = obj_class;
|
||||
res.prob = prob;
|
||||
res.x = x0;
|
||||
res.y = y0;
|
||||
res.w = x1 - x0;
|
||||
res.h = y1 - y0;
|
||||
if(mAP)
|
||||
for(int c=0; c<classes; c++)
|
||||
res.probs.push_back(dets[j].prob[c]);
|
||||
detected.push_back(res);
|
||||
}
|
||||
}
|
||||
batchDetected.push_back(detected);
|
||||
}
|
||||
|
||||
+2
-4
@@ -342,16 +342,14 @@ void printJsonCOCOFormat(std::ofstream *out_file, const std::string image_path,
|
||||
//min threshold confidence is set in DetectionNN.h
|
||||
if (bbox[i].probs[j] > 0) {
|
||||
|
||||
*out_file << std::fixed << std::setprecision(6) <<
|
||||
"{\"image_id\":" << image_id <<
|
||||
*out_file << "{\"image_id\":" << image_id <<
|
||||
", \"category_id\":" << coco_ids[j] <<
|
||||
", \"bbox\":[" << bx << ", " << by << ", " << bw << ", " << bh <<
|
||||
"], \"score\":" << bbox[i].probs[j] << "},\n";
|
||||
}
|
||||
}
|
||||
else
|
||||
*out_file << std::fixed << std::setprecision(6) <<
|
||||
"{\"image_id\":" << image_id <<
|
||||
*out_file << "{\"image_id\":" << image_id <<
|
||||
", \"category_id\":" << coco_ids[bbox[i].cl] <<
|
||||
", \"bbox\":[" << bx << ", " << by << ", " << bw << ", " << bh <<
|
||||
"], \"score\":" << bbox[i].prob << "},\n";
|
||||
|
||||
Reference in New Issue
Block a user