diff --git a/CMakeLists.txt b/CMakeLists.txt index 47780ea..dc9ac81 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -82,11 +82,8 @@ target_link_libraries(test_yolo3_berkeley tkDNN) add_executable(test_rtinference tests/test_rtinference/rtinference.cpp) target_link_libraries(test_rtinference tkDNN) -add_executable(detection demo/detection/detection.cpp) -target_link_libraries(detection tkDNN) - -add_executable(live demo/live/live.cpp) -target_link_libraries(live tkDNN) +add_executable(yolo3_demo demo/demo/demo.cpp) +target_link_libraries(yolo3_demo tkDNN) #install #if (CMAKE_INSTALL_PREFIX_INITIALIZED_TO_DEFAULT) diff --git a/README.md b/README.md index d382812..ae0d203 100644 --- a/README.md +++ b/README.md @@ -1,11 +1,11 @@ # tkDNN -tkDNN is a Deep Neural Network library built with cuDNN primitives specifically thought to work on NVIDIA TK1 board.
+tkDNN is a Deep Neural Network library built with cuDNN primitives specifically thought to work on NVIDIA TK1(and all successive) board.
The main scope is to do high performance inference on already trained models. this branch actually work on every NVIDIA GPU that support the dependencies: -* CUDA 8 -* CUDNN 6 -* TENSORRT 2 +* CUDA 9 +* CUDNN 7.105 +* TENSORRT 4.02 ## Workflow The recommended workflow follow these step: @@ -32,18 +32,20 @@ Assumiung you have correctly builded the library these are the test ready to exe * test_mnistRT: the mnist network hardcoded in using tensorRT apis (TENSORRT only) * test_yolo: YOLO detection network (CUDNN and TENSORRT) * test_yolo_tiny: smaller version of YOLO (CUDNN and TENSRRT) +* test_yolo3_berkeley: our yolo3 version trained with BDD100K dateset -## Live detection +## yolo3 berkeley demo detection For the live detection you need to precompile the tensorRT file by luncing the desidered network test, this is the recommended process: ``` export TKDNN_MODE=FP16 # set the half floating point optimization -rm yolo.rt # be sure to delete(or move) old tensorRT files -./test_yolo # run the yolo test (is slow) +rm yolo3_berkeley.rt # be sure to delete(or move) old tensorRT files +./test_yolo3_berkeley # run the yolo test (is slow) # with f16 inference the result will be a bit incorrect ``` -this will genereate a yolo.rt file that can be used for live detection: +this will genereate a yolo3_berkeley.rt file that can be used for live detection: ``` -./live yolo.rt 1 -s -t0.3 # launch detection on device 1 with 0.3 thresh +./demo # launch detection on a demo video +./demo /dev/video0 # launch detection on device 0 ``` diff --git a/demo/demo/demo.cpp b/demo/demo/demo.cpp new file mode 100644 index 0000000..7a0d547 --- /dev/null +++ b/demo/demo/demo.cpp @@ -0,0 +1,75 @@ +#include +#include +#include /* srand, rand */ +#include +#include +#include "utils.h" + +#include +#include +#include + +#include "Yolo3Detection.h" + +bool gRun; + +void sig_handler(int signo) { + std::cout<<"request gateway stop\n"; + gRun = false; +} + +int main(int argc, char *argv[]) { + + std::cout<<"detection\n"; + signal(SIGINT, sig_handler); + + Yolo3Detection yolo; + yolo.init("./"); + + gRun = true; + + char *input = "../demo/yolo_test.mp4"; + if(argc > 1) + input = argv[1]; + + cv::VideoCapture cap(input); + if(!cap.isOpened()) + gRun = false; + else + std::cout<<"camera started\n"; + + cv::Mat frame; + cv::namedWindow("detection", cv::WINDOW_NORMAL); + cv::resizeWindow("detection", 544*1.2, 320*1.2); + + while(gRun) { + cap >> frame; + if(!frame.data) { + continue; + } + + yolo.update(frame); + + // draw dets + for(int i=0; i -#include "tkdnn.h" -#include /* srand, rand */ -#include - -#include -#include -#include - -const char *reg_bias = "../tests/yolo/layers/g31.bin"; - -int prob_sort(const void *pa, const void *pb) { - tk::dnn::box a = *(tk::dnn::box *)pa; - tk::dnn::box b = *(tk::dnn::box *)pb; - float diff = a.prob - b.prob; - if(diff < 0) return 1; - else if(diff > 0) return -1; - return 0; -} - -cv::Mat GetSquareImage(const cv::Mat& img, int target_width) { - int width = img.cols, height = img.rows; - - cv::Mat square = cv::Mat::zeros( target_width, target_width, img.type() ); - - int max_dim = ( width >= height ) ? width : height; - float scale = ( ( float ) target_width ) / max_dim; - cv::Rect roi; - if ( width >= height ) - { - roi.width = target_width; - roi.x = 0; - roi.height = height * scale; - roi.y = ( target_width - roi.height ) / 2; - } - else - { - roi.y = 0; - roi.height = target_width; - roi.width = width * scale; - roi.x = ( target_width - roi.width ) / 2; - } - - cv::resize( img, square( roi ), roi.size() ); - - return square; -} - -//return inference time -double compute_image( cv::Mat imageORIG, - tk::dnn::NetworkRT *netRT, tk::dnn::RegionInterpret *rI, - dnnType *input, dnnType *output) { - - //Resize with padding and convert to float - cv::Mat image = GetSquareImage(imageORIG, netRT->input_dim.w); - cv::Mat imageF; - image.convertTo(imageF, CV_32FC3, 1/255.0); - - //split channels - cv::Mat bgr[3]; //destination array - cv::split(imageF,bgr);//split source - - //write channels - int idx = 0; - memcpy((void*)&input[idx], (void*)bgr[2].data, imageF.rows*imageF.cols*sizeof(dnnType)); - idx = imageF.rows*imageF.cols; - memcpy((void*)&input[idx], (void*)bgr[1].data, imageF.rows*imageF.cols*sizeof(dnnType)); - idx *= 2; - memcpy((void*)&input[idx], (void*)bgr[0].data, imageF.rows*imageF.cols*sizeof(dnnType)); - - //DO INFERENCE - printCenteredTitle(" TENSORRT inference ", '=', 30); - TIMER_START - checkCuda( cudaMemcpyAsync(netRT->buffersRT[netRT->buf_input_idx], input, - netRT->input_dim.tot()*sizeof(float), - cudaMemcpyHostToDevice, netRT->stream)); - netRT->enqueue(); - checkCuda( cudaMemcpyAsync(output, netRT->buffersRT[netRT->buf_output_idx], - netRT->output_dim.tot()*sizeof(float), - cudaMemcpyDeviceToHost, netRT->stream)); - cudaStreamSynchronize(netRT->stream); - TIMER_STOP - - - rI->interpretData(output, imageORIG.cols, imageORIG.rows); - - return t_ns; -} - -int print_usage() { - std::cout<<"usage: ./detection net.rt validation_list.txt" - <<" [-t ] [-s] [-i ]\n" - <<" -t: set thresh value\n -s: show images as compute\n" - <<" -i: images to compute\n\n" - <<"> validation_list.txt format: \n" - <<" path/to/image.jpg path/to/label.txt\n" - <<"> label.txt format: \n" - <<" \n" - <<" x and y are the box center, " - <<"all values are relative to the image size\n\n"; - return 1; -} - - -int main(int argc, char *argv[]) { - - //params - char *tensor_path = NULL; - char *imageset_path = NULL; - float thresh = 0.3f; - bool show = false; - int iterations = INT_MAX; - - //parse params - int c; - while ((c = getopt (argc, argv, "t:si:")) != -1) { - switch(c) { - case 't': thresh = atof(optarg); break; - case 's': show = true; break; - case 'i': iterations = atoi(optarg); break; - case '?': - return print_usage(); - default: return print_usage(); - } - } - - if(argc - optind == 2) { - tensor_path = argv[optind]; - imageset_path = argv[optind+1]; - } else { - std::cout<<"not enough arguments.\n"; - return print_usage(); - } - //end parsing - - if(!fileExist(tensor_path)) - FatalError("unable to read serialRT file"); - //convert network to tensorRT - tk::dnn::NetworkRT netRT(NULL, tensor_path); - tk::dnn::RegionInterpret rI(netRT.input_dim, netRT.output_dim, 80, 4, 5, thresh, reg_bias); - - dnnType *input = new float[netRT.input_dim.tot()]; - dnnType *output = new float[netRT.output_dim.tot()]; - - std::string line; - std::ifstream imageset(imageset_path); - if(!imageset.is_open()) - FatalError("could not read imageset"); - - double mTime = 0; - float mAP = 0; - int processed_images; - - for(processed_images=1; - processed_images-1 < iterations && getline(imageset, line); - processed_images++) { - - std::string image_path = line.substr(0, line.find(" ")); - std::string label_path = line.substr(line.find(" ")+1, line.size()); - std::cout<>cl) { - labels>>x>>y>>w>>h; - w *= img.cols; x *= img.cols; - h *= img.rows; y *= img.rows; - std::cout<=1; i--) { //for each detected evaluate sub group - - int prec = 0; - for(int j=0; j 0.6f && rI.res_boxes[j].cl == gt[z].cl) { - prec++; - break; - } - } - } - - AP += float(prec)/i; - } - AP = AP/gt_n; - std::cout<<"AP: "< -#include "tkdnn.h" -#include /* srand, rand */ -#include - -#include -#include -#include - -#define VOC - -#ifdef VOC -const char *reg_bias = "../tests/yolo_voc/layers/g31.bin"; -#define CLASS 20 -#else -const char *reg_bias = "../tests/yolo/layers/g31.bin"; -#define CLASS 80 -#endif - -int prob_sort(const void *pa, const void *pb) { - tk::dnn::box a = *(tk::dnn::box *)pa; - tk::dnn::box b = *(tk::dnn::box *)pb; - float diff = a.prob - b.prob; - if(diff < 0) return 1; - else if(diff > 0) return -1; - return 0; -} - -cv::Mat GetSquareImage(const cv::Mat& img, int target_width) { - int width = img.cols, height = img.rows; - - cv::Mat square = cv::Mat::zeros( target_width, target_width, img.type() ); - - int max_dim = ( width >= height ) ? width : height; - float scale = ( ( float ) target_width ) / max_dim; - cv::Rect roi; - if ( width >= height ) - { - roi.width = target_width; - roi.x = 0; - roi.height = height * scale; - roi.y = ( target_width - roi.height ) / 2; - } - else - { - roi.y = 0; - roi.height = target_width; - roi.width = width * scale; - roi.x = ( target_width - roi.width ) / 2; - } - - cv::resize( img, square( roi ), roi.size() ); - - return square; -} - -//return inference time -double compute_image( cv::Mat imageORIG, - tk::dnn::NetworkRT *netRT, tk::dnn::RegionInterpret *rI, - dnnType *input, dnnType *output) { - TIMER_START - - //Resize with padding and convert to float - cv::Mat image = GetSquareImage(imageORIG, netRT->input_dim.w); - cv::Mat imageF; - image.convertTo(imageF, CV_32FC3, 1/255.0); - - //split channels - cv::Mat bgr[3]; //destination array - cv::split(imageF,bgr);//split source - - //write channels - int idx = 0; - memcpy((void*)&input[idx], (void*)bgr[2].data, imageF.rows*imageF.cols*sizeof(dnnType)); - idx = imageF.rows*imageF.cols; - memcpy((void*)&input[idx], (void*)bgr[1].data, imageF.rows*imageF.cols*sizeof(dnnType)); - idx *= 2; - memcpy((void*)&input[idx], (void*)bgr[0].data, imageF.rows*imageF.cols*sizeof(dnnType)); - - //DO INFERENCE - checkCuda( cudaMemcpyAsync(netRT->buffersRT[netRT->buf_input_idx], input, - netRT->input_dim.tot()*sizeof(float), - cudaMemcpyHostToDevice, netRT->stream)); - netRT->enqueue(); - checkCuda( cudaMemcpyAsync(output, netRT->buffersRT[netRT->buf_output_idx], - netRT->output_dim.tot()*sizeof(float), - cudaMemcpyDeviceToHost, netRT->stream)); - cudaStreamSynchronize(netRT->stream); - - - - rI->interpretData(output, imageORIG.cols, imageORIG.rows); - - TIMER_STOP - return t_ns; -} - - - -int print_usage() { - std::cout<<"usage: ./live net.rt input_uri\n"; - return 1; -} - - - - -int main(int argc, char *argv[]) { - - //params - char *tensor_path = NULL; - char *device = 0; - float thresh = 0.3f; - bool show = false; - - //parse params - int c; - while ((c = getopt (argc, argv, "t:si:")) != -1) { - switch(c) { - case 't': thresh = atof(optarg); break; - case 's': show = true; break; - case '?': - return print_usage(); - default: return print_usage(); - } - } - - if(argc - optind == 2) { - tensor_path = argv[optind]; - device = argv[optind+1]; - } else { - std::cout<<"not enough arguments.\n"; - return print_usage(); - } - //end parsing - - std::cout<<"open video stream on device: "<> img; - - if(!img.data) - FatalError("Could not open image"); - std::cout<<"Image size: ("< +#include +#include /* srand, rand */ +#include +#include +#include "utils.h" + +#include +#include +#include + +#include + +/** + * + * @author Francesco Gatti + */ +class Yolo3Detection { + + private: + tk::dnn::NetworkRT *netRT = nullptr; + tk::dnn::Yolo* yolo[3]; + dnnType *input, *input_d; + + int ndets = 0; + tk::dnn::Yolo::detection *dets = nullptr; + + cv::Mat imageF; + cv::Mat bgr[3]; + + public: + static const int classes = 10; + static const int num = 3; + float thresh = 0.3; + cv::Scalar colors[classes]; + + // this is filled with results + std::vector detected; + + Yolo3Detection() {} + + virtual ~Yolo3Detection() {} + + /** + * Method used for inizialize the class + * + * @return Success of the initialization + */ + bool init(std::string tensor_path); + + void update(cv::Mat &frame); + +}; diff --git a/src/Yolo3Detection.cpp b/src/Yolo3Detection.cpp new file mode 100644 index 0000000..fd492b6 --- /dev/null +++ b/src/Yolo3Detection.cpp @@ -0,0 +1,128 @@ +#include "Yolo3Detection.h" + +bool Yolo3Detection::init(std::string tensor_folder) { + + //const char *tensor_path = "../data/yolo3/yolo3_berkeley.rt"; + + // class colors precompute + for(int c=0; c 1) r = 1; + if(g > 1) g = 1; + if(b > 1) b = 1; + //std::cout<input_dim = yolo[0]->output_dim = tk::dnn::dataDim_t(1, 45, 10, 17); + yolo[1] = new tk::dnn::Yolo(nullptr, classes, num, (tensor_folder + "/yolo3_1.bin").c_str() ); // yolo without input and bias + yolo[1]->input_dim = yolo[1]->output_dim = tk::dnn::dataDim_t(1, 45, 20, 34); + yolo[2] = new tk::dnn::Yolo(nullptr, classes, num, (tensor_folder + "/yolo3_2.bin").c_str() ); // yolo without input and bias + yolo[2]->input_dim = yolo[2]->output_dim = tk::dnn::dataDim_t(1, 45, 40, 68); + + dets = tk::dnn::Yolo::allocateDetections(tk::dnn::Yolo::MAX_DETECTIONS, classes); + + checkCuda(cudaMallocHost(&input, sizeof(dnnType)*netRT->input_dim.tot())); + checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot())); + + return true; +} + + +void Yolo3Detection::update(cv::Mat &imageORIG) { + + if(!imageORIG.data) { + std::cout<<"YOLO: NO IMAGE DATA\n"; + return; + } + + resize(imageORIG, imageORIG, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); + imageORIG.convertTo(imageF, CV_32FC3, 1/255.0); + + //split channels + cv::split(imageF,bgr);//split source + + //write channels + int idx = 0; + memcpy((void*)&input[idx], (void*)bgr[2].data, imageF.rows*imageF.cols*sizeof(dnnType)); + idx = imageF.rows*imageF.cols; + memcpy((void*)&input[idx], (void*)bgr[1].data, imageF.rows*imageF.cols*sizeof(dnnType)); + idx *= 2; + memcpy((void*)&input[idx], (void*)bgr[0].data, imageF.rows*imageF.cols*sizeof(dnnType)); + + //DO INFERENCE + dnnType *rt_out[3]; + tk::dnn::dataDim_t dim = netRT->input_dim; + checkCuda(cudaMemcpyAsync(input_d, input, dim.tot()*sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream)); + + printCenteredTitle(" TENSORRT inference ", '=', 30); { + dim.print(); + TIMER_START + netRT->infer(dim, input_d); + TIMER_STOP + dim.print(); + } + + TIMER_START + // compute dets + ndets = 0; + for(int i=0; i<3; i++) { + rt_out[i] = (dnnType*)netRT->buffersRT[i+1]; + yolo[i]->dstData = rt_out[i]; + yolo[i]->computeDetections(dets, ndets, netRT->input_dim.w, netRT->input_dim.h, netRT->input_dim.w, netRT->input_dim.h, thresh); + } + tk::dnn::Yolo::mergeDetections(dets, ndets, classes); + TIMER_STOP + + + float xRatio = float(imageORIG.cols) / float(netRT->input_dim.w); + float yRatio = float(imageORIG.rows) / float(netRT->input_dim.h); + + // fill detected + detected.clear(); + for(int j=0; j= thresh) { + obj_class = c; + prob = dets[j].prob[c]; + } + } + + if(obj_class >= 0) { + //std::cout<