diff --git a/CMakeLists.txt b/CMakeLists.txt index 225c9e1..f305500 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -80,6 +80,9 @@ target_link_libraries(test_rtinference tkDNN) add_executable(detection demo/detection/detection.cpp) target_link_libraries(detection tkDNN) +add_executable(live demo/live/live.cpp) +target_link_libraries(live tkDNN) + #install if (CMAKE_INSTALL_PREFIX_INITIALIZED_TO_DEFAULT) set (CMAKE_INSTALL_PREFIX "${CMAKE_BINARY_DIR}/install" diff --git a/README.md b/README.md index 0f08d93..557736a 100644 --- a/README.md +++ b/README.md @@ -33,3 +33,17 @@ Assumiung you have correctly builded the library these are the test ready to exe * test_yolo: YOLO detection network (CUDNN and TENSORRT) * test_yolo_tiny: smaller version of YOLO (CUDNN and TENSRRT) +## Live detection +For the live detection you need to precompile the tensorRT file by luncing the desidered network test, this is the recommended process: +``` +export TKDNN_MODE=FP16 # set the half floating point optimization +rm yolo.rt # be sure to delete(or move) old tensorRT files +./test_yolo # run the yolo test (is slow) +# with f16 inference the result will be a bit incorrect +``` +this will genereate a yolo.rt file that can be used for live detection: +``` +./live yolo.rt 1 -s -t0.3 # launch detection on device 1 with 0.3 thresh +``` + + diff --git a/demo/live/live.cpp b/demo/live/live.cpp new file mode 100644 index 0000000..5c1a833 --- /dev/null +++ b/demo/live/live.cpp @@ -0,0 +1,184 @@ +#include +#include "tkdnn.h" +#include /* srand, rand */ +#include + +#include +#include +#include + +const char *reg_bias = "../tests/yolo/layers/g31.bin"; + +int prob_sort(const void *pa, const void *pb) { + tkDNN::box a = *(tkDNN::box *)pa; + tkDNN::box b = *(tkDNN::box *)pb; + float diff = a.prob - b.prob; + if(diff < 0) return 1; + else if(diff > 0) return -1; + return 0; +} + +cv::Mat GetSquareImage(const cv::Mat& img, int target_width) { + int width = img.cols, height = img.rows; + + cv::Mat square = cv::Mat::zeros( target_width, target_width, img.type() ); + + int max_dim = ( width >= height ) ? width : height; + float scale = ( ( float ) target_width ) / max_dim; + cv::Rect roi; + if ( width >= height ) + { + roi.width = target_width; + roi.x = 0; + roi.height = height * scale; + roi.y = ( target_width - roi.height ) / 2; + } + else + { + roi.y = 0; + roi.height = target_width; + roi.width = width * scale; + roi.x = ( target_width - roi.width ) / 2; + } + + cv::resize( img, square( roi ), roi.size() ); + + return square; +} + +//return inference time +double compute_image( cv::Mat imageORIG, + tkDNN::NetworkRT *netRT, tkDNN::RegionInterpret *rI, + dnnType *input, dnnType *output) { + TIMER_START + + //Resize with padding and convert to float + cv::Mat image = GetSquareImage(imageORIG, netRT->input_dim.w); + cv::Mat imageF; + image.convertTo(imageF, CV_32FC3, 1/255.0); + + //split channels + cv::Mat bgr[3]; //destination array + cv::split(imageF,bgr);//split source + + //write channels + int idx = 0; + memcpy((void*)&input[idx], (void*)bgr[2].data, imageF.rows*imageF.cols*sizeof(dnnType)); + idx = imageF.rows*imageF.cols; + memcpy((void*)&input[idx], (void*)bgr[1].data, imageF.rows*imageF.cols*sizeof(dnnType)); + idx *= 2; + memcpy((void*)&input[idx], (void*)bgr[0].data, imageF.rows*imageF.cols*sizeof(dnnType)); + + //DO INFERENCE + checkCuda( cudaMemcpyAsync(netRT->buffersRT[netRT->buf_input_idx], input, + netRT->input_dim.tot()*sizeof(float), + cudaMemcpyHostToDevice, netRT->stream)); + netRT->enqueue(); + checkCuda( cudaMemcpyAsync(output, netRT->buffersRT[netRT->buf_output_idx], + netRT->output_dim.tot()*sizeof(float), + cudaMemcpyDeviceToHost, netRT->stream)); + cudaStreamSynchronize(netRT->stream); + + + + rI->interpretData(output, imageORIG.cols, imageORIG.rows); + + TIMER_STOP + return t_ns; +} + + + +int print_usage() { + std::cout<<"usage: ./live net.rt camera_idx\n"; + return 1; +} + + + + +int main(int argc, char *argv[]) { + + //params + char *tensor_path = NULL; + int device = 0; + float thresh = 0.3f; + bool show = false; + + //parse params + int c; + while ((c = getopt (argc, argv, "t:si:")) != -1) { + switch(c) { + case 't': thresh = atof(optarg); break; + case 's': show = true; break; + case '?': + return print_usage(); + default: return print_usage(); + } + } + + if(argc - optind == 2) { + tensor_path = argv[optind]; + device = atoi(argv[optind+1]); + } else { + std::cout<<"not enough arguments.\n"; + return print_usage(); + } + //end parsing + + std::cout<<"open video stream on device: "<> img; + + if(!img.data) + FatalError("Could not open image"); + std::cout<<"Image size: ("<