#include #include "tkdnn.h" #include /* srand, rand */ #include #include #include #include #define VOC #ifdef VOC const char *reg_bias = "../tests/yolo_voc/layers/g31.bin"; #define CLASS 20 #else const char *reg_bias = "../tests/yolo/layers/g31.bin"; #define CLASS 80 #endif int prob_sort(const void *pa, const void *pb) { tkDNN::box a = *(tkDNN::box *)pa; tkDNN::box b = *(tkDNN::box *)pb; float diff = a.prob - b.prob; if(diff < 0) return 1; else if(diff > 0) return -1; return 0; } cv::Mat GetSquareImage(const cv::Mat& img, int target_width) { int width = img.cols, height = img.rows; cv::Mat square = cv::Mat::zeros( target_width, target_width, img.type() ); int max_dim = ( width >= height ) ? width : height; float scale = ( ( float ) target_width ) / max_dim; cv::Rect roi; if ( width >= height ) { roi.width = target_width; roi.x = 0; roi.height = height * scale; roi.y = ( target_width - roi.height ) / 2; } else { roi.y = 0; roi.height = target_width; roi.width = width * scale; roi.x = ( target_width - roi.width ) / 2; } cv::resize( img, square( roi ), roi.size() ); return square; } //return inference time double compute_image( cv::Mat imageORIG, tkDNN::NetworkRT *netRT, tkDNN::RegionInterpret *rI, dnnType *input, dnnType *output) { TIMER_START //Resize with padding and convert to float cv::Mat image = GetSquareImage(imageORIG, netRT->input_dim.w); cv::Mat imageF; image.convertTo(imageF, CV_32FC3, 1/255.0); //split channels cv::Mat bgr[3]; //destination array cv::split(imageF,bgr);//split source //write channels int idx = 0; memcpy((void*)&input[idx], (void*)bgr[2].data, imageF.rows*imageF.cols*sizeof(dnnType)); idx = imageF.rows*imageF.cols; memcpy((void*)&input[idx], (void*)bgr[1].data, imageF.rows*imageF.cols*sizeof(dnnType)); idx *= 2; memcpy((void*)&input[idx], (void*)bgr[0].data, imageF.rows*imageF.cols*sizeof(dnnType)); //DO INFERENCE checkCuda( cudaMemcpyAsync(netRT->buffersRT[netRT->buf_input_idx], input, netRT->input_dim.tot()*sizeof(float), cudaMemcpyHostToDevice, netRT->stream)); netRT->enqueue(); checkCuda( cudaMemcpyAsync(output, netRT->buffersRT[netRT->buf_output_idx], netRT->output_dim.tot()*sizeof(float), cudaMemcpyDeviceToHost, netRT->stream)); cudaStreamSynchronize(netRT->stream); rI->interpretData(output, imageORIG.cols, imageORIG.rows); TIMER_STOP return t_ns; } int print_usage() { std::cout<<"usage: ./live net.rt camera_idx\n"; return 1; } int main(int argc, char *argv[]) { //params char *tensor_path = NULL; int device = 0; float thresh = 0.3f; bool show = false; //parse params int c; while ((c = getopt (argc, argv, "t:si:")) != -1) { switch(c) { case 't': thresh = atof(optarg); break; case 's': show = true; break; case '?': return print_usage(); default: return print_usage(); } } if(argc - optind == 2) { tensor_path = argv[optind]; device = atoi(argv[optind+1]); } else { std::cout<<"not enough arguments.\n"; return print_usage(); } //end parsing //std::cout<<"open video stream on device: "<> img; if(!img.data) FatalError("Could not open image"); std::cout<<"Image size: ("<