#include "Yolo3Detection.h" namespace tk { namespace dnn { bool Yolo3Detection::init(const std::string& tensor_path, const int n_classes) { //convert network to tensorRT std::cout<<(tensor_path).c_str()<<"\n"; netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); if(netRT->pluginFactory->n_yolos < 2 ) { FatalError("this is not yolo3"); } for(int i=0; ipluginFactory->n_yolos; i++) { YoloRT *yRT = netRT->pluginFactory->yolos[i]; classes = yRT->classes; num = yRT->num; nMasks = yRT->n_masks; // make a yolo layer for interpret predictions yolo[i] = new tk::dnn::Yolo(nullptr, classes, nMasks, ""); // yolo without input and bias yolo[i]->mask_h = new dnnType[nMasks]; yolo[i]->bias_h = new dnnType[num*nMasks*2]; memcpy(yolo[i]->mask_h, yRT->mask, sizeof(dnnType)*nMasks); memcpy(yolo[i]->bias_h, yRT->bias, sizeof(dnnType)*num*nMasks*2); yolo[i]->input_dim = yolo[i]->output_dim = tk::dnn::dataDim_t(1, yRT->c, yRT->h, yRT->w); yolo[i]->classesNames = yRT->classesNames; } dets = tk::dnn::Yolo::allocateDetections(tk::dnn::Yolo::MAX_DETECTIONS, classes); #ifndef OPENCV_CUDACONTRIB checkCuda(cudaMallocHost(&input, sizeof(dnnType)*netRT->input_dim.tot())); #endif checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot())); // class colors precompute for(int c=0; cclassesNames; return true; } void Yolo3Detection::preprocess(cv::Mat &frame){ #ifdef OPENCV_CUDACONTRIB cv::cuda::GpuMat orig_img, img_resized; orig_img = cv::cuda::GpuMat(frame); cv::cuda::resize(orig_img, img_resized, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); img_resized.convertTo(imagePreproc, CV_32FC3, 1/255.0); //split channels cv::cuda::split(imagePreproc,bgr);//split source //write channels for(int i=0; iinput_dim.c; i++) { int size = imagePreproc.rows * imagePreproc.cols; int ch = netRT->input_dim.c-1 -i; bgr[ch].download(bgr_h); //TODO: don't copy back on CPU checkCuda( cudaMemcpy(input_d + i*size, (float*)bgr_h.data, size*sizeof(dnnType), cudaMemcpyHostToDevice)); } #else cv::resize(frame, frame, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); frame.convertTo(imagePreproc, CV_32FC3, 1/255.0); //split channels cv::split(imagePreproc,bgr);//split source //write channels for(int i=0; iinput_dim.c; i++) { int idx = i*imagePreproc.rows*imagePreproc.cols; int ch = netRT->input_dim.c-1 -i; memcpy((void*)&input[idx], (void*)bgr[ch].data, imagePreproc.rows*imagePreproc.cols*sizeof(dnnType)); } checkCuda(cudaMemcpyAsync(input_d, input, netRT->input_dim.tot()*sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream)); #endif } void Yolo3Detection::postprocess(){ //get yolo outputs dnnType *rt_out[netRT->pluginFactory->n_yolos]; for(int i=0; ipluginFactory->n_yolos; i++) { rt_out[i] = (dnnType*)netRT->buffersRT[i+1]; } float x_ratio = float(originalSize.width) / float(netRT->input_dim.w); float y_ratio = float(originalSize.height) / float(netRT->input_dim.h); // compute dets nDets = 0; for(int i=0; ipluginFactory->n_yolos; i++) { yolo[i]->dstData = rt_out[i]; yolo[i]->computeDetections(dets, nDets, netRT->input_dim.w, netRT->input_dim.h, confThreshold); } tk::dnn::Yolo::mergeDetections(dets, nDets, classes); // fill detected detected.clear(); for(int j=0; j= confThreshold) { obj_class = c; prob = dets[j].prob[c]; } } if(obj_class >= 0) { // convert to image coords x0 = x_ratio*x0; x1 = x_ratio*x1; y0 = y_ratio*y0; y1 = y_ratio*y1; tk::dnn::box res; res.cl = obj_class; res.prob = prob; res.x = x0; res.y = y0; res.w = x1 - x0; res.h = y1 - y0; detected.push_back(res); } } } tk::dnn::Yolo* Yolo3Detection::getYoloLayer(int n) { if(n<3) return yolo[n]; else return nullptr; } }}