diff --git a/include/tkDNN/Yolo3Detection.h b/include/tkDNN/Yolo3Detection.h index 6d38514..317dc01 100644 --- a/include/tkDNN/Yolo3Detection.h +++ b/include/tkDNN/Yolo3Detection.h @@ -13,6 +13,7 @@ private: int num = 0; int nMasks = 0; int nDets = 0; + bool letterbox = false; tk::dnn::Yolo::detection *dets = nullptr; tk::dnn::Yolo* yolo[3]; @@ -21,7 +22,7 @@ private: cv::Mat bgr_h; public: - Yolo3Detection() {}; + Yolo3Detection(const bool letter_box=false) :letterbox(letter_box){} ~Yolo3Detection() {}; bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1); diff --git a/src/Yolo3Detection.cpp b/src/Yolo3Detection.cpp index 606c6d1..07276de 100644 --- a/src/Yolo3Detection.cpp +++ b/src/Yolo3Detection.cpp @@ -52,28 +52,80 @@ bool Yolo3Detection::init(const std::string& tensor_path, const int n_classes, c return true; } -void Yolo3Detection::preprocess(cv::Mat &frame, const int bi){ -#ifdef OPENCV_CUDACONTRIB - cv::cuda::GpuMat orig_img, img_resized; - orig_img = cv::cuda::GpuMat(frame); - cv::cuda::resize(orig_img, img_resized, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); - - img_resized.convertTo(imagePreproc, CV_32FC3, 1/255.0); - - //split channels - cv::cuda::split(imagePreproc,bgr);//split source - - //write channels - for(int i=0; iinput_dim.c; i++) { - int size = imagePreproc.rows * imagePreproc.cols; - int ch = netRT->input_dim.c-1 -i; - bgr[ch].download(bgr_h); //TODO: don't copy back on CPU - checkCuda( cudaMemcpy(input_d + i*size + netRT->input_dim.tot()*bi, (float*)bgr_h.data, size*sizeof(dnnType), cudaMemcpyHostToDevice)); +cv::Mat resize_image(cv::Mat im, int w, int h) +{ + cv::Mat resized = cv::Mat(cv::Size(w,h), CV_32FC3, cv::Scalar(0) ); + cv::Mat part = cv::Mat(cv::Size(w,im.rows), CV_32FC3, cv::Scalar(0) ); + int r, c, k; + float w_scale = (float)(im.cols - 1) / (w - 1); + float h_scale = (float)(im.rows - 1) / (h - 1); + + for(k = 0; k < im.channels(); ++k){ + for(r = 0; r < im.rows; ++r){ + for(c = 0; c < w; ++c){ + float val = 0; + if(c == w-1 || im.cols == 1){ + val = im.at(r, im.cols-1)[k]; + } else { + float sx = c*w_scale; + int ix = (int) sx; + float dx = sx - ix; + val = (1 - dx) * im.at(r, ix)[k] + dx * im.at(r,ix+1)[k]; + } + part.at(r,c)[k] = val; + } + } } -#else - cv::resize(frame, frame, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); + for(k = 0; k < im.channels(); ++k){ + for(r = 0; r < h; ++r){ + float sy = r*h_scale; + int iy = (int) sy; + float dy = sy - iy; + for(c = 0; c < w; ++c){ + float val = (1-dy) * part.at(iy, c)[k]; + resized.at(r, c)[k] = val; + } + if(r == h-1 || im.rows == 1) continue; + for(c = 0; c < w; ++c){ + float val = dy * part.at(iy+1, c)[k]; + resized.at(r,c)[k] += val; + } + } + } + + return resized; +} + +void Yolo3Detection::preprocess(cv::Mat &frame, const int bi){ frame.convertTo(imagePreproc, CV_32FC3, 1/255.0); + if(letterbox){ + int im_w = frame.cols; + int im_h = frame.rows; + int net_w = netRT->input_dim.w; + int net_h = netRT->input_dim.h; + if(net_w == net_h && letterbox){ + float ratio = ( im_w > im_h ) ? float(im_w)/float(net_w) : float(im_h)/float(net_h); + + int new_h = im_h/ratio; + int new_w = im_w/ratio; + + imagePreproc = resize_image(imagePreproc, new_w, new_h); + + cv::Mat borders; + int top = (net_h - new_h)/2; + int bottom = (net_h - new_h) - top; + int left = (net_w - new_w)/2; + int right = (net_w - new_w) - left; + + cv::copyMakeBorder(imagePreproc,imagePreproc, top, bottom, left, right, cv::BORDER_CONSTANT, cv::Scalar(0.5,0.5,0.5)); + } + else + FatalError("letterbox not spported with h!=w"); + } + else + imagePreproc = resize_image(imagePreproc, netRT->input_dim.w, netRT->input_dim.h); + //split channels cv::split(imagePreproc,bgr);//split source @@ -84,7 +136,6 @@ void Yolo3Detection::preprocess(cv::Mat &frame, const int bi){ memcpy((void*)&input[idx + netRT->input_dim.tot()*bi], (void*)bgr[ch].data, imagePreproc.rows*imagePreproc.cols*sizeof(dnnType)); } checkCuda(cudaMemcpyAsync(input_d + netRT->input_dim.tot()*bi, input + netRT->input_dim.tot()*bi, netRT->input_dim.tot()*sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream)); -#endif } void Yolo3Detection::postprocess(const int bi, const bool mAP){ @@ -105,14 +156,45 @@ void Yolo3Detection::postprocess(const int bi, const bool mAP){ } tk::dnn::Yolo::mergeDetections(dets, nDets, classes); + int im_w = originalSize[bi].width; + int im_h = originalSize[bi].height; + int net_w = netRT->input_dim.w; + int net_h = netRT->input_dim.h; + int new_h, new_w; + + int top = 0, left = 0; + + if(letterbox){ + float ratio = ( im_w > im_h ) ? float(im_w)/float(net_w) : float(im_h)/float(net_h); + x_ratio = ratio; + y_ratio = ratio; + std::cout< 0) { - *out_file << "{\"image_id\":" << image_id << + *out_file << std::fixed << std::setprecision(6) << + "{\"image_id\":" << image_id << ", \"category_id\":" << coco_ids[j] << ", \"bbox\":[" << bx << ", " << by << ", " << bw << ", " << bh << "], \"score\":" << bbox[i].probs[j] << "},\n"; } } else - *out_file << "{\"image_id\":" << image_id << + *out_file << std::fixed << std::setprecision(6) << + "{\"image_id\":" << image_id << ", \"category_id\":" << coco_ids[bbox[i].cl] << ", \"bbox\":[" << bx << ", " << by << ", " << bw << ", " << bh << "], \"score\":" << bbox[i].prob << "},\n";