Add OPENCV_CUDA define, to allow having preprocess both in CPU and GPU (Mobilenet and Centernet)
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
This commit is contained in:
@@ -35,9 +35,7 @@ class CenternetDetection {
|
|||||||
int ndets = 0;
|
int ndets = 0;
|
||||||
// tk::dnn::Yolo::detection *dets = nullptr;
|
// tk::dnn::Yolo::detection *dets = nullptr;
|
||||||
|
|
||||||
cv::Mat imageF;
|
cv::Mat imageOrig;
|
||||||
cv::cuda::GpuMat imageF1_d, imageF2_d;
|
|
||||||
cv::cuda::GpuMat bgr[3];
|
|
||||||
// std::vector< cv::cuda::GpuMat > bgr;
|
// std::vector< cv::cuda::GpuMat > bgr;
|
||||||
|
|
||||||
// variable to test cnet on dog pictures
|
// variable to test cnet on dog pictures
|
||||||
@@ -71,8 +69,16 @@ class CenternetDetection {
|
|||||||
|
|
||||||
float *target_coords;
|
float *target_coords;
|
||||||
|
|
||||||
float *mean_d;
|
|
||||||
float *stddev_d;
|
|
||||||
|
#ifdef OPENCV_CUDA
|
||||||
|
float *mean_d;
|
||||||
|
float *stddev_d;
|
||||||
|
#else
|
||||||
|
cv::Vec<float, 3> mean;
|
||||||
|
cv::Vec<float, 3> stddev;
|
||||||
|
dnnType *input;
|
||||||
|
#endif
|
||||||
|
|
||||||
float *d_ptrs;
|
float *d_ptrs;
|
||||||
|
|
||||||
@@ -88,6 +94,8 @@ class CenternetDetection {
|
|||||||
// pointer used in the kernels
|
// pointer used in the kernels
|
||||||
float *src_out;
|
float *src_out;
|
||||||
int *ids_out;
|
int *ids_out;
|
||||||
|
|
||||||
|
void preprocess();
|
||||||
public:
|
public:
|
||||||
dnnType *rt_out[4];
|
dnnType *rt_out[4];
|
||||||
|
|
||||||
@@ -97,7 +105,7 @@ class CenternetDetection {
|
|||||||
int classes = 80;
|
int classes = 80;
|
||||||
int num = 0;
|
int num = 0;
|
||||||
int n_masks = 0;
|
int n_masks = 0;
|
||||||
float thresh = 0.0;
|
float thresh = 0.3;
|
||||||
cv::Scalar colors[256];
|
cv::Scalar colors[256];
|
||||||
|
|
||||||
// this is filled with results
|
// this is filled with results
|
||||||
|
|||||||
@@ -83,7 +83,7 @@ private:
|
|||||||
void generate_ssd_priors(const SSDSpec *specs, const int n_specs, bool clamp = true);
|
void generate_ssd_priors(const SSDSpec *specs, const int n_specs, bool clamp = true);
|
||||||
void convert_locatios_to_boxes_and_center();
|
void convert_locatios_to_boxes_and_center();
|
||||||
float iou(const tk::dnn::box &a, const tk::dnn::box &b);
|
float iou(const tk::dnn::box &a, const tk::dnn::box &b);
|
||||||
void preprocess(const bool gpu = true);
|
void preprocess();
|
||||||
std::vector<tk::dnn::box> postprocess(const int width, const int height);
|
std::vector<tk::dnn::box> postprocess(const int width, const int height);
|
||||||
float get_color2(int c, int x, int max);
|
float get_color2(int c, int x, int max);
|
||||||
|
|
||||||
|
|||||||
@@ -14,6 +14,8 @@
|
|||||||
|
|
||||||
#define dnnType float
|
#define dnnType float
|
||||||
|
|
||||||
|
#define OPENCV_CUDA
|
||||||
|
|
||||||
// Colored output
|
// Colored output
|
||||||
#define COL_END "\033[0m"
|
#define COL_END "\033[0m"
|
||||||
|
|
||||||
|
|||||||
+90
-21
@@ -103,6 +103,8 @@ bool CenternetDetection::init(std::string tensor_path) {
|
|||||||
|
|
||||||
checkCuda( cudaMallocHost(&target_coords, 4 * K *sizeof(float)) );
|
checkCuda( cudaMallocHost(&target_coords, 4 * K *sizeof(float)) );
|
||||||
|
|
||||||
|
#ifdef OPENCV_CUDA
|
||||||
|
|
||||||
checkCuda( cudaMalloc(&mean_d, 3 * sizeof(float)) );
|
checkCuda( cudaMalloc(&mean_d, 3 * sizeof(float)) );
|
||||||
checkCuda( cudaMalloc(&stddev_d, 3 * sizeof(float)) );
|
checkCuda( cudaMalloc(&stddev_d, 3 * sizeof(float)) );
|
||||||
float mean[3] = {0.408, 0.447, 0.47};
|
float mean[3] = {0.408, 0.447, 0.47};
|
||||||
@@ -110,6 +112,11 @@ bool CenternetDetection::init(std::string tensor_path) {
|
|||||||
|
|
||||||
checkCuda(cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice));
|
checkCuda(cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice));
|
||||||
checkCuda(cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice));
|
checkCuda(cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice));
|
||||||
|
#else
|
||||||
|
checkCuda(cudaMallocHost(&input, sizeof(dnnType)*netRT->input_dim.tot()));
|
||||||
|
mean << 0.408, 0.447, 0.47;
|
||||||
|
stddev << 0.289, 0.274, 0.278;
|
||||||
|
#endif
|
||||||
|
|
||||||
checkCuda( cudaMalloc(&d_ptrs, dim.c * dim.h*dim.w * sizeof(float)) );
|
checkCuda( cudaMalloc(&d_ptrs, dim.c * dim.h*dim.w * sizeof(float)) );
|
||||||
// mean << 0.408, 0.447, 0.47;
|
// mean << 0.408, 0.447, 0.47;
|
||||||
@@ -130,7 +137,7 @@ bool CenternetDetection::init(std::string tensor_path) {
|
|||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
cv::Mat CenternetDetection::draw(cv::Mat &imageORIG) {
|
cv::Mat CenternetDetection::draw(cv::Mat &imageOrig) {
|
||||||
|
|
||||||
tk::dnn::box b;
|
tk::dnn::box b;
|
||||||
int x0, w, x1, y0, h, y1;
|
int x0, w, x1, y0, h, y1;
|
||||||
@@ -158,33 +165,30 @@ cv::Mat CenternetDetection::draw(cv::Mat &imageORIG) {
|
|||||||
y1 = b.y + h;
|
y1 = b.y + h;
|
||||||
objClass = b.cl;
|
objClass = b.cl;
|
||||||
det_class = coco_class_name[objClass];
|
det_class = coco_class_name[objClass];
|
||||||
cv::rectangle(imageORIG, cv::Point(x0, y0), cv::Point(x1, y1), colors[objClass], 2);
|
cv::rectangle(imageOrig, cv::Point(x0, y0), cv::Point(x1, y1), colors[objClass], 2);
|
||||||
// draw label
|
// draw label
|
||||||
cv::Size textSize = getTextSize(det_class, cv::FONT_HERSHEY_SIMPLEX, fontScale, thickness, &baseline);
|
cv::Size textSize = getTextSize(det_class, cv::FONT_HERSHEY_SIMPLEX, fontScale, thickness, &baseline);
|
||||||
cv::rectangle(imageORIG, cv::Point(x0, y0), cv::Point((x0 + textSize.width - 2), (y0 - textSize.height - 2)), colors[b.cl], -1);
|
cv::rectangle(imageOrig, cv::Point(x0, y0), cv::Point((x0 + textSize.width - 2), (y0 - textSize.height - 2)), colors[b.cl], -1);
|
||||||
cv::putText(imageORIG, det_class, cv::Point(x0, (y0 - (baseline / 2))), cv::FONT_HERSHEY_SIMPLEX, fontScale, cv::Scalar(255, 255, 255), thickness);
|
cv::putText(imageOrig, det_class, cv::Point(x0, (y0 - (baseline / 2))), cv::FONT_HERSHEY_SIMPLEX, fontScale, cv::Scalar(255, 255, 255), thickness);
|
||||||
|
|
||||||
}
|
}
|
||||||
return imageORIG;
|
return imageOrig;
|
||||||
// cv::namedWindow("cnet", cv::WINDOW_NORMAL);
|
// cv::namedWindow("cnet", cv::WINDOW_NORMAL);
|
||||||
// cv::imshow("cnet", imageOrig);
|
// cv::imshow("cnet", imageOrig);
|
||||||
// cv::waitKey(10000);
|
// cv::waitKey(10000);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
void CenternetDetection::update(cv::Mat &imageORIG) {
|
void CenternetDetection::preprocess()
|
||||||
|
{
|
||||||
if(!imageORIG.data) {
|
|
||||||
std::cout<<"CENTERNET: NO IMAGE DATA\n";
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
TIMER_START
|
|
||||||
auto start_t = std::chrono::steady_clock::now();
|
auto start_t = std::chrono::steady_clock::now();
|
||||||
auto step_t = std::chrono::steady_clock::now();
|
auto step_t = std::chrono::steady_clock::now();
|
||||||
auto end_t = std::chrono::steady_clock::now();
|
auto end_t = std::chrono::steady_clock::now();
|
||||||
// -----------------------------------pre-process ------------------------------------------
|
|
||||||
|
// -----------------------------------pre-process ------------------------------------------
|
||||||
// it will resize the images to `224 x 224` in GETTING_STARTED.md
|
// it will resize the images to `224 x 224` in GETTING_STARTED.md
|
||||||
cv::Size sz = imageORIG.size();
|
cv::Size sz = imageOrig.size();
|
||||||
std::cout<<"image: "<<sz.width<<", "<<sz.height<<std::endl;
|
std::cout<<"image: "<<sz.width<<", "<<sz.height<<std::endl;
|
||||||
|
|
||||||
float scale = 1.0;
|
float scale = 1.0;
|
||||||
@@ -232,9 +236,12 @@ void CenternetDetection::update(cv::Mat &imageORIG) {
|
|||||||
step_t = end_t;
|
step_t = end_t;
|
||||||
}
|
}
|
||||||
sz_old = sz;
|
sz_old = sz;
|
||||||
|
#ifdef OPENCV_CUDA
|
||||||
cv::cuda::GpuMat im_Orig;
|
cv::cuda::GpuMat im_Orig;
|
||||||
im_Orig = cv::cuda::GpuMat(imageORIG);
|
cv::cuda::GpuMat imageF1_d, imageF2_d;
|
||||||
// cv::cuda::resize (im_Orig, imageF1_d, cv::Size(new_width, new_height));
|
|
||||||
|
im_Orig = cv::cuda::GpuMat(imageOrig);
|
||||||
|
cv::cuda::resize (im_Orig, imageF1_d, cv::Size(new_width, new_height));
|
||||||
checkCuda( cudaDeviceSynchronize() );
|
checkCuda( cudaDeviceSynchronize() );
|
||||||
|
|
||||||
sz = imageF1_d.size();
|
sz = imageF1_d.size();
|
||||||
@@ -243,11 +250,9 @@ void CenternetDetection::update(cv::Mat &imageORIG) {
|
|||||||
std::cout << " TIME resize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
std::cout << " TIME resize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||||
step_t = end_t;
|
step_t = end_t;
|
||||||
|
|
||||||
// cv::cuda::warpAffine(imageF1_d, imageF2_d, trans, cv::Size(inp_width, inp_height), cv::INTER_LINEAR );
|
cv::cuda::warpAffine(imageF1_d, imageF2_d, trans, cv::Size(inp_width, inp_height), cv::INTER_LINEAR );
|
||||||
checkCuda( cudaDeviceSynchronize() );
|
checkCuda( cudaDeviceSynchronize() );
|
||||||
end_t = std::chrono::steady_clock::now();
|
|
||||||
std::cout << " TIME warpAffine: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
|
||||||
step_t = end_t;
|
|
||||||
|
|
||||||
imageF2_d.convertTo(imageF1_d, CV_32FC3, 1/255.0);
|
imageF2_d.convertTo(imageF1_d, CV_32FC3, 1/255.0);
|
||||||
checkCuda( cudaDeviceSynchronize() );
|
checkCuda( cudaDeviceSynchronize() );
|
||||||
@@ -256,7 +261,8 @@ void CenternetDetection::update(cv::Mat &imageORIG) {
|
|||||||
step_t = end_t;
|
step_t = end_t;
|
||||||
|
|
||||||
dim2 = dim;
|
dim2 = dim;
|
||||||
// cv::cuda::split(imageF1_d,bgr);//split source
|
cv::cuda::GpuMat bgr[3];
|
||||||
|
cv::cuda::split(imageF1_d,bgr);//split source
|
||||||
end_t = std::chrono::steady_clock::now();
|
end_t = std::chrono::steady_clock::now();
|
||||||
std::cout << " TIME split: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
std::cout << " TIME split: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||||
step_t = end_t;
|
step_t = end_t;
|
||||||
@@ -275,6 +281,69 @@ void CenternetDetection::update(cv::Mat &imageORIG) {
|
|||||||
end_t = std::chrono::steady_clock::now();
|
end_t = std::chrono::steady_clock::now();
|
||||||
std::cout << " TIME Memcpy to input_d: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
std::cout << " TIME Memcpy to input_d: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||||
step_t = end_t;
|
step_t = end_t;
|
||||||
|
#else
|
||||||
|
|
||||||
|
cv::Mat imageF;
|
||||||
|
resize(imageOrig, imageF, cv::Size(new_width, new_height));
|
||||||
|
sz = imageF.size();
|
||||||
|
std::cout<<"size: "<<sz.height<<" "<<sz.width<<" - "<<std::endl;
|
||||||
|
end_t = std::chrono::steady_clock::now();
|
||||||
|
std::cout << " TIME resize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||||
|
step_t = end_t;
|
||||||
|
|
||||||
|
cv::Mat trans = cv::getAffineTransform( src, dst );
|
||||||
|
cv::warpAffine(imageF, imageF, trans, cv::Size(inp_width, inp_height), cv::INTER_LINEAR );
|
||||||
|
end_t = std::chrono::steady_clock::now();
|
||||||
|
std::cout << " TIME warpAffine: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||||
|
step_t = end_t;
|
||||||
|
|
||||||
|
sz = imageF.size();
|
||||||
|
std::cout<<"size: "<<sz.height<<" "<<sz.width<<" - "<<std::endl;
|
||||||
|
imageF.convertTo(imageF, CV_32FC3, 1/255.0);
|
||||||
|
end_t = std::chrono::steady_clock::now();
|
||||||
|
std::cout << " TIME convertto: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||||
|
step_t = end_t;
|
||||||
|
|
||||||
|
dim2 = dim;
|
||||||
|
|
||||||
|
//split channels
|
||||||
|
cv::Mat bgr[3];
|
||||||
|
cv::split(imageF,bgr);//split source
|
||||||
|
|
||||||
|
for(int i=0; i<3; i++){
|
||||||
|
bgr[i] = bgr[i] - mean[i];
|
||||||
|
bgr[i] = bgr[i] / stddev[i];
|
||||||
|
}
|
||||||
|
|
||||||
|
//write channels
|
||||||
|
for(int i=0; i<dim2.c; i++) {
|
||||||
|
int idx = i*imageF.rows*imageF.cols;
|
||||||
|
int ch = dim2.c-3 +i;
|
||||||
|
// std::cout<<"i: "<<i<<", idx: "<<idx<<", ch: "<<ch<<std::endl;
|
||||||
|
memcpy((void*)&input[idx], (void*)bgr[ch].data, imageF.rows*imageF.cols*sizeof(dnnType));
|
||||||
|
}
|
||||||
|
|
||||||
|
checkCuda(cudaMemcpyAsync(input_d, input, dim2.tot()*sizeof(dnnType), cudaMemcpyHostToDevice));
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
#endif
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
void CenternetDetection::update(cv::Mat &image_orig) {
|
||||||
|
|
||||||
|
imageOrig = image_orig;
|
||||||
|
if(!imageOrig.data) {
|
||||||
|
std::cout<<"CENTERNET: NO IMAGE DATA\n";
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
TIMER_START
|
||||||
|
auto start_t = std::chrono::steady_clock::now();
|
||||||
|
auto step_t = std::chrono::steady_clock::now();
|
||||||
|
auto end_t = std::chrono::steady_clock::now();
|
||||||
|
|
||||||
|
preprocess();
|
||||||
|
|
||||||
printCenteredTitle(" TENSORRT inference ", '=', 30); {
|
printCenteredTitle(" TENSORRT inference ", '=', 30); {
|
||||||
dim2.print();
|
dim2.print();
|
||||||
|
|||||||
@@ -280,10 +280,10 @@ cv::Mat MobilenetDetection::draw()
|
|||||||
return origImg;
|
return origImg;
|
||||||
}
|
}
|
||||||
|
|
||||||
void MobilenetDetection::preprocess(const bool gpu)
|
void MobilenetDetection::preprocess()
|
||||||
{
|
{
|
||||||
std::cout<<"preprocess"<<std::endl;
|
std::cout<<"preprocess"<<std::endl;
|
||||||
if(gpu){
|
#ifdef OPENCV_CUDA
|
||||||
//move original image on GPU
|
//move original image on GPU
|
||||||
cv::cuda::GpuMat im_Orig, frame_resize, frame_nomean, frame_scaled;
|
cv::cuda::GpuMat im_Orig, frame_resize, frame_nomean, frame_scaled;
|
||||||
im_Orig = cv::cuda::GpuMat(origImg);
|
im_Orig = cv::cuda::GpuMat(origImg);
|
||||||
@@ -301,8 +301,7 @@ void MobilenetDetection::preprocess(const bool gpu)
|
|||||||
int idx = i * frame_scaled.rows * frame_scaled.cols;
|
int idx = i * frame_scaled.rows * frame_scaled.cols;
|
||||||
checkCuda( cudaMemcpy((void *)&input_d[idx], (void *)bgr[i].data, frame_scaled.rows * frame_scaled.cols* sizeof(float), cudaMemcpyDeviceToDevice) );
|
checkCuda( cudaMemcpy((void *)&input_d[idx], (void *)bgr[i].data, frame_scaled.rows * frame_scaled.cols* sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||||
}
|
}
|
||||||
}
|
#else
|
||||||
else{
|
|
||||||
//resize image, remove mean, divide by std
|
//resize image, remove mean, divide by std
|
||||||
cv::Mat frame_resize, frame_nomean, frame_scaled;
|
cv::Mat frame_resize, frame_nomean, frame_scaled;
|
||||||
resize(origImg, frame_resize, cv::Size(netRT->input_dim.w, netRT->input_dim.h));
|
resize(origImg, frame_resize, cv::Size(netRT->input_dim.w, netRT->input_dim.h));
|
||||||
@@ -317,7 +316,7 @@ void MobilenetDetection::preprocess(const bool gpu)
|
|||||||
memcpy((void *)&input[idx], (void *)bgr[i].data, frame_scaled.rows * frame_scaled.cols * sizeof(dnnType));
|
memcpy((void *)&input[idx], (void *)bgr[i].data, frame_scaled.rows * frame_scaled.cols * sizeof(dnnType));
|
||||||
}
|
}
|
||||||
checkCuda(cudaMemcpyAsync(input_d, input, netRT->input_dim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream));
|
checkCuda(cudaMemcpyAsync(input_d, input, netRT->input_dim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream));
|
||||||
}
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
void MobilenetDetection::update(cv::Mat &img)
|
void MobilenetDetection::update(cv::Mat &img)
|
||||||
|
|||||||
Reference in New Issue
Block a user