From 7f239efdc0820f5707d923a7ce873b0bec975bfb Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Tue, 21 Jan 2020 12:50:18 +0100 Subject: [PATCH] Centernet: fix pooling problem, add centrnet demo Signed-off-by: Micaela Verucchi Signed-off-by: Davide Sapienza --- CMakeLists.txt | 5 +- demo/demo/demo_centernet.cpp | 88 +++ demo/demo/{demo.cpp => demo_yolo3.cpp} | 50 +- include/tkDNN/CenternetDetection.h | 3 + include/tkDNN/pluginsRT/ActivationSigmoidRT.h | 60 ++ include/tkDNN/pluginsRT/DeformableConvRT.h | 1 - src/CenternetDetection.cpp | 251 ++----- src/NetworkRT.cpp | 3 +- src/Pooling.cpp | 4 +- src/kernels/activation_sigmoid.cu | 16 +- src/sorting.cu | 2 +- tests/resnet101_cnet/resnet101_cnet.cpp | 681 +----------------- 12 files changed, 229 insertions(+), 935 deletions(-) create mode 100644 demo/demo/demo_centernet.cpp rename demo/demo/{demo.cpp => demo_yolo3.cpp} (57%) create mode 100644 include/tkDNN/pluginsRT/ActivationSigmoidRT.h diff --git a/CMakeLists.txt b/CMakeLists.txt index 7af66c0..8b847d0 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -105,9 +105,12 @@ target_link_libraries(test_resnet101_cnet tkDNN) add_executable(test_rtinference tests/test_rtinference/rtinference.cpp) target_link_libraries(test_rtinference tkDNN) -add_executable(yolo3_demo demo/demo/demo.cpp) +add_executable(yolo3_demo demo/demo/demo_yolo3.cpp) target_link_libraries(yolo3_demo tkDNN) +add_executable(centernet_demo demo/demo/demo_centernet.cpp) +target_link_libraries(centernet_demo tkDNN) + #------------------------------------------------------------------------------- # Install diff --git a/demo/demo/demo_centernet.cpp b/demo/demo/demo_centernet.cpp new file mode 100644 index 0000000..0f9c86e --- /dev/null +++ b/demo/demo/demo_centernet.cpp @@ -0,0 +1,88 @@ +#include +#include +#include /* srand, rand */ +#include +#include +#include "utils.h" + +#include +#include +#include +#include + +#include "CenternetDetection.h" + +bool gRun; +bool SAVE_RESULT = false; + +void sig_handler(int signo) { + std::cout<<"request gateway stop\n"; + gRun = false; +} + +int main(int argc, char *argv[]) { + + std::cout<<"detection\n"; + signal(SIGINT, sig_handler); + + + char *net = "resnet101_cnet.rt"; + if(argc > 1) + net = argv[1]; + char *input = "../demo/yolo_test.mp4"; + if(argc > 2) + input = argv[2]; + + tk::dnn::CenternetDetection cnet; + cnet.init(net); + + gRun = true; + + cv::VideoCapture cap(input); + if(!cap.isOpened()) + gRun = false; + else + std::cout<<"camera started\n"; + + + cv::VideoWriter resultVideo; + if(SAVE_RESULT) { + int w = cap.get(cv::CAP_PROP_FRAME_WIDTH); + int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT); + resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h)); + } + + cv::Mat frame; + cv::Mat dnn_input; + cv::namedWindow("detection", cv::WINDOW_NORMAL); + + while(gRun) { + cap >> frame; + if(!frame.data) { + break; + } + + // this will be resized to the net format + dnn_input = frame.clone(); + // TODO: async infer + cnet.update(dnn_input); + // draw dets + frame = cnet.draw(dnn_input); + + cv::imshow("detection", frame); + cv::waitKey(1); + if(SAVE_RESULT) + resultVideo << frame; + } + + std::cout<<"detection end\n"; + + + std::cout< #include -// #include "Yolo3Detection.h" -#include "CenternetDetection.h" +#include "Yolo3Detection.h" bool gRun; bool SAVE_RESULT = false; @@ -27,15 +26,14 @@ int main(int argc, char *argv[]) { signal(SIGINT, sig_handler); - char *net = "resnet101_cnet.rt"; + char *net = "yolo3_berkeley.rt"; if(argc > 1) net = argv[1]; char *input = "../demo/yolo_test.mp4"; if(argc > 2) input = argv[2]; - // tk::dnn::Yolo3Detection yolo; - tk::dnn::CenternetDetection yolo; + tk::dnn::Yolo3Detection yolo; yolo.init(net); gRun = true; @@ -69,30 +67,28 @@ int main(int argc, char *argv[]) { // TODO: async infer yolo.update(dnn_input); - frame = yolo.draw(dnn_input); - // // draw dets - // for(int i=0; iclassesNames[b.cl]; - // float prob = b.prob; + // draw dets + for(int i=0; iclassesNames[b.cl]; + float prob = b.prob; - // // std::cout< coco_class_name; + // keep track of inference times (ms) + std::vector stats; + CenternetDetection() {} virtual ~CenternetDetection() {} diff --git a/include/tkDNN/pluginsRT/ActivationSigmoidRT.h b/include/tkDNN/pluginsRT/ActivationSigmoidRT.h new file mode 100644 index 0000000..4435f08 --- /dev/null +++ b/include/tkDNN/pluginsRT/ActivationSigmoidRT.h @@ -0,0 +1,60 @@ +#include +#include "../kernels.h" + +class ActivationSigmoidRT : public IPlugin { + +public: + ActivationSigmoidRT() { + + + } + + ~ActivationSigmoidRT(){ + + } + + int getNbOutputs() const override { + return 1; + } + + Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override { + return inputs[0]; + } + + void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override { + size = 1; + for(int i=0; i(inputs[0]), + reinterpret_cast(outputs[0]), size, stream); + return 0; + } + + + virtual size_t getSerializationSize() override { + return 1*sizeof(int); + } + + virtual void serialize(void* buffer) override { + char *buf = reinterpret_cast(buffer); + tk::dnn::writeBUF(buf, size); + } + + int size; +}; diff --git a/include/tkDNN/pluginsRT/DeformableConvRT.h b/include/tkDNN/pluginsRT/DeformableConvRT.h index b236095..dca1020 100644 --- a/include/tkDNN/pluginsRT/DeformableConvRT.h +++ b/include/tkDNN/pluginsRT/DeformableConvRT.h @@ -90,7 +90,6 @@ public: } virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override { -std::cout<<"LOL\n"; dnnType *srcData = (dnnType*)reinterpret_cast(inputs[0]); dnnType *output_conv = (dnnType*)reinterpret_cast(inputs[1]); diff --git a/src/CenternetDetection.cpp b/src/CenternetDetection.cpp index 59bf827..fbc9832 100644 --- a/src/CenternetDetection.cpp +++ b/src/CenternetDetection.cpp @@ -229,58 +229,50 @@ void CenternetDetection::update(cv::Mat &imageORIG) { src.at(2,1)=src.at(1,1) + (src.at(0,0)-src.at(1,0) ); dst.at(2,0)=dst.at(1,0) + (-dst.at(0,1)+dst.at(1,1) ); dst.at(2,1)=dst.at(1,1) + (dst.at(0,0)-dst.at(1,0) ); - // std::cout<<"src: "<(end_t - step_t).count() << " ms" << std::endl; + std::cout << " TIME gett affine trans: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; + + resize(imageORIG, imageF, cv::Size(new_width, new_height)); sz = imageF.size(); std::cout<<"size: "<(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; - cv::warpAffine(imageF, imageF, trans, cv::Size(inp_width, inp_height), cv::INTER_LINEAR ); + + cv::warpAffine(imageF, imageF, trans, cv::Size(inp_width, inp_height), cv::INTER_LINEAR ); end_t = std::chrono::steady_clock::now(); std::cout << " TIME warpAffine: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; + sz = imageF.size(); std::cout<<"size: "<(end_t - step_t).count() << " ms" << std::endl; - step_t = end_t; - std::cout<<"mean: "<(end_t - step_t).count() << " us" << std::endl; - step_t = end_t; - - //split channels - cv::split(imageF,bgr);//split source end_t = std::chrono::steady_clock::now(); - std::cout << " TIME split: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; + std::cout << " TIME convert: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; + dim2 = dim; + + //split channels + cv::split(imageF,bgr);//split source + for(int i=0; i<3; i++){ bgr[i] = bgr[i] - mean[i]; bgr[i] = bgr[i] / stddev[i]; } - end_t = std::chrono::steady_clock::now(); - std::cout << " TIME mean std: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; - step_t = end_t; - //write channels for(int i=0; iinfer(dim2, input_d); TIMER_STOP dim2.print(); + + stats.push_back(t_ns); } // checkResult(dim2.tot(), input_h, input); - std::cout<<" --- pre-process ---\n"; - end_t = std::chrono::steady_clock::now(); - std::cout << " TIME : " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; - step_t = end_t; + step_t = std::chrono::steady_clock::now(); + // ------------------------------------ process -------------------------------------------- rt_out[0] = (dnnType *)netRT->buffersRT[1]; rt_out[1] = (dnnType *)netRT->buffersRT[2]; rt_out[2] = (dnnType *)netRT->buffersRT[3]; rt_out[3] = (dnnType *)netRT->buffersRT[4]; - activationSIGMOIDForward(rt_out[0], rt_out[0], dim_hm.tot()); checkCuda( cudaDeviceSynchronize() ); - end_t = std::chrono::steady_clock::now(); - std::cout << " TIME sigmoid : " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; - step_t = end_t; subtractWithThreshold(rt_out[0], rt_out[0] + dim_hm.tot(), rt_out[1], rt_out[0]); - float *prova; - checkCuda( cudaMallocHost(&prova, K*sizeof(float)) ); - checkCuda( cudaMemcpy(prova, rt_out[0], K*sizeof(float), cudaMemcpyDeviceToHost) ); - std::cout<<"heat:\n"; - for(int i=0; i toll || hm_h[i]-hmax_h[i] < -toll){ - // hm_h[i] = 0.0f; - // } - // } - // checkCuda( cudaFreeHost(hmax_h) ); - std::cout<<" --- hmax ---\n"; end_t = std::chrono::steady_clock::now(); - std::cout << " TIME : " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; + std::cout << " TIME threshold: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; // ----------- nms end // ----------- topk - - // thrust::device_vector ids_d; - // int ids[dim_hm.h * dim_hm.w]; - // for(int i=0; i ids2( dim_hm.h * dim_hm.w ); - // for(int i=0; i dim_hm.h * dim_hm.w){ printf ("Error topk (K is too large)\n"); return; } - checkCuda( cudaMemcpy(ids_d, ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyHostToDevice) ); - // checkCuda( cudaMemcpy(ids_2d, ids_2, dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyHostToDevice) ); - - - // sortAndTopKonDevice(rt_out[0], ids_2d, topk_scores, topk_inds_ , topk_ys_ , topk_xs_ ,dim_hm.h * dim_hm.w, K, dim_hm.c); - // checkCuda( cudaDeviceSynchronize() ); - // for(int i=0; ioutput_dim.h * hm->output_dim.w elements for each channel and sort it. Then find the first 100 elements - // // memcpy(ids2, ids, dim_hm.h * dim_hm.w); - // sort(rt_out[0]+ i * dim_hm.h * dim_hm.w, - // rt_out[0]+ i * dim_hm.h * dim_hm.w + dim_hm.h * dim_hm.w, - // ids_d); - // // end_t = std::chrono::steady_clock::now(); - // // std::cout << " TIME sort channel "<(end_t - step_t).count() << " ms" << std::endl; - // // step_t = end_t; - // topk(rt_out[0]+ i * dim_hm.h * dim_hm.w, ids_d, K, topk_scores + i*K, - // topk_inds_ + i*K, topk_ys_ + i*K, topk_xs_ + i*K); - // // checkCuda( cudaMemcpy(ids2, ids2_d, dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyDeviceToHost) ); - - // // for (int j=0; j(end_t - step_t).count() << " ms" << std::endl; - // // step_t = end_t; - - // } - // checkCuda( cudaFree(ids_d )); - std::cout<<" --- a 100 ---\n"; - end_t = std::chrono::steady_clock::now(); - std::cout << " TIME sort topk on 80 channel: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; - step_t = end_t; - // final - - // sort(topk_scores, - // topk_scores + dim_hm.c * K, - // topk_inds_); sort(rt_out[0], rt_out[0]+dim_hm.tot(), ids_d); checkCuda( cudaDeviceSynchronize() ); - int *topk_inds; - checkCuda( cudaMallocHost(&topk_inds, K*sizeof(int)) ); - // checkCuda( cudaMemcpy(topk_inds, ids_d, K*sizeof(int), cudaMemcpyDeviceToHost) ); - // for(int i=0; i(end_t - step_t).count() << " ms" << std::endl; + std::cout << " TIME sort: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; - // topk(topk_scores, topk_inds_, K, scores_d, - // topk_inds_d, topk_ys_d, topk_xs_d); topk(rt_out[0], ids_d, K, scores_d, topk_inds_d, topk_ys_d, topk_xs_d); - checkCuda( cudaDeviceSynchronize() ); + checkCuda( cudaDeviceSynchronize() ); + end_t = std::chrono::steady_clock::now(); - std::cout << " TIME topk channel: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; + std::cout << " TIME topk: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; - - - checkCuda( cudaMemcpy(topk_inds, topk_inds_d, K*sizeof(int), cudaMemcpyDeviceToHost) ); - for(int i=0; i(end_t - step_t).count() << " ms" << std::endl; + step_t = end_t; + checkCuda( cudaMemcpy(topk_xs_d, (float *)inttopk_xs_d, K*sizeof(float), cudaMemcpyDeviceToDevice) ); checkCuda( cudaMemcpy(topk_ys_d, (float *)inttopk_ys_d, K*sizeof(float), cudaMemcpyDeviceToDevice) ); checkCuda( cudaMemcpy(clses, clses_d, K*sizeof(int), cudaMemcpyDeviceToHost) ); - std::cout<<"\ntopk_ids: \n"; - checkCuda( cudaMemcpy(topk_inds, topk_inds_d, K*sizeof(int), cudaMemcpyDeviceToHost) ); - for(int i=0; i(end_t - step_t).count() << " ms" << std::endl; - step_t = end_t; + // ----------- topk end - // dnnType *reg_aus; - // checkCuda( cudaMallocHost(®_aus, dim_reg.tot()*sizeof(dnnType)) ); - // checkCuda( cudaMemcpy(reg_aus, rt_out[3], dim_reg.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) ); - - // for(int i = 0; i < K; i++){ - // topk_xs[i] = topk_xs[i] + reg_aus[topk_inds[i]]; - // topk_ys[i] = topk_ys[i] + reg_aus[topk_inds[i]+dim_reg.h*dim_reg.w]; - // } topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3]); // checkCuda( cudaDeviceSynchronize() ); + end_t = std::chrono::steady_clock::now(); std::cout << " TIME add offset: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; - // checkCuda( cudaFreeHost(reg_aus) ); - - // dnnType *wh_aus; - // checkCuda( cudaMemcpy(wh_aus, rt_out[2], dim_wh.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) ); bboxes(topk_inds_d, K, dim_wh.h*dim_wh.w, topk_xs_d, topk_ys_d, rt_out[2], bbx0_d, bbx1_d, bby0_d, bby1_d); // checkCuda( cudaDeviceSynchronize() ); + checkCuda( cudaMemcpy(bbx0, bbx0_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); checkCuda( cudaMemcpy(bby0, bby0_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); checkCuda( cudaMemcpy(bbx1, bbx1_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); checkCuda( cudaMemcpy(bby1, bby1_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); - // for(int i = 0; i < K; i++){ - // bboxes[i * 4] = topk_xs[i] - wh_aus[topk_inds[i]] / 2; - // bboxes[i * 4 + 1] = topk_ys[i] - wh_aus[topk_inds[i]+dim_reg.h*dim_reg.w] / 2; - // bboxes[i * 4 + 2] = topk_xs[i] + wh_aus[topk_inds[i]] / 2; - // bboxes[i * 4 + 3] = topk_ys[i] + wh_aus[topk_inds[i]+dim_reg.h*dim_reg.w] / 2; - // } - // for(int i = 0; i < K; i++){ - // std::cout<<"-----\n(x0, y0) = ("<(end_t - step_t).count() << " ms" << std::endl; + std::cout << " TIME bboxes: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; - // servono [bboxes, scores, clses] - // checkCuda( cudaDeviceSynchronize() ); - std::cout<<" --- process ---\n"; - end_t = std::chrono::steady_clock::now(); - std::cout << " TIME : " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; - step_t = end_t; // ---------------------------------- post-process ----------------------------------------- // --------- ctdet_post_process @@ -548,12 +384,13 @@ void CenternetDetection::update(cv::Mat &imageORIG) { cv::Mat trans2(cv::Size(3,2), CV_32F); trans2 = cv::getAffineTransform( dst, src ); + end_t = std::chrono::steady_clock::now(); std::cout << " TIME getAffineTrans 2: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; + cv::Mat new_pt1(cv::Size(1,2), CV_32F); - cv::Mat new_pt2(cv::Size(1,2), CV_32F); - + cv::Mat new_pt2(cv::Size(1,2), CV_32F); for(int i = 0; i(0,0)=static_cast(trans2.at(0,0))*bbx0[i] + @@ -570,23 +407,18 @@ void CenternetDetection::update(cv::Mat &imageORIG) { static_cast(trans2.at(1,1))*bby1[i] + static_cast(trans2.at(1,2))*1.0; - // std::cout<<"\n new: "<(0,0); target_coords[i*4+1] = new_pt1.at(0,1); target_coords[i*4+2] = new_pt2.at(0,0); target_coords[i*4+3] = new_pt2.at(0,1); - // std::cout<(0,0)<<", "<(0,1)<<", "<(0,0)<<", "<(0,1)< thresh){ - std::cout<<"th: "<(end_t - step_t).count() << " ms" << std::endl; + std::cout << " TIME detections: " << std::chrono::duration_cast(end_t - step_t).count() << " ms" << std::endl; step_t = end_t; + std::cout<<"TOTAL: \n"; TIMER_STOP } diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index bec7b17..0f2d621 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -290,8 +290,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Pooling *l) { if(l->pool_mode == tkdnnPoolingMode_t::POOLING_AVERAGE_EXCLUDE_PADDING) ptype = PoolingType::kMAX_AVERAGE_BLEND; - // if(l->input_dim.h % 2 == 1 && l->input_dim.w % 2 == 1) - if(l->input_dim.h == l->output_dim.h && l->input_dim.w == l->output_dim.w) + if(l->paddingH == 0 && l->paddingW == 0 && l->input_dim.h == l->output_dim.h && l->input_dim.w == l->output_dim.w) { IPlugin *plugin = new ResizeLayerRT( l->output_dim.c,l->output_dim.h+1,l->output_dim.w+1 ); IPluginLayer *lRT = networkRT->addPlugin(&input, 1, *plugin); diff --git a/src/Pooling.cpp b/src/Pooling.cpp index 1ad0664..cc498df 100644 --- a/src/Pooling.cpp +++ b/src/Pooling.cpp @@ -55,8 +55,8 @@ Pooling::Pooling( Network *net, int winH, int winW, int strideH, int strideW, int padW = paddingW == 0? winW -1 : paddingW; if(final){ - h = (h + padH - winH)/strideH +1 +1; - w = (w + padW - winW)/strideW +1 +1; + h = (h + 2*paddingH - winH)/strideH +1 ; + w = (w + 2*paddingW - winW)/strideW +1; } else{ h = (h + padH - winH)/strideH +1; diff --git a/src/kernels/activation_sigmoid.cu b/src/kernels/activation_sigmoid.cu index ba6c997..400a948 100644 --- a/src/kernels/activation_sigmoid.cu +++ b/src/kernels/activation_sigmoid.cu @@ -1,21 +1,13 @@ #include "kernels.h" - -__device__ -__forceinline__ -double sigmoid (double a) -{ - return 1.0 / (1.0 + exp (-a)); -} +#include __global__ void activation_sigmoid(dnnType *input, dnnType *output, int size) { - int stride = gridDim.x * blockDim.x; - int tid = blockDim.x * blockIdx.x + threadIdx.x; - for (int i = tid; i < size; i += stride) { - output[i] = sigmoid (input[i]); - } + int i = blockDim.x * blockIdx.x + threadIdx.x; + if(i < size) + output[i] = 1.0f / (1.0f + exp (-input[i])); } diff --git a/src/sorting.cu b/src/sorting.cu index e007be1..fd62cec 100644 --- a/src/sorting.cu +++ b/src/sorting.cu @@ -45,7 +45,7 @@ struct threshold : public thrust::binary_function { __host__ __device__ float operator()(float x, float y) { - float toll = 1e-6; + double toll = 1e-6; if(fabsf(x-y)>toll) return 0.0f; else diff --git a/tests/resnet101_cnet/resnet101_cnet.cpp b/tests/resnet101_cnet/resnet101_cnet.cpp index 061e4b6..e05f517 100644 --- a/tests/resnet101_cnet/resnet101_cnet.cpp +++ b/tests/resnet101_cnet/resnet101_cnet.cpp @@ -183,53 +183,6 @@ const char *output_bin[]={ "../tests/resnet101_cnet/debug/wh.bin", "../tests/resnet101_cnet/debug/reg.bin"}; - - -std::vector sort_indexes(const std::vector &v) { - - // initialize original index locations - std::vector idx(v.size()); - iota(idx.begin(), idx.end(), 0); - - // sort indexes based on comparing values in v - sort(idx.begin(), idx.end(), - [&v](size_t i1, size_t i2) {return v[i1] > v[i2];}); - - return idx; -} - -float _colors[6][3] = { {1,0,1}, {0,0,1},{0,1,1},{0,1,0},{1,1,0},{1,0,0} }; -float get_color(int c, int x, int max) -{ - float ratio = ((float)x/max)*5; - int i = floor(ratio); - int j = ceil(ratio); - ratio -= i; - float r = (1-ratio) * _colors[i % 6][c % 3] + ratio*_colors[j % 6][c % 3]; - //printf("%f\n", r); - return r; -} - -int computeDetections(dnnType *hm_d, dnnType *wh_d, dnnType *reg_d, int hm_dim, int wh_dim, int reg_dim, bool cat_spec_wh, int k){ - // _nms - int kernel = 3; - int pad = (kernel - 1)/2; - std::cout<<"computeDetections\n"; - // dnnType *hmax; - // tk::dnn::Pooling maxpool(&hmax, 3, 3, 2, 2, 1, 1, tk::dnn::POOLING_MAX) - // = (dnnType *) - - // net.functional.max_pool2d( - // heat, (kernel, kernel), stride=1, padding=pad) - // keep = (hmax == heat).float() - // return heat * keep -} - -int process(dnnType *hm_d, dnnType *wh_d, dnnType *reg_d, int hm_dim, int wh_dim, int reg_dim){ - std::cout<<"process\n"; - // computeDetections(hm_d, wh_d, reg_d, hm_dim, wh_dim, reg_dim, false, 100); -} - int main() { @@ -450,637 +403,5 @@ int main() checkResult(odim, rt_out, out); std::cout << "CUDNN vs TRT "; checkResult(odim, cudnn_out, rt_out); - } - - TIMER_START - - // -------- transofrm compose - cv::Mat imageOrig = cv::imread("/media/davide/DATA/shared_home/Projects/Professionale/repos/photo_2020-01-14_09-56-07.jpg"); - cv::Mat imageF; - imageOrig.convertTo(imageF, CV_32FC3, 1/255.0); - cv::Mat image; - cv::Size sz = imageF.size(); - std::cout<<"image: "<output_dim.tot()*sizeof(dnnType)) ); - - dnnType *rt_out[4]; - rt_out[0] = (dnnType *)netRT.buffersRT[1]; - rt_out[1] = (dnnType *)netRT.buffersRT[2]; - rt_out[2] = (dnnType *)netRT.buffersRT[3]; - rt_out[3] = (dnnType *)netRT.buffersRT[4]; - - // checkCuda( cudaMemcpy(hm_h, rt_out[0], hm->output_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) ); - std::cout<<"hm\n"; - hm->output_dim.print(); - // for(int i=0; ioutput_dim.tot(); i++ ){ - // std::cout<output_dim.tot()); - checkCuda( cudaDeviceSynchronize() ); - - - checkCuda( cudaMemcpy(hm_h, rt_out[0], hm->output_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) ); - std::cout<<"hm\n"; - hm->output_dim.print(); - // for(int i=0; ioutput_dim.tot(); i++ ){ - // std::cout<infer(hmax->input_dim.tot(), rt_out[0]); - // keep = (hmax == heat).float() - // return heat * keep - - dnnType *hmax_h; - checkCuda( cudaMallocHost(&hmax_h, hmax->output_dim.tot()*sizeof(dnnType)) ); - checkCuda( cudaMemcpy(hmax_h, rt_out[1], hmax->output_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) ); - - std::cout<<"hmax\n"; - hmax->output_dim.print(); - // for(int i=0; ioutput_dim.tot(); i++ ){ - // std::cout<output_dim.print(); - std::cout<<"hmax:\n"; - hmax->output_dim.print(); - // return 0; - float toll = 0.000001; - for(int i=0; i < hm->output_dim.tot(); i++){ - if(hm_h[i]-hmax_h[i] > toll || hm_h[i]-hmax_h[i] < -toll){ - hm_h[i] = 0.0f; - } - } - // std::cout<<"\n"; - // for(int i=0; ioutput_dim.tot(); i++ ){ - // std::cout<dstData, hm_h, hm->output_dim.tot()*sizeof(dnnType), cudaMemcpyHostToDevice) ); - checkCuda( cudaFreeHost(hmax_h) ); - // ----------- nms end - // ----------- topk - int K = 100; - int width = 56; // TODO - float *topk_scores; - int *topk_inds_; - float *topk_ys_; - float *topk_xs_; - std::cout<<"mah: "<output_dim.c * K<output_dim.c * K *sizeof(float)) ); - checkCuda( cudaMallocHost(&topk_inds_, hm->output_dim.c * K *sizeof(int)) ); - checkCuda( cudaMallocHost(&topk_ys_, hm->output_dim.c * K *sizeof(float)) ); - checkCuda( cudaMallocHost(&topk_xs_, hm->output_dim.c * K *sizeof(float)) ); - std::cout<<"1\n"; - dnnType *hm_aus; - checkCuda( cudaMallocHost(&hm_aus, hm->output_dim.h * hm->output_dim.w *sizeof(dnnType)) ); - std::cout<<"2\n"; - int count; - std::vector v = {2.0, 3.0, 9.0}; - for (auto i: sort_indexes(v)) { - std::cout << i<< "--" <output_dim.c; i++){ - count = 0; - // get the hm->output_dim.h * hm->output_dim.w elements for each channel and sort it. Then find the first 100 elements - checkCuda( cudaMemcpy(hm_aus, hm_h + i * hm->output_dim.h * hm->output_dim.w, - hm->output_dim.h * hm->output_dim.w * sizeof(dnnType), cudaMemcpyHostToHost) ); - // std::cout<<"top scores: "<output_dim.h * hm->output_dim.w<<"\n"; - // for(int k=0; koutput_dim.h * hm->output_dim.w; k++) - // std::cout< my_vector {arr, arr + arr_length} - std::vector my_vector{hm_aus, hm_aus + hm->output_dim.h * hm->output_dim.w}; - for (auto j: sort_indexes(my_vector)) { - // std::cout <<"j: "< "<< hm_aus[j] << std::endl; - topk_scores[i*K + count] = hm_aus[j]; - topk_inds_[i*K +count] = j; - topk_ys_[i*K +count] = (int)(j / width); - topk_xs_[i*K +count] = (int)(j % width); - if(++count == K) - break; - } - } - std::cout<<"topk_xs_[0]: "<output_dim.c * K; i++) - std::cout< my_vector{topk_scores, topk_scores + hm->output_dim.c * K }; - for (auto j: sort_indexes(my_vector)) { - // std::cout <<"j: "< "<< hm_aus[j] << std::endl; - scores[count] = topk_scores[j]; - clses[count] = (int)(j / K); - topk_inds[count] = topk_inds_[j]; - topk_ys[count] = topk_ys_[j]; - topk_xs[count] = topk_xs_[j]; - if(++count == K) - break; - } - checkCuda( cudaFreeHost(topk_scores) ); - checkCuda( cudaFreeHost(topk_inds_) ); - checkCuda( cudaFreeHost(topk_ys_) ); - checkCuda( cudaFreeHost(topk_xs_) ); - std::cout<<"5\n"; - // ----------- topk end - std::cout<<"topk_xs[0]: "<output_dim.tot()*sizeof(dnnType)) ); - checkCuda( cudaMemcpy(reg_aus, rt_out[3], reg->output_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) ); - std::cout<<"reg:\n"; - reg->output_dim.print(); - // for(int i=0; ioutput_dim.tot(); i++ ){ - // std::cout<output_dim.h*reg->output_dim.w]; - } - std::cout<<"topk_xs[0]: "<output_dim.tot()*sizeof(dnnType)) ); - checkCuda( cudaMallocHost(&bboxes, 4 * K *sizeof(dnnType)) ); - checkCuda( cudaMemcpy(wh_aus, rt_out[2], wh->output_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) ); - std::cout<<"7\n"; - for(int i = 0; i< K; i++) - std::cout<output_dim.h*reg->output_dim.w] / 2; - bboxes[i * 4 + 2] = topk_xs[i] + wh_aus[topk_inds[i]] / 2; - bboxes[i * 4 + 3] = topk_ys[i] + wh_aus[topk_inds[i]+reg->output_dim.h*reg->output_dim.w] / 2; - } - ////////////////// fin qui ok - - checkCuda( cudaFreeHost(wh_aus) ); - checkCuda( cudaFreeHost(topk_inds) ); - checkCuda( cudaFreeHost(topk_ys) ); - checkCuda( cudaFreeHost(topk_xs) ); - - std::cout<<"8\n"; - float *detections; - std::cout<<"bboxes:\n"; - for(int i = 0; i < K+1; i++){ - std::cout<(0,0)=c[0]; - src.at(0,1)=c[1]; - src.at(1,0)=c[0]; - src.at(1,1)=c[1] + s[0] * -0.5; - dst.at(0,0)=width * 0.5; - dst.at(0,1)=width * 0.5; - dst.at(1,0)=width * 0.5; - dst.at(1,1)=width * 0.5 + width * -0.5; - - src.at(2,0)=src.at(1,0) + (-src.at(0,1)+src.at(1,1) ); - src.at(2,1)=src.at(1,1) + (src.at(0,0)-src.at(1,0) ); - dst.at(2,0)=dst.at(1,0) + (-dst.at(0,1)+dst.at(1,1) ); - dst.at(2,1)=dst.at(1,1) + (dst.at(0,0)-dst.at(1,0) ); - std::cout<<"src: "<(0,0)<<" - "<(0,1)<<" - "<(0,2)<<"\n"<(1,0)<<" - "<(1,1)<<" - "<(1,2)<(0,0)=detections[i*4]; - // new_pt1.at(0,1)=detections[i*4+1]; - // new_pt1.at(0,2)=1.0; - // new_pt1 << detections[i], detections[i+K], 1.0; - // std::cout<<"----\ni: "<(0,0)=static_cast(trans2.at(0,0))*detections[i*4] + - static_cast(trans2.at(0,1))*detections[i*4+1] + - static_cast(trans2.at(0,2))*1.0; - new_pt1.at(0,1)=static_cast(trans2.at(1,0))*detections[i*4] + - static_cast(trans2.at(1,1))*detections[i*4+1] + - static_cast(trans2.at(1,2))*1.0; - - new_pt2.at(0,0)=static_cast(trans2.at(0,0))*detections[i*4+2] + - static_cast(trans2.at(0,1))*detections[i*4+3] + - static_cast(trans2.at(0,2))*1.0; - new_pt2.at(0,1)=static_cast(trans2.at(1,0))*detections[i*4+2] + - static_cast(trans2.at(1,1))*detections[i*4+3] + - static_cast(trans2.at(1,2))*1.0; - - - // std::cout<<"\n new: "<(0,0); - target_coords[i*4+1] = new_pt1.at(0,1); - target_coords[i*4+2] = new_pt2.at(0,0); - target_coords[i*4+3] = new_pt2.at(0,1); - // std::cout<(0,0)<<", "<(0,1)<<", "<(0,0)<<", "<(0,1)< coco_class_name(coco_class_name_, std::end( coco_class_name_ )); - int num_classes = 80; - float vis_threshold = 0.3; - // int *classes; - std::vector detected; - // checkCuda( cudaMallocHost(&classes, K *sizeof(int)) ); - // checkCuda( cudaMemcpy(classes, detections + 5 * K *sizeof(dnnType), K *sizeof(dnnType), cudaMemcpyHostToHost) ); - for(int i = 0; i i+1 (1:80); - - if(scores[j] > vis_threshold){ - std::cout<<"th: "<