diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp index e838395..1a93206 100644 --- a/demo/demo/demo3D.cpp +++ b/demo/demo/demo3D.cpp @@ -5,6 +5,7 @@ #include #include "CenternetDetection3D.h" +#include "CenternetDetection3DTrack.h" bool gRun; bool SAVE_RESULT = false; @@ -34,6 +35,7 @@ int main(int argc, char *argv[]) { n_classes = atoi(argv[4]); tk::dnn::CenternetDetection3D cnet; + tk::dnn::CenternetDetection3DTrack ctrack; tk::dnn::DetectionNN3D *detNN; @@ -42,6 +44,9 @@ int main(int argc, char *argv[]) { case 'c': detNN = &cnet; break; + case 't': + detNN = &ctrack; + break; default: FatalError("Network type not allowed (3rd parameter)\n"); } diff --git a/include/tkDNN/CenternetDetection3DTrack.h b/include/tkDNN/CenternetDetection3DTrack.h new file mode 100644 index 0000000..5149d70 --- /dev/null +++ b/include/tkDNN/CenternetDetection3DTrack.h @@ -0,0 +1,176 @@ +#ifndef CENTERNETDETECTION3DTRACK_H +#define CENTERNETDETECTION3DTRACK_H + +#include "kernels.h" +#include "utils.h" +#include "tkdnn.h" +#include +#include "opencv2/opencv.hpp" +#include +#include +#include // std::iota +#include // std::sort + +#include "DetectionNN3D.h" + +#include "kernelsThrust.h" + + +namespace tk { namespace dnn { + +struct detectionRes +{ + float score; + int cl; + cv::Mat ct, tr, bb0, bb1; + float dep; + float dim[3]; + float alpha; + float x,y,z; + float rot_y; + detectionRes() : ct(cv::Mat(cv::Size(1,2), CV_32F)), + tr(cv::Mat(cv::Size(1,2), CV_32F)), + bb0(cv::Mat(cv::Size(1,2), CV_32F)), + bb1(cv::Mat(cv::Size(1,2), CV_32F)) { } + ~detectionRes() { + ct.release(); + tr.release(); + bb0.release(); + bb1.release(); + } +}; + +struct trackingRes +{ + struct detectionRes det_res; + int tracking_id; + int age; + int active; + int color; +}; + +class CenternetDetection3DTrack : public DetectionNN3D +{ +private: + tk::dnn::dataDim_t dim; + tk::dnn::dataDim_t dim2; + tk::dnn::dataDim_t dim_hm; + tk::dnn::dataDim_t dim_wh; + tk::dnn::dataDim_t dim_reg; + tk::dnn::dataDim_t dim_track; + tk::dnn::dataDim_t dim_dep; + tk::dnn::dataDim_t dim_rot; + tk::dnn::dataDim_t dim_dim; + tk::dnn::dataDim_t dim_amodel_offset; + + /* preprocessing */ + #ifdef OPENCV_CUDACONTRIB + float *mean_d; + float *stddev_d; + #else + cv::Vec mean; + cv::Vec stddev; + dnnType *input; + #endif + float *d_ptrs; + + cv::Mat src; + cv::Mat dst; + cv::Mat dst2; + cv::Mat trans, trans2, trans_out; + + /* pre inf */ + bool iter0; + dnnType *input_pre_inf_d; + bool test_pre_inf = true; + dnnType *img_d, *hm_d; + tk::dnn::dataDim_t dim_in0; + tk::dnn::dataDim_t dim_in1; + dnnType *out_d; + + + /* postprocessing */ + int K = 100; + int width = 128;//56; // TODO + + // pointer used in the kernels + float *src_out; + int *ids_out; + + float *topk_scores; + int *topk_inds_; + float *topk_ys_; + float *topk_xs_; + int *ids_d, *ids_; + + float *ones; + + float *scores, *scores_d; + int *clses, *clses_d; + int *topk_inds_d; + float *topk_ys_d; + float *topk_xs_d; + int *inttopk_xs_d, *inttopk_ys_d; + + float *bbx0, *bby0, *bbx1, *bby1; + float *bbx0_d, *bby0_d, *bbx1_d, *bby1_d; + + int *intxs, *intys; + + float *track, *dep, *rot, *dim_, *wh, *amodel_offset; + float *track_d, *dep_d, *rot_d, *dim_d, *wh_d, *amodel_offset_d; + + float *target_coords; + + /* visualization */ + cv::Mat r; + cv::Mat calibs; + cv::Mat corners, pts3DHomo; + + std::vector> face_id; + cv::Scalar tr_colors[256]; + bool view2d = false; + + //processing + struct threshold op; + float out_thresh = 0.1; + float new_thresh = 0.3; + float vis_thresh = 0.3; + float peakThreshold = 0.2; + float centerThreshold = 0.3; //default 0.5 + + + //detections + std::vector det_res; + int count_det; + //tracks + std::vector tr_res; + int count_tr; + int track_id=0; + + + bool init_preprocessing(); + bool init_pre_inf(); + bool init_postprocessing(); + bool init_visualization(const int n_classes); + void pre_inf(); + void _get_additional_inputs(); + cv::Mat transform_preds_with_trans(float x1, float x2); + void tracking(); + +public: + tk::dnn::Network *pre_phase_net = nullptr; + CenternetDetection3DTrack() {}; + ~CenternetDetection3DTrack() {}; + bool init(const std::string& tensor_path, const int n_classes=3); + void preprocess(cv::Mat &frame); + void postprocess(); + cv::Mat draw(cv::Mat &frame); +}; + + +} // namespace dnn +} // namespace tk + + +#endif /*CENTERNETDETECTION3DTRACK_H*/ \ No newline at end of file diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp new file mode 100644 index 0000000..35488f7 --- /dev/null +++ b/src/CenternetDetection3DTrack.cpp @@ -0,0 +1,862 @@ +#include "CenternetDetection3DTrack.h" + + +namespace tk { namespace dnn { + +bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n_classes){ + std::cout<<(tensor_path).c_str()<<"\n"; + netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); + + dim = netRT->input_dim; + dim.c = 3; + + init_preprocessing(); + init_pre_inf(); + init_postprocessing(); + init_visualization(n_classes); + + count_tr = 0; +} + +bool CenternetDetection3DTrack::init_preprocessing(){ + //image transformation + src = cv::Mat(cv::Size(2,3), CV_32F); + dst = cv::Mat(cv::Size(2,3), CV_32F); + dst2 = cv::Mat(cv::Size(2,3), CV_32F); + trans = cv::Mat(cv::Size(3,2), CV_32F); + trans2 = cv::Mat(cv::Size(3,2), CV_32F); + trans_out = cv::Mat(cv::Size(3,2), CV_32F); + + dst2.at(0,0)=width * 0.5; + dst2.at(0,1)=width * 0.5; + dst2.at(1,0)=width * 0.5; + dst2.at(1,1)=width * 0.5 + width * -0.5; + + dst2.at(2,0)=dst2.at(1,0) + (-dst2.at(0,1)+dst2.at(1,1) ); + dst2.at(2,1)=dst2.at(1,1) + (dst2.at(0,0)-dst2.at(1,0) ); + + +#ifdef OPENCV_CUDACONTRIB + + checkCuda( cudaMalloc(&mean_d, 3 * sizeof(float)) ); + checkCuda( cudaMalloc(&stddev_d, 3 * sizeof(float)) ); + float mean[3] = {0.40789655, 0.44719303, 0.47026116}; + float stddev[3] = {0.2886383, 0.27408165, 0.27809834}; + + checkCuda(cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice)); + checkCuda(cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice)); +#else + checkCuda(cudaMallocHost(&input, sizeof(dnnType)*dim.tot())); + mean << 0.40789655, 0.44719303, 0.47026116; + stddev << 0.2886383, 0.27408165, 0.27809834; + +#endif + + checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot())); + checkCuda(cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot())); + checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) ); +} + +bool CenternetDetection3DTrack::init_pre_inf(){ + // initial steps: the first part of the network + const char *pre_img_conv1_bin = "/home/davide/Projects/repos/tkDNN/build/dla34_cnet3d_track/layers/base-pre_img_layer-0.bin"; + const char *pre_hm_conv1_bin = "dla34_cnet3d_track/layers/base-pre_hm_layer-0.bin"; + const char *conv1_bin = "dla34_cnet3d_track/layers/base-base_layer-0.bin"; + const char *conv2_bin = "dla34_cnet3d_track/layers/base-level0-0.bin"; + dim_in0 = tk::dnn::dataDim_t(1, 3, 512, 512, 1); + dim_in1 = tk::dnn::dataDim_t(1, 1, 512, 512, 1); + + + checkCuda( cudaMalloc(&out_d, netRT->input_dim.tot()*sizeof(dnnType)) ); + checkCuda( cudaMalloc(&img_d, dim_in0.tot()*sizeof(dnnType)) ); + checkCuda( cudaMalloc(&hm_d, dim_in1.tot()*sizeof(dnnType)) ); + // init to zeros hm + dnnType *hm_h; + checkCuda( cudaMallocHost(&hm_h, 1 * dim.h * dim.w*sizeof(dnnType)) ); + for(int i=0; i<1 * dim.h * dim.w; i++) + hm_h[i]=0.0f; + checkCuda( cudaMemcpy(hm_d, hm_h, 1 * dim.h * dim.w * sizeof(dnnType), cudaMemcpyHostToDevice) ); + checkCuda( cudaFreeHost(hm_h) ); + dnnType *i0_h, *i1_h, *i2_h; + // dnnType *i0_d, *i1_d, *i2_d; + + // const char *input_bin = "dla34_cnet3d_track/debug/input.bin"; + // const char *pre_img_bin = "dla34_cnet3d_track/debug/pre_imgages.bin"; + // const char *pre_hm_bin = "dla34_cnet3d_track/debug/pre_hms.bin"; + // readBinaryFile(pre_img_bin, dim_in0.tot(), &i0_h, &img_d); + // readBinaryFile(pre_hm_bin, dim_in1.tot(), &i1_h, &hm_d); + // readBinaryFile(input_bin, dim_in0.tot(), &i2_h, &input_pre_inf_d); + + pre_phase_net = new tk::dnn::Network(dim_in0); + //pre-img + tk::dnn::Input *in_pre_img = new tk::dnn::Input(pre_phase_net, dim_in0, img_d); + tk::dnn::Conv2d *pre_img_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_img_conv1_bin, true); + tk::dnn::Activation *pre_img_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + //pre-hm + tk::dnn::Input *in_pre_hm = new tk::dnn::Input(pre_phase_net, dim_in1, hm_d); + tk::dnn::Conv2d *pre_hm_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_hm_conv1_bin, true); + tk::dnn::Activation *pre_hm_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + // image input + tk::dnn::Input *input_image = new tk::dnn::Input(pre_phase_net, dim_in0, input_pre_inf_d); + tk::dnn::Conv2d *conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, conv1_bin, true); + tk::dnn::Activation *relu1 = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Shortcut *s0_input = new tk::dnn::Shortcut(pre_phase_net, pre_img_relu); + tk::dnn::Shortcut *s1_input = new tk::dnn::Shortcut(pre_phase_net, pre_hm_relu); + // output data + out_d = s1_input->dstData; + //print network model + pre_phase_net->print(); + + iter0=true; // in the first iteration the last input is equal to the current input. + return true; +} + +bool CenternetDetection3DTrack::init_postprocessing(){ + srand(0); //seed = 0 for random colors + + dim_hm = tk::dnn::dataDim_t(1, 10, 128, 128, 1); + dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_reg = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_track = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_dep = tk::dnn::dataDim_t(1, 1, 128, 128, 1); + dim_rot = tk::dnn::dataDim_t(1, 8, 128, 128, 1); + dim_dim = tk::dnn::dataDim_t(1, 3, 128, 128, 1); + dim_amodel_offset = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + + checkCuda( cudaMalloc(&topk_scores, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&topk_inds_, dim_hm.c * K *sizeof(int)) ); + checkCuda( cudaMalloc(&topk_ys_, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&topk_xs_, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&ids_d, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) ); + checkCuda( cudaMallocHost(&ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) ); + for(int i =0; i(0,0) = 633.0; + calibs.at(0,1) = 0.0; + calibs.at(0,2) = 0.0; //w/2 + calibs.at(0,3) = 0.0; + calibs.at(1,0) = 0.0; + calibs.at(1,1) = 633.0; + calibs.at(1,2) = 0.0; //h/2 + calibs.at(1,3) = 0.0; + calibs.at(2,0) = 0.0; + calibs.at(2,1) = 0.0; + calibs.at(2,2) = 1.0; + calibs.at(2,3) = 0.0; + + // Alloc array used in the kernel + checkCuda( cudaMalloc(&src_out, K *sizeof(float)) ); + checkCuda( cudaMalloc(&ids_out, K *sizeof(int)) ); +} + +bool CenternetDetection3DTrack::init_visualization(const int n_classes){ + classes = n_classes; + // const char *kitti_class_name[] = { + // "person", "car", "bicycle"}; + // classesNames = std::vector(kitti_class_name, std::end( kitti_class_name)); + + const char *class_name[] = {"car", "truck", "bus", "trailer", "construction_vehicle", "pedestrian", + "motorcycle", "bicycle", "traffic_cone", "barrier"}; + classesNames = std::vector(class_name, std::end( class_name)); + + // const char *coco_class_name[] = { + // "person", "bicycle", "car", "motorcycle", "airplane", + // "bus", "train", "truck", "boat", "traffic light", "fire hydrant", + // "stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse", + // "sheep", "cow", "elephant", "bear", "zebra", "giraffe", "backpack", + // "umbrella", "handbag", "tie", "suitcase", "frisbee", "skis", + // "snowboard", "sports ball", "kite", "baseball bat", "baseball glove", + // "skateboard", "surfboard", "tennis racket", "bottle", "wine glass", + // "cup", "fork", "knife", "spoon", "bowl", "banana", "apple", "sandwich", + // "orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake", + // "chair", "couch", "potted plant", "bed", "dining table", "toilet", "tv", + // "laptop", "mouse", "remote", "keyboard", "cell phone", "microwave", + // "oven", "toaster", "sink", "refrigerator", "book", "clock", "vase", + // "scissors", "teddy bear", "hair drier", "toothbrush" + // }; + // classesNames = std::vector(coco_class_name, std::end( coco_class_name)); + + for(int c=0; c(0,1) = 0.0; + r.at(1,0) = 0.0; + r.at(1,1) = 1.0; + r.at(1,2) = 0.0; + r.at(2,1) = 0.0; + + corners = cv::Mat(cv::Size(8,3), CV_32F); + corners.at(1,0) = 0.0; + corners.at(1,1) = 0.0; + corners.at(1,2) = 0.0; + corners.at(1,3) = 0.0; + + pts3DHomo = cv::Mat(cv::Size(8,4), CV_32F); + pts3DHomo.at(3,0) = 1.0; + pts3DHomo.at(3,1) = 1.0; + pts3DHomo.at(3,2) = 1.0; + pts3DHomo.at(3,3) = 1.0; + pts3DHomo.at(3,4) = 1.0; + pts3DHomo.at(3,5) = 1.0; + pts3DHomo.at(3,6) = 1.0; + pts3DHomo.at(3,7) = 1.0; + + face_id.push_back({0,1,5,4}); + face_id.push_back({1,2,6, 5}); + face_id.push_back({2,3,7,6}); + face_id.push_back({3,0,4,7}); + // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); +} + +void CenternetDetection3DTrack::_get_additional_inputs(){ + //None no additional input +} + +void CenternetDetection3DTrack::pre_inf(){ + TKDNN_TSTART + tk::dnn::dataDim_t dim_aus; + pre_phase_net->infer(dim_aus, nullptr); + TKDNN_TSTOP + checkCuda( cudaDeviceSynchronize() ); + checkCuda( cudaMemcpy(input_d, pre_phase_net->layers[pre_phase_net->num_layers-1]->dstData, netRT->input_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice) ); + checkCuda( cudaDeviceSynchronize() ); +} + +void CenternetDetection3DTrack::preprocess(cv::Mat &frame){ + // -----------------------------------pre-process ------------------------------------------ + + cv::Size sz = originalSize; + cv::Size sz_old; + float scale = 1.0; + float new_height = sz.height * scale; + float new_width = sz.width * scale; + if(sz.height != sz_old.height && sz.width != sz_old.width){ + calibs.at(0,2) = new_width / 2.0f; + calibs.at(1,2) = new_height /2.0f; + float c[] = {new_width / 2.0f, new_height /2.0f}; + float s[] = {dim.w, dim.h}; + // float s = new_width >= new_height ? new_width : new_height; + // ----------- get_affine_transform + // rot_rad = pi * 0 / 100 --> 0 + dim.print(); + src.at(0,0)=c[0]; + src.at(0,1)=c[1]; + src.at(1,0)=c[0]; + src.at(1,1)=c[1] + s[0] * -0.5; + dst.at(0,0)=dim.w * 0.5; + dst.at(0,1)=dim.h * 0.5; + dst.at(1,0)=dim.w * 0.5; + dst.at(1,1)=dim.h * 0.5 + dim.w * -0.5; + + src.at(2,0)=src.at(1,0) + (-src.at(0,1)+src.at(1,1) ); + src.at(2,1)=src.at(1,1) + (src.at(0,0)-src.at(1,0) ); + dst.at(2,0)=dst.at(1,0) + (-dst.at(0,1)+dst.at(1,1) ); + dst.at(2,1)=dst.at(1,1) + (dst.at(0,0)-dst.at(1,0) ); + + + trans = cv::getAffineTransform( src, dst ); + trans2 = cv::getAffineTransform( dst2, src ); + trans2.convertTo(trans_out, CV_32F); + } + sz_old = sz; +#ifdef OPENCV_CUDACONTRIB + std::cout<<"OPENCV CPMTROB\n"; + cv::cuda::GpuMat im_Orig; + cv::cuda::GpuMat imageF1_d, imageF2_d; + + im_Orig = cv::cuda::GpuMat(frame); + // cv::cuda::resize (im_Orig, imageF1_d, cv::Size(new_width, new_height)); + imageF1_d = im_Orig; + checkCuda( cudaDeviceSynchronize() ); + + sz = imageF1_d.size(); + + cv::cuda::warpAffine(imageF1_d, imageF2_d, trans, cv::Size(dim.w, dim.h), cv::INTER_LINEAR ); + checkCuda( cudaDeviceSynchronize() ); + + imageF2_d.convertTo(imageF1_d, CV_32FC3, 1/255.0); + checkCuda( cudaDeviceSynchronize() ); + + dim2 = dim; + cv::cuda::GpuMat bgr[3]; + cv::cuda::split(imageF1_d,bgr);//split source + + for(int i=0; i(0,0) = x1; + target_coords.at(0,1) = x2; + target_coords.at(0,2) = 1.0; + return trans_out * target_coords; +} + +void CenternetDetection3DTrack::tracking(){ + + float item_size[count_det]; + int item_cl[count_det]; + float dets[2*count_det]; + for(int i=0; i(0,0) - det_res[i].bb0.at(0,0)) * + (det_res[i].bb1.at(0,1) - det_res[i].bb0.at(0,1)); + item_cl[i] = det_res[i].cl; + dets[i*2] = det_res[i].ct.at(0,0); + dets[i*2+1] = det_res[i].ct.at(0,1); + } + + float track_size[count_tr]; + int track_cl[count_tr]; + float tracks[2*count_tr]; + for(int i=0; i(0,0) - tr_res[i].det_res.bb0.at(0,0)) * + (tr_res[i].det_res.bb1.at(0,1) - tr_res[i].det_res.bb0.at(0,1)); + track_cl[i] = tr_res[i].det_res.cl; + tracks[i*2] = tr_res[i].det_res.ct.at(0,0); + tracks[i*2+1] = tr_res[i].det_res.ct.at(0,1); + } + float dist[count_tr*count_det]; + bool invalid; + for(int i=0; i track_size[i] || dist[j*count_tr+i] > item_size[j] || item_cl[j] != track_cl[i]; + dist[j*count_tr+i] = dist[j*count_tr+i] + invalid * (1 << 18); + } + } + int matched_indices[2*count_tr]; + float min_tr; + int min_idtr=-1; + for(int i=0; i new_tr_res; + int id_new_tr=0; + for(int i=0; i new_thresh) { + count_tr_ ++; + struct trackingRes new_tr_res_; + new_tr_res_.det_res.score = det_res[i].score; + new_tr_res_.det_res.cl = det_res[i].cl; + new_tr_res_.det_res.ct = det_res[i].ct; + new_tr_res_.det_res.tr = det_res[i].tr; + new_tr_res_.det_res.bb0 = det_res[i].bb0; + new_tr_res_.det_res.bb1 = det_res[i].bb1; + new_tr_res_.det_res.dep = det_res[i].dep; + new_tr_res_.det_res.dim[0] = det_res[i].dim[0]; + new_tr_res_.det_res.dim[1] = det_res[i].dim[1]; + new_tr_res_.det_res.dim[2] = det_res[i].dim[2]; + new_tr_res_.det_res.alpha = det_res[i].alpha; + new_tr_res_.det_res.x = det_res[i].x; + new_tr_res_.det_res.y = det_res[i].y; + new_tr_res_.det_res.z = det_res[i].z; + new_tr_res_.det_res.rot_y = det_res[i].rot_y; + new_tr_res_.tracking_id = track_id++; + new_tr_res_.age = 1; + new_tr_res_.active = 1; + new_tr_res_.color = rand() % 256; + tr_res.push_back(new_tr_res_); + } + } + count_tr = count_tr_; + + if(track_id==1000) + track_id=0; + det_res.clear(); + +} + +void CenternetDetection3DTrack::postprocess(){ + dnnType *rt_out[9]; + rt_out[0] = (dnnType *)netRT->buffersRT[1]; + rt_out[1] = (dnnType *)netRT->buffersRT[2]; + rt_out[2] = (dnnType *)netRT->buffersRT[3]; + rt_out[3] = (dnnType *)netRT->buffersRT[4]; + rt_out[4] = (dnnType *)netRT->buffersRT[5]; + rt_out[5] = (dnnType *)netRT->buffersRT[6]; + rt_out[6] = (dnnType *)netRT->buffersRT[7]; + rt_out[7] = (dnnType *)netRT->buffersRT[8]; + rt_out[8] = (dnnType *)netRT->buffersRT[9]; + + // ------------------------------------ process -------------------------------------------- + + activationSIGMOIDForward(rt_out[0], rt_out[0], dim_hm.tot()); + checkCuda( cudaDeviceSynchronize() ); + + // output['dep'] = 1. / (output['dep'].sigmoid() + 1e-6) - 1. + activationSIGMOIDForward(rt_out[5], rt_out[5], dim_dep.tot()); + checkCuda( cudaDeviceSynchronize() ); + transformDep(ones, ones + dim_dep.tot(), rt_out[5], rt_out[5] + dim_dep.tot()); + checkCuda( cudaDeviceSynchronize() ); + + // nms + subtractWithThreshold(rt_out[0], rt_out[0] + dim_hm.tot(), rt_out[1], rt_out[0], op); + + // ----------- nms end + // ----------- topk + + if(K > dim_hm.h * dim_hm.w){ + printf ("Error topk (K is too large)\n"); + return; + } + + checkCuda( cudaMemcpy(ids_d, ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyHostToDevice) ); + + sort(rt_out[0],rt_out[0]+dim_hm.tot(),ids_d); + checkCuda( cudaDeviceSynchronize() ); + + topk(rt_out[0], ids_d, K, scores_d, topk_inds_d, topk_ys_d, topk_xs_d); + checkCuda( cudaDeviceSynchronize() ); + + checkCuda( cudaMemcpy(scores, scores_d, K *sizeof(float), cudaMemcpyDeviceToHost) ); + + topKxyclasses(topk_inds_d, topk_inds_d+K, K, width, dim_hm.w*dim_hm.h, clses_d, inttopk_xs_d, inttopk_ys_d); + checkCuda( cudaDeviceSynchronize() ); + checkCuda( cudaMemcpy(topk_xs_d, (float *)inttopk_xs_d, K*sizeof(float), cudaMemcpyDeviceToDevice) ); + checkCuda( cudaMemcpy(topk_ys_d, (float *)inttopk_ys_d, K*sizeof(float), cudaMemcpyDeviceToDevice) ); + + checkCuda( cudaMemcpy(intxs, inttopk_xs_d, K * sizeof(int), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(intys, inttopk_ys_d, K * sizeof(int), cudaMemcpyDeviceToHost) ); + + checkCuda( cudaMemcpy(clses, clses_d, K*sizeof(int), cudaMemcpyDeviceToHost) ); + + // ----------- topk end + + topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3], src_out, ids_out); + checkCuda( cudaDeviceSynchronize() ); + + bboxes(topk_inds_d, K, dim_wh.h*dim_wh.w, topk_xs_d, topk_ys_d, rt_out[2], bbx0_d, bbx1_d, bby0_d, bby1_d, src_out, ids_out); + checkCuda( cudaDeviceSynchronize() ); + checkCuda( cudaMemcpy(bbx0, bbx0_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(bby0, bby0_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(bbx1, bbx1_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(bby1, bby1_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + + //regression heads + // ['tracking', 'dep', 'rot', 'dim', 'amodel_offset', + // 'nuscenes_att', 'velocity'] + getRecordsFromTopKId(topk_inds_d, K, dim_track.c, dim_track.h * dim_track.w, rt_out[4], track_d, ids_out); + checkCuda( cudaMemcpy(track, track_d, K * dim_track.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_dep.c, dim_dep.h * dim_dep.w, rt_out[5], dep_d, ids_out); + checkCuda( cudaMemcpy(dep, dep_d, K * dim_dep.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_rot.c, dim_rot.h * dim_rot.w, rt_out[6], rot_d, ids_out); + checkCuda( cudaMemcpy(rot, rot_d, K * dim_rot.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_dim.c, dim_dim.h * dim_dim.w, rt_out[7], dim_d, ids_out); + checkCuda( cudaMemcpy(dim_, dim_d, K * dim_dim.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_amodel_offset.c, dim_amodel_offset.h * dim_amodel_offset.w, rt_out[8], amodel_offset_d, ids_out); + checkCuda( cudaMemcpy(amodel_offset, amodel_offset_d, K * dim_amodel_offset.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + // ---------------------------------- post-process ----------------------------------------- + + count_det = 0; + det_res.clear(); + for(int i = 0; i(2,3); + new_det_res.x = ((float)new_det_res.ct.at(0,0) * dep[i] - calibs.at(0,3) - calibs.at(0,2) * new_det_res.z) / calibs.at(0,0); + new_det_res.y = ((float)new_det_res.ct.at(0,1) * dep[i] - calibs.at(1,3) - calibs.at(1,2) * new_det_res.z) / calibs.at(1,1) + (dim_[i] / 2); + + // alpha2rot_y + // idx = rot[:, 1] > rot[:, 5] + // alpha1 = np.arctan2(rot[:, 2], rot[:, 3]) + (-0.5 * np.pi) + // alpha2 = np.arctan2(rot[:, 6], rot[:, 7]) + ( 0.5 * np.pi) + // return alpha1 * idx + alpha2 * (1 - idx) + if(rot[1*K + i] > rot[5*K + i]) + new_det_res.alpha = std::atan2(rot[2*K + i], rot[3*K + i]) -0.5 * M_PI; + else + new_det_res.alpha = std::atan2(rot[6*K + i], rot[7*K + i]) +0.5 * M_PI; + new_det_res.rot_y = (new_det_res.alpha + std::atan2((float)new_det_res.ct.at(0,0) - calibs.at(0,2), calibs.at(0,0))); + new_det_res.ct = new_det_res.ct + new_det_res.tr; //dest + det_res.push_back(new_det_res); + + } + // track step + tracking(); +} + +cv::Mat CenternetDetection3DTrack::draw(cv::Mat &frame) { + + float sc; + int id; + std::string txt; + int baseline = 0; + float font_scale = 0.8; + int thickness = 2; + for(int i=0; i vis_thresh){// && tr_res[i].active!=0) { + if(view2d) { + + + cv::rectangle(frame, cv::Point(tr_res[i].det_res.bb0.at(0,0), tr_res[i].det_res.bb0.at(0,1)), + cv::Point(tr_res[i].det_res.bb1.at(0,0), tr_res[i].det_res.bb1.at(0,1)), tr_colors[tr_res[i].color], thickness); + cv::rectangle(frame, cv::Point(tr_res[i].det_res.bb0.at(0,0), + tr_res[i].det_res.bb0.at(0,1) - text_size.height - thickness), + cv::Point(tr_res[i].det_res.bb0.at(0,0) + text_size.width, + tr_res[i].det_res.bb0.at(0,1)), tr_colors[tr_res[i].color], -1); + + cv::putText(frame, txt, cv::Point(tr_res[i].det_res.bb0.at(0,0), + tr_res[i].det_res.bb0.at(0,1) - thickness -1), + cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1); + + cv::arrowedLine(frame, cv::Point((int)tr_res[i].det_res.ct.at(0,0), + (int)tr_res[i].det_res.ct.at(0,1)), + cv::Point((int)(tr_res[i].det_res.ct.at(0,0) + tr_res[i].det_res.tr.at(0,0)), + (int)(tr_res[i].det_res.ct.at(0,1) + tr_res[i].det_res.tr.at(0,1))), + cv::Scalar(255, 0, 255), 2); + } + //3d + if(!view2d && tr_res[i].det_res.z > 1){ + r.at(0,0) = std::cos(tr_res[i].det_res.rot_y); + r.at(0,2) = std::sin(tr_res[i].det_res.rot_y); + r.at(2,0) = -std::sin(tr_res[i].det_res.rot_y); + r.at(2,2) = std::cos(tr_res[i].det_res.rot_y); + + corners.at(0,0) = tr_res[i].det_res.dim[2]/2; + corners.at(0,1) = tr_res[i].det_res.dim[2]/2; + corners.at(0,2) = -tr_res[i].det_res.dim[2]/2; + corners.at(0,3) = -tr_res[i].det_res.dim[2]/2; + corners.at(0,4) = tr_res[i].det_res.dim[2]/2; + corners.at(0,5) = tr_res[i].det_res.dim[2]/2; + corners.at(0,6) = -tr_res[i].det_res.dim[2]/2; + corners.at(0,7) = -tr_res[i].det_res.dim[2]/2; + + corners.at(1,4) = -tr_res[i].det_res.dim[0]; + corners.at(1,5) = -tr_res[i].det_res.dim[0]; + corners.at(1,6) = -tr_res[i].det_res.dim[0]; + corners.at(1,7) = -tr_res[i].det_res.dim[0]; + + corners.at(2,0) = tr_res[i].det_res.dim[1]/2; + corners.at(2,1) = -tr_res[i].det_res.dim[1]/2; + corners.at(2,2) = -tr_res[i].det_res.dim[1]/2; + corners.at(2,3) = tr_res[i].det_res.dim[1]/2; + corners.at(2,4) = tr_res[i].det_res.dim[1]/2; + corners.at(2,5) = -tr_res[i].det_res.dim[1]/2; + corners.at(2,6) = -tr_res[i].det_res.dim[1]/2; + corners.at(2,7) = tr_res[i].det_res.dim[1]/2; + + cv::Mat aus = r * corners; + + for(int k=0; k<8; k++) { + aus.at(0,k) += tr_res[i].det_res.x; + aus.at(1,k) += tr_res[i].det_res.y; + aus.at(2,k) += tr_res[i].det_res.z; + } + + // corners.copyTo(pts3DHomo(cv::Rect(0, 0, 8, 3))); + for(int k1=0; k1<3; k1++) { + for(int k2=0; k2<8; k2++) + pts3DHomo.at(k1,k2) = aus.at(k1,k2); + } + + aus.release(); + aus = calibs * pts3DHomo; + std::vector res_corners; + for(int k=0; k<8; k++) { + res_corners.push_back(aus.at(0,k) / aus.at(2,k)); + res_corners.push_back(aus.at(1,k) / aus.at(2,k)); + } + aus.release(); + for(int ind_f = 3; ind_f>=0; ind_f--) { + for(int j=0; j<4; j++) { + cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(j) * 2), + res_corners.at(face_id.at(ind_f).at(j) * 2 + 1)), + cv::Point(res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2), + res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), + tr_colors[tr_res[i].color], 2); + if(ind_f == 0) { + cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(0) * 2), + res_corners.at(face_id.at(ind_f).at(0) * 2 + 1)), + cv::Point(res_corners.at(face_id.at(ind_f).at(2) * 2), + res_corners.at(face_id.at(ind_f).at(2) * 2 + 1)), tr_colors[tr_res[i].color], 2); + cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(1) * 2), + res_corners.at(face_id.at(ind_f).at(1) * 2 + 1)), + cv::Point(res_corners.at(face_id.at(ind_f).at(3) * 2), + res_corners.at(face_id.at(ind_f).at(3) * 2 + 1)), tr_colors[tr_res[i].color], 2); + } + } + } + float bb0=(1 << 10), bb1=0, bb2=(1 << 10), bb3=0; + for(int k=0; k<8; k++) { + if(res_corners[2*k]bb1) + bb1=res_corners[2*k]; + if(res_corners[2*k+1]bb3) + bb3=res_corners[2*k+1]; + + } + cv::rectangle(frame, cv::Point(bb0, bb2), cv::Point(bb1, bb3), + tr_colors[tr_res[i].color], thickness); + cv::rectangle(frame, cv::Point(bb0, bb2 - text_size.height - thickness), + cv::Point(bb0 + text_size.width, bb2), tr_colors[tr_res[i].color], -1); + + cv::putText(frame, txt, cv::Point(bb0, bb2 - thickness -1), cv::FONT_HERSHEY_SIMPLEX, + font_scale, cv::Scalar(255, 255, 255), 1); + + cv::arrowedLine(frame, cv::Point((int)((bb0 + bb1)/2), (int)((bb2 + bb3)/2)), + cv::Point((int)((bb0 + bb1)/2 + tr_res[i].det_res.tr.at(0,0)), + (int)((bb2 + bb3)/2 + tr_res[i].det_res.tr.at(0,1))), + cv::Scalar(255, 0, 255), 2); + } + } + + } + return frame; +} + +}} + +