From 64098ad2441540a3bb5d12cd42f199c54ba6b4ba Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Fri, 22 May 2020 12:51:25 +0200 Subject: [PATCH 001/162] Add CenterNet based on DLA34 for 3D, CUDNN and TensorRT work Signed-off-by: Davide Sapienza --- CMakeLists.txt | 3 + tests/dla34_cnet3d/dla34_cnet3d.cpp | 562 ++++++++++++++++++++++++++++ 2 files changed, 565 insertions(+) create mode 100644 tests/dla34_cnet3d/dla34_cnet3d.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index 028d375..1ae4289 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -142,6 +142,9 @@ target_link_libraries(test_dla34 tkDNN) add_executable(test_dla34_cnet tests/dla34_cnet/dla34_cnet.cpp) target_link_libraries(test_dla34_cnet tkDNN) +add_executable(test_dla34_cnet3d tests/dla34_cnet3d/dla34_cnet3d.cpp) +target_link_libraries(test_dla34_cnet3d tkDNN) + add_executable(test_imuodom tests/imuodom/imuodom.cpp) target_link_libraries(test_imuodom tkDNN) ################################################################################ diff --git a/tests/dla34_cnet3d/dla34_cnet3d.cpp b/tests/dla34_cnet3d/dla34_cnet3d.cpp new file mode 100644 index 0000000..ecd5693 --- /dev/null +++ b/tests/dla34_cnet3d/dla34_cnet3d.cpp @@ -0,0 +1,562 @@ +#include +#include "tkdnn.h" + +const char *input_bin = "dla34_cnet3d/debug/input.bin"; +const char *conv1_bin = "dla34_cnet3d/layers/base-base_layer-0.bin"; +const char *conv2_bin = "dla34_cnet3d/layers/base-level0-0.bin"; +const char *conv3_bin = "dla34_cnet3d/layers/base-level1-0.bin"; +// s - stage, t - tree +const char *s1_t1_conv1_bin = "dla34_cnet3d/layers/base-level2-tree1-conv1.bin"; +const char *s1_t1_conv2_bin = "dla34_cnet3d/layers/base-level2-tree1-conv2.bin"; +const char *s1_t1_project = "dla34_cnet3d/layers/base-level2-project-0.bin"; +const char *s1_t2_conv1_bin = "dla34_cnet3d/layers/base-level2-tree2-conv1.bin"; +const char *s1_t2_conv2_bin = "dla34_cnet3d/layers/base-level2-tree2-conv2.bin"; +const char *s1_root_conv1_bin = "dla34_cnet3d/layers/base-level2-root-conv.bin"; +const char *s2_t1_t1_conv1_bin = "dla34_cnet3d/layers/base-level3-tree1-tree1-conv1.bin"; +const char *s2_t1_t1_conv2_bin = "dla34_cnet3d/layers/base-level3-tree1-tree1-conv2.bin"; +const char *s2_t1_t1_project = "dla34_cnet3d/layers/base-level3-tree1-project-0.bin"; +const char *s2_t1_t2_conv1_bin = "dla34_cnet3d/layers/base-level3-tree1-tree2-conv1.bin"; +const char *s2_t1_t2_conv2_bin = "dla34_cnet3d/layers/base-level3-tree1-tree2-conv2.bin"; +const char *s2_t1_root_conv1_bin = "dla34_cnet3d/layers/base-level3-tree1-root-conv.bin"; +const char *s2_t2_t1_conv1_bin = "dla34_cnet3d/layers/base-level3-tree2-tree1-conv1.bin"; +const char *s2_t2_t1_conv2_bin = "dla34_cnet3d/layers/base-level3-tree2-tree1-conv2.bin"; +const char *s2_t2_t2_conv1_bin = "dla34_cnet3d/layers/base-level3-tree2-tree2-conv1.bin"; +const char *s2_t2_t2_conv2_bin = "dla34_cnet3d/layers/base-level3-tree2-tree2-conv2.bin"; +const char *s2_t2_root_conv1_bin = "dla34_cnet3d/layers/base-level3-tree2-root-conv.bin"; +const char *s3_t1_t1_conv1_bin = "dla34_cnet3d/layers/base-level4-tree1-tree1-conv1.bin"; +const char *s3_t1_t1_conv2_bin = "dla34_cnet3d/layers/base-level4-tree1-tree1-conv2.bin"; +const char *s3_t1_t1_project = "dla34_cnet3d/layers/base-level4-tree1-project-0.bin"; +const char *s3_t1_t2_conv1_bin = "dla34_cnet3d/layers/base-level4-tree1-tree2-conv1.bin"; +const char *s3_t1_t2_conv2_bin = "dla34_cnet3d/layers/base-level4-tree1-tree2-conv2.bin"; +const char *s3_t1_root_conv1_bin = "dla34_cnet3d/layers/base-level4-tree1-root-conv.bin"; +const char *s3_t2_t1_conv1_bin = "dla34_cnet3d/layers/base-level4-tree2-tree1-conv1.bin"; +const char *s3_t2_t1_conv2_bin = "dla34_cnet3d/layers/base-level4-tree2-tree1-conv2.bin"; +const char *s3_t2_t2_conv1_bin = "dla34_cnet3d/layers/base-level4-tree2-tree2-conv1.bin"; +const char *s3_t2_t2_conv2_bin = "dla34_cnet3d/layers/base-level4-tree2-tree2-conv2.bin"; +const char *s3_t2_root_conv1_bin = "dla34_cnet3d/layers/base-level4-tree2-root-conv.bin"; +const char *s4_t1_conv1_bin = "dla34_cnet3d/layers/base-level5-tree1-conv1.bin"; +const char *s4_t1_conv2_bin = "dla34_cnet3d/layers/base-level5-tree1-conv2.bin"; +const char *s4_t1_project = "dla34_cnet3d/layers/base-level5-project-0.bin"; +const char *s4_t2_conv1_bin = "dla34_cnet3d/layers/base-level5-tree2-conv1.bin"; +const char *s4_t2_conv2_bin = "dla34_cnet3d/layers/base-level5-tree2-conv2.bin"; +const char *s4_root_conv1_bin = "dla34_cnet3d/layers/base-level5-root-conv.bin"; + +//final +// const char *fc_bin = "dla34_cnet3d/layers/output.bin"; + +const char *ida_0_p_1_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_0-proj_1-conv.bin"; +const char *ida_0_p_1_conv_bin = "dla34_cnet3d/layers/dla_up-ida_0-proj_1-conv-conv_offset_mask.bin"; +const char *ida_0_up_1_deconv_bin = "dla34_cnet3d/layers/dla_up-ida_0-up_1.bin"; +const char *ida_0_n_1_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_0-node_1-conv.bin"; +const char *ida_0_n_1_conv_bin = "dla34_cnet3d/layers/dla_up-ida_0-node_1-conv-conv_offset_mask.bin"; + +const char *ida_1_p_1_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_1-proj_1-conv.bin"; +const char *ida_1_p_1_conv_bin = "dla34_cnet3d/layers/dla_up-ida_1-proj_1-conv-conv_offset_mask.bin"; +const char *ida_1_up_1_deconv_bin = "dla34_cnet3d/layers/dla_up-ida_1-up_1.bin"; +const char *ida_1_n_1_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_1-node_1-conv.bin"; +const char *ida_1_n_1_conv_bin = "dla34_cnet3d/layers/dla_up-ida_1-node_1-conv-conv_offset_mask.bin"; +const char *ida_1_p_2_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_1-proj_2-conv.bin"; +const char *ida_1_p_2_conv_bin = "dla34_cnet3d/layers/dla_up-ida_1-proj_2-conv-conv_offset_mask.bin"; +const char *ida_1_up_2_deconv_bin = "dla34_cnet3d/layers/dla_up-ida_1-up_2.bin"; +const char *ida_1_n_2_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_1-node_2-conv.bin"; +const char *ida_1_n_2_conv_bin = "dla34_cnet3d/layers/dla_up-ida_1-node_2-conv-conv_offset_mask.bin"; + +const char *ida_2_p_1_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_2-proj_1-conv.bin"; +const char *ida_2_p_1_conv_bin = "dla34_cnet3d/layers/dla_up-ida_2-proj_1-conv-conv_offset_mask.bin"; +const char *ida_2_up_1_deconv_bin = "dla34_cnet3d/layers/dla_up-ida_2-up_1.bin"; +const char *ida_2_n_1_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_2-node_1-conv.bin"; +const char *ida_2_n_1_conv_bin = "dla34_cnet3d/layers/dla_up-ida_2-node_1-conv-conv_offset_mask.bin"; +const char *ida_2_p_2_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_2-proj_2-conv.bin"; +const char *ida_2_p_2_conv_bin = "dla34_cnet3d/layers/dla_up-ida_2-proj_2-conv-conv_offset_mask.bin"; +const char *ida_2_up_2_deconv_bin = "dla34_cnet3d/layers/dla_up-ida_2-up_2.bin"; +const char *ida_2_n_2_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_2-node_2-conv.bin"; +const char *ida_2_n_2_conv_bin = "dla34_cnet3d/layers/dla_up-ida_2-node_2-conv-conv_offset_mask.bin"; +const char *ida_2_p_3_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_2-proj_3-conv.bin"; +const char *ida_2_p_3_conv_bin = "dla34_cnet3d/layers/dla_up-ida_2-proj_3-conv-conv_offset_mask.bin"; +const char *ida_2_up_3_deconv_bin = "dla34_cnet3d/layers/dla_up-ida_2-up_3.bin"; +const char *ida_2_n_3_dcn_bin = "dla34_cnet3d/layers/dla_up-ida_2-node_3-conv.bin"; +const char *ida_2_n_3_conv_bin = "dla34_cnet3d/layers/dla_up-ida_2-node_3-conv-conv_offset_mask.bin"; + +const char *ida_up_p_1_dcn_bin = "dla34_cnet3d/layers/ida_up-proj_1-conv.bin"; +const char *ida_up_p_1_conv_bin = "dla34_cnet3d/layers/ida_up-proj_1-conv-conv_offset_mask.bin"; +const char *ida_up_up_1_deconv_bin = "dla34_cnet3d/layers/ida_up-up_1.bin"; +const char *ida_up_n_1_dcn_bin = "dla34_cnet3d/layers/ida_up-node_1-conv.bin"; +const char *ida_up_n_1_conv_bin = "dla34_cnet3d/layers/ida_up-node_1-conv-conv_offset_mask.bin"; +const char *ida_up_p_2_dcn_bin = "dla34_cnet3d/layers/ida_up-proj_2-conv.bin"; +const char *ida_up_p_2_conv_bin = "dla34_cnet3d/layers/ida_up-proj_2-conv-conv_offset_mask.bin"; +const char *ida_up_up_2_deconv_bin = "dla34_cnet3d/layers/ida_up-up_2.bin"; +const char *ida_up_n_2_dcn_bin = "dla34_cnet3d/layers/ida_up-node_2-conv.bin"; +const char *ida_up_n_2_conv_bin = "dla34_cnet3d/layers/ida_up-node_2-conv-conv_offset_mask.bin"; + +const char *hm_conv1_bin = "dla34_cnet3d/layers/hm-0.bin"; +const char *hm_conv2_bin = "dla34_cnet3d/layers/hm-2.bin"; +const char *wh_conv1_bin = "dla34_cnet3d/layers/wh-0.bin"; +const char *wh_conv2_bin = "dla34_cnet3d/layers/wh-2.bin"; +const char *reg_conv1_bin = "dla34_cnet3d/layers/reg-0.bin"; +const char *reg_conv2_bin = "dla34_cnet3d/layers/reg-2.bin"; +const char *dep_conv1_bin = "dla34_cnet3d/layers/dep-0.bin"; +const char *dep_conv2_bin = "dla34_cnet3d/layers/dep-2.bin"; +const char *rot_conv1_bin = "dla34_cnet3d/layers/rot-0.bin"; +const char *rot_conv2_bin = "dla34_cnet3d/layers/rot-2.bin"; +const char *dim_conv1_bin = "dla34_cnet3d/layers/dim-0.bin"; +const char *dim_conv2_bin = "dla34_cnet3d/layers/dim-2.bin"; + +const char *output_bin[]={ +"dla34_cnet3d/debug/hm.bin", +"dla34_cnet3d/debug/wh.bin", +"dla34_cnet3d/debug/reg.bin", +"dla34_cnet3d/debug/dep.bin", +"dla34_cnet3d/debug/rot.bin", +"dla34_cnet3d/debug/dim.bin"}; + +int main() +{ + + // downloadWeightsifDoNotExist(input_bin, "dla34_cnet3d", "https://cloud.hipert.unimore.it/s/KRZBbCQsKAtQwpZ/download"); + + // Network layout + tk::dnn::dataDim_t dim(1, 3, 512, 512, 1); + tk::dnn::Network net(dim); + tk::dnn::Layer *last1, *last2, *last3, *last4; + tk::dnn::Layer *base1, *base2, *base3, *base4, *base5, *base6, *ida1, *ida2_1, *ida2_2, *ida3_1, *ida3_2, *ida3_3, *idaup_1, *idaup_2; + + tk::dnn::Conv2d conv1(&net, 16, 7, 7, 1, 1, 3, 3, conv1_bin, true); + tk::dnn::Activation relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d conv2(&net, 16, 3, 3, 1, 1, 1, 1, conv2_bin, true); + tk::dnn::Activation relu2(&net, CUDNN_ACTIVATION_RELU); + base1 = &relu2; + + tk::dnn::Conv2d conv3(&net, 32, 3, 3, 2, 2, 1, 1, conv3_bin, true); + tk::dnn::Activation relu3(&net, CUDNN_ACTIVATION_RELU); + base2 = &relu3; + + // level 2 + // tree 1 + tk::dnn::Conv2d s1_t1_conv1(&net, 64, 3, 3, 2, 2, 1, 1, s1_t1_conv1_bin, true); + tk::dnn::Activation s1_t1_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s1_t1_conv2(&net, 64, 3, 3, 1, 1, 1, 1, s1_t1_conv2_bin, true); + last2 = &s1_t1_conv2; + + // get the basicblock input and apply maxpool conv2d and relu + tk::dnn::Layer *route_s1_t1_layers[1] = { base2 }; + tk::dnn::Route route_s1_t1(&net, route_s1_t1_layers, 1); + // downsample + tk::dnn::Pooling s1_t1_maxpool1(&net, 2, 2, 2, 2, 0, 0, tk::dnn::POOLING_MAX); + // project + tk::dnn::Conv2d s1_t1_residual1_conv1(&net, 64, 1, 1, 1, 1, 0, 0, s1_t1_project, true); + + tk::dnn::Shortcut s1_t1_s1(&net, last2); + tk::dnn::Activation s1_t1_relu(&net, CUDNN_ACTIVATION_RELU); + + last1 = &s1_t1_relu; + // tree 2 + tk::dnn::Conv2d s1_t2_conv1(&net, 64, 3, 3, 1, 1, 1, 1, s1_t2_conv1_bin, true); + tk::dnn::Activation s1_t2_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s1_t2_conv2(&net, 64, 3, 3, 1, 1, 1, 1, s1_t2_conv2_bin, true); + + tk::dnn::Shortcut s1_t2_s1(&net, last1); + tk::dnn::Activation s1_t2_relu(&net, CUDNN_ACTIVATION_RELU); + last2 = &s1_t2_relu; + + // root + // join last1 and net in single input 128, 56, 56 + tk::dnn::Layer *route_s1_root_layers[2] = { last2, last1 }; + tk::dnn::Route route_s1_root(&net, route_s1_root_layers, 2); + tk::dnn::Conv2d s1_root_conv1(&net, 64, 1, 1, 1, 1, 0, 0, s1_root_conv1_bin, true); + tk::dnn::Activation s1_root_relu(&net, CUDNN_ACTIVATION_RELU); + + base3 = &s1_root_relu; + + // level 3 + // tree 1 + // tree 1 + tk::dnn::Conv2d s2_t1_t1_conv1(&net, 128, 3, 3, 2, 2, 1, 1, s2_t1_t1_conv1_bin, true); + tk::dnn::Activation s2_t1_t1_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s2_t1_t1_conv2(&net, 128, 3, 3, 1, 1, 1, 1, s2_t1_t1_conv2_bin, true); + last2 = &s2_t1_t1_conv2; + + // get the basicblock input and apply maxpool conv2d and relu + tk::dnn::Layer *route_s2_t1_t1_layers[1] = { base3 }; + tk::dnn::Route route_s2_t1_t1(&net, route_s2_t1_t1_layers, 1); + // downsample + tk::dnn::Pooling s2_t1_t1_maxpool1(&net, 2, 2, 2, 2, 0, 0, tk::dnn::POOLING_MAX); + last4 = &s2_t1_t1_maxpool1; + // project + tk::dnn::Conv2d s2_t1_t1_residual1_conv1(&net, 128, 1, 1, 1, 1, 0, 0, s2_t1_t1_project, true); + + tk::dnn::Shortcut s2_t1_t1_s1(&net, last2); + tk::dnn::Activation s2_t1_t1_relu(&net, CUDNN_ACTIVATION_RELU); + + last1 = &s2_t1_t1_relu; + + // tree 2 + tk::dnn::Conv2d s2_t1_t2_conv1(&net, 128, 3, 3, 1, 1, 1, 1, s2_t1_t2_conv1_bin, true); + tk::dnn::Activation s2_t1_t2_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s2_t1_t2_conv2(&net, 128, 3, 3, 1, 1, 1, 1, s2_t1_t2_conv2_bin, true); + + tk::dnn::Shortcut s2_t1_t2_s1(&net, last1); + tk::dnn::Activation s2_t1_t2_relu(&net, CUDNN_ACTIVATION_RELU); + last2 = &s2_t1_t2_relu; + + // root + // join last1 and net in single input 128, 56, 56 + tk::dnn::Layer *route_s2_t1_root_layers[2] = { last2, last1 }; + tk::dnn::Route route_s2_t1_root(&net, route_s2_t1_root_layers, 2); + tk::dnn::Conv2d s2_t1_root_conv1(&net, 128, 1, 1, 1, 1, 0, 0, s2_t1_root_conv1_bin, true); + tk::dnn::Activation s2_t1_root_relu(&net, CUDNN_ACTIVATION_RELU); + + last1 = &s2_t1_root_relu; + last3 = &s2_t1_root_relu; + // tree 2 + // tree 1 + tk::dnn::Conv2d s2_t2_t1_conv1(&net, 128, 3, 3, 1, 1, 1, 1, s2_t2_t1_conv1_bin, true); + tk::dnn::Activation s2_t2_t1_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s2_t2_t1_conv2(&net, 128, 3, 3, 1, 1, 1, 1, s2_t2_t1_conv2_bin, true); + tk::dnn::Shortcut s2_t2_t1_s1(&net, last1); + tk::dnn::Activation s2_t2_t1_relu(&net, CUDNN_ACTIVATION_RELU); + + last1 = &s2_t2_t1_relu; + + // tree 2 + tk::dnn::Conv2d s2_t2_t2_conv1(&net, 128, 3, 3, 1, 1, 1, 1, s2_t2_t2_conv1_bin, true); + tk::dnn::Activation s2_t2_t2_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s2_t2_t2_conv2(&net, 128, 3, 3, 1, 1, 1, 1, s2_t2_t2_conv2_bin, true); + + tk::dnn::Shortcut s2_t2_t2_s1(&net, last1); + tk::dnn::Activation s2_t2_t2_relu(&net, CUDNN_ACTIVATION_RELU); + last2 = &s2_t2_t2_relu; + + // root + // join last1 and net in single input 128, 56, 56 + tk::dnn::Layer *route_s2_t2_root_layers[4] = { last2, last1, last4, last3}; + tk::dnn::Route route_s2_t2_root(&net, route_s2_t2_root_layers, 4); + tk::dnn::Conv2d s2_t2_root_conv1(&net, 128, 1, 1, 1, 1, 0, 0, s2_t2_root_conv1_bin, true); + tk::dnn::Activation s2_t2_root_relu(&net, CUDNN_ACTIVATION_RELU); + + base4 = &s2_t2_root_relu; + + // level 4 + // tree 1 + // tree 1 + tk::dnn::Conv2d s3_t1_t1_conv1(&net, 256, 3, 3, 2, 2, 1, 1, s3_t1_t1_conv1_bin, true); + tk::dnn::Activation s3_t1_t1_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s3_t1_t1_conv2(&net, 256, 3, 3, 1, 1, 1, 1, s3_t1_t1_conv2_bin, true); + last2 = &s3_t1_t1_conv2; + + // get the basicblock input and apply maxpool conv2d and relu + tk::dnn::Layer *route_s3_t1_t1_layers[1] = { base4 }; + tk::dnn::Route route_s3_t1_t1(&net, route_s3_t1_t1_layers, 1); + // downsample + tk::dnn::Pooling s3_t1_t1_maxpool1(&net, 2, 2, 2, 2, 0, 0, tk::dnn::POOLING_MAX); + last4 = &s3_t1_t1_maxpool1; + // project + tk::dnn::Conv2d s3_t1_t1_residual1_conv1(&net, 256, 1, 1, 1, 1, 0, 0, s3_t1_t1_project, true); + + tk::dnn::Shortcut s3_t1_t1_s1(&net, last2); + tk::dnn::Activation s3_t1_t1_relu(&net, CUDNN_ACTIVATION_RELU); + + last1 = &s3_t1_t1_relu; + + // tree 2 + tk::dnn::Conv2d s3_t1_t2_conv1(&net, 256, 3, 3, 1, 1, 1, 1, s3_t1_t2_conv1_bin, true); + tk::dnn::Activation s3_t1_t2_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s3_t1_t2_conv2(&net, 256, 3, 3, 1, 1, 1, 1, s3_t1_t2_conv2_bin, true); + + tk::dnn::Shortcut s3_t1_t2_s1(&net, last1); + tk::dnn::Activation s3_t1_t2_relu(&net, CUDNN_ACTIVATION_RELU); + last2 = &s3_t1_t2_relu; + + // root + // join last1 and net in single input 256, 56, 56 + tk::dnn::Layer *route_s3_t1_root_layers[2] = { last2, last1 }; + tk::dnn::Route route_s3_t1_root(&net, route_s3_t1_root_layers, 2); + tk::dnn::Conv2d s3_t1_root_conv1(&net, 256, 1, 1, 1, 1, 0, 0, s3_t1_root_conv1_bin, true); + tk::dnn::Activation s3_t1_root_relu(&net, CUDNN_ACTIVATION_RELU); + + last1 = &s3_t1_root_relu; + last3 = &s3_t1_root_relu; + // tree 2 + // tree 1 + tk::dnn::Conv2d s3_t2_t1_conv1(&net, 256, 3, 3, 1, 1, 1, 1, s3_t2_t1_conv1_bin, true); + tk::dnn::Activation s3_t2_t1_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s3_t2_t1_conv2(&net, 256, 3, 3, 1, 1, 1, 1, s3_t2_t1_conv2_bin, true); + tk::dnn::Shortcut s3_t2_t1_s1(&net, last1); + tk::dnn::Activation s3_t2_t1_relu(&net, CUDNN_ACTIVATION_RELU); + + last1 = &s3_t2_t1_relu; + + // tree 2 + tk::dnn::Conv2d s3_t2_t2_conv1(&net, 256, 3, 3, 1, 1, 1, 1, s3_t2_t2_conv1_bin, true); + tk::dnn::Activation s3_t2_t2_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s3_t2_t2_conv2(&net, 256, 3, 3, 1, 1, 1, 1, s3_t2_t2_conv2_bin, true); + + tk::dnn::Shortcut s3_t2_t2_s1(&net, last1); + tk::dnn::Activation s3_t2_t2_relu(&net, CUDNN_ACTIVATION_RELU); + last2 = &s3_t2_t2_relu; + + // root + // join last1 and net in single input 256, 56, 56 + tk::dnn::Layer *route_s3_t2_root_layers[4] = { last2, last1, last4, last3}; + tk::dnn::Route route_s3_t2_root(&net, route_s3_t2_root_layers, 4); + tk::dnn::Conv2d s3_t2_root_conv1(&net, 256, 1, 1, 1, 1, 0, 0, s3_t2_root_conv1_bin, true); + tk::dnn::Activation s3_t2_root_relu(&net, CUDNN_ACTIVATION_RELU); + + base5 = &s3_t2_root_relu; + + // level 5 + // tree 1 + tk::dnn::Conv2d s4_t1_conv1(&net, 512, 3, 3, 2, 2, 1, 1, s4_t1_conv1_bin, true); + tk::dnn::Activation s4_t1_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s4_t1_conv2(&net, 512, 3, 3, 1, 1, 1, 1, s4_t1_conv2_bin, true); + last2 = &s4_t1_conv2; + + // get the basicblock input and apply maxpool conv2d and relu + tk::dnn::Layer *route_s4_t1_layers[1] = { base5 }; + tk::dnn::Route route_s4_t1(&net, route_s4_t1_layers, 1); + // downsample + tk::dnn::Pooling s4_t1_maxpool1(&net, 2, 2, 2, 2, 0, 0, tk::dnn::POOLING_MAX); + last4 = &s4_t1_maxpool1; + // project + tk::dnn::Conv2d s4_t1_residual1_conv1(&net, 512, 1, 1, 1, 1, 0, 0, s4_t1_project, true); + + tk::dnn::Shortcut s4_t1_s1(&net, last2); + tk::dnn::Activation s4_t1_relu(&net, CUDNN_ACTIVATION_RELU); + + last1 = &s4_t1_relu; + + // tree 2 + tk::dnn::Conv2d s4_t2_conv1(&net, 512, 3, 3, 1, 1, 1, 1, s4_t2_conv1_bin, true); + tk::dnn::Activation s4_t2_relu1(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Conv2d s4_t2_conv2(&net, 512, 3, 3, 1, 1, 1, 1, s4_t2_conv2_bin, true); + + tk::dnn::Shortcut s4_t2_s1(&net, last1); + tk::dnn::Activation s4_t2_relu(&net, CUDNN_ACTIVATION_RELU); + last2 = &s4_t2_relu; + + // root + // join last1 and net in single input 128, 56, 56 + tk::dnn::Layer *route_s4_root_layers[3] = { last2, last1, last4 }; + tk::dnn::Route route_s4_root(&net, route_s4_root_layers, 3); + tk::dnn::Conv2d s4_root_conv1(&net, 512, 1, 1, 1, 1, 0, 0, s4_root_conv1_bin, true); + tk::dnn::Activation s4_root_relu(&net, CUDNN_ACTIVATION_RELU); + + base6 = &s4_root_relu; + + //final + // tk::dnn::Pooling avgpool(&net, 7, 7, 7, 7, 0, 0, tk::dnn::POOLING_AVERAGE); + // tk::dnn::Dense fc(&net, 1000, fc_bin); + + //ida 0 + tk::dnn::DeformConv2d ida_0_p_1_dcn(&net, 256, 1, 3, 3, 1, 1, 1, 1, ida_0_p_1_dcn_bin, ida_0_p_1_conv_bin, true); + tk::dnn::Activation ida_0_p_1_relu(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d ida_0_up_1_deconv(&net, 256, 4, 4, 2, 2, 1, 1, ida_0_up_1_deconv_bin, false, 256); + tk::dnn::Shortcut ida_0_shortcut(&net, base5); + tk::dnn::DeformConv2d ida_0_n_1_dcn(&net, 256, 1, 3, 3, 1, 1, 1, 1, ida_0_n_1_dcn_bin, ida_0_n_1_conv_bin, true); + tk::dnn::Activation ida_0_n_1_relu(&net, CUDNN_ACTIVATION_RELU); + ida1 = &ida_0_n_1_relu; + + //ida1-1 + tk::dnn::Layer *route_ida1_layers_1[1] = { base5 }; + tk::dnn::Route route_ida1_1(&net, route_ida1_layers_1, 1); + + tk::dnn::DeformConv2d ida_1_p_1_dcn(&net, 128, 1, 3, 3, 1, 1, 1, 1, ida_1_p_1_dcn_bin, ida_1_p_1_conv_bin, true); + tk::dnn::Activation ida_1_p_1_relu(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d ida_1_up_1_deconv(&net, 128, 4, 4, 2, 2, 1, 1, ida_1_up_1_deconv_bin, false, 128); + tk::dnn::Shortcut ida_1_shortcut1(&net, base4); + tk::dnn::DeformConv2d ida_1_n_1_dcn(&net, 128, 1, 3, 3, 1, 1, 1, 1, ida_1_n_1_dcn_bin, ida_1_n_1_conv_bin, true); + tk::dnn::Activation ida_1_n_1_relu(&net, CUDNN_ACTIVATION_RELU); + ida2_1 = &ida_1_n_1_relu; + + //ida1-2 + tk::dnn::Layer *route_ida1_layers_2[1] = { ida1 }; + tk::dnn::Route route_ida1_2(&net, route_ida1_layers_2, 1); + + tk::dnn::DeformConv2d ida_1_p_2_dcn(&net, 128, 1, 3, 3, 1, 1, 1, 1, ida_1_p_2_dcn_bin, ida_1_p_2_conv_bin, true); + tk::dnn::Activation ida_1_p_2_relu(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d ida_1_up_2_deconv(&net, 128, 4, 4, 2, 2, 1, 1, ida_1_up_2_deconv_bin, false, 128); + tk::dnn::Shortcut ida_1_shortcut2(&net, ida2_1); + tk::dnn::DeformConv2d ida_1_n_2_dcn(&net, 128, 1, 3, 3, 1, 1, 1, 1, ida_1_n_2_dcn_bin, ida_1_n_2_conv_bin, true); + tk::dnn::Activation ida_1_n_2_relu(&net, CUDNN_ACTIVATION_RELU); + ida2_2 = &ida_1_n_2_relu; + + //ida2-1 + tk::dnn::Layer *route_ida2_layers_1[1] = { base4 }; + tk::dnn::Route route_ida2_1(&net, route_ida2_layers_1, 1); + + tk::dnn::DeformConv2d ida_2_p_1_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_2_p_1_dcn_bin, ida_2_p_1_conv_bin, true); + tk::dnn::Activation ida_2_p_1_relu(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d ida_2_up_1_deconv(&net, 64, 4, 4, 2, 2, 1, 1, ida_2_up_1_deconv_bin, false, 64); + tk::dnn::Shortcut ida_2_shortcut1(&net, base3); + tk::dnn::DeformConv2d ida_2_n_1_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_2_n_1_dcn_bin, ida_2_n_1_conv_bin, true); + tk::dnn::Activation ida_2_n_1_relu(&net, CUDNN_ACTIVATION_RELU); + ida3_1 = &ida_2_n_1_relu; + + //ida2-2 + tk::dnn::Layer *route_ida2_layers_2[1] = { ida2_1 }; + tk::dnn::Route route_ida2_2(&net, route_ida2_layers_2, 1); + + tk::dnn::DeformConv2d ida_2_p_2_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_2_p_2_dcn_bin, ida_2_p_2_conv_bin, true); + tk::dnn::Activation ida_2_p_2_relu(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d ida_2_up_2_deconv(&net, 64, 4, 4, 2, 2, 1, 1, ida_2_up_2_deconv_bin, false, 64); + tk::dnn::Shortcut ida_2_shortcut2(&net, ida3_1); + tk::dnn::DeformConv2d ida_2_n_2_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_2_n_2_dcn_bin, ida_2_n_2_conv_bin, true); + tk::dnn::Activation ida_2_n_2_relu(&net, CUDNN_ACTIVATION_RELU); + ida3_2 = &ida_2_n_2_relu; + + //ida2-3 + tk::dnn::Layer *route_ida2_layers_3[1] = { ida2_2 }; + tk::dnn::Route route_ida2_3(&net, route_ida2_layers_3, 1); + + tk::dnn::DeformConv2d ida_2_p_3_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_2_p_3_dcn_bin, ida_2_p_3_conv_bin, true); + tk::dnn::Activation ida_2_p_3_relu(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d ida_2_up_3_deconv(&net, 64, 4, 4, 2, 2, 1, 1, ida_2_up_3_deconv_bin, false, 64); + tk::dnn::Shortcut ida_2_shortcut3(&net, ida3_2); + tk::dnn::DeformConv2d ida_2_n_3_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_2_n_3_dcn_bin, ida_2_n_3_conv_bin, true); + tk::dnn::Activation ida_2_n_3_relu(&net, CUDNN_ACTIVATION_RELU); + ida3_3 = &ida_2_n_3_relu; + + //idaup-1 + tk::dnn::Layer *route_idaup_layers_1[1] = { ida2_2 }; + tk::dnn::Route route_idaup_1(&net, route_idaup_layers_1, 1); + + tk::dnn::DeformConv2d idaup_p_1_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_up_p_1_dcn_bin, ida_up_p_1_conv_bin, true); + tk::dnn::Activation idaup_p_1_relu(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d idaup_up_1_deconv(&net, 64, 4, 4, 2, 2, 1, 1, ida_up_up_1_deconv_bin, false, 64); + tk::dnn::Shortcut idaup_shortcut1(&net, ida3_3); + tk::dnn::DeformConv2d idaup_n_1_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_up_n_1_dcn_bin, ida_up_n_1_conv_bin, true); + tk::dnn::Activation idaup_n_1_relu(&net, CUDNN_ACTIVATION_RELU); + idaup_1 = &idaup_n_1_relu; + + //idaup-2 + tk::dnn::Layer *route_idaup_layers_2[1] = { ida1 }; + tk::dnn::Route route_idaup_2(&net, route_idaup_layers_2, 1); + + tk::dnn::DeformConv2d idaup_p_2_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_up_p_2_dcn_bin, ida_up_p_2_conv_bin, true); + tk::dnn::Activation idaup_p_2_relu(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d idaup_up_2_deconv(&net, 64, 8, 8, 4, 4, 2, 2, ida_up_up_2_deconv_bin, false, 64); + tk::dnn::Shortcut idaup_shortcut2(&net, idaup_1); + tk::dnn::DeformConv2d idaup_n_2_dcn(&net, 64, 1, 3, 3, 1, 1, 1, 1, ida_up_n_2_dcn_bin, ida_up_n_2_conv_bin, true); + tk::dnn::Activation idaup_n_2_relu(&net, CUDNN_ACTIVATION_RELU); + idaup_2 = &idaup_n_2_relu; + + tk::dnn::Layer *route_1_0_layers[1] = { idaup_2 }; + + // hm + tk::dnn::Conv2d *hm_conv1 = new tk::dnn::Conv2d(&net, 256, 3, 3, 1, 1, 1, 1, hm_conv1_bin, false); + tk::dnn::Activation *hm_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *hm = new tk::dnn::Conv2d(&net, 3, 1, 1, 1, 1, 0, 0, hm_conv2_bin, false); + hm->setFinal(); + int kernel = 3; + int pad = (kernel - 1)/2; + tk::dnn::Activation *hm_sig = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_SIGMOID); + tk::dnn::Pooling *hmax = new tk::dnn::Pooling(&net, kernel, kernel, 1, 1, pad, pad, tk::dnn::POOLING_MAX); + hmax->setFinal(); + + // wh + tk::dnn::Route *route_1_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *wh_conv1 = new tk::dnn::Conv2d(&net, 256, 3, 3, 1, 1, 1, 1, wh_conv1_bin, false); + tk::dnn::Activation *wh_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *wh = new tk::dnn::Conv2d(&net, 2, 1, 1, 1, 1, 0, 0, wh_conv2_bin, false); + wh->setFinal(); + + // reg + tk::dnn::Route *route_2_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *reg_conv1 = new tk::dnn::Conv2d(&net, 256, 3, 3, 1, 1, 1, 1, reg_conv1_bin, false); + tk::dnn::Activation *reg_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *reg = new tk::dnn::Conv2d(&net, 2, 1, 1, 1, 1, 0, 0, reg_conv2_bin, false); + reg->setFinal(); + + // dep + tk::dnn::Route *route_3_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *dep_conv1 = new tk::dnn::Conv2d(&net, 256, 3, 3, 1, 1, 1, 1, dep_conv1_bin, false); + tk::dnn::Activation *dep_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *dep = new tk::dnn::Conv2d(&net, 1, 1, 1, 1, 1, 0, 0, dep_conv2_bin, false); + dep->setFinal(); + + // rot + tk::dnn::Route *route_4_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *rot_conv1 = new tk::dnn::Conv2d(&net, 256, 3, 3, 1, 1, 1, 1, rot_conv1_bin, false); + tk::dnn::Activation *rot_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *rot = new tk::dnn::Conv2d(&net, 8, 1, 1, 1, 1, 0, 0, rot_conv2_bin, false); + rot->setFinal(); + + // dim + tk::dnn::Route *route_5_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *dim_conv1 = new tk::dnn::Conv2d(&net, 256, 3, 3, 1, 1, 1, 1, dim_conv1_bin, false); + tk::dnn::Activation *dim_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *dim_ = new tk::dnn::Conv2d(&net, 3, 1, 1, 1, 1, 0, 0, dim_conv2_bin, false); + dim_->setFinal(); + + // Load input + dnnType *data; + dnnType *input_h; + readBinaryFile(input_bin, dim.tot(), &input_h, &data); + //printDeviceVector(64, data, true); + + //print network model + net.print(); + + //convert network to tensorRT + tk::dnn::NetworkRT netRT(&net, net.getNetworkRTName("dla34_cnet3d")); + + tk::dnn::dataDim_t dim1 = dim; //input dim + printCenteredTitle(" CUDNN inference ", '=', 30); + { + dim1.print(); + TIMER_START + net.infer(dim1, data); + TIMER_STOP + dim1.print(); + } + + tk::dnn::dataDim_t dim2 = dim; + printCenteredTitle(" TENSORRT inference ", '=', 30); + { + dim2.print(); + TIMER_START + netRT.infer(dim2, data); + TIMER_STOP + dim2.print(); + } + + tk::dnn::Layer *outs[6] = { hm, wh, reg, dep, rot, dim_ }; + int out_count = 1; + int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0; + for(int i=0; i<6; i++) { + printCenteredTitle((std::string(" RESNET CHECK RESULTS ") + std::to_string(i) + " ").c_str(), '=', 30); + + outs[i]->output_dim.print(); + + dnnType *out, *out_h; + int odim = outs[i]->output_dim.tot(); + readBinaryFile(output_bin[i], odim, &out_h, &out); + + dnnType *cudnn_out, *rt_out; + cudnn_out = outs[i]->dstData; + rt_out = (dnnType *)netRT.buffersRT[i+out_count]; + // there is the maxpool. It isn't an output but it is necessary for the process section + if(i==0) + out_count ++; + + std::cout<<"CUDNN vs correct"; + ret_cudnn |= checkResult(odim, cudnn_out, out) == 0 ? 0: ERROR_CUDNN; + std::cout<<"TRT vs correct"; + ret_tensorrt |= checkResult(odim, rt_out, out) == 0 ? 0 : ERROR_TENSORRT; + std::cout<<"CUDNN vs TRT "; + ret_cudnn_tensorrt |= checkResult(odim, cudnn_out, rt_out) == 0 ? 0 : ERROR_CUDNNvsTENSORRT; + } + return ret_cudnn | ret_tensorrt | ret_cudnn_tensorrt; +} -- 2.52.0 From ba8c28238433922d6699abfd82ee19875bb180e5 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Wed, 27 May 2020 18:03:09 +0200 Subject: [PATCH 002/162] Add 3D CenterNet detection class Signed-off-by: Davide Sapienza --- include/tkDNN/CenternetDetection3D.h | 100 ++++++ include/tkDNN/DetectionNN3D.h | 136 +++++++ include/tkDNN/Layer.h | 10 + include/tkDNN/kernelsThrust.h | 2 + src/CenternetDetection3D.cpp | 517 +++++++++++++++++++++++++++ src/kernels/postprocessing.cu | 14 + 6 files changed, 779 insertions(+) create mode 100644 include/tkDNN/CenternetDetection3D.h create mode 100644 include/tkDNN/DetectionNN3D.h create mode 100644 src/CenternetDetection3D.cpp diff --git a/include/tkDNN/CenternetDetection3D.h b/include/tkDNN/CenternetDetection3D.h new file mode 100644 index 0000000..7b9ea7a --- /dev/null +++ b/include/tkDNN/CenternetDetection3D.h @@ -0,0 +1,100 @@ +#ifndef CENTERNETDETECTION3D_H +#define CENTERNETDETECTION3D_H + +#include "kernels.h" +#include +#include "opencv2/opencv.hpp" +#include +#include +#include // std::iota +#include // std::sort + +#include "DetectionNN3D.h" + +#include "kernelsThrust.h" + + +namespace tk { namespace dnn { + +class CenternetDetection3D : public DetectionNN3D +{ +private: + tk::dnn::dataDim_t dim; + tk::dnn::dataDim_t dim2; + tk::dnn::dataDim_t dim_hm; + tk::dnn::dataDim_t dim_wh; + tk::dnn::dataDim_t dim_reg; + tk::dnn::dataDim_t dim_dep; + tk::dnn::dataDim_t dim_rot; + tk::dnn::dataDim_t dim_dim; + float *topk_scores; + int *topk_inds_; + float *topk_ys_; + float *topk_xs_; + int *ids_d, *ids_; + + float *ones; + + float *scores, *scores_d; + int *clses, *clses_d; + int *topk_inds_d; + float *topk_ys_d; + float *topk_xs_d; + int *inttopk_xs_d, *inttopk_ys_d; + + float *xs, *ys; + + float *dep, *rot, *dim_, *wh; + float *dep_d, *rot_d, *dim_d, *wh_d; + + float *target_coords; + + #ifdef OPENCV_CUDACONTRIB + float *mean_d; + float *stddev_d; + #else + cv::Vec mean; + cv::Vec stddev; + dnnType *input; + #endif + cv::Mat r; + cv::Mat calibs; + float *d_ptrs; + + cv::Mat src; + cv::Mat dst; + cv::Mat dst2; + cv::Mat trans, trans2; + //processing + int K = 100; + int width = 128;//56; // TODO + + // pointer used in the kernels + float *src_out; + int *ids_out; + + struct threshold op; + float peakThreshold = 0.2; + float centerThreshold = 0.3; //default 0.5 + cv::Mat corners, pts3DHomo; + + std::vector detected3D; + std::vectorcls3D; + std::vector> face_id; + +public: + CenternetDetection3D() {}; + ~CenternetDetection3D() {}; + + bool init(const std::string& tensor_path, const int n_classes=3); + void preprocess(cv::Mat &frame); + void postprocess(); + cv::Mat draw(cv::Mat &frame); +}; + + +} // namespace dnn +} // namespace tk + + +#endif /*CENTERNETDETECTION_H*/ \ No newline at end of file diff --git a/include/tkDNN/DetectionNN3D.h b/include/tkDNN/DetectionNN3D.h new file mode 100644 index 0000000..111d7e2 --- /dev/null +++ b/include/tkDNN/DetectionNN3D.h @@ -0,0 +1,136 @@ +#ifndef DETECTIONNN3D_H +#define DETECTIONNN3D_H + +#include +#include +#include +#include +#include +#include "utils.h" + +#include +#include +#include + +#include "tkdnn.h" + +// #define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib. + +#ifdef OPENCV_CUDACONTRIB +#include +#include +#endif + + +namespace tk { namespace dnn { + +class DetectionNN3D { + + protected: + tk::dnn::NetworkRT *netRT = nullptr; + dnnType *input_d; + + cv::Size originalSize; + + cv::Scalar colors[256]; + +#ifdef OPENCV_CUDACONTRIB + cv::cuda::GpuMat bgr[3]; + cv::cuda::GpuMat imagePreproc; +#else + cv::Mat bgr[3]; + cv::Mat imagePreproc; + dnnType *input; +#endif + + /** + * This method preprocess the image, before feeding it to the NN. + * + * @param frame original frame to adapt for inference. + */ + virtual void preprocess(cv::Mat &frame) = 0; + + /** + * This method postprocess the output of the NN to obtain the correct + * boundig boxes. + * + */ + virtual void postprocess() = 0; + + public: + int classes = 0; + float confThreshold = 0.3; /*threshold on the confidence of the boxes*/ + + std::vector detected; /*bounding boxes in output*/ + std::vector stats; /*keeps track of inference times (ms)*/ + std::vector classesNames; + + DetectionNN3D() {}; + ~DetectionNN3D(){}; + + /** + * Method used to inialize the class, allocate memory and compute + * needed data. + * + * @param tensor_path path to the rt file og the NN. + * @param n_classes number of classes for the given dataset. + * @return true if everything is correct, false otherwise. + */ + virtual bool init(const std::string& tensor_path, const int n_classes=3) = 0; + + /** + * Method to draw boundixg boxes and labels on a frame. + * + * @param frame orginal frame to draw bounding box on. + * @return frame with boundig boxes. + */ + virtual cv::Mat draw(cv::Mat &frame){}; + + /** + * This method performs the whole detection of the NN. + * + * @param frame frame to run detection on. + * @param save_times if set to true, preprocess, inference and postprocess times + * are saved on a csv file, otherwise not. + * @param times pointer to the output stream where to write times + */ + void update(cv::Mat &frame, bool save_times=false, std::ofstream *times=nullptr){ + if(!frame.data) + FatalError("No image data feed to detection"); + + if(save_times && times==nullptr) + FatalError("save_times set to true, but no valid ofstream given"); + + originalSize = frame.size(); + printCenteredTitle(" TENSORRT detection ", '=', 30); + { + TIMER_START + preprocess(frame); + TIMER_STOP + if(save_times) *times<input_dim; + { + dim.print(); + TIMER_START + netRT->infer(dim, input_d); + TIMER_STOP + dim.print(); + stats.push_back(t_ns); + if(save_times) *times< corners; + float prob; + + void print() + { + std::cout<<"\tcl: "<input_dim; + + const char *kitti_class_name[] = { + "person", "car", "bicycle"}; + classesNames = std::vector(kitti_class_name, std::end( kitti_class_name)); + + for(int c=0; cinput_dim.tot())); + + dim_hm = tk::dnn::dataDim_t(1, 3, 128, 128, 1); + dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_reg = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_dep = tk::dnn::dataDim_t(1, 1, 128, 128, 1); + dim_rot = tk::dnn::dataDim_t(1, 8, 128, 128, 1); + dim_dim = tk::dnn::dataDim_t(1, 3, 128, 128, 1); + + checkCuda( cudaMalloc(&topk_scores, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&topk_inds_, dim_hm.c * K *sizeof(int)) ); + checkCuda( cudaMalloc(&topk_ys_, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&topk_xs_, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&ids_d, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) ); + checkCuda( cudaMallocHost(&ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) ); + for(int i =0; iinput_dim.tot())); + mean << 0.485, 0.456, 0.406; + stddev << 0.229, 0.224, 0.225; +#endif + + calibs = cv::Mat(cv::Size(4,3), CV_32F); + calibs.at(0,0) = 707.0493; + calibs.at(0,1) = 0.0; + calibs.at(0,2) = 604.0814; + calibs.at(0,3) = 45.75831; + calibs.at(1,0) = 0.0; + calibs.at(1,1) = 707.0493; + calibs.at(1,2) = 180.5066; + calibs.at(1,3) = -0.3454157; + calibs.at(2,0) = 0.0; + calibs.at(2,1) = 0.0; + calibs.at(2,2) = 1.0; + calibs.at(2,3) = 0.004981016; + + r = cv::Mat(cv::Size(3,3), CV_32F); + r.at(0,1) = 0.0; + r.at(1,0) = 0.0; + r.at(1,1) = 1.0; + r.at(1,2) = 0.0; + r.at(2,1) = 0.0; + + corners = cv::Mat(cv::Size(8,3), CV_32F); + corners.at(1,0) = 0.0; + corners.at(1,1) = 0.0; + corners.at(1,2) = 0.0; + corners.at(1,3) = 0.0; + + pts3DHomo = cv::Mat(cv::Size(8,4), CV_32F); + pts3DHomo.at(3,0) = 1.0; + pts3DHomo.at(3,1) = 1.0; + pts3DHomo.at(3,2) = 1.0; + pts3DHomo.at(3,3) = 1.0; + pts3DHomo.at(3,4) = 1.0; + pts3DHomo.at(3,5) = 1.0; + pts3DHomo.at(3,6) = 1.0; + pts3DHomo.at(3,7) = 1.0; + + checkCuda( cudaMalloc(&d_ptrs, dim.c * dim.h*dim.w * sizeof(float)) ); + + // Alloc array used in the kernel + checkCuda( cudaMalloc(&src_out, K *sizeof(float)) ); + checkCuda( cudaMalloc(&ids_out, K *sizeof(int)) ); + + dst2.at(0,0)=width * 0.5; + dst2.at(0,1)=width * 0.5; + dst2.at(1,0)=width * 0.5; + dst2.at(1,1)=width * 0.5 + width * -0.5; + + dst2.at(2,0)=dst2.at(1,0) + (-dst2.at(0,1)+dst2.at(1,1) ); + dst2.at(2,1)=dst2.at(1,1) + (dst2.at(0,0)-dst2.at(1,0) ); + + face_id.push_back({0,1,5,4}); + face_id.push_back({1,2,6, 5}); + face_id.push_back({2,3,7,6}); + face_id.push_back({3,0,4,7}); + // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); +} + +void CenternetDetection3D::preprocess(cv::Mat &frame){ + // -----------------------------------pre-process ------------------------------------------ + + // auto start_t = std::chrono::steady_clock::now(); + // auto step_t = std::chrono::steady_clock::now(); + // auto end_t = std::chrono::steady_clock::now(); + cv::Size sz = originalSize; + // std::cout<<"image: "< 0 + + src.at(0,0)=c[0]; + src.at(0,1)=c[1]; + src.at(1,0)=c[0]; + src.at(1,1)=c[1] + s[0] * -0.5; + dst.at(0,0)=netRT->input_dim.w * 0.5; + dst.at(0,1)=netRT->input_dim.h * 0.5; + dst.at(1,0)=netRT->input_dim.w * 0.5; + dst.at(1,1)=netRT->input_dim.h * 0.5 + netRT->input_dim.w * -0.5; + + src.at(2,0)=src.at(1,0) + (-src.at(0,1)+src.at(1,1) ); + src.at(2,1)=src.at(1,1) + (src.at(0,0)-src.at(1,0) ); + dst.at(2,0)=dst.at(1,0) + (-dst.at(0,1)+dst.at(1,1) ); + dst.at(2,1)=dst.at(1,1) + (dst.at(0,0)-dst.at(1,0) ); + + trans = cv::getAffineTransform( src, dst ); + // end_t = std::chrono::steady_clock::now(); + // std::cout << " TIME gett affine trans: " << std::chrono::duration_cast(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + + trans2 = cv::getAffineTransform( dst2, src ); + // end_t = std::chrono::steady_clock::now(); + // std::cout << " TIME getAffineTrans 2: " << std::chrono::duration_cast(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + } + sz_old = sz; +#ifdef OPENCV_CUDACONTRIB + std::cout<<"OPENCV CPMTROB\n"; + cv::cuda::GpuMat im_Orig; + cv::cuda::GpuMat imageF1_d, imageF2_d; + + im_Orig = cv::cuda::GpuMat(frame); + // cv::cuda::resize (im_Orig, imageF1_d, cv::Size(new_width, new_height)); + imageF1_d = im_Orig; + checkCuda( cudaDeviceSynchronize() ); + + sz = imageF1_d.size(); + // std::cout<<"size: "<(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + + cv::cuda::warpAffine(imageF1_d, imageF2_d, trans, cv::Size(netRT->input_dim.w, netRT->input_dim.h), cv::INTER_LINEAR ); + checkCuda( cudaDeviceSynchronize() ); + + imageF2_d.convertTo(imageF1_d, CV_32FC3, 1/255.0); + checkCuda( cudaDeviceSynchronize() ); + // end_t = std::chrono::steady_clock::now(); + // std::cout << " TIME convert: " << std::chrono::duration_cast(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + + dim2 = dim; + cv::cuda::GpuMat bgr[3]; + cv::cuda::split(imageF1_d,bgr);//split source + // end_t = std::chrono::steady_clock::now(); + // std::cout << " TIME split: " << std::chrono::duration_cast(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + + for(int i=0; i(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + + checkCuda(cudaMemcpy(input_d, d_ptrs, dim2.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice)); + + // end_t = std::chrono::steady_clock::now(); + // std::cout << " TIME Memcpy to input_d: " << std::chrono::duration_cast(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; +#else + std::cout<<"NO OPENCV CPMTROB\n"; + cv::Mat imageF; + // resize(frame, imageF, cv::Size(new_width, new_height)); + imageF = frame; + sz = imageF.size(); + // std::cout<<"size: "<(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + + cv::Mat trans = cv::getAffineTransform( src, dst ); + cv::warpAffine(imageF, imageF, trans, cv::Size(netRT->input_dim.w, netRT->input_dim.h), cv::INTER_LINEAR ); + // end_t = std::chrono::steady_clock::now(); + // std::cout << " TIME warpAffine: " << std::chrono::duration_cast(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + + sz = imageF.size(); + // std::cout<<"size: "<(end_t - step_t).count() << " us" << std::endl; + // step_t = end_t; + dim2 = dim; + //split channels + cv::Mat bgr[3]; + cv::split(imageF,bgr);//split source + for(int i=0; i<3; i++){ + bgr[i] = bgr[i] - mean[i]; + bgr[i] = bgr[i] / stddev[i]; + } + + //write channels + for(int i=0; ibuffersRT[1]; + rt_out[1] = (dnnType *)netRT->buffersRT[2]; + rt_out[2] = (dnnType *)netRT->buffersRT[3]; + rt_out[3] = (dnnType *)netRT->buffersRT[4]; + rt_out[4] = (dnnType *)netRT->buffersRT[5]; + rt_out[5] = (dnnType *)netRT->buffersRT[6]; + rt_out[6] = (dnnType *)netRT->buffersRT[7]; + + // ------------------------------------ process -------------------------------------------- + activationSIGMOIDForward(rt_out[0], rt_out[0], dim_hm.tot()); + checkCuda( cudaDeviceSynchronize() ); + + // output['dep'] = 1. / (output['dep'].sigmoid() + 1e-6) - 1. + activationSIGMOIDForward(rt_out[4], rt_out[4], dim_dep.tot()); + checkCuda( cudaDeviceSynchronize() ); + transformDep(ones, ones + dim_dep.tot(), rt_out[4], rt_out[4] + dim_dep.tot()); + checkCuda( cudaDeviceSynchronize() ); + + subtractWithThreshold(rt_out[0], rt_out[0] + dim_hm.tot(), rt_out[1], rt_out[0], op); + + // ----------- nms end + // ----------- topk + + if(K > dim_hm.h * dim_hm.w){ + printf ("Error topk (K is too large)\n"); + return; + } + + checkCuda( cudaMemcpy(ids_d, ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyHostToDevice) ); + + sort(rt_out[0],rt_out[0]+dim_hm.tot(),ids_d); + checkCuda( cudaDeviceSynchronize() ); + + topk(rt_out[0], ids_d, K, scores_d, topk_inds_d, topk_ys_d, topk_xs_d); + checkCuda( cudaDeviceSynchronize() ); + + checkCuda( cudaMemcpy(scores, scores_d, K *sizeof(float), cudaMemcpyDeviceToHost) ); + + topKxyclasses(topk_inds_d, topk_inds_d+K, K, width, dim_hm.w*dim_hm.h, clses_d, inttopk_xs_d, inttopk_ys_d); + + checkCuda( cudaMemcpy(topk_xs_d, (float *)inttopk_xs_d, K*sizeof(float), cudaMemcpyDeviceToDevice) ); + checkCuda( cudaMemcpy(topk_ys_d, (float *)inttopk_ys_d, K*sizeof(float), cudaMemcpyDeviceToDevice) ); + + checkCuda( cudaMemcpy(clses, clses_d, K*sizeof(int), cudaMemcpyDeviceToHost) ); + + // ----------- topk end + + topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3], src_out, ids_out); + // checkCuda( cudaDeviceSynchronize() ); + + + getRecordsFromTopKId(topk_inds_d, K, dim_dep.c, dim_dep.h * dim_dep.w, rt_out[4], dep_d, ids_out); + checkCuda( cudaMemcpy(dep, dep_d, K * dim_dep.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_rot.c, dim_rot.h * dim_rot.w, rt_out[5], rot_d, ids_out); + checkCuda( cudaMemcpy(rot, rot_d, K * dim_rot.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_dim.c, dim_dim.h * dim_dim.w, rt_out[6], dim_d, ids_out); + checkCuda( cudaMemcpy(dim_, dim_d, K * dim_dim.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_wh.c, dim_wh.h * dim_wh.w, rt_out[2], wh_d, ids_out); + checkCuda( cudaMemcpy(wh, wh_d, K * dim_wh.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + checkCuda( cudaMemcpy(xs, topk_xs_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(ys, topk_ys_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + + // ---------------------------------- post-process ----------------------------------------- + + // ddd_post_process_2d + cv::Mat new_pt1(cv::Size(1,2), CV_32F); + cv::Mat new_pt2(cv::Size(1,2), CV_32F); + + for(int i = 0; i(0,0)=static_cast(trans2.at(0,0))*xs[i] + + static_cast(trans2.at(0,1))*ys[i] + + static_cast(trans2.at(0,2))*1.0; + new_pt1.at(0,1)=static_cast(trans2.at(1,0))*xs[i] + + static_cast(trans2.at(1,1))*ys[i] + + static_cast(trans2.at(1,2))*1.0; + + new_pt2.at(0,0)=static_cast(trans2.at(0,0))*wh[i] + + static_cast(trans2.at(0,1))*wh[K+i] + + static_cast(trans2.at(0,2))*1.0; + new_pt2.at(0,1)=static_cast(trans2.at(1,0))*wh[i] + + static_cast(trans2.at(1,1))*wh[K+i] + + static_cast(trans2.at(1,2))*1.0; + + target_coords[i*4] = new_pt1.at(0,0); + target_coords[i*4+1] = new_pt1.at(0,1); + target_coords[i*4+2] = new_pt2.at(0,0); + target_coords[i*4+3] = new_pt2.at(0,1); + } + + float alpha; + float x, y, z, rot_y; + detected3D.clear(); + for(int i = 0; i rot[5*K + j]) + alpha = std::atan2(rot[2*K + j], rot[3*K + j]) -0.5 * M_PI; + else + alpha = std::atan2(rot[6*K + j], rot[7*K + j]) +0.5 * M_PI; + + // unproject_2d_to_3d + z = dep[j] - calibs.at(2,3);// z = depth - P[2, 3] + x = (target_coords[j*4] * dep[j] - calibs.at(0,3) - calibs.at(0,2) * z) / calibs.at(0,0); + y = (target_coords[j*4+1] * dep[j] - calibs.at(1,3) - calibs.at(1,2) * z) / calibs.at(1,1) + (dim_[j] / 2); + // alpha2rot_y + rot_y = (alpha + std::atan2(target_coords[j*4] - calibs.at(0,2), calibs.at(0,0))); + if(rot_y>M_PI) + rot_y -= 2*M_PI; + if(rot_y peakThreshold) { + if(scores[j] > centerThreshold) { + if(z>0) { + // compute_box_3d + r.at(0,0) = std::cos(rot_y); + r.at(0,2) = std::sin(rot_y); + r.at(2,0) = -std::sin(rot_y); + r.at(2,2) = std::cos(rot_y); + + corners.at(0,0) = dim_[2*K+j]/2; + corners.at(0,1) = dim_[2*K+j]/2; + corners.at(0,2) = -dim_[2*K+j]/2; + corners.at(0,3) = -dim_[2*K+j]/2; + corners.at(0,4) = dim_[2*K+j]/2; + corners.at(0,5) = dim_[2*K+j]/2; + corners.at(0,6) = -dim_[2*K+j]/2; + corners.at(0,7) = -dim_[2*K+j]/2; + + corners.at(1,4) = -dim_[j]; + corners.at(1,5) = -dim_[j]; + corners.at(1,6) = -dim_[j]; + corners.at(1,7) = -dim_[j]; + + corners.at(2,0) = dim_[K+j]/2; + corners.at(2,1) = -dim_[K+j]/2; + corners.at(2,2) = -dim_[K+j]/2; + corners.at(2,3) = dim_[K+j]/2; + corners.at(2,4) = dim_[K+j]/2; + corners.at(2,5) = -dim_[K+j]/2; + corners.at(2,6) = -dim_[K+j]/2; + corners.at(2,7) = dim_[K+j]/2; + cv::Mat aus = r * corners; + + for(int k=0; k<8; k++) { + aus.at(0,k) += x; + aus.at(1,k) += y; + aus.at(2,k) += z; + } + // corners.copyTo(pts3DHomo(cv::Rect(0, 0, 8, 3))); + for(int k1=0; k1<3; k1++) { + for(int k2=0; k2<8; k2++) + pts3DHomo.at(k1,k2) = aus.at(k1,k2); + } + aus.release(); + aus = calibs * pts3DHomo; + + tk::dnn::box3D res; + for(int k=0; k<8; k++) { + res.corners.push_back(aus.at(0,k) / aus.at(2,k)); + res.corners.push_back(aus.at(1,k) / aus.at(2,k)); + } + res.cl = i; + res.prob = scores[j]; + res.print(); + detected3D.push_back(res); + } + } + } + } + } +} + +cv::Mat CenternetDetection3D::draw(cv::Mat &frame) { + tk::dnn::box3D b; + int x0, w, x1, y0, h, y1; + int objClass; + std::string det_class; + + int baseline = 0; + float font_scale = 0.5; + int thickness = 2; + + // draw dets + for(int i=0; i=0; ind_f--) { + for(int j=0; j<4; j++) { + cv::line(frame, cv::Point(b.corners.at(face_id.at(ind_f).at(j) * 2), + b.corners.at(face_id.at(ind_f).at(j) * 2 + 1)), + cv::Point(b.corners.at(face_id.at(ind_f).at((j+1)%4) * 2), + b.corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), + colors[b.cl], 2); + if(ind_f == 0) { + cv::line(frame, cv::Point(b.corners.at(face_id.at(ind_f).at(0) * 2), + b.corners.at(face_id.at(ind_f).at(0) * 2 + 1)), + cv::Point(b.corners.at(face_id.at(ind_f).at(2) * 2), + b.corners.at(face_id.at(ind_f).at(2) * 2 + 1)), colors[b.cl], 2); + cv::line(frame, cv::Point(b.corners.at(face_id.at(ind_f).at(1) * 2), + b.corners.at(face_id.at(ind_f).at(1) * 2 + 1)), + cv::Point(b.corners.at(face_id.at(ind_f).at(3) * 2), + b.corners.at(face_id.at(ind_f).at(3) * 2 + 1)), colors[b.cl], 2); + } + } + } + // draw label + cv::Size text_size = getTextSize(classesNames[b.cl], cv::FONT_HERSHEY_SIMPLEX, font_scale, thickness, &baseline); + cv::rectangle(frame, cv::Point(b.corners.at(face_id.at(0).at(0) * 2), + b.corners.at(face_id.at(0).at(0) * 2 + 1)), + cv::Point((b.corners.at(face_id.at(0).at(0) * 2) + text_size.width - 2), + (b.corners.at(face_id.at(0).at(0) * 2 + 1)) - text_size.height - 2), colors[b.cl], -1); + cv::putText(frame, classesNames[b.cl], cv::Point(b.corners.at(face_id.at(0).at(0) * 2), + b.corners.at(face_id.at(0).at(0) * 2 + 1) - (baseline / 2)), + cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), thickness); + } + return frame; +} + +}} + + diff --git a/src/kernels/postprocessing.cu b/src/kernels/postprocessing.cu index 3510200..88dbcc1 100644 --- a/src/kernels/postprocessing.cu +++ b/src/kernels/postprocessing.cu @@ -1,5 +1,11 @@ #include "kernelsThrust.h" +void transformDep(float *src_begin, float *src_end, float *dst_begin, float *dst_end) { + int e = exp(-6); + thrust::transform(thrust::device, dst_begin, dst_end, thrust::make_constant_iterator(e), dst_begin, thrust::plus()); + thrust::transform(thrust::device, src_begin, src_end, dst_begin, dst_begin, thrust::divides()); + thrust::transform(thrust::device, dst_begin, dst_end, thrust::make_constant_iterator(-1.0), dst_begin, thrust::plus()); +} void subtractWithThreshold(dnnType *src_begin, dnnType *src_end, dnnType *src2_begin, dnnType *src_out, struct threshold op){ thrust::transform(thrust::device, src_begin, src_end, src2_begin, src_out, op); @@ -51,6 +57,14 @@ void topKxyAddOffset(int * ids_begin, const int K, const int size, thrust::transform(thrust::device, intys_begin, intys_begin + K, src_out, ys_begin, thrust::plus()); } +void getRecordsFromTopKId(int * ids_begin, const int K, const int ch, const int size, dnnType *src_begin, float *src_out, int *ids_out) { + for(int i=0; i()); + thrust::gather(thrust::device, ids_out, ids_out + K, src_begin, src_out+i*K); + } +} + void bboxes(int * ids_begin, const int K, const int size, float *xs_begin, float *ys_begin, dnnType *src_begin, float *bbx0, float *bbx1, float *bby0, float *bby1, float *src_out, int *ids_out){ -- 2.52.0 From 6bdf47bae60acb200d40c33b9c6a85b2013b920f Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Wed, 27 May 2020 18:05:38 +0200 Subject: [PATCH 003/162] Add 3D demo program Signed-off-by: Davide Sapienza --- CMakeLists.txt | 3 ++ demo/demo/demo3D.cpp | 103 +++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 106 insertions(+) create mode 100644 demo/demo/demo3D.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index 1ae4289..0e75e3b 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -159,6 +159,9 @@ target_link_libraries(map_demo tkDNN) add_executable(demo demo/demo/demo.cpp) target_link_libraries(demo tkDNN) +add_executable(demo3D demo/demo/demo3D.cpp) +target_link_libraries(demo3D tkDNN) + #------------------------------------------------------------------------------- # Install #------------------------------------------------------------------------------- diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp new file mode 100644 index 0000000..e838395 --- /dev/null +++ b/demo/demo/demo3D.cpp @@ -0,0 +1,103 @@ +#include +#include +#include /* srand, rand */ +#include +#include + +#include "CenternetDetection3D.h" + +bool gRun; +bool SAVE_RESULT = false; + +void sig_handler(int signo) { + std::cout<<"request gateway stop\n"; + gRun = false; +} + +int main(int argc, char *argv[]) { + + std::cout<<"detection\n"; + signal(SIGINT, sig_handler); + + + std::string net = "dla34_cnet3d_fp32.rt"; + if(argc > 1) + net = argv[1]; + std::string input = "../demo/yolo_test.mp4"; + if(argc > 2) + input = argv[2]; + char ntype = 'c'; + if(argc > 3) + ntype = argv[3][0]; + int n_classes = 3; + if(argc > 4) + n_classes = atoi(argv[4]); + + tk::dnn::CenternetDetection3D cnet; + + tk::dnn::DetectionNN3D *detNN; + + switch(ntype) + { + case 'c': + detNN = &cnet; + break; + default: + FatalError("Network type not allowed (3rd parameter)\n"); + } + + detNN->init(net, n_classes); + + gRun = true; + + cv::VideoCapture cap(input); + if(!cap.isOpened()) + gRun = false; + else + std::cout<<"camera started\n"; + + cv::VideoWriter resultVideo; + if(SAVE_RESULT) { + int w = cap.get(cv::CAP_PROP_FRAME_WIDTH); + int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT); + resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h)); + } + + cv::Mat frame; + cv::Mat dnn_input; + cv::namedWindow("detection", cv::WINDOW_NORMAL); + + std::vector detected_bbox; + + while(gRun) { + cap >> frame; + if(!frame.data) { + break; + } + + // this will be resized to the net format + dnn_input = frame.clone(); + + //inference + detNN->update(dnn_input); + frame = detNN->draw(frame); + + cv::imshow("detection", frame); + cv::waitKey(1); + if(SAVE_RESULT) + resultVideo << frame; + } + + std::cout<<"detection end\n"; + double mean = 0; + + std::cout<stats.begin(), detNN->stats.end())<<" ms\n"; + std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())<<" ms\n"; + for(int i=0; istats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size(); + std::cout<<"Avg: "< Date: Wed, 27 May 2020 18:06:26 +0200 Subject: [PATCH 004/162] Update README Signed-off-by: Davide Sapienza --- README.md | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/README.md b/README.md index aeb3e02..d700d02 100644 --- a/README.md +++ b/README.md @@ -131,6 +131,19 @@ N.b. By default it is used FP32 inference ![demo](https://user-images.githubusercontent.com/11562617/72547657-540e7800-388d-11ea-83c6-49dfea2a0607.gif) +### Run the 3D demo + +To run the 3D object detection demo follow these steps (example with CenterNet based on DLA34): +``` +rm dla34_cnet3d_fp32.rt # be sure to delete(or move) old tensorRT files +./test_dla34_cnet3d # run the yolo test (is slow) +./demo3D dla34_cnet3d_fp32.rt ../demo/yolo_test.mp4 c +``` +The demo3D program takes the same parameters of the demo program: +``` +./demo +``` + ### FP16 inference To run the an object detection demo with FP16 inference follow these steps (example with yolov3): -- 2.52.0 From 7a677d5c10ed913593fefa38e6d4f3fc4065008e Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Fri, 29 May 2020 14:54:36 +0200 Subject: [PATCH 005/162] Add CenterNet based on Resnet101 for 3D, CUDNN and TensorRT work Signed-off-by: Davide Sapienza --- CMakeLists.txt | 3 + tests/resnet101_cnet3d/resnet101_cnet3d.cpp | 443 ++++++++++++++++++++ 2 files changed, 446 insertions(+) create mode 100644 tests/resnet101_cnet3d/resnet101_cnet3d.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index 0e75e3b..e5a2b26 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -136,6 +136,9 @@ target_link_libraries(test_bdd-csresnext50-panet-spp tkDNN) add_executable(test_resnet101_cnet tests/resnet101_cnet/resnet101_cnet.cpp) target_link_libraries(test_resnet101_cnet tkDNN) +add_executable(test_resnet101_cnet3d tests/resnet101_cnet3d/resnet101_cnet3d.cpp) +target_link_libraries(test_resnet101_cnet3d tkDNN) + add_executable(test_dla34 tests/dla34/dla34.cpp) target_link_libraries(test_dla34 tkDNN) diff --git a/tests/resnet101_cnet3d/resnet101_cnet3d.cpp b/tests/resnet101_cnet3d/resnet101_cnet3d.cpp new file mode 100644 index 0000000..0089a19 --- /dev/null +++ b/tests/resnet101_cnet3d/resnet101_cnet3d.cpp @@ -0,0 +1,443 @@ +#include + +#include "kernels.h" +#include "Yolo3Detection.h" +#include "tkdnn.h" +#include +#include // std::iota +#include // std::sort +// #include "utils.h" + +const char *input_bin = "resnet101_cnet3d/debug/input.bin"; +const char *conv1_bin = "resnet101_cnet3d/layers/conv1.bin"; + +//layer1 +const char *layer1_bin[]={ +"resnet101_cnet3d/layers/layer1-0-conv1.bin", +"resnet101_cnet3d/layers/layer1-0-conv2.bin", +"resnet101_cnet3d/layers/layer1-0-conv3.bin", +"resnet101_cnet3d/layers/layer1-0-downsample-0.bin", + +"resnet101_cnet3d/layers/layer1-1-conv1.bin", +"resnet101_cnet3d/layers/layer1-1-conv2.bin", +"resnet101_cnet3d/layers/layer1-1-conv3.bin", + +"resnet101_cnet3d/layers/layer1-2-conv1.bin", +"resnet101_cnet3d/layers/layer1-2-conv2.bin", +"resnet101_cnet3d/layers/layer1-2-conv3.bin"}; + + +//layer2 +const char *layer2_bin[]={ +"resnet101_cnet3d/layers/layer2-0-conv1.bin", +"resnet101_cnet3d/layers/layer2-0-conv2.bin", +"resnet101_cnet3d/layers/layer2-0-conv3.bin", +"resnet101_cnet3d/layers/layer2-0-downsample-0.bin", + +"resnet101_cnet3d/layers/layer2-1-conv1.bin", +"resnet101_cnet3d/layers/layer2-1-conv2.bin", +"resnet101_cnet3d/layers/layer2-1-conv3.bin", + +"resnet101_cnet3d/layers/layer2-2-conv1.bin", +"resnet101_cnet3d/layers/layer2-2-conv2.bin", +"resnet101_cnet3d/layers/layer2-2-conv3.bin", + +"resnet101_cnet3d/layers/layer2-3-conv1.bin", +"resnet101_cnet3d/layers/layer2-3-conv2.bin", +"resnet101_cnet3d/layers/layer2-3-conv3.bin" +}; +//layer3 +const char *layer3_bin[]={ +"resnet101_cnet3d/layers/layer3-0-conv1.bin", +"resnet101_cnet3d/layers/layer3-0-conv2.bin", +"resnet101_cnet3d/layers/layer3-0-conv3.bin", +"resnet101_cnet3d/layers/layer3-0-downsample-0.bin", + +"resnet101_cnet3d/layers/layer3-1-conv1.bin", +"resnet101_cnet3d/layers/layer3-1-conv2.bin", +"resnet101_cnet3d/layers/layer3-1-conv3.bin", + +"resnet101_cnet3d/layers/layer3-2-conv1.bin", +"resnet101_cnet3d/layers/layer3-2-conv2.bin", +"resnet101_cnet3d/layers/layer3-2-conv3.bin", + +"resnet101_cnet3d/layers/layer3-3-conv1.bin", +"resnet101_cnet3d/layers/layer3-3-conv2.bin", +"resnet101_cnet3d/layers/layer3-3-conv3.bin", + +"resnet101_cnet3d/layers/layer3-4-conv1.bin", +"resnet101_cnet3d/layers/layer3-4-conv2.bin", +"resnet101_cnet3d/layers/layer3-4-conv3.bin", + +"resnet101_cnet3d/layers/layer3-5-conv1.bin", +"resnet101_cnet3d/layers/layer3-5-conv2.bin", +"resnet101_cnet3d/layers/layer3-5-conv3.bin", + +"resnet101_cnet3d/layers/layer3-6-conv1.bin", +"resnet101_cnet3d/layers/layer3-6-conv2.bin", +"resnet101_cnet3d/layers/layer3-6-conv3.bin", + +"resnet101_cnet3d/layers/layer3-7-conv1.bin", +"resnet101_cnet3d/layers/layer3-7-conv2.bin", +"resnet101_cnet3d/layers/layer3-7-conv3.bin", + +"resnet101_cnet3d/layers/layer3-8-conv1.bin", +"resnet101_cnet3d/layers/layer3-8-conv2.bin", +"resnet101_cnet3d/layers/layer3-8-conv3.bin", + +"resnet101_cnet3d/layers/layer3-9-conv1.bin", +"resnet101_cnet3d/layers/layer3-9-conv2.bin", +"resnet101_cnet3d/layers/layer3-9-conv3.bin", + +"resnet101_cnet3d/layers/layer3-10-conv1.bin", +"resnet101_cnet3d/layers/layer3-10-conv2.bin", +"resnet101_cnet3d/layers/layer3-10-conv3.bin", + +"resnet101_cnet3d/layers/layer3-11-conv1.bin", +"resnet101_cnet3d/layers/layer3-11-conv2.bin", +"resnet101_cnet3d/layers/layer3-11-conv3.bin", + +"resnet101_cnet3d/layers/layer3-12-conv1.bin", +"resnet101_cnet3d/layers/layer3-12-conv2.bin", +"resnet101_cnet3d/layers/layer3-12-conv3.bin", + +"resnet101_cnet3d/layers/layer3-13-conv1.bin", +"resnet101_cnet3d/layers/layer3-13-conv2.bin", +"resnet101_cnet3d/layers/layer3-13-conv3.bin", + +"resnet101_cnet3d/layers/layer3-14-conv1.bin", +"resnet101_cnet3d/layers/layer3-14-conv2.bin", +"resnet101_cnet3d/layers/layer3-14-conv3.bin", + +"resnet101_cnet3d/layers/layer3-15-conv1.bin", +"resnet101_cnet3d/layers/layer3-15-conv2.bin", +"resnet101_cnet3d/layers/layer3-15-conv3.bin", + +"resnet101_cnet3d/layers/layer3-16-conv1.bin", +"resnet101_cnet3d/layers/layer3-16-conv2.bin", +"resnet101_cnet3d/layers/layer3-16-conv3.bin", + +"resnet101_cnet3d/layers/layer3-17-conv1.bin", +"resnet101_cnet3d/layers/layer3-17-conv2.bin", +"resnet101_cnet3d/layers/layer3-17-conv3.bin", + +"resnet101_cnet3d/layers/layer3-18-conv1.bin", +"resnet101_cnet3d/layers/layer3-18-conv2.bin", +"resnet101_cnet3d/layers/layer3-18-conv3.bin", + +"resnet101_cnet3d/layers/layer3-19-conv1.bin", +"resnet101_cnet3d/layers/layer3-19-conv2.bin", +"resnet101_cnet3d/layers/layer3-19-conv3.bin", + +"resnet101_cnet3d/layers/layer3-20-conv1.bin", +"resnet101_cnet3d/layers/layer3-20-conv2.bin", +"resnet101_cnet3d/layers/layer3-20-conv3.bin", + +"resnet101_cnet3d/layers/layer3-21-conv1.bin", +"resnet101_cnet3d/layers/layer3-21-conv2.bin", +"resnet101_cnet3d/layers/layer3-21-conv3.bin", + +"resnet101_cnet3d/layers/layer3-22-conv1.bin", +"resnet101_cnet3d/layers/layer3-22-conv2.bin", +"resnet101_cnet3d/layers/layer3-22-conv3.bin"}; + + +//layer4 +const char *layer4_bin[]={ +"resnet101_cnet3d/layers/layer4-0-conv1.bin", +"resnet101_cnet3d/layers/layer4-0-conv2.bin", +"resnet101_cnet3d/layers/layer4-0-conv3.bin", +"resnet101_cnet3d/layers/layer4-0-downsample-0.bin", + +"resnet101_cnet3d/layers/layer4-1-conv1.bin", +"resnet101_cnet3d/layers/layer4-1-conv2.bin", +"resnet101_cnet3d/layers/layer4-1-conv3.bin", + +"resnet101_cnet3d/layers/layer4-2-conv1.bin", +"resnet101_cnet3d/layers/layer4-2-conv2.bin", +"resnet101_cnet3d/layers/layer4-2-conv3.bin"}; + +const char *d_conv1_bin = "resnet101_cnet3d/layers/deconv_layers-0-conv_offset_mask.bin"; +const char *deform1_bin = "resnet101_cnet3d/layers/deconv_layers-0.bin"; +const char *deconv1_bin = "resnet101_cnet3d/layers/deconv_layers-3.bin"; + +const char *d_conv2_bin = "resnet101_cnet3d/layers/deconv_layers-6-conv_offset_mask.bin"; +const char *deform2_bin = "resnet101_cnet3d/layers/deconv_layers-6.bin"; +const char *deconv2_bin = "resnet101_cnet3d/layers/deconv_layers-9.bin"; + +const char *d_conv3_bin = "resnet101_cnet3d/layers/deconv_layers-12-conv_offset_mask.bin"; +const char *deform3_bin = "resnet101_cnet3d/layers/deconv_layers-12.bin"; +const char *deconv3_bin = "resnet101_cnet3d/layers/deconv_layers-15.bin"; + +const char *hm_conv1_bin = "resnet101_cnet3d/layers/hm-0.bin"; +const char *hm_conv2_bin = "resnet101_cnet3d/layers/hm-2.bin"; +const char *wh_conv1_bin = "resnet101_cnet3d/layers/wh-0.bin"; +const char *wh_conv2_bin = "resnet101_cnet3d/layers/wh-2.bin"; +const char *reg_conv1_bin = "resnet101_cnet3d/layers/reg-0.bin"; +const char *reg_conv2_bin = "resnet101_cnet3d/layers/reg-2.bin"; +const char *dep_conv1_bin = "resnet101_cnet3d/layers/dep-0.bin"; +const char *dep_conv2_bin = "resnet101_cnet3d/layers/dep-2.bin"; +const char *rot_conv1_bin = "resnet101_cnet3d/layers/rot-0.bin"; +const char *rot_conv2_bin = "resnet101_cnet3d/layers/rot-2.bin"; +const char *dim_conv1_bin = "resnet101_cnet3d/layers/dim-0.bin"; +const char *dim_conv2_bin = "resnet101_cnet3d/layers/dim-2.bin"; +//final +const char *fc_bin = "resnet101_cnet3d/layers/fc.bin"; + +const char *output_bin[]={ +"resnet101_cnet3d/debug/hm.bin", +"resnet101_cnet3d/debug/wh.bin", +"resnet101_cnet3d/debug/reg.bin", +"resnet101_cnet3d/debug/dep.bin", +"resnet101_cnet3d/debug/rot.bin", +"resnet101_cnet3d/debug/dim.bin"}; + +int main() +{ + // downloadWeightsifDoNotExist(input_bin, "resnet101_cnet3d", "https://cloud.hipert.unimore.it/s/5BTjHMWBcJk8g3i/download"); + + // Network layout + tk::dnn::dataDim_t dim(1, 3, 512, 512, 1); + tk::dnn::Network net(dim); + + tk::dnn::Conv2d conv1(&net, 64, 7, 7, 2, 2, 3, 3, conv1_bin, true); + tk::dnn::Activation relu3(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Pooling maxpool4(&net, 3, 3, 2, 2, 1, 1, tk::dnn::POOLING_MAX); + + + //layer 1 + int id_layer1_bin = 0; + tk::dnn::Layer *last = &maxpool4; + for(int i=0; i<3;i++) + { + tk::dnn::Conv2d *layer1_0_conv1 = new tk::dnn::Conv2d(&net, 64, 1, 1, 1, 1, 0, 0, layer1_bin[id_layer1_bin++], true); + tk::dnn::Activation *relu1_0_1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *layer1_0_conv2 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, layer1_bin[id_layer1_bin++], true); + tk::dnn::Activation *relu1_0_2 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *layer1_0_conv3 = new tk::dnn::Conv2d(&net, 256, 1, 1, 1, 1, 0, 0, layer1_bin[id_layer1_bin++], true); + if(i==0) { + tk::dnn::Layer *route_1_0_layers[1] = { last }; + tk::dnn::Route *route_1_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *layer1_0_downsample_0 = new tk::dnn::Conv2d(&net, 256, 1, 1, 1, 1, 0, 0, layer1_bin[id_layer1_bin++], true); + tk::dnn::Shortcut *s1_0 = new tk::dnn::Shortcut(&net, layer1_0_conv3); + } else { + tk::dnn::Shortcut *s1_0 = new tk::dnn::Shortcut(&net, last); + } + tk::dnn::Activation *layer1_0_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + last = layer1_0_relu; + } + + // layer 2 + int id_layer2_bin = 0; + for(int i=0; i<4;i++) + { + tk::dnn::Conv2d *layer1_0_conv1 = new tk::dnn::Conv2d(&net, 128, 1, 1, 1, 1, 0, 0, layer2_bin[id_layer2_bin++], true); + tk::dnn::Activation *relu1_0_1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *layer1_0_conv2; + if(i==0) + layer1_0_conv2 = new tk::dnn::Conv2d(&net, 128, 3, 3, 2, 2, 1, 1, layer2_bin[id_layer2_bin++], true); + else + layer1_0_conv2 = new tk::dnn::Conv2d(&net, 128, 3, 3, 1, 1, 1, 1, layer2_bin[id_layer2_bin++], true); + + tk::dnn::Activation *relu1_0_2 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *layer1_0_conv3 = new tk::dnn::Conv2d(&net, 512, 1, 1, 1, 1, 0, 0, layer2_bin[id_layer2_bin++], true); + if(i==0) + { + tk::dnn::Layer *route_1_0_layers[1] = { last }; + tk::dnn::Route *route_1_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *layer1_0_downsample_0 = new tk::dnn::Conv2d(&net, 512, 1, 1, 2, 2, 0, 0, layer2_bin[id_layer2_bin++], true); + tk::dnn::Shortcut *s1_0 = new tk::dnn::Shortcut(&net, layer1_0_conv3); + } + else + { + tk::dnn::Shortcut *s1_0 = new tk::dnn::Shortcut(&net, last); + } + tk::dnn::Activation *layer1_0_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + last = layer1_0_relu; + } + + // layer 3 + int id_layer3_bin = 0; + for(int i=0; i<23;i++) + { + tk::dnn::Conv2d *layer1_0_conv1 = new tk::dnn::Conv2d(&net, 256, 1, 1, 1, 1, 0, 0, layer3_bin[id_layer3_bin++], true); + tk::dnn::Activation *relu1_0_1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *layer1_0_conv2; + if(i==0) + layer1_0_conv2 = new tk::dnn::Conv2d(&net, 256, 3, 3, 2, 2, 1, 1, layer3_bin[id_layer3_bin++], true); + else + layer1_0_conv2 = new tk::dnn::Conv2d(&net, 256, 3, 3, 1, 1, 1, 1, layer3_bin[id_layer3_bin++], true); + + tk::dnn::Activation *relu1_0_2 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *layer1_0_conv3 = new tk::dnn::Conv2d(&net, 1024, 1, 1, 1, 1, 0, 0, layer3_bin[id_layer3_bin++], true); + if(i==0) + { + tk::dnn::Layer *route_1_0_layers[1] = { last }; + tk::dnn::Route *route_1_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *layer1_0_downsample_0 = new tk::dnn::Conv2d(&net, 1024, 1, 1, 2, 2, 0, 0, layer3_bin[id_layer3_bin++], true); + tk::dnn::Shortcut *s1_0 = new tk::dnn::Shortcut(&net, layer1_0_conv3); + } + else + { + tk::dnn::Shortcut *s1_0 = new tk::dnn::Shortcut(&net, last); + } + tk::dnn::Activation *layer1_0_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + last = layer1_0_relu; + } + + // layer 4 + int id_layer4_bin = 0; + for(int i=0; i<3;i++) + { + tk::dnn::Conv2d *layer1_0_conv1 = new tk::dnn::Conv2d(&net, 512, 1, 1, 1, 1, 0, 0, layer4_bin[id_layer4_bin++], true); + tk::dnn::Activation *relu1_0_1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *layer1_0_conv2; + if(i==0) + layer1_0_conv2 = new tk::dnn::Conv2d(&net, 512, 3, 3, 2, 2, 1, 1, layer4_bin[id_layer4_bin++], true); + else + layer1_0_conv2 = new tk::dnn::Conv2d(&net, 512, 3, 3, 1, 1, 1, 1, layer4_bin[id_layer4_bin++], true); + + tk::dnn::Activation *relu1_0_2 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *layer1_0_conv3 = new tk::dnn::Conv2d(&net, 2048, 1, 1, 1, 1, 0, 0, layer4_bin[id_layer4_bin++], true); + if(i==0) + { + tk::dnn::Layer *route_1_0_layers[1] = { last }; + tk::dnn::Route *route_1_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *layer1_0_downsample_0 = new tk::dnn::Conv2d(&net, 2048, 1, 1, 2, 2, 0, 0, layer4_bin[id_layer4_bin++], true); + tk::dnn::Shortcut *s1_0 = new tk::dnn::Shortcut(&net, layer1_0_conv3); + } + else + { + tk::dnn::Shortcut *s1_0 = new tk::dnn::Shortcut(&net, last); + } + tk::dnn::Activation *layer1_0_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + last = layer1_0_relu; + } + + tk::dnn::DeformConv2d *layer0_deform1 = new tk::dnn::DeformConv2d(&net, 256, 1, 3, 3, 1, 1, 1, 1, deform1_bin, d_conv1_bin, true); + tk::dnn::Activation *layer0_deform1_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d *layer0_deconv1 = new tk::dnn::DeConv2d(&net, 256, 4, 4, 2, 2, 1, 1, deconv1_bin, true); + tk::dnn::Activation *layer0_deconv1_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::DeformConv2d *layer1_deform1 = new tk::dnn::DeformConv2d(&net, 128, 1, 3, 3, 1, 1, 1, 1, deform2_bin, d_conv2_bin, true); + tk::dnn::Activation *layer1_deform1_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d *layer1_deconv1 = new tk::dnn::DeConv2d(&net, 128, 4, 4, 2, 2, 1, 1, deconv2_bin, true); + tk::dnn::Activation *layer1_deconv1_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::DeformConv2d *layer2_deform1 = new tk::dnn::DeformConv2d(&net, 64, 1, 3, 3, 1, 1, 1, 1, deform3_bin, d_conv3_bin, true); + tk::dnn::Activation *layer2_deform1_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::DeConv2d *layer2_deconv1 = new tk::dnn::DeConv2d(&net, 64, 4, 4, 2, 2, 1, 1, deconv3_bin, true); + tk::dnn::Activation *layer2_deconv1_relu = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Layer *route_1_0_layers[1] = { layer2_deconv1_relu }; + tk::dnn::Conv2d *hm_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, hm_conv1_bin, false); + tk::dnn::Activation *hm_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *hm = new tk::dnn::Conv2d(&net, 3, 1, 1, 1, 1, 0, 0, hm_conv2_bin, false); + hm->setFinal(); + int kernel = 3; + int pad = (kernel - 1)/2; + tk::dnn::Activation *hm_sig = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_SIGMOID); + tk::dnn::Pooling *hmax = new tk::dnn::Pooling(&net, kernel, kernel, 1, 1, pad, pad, tk::dnn::POOLING_MAX); + hmax->setFinal(); + + tk::dnn::Route *route_1_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *wh_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, wh_conv1_bin, false); + tk::dnn::Activation *wh_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *wh = new tk::dnn::Conv2d(&net, 2, 1, 1, 1, 1, 0, 0, wh_conv2_bin, false); + wh->setFinal(); + + tk::dnn::Route *route_2_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *reg_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, reg_conv1_bin, false); + tk::dnn::Activation *reg_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *reg = new tk::dnn::Conv2d(&net, 2, 1, 1, 1, 1, 0, 0, reg_conv2_bin, false); + reg->setFinal(); + + // dep + tk::dnn::Route *route_3_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *dep_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, dep_conv1_bin, false); + tk::dnn::Activation *dep_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *dep = new tk::dnn::Conv2d(&net, 1, 1, 1, 1, 1, 0, 0, dep_conv2_bin, false); + dep->setFinal(); + + // rot + tk::dnn::Route *route_4_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *rot_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, rot_conv1_bin, false); + tk::dnn::Activation *rot_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *rot = new tk::dnn::Conv2d(&net, 8, 1, 1, 1, 1, 0, 0, rot_conv2_bin, false); + rot->setFinal(); + + // dim + tk::dnn::Route *route_5_0 = new tk::dnn::Route(&net, route_1_0_layers, 1); + tk::dnn::Conv2d *dim_conv1 = new tk::dnn::Conv2d(&net, 64, 3, 3, 1, 1, 1, 1, dim_conv1_bin, false); + tk::dnn::Activation *dim_relu1 = new tk::dnn::Activation(&net, CUDNN_ACTIVATION_RELU); + tk::dnn::Conv2d *dim_ = new tk::dnn::Conv2d(&net, 3, 1, 1, 1, 1, 0, 0, dim_conv2_bin, false); + dim_->setFinal(); + + // Load input + dnnType *data; + dnnType *input_h; + readBinaryFile(input_bin, dim.tot(), &input_h, &data); + // printDeviceVector(64, data, true); + + //print network model + net.print(); + + //convert network to tensorRT + tk::dnn::NetworkRT netRT(&net, net.getNetworkRTName("resnet101_cnet3d")); + + + tk::dnn::dataDim_t dim1 = dim; //input dim + printCenteredTitle(" CUDNN inference ", '=', 30); + { + dim1.print(); + TIMER_START + net.infer(dim1, data); + TIMER_STOP + dim1.print(); + } + + // printDeviceVector(64, cudnn_out, true); + + tk::dnn::dataDim_t dim2 = dim; + printCenteredTitle(" TENSORRT inference ", '=', 30); + { + dim2.print(); + TIMER_START + netRT.infer(dim2, data); + TIMER_STOP + dim2.print(); + } + + tk::dnn::Layer *outs[6] = { hm, wh, reg, dep, rot, dim_ }; + int out_count = 1; + int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0; + for(int i=0; i<6; i++) { + printCenteredTitle((std::string(" RESNET CHECK RESULTS ") + std::to_string(i) + " ").c_str(), '=', 30); + + outs[i]->output_dim.print(); + + dnnType *out, *out_h; + int odim = outs[i]->output_dim.tot(); + readBinaryFile(output_bin[i], odim, &out_h, &out); + // std::cout<<"OUTPUT BIN:\n"; + // printDeviceVector(odim, cudnn_out, true); + // std::cout<<"FILE BIN:\n"; + // printDeviceVector(odim, out, true); + + dnnType *cudnn_out, *rt_out; + cudnn_out = outs[i]->dstData; + rt_out = (dnnType *)netRT.buffersRT[i+out_count]; + // there is the maxpool. It isn't an output but it is necessary for the process section + if(i==0) + out_count ++; + + std::cout<<"CUDNN vs correct"; + ret_cudnn |= checkResult(odim, cudnn_out, out) == 0 ? 0: ERROR_CUDNN; + std::cout<<"TRT vs correct"; + ret_tensorrt |= checkResult(odim, rt_out, out) == 0 ? 0 : ERROR_TENSORRT; + std::cout<<"CUDNN vs TRT "; + ret_cudnn_tensorrt |= checkResult(odim, cudnn_out, rt_out) == 0 ? 0 : ERROR_CUDNNvsTENSORRT; + } + return ret_cudnn | ret_tensorrt | ret_cudnn_tensorrt; +} -- 2.52.0 From 3d2405323b0baa54e4796b47dc0c5e9d39e46dfc Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Fri, 29 May 2020 15:54:22 +0200 Subject: [PATCH 006/162] Add the downloading CenterNet weights and outputs for 3D Signed-off-by: Davide Sapienza --- tests/dla34_cnet3d/dla34_cnet3d.cpp | 2 +- tests/resnet101_cnet3d/resnet101_cnet3d.cpp | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/dla34_cnet3d/dla34_cnet3d.cpp b/tests/dla34_cnet3d/dla34_cnet3d.cpp index ecd5693..9766b45 100644 --- a/tests/dla34_cnet3d/dla34_cnet3d.cpp +++ b/tests/dla34_cnet3d/dla34_cnet3d.cpp @@ -112,7 +112,7 @@ const char *output_bin[]={ int main() { - // downloadWeightsifDoNotExist(input_bin, "dla34_cnet3d", "https://cloud.hipert.unimore.it/s/KRZBbCQsKAtQwpZ/download"); + downloadWeightsifDoNotExist(input_bin, "dla34_cnet3d", "https://cloud.hipert.unimore.it/s/2MDyWGzQsTKMjmR/download"); // Network layout tk::dnn::dataDim_t dim(1, 3, 512, 512, 1); diff --git a/tests/resnet101_cnet3d/resnet101_cnet3d.cpp b/tests/resnet101_cnet3d/resnet101_cnet3d.cpp index 0089a19..5d084be 100644 --- a/tests/resnet101_cnet3d/resnet101_cnet3d.cpp +++ b/tests/resnet101_cnet3d/resnet101_cnet3d.cpp @@ -194,7 +194,7 @@ const char *output_bin[]={ int main() { - // downloadWeightsifDoNotExist(input_bin, "resnet101_cnet3d", "https://cloud.hipert.unimore.it/s/5BTjHMWBcJk8g3i/download"); + downloadWeightsifDoNotExist(input_bin, "resnet101_cnet3d", "https://cloud.hipert.unimore.it/s/xH5oH9t5wdnktYf/download"); // Network layout tk::dnn::dataDim_t dim(1, 3, 512, 512, 1); -- 2.52.0 From c4e955eab54332eb8483629b0df42cd0ed94fae0 Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Thu, 11 Jun 2020 12:43:34 +0200 Subject: [PATCH 007/162] Add different yolov4 size tests Signed-off-by: Micaela Verucchi --- include/tkDNN/utils.h | 2 +- src/utils.cpp | 16 +- tests/darknet/cfg/yolo4_320.cfg | 1156 +++++++++++++++++ .../darknet/cfg/{yolo4.cfg => yolo4_416.cfg} | 0 tests/darknet/cfg/yolo4_512.cfg | 1156 +++++++++++++++++ tests/darknet/cfg/yolo4_608.cfg | 1156 +++++++++++++++++ tests/darknet/{yolo4.cpp => yolo4_320.cpp} | 6 +- tests/darknet/yolo4_416.cpp | 34 + tests/darknet/yolo4_512.cpp | 34 + tests/darknet/yolo4_608.cpp | 34 + tests/test_rtinference/rtinference.cpp | 16 +- 11 files changed, 3596 insertions(+), 14 deletions(-) create mode 100644 tests/darknet/cfg/yolo4_320.cfg rename tests/darknet/cfg/{yolo4.cfg => yolo4_416.cfg} (100%) create mode 100644 tests/darknet/cfg/yolo4_512.cfg create mode 100644 tests/darknet/cfg/yolo4_608.cfg rename tests/darknet/{yolo4.cpp => yolo4_320.cpp} (85%) create mode 100644 tests/darknet/yolo4_416.cpp create mode 100644 tests/darknet/yolo4_512.cpp create mode 100644 tests/darknet/yolo4_608.cpp diff --git a/include/tkDNN/utils.h b/include/tkDNN/utils.h index aa73e9e..bca99f8 100644 --- a/include/tkDNN/utils.h +++ b/include/tkDNN/utils.h @@ -105,7 +105,7 @@ void printCenteredTitle(const char *title, char fill, int dim = 30); bool fileExist(const char *fname); void downloadWeightsifDoNotExist(const std::string& input_bin, const std::string& test_folder, const std::string& weights_url); void readBinaryFile(std::string fname, int size, dnnType** data_h, dnnType** data_d, int seek = 0); -int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device = true, int limit = 10); +int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device = true, int limit = 10, bool verbose=true); void printDeviceVector(int size, dnnType* vec_d, bool device = true); float getColor(const int c, const int x, const int max); void resize(int size, dnnType **data); diff --git a/src/utils.cpp b/src/utils.cpp index 65030f0..1ab57ad 100644 --- a/src/utils.cpp +++ b/src/utils.cpp @@ -83,7 +83,7 @@ void printDeviceVector(int size, dnnType* vec_d, bool device){ delete [] vec; } -int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device, int limit) { +int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device, int limit, bool verbose) { dnnType *data_h, *correct_h; const float eps = 0.02f; @@ -117,13 +117,15 @@ int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device, int delete [] correct_h; } - std::cout<<" | "; - if(diffs == 0) - std::cout< input_bins = { bin_path + "/layers/input.bin" }; @@ -15,9 +15,9 @@ int main() { bin_path + "/debug/layer161_out.bin" }; std::string wgs_path = bin_path + "/layers"; - std::string cfg_path = "../tests/darknet/cfg/yolo4.cfg"; + std::string cfg_path = "../tests/darknet/cfg/yolo4_320.cfg"; std::string name_path = "../tests/darknet/names/coco.names"; - downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download"); + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/64PHAwrM6RCZbiR/download"); // parse darknet network tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); diff --git a/tests/darknet/yolo4_416.cpp b/tests/darknet/yolo4_416.cpp new file mode 100644 index 0000000..984b265 --- /dev/null +++ b/tests/darknet/yolo4_416.cpp @@ -0,0 +1,34 @@ +#include +#include +#include "tkdnn.h" +#include "test.h" +#include "DarknetParser.h" + +int main() { + std::string bin_path = "yolo4_416"; + std::vector input_bins = { + bin_path + "/layers/input.bin" + }; + std::vector output_bins = { + bin_path + "/debug/layer139_out.bin", + bin_path + "/debug/layer150_out.bin", + bin_path + "/debug/layer161_out.bin" + }; + std::string wgs_path = bin_path + "/layers"; + std::string cfg_path = "../tests/darknet/cfg/yolo4_416.cfg"; + std::string name_path = "../tests/darknet/names/coco.names"; + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/982LxTQcNQfFQc4/download"); + + // parse darknet network + tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); + net->print(); + + //convert network to tensorRT + tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str())); + + int ret = testInference(input_bins, output_bins, net, netRT); + net->releaseLayers(); + delete net; + delete netRT; + return ret; +} diff --git a/tests/darknet/yolo4_512.cpp b/tests/darknet/yolo4_512.cpp new file mode 100644 index 0000000..414e9be --- /dev/null +++ b/tests/darknet/yolo4_512.cpp @@ -0,0 +1,34 @@ +#include +#include +#include "tkdnn.h" +#include "test.h" +#include "DarknetParser.h" + +int main() { + std::string bin_path = "yolo4_512"; + std::vector input_bins = { + bin_path + "/layers/input.bin" + }; + std::vector output_bins = { + bin_path + "/debug/layer139_out.bin", + bin_path + "/debug/layer150_out.bin", + bin_path + "/debug/layer161_out.bin" + }; + std::string wgs_path = bin_path + "/layers"; + std::string cfg_path = "../tests/darknet/cfg/yolo4_512.cfg"; + std::string name_path = "../tests/darknet/names/coco.names"; + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/XN3FNXs3fnMaK5i/download"); + + // parse darknet network + tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); + net->print(); + + //convert network to tensorRT + tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str())); + + int ret = testInference(input_bins, output_bins, net, netRT); + net->releaseLayers(); + delete net; + delete netRT; + return ret; +} diff --git a/tests/darknet/yolo4_608.cpp b/tests/darknet/yolo4_608.cpp new file mode 100644 index 0000000..dda084f --- /dev/null +++ b/tests/darknet/yolo4_608.cpp @@ -0,0 +1,34 @@ +#include +#include +#include "tkdnn.h" +#include "test.h" +#include "DarknetParser.h" + +int main() { + std::string bin_path = "yolo4_608"; + std::vector input_bins = { + bin_path + "/layers/input.bin" + }; + std::vector output_bins = { + bin_path + "/debug/layer139_out.bin", + bin_path + "/debug/layer150_out.bin", + bin_path + "/debug/layer161_out.bin" + }; + std::string wgs_path = bin_path + "/layers"; + std::string cfg_path = "../tests/darknet/cfg/yolo4_608.cfg"; + std::string name_path = "../tests/darknet/names/coco.names"; + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/Bg9r7kqDFJiFB4c/download"); + + // parse darknet network + tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); + net->print(); + + //convert network to tensorRT + tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str())); + + int ret = testInference(input_bins, output_bins, net, netRT); + net->releaseLayers(); + delete net; + delete netRT; + return ret; +} diff --git a/tests/test_rtinference/rtinference.cpp b/tests/test_rtinference/rtinference.cpp index a629168..e128ae8 100644 --- a/tests/test_rtinference/rtinference.cpp +++ b/tests/test_rtinference/rtinference.cpp @@ -1,4 +1,5 @@ #include +#include #include "tkdnn.h" #include /* srand, rand */ @@ -29,6 +30,7 @@ int main(int argc, char *argv[]) { int ret_tensorrt = 0; std::cout<<"Testing with batchsize: "< stats; printCenteredTitle(" TENSORRT inference ", '=', 30); float total_time = 0; for(int i=0; i<1200; i++) { @@ -46,18 +48,26 @@ int main(int argc, char *argv[]) { netRT.infer(dim, input_d); TKDNN_TSTOP total_time+= t_ns; + if(i> 1) + stats.push_back(t_ns); // control output - std::cout<<"Output Buffers: "< Date: Thu, 11 Jun 2020 16:09:01 +0200 Subject: [PATCH 008/162] Add script for inference FPS Signed-off-by: Micaela Verucchi --- scripts/test_inference.sh | 51 ++++++++++++++++++++++++++ tests/test_rtinference/rtinference.cpp | 25 +++++++++++-- 2 files changed, 73 insertions(+), 3 deletions(-) create mode 100644 scripts/test_inference.sh diff --git a/scripts/test_inference.sh b/scripts/test_inference.sh new file mode 100644 index 0000000..fe8dac3 --- /dev/null +++ b/scripts/test_inference.sh @@ -0,0 +1,51 @@ +#!/bin/bash + +function test_inference { + ./test_$1 + ./test_rtinference $1_$2.rt 1 + ./test_rtinference $1_$2.rt 4 +} + +sudo jeston_clock + +# modes=( 1 ) # only FP32 +# modes=( 1 2 ) # FP32 and FP16 +modes=( 1 2 3 ) # FP32, FP16 and INT8 + +rm times_rtinference.csv +for i in "${modes[@]}" +do + rm *rt + if [ $i -eq 1 ] + then + export TKDNN_MODE=FP32 + mode=fp32 + echo -e "${ORANGE}Test FP32${NC}" + fi + if [ $i -eq 2 ] + then + export TKDNN_MODE=FP16 + mode=fp16 + echo -e "${ORANGE}Test FP16${NC}" + fi + if [ $i -eq 3 ] + then + export TKDNN_MODE=INT8 + export TKDNN_CALIB_LABEL_PATH=../demo/COCO_val2017/all_labels.txt + export TKDNN_CALIB_IMG_PATH=../demo/COCO_val2017/all_images.txt + mode=int8 + echo -e "${ORANGE}Test INT8${NC}" + + fi + + export TKDNN_BATCHSIZE=4 + echo -e "${ORANGE}Batch $TKDNN_BATCHSIZE ${NC}" + + test_inference yolo4_320 $mode + test_inference yolo4_416 $mode + test_inference yolo4_512 $mode + test_inference yolo4_608 $mode +done + + + diff --git a/tests/test_rtinference/rtinference.cpp b/tests/test_rtinference/rtinference.cpp index e128ae8..76c2a33 100644 --- a/tests/test_rtinference/rtinference.cpp +++ b/tests/test_rtinference/rtinference.cpp @@ -18,6 +18,8 @@ int main(int argc, char *argv[]) { //convert network to tensorRT tk::dnn::NetworkRT netRT(NULL, argv[1]); + + tk::dnn::dataDim_t idim = netRT.input_dim; tk::dnn::dataDim_t odim = netRT.output_dim; @@ -63,11 +65,28 @@ int main(int argc, char *argv[]) { } } } - std::cout<<"Min: "<<*std::min_element(stats.begin(), stats.end())/BATCH_SIZE<<" ms\n"; - std::cout<<"Max: "<<*std::max_element(stats.begin(), stats.end())/BATCH_SIZE<<" ms\n"; + + double min = *std::min_element(stats.begin(), stats.end())/BATCH_SIZE; + double max = *std::max_element(stats.begin(), stats.end())/BATCH_SIZE; double mean =0; for(int i=0; i Date: Thu, 11 Jun 2020 20:54:17 +0200 Subject: [PATCH 009/162] New yolo4_512 download link Signed-off-by: Micaela Verucchi --- tests/darknet/yolo4_512.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/darknet/yolo4_512.cpp b/tests/darknet/yolo4_512.cpp index 414e9be..df3c2d0 100644 --- a/tests/darknet/yolo4_512.cpp +++ b/tests/darknet/yolo4_512.cpp @@ -17,7 +17,7 @@ int main() { std::string wgs_path = bin_path + "/layers"; std::string cfg_path = "../tests/darknet/cfg/yolo4_512.cfg"; std::string name_path = "../tests/darknet/names/coco.names"; - downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/XN3FNXs3fnMaK5i/download"); + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/fjFDqFmiSARKxFe/download"); // parse darknet network tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); -- 2.52.0 From 9f1e30eaa9c637e9ef30ab78e9f4809f4caac4e2 Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Fri, 19 Jun 2020 18:55:42 +0200 Subject: [PATCH 010/162] Add shelfnet. Resnet18backbone works Signed-off-by: Micaela Verucchi --- CMakeLists.txt | 4 + src/kernels/activation_leaky.cu | 2 +- src/utils.cpp | 1 + tests/shelfnet/shelfnet.cpp | 213 ++++++++++++++++++++++++++++++++ 4 files changed, 219 insertions(+), 1 deletion(-) create mode 100644 tests/shelfnet/shelfnet.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index 8c8619d..a2a8a06 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -103,6 +103,10 @@ target_link_libraries(test_resnet101_cnet tkDNN) add_executable(test_dla34_cnet tests/centernet/dla34_cnet/dla34_cnet.cpp) target_link_libraries(test_dla34_cnet tkDNN) +# SHELFNET +add_executable(test_shelfnet tests/shelfnet/shelfnet.cpp) +target_link_libraries(test_shelfnet tkDNN) + # DEMOS add_executable(test_rtinference tests/test_rtinference/rtinference.cpp) target_link_libraries(test_rtinference tkDNN) diff --git a/src/kernels/activation_leaky.cu b/src/kernels/activation_leaky.cu index a9029ad..47dc18f 100644 --- a/src/kernels/activation_leaky.cu +++ b/src/kernels/activation_leaky.cu @@ -9,7 +9,7 @@ void activation_leaky(dnnType *input, dnnType *output, int size) { if (input[i]>0) output[i] = input[i]; else - output[i] = 0.1f*input[i]; + output[i] = 0.01f*input[i]; //FIME!! } } diff --git a/src/utils.cpp b/src/utils.cpp index 65030f0..87aad06 100644 --- a/src/utils.cpp +++ b/src/utils.cpp @@ -102,6 +102,7 @@ int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device, int } int diffs = 0; for(int i=0; i eps) { diffs += 1; diff --git a/tests/shelfnet/shelfnet.cpp b/tests/shelfnet/shelfnet.cpp new file mode 100644 index 0000000..3f8834a --- /dev/null +++ b/tests/shelfnet/shelfnet.cpp @@ -0,0 +1,213 @@ +#include +#include "tkdnn.h" + + +const char *output_bin1 = "shelfnet/debug/classification_headers-5.bin"; +const char *output_bin2 = "shelfnet/debug/regression_headers-5.bin"; +const char *input_bin = "shelfnet/debug/input.bin"; + +const char *backbone[] = { + "shelfnet/layers/backbone-conv1.bin", + "shelfnet/layers/backbone-layer1-0-conv1.bin", + "shelfnet/layers/backbone-layer1-0-conv2.bin", + "shelfnet/layers/backbone-layer1-1-conv1.bin", + "shelfnet/layers/backbone-layer1-1-conv2.bin", + "shelfnet/layers/backbone-layer2-0-conv1.bin", + "shelfnet/layers/backbone-layer2-0-conv2.bin", + "shelfnet/layers/backbone-layer2-0-downsample-0.bin", + "shelfnet/layers/backbone-layer2-1-conv1.bin", + "shelfnet/layers/backbone-layer2-1-conv2.bin", + "shelfnet/layers/backbone-layer3-0-conv1.bin", + "shelfnet/layers/backbone-layer3-0-conv2.bin", + "shelfnet/layers/backbone-layer3-0-downsample-0.bin", + "shelfnet/layers/backbone-layer3-1-conv1.bin", + "shelfnet/layers/backbone-layer3-1-conv2.bin", + "shelfnet/layers/backbone-layer4-0-conv1.bin", + "shelfnet/layers/backbone-layer4-0-conv2.bin", + "shelfnet/layers/backbone-layer4-0-downsample-0.bin", + "shelfnet/layers/backbone-layer4-1-conv1.bin", + "shelfnet/layers/backbone-layer4-1-conv2.bin"}; + +const char *conv_out[] = { + "shelfnet/layers/conv_out16-conv-conv.bin", + "shelfnet/layers/conv_out16-conv_out.bin", + "shelfnet/layers/conv_out32-conv-conv.bin", + "shelfnet/layers/conv_out32-conv_out.bin", + "shelfnet/layers/conv_out-conv-conv.bin", + "shelfnet/layers/conv_out-conv_out.bin"}; + +const char *decoder[] = { + "shelfnet/layers/decoder-bottom-conv1.bin", + "shelfnet/layers/decoder-up_conv_list-0-conv_atten.bin", + "shelfnet/layers/decoder-up_conv_list-0-conv-conv.bin", + "shelfnet/layers/decoder-up_conv_list-1-conv_atten.bin", + "shelfnet/layers/decoder-up_conv_list-1-conv-conv.bin", + "shelfnet/layers/decoder-up_dense_list-0-conv.bin", + "shelfnet/layers/decoder-up_dense_list-1-conv.bin"}; + + +const char *ladder[] = { + "shelfnet/layers/ladder-bottom-conv1.bin", + "shelfnet/layers/ladder-down_conv_list-0.bin", + "shelfnet/layers/ladder-down_conv_list-1.bin", + "shelfnet/layers/ladder-down_module_list-0-conv1.bin", + "shelfnet/layers/ladder-down_module_list-1-conv1.bin", + "shelfnet/layers/ladder-inconv-conv1.bin", + "shelfnet/layers/ladder-up_conv_list-0-conv_atten.bin", + "shelfnet/layers/ladder-up_conv_list-0-conv-conv.bin", + "shelfnet/layers/ladder-up_conv_list-1-conv_atten.bin", + "shelfnet/layers/ladder-up_conv_list-1-conv-conv.bin", + "shelfnet/layers/ladder-up_dense_list-0-conv.bin", + "shelfnet/layers/ladder-up_dense_list-1-conv.bin"}; + +const char *trans[] = { + "shelfnet/layers/trans1-conv.bin", + "shelfnet/layers/trans2-conv.bin", + "shelfnet/layers/trans3-conv.bin"}; +int main() +{ + + // downloadWeightsifDoNotExist(input_bin, "shelfnet", "https://cloud.hipert.unimore.it/s/x4ZfxBKN23zAJQp/download"); + + int classes = 19; + + // Network layout + tk::dnn::dataDim_t dim(1, 3, 1024, 1024, 1); + tk::dnn::Network net(dim); + + int bi = 0; + new tk::dnn::Conv2d(&net, 64, 7, 7, 2, 2, 3, 3, backbone[bi++], true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY); + tk::dnn::Layer* last = new tk::dnn::Pooling (&net, 3, 3, 2, 2, 1, 1, tk::dnn::POOLING_MAX); + + + + for(int i=0; i<2; ++i){ + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY); + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true); + new tk::dnn::Shortcut(&net, last); + last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + } + + std::vector features; + for(int i=0;i<3;++i){ + int out_channel = pow(2,7+i); + std::cout< batch_frame; + std::vector batch_dnn_input; + + while(gRun) { + batch_dnn_input.clear(); + batch_frame.clear(); + + for(int bi=0; bi< n_batch; ++bi){ + cap >> frame; + if(!frame.data) + break; + + batch_frame.push_back(frame); + + // this will be resized to the net format + batch_dnn_input.push_back(frame.clone()); + } + if(!frame.data) + break; + + //inference + segNN.update(batch_dnn_input, n_batch); + segNN.draw(); + + if(show){ + for(int bi=0; bi< n_batch; ++bi){ + cv::imshow("segmentation", batch_frame[bi]); + cv::waitKey(1); + } + } + if(n_batch == 1 && SAVE_RESULT) + resultVideo << frame; + } + + std::cout<<"segmentation end\n"; + double mean = 0; + + std::cout< +#include +#include +#include +#include +#include "utils.h" + +#include +#include +#include +#include + + + +#include "tkdnn.h" +#include "NetworkViz.h" + +namespace tk { namespace dnn { + +class SegmentationNN { + + protected: + tk::dnn::NetworkRT *netRT = nullptr; + int nBatches = 1; + + std::vector originalSize; + std::vector masks; + cv::Mat bgr[3]; + dnnType *input; + dnnType *input_d; + float* confidences_h; + + /** + * This method preprocess the image, before feeding it to the NN. + * + * @param frame original frame to adapt for inference. + * @param bi batch index + */ + void preprocess(cv::Mat &frame, const int bi=0) { + + frame.convertTo(frame, CV_32FC3, 1 / 255.0, 0); + + cv::split(frame, bgr); + float mean[] = {0.485, 0.456, 0.406}; + float stddev[] = {0.229, 0.224, 0.225}; + for(int i=0; i<3; i++){ + bgr[2-i] -= mean[i]; + bgr[2-i] /= stddev[i]; + } + cv::merge(bgr, 3, frame); + + int crop_size = netRT->input_dim.w; + int H = frame.rows; + int W = frame.cols; + cv::Mat frame_cropped; + + cv::Mat mask(frame.size(), CV_8UC3, cv::Scalar(255,255,255)); + + if(H != W){ + if(H < W){ + int top = (W - H)/2; + int bottom = W - top - H; + cv::copyMakeBorder(frame, frame_cropped, top, bottom, 0, 0, cv::BORDER_CONSTANT, cv::Scalar(0,0,0) ); + cv::copyMakeBorder(mask, mask, top, bottom, 0, 0, cv::BORDER_CONSTANT, cv::Scalar(0,0,0) ); + } + else{ + int left = (H - W)/2; + int right = H - left - W; + cv::copyMakeBorder(frame, frame_cropped, 0, 0, left, right, cv::BORDER_CONSTANT, cv::Scalar(0,0,0) ); + cv::copyMakeBorder(mask, mask, 0, 0, left, right, cv::BORDER_CONSTANT, cv::Scalar(0,0,0) ); + } + } + + resize(frame_cropped, frame_cropped, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); + resize(mask, mask, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); + masks[bi] = mask.clone(); + + cv::split(frame_cropped, bgr); + for (int i = 0; i < netRT->input_dim.c; i++){ + int idx = i * frame_cropped.rows * frame_cropped.cols; + int ch = netRT->input_dim.c-1 -i; + memcpy((void *)&input[idx + netRT->input_dim.tot()*bi], (void *)bgr[ch].data, frame_cropped.rows * frame_cropped.cols * sizeof(dnnType)); + } + checkCuda(cudaMemcpyAsync(input_d+ netRT->input_dim.tot()*bi, input + netRT->input_dim.tot()*bi, netRT->input_dim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream)); + } + + /** + * This method postprocess the output of the NN to obtain the correct + * boundig boxes. + * + * @param bi batch index + */ + void postprocess(const int bi=0) { + dnnType *rt_out = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi; + + dataDim_t odim = netRT->output_dim; + + + checkCuda(cudaMemcpy(confidences_h, rt_out, odim.tot() * sizeof(float), cudaMemcpyDeviceToHost)); + + for(int i=0;i max_conf){ + max_conf = cur_conf; + max_id = k; + } + } + confidences_h[bi*odim.tot()+0*odim.h*odim.w+i*odim.h+j] = max_id; + } + } + dataDim_t vdim = odim; + vdim.c = 1; + segmented[bi] = vizData2Mat(confidences_h, vdim, 1024, 0, 18); + }; + + public: + int classes = 0; + std::vector stats; /*keeps track of inference times (ms)*/ + std::vector classesNames; + std::vector segmented; + + SegmentationNN() {}; + ~SegmentationNN(){}; + + /** + * Method used to inialize the class, allocate memory and compute + * needed data. + * + * @param tensor_path path to the rt file og the NN. + * @param n_classes number of classes for the given dataset. + * @param n_batches maximum number of batches to use in inference + * @return true if everything is correct, false otherwise. + */ + bool init(const std::string& tensor_path, const int n_classes=19, const int n_batches=1){ + std::cout<<(tensor_path).c_str()<<"\n"; + if(!fileExist(tensor_path.c_str())) + FatalError("This file do not exists" + tensor_path ); + + netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str()); + classes = n_classes; + nBatches = n_batches; + + checkCuda(cudaMallocHost(&input, sizeof(dnnType) * netRT->input_dim.tot() * nBatches)); + checkCuda(cudaMalloc(&input_d, sizeof(dnnType) * netRT->input_dim.tot() * nBatches)); + + confidences_h = (float *)malloc(netRT->output_dim.tot() * sizeof(float)); + + segmented.resize(nBatches); + masks.resize(nBatches); + } + + /** + * This method performs the whole detection of the NN. + * + * @param frames frames to run detection on. + * @param cur_batches number of batches to use in inference + * @param save_times if set to true, preprocess, inference and postprocess times + * are saved on a csv file, otherwise not. + * @param times pointer to the output stream where to write times + * @param mAP set to true only if all the probabilities for a bounding + * box are needed, as in some cases for the mAP calculation + */ + void update(std::vector& frames, const int cur_batches=1){ + if(cur_batches > nBatches) + FatalError("A batch size greater than nBatches cannot be used"); + + originalSize.clear(); + if(TKDNN_VERBOSE) printCenteredTitle(" TENSORRT detection ", '=', 30); + { + TKDNN_TSTART + for(int bi=0; biinput_dim; + dim.n = cur_batches; + { + if(TKDNN_VERBOSE) dim.print(); + TKDNN_TSTART + netRT->infer(dim, input_d); + TKDNN_TSTOP + if(TKDNN_VERBOSE) dim.print(); + stats.push_back(t_ns); + } + + { + TKDNN_TSTART + for(int bi=0; bi> $out_file print_output $? imuodom + test_net shelfnet test_net yolo4 test_net yolo4_berkeley test_net yolo3 diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index 71662d3..92a95f2 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -483,6 +483,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Resize *l) { IResizeLayer *lRT = networkRT->addResize(*input); //default is kNEAREST checkNULL(lRT); Dims d{}; + lRT->setResizeMode(ResizeMode(l->mode)); lRT->setOutputDimensions(DimsCHW{l->output_dim.c, l->output_dim.h, l->output_dim.w}); return lRT; } @@ -514,7 +515,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Shortcut *l) { ITensor *back_tens = tensors[l->backLayer]; - if(false) //l->backLayer->output_dim.c == l->output_dim.c && !l->mul) FIXME + if(l->backLayer->output_dim.c == l->output_dim.c && !l->mul) { IElementWiseLayer *lRT = networkRT->addElementWise(*input, *back_tens, ElementWiseOperation::kSUM); checkNULL(lRT); diff --git a/src/NetworkViz.cpp b/src/NetworkViz.cpp index 6ac274c..842a26e 100644 --- a/src/NetworkViz.cpp +++ b/src/NetworkViz.cpp @@ -6,23 +6,22 @@ namespace tk { namespace dnn { -cv::Mat vizFloat2colorMap(cv::Mat map) { +cv::Mat vizFloat2colorMap(cv::Mat map,double min, double max) { + + if(min == 0 && max == 0) + cv::minMaxIdx(map, &min, &max); - double min; - double max; - cv::minMaxIdx(map, &min, &max); cv::Mat adjMap; // expand your range to 0..255. Similar to histEq(); map.convertTo(adjMap,CV_8UC1, 255 / (max-min), -min); //return adjMap; - cv::Mat falseColorsMap; - applyColorMap(adjMap, falseColorsMap, cv::COLORMAP_HOT); + applyColorMap(adjMap, falseColorsMap, cv::COLORMAP_VIRIDIS); return falseColorsMap; } -cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int imgdim) { +cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int imgdim, double min, double max) { dnnType *data = nullptr; // copy to CPU @@ -38,7 +37,7 @@ cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int imgdim) { cv::Mat grid = cv::Mat(gridSize, CV_8UC3, cv::Scalar(0)); for(int i=0; imode = mode; if(fixed){ output_dim.c = scale_c; output_dim.h = scale_h; diff --git a/tests/shelfnet/shelfnet.cpp b/tests/shelfnet/shelfnet.cpp index ab28d90..7cc1b42 100644 --- a/tests/shelfnet/shelfnet.cpp +++ b/tests/shelfnet/shelfnet.cpp @@ -191,7 +191,7 @@ int main() down_out.push_back(l_last); new tk::dnn::Conv2d (&net, out_channel*2, 3, 3, 2, 2, 1, 1, ladder[li++], false); - last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.0f); //should be ReLU } new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); @@ -231,12 +231,12 @@ int main() new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, conv_out[ci++], true); new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); new tk::dnn::Conv2d (&net, 19, 3, 3, 1, 1, 1, 1, conv_out[ci++], false); - // /*up_out[i] =*/ new tk::dnn::Resize(&net, 19, net.input_dim.h, net.input_dim.w, true); + /*up_out[i] =*/ new tk::dnn::Resize(&net, 19, net.input_dim.h, net.input_dim.w, true, tk::dnn::ResizeMode_t::LINEAR); // } - // new tk::dnn::Softmax(&net); + new tk::dnn::Softmax(&net); - const char *output_bin = "shelfnet/debug/conv_out-conv_out.bin"; + const char *output_bin = "shelfnet/debug/softmax.bin"; // Load input dnnType *data; diff --git a/tests/test_rtinference/rtinference.cpp b/tests/test_rtinference/rtinference.cpp index a629168..3eacc37 100644 --- a/tests/test_rtinference/rtinference.cpp +++ b/tests/test_rtinference/rtinference.cpp @@ -31,7 +31,7 @@ int main(int argc, char *argv[]) { std::cout<<"Testing with batchsize: "< batch_frame; std::vector batch_dnn_input; @@ -85,14 +82,8 @@ int main(int argc, char *argv[]) { //inference segNN.update(batch_dnn_input, n_batch); - segNN.draw(); + frame = segNN.draw(); - if(show){ - for(int bi=0; bi< n_batch; ++bi){ - cv::imshow("segmentation", batch_frame[bi]); - cv::waitKey(1); - } - } if(n_batch == 1 && SAVE_RESULT) resultVideo << frame; } diff --git a/include/tkDNN/SegmentationNN.h b/include/tkDNN/SegmentationNN.h index 93be2ff..0289d61 100644 --- a/include/tkDNN/SegmentationNN.h +++ b/include/tkDNN/SegmentationNN.h @@ -17,6 +17,7 @@ #include "tkdnn.h" #include "NetworkViz.h" +#include "kernelsThrust.h" namespace tk { namespace dnn { @@ -33,6 +34,12 @@ class SegmentationNN { dnnType *input_d; float* confidences_h; + float * tmpInputData_d; + float *tmpOutData_d; + float *tmpOutData_h; + + cublasHandle_t cublasHandle; + /** * This method preprocess the image, before feeding it to the NN. * @@ -97,28 +104,14 @@ class SegmentationNN { dnnType *rt_out = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi; dataDim_t odim = netRT->output_dim; - - checkCuda(cudaMemcpy(confidences_h, rt_out, odim.tot() * sizeof(float), cudaMemcpyDeviceToHost)); + matrixTranspose(cublasHandle, rt_out, tmpInputData_d, odim.c, odim.w*odim.h); + maxElem(tmpInputData_d, tmpOutData_d, odim.c, odim.h, odim.w); + checkCuda(cudaMemcpy(tmpOutData_h, tmpOutData_d, odim.w*odim.h * sizeof(float), cudaMemcpyDeviceToHost)); - for(int i=0;i max_conf){ - max_conf = cur_conf; - max_id = k; - } - } - confidences_h[bi*odim.tot()+0*odim.h*odim.w+i*odim.h+j] = max_id; - } - } dataDim_t vdim = odim; vdim.c = 1; - segmented[bi] = vizData2Mat(confidences_h, vdim, 1024, 0, 18); + segmented[bi] = vizData2Mat(tmpOutData_h, vdim, 1024, 0, 18); }; public: @@ -127,8 +120,12 @@ class SegmentationNN { std::vector classesNames; std::vector segmented; - SegmentationNN() {}; - ~SegmentationNN(){}; + SegmentationNN() { + checkERROR( cublasCreate(&cublasHandle) ); + }; + ~SegmentationNN(){ + checkERROR( cublasDestroy(cublasHandle) ); + }; /** * Method used to inialize the class, allocate memory and compute @@ -151,7 +148,12 @@ class SegmentationNN { checkCuda(cudaMallocHost(&input, sizeof(dnnType) * netRT->input_dim.tot() * nBatches)); checkCuda(cudaMalloc(&input_d, sizeof(dnnType) * netRT->input_dim.tot() * nBatches)); - confidences_h = (float *)malloc(netRT->output_dim.tot() * sizeof(float)); + dataDim_t odim = netRT->output_dim; + + checkCuda(cudaMallocHost(&confidences_h, sizeof(float) * odim.tot())); + checkCuda(cudaMalloc(&tmpInputData_d, sizeof(float) * odim.tot())); + checkCuda(cudaMalloc(&tmpOutData_d, sizeof(float) * odim.w*odim.h)); + checkCuda(cudaMallocHost(&tmpOutData_h, sizeof(float) * odim.w*odim.h)); segmented.resize(nBatches); masks.resize(nBatches); @@ -208,7 +210,7 @@ class SegmentationNN { /** * Method to draw boundixg boxes and labels on a frame. */ - void draw(const int cur_batches=1) { + cv::Mat draw(const int cur_batches=1) { for(int i=0; i #include #include #include @@ -9,6 +10,8 @@ #include #include #include +#include + #include "tkdnn.h" @@ -36,4 +39,6 @@ void topKxyAddOffset(int * ids_begin, const int K, const int size, int *intxs_be void bboxes(int * ids_begin, const int K, const int size, float *xs_begin, float *ys_begin, dnnType *src_begin, float *bbx0, float *bbx1, float *bby0, float *bby1, float *src_out, int *ids_out); +void maxElem(dnnType *src_begin, dnnType *dst_begin, const int c, const int h, const int w); + #endif //KERNELSTHRUST_H \ No newline at end of file diff --git a/src/kernels/postprocessing.cu b/src/kernels/postprocessing.cu index 3510200..53234a2 100644 --- a/src/kernels/postprocessing.cu +++ b/src/kernels/postprocessing.cu @@ -34,6 +34,25 @@ void sortAndTopKonDevice(dnnType *src_begin, int *idsrc, float *topk_scores, int sortAndTopK_kernel<<>>(src_begin, idsrc, topk_scores, topk_inds, topk_ys, topk_xs, size, K); } +__global__ +void maxElem_kernel(float *src_begin, float *dst_begin, const int n_classes, const int size){ + int i = blockDim.x*blockIdx.x + threadIdx.x; + if (i > size) + return; + + thrust::device_ptr dPbeg ( &src_begin[i*n_classes] ) ; + thrust::device_ptr dPend = dPbeg + n_classes; + thrust::device_ptr result = thrust::max_element(thrust::device,dPbeg, dPend); + + dst_begin[i] = result - dPbeg; +} + +void maxElem(dnnType *src_begin, dnnType *dst_begin, const int c, const int h, const int w){ + int blocks = (h*w)/32+1; + int threads = 32; + maxElem_kernel<<>>(src_begin, dst_begin, c, h*w); +} + void topKxyclasses(int *ids_begin, int *ids_end, const int K, const int size, const int wh, int *clses, int *xs, int *ys){ thrust::transform(thrust::device, ids_begin, ids_end, thrust::make_constant_iterator(wh), clses, thrust::divides()); thrust::transform(thrust::device, ids_begin, ids_end, thrust::make_constant_iterator(wh), ids_begin, thrust::modulus()); -- 2.52.0 From 3bf954750273e342fa7d42ab95fe10068f070fab Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Mon, 29 Jun 2020 15:11:52 +0200 Subject: [PATCH 015/162] Improve preprocessing Signed-off-by: Micaela Verucchi --- include/tkDNN/SegmentationNN.h | 46 ++++++++++++++++++---------------- 1 file changed, 24 insertions(+), 22 deletions(-) diff --git a/include/tkDNN/SegmentationNN.h b/include/tkDNN/SegmentationNN.h index 0289d61..87c7b8c 100644 --- a/include/tkDNN/SegmentationNN.h +++ b/include/tkDNN/SegmentationNN.h @@ -13,8 +13,6 @@ #include #include - - #include "tkdnn.h" #include "NetworkViz.h" #include "kernelsThrust.h" @@ -38,6 +36,8 @@ class SegmentationNN { float *tmpOutData_d; float *tmpOutData_h; + float *mean_d, *stddev_d; + cublasHandle_t cublasHandle; /** @@ -47,23 +47,10 @@ class SegmentationNN { * @param bi batch index */ void preprocess(cv::Mat &frame, const int bi=0) { - frame.convertTo(frame, CV_32FC3, 1 / 255.0, 0); - - cv::split(frame, bgr); - float mean[] = {0.485, 0.456, 0.406}; - float stddev[] = {0.229, 0.224, 0.225}; - for(int i=0; i<3; i++){ - bgr[2-i] -= mean[i]; - bgr[2-i] /= stddev[i]; - } - cv::merge(bgr, 3, frame); - - int crop_size = netRT->input_dim.w; int H = frame.rows; int W = frame.cols; cv::Mat frame_cropped; - cv::Mat mask(frame.size(), CV_8UC3, cv::Scalar(255,255,255)); if(H != W){ @@ -81,17 +68,22 @@ class SegmentationNN { } } - resize(frame_cropped, frame_cropped, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); - resize(mask, mask, cv::Size(netRT->input_dim.w, netRT->input_dim.h)); - masks[bi] = mask.clone(); + tk::dnn::dataDim_t idim = netRT->input_dim; + + resize(frame_cropped, frame_cropped, cv::Size(idim.w, idim.h)); + resize(mask, mask, cv::Size(idim.w, idim.h)); + masks[bi] = mask; cv::split(frame_cropped, bgr); - for (int i = 0; i < netRT->input_dim.c; i++){ + for (int i = 0; i < idim.c; i++){ int idx = i * frame_cropped.rows * frame_cropped.cols; - int ch = netRT->input_dim.c-1 -i; - memcpy((void *)&input[idx + netRT->input_dim.tot()*bi], (void *)bgr[ch].data, frame_cropped.rows * frame_cropped.cols * sizeof(dnnType)); + int ch = idim.c-1 -i; + memcpy((void *)&input[idx + idim.tot()*bi], (void *)bgr[ch].data, frame_cropped.rows * frame_cropped.cols * sizeof(dnnType)); } - checkCuda(cudaMemcpyAsync(input_d+ netRT->input_dim.tot()*bi, input + netRT->input_dim.tot()*bi, netRT->input_dim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream)); + + checkCuda(cudaMemcpyAsync(input_d+ idim.tot()*bi, input + idim.tot()*bi, idim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream)); + + normalize(input_d + idim.tot()*bi, idim.c, idim.h, idim.w, mean_d, stddev_d); } /** @@ -157,6 +149,16 @@ class SegmentationNN { segmented.resize(nBatches); masks.resize(nBatches); + + std::vector mean = {0.485, 0.456, 0.406}; + std::vector stddev = {0.229, 0.224, 0.225}; + + checkCuda(cudaMalloc(&mean_d, sizeof(float) * mean.size())); + checkCuda(cudaMalloc(&stddev_d, sizeof(float) * stddev.size())); + + checkCuda(cudaMemcpyAsync(mean_d, mean.data(), mean.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream)); + checkCuda(cudaMemcpyAsync(stddev_d, stddev.data(), stddev.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream)); + } /** -- 2.52.0 From a5cc4e3edada046902cb9fd3eea0ad1efe59970a Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Wed, 1 Jul 2020 10:53:24 +0200 Subject: [PATCH 016/162] Add berkeley test, add weights for shelfnets Signed-off-by: Micaela Verucchi --- CMakeLists.txt | 3 + demo/demo/seg_demo.cpp | 9 +- tests/shelfnet/shelfnet.cpp | 2 +- tests/shelfnet/shelfnet_berkeley.cpp | 295 +++++++++++++++++++++++++++ 4 files changed, 305 insertions(+), 4 deletions(-) create mode 100644 tests/shelfnet/shelfnet_berkeley.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index 88e94e3..9e9c27c 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -107,6 +107,9 @@ target_link_libraries(test_dla34_cnet tkDNN) add_executable(test_shelfnet tests/shelfnet/shelfnet.cpp) target_link_libraries(test_shelfnet tkDNN) +add_executable(test_shelfnet_berkeley tests/shelfnet/shelfnet_berkeley.cpp) +target_link_libraries(test_shelfnet_berkeley tkDNN) + # DEMOS add_executable(test_rtinference tests/test_rtinference/rtinference.cpp) target_link_libraries(test_rtinference tkDNN) diff --git a/demo/demo/seg_demo.cpp b/demo/demo/seg_demo.cpp index c1ee43b..0737ad7 100644 --- a/demo/demo/seg_demo.cpp +++ b/demo/demo/seg_demo.cpp @@ -29,9 +29,12 @@ int main(int argc, char *argv[]) { int n_batch = 1; if(argc > 3) n_batch = atoi(argv[3]); - bool show = false; + int n_classes = 19; if(argc > 4) - show = atoi(argv[4]); + n_classes = atoi(argv[4]); + bool show = false; + if(argc > 5) + show = atoi(argv[5]); if(n_batch < 1 || n_batch > 64) FatalError("Batch dim not supported"); @@ -39,7 +42,7 @@ int main(int argc, char *argv[]) { if(!show) SAVE_RESULT = true; - int n_classes = 19; + tk::dnn::SegmentationNN segNN; segNN.init(net, n_classes, n_batch); diff --git a/tests/shelfnet/shelfnet.cpp b/tests/shelfnet/shelfnet.cpp index 7cc1b42..48cad04 100644 --- a/tests/shelfnet/shelfnet.cpp +++ b/tests/shelfnet/shelfnet.cpp @@ -83,7 +83,7 @@ const char *trans[] = { int main() { - // downloadWeightsifDoNotExist(input_bin, "shelfnet", "https://cloud.hipert.unimore.it/s/x4ZfxBKN23zAJQp/download"); + downloadWeightsifDoNotExist(input_bin, "shelfnet", "https://cloud.hipert.unimore.it/s/mEDZMRJaGCFWSJF/download"); int classes = 19; diff --git a/tests/shelfnet/shelfnet_berkeley.cpp b/tests/shelfnet/shelfnet_berkeley.cpp new file mode 100644 index 0000000..fe81191 --- /dev/null +++ b/tests/shelfnet/shelfnet_berkeley.cpp @@ -0,0 +1,295 @@ +#include +#include +#include + +#include "tkdnn.h" +#include "NetworkViz.h" + + +const char *input_bin = "shelfnet_berkeley/debug/input.bin"; + +const char *backbone[] = { + "shelfnet_berkeley/layers/backbone-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer1-0-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer1-0-conv2.bin", + "shelfnet_berkeley/layers/backbone-layer1-1-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer1-1-conv2.bin", + "shelfnet_berkeley/layers/backbone-layer2-0-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer2-0-conv2.bin", + "shelfnet_berkeley/layers/backbone-layer2-0-downsample-0.bin", + "shelfnet_berkeley/layers/backbone-layer2-1-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer2-1-conv2.bin", + "shelfnet_berkeley/layers/backbone-layer3-0-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer3-0-conv2.bin", + "shelfnet_berkeley/layers/backbone-layer3-0-downsample-0.bin", + "shelfnet_berkeley/layers/backbone-layer3-1-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer3-1-conv2.bin", + "shelfnet_berkeley/layers/backbone-layer4-0-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer4-0-conv2.bin", + "shelfnet_berkeley/layers/backbone-layer4-0-downsample-0.bin", + "shelfnet_berkeley/layers/backbone-layer4-1-conv1.bin", + "shelfnet_berkeley/layers/backbone-layer4-1-conv2.bin"}; + +const char *conv_out[] = { + "shelfnet_berkeley/layers/conv_out-conv-conv.bin", + "shelfnet_berkeley/layers/conv_out-conv_out.bin", + "shelfnet_berkeley/layers/conv_out16-conv-conv.bin", + "shelfnet_berkeley/layers/conv_out16-conv_out.bin", + "shelfnet_berkeley/layers/conv_out32-conv-conv.bin", + "shelfnet_berkeley/layers/conv_out32-conv_out.bin" + }; + +const char *decoder[] = { + "shelfnet_berkeley/layers/decoder-bottom-conv1.bin", + "shelfnet_berkeley/layers/decoder-bottom-conv12.bin", + "shelfnet_berkeley/layers/decoder-up_conv_list-0-conv-conv.bin", + "shelfnet_berkeley/layers/decoder-up_conv_list-0-conv_atten.bin", + "shelfnet_berkeley/layers/decoder-up_dense_list-0-conv.bin", + "shelfnet_berkeley/layers/decoder-up_conv_list-1-conv-conv.bin", + "shelfnet_berkeley/layers/decoder-up_conv_list-1-conv_atten.bin", + "shelfnet_berkeley/layers/decoder-up_dense_list-1-conv.bin" + }; + + +const char *ladder[] = { + "shelfnet_berkeley/layers/ladder-inconv-conv1.bin", + "shelfnet_berkeley/layers/ladder-inconv-conv12.bin", + "shelfnet_berkeley/layers/ladder-down_module_list-0-conv1.bin", + "shelfnet_berkeley/layers/ladder-down_module_list-0-conv12.bin", + "shelfnet_berkeley/layers/ladder-down_conv_list-0.bin", + + "shelfnet_berkeley/layers/ladder-down_module_list-1-conv1.bin", + "shelfnet_berkeley/layers/ladder-down_module_list-1-conv12.bin", + "shelfnet_berkeley/layers/ladder-down_conv_list-1.bin", + + "shelfnet_berkeley/layers/ladder-bottom-conv1.bin", + "shelfnet_berkeley/layers/ladder-bottom-conv12.bin", + + + + "shelfnet_berkeley/layers/ladder-up_conv_list-0-conv-conv.bin", + "shelfnet_berkeley/layers/ladder-up_conv_list-0-conv_atten.bin", + "shelfnet_berkeley/layers/ladder-up_dense_list-0-conv.bin", + + + "shelfnet_berkeley/layers/ladder-up_conv_list-1-conv-conv.bin", + "shelfnet_berkeley/layers/ladder-up_conv_list-1-conv_atten.bin", + "shelfnet_berkeley/layers/ladder-up_dense_list-1-conv.bin"}; + +const char *trans[] = { + "shelfnet_berkeley/layers/trans1-conv.bin", + "shelfnet_berkeley/layers/trans2-conv.bin", + "shelfnet_berkeley/layers/trans3-conv.bin"}; +int main() +{ + + downloadWeightsifDoNotExist(input_bin, "shelfnet_berkeley", "https://cloud.hipert.unimore.it/s/m92e7QdD9gYMF7f/download"); + + int classes = 20; + + // Network layout + tk::dnn::dataDim_t dim(1, 3, 1024, 1024, 1); + tk::dnn::Network net(dim); + + int bi = 0, di = 0, li = 0, ci = 0; + new tk::dnn::Conv2d(&net, 64, 7, 7, 2, 2, 3, 3, backbone[bi++], true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + tk::dnn::Layer* last = new tk::dnn::Pooling (&net, 3, 3, 2, 2, 1, 1, tk::dnn::POOLING_MAX); + + + + for(int i=0; i<2; ++i){ + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true); + new tk::dnn::Shortcut(&net, last); + last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + } + + std::vector features; + for(int i=0;i<3;++i){ + int out_channel = pow(2,7+i); + std::cout< up_out; + //bottom + new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, decoder[di++], true, false, 1, true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, decoder[di++], true, false, 1, true); + new tk::dnn::Shortcut(&net, last); + last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + up_out.push_back(last); + + for(int i=0; i<2; ++i){ + int out_channel = pow(2,7-i); + //up-conv + std::cout<output_dim.w, last->output_dim.h, last->output_dim.w, last->output_dim.h, 0, 0, tk::dnn::POOLING_AVERAGE); + new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, decoder[di++], true); + + tk::dnn::Layer* act = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_SIGMOID); + new tk::dnn::Route(&net, &last, 1); + new tk::dnn::Shortcut(&net, act, true); + + //interpolate + new tk::dnn::Resize(&net, 1,2,2); + new tk::dnn::Shortcut(&net, features[1-i]); + + //up-dense + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, decoder[di++], true); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + up_out.push_back(last); + } + + //LADDER + + std::vector down_out; + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Shortcut(&net, last); + new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + + for(int i=0; i<2;++i){ + int out_channel = pow(2,6+i); + tk::dnn::Layer* l_last = new tk::dnn::Shortcut(&net, up_out[2-i]); + + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Shortcut(&net, l_last); + l_last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + down_out.push_back(l_last); + + new tk::dnn::Conv2d (&net, out_channel*2, 3, 3, 2, 2, 1, 1, ladder[li++], false); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.0f); //should be ReLU + } + + new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Shortcut(&net, last); + last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + up_out.clear(); + up_out.push_back(last); + + for(int i=0; i<2; ++i){ + int out_channel = pow(2,7-i); + //up-conv + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + + new tk::dnn::Pooling(&net, last->output_dim.w, last->output_dim.h, last->output_dim.w, last->output_dim.h, 0, 0, tk::dnn::POOLING_AVERAGE); + new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, ladder[li++], true); + + tk::dnn::Layer* act = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_SIGMOID); + new tk::dnn::Route(&net, &last, 1); + new tk::dnn::Shortcut(&net, act, true); + + //interpolate + new tk::dnn::Resize(&net, 1,2,2); + new tk::dnn::Shortcut(&net, down_out[1-i]); + + // //up-dense + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + up_out.push_back(last); + } + + + // for(int i=2;i>=0;--i){ + // new tk::dnn::Route(&net, &up_out[i], 1); + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, conv_out[ci++], true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, classes, 3, 3, 1, 1, 1, 1, conv_out[ci++], false); + /*up_out[i] =*/ new tk::dnn::Resize(&net, classes, net.input_dim.h, net.input_dim.w, true, tk::dnn::ResizeMode_t::LINEAR); + // } + + new tk::dnn::Softmax(&net); + + const char *output_bin = "shelfnet_berkeley/debug/softmax.bin"; + + // Load input + dnnType *data; + dnnType *input_h; + readBinaryFile(input_bin, dim.tot(), &input_h, &data); + std::cout<<"Input:"< batch_frame; - std::vector batch_dnn_input; - int height = 0, width = 0; - - while(gRun) { - batch_dnn_input.clear(); - batch_frame.clear(); - - for(int bi=0; bi< n_batch; ++bi){ + cv::Mat frame; + while(gRun) { cap >> frame; if(!frame.data) break; height = frame.rows; width = frame.cols; - batch_frame.push_back(frame); - // this will be resized to the net format - batch_dnn_input.push_back(frame.clone()); - } - if(!frame.data) - break; - - //inference - segNN.update(batch_dnn_input, n_batch); - frame = segNN.draw(); + //inference + segNN.updateOriginal(frame); + if(show) + segNN.draw(); - if(n_batch == 1 && SAVE_RESULT) - resultVideo << frame; + if(SAVE_RESULT) + resultVideo << segNN.segmented[0]; + } } std::cout<<"segmentation end\n"; double mean = 0, mean_pre = 0, mean_post = 0; std::cout< splitted_frames; + int H, W, net_H, net_W; + int top = 0, bottom = 0, left = 0, right = 0; + std::vector> pos; + + { + TKDNN_TSTART + cv::Size original_size = frame.size(); + + frame.convertTo(frame, CV_32FC3, 1 / 255.0, 0); + H = frame.rows; + W = frame.cols; + net_H = netRT->input_dim.h; + net_W = netRT->input_dim.w; + + cv::Mat frame_cropped; + + if( H <= net_H && W <= net_W ){ // smaller size wrt network + top = (net_H - H)/2; + bottom = net_H - H - top ; + left = (net_W - W)/2; + right = net_W - W - left ; + cv::copyMakeBorder(frame, frame_cropped, top, bottom, left, right, cv::BORDER_CONSTANT, cv::Scalar(0,0,0) ); + splitted_frames.push_back(frame_cropped); + } + else{ //bigger size wrt network + + + if(H < net_H || W < net_W){ + if(H < net_H){ + top = (net_H - H)/2; + bottom = net_H - H - top ; + } + else{ + left = (net_W - W)/2; + right = net_W - W - left ; + } + cv::copyMakeBorder(frame, frame_cropped, top, bottom, left, right, cv::BORDER_CONSTANT, cv::Scalar(0,0,0)); + } + + for(int x=0; x+net_W<=W ;){ + for(int y=0; y+net_H <=H ; ){ + cv::Rect roi(x, y, net_W, net_H); + cv::Mat image_roi = frame(roi); + splitted_frames.push_back(image_roi); + pos.push_back(std::make_pair(x,y)); + + y += net_H; + if(y == H) + break; + if(y + net_H > H) y = H - net_H; + } + x += net_W; + if(x == W) + break; + if(x + net_W > W) x = W - net_W; + } + } + + tk::dnn::dataDim_t idim = netRT->input_dim; + + if(splitted_frames.size()> nBatches) + FatalError(std::to_string(splitted_frames.size()) + " min batches required"); + + for(int bi=0; bistream)); + normalize(input_d + idim.tot()*bi, idim.c, idim.h, idim.w, mean_d, stddev_d); + } + TKDNN_TSTOP + stats_pre.push_back(t_ns); + } + + tk::dnn::dataDim_t dim = netRT->input_dim; + dim.n = splitted_frames.size(); + { + if(TKDNN_VERBOSE) dim.print(); + TKDNN_TSTART + netRT->infer(dim, input_d); + TKDNN_TSTOP + if(TKDNN_VERBOSE) dim.print(); + stats.push_back(t_ns); + } + + dataDim_t odim = netRT->output_dim; + + std::vector out_img; + + { + TKDNN_TSTART + + for(int bi=0; bibuffersRT[1]+ netRT->buffersDIM[1].tot()*bi; + + matrixTranspose(cublasHandle, rt_out, tmpInputData_d, odim.c, odim.w*odim.h); + maxElem(tmpInputData_d, tmpOutData_d, odim.c, odim.h, odim.w); + checkCuda(cudaMemcpy(tmpOutData_h, tmpOutData_d, odim.w*odim.h * sizeof(float), cudaMemcpyDeviceToHost)); + + dataDim_t vdim = odim; + vdim.c = 1; + + cv::Mat colored; + + if(apply_colormap) + colored = vizData2Mat(tmpOutData_h, vdim, 1024, 0, 18); + else{ + cv::Mat colored_fp32 (cv::Size(odim.w, odim.h),CV_32FC1, tmpOutData_h); + colored_fp32.convertTo(colored, CV_8UC1); + } + out_img.push_back(colored); + } + + + cv::Mat seg(frame.size(), out_img[0].type()); + if(out_img.size() == 1) + { + cv::Rect roi(left, top, W, H); + seg = out_img[0](roi); + } + else{ + int bi=0; + + if(top == 0 && left == 0){ + + for(int i=0; i Date: Mon, 23 Nov 2020 13:07:54 +0100 Subject: [PATCH 022/162] Add computation of #parameters, #MACC, and max feature map size in the tests Signed-off-by: Micaela Verucchi --- include/tkDNN/Layer.h | 4 ++++ include/tkDNN/Network.h | 1 + src/Conv2d.cpp | 5 +++++ src/DeformConv2d.cpp | 6 ++++++ src/Layer.cpp | 2 ++ src/LayerWgs.cpp | 5 ++++- src/Network.cpp | 36 ++++++++++++++++++++++++++++++++++++ 7 files changed, 58 insertions(+), 1 deletion(-) diff --git a/include/tkDNN/Layer.h b/include/tkDNN/Layer.h index 790a431..91a6af0 100644 --- a/include/tkDNN/Layer.h +++ b/include/tkDNN/Layer.h @@ -54,6 +54,10 @@ public: int id = 0; bool final; //if the layer is the final one + uint n_params = 0; + uint feature_map_size = 0; + long unsigned MACC = 0; + std::string getLayerName() { layerType_t type = getLayerType(); diff --git a/include/tkDNN/Network.h b/include/tkDNN/Network.h index b78acff..6edf248 100644 --- a/include/tkDNN/Network.h +++ b/include/tkDNN/Network.h @@ -50,6 +50,7 @@ public: bool addLayer(Layer *l); void print(); const char *getNetworkRTName(const char *network_name); + void adjustFeatureMapSizeWithShortcuts(); cudnnDataType_t dataType; cudnnTensorFormat_t tensorFormat; diff --git a/src/Conv2d.cpp b/src/Conv2d.cpp index b57cf58..2901a70 100644 --- a/src/Conv2d.cpp +++ b/src/Conv2d.cpp @@ -166,6 +166,11 @@ Conv2d::Conv2d( Network *net, int out_ch, int kernelH, int kernelW, } initCUDNN(deConv); + if(this->groups != 1) + MACC = kernelH*kernelW*output_dim.c*output_dim.w*output_dim.h; + else + MACC = input_dim.c*kernelH*kernelW*output_dim.c*output_dim.w*output_dim.h; + // allocate warkspace if (ws_sizeInBytes!=0) { checkCuda( cudaMalloc(&workSpace, ws_sizeInBytes) ); diff --git a/src/DeformConv2d.cpp b/src/DeformConv2d.cpp index dbb71e1..826cbb0 100644 --- a/src/DeformConv2d.cpp +++ b/src/DeformConv2d.cpp @@ -73,6 +73,12 @@ DeformConv2d::DeformConv2d( Network *net, int out_ch, int deformable_group, int output_dim.c = out_ch; initCUDNN(); + + if(this->deformableGroup != 1) + MACC = kernelH*kernelW*output_dim.c*output_dim.w*output_dim.h; + else + MACC = input_dim.c*kernelH*kernelW*output_dim.c*output_dim.w*output_dim.h; + //allocate data for infer result checkCuda( cudaMalloc(&dstData, output_dim.tot()*sizeof(dnnType)) ); } diff --git a/src/Layer.cpp b/src/Layer.cpp index a355b90..7e13a48 100644 --- a/src/Layer.cpp +++ b/src/Layer.cpp @@ -18,6 +18,8 @@ Layer::Layer(Network *net) { if(!net->addLayer(this)) FatalError("Net reached max number of layers"); } + + feature_map_size = input_dim.tot() + output_dim.tot(); } Layer::~Layer() { diff --git a/src/LayerWgs.cpp b/src/LayerWgs.cpp index a761327..2f875a4 100644 --- a/src/LayerWgs.cpp +++ b/src/LayerWgs.cpp @@ -19,6 +19,8 @@ LayerWgs::LayerWgs(Network *net, int inputs, int outputs, int seek = 0; readBinaryFile(weights_path.c_str(), inputs*outputs*kh*kw*kl, &data_h, &data_d, seek); seek += inputs*outputs*kh*kw*kl; + n_params = seek; + this->additional_bias = additional_bias; if(additional_bias) { readBinaryFile(weights_path.c_str(), outputs, &bias2_h, &bias2_d, seek); @@ -26,15 +28,16 @@ LayerWgs::LayerWgs(Network *net, int inputs, int outputs, } readBinaryFile(weights_path.c_str(), outputs, &bias_h, &bias_d, seek); + seek += outputs; this->batchnorm = batchnorm; if(batchnorm) { - seek += outputs; readBinaryFile(weights_path.c_str(), outputs, &scales_h, &scales_d, seek); seek += outputs; readBinaryFile(weights_path.c_str(), outputs, &mean_h, &mean_d, seek); seek += outputs; readBinaryFile(weights_path.c_str(), outputs, &variance_h, &variance_d, seek); + seek += outputs; float eps = TKDNN_BN_MIN_EPSILON; diff --git a/src/Network.cpp b/src/Network.cpp index 7fa291f..2c3c8c7 100644 --- a/src/Network.cpp +++ b/src/Network.cpp @@ -96,6 +96,28 @@ dataDim_t Network::getOutputDim() { return layers[num_layers-1]->output_dim; } +void Network::adjustFeatureMapSizeWithShortcuts(){ + layerType_t layer_type; + int shortcutted_idx; + + for(int i=0; igetLayerType(); + if(layer_type == LAYER_SHORTCUT){ + shortcutted_idx = -1; + for(int j=0; j(layers[i])->backLayer == layers[j]){ + shortcutted_idx = j; + break; + } + } + if(shortcutted_idx == -1) + FatalError("Problem when computing featuer_map_size with shortcuts"); + for(int j=shortcutted_idx+1; jfeature_map_size += layers[shortcutted_idx]->output_dim.tot(); + } + } +} + void Network::print() { printCenteredTitle(" NETWORK MODEL ", '=', 60); @@ -106,10 +128,21 @@ void Network::print() { std::cout.width(16); std::cout<input_dim; dataDim_t out = layers[i]->output_dim; + tot_params += layers[i]->n_params; + tot_MACC += layers[i]->MACC; + if(layers[i]->feature_map_size> max_feature_map_size) + max_feature_map_size = layers[i]->feature_map_size; + std::cout.width(3); std::cout<getLayerName(); @@ -128,6 +161,9 @@ void Network::print() { } printCenteredTitle("", '=', 60); std::cout<<"\n"; + std::cout<<"N params: "< Date: Tue, 24 Nov 2020 12:37:33 +0100 Subject: [PATCH 023/162] Add shelfnet_mapillary, README_seg, resize of input Signed-off-by: Micaela Verucchi --- CMakeLists.txt | 3 + README_seg.md | 56 +++++ demo/demo/seg_demo.cpp | 48 ++++- include/tkDNN/NetworkViz.h | 4 +- include/tkDNN/SegmentationNN.h | 8 +- src/NetworkViz.cpp | 108 +++++++++- tests/shelfnet/shelfnet_mapillary.cpp | 295 ++++++++++++++++++++++++++ 7 files changed, 498 insertions(+), 24 deletions(-) create mode 100644 README_seg.md create mode 100644 tests/shelfnet/shelfnet_mapillary.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index 9e9c27c..c292300 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -110,6 +110,9 @@ target_link_libraries(test_shelfnet tkDNN) add_executable(test_shelfnet_berkeley tests/shelfnet/shelfnet_berkeley.cpp) target_link_libraries(test_shelfnet_berkeley tkDNN) +add_executable(test_shelfnet_mapillary tests/shelfnet/shelfnet_mapillary.cpp) +target_link_libraries(test_shelfnet_mapillary tkDNN) + # DEMOS add_executable(test_rtinference tests/test_rtinference/rtinference.cpp) target_link_libraries(test_rtinference tkDNN) diff --git a/README_seg.md b/README_seg.md new file mode 100644 index 0000000..a9afcd0 --- /dev/null +++ b/README_seg.md @@ -0,0 +1,56 @@ +# Semantic Segmentation with tkDNN + +Currently tkDNN supports only ShelfNet as semantic segmentation network. + +## Export weights from Shelfnet +To get the weights needed to run Mobilenet tests use [this](https://git.hipert.unimore.it/mverucchi/shelfnet) fork of a Pytorch implementation of Shelfnet network. + +``` +git clone https://git.hipert.unimore.it/mverucchi/shelfnet +cd shelfnet +cd ShelfNet18_realtime +conda env create --file shelfnet_env.yml +conda activate shelfnet +mkdir layer debug +python export.py +``` + + +## Run the demo + +To run the semantic segmentation demo follow these steps (example with shelfnet_mapillary): +``` +rm shelfnet_mapillary_fp32.rt # be sure to delete(or move) old tensorRT files +export TKDNN_BATCHSIZE=4 # be sure you have batch size > than 1 if you want to run inference on images bigger than 1024 +./test_shelfnet_mapillary # run the yolo test (is slow) +./demo shelfnet_mapillary_fp32.rt ../demo/yolo_test.mp4 1 15 +``` +In general the demo program takes the following parameters: +``` +./seg_demo +``` +where +* `````` is the rt file generated by a test +* ```<``` is the path to a video file or a camera input +* `````` number of batches to use in inference (N.B. you should first export TKDNN_BATCHSIZE to the required n_batches and create again the rt file for the network). +* ``````is the number of classes the network is trained on +* `````` if set to 0 the demo will not resize the input frames, but use it as it is, otherwise it will resize it. +* `````` is `````` is set to 1, then the input frames will be proportionally resized using `````` as width baseline. +* `````` if set to 0 the demo will not show the visualization but save the video into result.mp4 (if n-batches ==1) +* `````` if set to 0 (deafult) the demo will run, otherwise the evaluation of a dataset will run and the output of the segmentation will be saved. Attention: this is under development and paths are embedded, so change them in the code in advance. + +N.b. By default it is used FP32 inference + + + +## Existing tests and supported networks + +| Test Name | Network | Dataset | N Classes | Input size | Weights | +| :---------------- | :-------------------------------------------- | :-----------------------------------------------------------: | :-------: | :-----------: | :------------------------------------------------------------------------ | +| shelfnet | ShelfNet18_realtime1 | [Cityscapes](https://www.cityscapes-dataset.com/) | 19 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/mEDZMRJaGCFWSJF/download) | +| shelfnet_berkeley | ShelfNet18_realtime1 | [DeepDrive](https://bdd-data.berkeley.edu/) | 20 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/m92e7QdD9gYMF7f/download) | +| shelfnet_mapillary | ShelfNet18_realtime1 | [Mapillary Vistas](https://www.mapillary.com/dataset/vistas?pKey=aFWuj_m4nGoq3-tDz5KAqQ)* | 15 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/6WnZCKLjik7xrny/download) | + +1. Zhuang, Juntang, et al. "ShelfNet for fast semantic segmentation." Proceedings of the IEEE International Conference on Computer Vision Workshops. 2019. + +*. Mapillary Vistas has originally 66 classes, but we reduced them to 15 to improve the results on the categories of our interest. \ No newline at end of file diff --git a/demo/demo/seg_demo.cpp b/demo/demo/seg_demo.cpp index 35b4fa8..bd31a1a 100644 --- a/demo/demo/seg_demo.cpp +++ b/demo/demo/seg_demo.cpp @@ -48,22 +48,37 @@ int main(int argc, char *argv[]) { int n_classes = 19; if(argc > 4) n_classes = atoi(argv[4]); - bool show = true; + bool resize = false; if(argc > 5) - show = atoi(argv[5]); - bool write_pred = false; + resize = atoi(argv[5]); + int baseline_resize = 1024; if(argc > 6) - write_pred = atoi(argv[6]); - + baseline_resize = atoi(argv[6]); + bool show = true; + if(argc > 7) + show = atoi(argv[7]); + bool write_pred = false; + if(argc > 8) + write_pred = atoi(argv[8]); + if(resize && (baseline_resize < 0 || baseline_resize > 5000)) + FatalError("Problem with baseline resize") if(n_batch < 1 || n_batch > 64) FatalError("Batch dim not supported"); + std::string net_name; + removePathAndExtension(net, net_name); + bool mapillary_15 = false; //TODO change me pls + if(n_classes == 15 && net_name == "shelfnet_mapillary_fp32") + mapillary_15 = true; + + //net initialization tk::dnn::SegmentationNN segNN; segNN.init(net, n_classes, n_batch); int height = 0, width = 0; - + int basewidth=baseline_resize, hsize; + if(write_pred){ std::string gt_folder = "../demo/CityScapes_val/images/"; std::string images_names = "../demo/CityScapes_val/all_images.txt"; @@ -85,9 +100,16 @@ int main(int argc, char *argv[]) { cv::VideoWriter resultVideo; if(SAVE_RESULT) { - int w = cap.get(cv::CAP_PROP_FRAME_WIDTH); - int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT); - resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(1024, 1024)); + int w,h; + if(resize){ + w = basewidth; + h = int((float(cap.get(cv::CAP_PROP_FRAME_HEIGHT))*float(basewidth/float(cap.get(cv::CAP_PROP_FRAME_WIDTH))))); + } + else{ + w = cap.get(cv::CAP_PROP_FRAME_WIDTH); + h = cap.get(cv::CAP_PROP_FRAME_HEIGHT); + } + resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h)); } cv::Mat frame; @@ -95,11 +117,17 @@ int main(int argc, char *argv[]) { cap >> frame; if(!frame.data) break; + + if(resize){ + hsize = int((float(frame.rows)*float(basewidth/float(frame.cols)))); + cv::resize(frame, frame, cv::Size(basewidth, hsize)); + } + height = frame.rows; width = frame.cols; //inference - segNN.updateOriginal(frame); + segNN.updateOriginal(frame, true, mapillary_15); if(show) segNN.draw(); diff --git a/include/tkDNN/NetworkViz.h b/include/tkDNN/NetworkViz.h index 94468fc..2e2c3f4 100644 --- a/include/tkDNN/NetworkViz.h +++ b/include/tkDNN/NetworkViz.h @@ -5,8 +5,8 @@ namespace tk { namespace dnn { -cv::Mat vizFloat2colorMap(cv::Mat map, double min=0, double max=0); -cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int imgdim, double min=0, double max=0); +cv::Mat vizFloat2colorMap(cv::Mat map, double min=0, double max=0, bool mapillary_15=false); +cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int imgdim, double min=0, double max=0, bool mapillary_15=false); cv::Mat vizLayer2Mat(tk::dnn::Network *net, int layer, int imgdim = 1000); }} diff --git a/include/tkDNN/SegmentationNN.h b/include/tkDNN/SegmentationNN.h index edede43..ec6ac79 100644 --- a/include/tkDNN/SegmentationNN.h +++ b/include/tkDNN/SegmentationNN.h @@ -97,7 +97,7 @@ class SegmentationNN { * * @param bi batch index */ - void postprocess(const int bi=0, bool appy_colormap = true) { + void postprocess(const int bi=0, bool appy_colormap = true, bool mapillary_15=false) { dnnType *rt_out = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi; dataDim_t odim = netRT->output_dim; @@ -112,7 +112,7 @@ class SegmentationNN { cv::Mat colored; if(appy_colormap) - colored = vizData2Mat(tmpOutData_h, vdim, 1024, 0, 18); + colored = vizData2Mat(tmpOutData_h, vdim, 1024, 0, classes, mapillary_15); else{ cv::Mat colored_fp32 (cv::Size(odim.w, odim.h),CV_32FC1, tmpOutData_h); colored_fp32.convertTo(colored, CV_8UC1); @@ -234,7 +234,7 @@ class SegmentationNN { } } - void updateOriginal(cv::Mat frame, bool apply_colormap=true){ + void updateOriginal(cv::Mat frame, bool apply_colormap=true, bool mapillary_15=false){ std::vector splitted_frames; int H, W, net_H, net_W; @@ -347,7 +347,7 @@ class SegmentationNN { cv::Mat colored; if(apply_colormap) - colored = vizData2Mat(tmpOutData_h, vdim, 1024, 0, 18); + colored = vizData2Mat(tmpOutData_h, vdim, 1024, 0, classes, mapillary_15); else{ cv::Mat colored_fp32 (cv::Size(odim.w, odim.h),CV_32FC1, tmpOutData_h); colored_fp32.convertTo(colored, CV_8UC1); diff --git a/src/NetworkViz.cpp b/src/NetworkViz.cpp index 842a26e..4a3f113 100644 --- a/src/NetworkViz.cpp +++ b/src/NetworkViz.cpp @@ -6,22 +6,114 @@ namespace tk { namespace dnn { -cv::Mat vizFloat2colorMap(cv::Mat map,double min, double max) { +cv::Mat mapillary_15_map(cv::Mat adjMap){ + + // cv::imshow("test", adjMap); + // cv::waitKey(0); + cv::Mat M1(1, 256, CV_8UC1), M2(1, 256, CV_8UC1), M3(1, 256, CV_8UC1); + + M3.at(0)=165; + M2.at(0)=42; + M1.at(0)=45; + + M3.at(1)=196; + M2.at(1)=196; + M1.at(1)=196; + + M3.at(2)=90; + M2.at(2)=120; + M1.at(2)=150; + + M3.at(3)=128; + M2.at(3)=64; + M1.at(3)=128; + + M3.at(4)=70; + M2.at(4)=70; + M1.at(4)=70; + + M3.at(5)=220; + M2.at(5)=20; + M1.at(5)=60; + + M3.at(6)=255; + M2.at(6)=255; + M1.at(6)=255; + + M3.at(7)=107; + M2.at(7)=142; + M1.at(7)=35; + + M3.at(8)=70; + M2.at(8)=130; + M1.at(8)=180; + + M3.at(9)=220; + M2.at(9)=220; + M1.at(9)=220; + + M3.at(10)=153; + M2.at(10)=153; + M1.at(10)=153; + + M3.at(11)=128; + M2.at(11)=128; + M1.at(11)=128; + + M3.at(12)=119; + M2.at(12)=11; + M1.at(12)=32; + + M3.at(13)=0; + M2.at(13)=0; + M1.at(13)=142; + + for(int i=14;i<256;i++) + { + M1.at(i)=0; + M2.at(i)=0; + M3.at(i)=0; + } + + cv::Mat r1,r2,r3; + + cv::LUT(adjMap,M1,r1); + cv::LUT(adjMap,M2,r2); + cv::LUT(adjMap,M3,r3); + + std::vector planes; + planes.push_back(r1); + planes.push_back(r2); + planes.push_back(r3); + + cv::Mat dst; + cv::merge(planes,dst); + return dst; + + +} + +cv::Mat vizFloat2colorMap(cv::Mat map,double min, double max, bool mapillary_15) { if(min == 0 && max == 0) cv::minMaxIdx(map, &min, &max); cv::Mat adjMap; - // expand your range to 0..255. Similar to histEq(); - map.convertTo(adjMap,CV_8UC1, 255 / (max-min), -min); - //return adjMap; - cv::Mat falseColorsMap; - applyColorMap(adjMap, falseColorsMap, cv::COLORMAP_VIRIDIS); + + if(mapillary_15){ + map.convertTo(adjMap,CV_8UC1); + falseColorsMap = mapillary_15_map(adjMap); + } + else{ + // expand your range to 0..255. Similar to histEq(); + map.convertTo(adjMap,CV_8UC1, 255 / (max-min), -min); + applyColorMap(adjMap, falseColorsMap, cv::COLORMAP_JET); + } return falseColorsMap; } -cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int imgdim, double min, double max) { +cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int imgdim, double min, double max, bool mapillary_15) { dnnType *data = nullptr; // copy to CPU @@ -37,7 +129,7 @@ cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int imgdim, doub cv::Mat grid = cv::Mat(gridSize, CV_8UC3, cv::Scalar(0)); for(int i=0; i +#include +#include + +#include "tkdnn.h" +#include "NetworkViz.h" + + +const char *input_bin = "shelfnet_mapillary/debug/input.bin"; + +const char *backbone[] = { + "shelfnet_mapillary/layers/backbone-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer1-0-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer1-0-conv2.bin", + "shelfnet_mapillary/layers/backbone-layer1-1-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer1-1-conv2.bin", + "shelfnet_mapillary/layers/backbone-layer2-0-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer2-0-conv2.bin", + "shelfnet_mapillary/layers/backbone-layer2-0-downsample-0.bin", + "shelfnet_mapillary/layers/backbone-layer2-1-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer2-1-conv2.bin", + "shelfnet_mapillary/layers/backbone-layer3-0-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer3-0-conv2.bin", + "shelfnet_mapillary/layers/backbone-layer3-0-downsample-0.bin", + "shelfnet_mapillary/layers/backbone-layer3-1-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer3-1-conv2.bin", + "shelfnet_mapillary/layers/backbone-layer4-0-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer4-0-conv2.bin", + "shelfnet_mapillary/layers/backbone-layer4-0-downsample-0.bin", + "shelfnet_mapillary/layers/backbone-layer4-1-conv1.bin", + "shelfnet_mapillary/layers/backbone-layer4-1-conv2.bin"}; + +const char *conv_out[] = { + "shelfnet_mapillary/layers/conv_out-conv-conv.bin", + "shelfnet_mapillary/layers/conv_out-conv_out.bin", + "shelfnet_mapillary/layers/conv_out16-conv-conv.bin", + "shelfnet_mapillary/layers/conv_out16-conv_out.bin", + "shelfnet_mapillary/layers/conv_out32-conv-conv.bin", + "shelfnet_mapillary/layers/conv_out32-conv_out.bin" + }; + +const char *decoder[] = { + "shelfnet_mapillary/layers/decoder-bottom-conv1.bin", + "shelfnet_mapillary/layers/decoder-bottom-conv12.bin", + "shelfnet_mapillary/layers/decoder-up_conv_list-0-conv-conv.bin", + "shelfnet_mapillary/layers/decoder-up_conv_list-0-conv_atten.bin", + "shelfnet_mapillary/layers/decoder-up_dense_list-0-conv.bin", + "shelfnet_mapillary/layers/decoder-up_conv_list-1-conv-conv.bin", + "shelfnet_mapillary/layers/decoder-up_conv_list-1-conv_atten.bin", + "shelfnet_mapillary/layers/decoder-up_dense_list-1-conv.bin" + }; + + +const char *ladder[] = { + "shelfnet_mapillary/layers/ladder-inconv-conv1.bin", + "shelfnet_mapillary/layers/ladder-inconv-conv12.bin", + "shelfnet_mapillary/layers/ladder-down_module_list-0-conv1.bin", + "shelfnet_mapillary/layers/ladder-down_module_list-0-conv12.bin", + "shelfnet_mapillary/layers/ladder-down_conv_list-0.bin", + + "shelfnet_mapillary/layers/ladder-down_module_list-1-conv1.bin", + "shelfnet_mapillary/layers/ladder-down_module_list-1-conv12.bin", + "shelfnet_mapillary/layers/ladder-down_conv_list-1.bin", + + "shelfnet_mapillary/layers/ladder-bottom-conv1.bin", + "shelfnet_mapillary/layers/ladder-bottom-conv12.bin", + + + + "shelfnet_mapillary/layers/ladder-up_conv_list-0-conv-conv.bin", + "shelfnet_mapillary/layers/ladder-up_conv_list-0-conv_atten.bin", + "shelfnet_mapillary/layers/ladder-up_dense_list-0-conv.bin", + + + "shelfnet_mapillary/layers/ladder-up_conv_list-1-conv-conv.bin", + "shelfnet_mapillary/layers/ladder-up_conv_list-1-conv_atten.bin", + "shelfnet_mapillary/layers/ladder-up_dense_list-1-conv.bin"}; + +const char *trans[] = { + "shelfnet_mapillary/layers/trans1-conv.bin", + "shelfnet_mapillary/layers/trans2-conv.bin", + "shelfnet_mapillary/layers/trans3-conv.bin"}; +int main() +{ + + downloadWeightsifDoNotExist(input_bin, "shelfnet_mapillary", "https://cloud.hipert.unimore.it/s/6WnZCKLjik7xrny/download"); + + int classes = 15; + + // Network layout + tk::dnn::dataDim_t dim(1, 3, 1024, 1024, 1); + tk::dnn::Network net(dim); + + int bi = 0, di = 0, li = 0, ci = 0; + new tk::dnn::Conv2d(&net, 64, 7, 7, 2, 2, 3, 3, backbone[bi++], true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + tk::dnn::Layer* last = new tk::dnn::Pooling (&net, 3, 3, 2, 2, 1, 1, tk::dnn::POOLING_MAX); + + + + for(int i=0; i<2; ++i){ + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true); + new tk::dnn::Shortcut(&net, last); + last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + } + + std::vector features; + for(int i=0;i<3;++i){ + int out_channel = pow(2,7+i); + std::cout< up_out; + //bottom + new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, decoder[di++], true, false, 1, true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, decoder[di++], true, false, 1, true); + new tk::dnn::Shortcut(&net, last); + last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + up_out.push_back(last); + + for(int i=0; i<2; ++i){ + int out_channel = pow(2,7-i); + //up-conv + std::cout<output_dim.w, last->output_dim.h, last->output_dim.w, last->output_dim.h, 0, 0, tk::dnn::POOLING_AVERAGE); + new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, decoder[di++], true); + + tk::dnn::Layer* act = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_SIGMOID); + new tk::dnn::Route(&net, &last, 1); + new tk::dnn::Shortcut(&net, act, true); + + //interpolate + new tk::dnn::Resize(&net, 1,2,2); + new tk::dnn::Shortcut(&net, features[1-i]); + + //up-dense + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, decoder[di++], true); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + up_out.push_back(last); + } + + //LADDER + + std::vector down_out; + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Shortcut(&net, last); + new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + + for(int i=0; i<2;++i){ + int out_channel = pow(2,6+i); + tk::dnn::Layer* l_last = new tk::dnn::Shortcut(&net, up_out[2-i]); + + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Shortcut(&net, l_last); + l_last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + down_out.push_back(l_last); + + new tk::dnn::Conv2d (&net, out_channel*2, 3, 3, 2, 2, 1, 1, ladder[li++], false); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.0f); //should be ReLU + } + + new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true); + new tk::dnn::Shortcut(&net, last); + last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU); + up_out.clear(); + up_out.push_back(last); + + for(int i=0; i<2; ++i){ + int out_channel = pow(2,7-i); + //up-conv + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + + new tk::dnn::Pooling(&net, last->output_dim.w, last->output_dim.h, last->output_dim.w, last->output_dim.h, 0, 0, tk::dnn::POOLING_AVERAGE); + new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, ladder[li++], true); + + tk::dnn::Layer* act = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_SIGMOID); + new tk::dnn::Route(&net, &last, 1); + new tk::dnn::Shortcut(&net, act, true); + + //interpolate + new tk::dnn::Resize(&net, 1,2,2); + new tk::dnn::Shortcut(&net, down_out[1-i]); + + // //up-dense + new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true); + last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + up_out.push_back(last); + } + + + // for(int i=2;i>=0;--i){ + // new tk::dnn::Route(&net, &up_out[i], 1); + new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, conv_out[ci++], true); + new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01); + new tk::dnn::Conv2d (&net, classes, 3, 3, 1, 1, 1, 1, conv_out[ci++], false); + /*up_out[i] =*/ new tk::dnn::Resize(&net, classes, net.input_dim.h, net.input_dim.w, true, tk::dnn::ResizeMode_t::LINEAR); + // } + + new tk::dnn::Softmax(&net); + + const char *output_bin = "shelfnet_mapillary/debug/softmax.bin"; + + // Load input + dnnType *data; + dnnType *input_h; + readBinaryFile(input_bin, dim.tot(), &input_h, &data); + std::cout<<"Input:"<output_dim.print(); + + dnnType *out, *out_h; + int odim = outs[i]->output_dim.tot(); + readBinaryFile(output_bin[i], odim, &out_h, &out); + + dnnType *cudnn_out, *rt_out; + cudnn_out = outs[i]->dstData; + rt_out = (dnnType *)netRT.buffersRT[i+out_count]; + // there is the maxpool. It isn't an output but it is necessary for the process section + if(i==0) + out_count ++; + + std::cout<<"CUDNN vs correct"; + ret_cudnn |= checkResult(odim, cudnn_out, out) == 0 ? 0: ERROR_CUDNN; + std::cout<<"TRT vs correct"; + ret_tensorrt |= checkResult(odim, rt_out, out) == 0 ? 0 : ERROR_TENSORRT; + std::cout<<"CUDNN vs TRT "; + ret_cudnn_tensorrt |= checkResult(odim, cudnn_out, rt_out) == 0 ? 0 : ERROR_CUDNNvsTENSORRT; + } + return ret_cudnn | ret_tensorrt | ret_cudnn_tensorrt; +} -- 2.52.0 From 48ecebe6dd187b440f1f6668de4d5634ddba9357 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Mon, 7 Dec 2020 18:45:29 +0100 Subject: [PATCH 028/162] Add CenterTrack pre, post, visualization and demo. Signed-off-by: Davide Sapienza --- demo/demo/demo3D.cpp | 5 + include/tkDNN/CenternetDetection3DTrack.h | 176 +++++ src/CenternetDetection3DTrack.cpp | 862 ++++++++++++++++++++++ 3 files changed, 1043 insertions(+) create mode 100644 include/tkDNN/CenternetDetection3DTrack.h create mode 100644 src/CenternetDetection3DTrack.cpp diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp index e838395..1a93206 100644 --- a/demo/demo/demo3D.cpp +++ b/demo/demo/demo3D.cpp @@ -5,6 +5,7 @@ #include #include "CenternetDetection3D.h" +#include "CenternetDetection3DTrack.h" bool gRun; bool SAVE_RESULT = false; @@ -34,6 +35,7 @@ int main(int argc, char *argv[]) { n_classes = atoi(argv[4]); tk::dnn::CenternetDetection3D cnet; + tk::dnn::CenternetDetection3DTrack ctrack; tk::dnn::DetectionNN3D *detNN; @@ -42,6 +44,9 @@ int main(int argc, char *argv[]) { case 'c': detNN = &cnet; break; + case 't': + detNN = &ctrack; + break; default: FatalError("Network type not allowed (3rd parameter)\n"); } diff --git a/include/tkDNN/CenternetDetection3DTrack.h b/include/tkDNN/CenternetDetection3DTrack.h new file mode 100644 index 0000000..5149d70 --- /dev/null +++ b/include/tkDNN/CenternetDetection3DTrack.h @@ -0,0 +1,176 @@ +#ifndef CENTERNETDETECTION3DTRACK_H +#define CENTERNETDETECTION3DTRACK_H + +#include "kernels.h" +#include "utils.h" +#include "tkdnn.h" +#include +#include "opencv2/opencv.hpp" +#include +#include +#include // std::iota +#include // std::sort + +#include "DetectionNN3D.h" + +#include "kernelsThrust.h" + + +namespace tk { namespace dnn { + +struct detectionRes +{ + float score; + int cl; + cv::Mat ct, tr, bb0, bb1; + float dep; + float dim[3]; + float alpha; + float x,y,z; + float rot_y; + detectionRes() : ct(cv::Mat(cv::Size(1,2), CV_32F)), + tr(cv::Mat(cv::Size(1,2), CV_32F)), + bb0(cv::Mat(cv::Size(1,2), CV_32F)), + bb1(cv::Mat(cv::Size(1,2), CV_32F)) { } + ~detectionRes() { + ct.release(); + tr.release(); + bb0.release(); + bb1.release(); + } +}; + +struct trackingRes +{ + struct detectionRes det_res; + int tracking_id; + int age; + int active; + int color; +}; + +class CenternetDetection3DTrack : public DetectionNN3D +{ +private: + tk::dnn::dataDim_t dim; + tk::dnn::dataDim_t dim2; + tk::dnn::dataDim_t dim_hm; + tk::dnn::dataDim_t dim_wh; + tk::dnn::dataDim_t dim_reg; + tk::dnn::dataDim_t dim_track; + tk::dnn::dataDim_t dim_dep; + tk::dnn::dataDim_t dim_rot; + tk::dnn::dataDim_t dim_dim; + tk::dnn::dataDim_t dim_amodel_offset; + + /* preprocessing */ + #ifdef OPENCV_CUDACONTRIB + float *mean_d; + float *stddev_d; + #else + cv::Vec mean; + cv::Vec stddev; + dnnType *input; + #endif + float *d_ptrs; + + cv::Mat src; + cv::Mat dst; + cv::Mat dst2; + cv::Mat trans, trans2, trans_out; + + /* pre inf */ + bool iter0; + dnnType *input_pre_inf_d; + bool test_pre_inf = true; + dnnType *img_d, *hm_d; + tk::dnn::dataDim_t dim_in0; + tk::dnn::dataDim_t dim_in1; + dnnType *out_d; + + + /* postprocessing */ + int K = 100; + int width = 128;//56; // TODO + + // pointer used in the kernels + float *src_out; + int *ids_out; + + float *topk_scores; + int *topk_inds_; + float *topk_ys_; + float *topk_xs_; + int *ids_d, *ids_; + + float *ones; + + float *scores, *scores_d; + int *clses, *clses_d; + int *topk_inds_d; + float *topk_ys_d; + float *topk_xs_d; + int *inttopk_xs_d, *inttopk_ys_d; + + float *bbx0, *bby0, *bbx1, *bby1; + float *bbx0_d, *bby0_d, *bbx1_d, *bby1_d; + + int *intxs, *intys; + + float *track, *dep, *rot, *dim_, *wh, *amodel_offset; + float *track_d, *dep_d, *rot_d, *dim_d, *wh_d, *amodel_offset_d; + + float *target_coords; + + /* visualization */ + cv::Mat r; + cv::Mat calibs; + cv::Mat corners, pts3DHomo; + + std::vector> face_id; + cv::Scalar tr_colors[256]; + bool view2d = false; + + //processing + struct threshold op; + float out_thresh = 0.1; + float new_thresh = 0.3; + float vis_thresh = 0.3; + float peakThreshold = 0.2; + float centerThreshold = 0.3; //default 0.5 + + + //detections + std::vector det_res; + int count_det; + //tracks + std::vector tr_res; + int count_tr; + int track_id=0; + + + bool init_preprocessing(); + bool init_pre_inf(); + bool init_postprocessing(); + bool init_visualization(const int n_classes); + void pre_inf(); + void _get_additional_inputs(); + cv::Mat transform_preds_with_trans(float x1, float x2); + void tracking(); + +public: + tk::dnn::Network *pre_phase_net = nullptr; + CenternetDetection3DTrack() {}; + ~CenternetDetection3DTrack() {}; + bool init(const std::string& tensor_path, const int n_classes=3); + void preprocess(cv::Mat &frame); + void postprocess(); + cv::Mat draw(cv::Mat &frame); +}; + + +} // namespace dnn +} // namespace tk + + +#endif /*CENTERNETDETECTION3DTRACK_H*/ \ No newline at end of file diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp new file mode 100644 index 0000000..35488f7 --- /dev/null +++ b/src/CenternetDetection3DTrack.cpp @@ -0,0 +1,862 @@ +#include "CenternetDetection3DTrack.h" + + +namespace tk { namespace dnn { + +bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n_classes){ + std::cout<<(tensor_path).c_str()<<"\n"; + netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); + + dim = netRT->input_dim; + dim.c = 3; + + init_preprocessing(); + init_pre_inf(); + init_postprocessing(); + init_visualization(n_classes); + + count_tr = 0; +} + +bool CenternetDetection3DTrack::init_preprocessing(){ + //image transformation + src = cv::Mat(cv::Size(2,3), CV_32F); + dst = cv::Mat(cv::Size(2,3), CV_32F); + dst2 = cv::Mat(cv::Size(2,3), CV_32F); + trans = cv::Mat(cv::Size(3,2), CV_32F); + trans2 = cv::Mat(cv::Size(3,2), CV_32F); + trans_out = cv::Mat(cv::Size(3,2), CV_32F); + + dst2.at(0,0)=width * 0.5; + dst2.at(0,1)=width * 0.5; + dst2.at(1,0)=width * 0.5; + dst2.at(1,1)=width * 0.5 + width * -0.5; + + dst2.at(2,0)=dst2.at(1,0) + (-dst2.at(0,1)+dst2.at(1,1) ); + dst2.at(2,1)=dst2.at(1,1) + (dst2.at(0,0)-dst2.at(1,0) ); + + +#ifdef OPENCV_CUDACONTRIB + + checkCuda( cudaMalloc(&mean_d, 3 * sizeof(float)) ); + checkCuda( cudaMalloc(&stddev_d, 3 * sizeof(float)) ); + float mean[3] = {0.40789655, 0.44719303, 0.47026116}; + float stddev[3] = {0.2886383, 0.27408165, 0.27809834}; + + checkCuda(cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice)); + checkCuda(cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice)); +#else + checkCuda(cudaMallocHost(&input, sizeof(dnnType)*dim.tot())); + mean << 0.40789655, 0.44719303, 0.47026116; + stddev << 0.2886383, 0.27408165, 0.27809834; + +#endif + + checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot())); + checkCuda(cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot())); + checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) ); +} + +bool CenternetDetection3DTrack::init_pre_inf(){ + // initial steps: the first part of the network + const char *pre_img_conv1_bin = "/home/davide/Projects/repos/tkDNN/build/dla34_cnet3d_track/layers/base-pre_img_layer-0.bin"; + const char *pre_hm_conv1_bin = "dla34_cnet3d_track/layers/base-pre_hm_layer-0.bin"; + const char *conv1_bin = "dla34_cnet3d_track/layers/base-base_layer-0.bin"; + const char *conv2_bin = "dla34_cnet3d_track/layers/base-level0-0.bin"; + dim_in0 = tk::dnn::dataDim_t(1, 3, 512, 512, 1); + dim_in1 = tk::dnn::dataDim_t(1, 1, 512, 512, 1); + + + checkCuda( cudaMalloc(&out_d, netRT->input_dim.tot()*sizeof(dnnType)) ); + checkCuda( cudaMalloc(&img_d, dim_in0.tot()*sizeof(dnnType)) ); + checkCuda( cudaMalloc(&hm_d, dim_in1.tot()*sizeof(dnnType)) ); + // init to zeros hm + dnnType *hm_h; + checkCuda( cudaMallocHost(&hm_h, 1 * dim.h * dim.w*sizeof(dnnType)) ); + for(int i=0; i<1 * dim.h * dim.w; i++) + hm_h[i]=0.0f; + checkCuda( cudaMemcpy(hm_d, hm_h, 1 * dim.h * dim.w * sizeof(dnnType), cudaMemcpyHostToDevice) ); + checkCuda( cudaFreeHost(hm_h) ); + dnnType *i0_h, *i1_h, *i2_h; + // dnnType *i0_d, *i1_d, *i2_d; + + // const char *input_bin = "dla34_cnet3d_track/debug/input.bin"; + // const char *pre_img_bin = "dla34_cnet3d_track/debug/pre_imgages.bin"; + // const char *pre_hm_bin = "dla34_cnet3d_track/debug/pre_hms.bin"; + // readBinaryFile(pre_img_bin, dim_in0.tot(), &i0_h, &img_d); + // readBinaryFile(pre_hm_bin, dim_in1.tot(), &i1_h, &hm_d); + // readBinaryFile(input_bin, dim_in0.tot(), &i2_h, &input_pre_inf_d); + + pre_phase_net = new tk::dnn::Network(dim_in0); + //pre-img + tk::dnn::Input *in_pre_img = new tk::dnn::Input(pre_phase_net, dim_in0, img_d); + tk::dnn::Conv2d *pre_img_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_img_conv1_bin, true); + tk::dnn::Activation *pre_img_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + //pre-hm + tk::dnn::Input *in_pre_hm = new tk::dnn::Input(pre_phase_net, dim_in1, hm_d); + tk::dnn::Conv2d *pre_hm_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_hm_conv1_bin, true); + tk::dnn::Activation *pre_hm_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + // image input + tk::dnn::Input *input_image = new tk::dnn::Input(pre_phase_net, dim_in0, input_pre_inf_d); + tk::dnn::Conv2d *conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, conv1_bin, true); + tk::dnn::Activation *relu1 = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + + tk::dnn::Shortcut *s0_input = new tk::dnn::Shortcut(pre_phase_net, pre_img_relu); + tk::dnn::Shortcut *s1_input = new tk::dnn::Shortcut(pre_phase_net, pre_hm_relu); + // output data + out_d = s1_input->dstData; + //print network model + pre_phase_net->print(); + + iter0=true; // in the first iteration the last input is equal to the current input. + return true; +} + +bool CenternetDetection3DTrack::init_postprocessing(){ + srand(0); //seed = 0 for random colors + + dim_hm = tk::dnn::dataDim_t(1, 10, 128, 128, 1); + dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_reg = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_track = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_dep = tk::dnn::dataDim_t(1, 1, 128, 128, 1); + dim_rot = tk::dnn::dataDim_t(1, 8, 128, 128, 1); + dim_dim = tk::dnn::dataDim_t(1, 3, 128, 128, 1); + dim_amodel_offset = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + + checkCuda( cudaMalloc(&topk_scores, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&topk_inds_, dim_hm.c * K *sizeof(int)) ); + checkCuda( cudaMalloc(&topk_ys_, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&topk_xs_, dim_hm.c * K *sizeof(float)) ); + checkCuda( cudaMalloc(&ids_d, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) ); + checkCuda( cudaMallocHost(&ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) ); + for(int i =0; i(0,0) = 633.0; + calibs.at(0,1) = 0.0; + calibs.at(0,2) = 0.0; //w/2 + calibs.at(0,3) = 0.0; + calibs.at(1,0) = 0.0; + calibs.at(1,1) = 633.0; + calibs.at(1,2) = 0.0; //h/2 + calibs.at(1,3) = 0.0; + calibs.at(2,0) = 0.0; + calibs.at(2,1) = 0.0; + calibs.at(2,2) = 1.0; + calibs.at(2,3) = 0.0; + + // Alloc array used in the kernel + checkCuda( cudaMalloc(&src_out, K *sizeof(float)) ); + checkCuda( cudaMalloc(&ids_out, K *sizeof(int)) ); +} + +bool CenternetDetection3DTrack::init_visualization(const int n_classes){ + classes = n_classes; + // const char *kitti_class_name[] = { + // "person", "car", "bicycle"}; + // classesNames = std::vector(kitti_class_name, std::end( kitti_class_name)); + + const char *class_name[] = {"car", "truck", "bus", "trailer", "construction_vehicle", "pedestrian", + "motorcycle", "bicycle", "traffic_cone", "barrier"}; + classesNames = std::vector(class_name, std::end( class_name)); + + // const char *coco_class_name[] = { + // "person", "bicycle", "car", "motorcycle", "airplane", + // "bus", "train", "truck", "boat", "traffic light", "fire hydrant", + // "stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse", + // "sheep", "cow", "elephant", "bear", "zebra", "giraffe", "backpack", + // "umbrella", "handbag", "tie", "suitcase", "frisbee", "skis", + // "snowboard", "sports ball", "kite", "baseball bat", "baseball glove", + // "skateboard", "surfboard", "tennis racket", "bottle", "wine glass", + // "cup", "fork", "knife", "spoon", "bowl", "banana", "apple", "sandwich", + // "orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake", + // "chair", "couch", "potted plant", "bed", "dining table", "toilet", "tv", + // "laptop", "mouse", "remote", "keyboard", "cell phone", "microwave", + // "oven", "toaster", "sink", "refrigerator", "book", "clock", "vase", + // "scissors", "teddy bear", "hair drier", "toothbrush" + // }; + // classesNames = std::vector(coco_class_name, std::end( coco_class_name)); + + for(int c=0; c(0,1) = 0.0; + r.at(1,0) = 0.0; + r.at(1,1) = 1.0; + r.at(1,2) = 0.0; + r.at(2,1) = 0.0; + + corners = cv::Mat(cv::Size(8,3), CV_32F); + corners.at(1,0) = 0.0; + corners.at(1,1) = 0.0; + corners.at(1,2) = 0.0; + corners.at(1,3) = 0.0; + + pts3DHomo = cv::Mat(cv::Size(8,4), CV_32F); + pts3DHomo.at(3,0) = 1.0; + pts3DHomo.at(3,1) = 1.0; + pts3DHomo.at(3,2) = 1.0; + pts3DHomo.at(3,3) = 1.0; + pts3DHomo.at(3,4) = 1.0; + pts3DHomo.at(3,5) = 1.0; + pts3DHomo.at(3,6) = 1.0; + pts3DHomo.at(3,7) = 1.0; + + face_id.push_back({0,1,5,4}); + face_id.push_back({1,2,6, 5}); + face_id.push_back({2,3,7,6}); + face_id.push_back({3,0,4,7}); + // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); +} + +void CenternetDetection3DTrack::_get_additional_inputs(){ + //None no additional input +} + +void CenternetDetection3DTrack::pre_inf(){ + TKDNN_TSTART + tk::dnn::dataDim_t dim_aus; + pre_phase_net->infer(dim_aus, nullptr); + TKDNN_TSTOP + checkCuda( cudaDeviceSynchronize() ); + checkCuda( cudaMemcpy(input_d, pre_phase_net->layers[pre_phase_net->num_layers-1]->dstData, netRT->input_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice) ); + checkCuda( cudaDeviceSynchronize() ); +} + +void CenternetDetection3DTrack::preprocess(cv::Mat &frame){ + // -----------------------------------pre-process ------------------------------------------ + + cv::Size sz = originalSize; + cv::Size sz_old; + float scale = 1.0; + float new_height = sz.height * scale; + float new_width = sz.width * scale; + if(sz.height != sz_old.height && sz.width != sz_old.width){ + calibs.at(0,2) = new_width / 2.0f; + calibs.at(1,2) = new_height /2.0f; + float c[] = {new_width / 2.0f, new_height /2.0f}; + float s[] = {dim.w, dim.h}; + // float s = new_width >= new_height ? new_width : new_height; + // ----------- get_affine_transform + // rot_rad = pi * 0 / 100 --> 0 + dim.print(); + src.at(0,0)=c[0]; + src.at(0,1)=c[1]; + src.at(1,0)=c[0]; + src.at(1,1)=c[1] + s[0] * -0.5; + dst.at(0,0)=dim.w * 0.5; + dst.at(0,1)=dim.h * 0.5; + dst.at(1,0)=dim.w * 0.5; + dst.at(1,1)=dim.h * 0.5 + dim.w * -0.5; + + src.at(2,0)=src.at(1,0) + (-src.at(0,1)+src.at(1,1) ); + src.at(2,1)=src.at(1,1) + (src.at(0,0)-src.at(1,0) ); + dst.at(2,0)=dst.at(1,0) + (-dst.at(0,1)+dst.at(1,1) ); + dst.at(2,1)=dst.at(1,1) + (dst.at(0,0)-dst.at(1,0) ); + + + trans = cv::getAffineTransform( src, dst ); + trans2 = cv::getAffineTransform( dst2, src ); + trans2.convertTo(trans_out, CV_32F); + } + sz_old = sz; +#ifdef OPENCV_CUDACONTRIB + std::cout<<"OPENCV CPMTROB\n"; + cv::cuda::GpuMat im_Orig; + cv::cuda::GpuMat imageF1_d, imageF2_d; + + im_Orig = cv::cuda::GpuMat(frame); + // cv::cuda::resize (im_Orig, imageF1_d, cv::Size(new_width, new_height)); + imageF1_d = im_Orig; + checkCuda( cudaDeviceSynchronize() ); + + sz = imageF1_d.size(); + + cv::cuda::warpAffine(imageF1_d, imageF2_d, trans, cv::Size(dim.w, dim.h), cv::INTER_LINEAR ); + checkCuda( cudaDeviceSynchronize() ); + + imageF2_d.convertTo(imageF1_d, CV_32FC3, 1/255.0); + checkCuda( cudaDeviceSynchronize() ); + + dim2 = dim; + cv::cuda::GpuMat bgr[3]; + cv::cuda::split(imageF1_d,bgr);//split source + + for(int i=0; i(0,0) = x1; + target_coords.at(0,1) = x2; + target_coords.at(0,2) = 1.0; + return trans_out * target_coords; +} + +void CenternetDetection3DTrack::tracking(){ + + float item_size[count_det]; + int item_cl[count_det]; + float dets[2*count_det]; + for(int i=0; i(0,0) - det_res[i].bb0.at(0,0)) * + (det_res[i].bb1.at(0,1) - det_res[i].bb0.at(0,1)); + item_cl[i] = det_res[i].cl; + dets[i*2] = det_res[i].ct.at(0,0); + dets[i*2+1] = det_res[i].ct.at(0,1); + } + + float track_size[count_tr]; + int track_cl[count_tr]; + float tracks[2*count_tr]; + for(int i=0; i(0,0) - tr_res[i].det_res.bb0.at(0,0)) * + (tr_res[i].det_res.bb1.at(0,1) - tr_res[i].det_res.bb0.at(0,1)); + track_cl[i] = tr_res[i].det_res.cl; + tracks[i*2] = tr_res[i].det_res.ct.at(0,0); + tracks[i*2+1] = tr_res[i].det_res.ct.at(0,1); + } + float dist[count_tr*count_det]; + bool invalid; + for(int i=0; i track_size[i] || dist[j*count_tr+i] > item_size[j] || item_cl[j] != track_cl[i]; + dist[j*count_tr+i] = dist[j*count_tr+i] + invalid * (1 << 18); + } + } + int matched_indices[2*count_tr]; + float min_tr; + int min_idtr=-1; + for(int i=0; i new_tr_res; + int id_new_tr=0; + for(int i=0; i new_thresh) { + count_tr_ ++; + struct trackingRes new_tr_res_; + new_tr_res_.det_res.score = det_res[i].score; + new_tr_res_.det_res.cl = det_res[i].cl; + new_tr_res_.det_res.ct = det_res[i].ct; + new_tr_res_.det_res.tr = det_res[i].tr; + new_tr_res_.det_res.bb0 = det_res[i].bb0; + new_tr_res_.det_res.bb1 = det_res[i].bb1; + new_tr_res_.det_res.dep = det_res[i].dep; + new_tr_res_.det_res.dim[0] = det_res[i].dim[0]; + new_tr_res_.det_res.dim[1] = det_res[i].dim[1]; + new_tr_res_.det_res.dim[2] = det_res[i].dim[2]; + new_tr_res_.det_res.alpha = det_res[i].alpha; + new_tr_res_.det_res.x = det_res[i].x; + new_tr_res_.det_res.y = det_res[i].y; + new_tr_res_.det_res.z = det_res[i].z; + new_tr_res_.det_res.rot_y = det_res[i].rot_y; + new_tr_res_.tracking_id = track_id++; + new_tr_res_.age = 1; + new_tr_res_.active = 1; + new_tr_res_.color = rand() % 256; + tr_res.push_back(new_tr_res_); + } + } + count_tr = count_tr_; + + if(track_id==1000) + track_id=0; + det_res.clear(); + +} + +void CenternetDetection3DTrack::postprocess(){ + dnnType *rt_out[9]; + rt_out[0] = (dnnType *)netRT->buffersRT[1]; + rt_out[1] = (dnnType *)netRT->buffersRT[2]; + rt_out[2] = (dnnType *)netRT->buffersRT[3]; + rt_out[3] = (dnnType *)netRT->buffersRT[4]; + rt_out[4] = (dnnType *)netRT->buffersRT[5]; + rt_out[5] = (dnnType *)netRT->buffersRT[6]; + rt_out[6] = (dnnType *)netRT->buffersRT[7]; + rt_out[7] = (dnnType *)netRT->buffersRT[8]; + rt_out[8] = (dnnType *)netRT->buffersRT[9]; + + // ------------------------------------ process -------------------------------------------- + + activationSIGMOIDForward(rt_out[0], rt_out[0], dim_hm.tot()); + checkCuda( cudaDeviceSynchronize() ); + + // output['dep'] = 1. / (output['dep'].sigmoid() + 1e-6) - 1. + activationSIGMOIDForward(rt_out[5], rt_out[5], dim_dep.tot()); + checkCuda( cudaDeviceSynchronize() ); + transformDep(ones, ones + dim_dep.tot(), rt_out[5], rt_out[5] + dim_dep.tot()); + checkCuda( cudaDeviceSynchronize() ); + + // nms + subtractWithThreshold(rt_out[0], rt_out[0] + dim_hm.tot(), rt_out[1], rt_out[0], op); + + // ----------- nms end + // ----------- topk + + if(K > dim_hm.h * dim_hm.w){ + printf ("Error topk (K is too large)\n"); + return; + } + + checkCuda( cudaMemcpy(ids_d, ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyHostToDevice) ); + + sort(rt_out[0],rt_out[0]+dim_hm.tot(),ids_d); + checkCuda( cudaDeviceSynchronize() ); + + topk(rt_out[0], ids_d, K, scores_d, topk_inds_d, topk_ys_d, topk_xs_d); + checkCuda( cudaDeviceSynchronize() ); + + checkCuda( cudaMemcpy(scores, scores_d, K *sizeof(float), cudaMemcpyDeviceToHost) ); + + topKxyclasses(topk_inds_d, topk_inds_d+K, K, width, dim_hm.w*dim_hm.h, clses_d, inttopk_xs_d, inttopk_ys_d); + checkCuda( cudaDeviceSynchronize() ); + checkCuda( cudaMemcpy(topk_xs_d, (float *)inttopk_xs_d, K*sizeof(float), cudaMemcpyDeviceToDevice) ); + checkCuda( cudaMemcpy(topk_ys_d, (float *)inttopk_ys_d, K*sizeof(float), cudaMemcpyDeviceToDevice) ); + + checkCuda( cudaMemcpy(intxs, inttopk_xs_d, K * sizeof(int), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(intys, inttopk_ys_d, K * sizeof(int), cudaMemcpyDeviceToHost) ); + + checkCuda( cudaMemcpy(clses, clses_d, K*sizeof(int), cudaMemcpyDeviceToHost) ); + + // ----------- topk end + + topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3], src_out, ids_out); + checkCuda( cudaDeviceSynchronize() ); + + bboxes(topk_inds_d, K, dim_wh.h*dim_wh.w, topk_xs_d, topk_ys_d, rt_out[2], bbx0_d, bbx1_d, bby0_d, bby1_d, src_out, ids_out); + checkCuda( cudaDeviceSynchronize() ); + checkCuda( cudaMemcpy(bbx0, bbx0_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(bby0, bby0_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(bbx1, bbx1_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + checkCuda( cudaMemcpy(bby1, bby1_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); + + //regression heads + // ['tracking', 'dep', 'rot', 'dim', 'amodel_offset', + // 'nuscenes_att', 'velocity'] + getRecordsFromTopKId(topk_inds_d, K, dim_track.c, dim_track.h * dim_track.w, rt_out[4], track_d, ids_out); + checkCuda( cudaMemcpy(track, track_d, K * dim_track.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_dep.c, dim_dep.h * dim_dep.w, rt_out[5], dep_d, ids_out); + checkCuda( cudaMemcpy(dep, dep_d, K * dim_dep.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_rot.c, dim_rot.h * dim_rot.w, rt_out[6], rot_d, ids_out); + checkCuda( cudaMemcpy(rot, rot_d, K * dim_rot.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_dim.c, dim_dim.h * dim_dim.w, rt_out[7], dim_d, ids_out); + checkCuda( cudaMemcpy(dim_, dim_d, K * dim_dim.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + getRecordsFromTopKId(topk_inds_d, K, dim_amodel_offset.c, dim_amodel_offset.h * dim_amodel_offset.w, rt_out[8], amodel_offset_d, ids_out); + checkCuda( cudaMemcpy(amodel_offset, amodel_offset_d, K * dim_amodel_offset.c * sizeof(float), cudaMemcpyDeviceToHost) ); + + // ---------------------------------- post-process ----------------------------------------- + + count_det = 0; + det_res.clear(); + for(int i = 0; i(2,3); + new_det_res.x = ((float)new_det_res.ct.at(0,0) * dep[i] - calibs.at(0,3) - calibs.at(0,2) * new_det_res.z) / calibs.at(0,0); + new_det_res.y = ((float)new_det_res.ct.at(0,1) * dep[i] - calibs.at(1,3) - calibs.at(1,2) * new_det_res.z) / calibs.at(1,1) + (dim_[i] / 2); + + // alpha2rot_y + // idx = rot[:, 1] > rot[:, 5] + // alpha1 = np.arctan2(rot[:, 2], rot[:, 3]) + (-0.5 * np.pi) + // alpha2 = np.arctan2(rot[:, 6], rot[:, 7]) + ( 0.5 * np.pi) + // return alpha1 * idx + alpha2 * (1 - idx) + if(rot[1*K + i] > rot[5*K + i]) + new_det_res.alpha = std::atan2(rot[2*K + i], rot[3*K + i]) -0.5 * M_PI; + else + new_det_res.alpha = std::atan2(rot[6*K + i], rot[7*K + i]) +0.5 * M_PI; + new_det_res.rot_y = (new_det_res.alpha + std::atan2((float)new_det_res.ct.at(0,0) - calibs.at(0,2), calibs.at(0,0))); + new_det_res.ct = new_det_res.ct + new_det_res.tr; //dest + det_res.push_back(new_det_res); + + } + // track step + tracking(); +} + +cv::Mat CenternetDetection3DTrack::draw(cv::Mat &frame) { + + float sc; + int id; + std::string txt; + int baseline = 0; + float font_scale = 0.8; + int thickness = 2; + for(int i=0; i vis_thresh){// && tr_res[i].active!=0) { + if(view2d) { + + + cv::rectangle(frame, cv::Point(tr_res[i].det_res.bb0.at(0,0), tr_res[i].det_res.bb0.at(0,1)), + cv::Point(tr_res[i].det_res.bb1.at(0,0), tr_res[i].det_res.bb1.at(0,1)), tr_colors[tr_res[i].color], thickness); + cv::rectangle(frame, cv::Point(tr_res[i].det_res.bb0.at(0,0), + tr_res[i].det_res.bb0.at(0,1) - text_size.height - thickness), + cv::Point(tr_res[i].det_res.bb0.at(0,0) + text_size.width, + tr_res[i].det_res.bb0.at(0,1)), tr_colors[tr_res[i].color], -1); + + cv::putText(frame, txt, cv::Point(tr_res[i].det_res.bb0.at(0,0), + tr_res[i].det_res.bb0.at(0,1) - thickness -1), + cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1); + + cv::arrowedLine(frame, cv::Point((int)tr_res[i].det_res.ct.at(0,0), + (int)tr_res[i].det_res.ct.at(0,1)), + cv::Point((int)(tr_res[i].det_res.ct.at(0,0) + tr_res[i].det_res.tr.at(0,0)), + (int)(tr_res[i].det_res.ct.at(0,1) + tr_res[i].det_res.tr.at(0,1))), + cv::Scalar(255, 0, 255), 2); + } + //3d + if(!view2d && tr_res[i].det_res.z > 1){ + r.at(0,0) = std::cos(tr_res[i].det_res.rot_y); + r.at(0,2) = std::sin(tr_res[i].det_res.rot_y); + r.at(2,0) = -std::sin(tr_res[i].det_res.rot_y); + r.at(2,2) = std::cos(tr_res[i].det_res.rot_y); + + corners.at(0,0) = tr_res[i].det_res.dim[2]/2; + corners.at(0,1) = tr_res[i].det_res.dim[2]/2; + corners.at(0,2) = -tr_res[i].det_res.dim[2]/2; + corners.at(0,3) = -tr_res[i].det_res.dim[2]/2; + corners.at(0,4) = tr_res[i].det_res.dim[2]/2; + corners.at(0,5) = tr_res[i].det_res.dim[2]/2; + corners.at(0,6) = -tr_res[i].det_res.dim[2]/2; + corners.at(0,7) = -tr_res[i].det_res.dim[2]/2; + + corners.at(1,4) = -tr_res[i].det_res.dim[0]; + corners.at(1,5) = -tr_res[i].det_res.dim[0]; + corners.at(1,6) = -tr_res[i].det_res.dim[0]; + corners.at(1,7) = -tr_res[i].det_res.dim[0]; + + corners.at(2,0) = tr_res[i].det_res.dim[1]/2; + corners.at(2,1) = -tr_res[i].det_res.dim[1]/2; + corners.at(2,2) = -tr_res[i].det_res.dim[1]/2; + corners.at(2,3) = tr_res[i].det_res.dim[1]/2; + corners.at(2,4) = tr_res[i].det_res.dim[1]/2; + corners.at(2,5) = -tr_res[i].det_res.dim[1]/2; + corners.at(2,6) = -tr_res[i].det_res.dim[1]/2; + corners.at(2,7) = tr_res[i].det_res.dim[1]/2; + + cv::Mat aus = r * corners; + + for(int k=0; k<8; k++) { + aus.at(0,k) += tr_res[i].det_res.x; + aus.at(1,k) += tr_res[i].det_res.y; + aus.at(2,k) += tr_res[i].det_res.z; + } + + // corners.copyTo(pts3DHomo(cv::Rect(0, 0, 8, 3))); + for(int k1=0; k1<3; k1++) { + for(int k2=0; k2<8; k2++) + pts3DHomo.at(k1,k2) = aus.at(k1,k2); + } + + aus.release(); + aus = calibs * pts3DHomo; + std::vector res_corners; + for(int k=0; k<8; k++) { + res_corners.push_back(aus.at(0,k) / aus.at(2,k)); + res_corners.push_back(aus.at(1,k) / aus.at(2,k)); + } + aus.release(); + for(int ind_f = 3; ind_f>=0; ind_f--) { + for(int j=0; j<4; j++) { + cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(j) * 2), + res_corners.at(face_id.at(ind_f).at(j) * 2 + 1)), + cv::Point(res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2), + res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), + tr_colors[tr_res[i].color], 2); + if(ind_f == 0) { + cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(0) * 2), + res_corners.at(face_id.at(ind_f).at(0) * 2 + 1)), + cv::Point(res_corners.at(face_id.at(ind_f).at(2) * 2), + res_corners.at(face_id.at(ind_f).at(2) * 2 + 1)), tr_colors[tr_res[i].color], 2); + cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(1) * 2), + res_corners.at(face_id.at(ind_f).at(1) * 2 + 1)), + cv::Point(res_corners.at(face_id.at(ind_f).at(3) * 2), + res_corners.at(face_id.at(ind_f).at(3) * 2 + 1)), tr_colors[tr_res[i].color], 2); + } + } + } + float bb0=(1 << 10), bb1=0, bb2=(1 << 10), bb3=0; + for(int k=0; k<8; k++) { + if(res_corners[2*k]bb1) + bb1=res_corners[2*k]; + if(res_corners[2*k+1]bb3) + bb3=res_corners[2*k+1]; + + } + cv::rectangle(frame, cv::Point(bb0, bb2), cv::Point(bb1, bb3), + tr_colors[tr_res[i].color], thickness); + cv::rectangle(frame, cv::Point(bb0, bb2 - text_size.height - thickness), + cv::Point(bb0 + text_size.width, bb2), tr_colors[tr_res[i].color], -1); + + cv::putText(frame, txt, cv::Point(bb0, bb2 - thickness -1), cv::FONT_HERSHEY_SIMPLEX, + font_scale, cv::Scalar(255, 255, 255), 1); + + cv::arrowedLine(frame, cv::Point((int)((bb0 + bb1)/2), (int)((bb2 + bb3)/2)), + cv::Point((int)((bb0 + bb1)/2 + tr_res[i].det_res.tr.at(0,0)), + (int)((bb2 + bb3)/2 + tr_res[i].det_res.tr.at(0,1))), + cv::Scalar(255, 0, 255), 2); + } + } + + } + return frame; +} + +}} + + -- 2.52.0 From 4543df853390c5512653419905b3906733b3b378 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Mon, 7 Dec 2020 18:56:07 +0100 Subject: [PATCH 029/162] Update readme. Signed-off-by: Davide Sapienza --- README.md | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/README.md b/README.md index 234e66b..3e4ceac 100644 --- a/README.md +++ b/README.md @@ -169,6 +169,16 @@ cd pytorch-ssd conda env create -f env_mobv2ssd.yml python run_ssd_live_demo.py mb2-ssd-lite ``` +### 5)Export weights for CenterTrack +To get the weights needed to run CenterTrack tests use [this](https://github.com/sapienzadavide/CenterTrack.git) fork of the original CenterTrack. +``` +git clone https://github.com/sapienzadavide/CenterTrack.git +``` +* follow the instruction in the README.md and INSTALL.md + +``` +python demo.py tracking,ddd --load_model ../models/nuScenes_3Dtracking.pth --dataset nuscenes --pre_hm --track_thresh 0.1 --demo /path/to/image/or/folder/or/video/or/webcam --test_focal_length 633 --exp_wo --exp_wo_dim 512 --input_h 512 --input_w 512 +``` ## Darknet Parser tkDNN implement and easy parser for darknet cfg files, a network can be converted with *tk::dnn::darknetParser*: @@ -246,6 +256,15 @@ The demo3D program takes the same parameters of the demo program: ./demo ``` +#### Run the 3D OD-tracking demo + +To run the 3D object detection & tracking demo follow these steps (example with CenterTrack based on DLA34): +``` +rm dla34_cnet3d_track_fp32.rt # be sure to delete(or move) old tensorRT files +./test_dla34_cnet3d_track # run the yolo test (is slow) +./demo3D dla34_cnet3d_track_fp32.rt ../demo/yolo_test.mp4 t +``` + ### FP16 inference To run the an object detection demo with FP16 inference follow these steps (example with yolov3): -- 2.52.0 From dbc052865c0c2c74994786bd62938f44d4b5a674 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Wed, 9 Dec 2020 18:04:42 +0100 Subject: [PATCH 030/162] Fix a wrong path in CenternetDetection3DTrack.cpp Signed-off-by: Davide Sapienza --- src/CenternetDetection3DTrack.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp index 35488f7..6b34571 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenternetDetection3DTrack.cpp @@ -59,7 +59,7 @@ bool CenternetDetection3DTrack::init_preprocessing(){ bool CenternetDetection3DTrack::init_pre_inf(){ // initial steps: the first part of the network - const char *pre_img_conv1_bin = "/home/davide/Projects/repos/tkDNN/build/dla34_cnet3d_track/layers/base-pre_img_layer-0.bin"; + const char *pre_img_conv1_bin = "dla34_cnet3d_track/layers/base-pre_img_layer-0.bin"; const char *pre_hm_conv1_bin = "dla34_cnet3d_track/layers/base-pre_hm_layer-0.bin"; const char *conv1_bin = "dla34_cnet3d_track/layers/base-base_layer-0.bin"; const char *conv2_bin = "dla34_cnet3d_track/layers/base-level0-0.bin"; -- 2.52.0 From 9e1d7b3bb42f870b417a3229bf495e575377aed7 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Wed, 9 Dec 2020 18:05:54 +0100 Subject: [PATCH 031/162] Update demo3d with show flag Signed-off-by: Davide Sapienza --- demo/demo/demo3D.cpp | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp index 1a93206..609a21a 100644 --- a/demo/demo/demo3D.cpp +++ b/demo/demo/demo3D.cpp @@ -33,6 +33,12 @@ int main(int argc, char *argv[]) { int n_classes = 3; if(argc > 4) n_classes = atoi(argv[4]); + bool show = false; + if(argc > 5) + show = atoi(argv[5]); + + if(!show) + SAVE_RESULT = true; tk::dnn::CenternetDetection3D cnet; tk::dnn::CenternetDetection3DTrack ctrack; @@ -70,7 +76,8 @@ int main(int argc, char *argv[]) { cv::Mat frame; cv::Mat dnn_input; - cv::namedWindow("detection", cv::WINDOW_NORMAL); + if(show) + cv::namedWindow("detection", cv::WINDOW_NORMAL); std::vector detected_bbox; @@ -86,9 +93,11 @@ int main(int argc, char *argv[]) { //inference detNN->update(dnn_input); frame = detNN->draw(frame); - - cv::imshow("detection", frame); - cv::waitKey(1); + + if(show) { + cv::imshow("detection", frame); + cv::waitKey(1); + } if(SAVE_RESULT) resultVideo << frame; } -- 2.52.0 From 1cfa199ee6b301fafd707b7e0d600423b73357d0 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Wed, 9 Dec 2020 19:43:51 +0100 Subject: [PATCH 032/162] Add pre-processing and post-processing stats Signed-off-by: Davide Sapienza --- demo/demo/demo3D.cpp | 13 +++++++++++++ include/tkDNN/DetectionNN3D.h | 4 +++- 2 files changed, 16 insertions(+), 1 deletion(-) diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp index 609a21a..4286faa 100644 --- a/demo/demo/demo3D.cpp +++ b/demo/demo/demo3D.cpp @@ -105,11 +105,24 @@ int main(int argc, char *argv[]) { std::cout<<"detection end\n"; double mean = 0; + std::cout<pre_stats.begin(), detNN->pre_stats.end())<<" ms\n"; + std::cout<<"Max: "<<*std::max_element(detNN->pre_stats.begin(), detNN->pre_stats.end())<<" ms\n"; + for(int i=0; ipre_stats.size(); i++) mean += detNN->pre_stats[i]; mean /= detNN->pre_stats.size(); + std::cout<<"Avg: "<stats.begin(), detNN->stats.end())<<" ms\n"; std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())<<" ms\n"; for(int i=0; istats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size(); std::cout<<"Avg: "<post_stats.begin(), detNN->post_stats.end())<<" ms\n"; + std::cout<<"Max: "<<*std::max_element(detNN->post_stats.begin(), detNN->post_stats.end())<<" ms\n"; + for(int i=0; ipost_stats.size(); i++) mean += detNN->post_stats[i]; mean /= detNN->post_stats.size(); + std::cout<<"Avg: "< detected; /*bounding boxes in output*/ - std::vector stats; /*keeps track of inference times (ms)*/ + std::vector pre_stats, stats, post_stats, visual_stats; /*keeps track of inference times (ms)*/ std::vector classesNames; DetectionNN3D() {}; @@ -107,6 +107,7 @@ class DetectionNN3D { TKDNN_TSTART preprocess(frame); TKDNN_TSTOP + pre_stats.push_back(t_ns); if(save_times) *times< Date: Fri, 11 Dec 2020 23:26:45 +0900 Subject: [PATCH 033/162] Use correct BN MIN Epsilon --- include/tkDNN/Layer.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/tkDNN/Layer.h b/include/tkDNN/Layer.h index 25c4565..9330f8d 100644 --- a/include/tkDNN/Layer.h +++ b/include/tkDNN/Layer.h @@ -32,7 +32,7 @@ enum layerType_t { LAYER_YOLO }; -#define TKDNN_BN_MIN_EPSILON 1e-5 +#define TKDNN_BN_MIN_EPSILON 1e-6 /** Simple layer Father class -- 2.52.0 From 59b0f434a78244da6e6bcea6e8296c61b65abd3a Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Mon, 14 Dec 2020 15:32:35 +0100 Subject: [PATCH 034/162] Fix a bug in the 3D bounding boxes. Signed-off-by: Davide Sapienza --- src/CenternetDetection3DTrack.cpp | 33 ++++++++++++++++--------------- 1 file changed, 17 insertions(+), 16 deletions(-) diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp index 6b34571..ef3e161 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenternetDetection3DTrack.cpp @@ -267,8 +267,8 @@ bool CenternetDetection3DTrack::init_visualization(const int n_classes){ face_id.push_back({0,1,5,4}); face_id.push_back({1,2,6, 5}); - face_id.push_back({2,3,7,6}); face_id.push_back({3,0,4,7}); + face_id.push_back({2,3,7,6}); // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); } @@ -809,20 +809,20 @@ cv::Mat CenternetDetection3DTrack::draw(cv::Mat &frame) { aus.release(); for(int ind_f = 3; ind_f>=0; ind_f--) { for(int j=0; j<4; j++) { - cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(j) * 2), - res_corners.at(face_id.at(ind_f).at(j) * 2 + 1)), - cv::Point(res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2), - res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), + cv::line(frame, cv::Point((int)res_corners.at(face_id.at(ind_f).at(j) * 2), + (int)res_corners.at(face_id.at(ind_f).at(j) * 2 + 1)), + cv::Point((int)res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2), + (int)res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), tr_colors[tr_res[i].color], 2); - if(ind_f == 0) { - cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(0) * 2), - res_corners.at(face_id.at(ind_f).at(0) * 2 + 1)), - cv::Point(res_corners.at(face_id.at(ind_f).at(2) * 2), - res_corners.at(face_id.at(ind_f).at(2) * 2 + 1)), tr_colors[tr_res[i].color], 2); - cv::line(frame, cv::Point(res_corners.at(face_id.at(ind_f).at(1) * 2), - res_corners.at(face_id.at(ind_f).at(1) * 2 + 1)), - cv::Point(res_corners.at(face_id.at(ind_f).at(3) * 2), - res_corners.at(face_id.at(ind_f).at(3) * 2 + 1)), tr_colors[tr_res[i].color], 2); + if(ind_f == 0 && j==3) { + cv::line(frame, cv::Point((int)res_corners.at(face_id.at(ind_f).at(0) * 2), + (int)res_corners.at(face_id.at(ind_f).at(0) * 2 + 1)), + cv::Point((int)res_corners.at(face_id.at(ind_f).at(2) * 2), + (int)res_corners.at(face_id.at(ind_f).at(2) * 2 + 1)), tr_colors[tr_res[i].color], 2); + cv::line(frame, cv::Point((int)res_corners.at(face_id.at(ind_f).at(1) * 2), + (int)res_corners.at(face_id.at(ind_f).at(1) * 2 + 1)), + cv::Point((int)res_corners.at(face_id.at(ind_f).at(3) * 2), + (int)res_corners.at(face_id.at(ind_f).at(3) * 2 + 1)), tr_colors[tr_res[i].color], 2); } } } @@ -838,8 +838,9 @@ cv::Mat CenternetDetection3DTrack::draw(cv::Mat &frame) { bb3=res_corners[2*k+1]; } - cv::rectangle(frame, cv::Point(bb0, bb2), cv::Point(bb1, bb3), - tr_colors[tr_res[i].color], thickness); + // if(not no_bbox): + // cv::rectangle(frame, cv::Point(bb0, bb2), cv::Point(bb1, bb3), + // tr_colors[tr_res[i].color], thickness); cv::rectangle(frame, cv::Point(bb0, bb2 - text_size.height - thickness), cv::Point(bb0 + text_size.width, bb2), tr_colors[tr_res[i].color], -1); -- 2.52.0 From 56feb54377c0678e42077fabbf84ce2fc138f4c5 Mon Sep 17 00:00:00 2001 From: perseusdg Date: Thu, 21 Jan 2021 00:22:15 +0400 Subject: [PATCH 035/162] able to build kernels as shared object file(dll),and minor changes to lstm.cpp and utils.cpp to overcome minor msvc build errors --- CMakeLists.txt | 11 +++++++++-- include/tkDNN/DetectionNN.h | 5 +++++ include/tkDNN/ImuOdom.h | 6 ++++++ include/tkDNN/Int8BatchStream.h | 7 ++++++- include/tkDNN/utils.h | 8 ++++++++ src/LSTM.cpp | 13 +++++++++---- src/utils.cpp | 7 ++++++- 7 files changed, 49 insertions(+), 8 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 8c8619d..03e26c5 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -2,7 +2,13 @@ cmake_minimum_required(VERSION 3.5) project (tkDNN) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) +if(LINUX) set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable") +endif() +if(WIN32) +set(CMAKE_CXX_STANDARD 14) +set(CMAKE_CXX_FLAGS "/O2 /FS ") +endif() include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN) # project specific flags @@ -18,7 +24,7 @@ add_definitions(-DTKDNN_PATH="${CMAKE_CURRENT_SOURCE_DIR}") find_package(CUDA 9.0 REQUIRED) SET(CUDA_SEPARABLE_COMPILATION ON) #set(CUDA_NVCC_FLAGS "${CUDA_NVCC_FLAGS} -arch=sm_30 --compiler-options '-fPIC'") -set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32) +set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32 -arch=sm_61 ) find_package(CUDNN REQUIRED) include_directories(${CUDNN_INCLUDE_DIR}) @@ -28,6 +34,7 @@ include_directories(${CUDNN_INCLUDE_DIR}) file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu") cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${CUDNN_INCLUDE_DIRS}) cuda_add_library(kernels SHARED ${tkdnn_CUSRC}) +target_link_libraries(kernels ${CUDA_CUBLAS_LIBRARIES}) #------------------------------------------------------------------------------- @@ -48,7 +55,7 @@ set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV") file(GLOB tkdnn_SRC "src/*.cpp") set(tkdnn_LIBS kernels ${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDNN_LIBRARIES} ${OpenCV_LIBS} yaml-cpp) -set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11") +set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS}") include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${OPENCV_INCLUDE_DIRS} ${NVINFER_INCLUDES}) add_library(tkDNN SHARED ${tkdnn_SRC}) target_link_libraries(tkDNN ${tkdnn_LIBS}) diff --git a/include/tkDNN/DetectionNN.h b/include/tkDNN/DetectionNN.h index 0498d41..9ce33c1 100644 --- a/include/tkDNN/DetectionNN.h +++ b/include/tkDNN/DetectionNN.h @@ -4,7 +4,12 @@ #include #include #include +#ifdef __linux__ #include +#elif _WIN32 +#include +#endif + #include #include "utils.h" diff --git a/include/tkDNN/ImuOdom.h b/include/tkDNN/ImuOdom.h index 6d8d4cb..aace012 100644 --- a/include/tkDNN/ImuOdom.h +++ b/include/tkDNN/ImuOdom.h @@ -1,7 +1,13 @@ #include #include #include /* srand, rand */ + +#ifdef __linux__ #include +#elif _WIN32 +#include +#endif + #include #include #include "utils.h" diff --git a/include/tkDNN/Int8BatchStream.h b/include/tkDNN/Int8BatchStream.h index 4349c1f..7d2cef5 100644 --- a/include/tkDNN/Int8BatchStream.h +++ b/include/tkDNN/Int8BatchStream.h @@ -11,8 +11,13 @@ #include #include #include -#include +#include +#ifdef __linux__ #include +#elif _WIN32 +#include +#endif + #include #include "NvInfer.h" diff --git a/include/tkDNN/utils.h b/include/tkDNN/utils.h index 538a3f3..cb3c18f 100644 --- a/include/tkDNN/utils.h +++ b/include/tkDNN/utils.h @@ -12,7 +12,12 @@ #include #include +#ifdef __linux__ #include +#elif _WIN32 +#include +#endif + #include @@ -39,6 +44,7 @@ #define TKDNN_VERBOSE 0 // Simple Timer +#ifdef __linux__ #define TKDNN_TSTART timespec start, end; \ clock_gettime(CLOCK_MONOTONIC, &start); @@ -48,6 +54,8 @@ if(show) std::cout< 7 - checkCUDNN(cudnnSetRNNDescriptor_v6(net->cudnnHandle, + checkCUDNN(cudnnSetRNNDescriptor_v6(net->cudnnHandle,rnnDesc, stateSize, numLayers, dropoutDesc, + cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT, + //(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL), + cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL, + cudnnRNNMode_t::CUDNN_LSTM, + cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD, + net->dataType)); #else - checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle, -#endif - rnnDesc, stateSize, numLayers, dropoutDesc, + checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle,rnnDesc, stateSize, numLayers, dropoutDesc, cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT, //(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL), cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL, cudnnRNNMode_t::CUDNN_LSTM, cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD, net->dataType)); +#endif // Get temp space sizes diff --git a/src/utils.cpp b/src/utils.cpp index 65030f0..e143cce 100644 --- a/src/utils.cpp +++ b/src/utils.cpp @@ -170,6 +170,7 @@ void getMemUsage(double& vm_usage_kb, double& resident_set_kb){ using std::ios_base; using std::ifstream; using std::string; + SYSTEM_INFO sysInfo; vm_usage_kb = 0.0; resident_set_kb = 0.0; @@ -191,8 +192,12 @@ void getMemUsage(double& vm_usage_kb, double& resident_set_kb){ >> O >> itrealvalue >> starttime >> vsize >> rss; stat_stream.close(); - +#ifdef __linux__ long page_size_kb = sysconf(_SC_PAGE_SIZE) / 1024; // in case x86-64 is configured to use 2MB pages +#elif _WIN32 + long page_size_kb = sysInfo.dwPageSize/1024; +#endif + vm_usage_kb = vsize / 1024.0; resident_set_kb = rss * page_size_kb; } -- 2.52.0 From 512acd8cba99c7ec21d542b32a2881671da39cba Mon Sep 17 00:00:00 2001 From: perseusdg Date: Thu, 21 Jan 2021 09:29:27 +0400 Subject: [PATCH 036/162] minor fixes --- Issues.md | 1 + demo/demo/map.cpp | 5 +++++ include/tkDNN/utils.h | 14 +++++++++++--- src/Yolo.cpp | 9 +++++---- 4 files changed, 22 insertions(+), 7 deletions(-) create mode 100644 Issues.md diff --git a/Issues.md b/Issues.md new file mode 100644 index 0000000..4875b13 --- /dev/null +++ b/Issues.md @@ -0,0 +1 @@ +1)error C2131 @ Yolo3Detection.cpp(97) -> expression doesnt evaluate to a constant caused to read of variable outside its lifetime \ No newline at end of file diff --git a/demo/demo/map.cpp b/demo/demo/map.cpp index 356e35a..8e363ee 100644 --- a/demo/demo/map.cpp +++ b/demo/demo/map.cpp @@ -2,7 +2,12 @@ #include #include #include /* srand, rand */ +#ifdef __linux__ #include +#elif _WIN32 +#include +#endif + #include #include "utils.h" diff --git a/include/tkDNN/utils.h b/include/tkDNN/utils.h index cb3c18f..1404509 100644 --- a/include/tkDNN/utils.h +++ b/include/tkDNN/utils.h @@ -15,10 +15,12 @@ #ifdef __linux__ #include #elif _WIN32 -#include -#endif +#define NOMINMAX +#include +#endif #include +#include #define dnnType float @@ -55,7 +57,13 @@ #define TKDNN_TSTOP TKDNN_TSTOP_C(COL_CYANB, TKDNN_VERBOSE) #elif _WIN32 -#endif +#define TKDNN_TSTART auto start = std::chrono::high_resolution_clock::now(); +#define TKDNN_TSTOP auto stop = std::chrono::high_resolution_clock::now(); \ +std::chrono::duration duration = stop -start; \ +auto time_ms = std::chrono::duration_cast(duration);\ +double t_ns = time_ms.count(); +#endif + /******************************************************** * Prints the error message, and exits diff --git a/src/Yolo.cpp b/src/Yolo.cpp index 9737e74..6d0b546 100644 --- a/src/Yolo.cpp +++ b/src/Yolo.cpp @@ -9,6 +9,7 @@ #include "Layer.h" #include "kernels.h" + namespace tk { namespace dnn { Yolo::Yolo(Network *net, int classes, int num, std::string fname_weights, int n_masks, float scale_xy, double nms_thresh, nmsKind_t nsm_kind, int new_coords) : @@ -209,10 +210,10 @@ float yolo_box_iou(Yolo::box a, Yolo::box b) } void box_c(const Yolo::box a, const Yolo::box b, float& top, float& bot, float& left, float& right) { - top = std::min(a.y - a.h / 2, b.y - b.h / 2); - bot = std::max(a.y + a.h / 2, b.y + b.h / 2); - left = std::min(a.x - a.w / 2, b.x - b.w / 2); - right = std::max(a.x + a.w / 2, b.x + b.w / 2); + top = (std::min)(a.y - a.h / 2, b.y - b.h / 2); + bot = (std::max)(a.y + a.h / 2, b.y + b.h / 2); + left = (std::min)(a.x - a.w / 2, b.x - b.w / 2); + right = (std::max)(a.x + a.w / 2, b.x + b.w / 2); } // https://github.com/Zzh-tju/DIoU-darknet -- 2.52.0 From adac8576b0faf515ad3f459b1f50fd16cef6d64d Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Fri, 22 Jan 2021 17:57:42 +0100 Subject: [PATCH 037/162] Add support to Scaled-YOLO4, update Yolov4x-mish (tested) Signed-off-by: Micaela Verucchi --- README.md | 4 +- include/tkDNN/Layer.h | 7 +- include/tkDNN/NetworkRT.h | 1 + .../tkDNN/pluginsRT/ActivationLogisticRT.h | 60 + include/tkDNN/pluginsRT/YoloRT.h | 14 +- scripts/test_all_tests.sh | 5 +- src/Activation.cpp | 4 + src/DarknetParser.cpp | 1 + src/NetworkRT.cpp | 13 +- src/Yolo.cpp | 21 +- tests/darknet/cfg/yolo4-csp.cfg | 1279 +++++++++++++++++ tests/darknet/cfg/yolo4x.cfg | 21 +- tests/darknet/yolo4-csp.cpp | 36 + tests/darknet/yolo4x.cpp | 2 +- 14 files changed, 1441 insertions(+), 27 deletions(-) create mode 100644 include/tkDNN/pluginsRT/ActivationLogisticRT.h create mode 100644 tests/darknet/cfg/yolo4-csp.cfg create mode 100644 tests/darknet/yolo4-csp.cpp diff --git a/README.md b/README.md index 84e0037..d9927c0 100644 --- a/README.md +++ b/README.md @@ -353,7 +353,8 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing | yolo4 | Yolov4 8 | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) | | yolo4_berkeley | Yolov4 8 | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 540x320 | [weights](https://cloud.hipert.unimore.it/s/nkWFa5fgb4NTdnB/download) | | yolo4tiny | Yolov4 tiny 9 | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) | -| yolo4x | Yolov4x-mish 9 | [COCO 2017](http://cocodataset.org/) | 80 | 672x672 | [weights](https://cloud.hipert.unimore.it/s/BLPpiAigZJLorQD/download) | +| yolo4x | Yolov4x-mish 9 | [COCO 2017](http://cocodataset.org/) | 80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) | +| yolo4x-cps | Scaled Yolov4 10 | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) | ## References @@ -367,3 +368,4 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing 7. Wang, Chien-Yao, et al. "CSPNet: A New Backbone that can Enhance Learning Capability of CNN." arXiv preprint arXiv:1911.11929 (2019). 8. Bochkovskiy, Alexey, Chien-Yao Wang, and Hong-Yuan Mark Liao. "YOLOv4: Optimal Speed and Accuracy of Object Detection." arXiv preprint arXiv:2004.10934 (2020). 9. Bochkovskiy, Alexey, "Yolo v4, v3 and v2 for Windows and Linux" (https://github.com/AlexeyAB/darknet) +10. Wang, Chien-Yao, Alexey Bochkovskiy, and Hong-Yuan Mark Liao. "Scaled-YOLOv4: Scaling Cross Stage Partial Network." arXiv preprint arXiv:2011.08036 (2020). diff --git a/include/tkDNN/Layer.h b/include/tkDNN/Layer.h index 25c4565..e097372 100644 --- a/include/tkDNN/Layer.h +++ b/include/tkDNN/Layer.h @@ -19,6 +19,7 @@ enum layerType_t { LAYER_ACTIVATION_CRELU, LAYER_ACTIVATION_LEAKY, LAYER_ACTIVATION_MISH, + LAYER_ACTIVATION_LOGISTIC, LAYER_FLATTEN, LAYER_RESHAPE, LAYER_MULADD, @@ -68,6 +69,7 @@ public: case LAYER_ACTIVATION_CRELU: return "ActivationCReLU"; case LAYER_ACTIVATION_LEAKY: return "ActivationLeaky"; case LAYER_ACTIVATION_MISH: return "ActivationMish"; + case LAYER_ACTIVATION_LOGISTIC: return "ActivationLogistic"; case LAYER_FLATTEN: return "Flatten"; case LAYER_RESHAPE: return "Reshape"; case LAYER_MULADD: return "MulAdd"; @@ -212,7 +214,8 @@ public: typedef enum { ACTIVATION_ELU = 100, ACTIVATION_LEAKY = 101, - ACTIVATION_MISH = 102 + ACTIVATION_MISH = 102, + ACTIVATION_LOGISTIC = 103 } tkdnnActivationMode_t; /** @@ -233,6 +236,8 @@ public: return LAYER_ACTIVATION_LEAKY; else if (act_mode == ACTIVATION_MISH) return LAYER_ACTIVATION_MISH; + else if (act_mode == ACTIVATION_LOGISTIC) + return LAYER_ACTIVATION_LOGISTIC; else return LAYER_ACTIVATION; }; diff --git a/include/tkDNN/NetworkRT.h b/include/tkDNN/NetworkRT.h index 4c6c816..4fe2e0e 100644 --- a/include/tkDNN/NetworkRT.h +++ b/include/tkDNN/NetworkRT.h @@ -24,6 +24,7 @@ template T readBUF(const char*& buffer) using namespace nvinfer1; #include "pluginsRT/ActivationLeakyRT.h" +#include "pluginsRT/ActivationLogisticRT.h" #include "pluginsRT/ActivationReLUCeilingRT.h" #include "pluginsRT/ActivationMishRT.h" #include "pluginsRT/ReorgRT.h" diff --git a/include/tkDNN/pluginsRT/ActivationLogisticRT.h b/include/tkDNN/pluginsRT/ActivationLogisticRT.h new file mode 100644 index 0000000..a1ceb6b --- /dev/null +++ b/include/tkDNN/pluginsRT/ActivationLogisticRT.h @@ -0,0 +1,60 @@ +#include +#include "../kernels.h" + +class ActivationLogisticRT : public IPlugin { + +public: + ActivationLogisticRT() { + + + } + + ~ActivationLogisticRT(){ + + } + + int getNbOutputs() const override { + return 1; + } + + Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override { + return inputs[0]; + } + + void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override { + size = 1; + for(int i=0; i(inputs[0]), + reinterpret_cast(outputs[0]), batchSize*size, stream); + return 0; + } + + + virtual size_t getSerializationSize() override { + return 1*sizeof(int); + } + + virtual void serialize(void* buffer) override { + char *buf = reinterpret_cast(buffer); + tk::dnn::writeBUF(buf, size); + } + + int size; +}; diff --git a/include/tkDNN/pluginsRT/YoloRT.h b/include/tkDNN/pluginsRT/YoloRT.h index 9af8587..f2bdaa1 100644 --- a/include/tkDNN/pluginsRT/YoloRT.h +++ b/include/tkDNN/pluginsRT/YoloRT.h @@ -67,15 +67,17 @@ public: for (int b = 0; b < batchSize; ++b){ for(int n = 0; n < n_masks; ++n){ int index = entry_index(b, n*w*h, 0); - if (new_coords == 1) - activationLOGISTICForward(srcData + index, dstData + index, 4*w*h, stream); //x,y,w,h - else + if (new_coords == 1){ + if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); + } + else{ activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); //x,y - if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); + if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); - index = entry_index(b, n*w*h, 4); - activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream); + index = entry_index(b, n*w*h, 4); + activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream); + } } } diff --git a/scripts/test_all_tests.sh b/scripts/test_all_tests.sh index af04aff..6aab775 100644 --- a/scripts/test_all_tests.sh +++ b/scripts/test_all_tests.sh @@ -69,10 +69,11 @@ do echo -e "${ORANGE}Batch $TKDNN_BATCHSIZE ${NC}" test_net mnist - ./test_imuodom &>> $out_file - print_output $? imuodom + # ./test_imuodom &>> $out_file + # print_output $? imuodom test_net yolo4 + test_net yolo4-csp test_net yolo4x test_net yolo4_berkeley test_net yolo4tiny diff --git a/src/Activation.cpp b/src/Activation.cpp index 28c7624..a8642e7 100644 --- a/src/Activation.cpp +++ b/src/Activation.cpp @@ -52,6 +52,10 @@ dnnType* Activation::infer(dataDim_t &dim, dnnType* srcData) { else if(act_mode == ACTIVATION_MISH) { activationMishForward(srcData, dstData, dim.tot()); + } + else if(act_mode == ACTIVATION_LOGISTIC) { + activationLOGISTICForward(srcData, dstData, dim.tot()); + } else { dnnType alpha = dnnType(1); dnnType beta = dnnType(0); diff --git a/src/DarknetParser.cpp b/src/DarknetParser.cpp index 7b5410c..69b6b29 100644 --- a/src/DarknetParser.cpp +++ b/src/DarknetParser.cpp @@ -187,6 +187,7 @@ namespace tk { namespace dnn { if(f.activation == "relu") act = tkdnnActivationMode_t(CUDNN_ACTIVATION_RELU); else if(f.activation == "leaky") act = tk::dnn::ACTIVATION_LEAKY; else if(f.activation == "mish") act = tk::dnn::ACTIVATION_MISH; + else if(f.activation == "logistic") act = tk::dnn::ACTIVATION_LOGISTIC; else { FatalError("activation not supported: " + f.activation); } netLayers[netLayers.size()-1] = new tk::dnn::Activation(net, act); }; diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index 501ade4..c915ba5 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -226,7 +226,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Layer *l) { return convert_layer(input, (Conv2d*) l); if(type == LAYER_POOLING) return convert_layer(input, (Pooling*) l); - if(type == LAYER_ACTIVATION || type == LAYER_ACTIVATION_CRELU || type == LAYER_ACTIVATION_LEAKY || type == LAYER_ACTIVATION_MISH) + if(type == LAYER_ACTIVATION || type == LAYER_ACTIVATION_CRELU || type == LAYER_ACTIVATION_LEAKY || type == LAYER_ACTIVATION_MISH || type == LAYER_ACTIVATION_LOGISTIC) return convert_layer(input, (Activation*) l); if(type == LAYER_SOFTMAX) return convert_layer(input, (Softmax*) l); @@ -421,6 +421,12 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Activation *l) { checkNULL(lRT); return lRT; } + else if(l->act_mode == ACTIVATION_LOGISTIC) { + IPlugin *plugin = new ActivationLogisticRT(); + IPluginLayer *lRT = networkRT->addPlugin(&input, 1, *plugin); + checkNULL(lRT); + return lRT; + } else { FatalError("this Activation mode is not yet implemented"); return NULL; @@ -653,6 +659,11 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa a->size = readBUF(buf); return a; } + if(name.find("ActivationLogistic") == 0) { + ActivationLogisticRT *a = new ActivationLogisticRT(); + a->size = readBUF(buf); + return a; + } if(name.find("ActivationCReLU") == 0) { ActivationReLUCeiling *a = new ActivationReLUCeiling(readBUF(buf)); a->size = readBUF(buf); diff --git a/src/Yolo.cpp b/src/Yolo.cpp index 9737e74..61ed3cf 100644 --- a/src/Yolo.cpp +++ b/src/Yolo.cpp @@ -72,8 +72,8 @@ Yolo::box get_yolo_box(float *x, float *biases, int n, int index, int i, int j, b.h = exp(x[index + 3*stride]) * biases[2*n+1] / h; } else{ - b.x = (i + x[index + 0 * stride] * 2 - 0.5) / lw; - b.y = (j + x[index + 1 * stride] * 2 - 0.5) / lh; + b.x = (i + x[index + 0 * stride] ) / lw; + b.y = (j + x[index + 1 * stride] ) / lh; b.w = x[index + 2 * stride] * x[index + 2 * stride] * 4 * biases[2 * n] / w; b.h = x[index + 3 * stride] * x[index + 3 * stride] * 4 * biases[2 * n + 1] / h; } @@ -87,15 +87,18 @@ dnnType* Yolo::infer(dataDim_t &dim, dnnType* srcData) { for (int b = 0; b < dim.n; ++b){ for(int n = 0; n < n_masks; ++n){ int index = entry_index(b, n*dim.w*dim.h, 0, classes, input_dim, output_dim); - if (new_coords == 1) - activationLOGISTICForward(srcData + index, dstData + index, 4*dim.w*dim.h); - else + std::cout<<"new_coords"<scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); + } + else{ activationLOGISTICForward(srcData + index, dstData + index, 2*dim.w*dim.h); - if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); - - index = entry_index(b, n*dim.w*dim.h, 4, classes, input_dim, output_dim); - activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*dim.w*dim.h); + if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); + + index = entry_index(b, n*dim.w*dim.h, 4, classes, input_dim, output_dim); + activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*dim.w*dim.h); + } } } diff --git a/tests/darknet/cfg/yolo4-csp.cfg b/tests/darknet/cfg/yolo4-csp.cfg new file mode 100644 index 0000000..691ec03 --- /dev/null +++ b/tests/darknet/cfg/yolo4-csp.cfg @@ -0,0 +1,1279 @@ +[net] +# Testing +#batch=1 +#subdivisions=1 +# Training +batch=64 +subdivisions=8 +width=512 +height=512 +channels=3 +momentum=0.949 +decay=0.0005 +angle=0 +saturation = 1.5 +exposure = 1.5 +hue=.1 + +learning_rate=0.001 +burn_in=1000 +max_batches = 500500 +policy=steps +steps=400000,450000 +scales=.1,.1 + +mosaic=1 + +letter_box=1 + +ema_alpha=0.9998 + +#optimized_memory=1 + +#23:104x104 54:52x52 85:26x26 104:13x13 for 416 + + + +[convolutional] +batch_normalize=1 +filters=32 +size=3 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=2 +pad=1 +activation=mish + +#[convolutional] +#batch_normalize=1 +#filters=64 +#size=1 +#stride=1 +#pad=1 +#activation=mish + +#[route] +#layers = -2 + +#[convolutional] +#batch_normalize=1 +#filters=64 +#size=1 +#stride=1 +#pad=1 +#activation=mish + +[convolutional] +batch_normalize=1 +filters=32 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +#[convolutional] +#batch_normalize=1 +#filters=64 +#size=1 +#stride=1 +#pad=1 +#activation=mish + +#[route] +#layers = -1,-7 + +#[convolutional] +#batch_normalize=1 +#filters=64 +#size=1 +#stride=1 +#pad=1 +#activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-10 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-28 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-28 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=1024 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-16 + +[convolutional] +batch_normalize=1 +filters=1024 +size=1 +stride=1 +pad=1 +activation=mish + +########################## + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +### SPP ### +[maxpool] +stride=1 +size=5 + +[route] +layers=-2 + +[maxpool] +stride=1 +size=9 + +[route] +layers=-4 + +[maxpool] +stride=1 +size=13 + +[route] +layers=-1,-3,-5,-6 +### End SPP ### + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[route] +layers = -1, -13 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[upsample] +stride=2 + +[route] +layers = 79 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[route] +layers = -1, -6 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[upsample] +stride=2 + +[route] +layers = 48 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=128 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=128 +activation=mish + +[route] +layers = -1, -6 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +########################## + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=logistic + + +[yolo] +mask = 0,1,2 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +scale_x_y = 2.0 +objectness_smooth=0 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=4.0 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 +max_delta=5 + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=256 +activation=mish + +[route] +layers = -1, -20 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[route] +layers = -1,-6 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=logistic + + +[yolo] +mask = 3,4,5 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +scale_x_y = 2.0 +objectness_smooth=1 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=1.0 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 +max_delta=5 + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=512 +activation=mish + +[route] +layers = -1, -49 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[route] +layers = -1,-6 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=logistic + + +[yolo] +mask = 6,7,8 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +scale_x_y = 2.0 +objectness_smooth=1 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=0.4 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 +max_delta=2 diff --git a/tests/darknet/cfg/yolo4x.cfg b/tests/darknet/cfg/yolo4x.cfg index 89f2564..2ff854f 100644 --- a/tests/darknet/cfg/yolo4x.cfg +++ b/tests/darknet/cfg/yolo4x.cfg @@ -5,8 +5,8 @@ # Training batch=64 subdivisions=8 -width=672 -height=672 +width=640 +height=640 channels=3 momentum=0.949 decay=0.0005 @@ -15,7 +15,7 @@ saturation = 1.5 exposure = 1.5 hue=.1 -learning_rate=0.00261 +learning_rate=0.001 burn_in=1000 max_batches = 500500 policy=steps @@ -26,6 +26,8 @@ mosaic=1 letter_box=1 +#optimized_memory=1 + [convolutional] batch_normalize=1 filters=32 @@ -1131,6 +1133,7 @@ size=1 stride=1 pad=1 activation=mish +stopbackward=800 ########################## @@ -1147,7 +1150,7 @@ size=1 stride=1 pad=1 filters=255 -activation=linear +activation=logistic [yolo] @@ -1156,6 +1159,7 @@ anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 4 classes=80 num=9 jitter=.1 +scale_x_y = 2.0 objectness_smooth=0 ignore_thresh = .7 truth_thresh = 1 @@ -1169,6 +1173,7 @@ iou_loss=ciou nms_kind=diounms beta_nms=0.6 new_coords=1 +max_delta=5 [route] layers = -4 @@ -1275,7 +1280,7 @@ size=1 stride=1 pad=1 filters=255 -activation=linear +activation=logistic [yolo] @@ -1284,6 +1289,7 @@ anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 4 classes=80 num=9 jitter=.1 +scale_x_y = 2.0 objectness_smooth=1 ignore_thresh = .7 truth_thresh = 1 @@ -1297,6 +1303,7 @@ iou_loss=ciou nms_kind=diounms beta_nms=0.6 new_coords=1 +max_delta=5 [route] layers = -4 @@ -1403,7 +1410,7 @@ size=1 stride=1 pad=1 filters=255 -activation=linear +activation=logistic [yolo] @@ -1412,6 +1419,7 @@ anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 4 classes=80 num=9 jitter=.1 +scale_x_y = 2.0 objectness_smooth=1 ignore_thresh = .7 truth_thresh = 1 @@ -1425,3 +1433,4 @@ iou_loss=ciou nms_kind=diounms beta_nms=0.6 new_coords=1 +max_delta=2 diff --git a/tests/darknet/yolo4-csp.cpp b/tests/darknet/yolo4-csp.cpp new file mode 100644 index 0000000..af8a7fc --- /dev/null +++ b/tests/darknet/yolo4-csp.cpp @@ -0,0 +1,36 @@ +#include +#include +#include "tkdnn.h" +#include "test.h" +#include "DarknetParser.h" + +int main() { + std::string bin_path = "yolo4-csp"; + std::vector input_bins = { + bin_path + "/layers/input.bin" + }; + std::vector output_bins = { + bin_path + "/debug/layer144_out.bin", + bin_path + "/debug/layer159_out.bin", + bin_path + "/debug/layer174_out.bin" + }; + std::string wgs_path = bin_path + "/layers"; + std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4-csp.cfg"; + std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names"; + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download"); + + + + // parse darknet network + tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); + net->print(); + + //convert network to tensorRT + tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str())); + + int ret = testInference(input_bins, output_bins, net, netRT); + net->releaseLayers(); + delete net; + delete netRT; + return ret; +} diff --git a/tests/darknet/yolo4x.cpp b/tests/darknet/yolo4x.cpp index b9ad003..8df1aef 100644 --- a/tests/darknet/yolo4x.cpp +++ b/tests/darknet/yolo4x.cpp @@ -17,7 +17,7 @@ int main() { std::string wgs_path = bin_path + "/layers"; std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4x.cfg"; std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names"; - downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/BLPpiAigZJLorQD/download"); + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download"); -- 2.52.0 From fb52444cdc06b6a3868dd13cdd2cc0d0601cc6c0 Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Mon, 25 Jan 2021 04:12:05 +0530 Subject: [PATCH 038/162] Replaced dynamic arrays with std::vector ,works on linux ..needs to be tested on windows after clearing up the lnk2019 error --- CMakeLists.txt | 4 ++-- src/Yolo3Detection.cpp | 7 ++++--- src/utils.cpp | 2 +- 3 files changed, 7 insertions(+), 6 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 03e26c5..c11b56f 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -2,13 +2,13 @@ cmake_minimum_required(VERSION 3.5) project (tkDNN) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) -if(LINUX) +if(UNIX) set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable") endif() if(WIN32) set(CMAKE_CXX_STANDARD 14) set(CMAKE_CXX_FLAGS "/O2 /FS ") -endif() +endif(WIN32) include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN) # project specific flags diff --git a/src/Yolo3Detection.cpp b/src/Yolo3Detection.cpp index b94eea9..0c638e6 100644 --- a/src/Yolo3Detection.cpp +++ b/src/Yolo3Detection.cpp @@ -94,9 +94,10 @@ void Yolo3Detection::preprocess(cv::Mat &frame, const int bi){ void Yolo3Detection::postprocess(const int bi, const bool mAP){ //get yolo outputs - dnnType *rt_out[netRT->pluginFactory->n_yolos]; - for(int i=0; ipluginFactory->n_yolos; i++) - rt_out[i] = (dnnType*)netRT->buffersRT[i+1] + netRT->buffersDIM[i+1].tot()*bi; + std::vector rt_out; + //dnnType *rt_out[netRT->pluginFactory->n_yolos]; + for(int i=0; ipluginFactory->n_yolos; i++) + rt_out.push_back((dnnType*)netRT->buffersRT[i+1] + netRT->buffersDIM[i+1].tot()*bi); float x_ratio = float(originalSize[bi].width) / float(netRT->input_dim.w); float y_ratio = float(originalSize[bi].height) / float(netRT->input_dim.h); diff --git a/src/utils.cpp b/src/utils.cpp index e143cce..bbdf516 100644 --- a/src/utils.cpp +++ b/src/utils.cpp @@ -170,7 +170,6 @@ void getMemUsage(double& vm_usage_kb, double& resident_set_kb){ using std::ios_base; using std::ifstream; using std::string; - SYSTEM_INFO sysInfo; vm_usage_kb = 0.0; resident_set_kb = 0.0; @@ -195,6 +194,7 @@ void getMemUsage(double& vm_usage_kb, double& resident_set_kb){ #ifdef __linux__ long page_size_kb = sysconf(_SC_PAGE_SIZE) / 1024; // in case x86-64 is configured to use 2MB pages #elif _WIN32 +SYSTEM_INFO sysInfo; long page_size_kb = sysInfo.dwPageSize/1024; #endif -- 2.52.0 From 2d4dececb683ffa28f7f8aaf72a2e645e8c45adb Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Tue, 26 Jan 2021 20:17:43 +0400 Subject: [PATCH 039/162] Builds on windows successfully,issues with deserialization and downloading weights --- .gitignore | 3 ++- CMakeLists.txt | 9 +++++---- demo/demo/demo.cpp | 2 +- include/tkDNN/test.h | 3 ++- 4 files changed, 10 insertions(+), 7 deletions(-) diff --git a/.gitignore b/.gitignore index b56526f..5be5a73 100644 --- a/.gitignore +++ b/.gitignore @@ -12,5 +12,6 @@ build/ *.hdf5 *.pk *.table +cmake-build-release/ demo/COCO_val2017 -demo/BDD100K_val \ No newline at end of file +demo/BDD100K_val diff --git a/CMakeLists.txt b/CMakeLists.txt index c11b56f..4904e5d 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,13 +1,14 @@ cmake_minimum_required(VERSION 3.5) -project (tkDNN) +project (tkDNN CUDA) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) if(UNIX) set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable") endif() if(WIN32) set(CMAKE_CXX_STANDARD 14) -set(CMAKE_CXX_FLAGS "/O2 /FS ") +set(CMAKE_CXX_FLAGS "/O2 ") +set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON) endif(WIN32) include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN) @@ -33,7 +34,7 @@ include_directories(${CUDNN_INCLUDE_DIR}) # compile file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu") cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${CUDNN_INCLUDE_DIRS}) -cuda_add_library(kernels SHARED ${tkdnn_CUSRC}) +add_library(kernels SHARED ${tkdnn_CUSRC}) target_link_libraries(kernels ${CUDA_CUBLAS_LIBRARIES}) @@ -47,7 +48,7 @@ find_package(OpenCV REQUIRED) set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV") # gives problems in cross-compiling, probably malformed cmake config -#find_package(yaml-cpp REQUIRED) +find_package(yaml-cpp REQUIRED) #------------------------------------------------------------------------------- # Build Libraries diff --git a/demo/demo/demo.cpp b/demo/demo/demo.cpp index 9f50d0b..609affe 100644 --- a/demo/demo/demo.cpp +++ b/demo/demo/demo.cpp @@ -1,7 +1,7 @@ #include #include #include /* srand, rand */ -#include +//#include #include #include "CenternetDetection.h" diff --git a/include/tkDNN/test.h b/include/tkDNN/test.h index 13c943e..f1d44bd 100644 --- a/include/tkDNN/test.h +++ b/include/tkDNN/test.h @@ -29,7 +29,8 @@ int testInference(std::vector input_bins, std::vector readBinaryFile(input_bins[0], net->input_dim.tot(), &input_h, &data); // outputs - dnnType *cudnn_out[outputs.size()], *rt_out[outputs.size()]; + //dnnType *cudnn_out[outputs.size()], *rt_out[outputs.size()]; + std::vector cudnn_out,rt_out; tk::dnn::dataDim_t dim1 = net->input_dim; //input dim printCenteredTitle(" CUDNN inference ", '=', 30); { -- 2.52.0 From 4a9031433399b6dbf5d08a6c963f8e3f3a829b72 Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Tue, 26 Jan 2021 23:11:58 +0530 Subject: [PATCH 040/162] minor fixes in test.h --- CMakeLists.txt | 4 ++-- include/tkDNN/test.h | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 4904e5d..c173eca 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,6 +1,6 @@ cmake_minimum_required(VERSION 3.5) -project (tkDNN CUDA) +project (tkDNN) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) if(UNIX) set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable") @@ -34,7 +34,7 @@ include_directories(${CUDNN_INCLUDE_DIR}) # compile file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu") cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${CUDNN_INCLUDE_DIRS}) -add_library(kernels SHARED ${tkdnn_CUSRC}) +cuda_add_library(kernels SHARED ${tkdnn_CUSRC}) target_link_libraries(kernels ${CUDA_CUBLAS_LIBRARIES}) diff --git a/include/tkDNN/test.h b/include/tkDNN/test.h index f1d44bd..e842269 100644 --- a/include/tkDNN/test.h +++ b/include/tkDNN/test.h @@ -40,7 +40,7 @@ int testInference(std::vector input_bins, std::vector TKDNN_TSTOP dim1.print(); } - for(int i=0; idstData; + for(int i=0; idstData); if(netRT != nullptr) { tk::dnn::dataDim_t dim2 = net->input_dim; @@ -51,7 +51,7 @@ int testInference(std::vector input_bins, std::vector TKDNN_TSTOP dim2.print(); } - for(int i=0; ibuffersRT[i+1]; + for(int i=0; ibuffersRT[i+1]); } int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0; -- 2.52.0 From f055341af6cf6a3fb914a95a9aa2561fe3d81df7 Mon Sep 17 00:00:00 2001 From: Ricky Medrano Date: Tue, 9 Feb 2021 07:57:57 -0800 Subject: [PATCH 041/162] Minor Readme Changes Added Logistic as a viable activation you can use. Added conf-thresh as the 7th parameter in the ./demo call. --- README.md | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index d9927c0..5055e99 100644 --- a/README.md +++ b/README.md @@ -195,6 +195,7 @@ All models from darknet are now parsed directly from cfg, you still need to expo relu leaky mish + logistic ## Run the demo @@ -217,7 +218,7 @@ Once you have successfully created your rt file, run the demo: ``` In general the demo program takes 7 parameters: ``` -./demo +./demo ``` where * `````` is the rt file generated by a test -- 2.52.0 From 4b3731928c63f802a2804924463c57d33cf244d1 Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Thu, 25 Feb 2021 09:57:10 +0100 Subject: [PATCH 042/162] Add yolo4_320_coco2 (pedestrian and stop sign) Signed-off-by: Micaela Verucchi --- tests/darknet/cfg/yolo4_320_coco2.cfg | 1158 +++++++++++++++++++++++++ tests/darknet/names/coco2.names | 2 + tests/darknet/yolo4_320_coco2.cpp | 34 + 3 files changed, 1194 insertions(+) create mode 100644 tests/darknet/cfg/yolo4_320_coco2.cfg create mode 100644 tests/darknet/names/coco2.names create mode 100644 tests/darknet/yolo4_320_coco2.cpp diff --git a/tests/darknet/cfg/yolo4_320_coco2.cfg b/tests/darknet/cfg/yolo4_320_coco2.cfg new file mode 100644 index 0000000..9585fca --- /dev/null +++ b/tests/darknet/cfg/yolo4_320_coco2.cfg @@ -0,0 +1,1158 @@ +[net] +batch=64 +subdivisions=32 +# Training +#width=512 +#height=512 +width=320 +height=320 +channels=3 +momentum=0.949 +decay=0.0005 +angle=0 +saturation = 1.5 +exposure = 1.5 +hue=.1 + +learning_rate=0.0013 +burn_in=1000 +max_batches = 6000 +policy=steps +steps=4800,5400 +scales=.1,.1 + +#cutmix=1 +mosaic=1 + +#:104x104 54:52x52 85:26x26 104:13x13 for 416 + +[convolutional] +batch_normalize=1 +filters=32 +size=3 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=32 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-7 + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-10 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-28 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-28 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=1024 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-16 + +[convolutional] +batch_normalize=1 +filters=1024 +size=1 +stride=1 +pad=1 +activation=mish + +########################## + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +### SPP ### +[maxpool] +stride=1 +size=5 + +[route] +layers=-2 + +[maxpool] +stride=1 +size=9 + +[route] +layers=-4 + +[maxpool] +stride=1 +size=13 + +[route] +layers=-1,-3,-5,-6 +### End SPP ### + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[upsample] +stride=2 + +[route] +layers = 85 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +[upsample] +stride=2 + +[route] +layers = 54 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +########################## + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=leaky + +[convolutional] +size=1 +stride=1 +pad=1 +filters=21 +activation=linear + + +[yolo] +mask = 0,1,2 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=2 +num=9 +jitter=.3 +ignore_thresh = .7 +truth_thresh = 1 +scale_x_y = 1.2 +iou_thresh=0.213 +cls_normalizer=1.0 +iou_normalizer=0.07 +iou_loss=ciou +nms_kind=greedynms +beta_nms=0.6 +max_delta=5 + + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=256 +activation=leaky + +[route] +layers = -1, -16 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +size=1 +stride=1 +pad=1 +filters=21 +activation=linear + + +[yolo] +mask = 3,4,5 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=2 +num=9 +jitter=.3 +ignore_thresh = .7 +truth_thresh = 1 +scale_x_y = 1.1 +iou_thresh=0.213 +cls_normalizer=1.0 +iou_normalizer=0.07 +iou_loss=ciou +nms_kind=greedynms +beta_nms=0.6 +max_delta=5 + + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=512 +activation=leaky + +[route] +layers = -1, -37 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +size=1 +stride=1 +pad=1 +filters=21 +activation=linear + + +[yolo] +mask = 6,7,8 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=2 +num=9 +jitter=.3 +ignore_thresh = .7 +truth_thresh = 1 +random=1 +scale_x_y = 1.05 +iou_thresh=0.213 +cls_normalizer=1.0 +iou_normalizer=0.07 +iou_loss=ciou +nms_kind=greedynms +beta_nms=0.6 +max_delta=5 + diff --git a/tests/darknet/names/coco2.names b/tests/darknet/names/coco2.names new file mode 100644 index 0000000..e2f8903 --- /dev/null +++ b/tests/darknet/names/coco2.names @@ -0,0 +1,2 @@ +person +stop sign \ No newline at end of file diff --git a/tests/darknet/yolo4_320_coco2.cpp b/tests/darknet/yolo4_320_coco2.cpp new file mode 100644 index 0000000..877e604 --- /dev/null +++ b/tests/darknet/yolo4_320_coco2.cpp @@ -0,0 +1,34 @@ +#include +#include +#include "tkdnn.h" +#include "test.h" +#include "DarknetParser.h" + +int main() { + std::string bin_path = "yolo4_320_coco2"; + std::vector input_bins = { + bin_path + "/layers/input.bin" + }; + std::vector output_bins = { + bin_path + "/debug/layer139_out.bin", + bin_path + "/debug/layer150_out.bin", + bin_path + "/debug/layer161_out.bin" + }; + std::string wgs_path = bin_path + "/layers"; + std::string cfg_path = "../tests/darknet/cfg/yolo4_320_coco2.cfg"; + std::string name_path = "../tests/darknet/names/coco2.names"; + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/f3wk99iG5y7tEr8/download"); + + // parse darknet network + tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); + net->print(); + + //convert network to tensorRT + tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str())); + + int ret = testInference(input_bins, output_bins, net, netRT); + net->releaseLayers(); + delete net; + delete netRT; + return ret; +} -- 2.52.0 From 6aa8666be54ce7726654afd3062704555a836161 Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Wed, 10 Mar 2021 12:59:04 +0530 Subject: [PATCH 043/162] Commits for msvc 16.9 --- CMakeLists.txt | 2 +- demo/demo/demo.cpp | 2 +- src/kernels/deformable_conv.cu | 2 +- src/utils.cpp | 11 ++++++++++- 4 files changed, 13 insertions(+), 4 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index c173eca..77425d3 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -7,7 +7,7 @@ set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declara endif() if(WIN32) set(CMAKE_CXX_STANDARD 14) -set(CMAKE_CXX_FLAGS "/O2 ") +set(CMAKE_CXX_FLAGS "/O2 /FS ") set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON) endif(WIN32) include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN) diff --git a/demo/demo/demo.cpp b/demo/demo/demo.cpp index 609affe..c622f2b 100644 --- a/demo/demo/demo.cpp +++ b/demo/demo/demo.cpp @@ -22,7 +22,7 @@ int main(int argc, char *argv[]) { signal(SIGINT, sig_handler); - std::string net = "yolo3_berkeley.rt"; + std::string net = "yolo4tiny_fp32.rt"; if(argc > 1) net = argv[1]; std::string input = "../demo/yolo_test.mp4"; diff --git a/src/kernels/deformable_conv.cu b/src/kernels/deformable_conv.cu index 592c538..4dbc552 100644 --- a/src/kernels/deformable_conv.cu +++ b/src/kernels/deformable_conv.cu @@ -18,7 +18,7 @@ inline int GET_BLOCKS(const int N) } -__device__ float dmcn_im2col_bilinear(const float *bottom_data, const int data_width, +__device__ __host__ float dmcn_im2col_bilinear(const float *bottom_data, const int data_width, const int height, const int width, float h, float w) { int h_low = floor(h); int w_low = floor(w); diff --git a/src/utils.cpp b/src/utils.cpp index bbdf516..52775df 100644 --- a/src/utils.cpp +++ b/src/utils.cpp @@ -23,14 +23,23 @@ bool fileExist(const char *fname) { void downloadWeightsifDoNotExist(const std::string& input_bin, const std::string& test_folder, const std::string& weights_url){ if(!fileExist(input_bin.c_str())){ std::string mkdir_cmd = "mkdir " + test_folder; - std::string wget_cmd = "wget " + weights_url + " -O " + test_folder + "/weights.zip"; + std::string wget_cmd = "curl " + weights_url + " --output " + test_folder + "/weights.zip"; +#ifdef __linux__ std::string unzip_cmd = "unzip " + test_folder + "/weights.zip -d" + test_folder; std::string rm_cmd = "rm " + test_folder + "/weights.zip"; + +#elif _WIN32 + + std::string unzip_cmd = "7z x " + test_folder + "/weights.zip -o" + test_folder; +#endif int err = 0; err = system(mkdir_cmd.c_str()); err = system(wget_cmd.c_str()); err = system(unzip_cmd.c_str()); +#ifdef __linux__ err = system(rm_cmd.c_str()); +#endif + } } -- 2.52.0 From 304ab49897938beb3ea6619fb5720bc1aae1280d Mon Sep 17 00:00:00 2001 From: Harshvardhan Chandirasekar Date: Wed, 17 Mar 2021 22:26:26 +0530 Subject: [PATCH 044/162] shared_ptr migrations --- CMakeLists.txt | 2 +- include/tkDNN/NetworkRT.h | 7 ++++++- include/tkDNN/utils.h | 11 +++++++++++ src/NetworkRT.cpp | 11 +++++++---- 4 files changed, 25 insertions(+), 6 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 77425d3..f478cab 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -3,7 +3,7 @@ cmake_minimum_required(VERSION 3.5) project (tkDNN) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) if(UNIX) -set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable") +set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++14 -fPIC -Wno-deprecated-declarations -Wno-unused-variable") endif() if(WIN32) set(CMAKE_CXX_STANDARD 14) diff --git a/include/tkDNN/NetworkRT.h b/include/tkDNN/NetworkRT.h index 4c6c816..ca012fb 100644 --- a/include/tkDNN/NetworkRT.h +++ b/include/tkDNN/NetworkRT.h @@ -6,6 +6,7 @@ #include "Network.h" #include "Layer.h" #include "NvInfer.h" +#include namespace tk { namespace dnn { @@ -59,7 +60,8 @@ public: #if NV_TENSORRT_MAJOR >= 6 nvinfer1::IBuilderConfig *configRT; #endif - nvinfer1::ICudaEngine *engineRT; + std::shared_ptr engineRT; + //nvinfer1::ICudaEngine *engineRT; nvinfer1::IExecutionContext *contextRT; const static int MAX_BUFFERS_RT = 10; @@ -114,6 +116,9 @@ public: bool serialize(const char *filename); bool deserialize(const char *filename); + + + }; }} diff --git a/include/tkDNN/utils.h b/include/tkDNN/utils.h index 1404509..edc770e 100644 --- a/include/tkDNN/utils.h +++ b/include/tkDNN/utils.h @@ -110,6 +110,17 @@ double t_ns = time_ms.count(); FatalError(_error.str()); \ } \ } +struct InferDeleter +{ + template + void operator()(T* obj) const + { + if (obj) + { + obj->destroy(); + } + } +}; typedef enum { ERROR_CUDNN = 2, diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index 501ade4..4006caf 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -137,9 +137,11 @@ NetworkRT::NetworkRT(Network *net, const char *name) { printCudaMemUsage(); std::cout<<"Building tensorRT cuda engine...\n"; #if NV_TENSORRT_MAJOR >= 6 - engineRT = builderRT->buildEngineWithConfig(*networkRT, *configRT); + //engineRT = builderRT->buildEngineWithConfig(*networkRT, *configRT); + engineRT = std::shared_ptr(builderRT->buildEngineWithConfig(*networkRT,*configRT),InferDeleter()); #else - engineRT = builderRT->buildCudaEngine(*networkRT); + //engineRT = builderRT->buildCudaEngine(*networkRT); + engineRT = std::shared_ptr(builderRT->buildCudaEngine(*networkRT)); #endif if(engineRT == nullptr) FatalError("cloud not build cuda engine") @@ -561,7 +563,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, DeformConv2d *l) { IPluginLayer *lRT = networkRT->addPlugin(inputs, 2, *plugin); checkNULL(lRT); lRT->setName( ("Deformable" + std::to_string(l->id)).c_str() ); - delete(inputs); + delete[](inputs); // batchnorm void *bias_b, *power_b, *mean_b, *variance_b, *scales_b; if(dtRT == DataType::kHALF) { @@ -629,7 +631,8 @@ bool NetworkRT::deserialize(const char *filename) { pluginFactory = new PluginFactory(); runtimeRT = createInferRuntime(loggerRT); - engineRT = runtimeRT->deserializeCudaEngine(gieModelStream, size, (IPluginFactory *) pluginFactory); + //engineRT = runtimeRT->deserializeCudaEngine(gieModelStream, size, (IPluginFactory *) pluginFactory); + engineRT = std::shared_ptr(runtimeRT->deserializeCudaEngine(gieModelStream,size,(IPluginFactory*)pluginFactory),InferDeleter()); //if (gieModelStream) delete [] gieModelStream; return true; -- 2.52.0 From 06787a931fd8897ce8400d62c60393b8ef5fdc85 Mon Sep 17 00:00:00 2001 From: Harshvardhan Chandirasekar Date: Wed, 17 Mar 2021 23:06:50 +0530 Subject: [PATCH 045/162] minor migrations --- include/tkDNN/NetworkRT.h | 6 ++++-- src/NetworkRT.cpp | 6 ++++-- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/include/tkDNN/NetworkRT.h b/include/tkDNN/NetworkRT.h index ca012fb..a50cd8c 100644 --- a/include/tkDNN/NetworkRT.h +++ b/include/tkDNN/NetworkRT.h @@ -54,8 +54,10 @@ class NetworkRT { public: nvinfer1::DataType dtRT; - nvinfer1::IBuilder *builderRT; - nvinfer1::IRuntime *runtimeRT; + //nvinfer1::IBuilder *builderRT; + std::unique_ptr builderRT; + //nvinfer1::IRuntime *runtimeRT; + std::unique_ptr runtimeRT; nvinfer1::INetworkDefinition *networkRT; #if NV_TENSORRT_MAJOR >= 6 nvinfer1::IBuilderConfig *configRT; diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index 4006caf..e6eb778 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -33,7 +33,8 @@ NetworkRT::NetworkRT(Network *net, const char *name) { float(NV_TENSORRT_PATCH)/100; std::cout<<"New NetworkRT (TensorRT v"<(createInferBuilder(loggerRT)); + //builderRT = createInferBuilder(loggerRT); std::cout<<"Float16 support: "<platformHasFastFp16()<<"\n"; std::cout<<"Int8 support: "<platformHasFastInt8()<<"\n"; #if NV_TENSORRT_MAJOR >= 5 @@ -630,7 +631,8 @@ bool NetworkRT::deserialize(const char *filename) { } pluginFactory = new PluginFactory(); - runtimeRT = createInferRuntime(loggerRT); + //runtimeRT = createInferRuntime(loggerRT); + runtimeRT = std::unique_ptr(createInferRuntime(loggerRT)); //engineRT = runtimeRT->deserializeCudaEngine(gieModelStream, size, (IPluginFactory *) pluginFactory); engineRT = std::shared_ptr(runtimeRT->deserializeCudaEngine(gieModelStream,size,(IPluginFactory*)pluginFactory),InferDeleter()); //if (gieModelStream) delete [] gieModelStream; -- 2.52.0 From e94e1f7622d5bc4479036b39e012213acf9c1fdb Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Wed, 24 Mar 2021 22:26:46 +0530 Subject: [PATCH 046/162] tkdnn first patch for windows --- .gitignore | 1 + demo/demo/demo.cpp | 2 +- include/tkDNN/DetectionNN.h | 1 - include/tkDNN/ImuOdom.h | 2 + include/tkDNN/pluginsRT/ActivationLeakyRT.h | 3 +- include/tkDNN/pluginsRT/ActivationMishRT.h | 3 +- .../tkDNN/pluginsRT/ActivationReLUCeilingRT.h | 3 +- include/tkDNN/pluginsRT/ActivationSigmoidRT.h | 3 +- include/tkDNN/pluginsRT/DeformableConvRT.h | 3 +- include/tkDNN/pluginsRT/FlattenConcatRT.h | 3 +- .../tkDNN/pluginsRT/MaxPoolingFixedSizeRT.h | 3 +- include/tkDNN/pluginsRT/RegionRT.h | 3 +- include/tkDNN/pluginsRT/ReorgRT.h | 3 +- include/tkDNN/pluginsRT/ReshapeRT.h | 3 +- include/tkDNN/pluginsRT/ResizeLayerRT.h | 3 +- include/tkDNN/pluginsRT/RouteRT.h | 3 +- include/tkDNN/pluginsRT/ShortcutRT.h | 3 +- include/tkDNN/pluginsRT/UpsampleRT.h | 5 +- include/tkDNN/pluginsRT/YoloRT.h | 36 +++--- src/NetworkRT.cpp | 119 ++++++++++++++---- 20 files changed, 151 insertions(+), 54 deletions(-) diff --git a/.gitignore b/.gitignore index 5be5a73..7e5c4ca 100644 --- a/.gitignore +++ b/.gitignore @@ -15,3 +15,4 @@ build/ cmake-build-release/ demo/COCO_val2017 demo/BDD100K_val +/.vs diff --git a/demo/demo/demo.cpp b/demo/demo/demo.cpp index c622f2b..f97ead9 100644 --- a/demo/demo/demo.cpp +++ b/demo/demo/demo.cpp @@ -131,7 +131,7 @@ int main(int argc, char *argv[]) { double mean = 0; std::cout<stats.begin(), detNN->stats.end())/n_batch<<" ms\n"; + std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n"; std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n"; for(int i=0; istats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size(); std::cout<<"Avg: "< #elif _WIN32 +#define _USE_MATH_DEFINES +#include #include #endif diff --git a/include/tkDNN/pluginsRT/ActivationLeakyRT.h b/include/tkDNN/pluginsRT/ActivationLeakyRT.h index d3f66fb..9e26b2b 100644 --- a/include/tkDNN/pluginsRT/ActivationLeakyRT.h +++ b/include/tkDNN/pluginsRT/ActivationLeakyRT.h @@ -52,8 +52,9 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, size); + assert(buf == a + getSerializationSize()); } int size; diff --git a/include/tkDNN/pluginsRT/ActivationMishRT.h b/include/tkDNN/pluginsRT/ActivationMishRT.h index 1744ab0..5d660af 100644 --- a/include/tkDNN/pluginsRT/ActivationMishRT.h +++ b/include/tkDNN/pluginsRT/ActivationMishRT.h @@ -52,8 +52,9 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, size); + assert(buf == a + getSerializationSize()); } int size; diff --git a/include/tkDNN/pluginsRT/ActivationReLUCeilingRT.h b/include/tkDNN/pluginsRT/ActivationReLUCeilingRT.h index 286f22e..50ceb81 100644 --- a/include/tkDNN/pluginsRT/ActivationReLUCeilingRT.h +++ b/include/tkDNN/pluginsRT/ActivationReLUCeilingRT.h @@ -51,9 +51,10 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, ceiling); tk::dnn::writeBUF(buf, size); + assert(buf = a + getSerializationSize()); } diff --git a/include/tkDNN/pluginsRT/ActivationSigmoidRT.h b/include/tkDNN/pluginsRT/ActivationSigmoidRT.h index 1d47136..bcc58c7 100644 --- a/include/tkDNN/pluginsRT/ActivationSigmoidRT.h +++ b/include/tkDNN/pluginsRT/ActivationSigmoidRT.h @@ -52,8 +52,9 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, size); + assert(buf == a + getSerializationSize()); } int size; diff --git a/include/tkDNN/pluginsRT/DeformableConvRT.h b/include/tkDNN/pluginsRT/DeformableConvRT.h index 225a24e..5cb2bab 100644 --- a/include/tkDNN/pluginsRT/DeformableConvRT.h +++ b/include/tkDNN/pluginsRT/DeformableConvRT.h @@ -116,7 +116,7 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, chunk_dim); tk::dnn::writeBUF(buf, kh); tk::dnn::writeBUF(buf, kw); @@ -163,6 +163,7 @@ public: for(int i=0; i(buffer); + char *buf = reinterpret_cast(buffer),*a = buf; tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, rows); tk::dnn::writeBUF(buf, cols); + assert(buf == a + getSerializationSize()); } int c, h, w; diff --git a/include/tkDNN/pluginsRT/MaxPoolingFixedSizeRT.h b/include/tkDNN/pluginsRT/MaxPoolingFixedSizeRT.h index 911fca2..0899a34 100644 --- a/include/tkDNN/pluginsRT/MaxPoolingFixedSizeRT.h +++ b/include/tkDNN/pluginsRT/MaxPoolingFixedSizeRT.h @@ -55,7 +55,7 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, this->c); tk::dnn::writeBUF(buf, this->h); @@ -65,6 +65,7 @@ public: tk::dnn::writeBUF(buf, this->stride_W); tk::dnn::writeBUF(buf, this->winSize); tk::dnn::writeBUF(buf, this->padding); + assert(buf == a + getSerializationSize()); } int n, c, h, w; diff --git a/include/tkDNN/pluginsRT/RegionRT.h b/include/tkDNN/pluginsRT/RegionRT.h index f0d127e..8487652 100644 --- a/include/tkDNN/pluginsRT/RegionRT.h +++ b/include/tkDNN/pluginsRT/RegionRT.h @@ -73,13 +73,14 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, classes); tk::dnn::writeBUF(buf, coords); tk::dnn::writeBUF(buf, num); tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); + assert(buf == a + getSerializationSize()); } int c, h, w; diff --git a/include/tkDNN/pluginsRT/ReorgRT.h b/include/tkDNN/pluginsRT/ReorgRT.h index ee85718..c1b529a 100644 --- a/include/tkDNN/pluginsRT/ReorgRT.h +++ b/include/tkDNN/pluginsRT/ReorgRT.h @@ -52,11 +52,12 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, stride); tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); + assert(buf == a + getSerializationSize()); } int c, h, w, stride; diff --git a/include/tkDNN/pluginsRT/ReshapeRT.h b/include/tkDNN/pluginsRT/ReshapeRT.h index 97030db..37017c7 100644 --- a/include/tkDNN/pluginsRT/ReshapeRT.h +++ b/include/tkDNN/pluginsRT/ReshapeRT.h @@ -50,11 +50,12 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a = buf; tk::dnn::writeBUF(buf, n); tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); + assert(buf == a + getSerializationSize()); } int n, c, h, w; diff --git a/include/tkDNN/pluginsRT/ResizeLayerRT.h b/include/tkDNN/pluginsRT/ResizeLayerRT.h index ae87dbf..cde52bf 100644 --- a/include/tkDNN/pluginsRT/ResizeLayerRT.h +++ b/include/tkDNN/pluginsRT/ResizeLayerRT.h @@ -52,7 +52,7 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, o_c); tk::dnn::writeBUF(buf, o_h); @@ -61,6 +61,7 @@ public: tk::dnn::writeBUF(buf, i_c); tk::dnn::writeBUF(buf, i_h); tk::dnn::writeBUF(buf, i_w); + assert(buf == a + getSerializationSize()); } int i_c, i_h, i_w, o_c, o_h, o_w; diff --git a/include/tkDNN/pluginsRT/RouteRT.h b/include/tkDNN/pluginsRT/RouteRT.h index 23f30b7..5a8c170 100644 --- a/include/tkDNN/pluginsRT/RouteRT.h +++ b/include/tkDNN/pluginsRT/RouteRT.h @@ -75,7 +75,7 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, groups); tk::dnn::writeBUF(buf, group_id); tk::dnn::writeBUF(buf, in); @@ -85,6 +85,7 @@ public: tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); + assert(buf == a + getSerializationSize()); } static const int MAX_INPUTS = 4; diff --git a/include/tkDNN/pluginsRT/ShortcutRT.h b/include/tkDNN/pluginsRT/ShortcutRT.h index 3eadd3f..17f050f 100644 --- a/include/tkDNN/pluginsRT/ShortcutRT.h +++ b/include/tkDNN/pluginsRT/ShortcutRT.h @@ -59,13 +59,14 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, bc); tk::dnn::writeBUF(buf, bh); tk::dnn::writeBUF(buf, bw); tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); + assert(buf == a + getSerializationSize()); } diff --git a/include/tkDNN/pluginsRT/UpsampleRT.h b/include/tkDNN/pluginsRT/UpsampleRT.h index 7a62abc..5350b7e 100644 --- a/include/tkDNN/pluginsRT/UpsampleRT.h +++ b/include/tkDNN/pluginsRT/UpsampleRT.h @@ -54,11 +54,14 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); + char *buf = reinterpret_cast(buffer),*a=buf; tk::dnn::writeBUF(buf, stride); tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); + std::cout << "Upsample Serialization SIze" << getSerializationSize() << std::endl; + + assert(buf == a + getSerializationSize()); } int c, h, w, stride; diff --git a/include/tkDNN/pluginsRT/YoloRT.h b/include/tkDNN/pluginsRT/YoloRT.h index 9af8587..451d99f 100644 --- a/include/tkDNN/pluginsRT/YoloRT.h +++ b/include/tkDNN/pluginsRT/YoloRT.h @@ -89,21 +89,25 @@ public: } virtual void serialize(void* buffer) override { - char *buf = reinterpret_cast(buffer); - tk::dnn::writeBUF(buf, classes); - tk::dnn::writeBUF(buf, num); - tk::dnn::writeBUF(buf, n_masks); - tk::dnn::writeBUF(buf, scaleXY); - tk::dnn::writeBUF(buf, nms_thresh); - tk::dnn::writeBUF(buf, nms_kind); - tk::dnn::writeBUF(buf, new_coords); - tk::dnn::writeBUF(buf, c); - tk::dnn::writeBUF(buf, h); - tk::dnn::writeBUF(buf, w); - for(int i=0; i(buffer),*a=buf; + tk::dnn::writeBUF(buf, classes); std::cout << "Classes :" << classes << std::endl; + tk::dnn::writeBUF(buf, num); std::cout << "Num : " << num << std::endl; + tk::dnn::writeBUF(buf, n_masks); std::cout << "N_Masks" << n_masks << std::endl; + tk::dnn::writeBUF(buf, scaleXY); std::cout << "ScaleXY :" << scaleXY << std::endl; + tk::dnn::writeBUF(buf, nms_thresh); std::cout << "nms_thresh :" << nms_thresh << std::endl; + tk::dnn::writeBUF(buf, nms_kind); std::cout << "nms_kind : " << nms_kind << std::endl; + tk::dnn::writeBUF(buf, new_coords); std::cout << "new_coords : " << new_coords << std::endl; + tk::dnn::writeBUF(buf, c); std::cout << "C : " << c << std::endl; + tk::dnn::writeBUF(buf, h); std::cout << "H : " << h << std::endl; + tk::dnn::writeBUF(buf, w); std::cout << "C : " << c << std::endl; + for (int i = 0; i < n_masks; i++) + { + tk::dnn::writeBUF(buf, mask[i]); std::cout << "mask[i] : " << mask[i] << std::endl; + } + for (int i = 0; i < n_masks * 2 * num; i++) + { + tk::dnn::writeBUF(buf, bias[i]); std::cout << "bias[i] : " << bias[i] << std::endl; + } // save classes names for(int i=0; i(serialData); + const char * buf = reinterpret_cast(serialData),*bufCheck = buf; std::string name(layerName); - //std::cout<size = readBUF(buf); + assert(buf == bufCheck + serialLength); return a; } if(name.find("ActivationMish") == 0) { ActivationMishRT *a = new ActivationMishRT(); a->size = readBUF(buf); + assert(buf == bufCheck + serialLength); return a; } if(name.find("ActivationCReLU") == 0) { - ActivationReLUCeiling *a = new ActivationReLUCeiling(readBUF(buf)); + float activationReluTemp = readBUF(buf); + //ActivationReLUCeiling *a = new ActivationReLUCeiling(readBUF(buf)); + ActivationReLUCeiling* a = new ActivationReLUCeiling(activationReluTemp); a->size = readBUF(buf); + assert(buf == bufCheck + serialLength); return a; } if(name.find("Region") == 0) { - RegionRT *r = new RegionRT(readBUF(buf), //classes + int classesTemp = readBUF(buf); + int coordsTemp = readBUF(buf); + int numTemp = readBUF(buf); + /*RegionRT *r = new RegionRT(readBUF(buf), //classes readBUF(buf), //coords - readBUF(buf)); //num + readBUF(buf)); //num8*/ + RegionRT* r = new RegionRT(classesTemp, coordsTemp, numTemp); r->c = readBUF(buf); r->h = readBUF(buf); r->w = readBUF(buf); + assert(buf == bufCheck + serialLength); return r; } if(name.find("Reorg") == 0) { - ReorgRT *r = new ReorgRT(readBUF(buf)); //stride + int strideTemp = readBUF(buf); + //ReorgRT *r = new ReorgRT(readBUF(buf)); //stride + ReorgRT *r = new ReorgRT(strideTemp); r->c = readBUF(buf); r->h = readBUF(buf); r->w = readBUF(buf); + assert(buf == bufCheck + serialLength); return r; } @@ -695,27 +708,46 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa r->h = readBUF(buf); r->w = readBUF(buf); return r; + assert(buf == bufCheck + serialLength); } if(name.find("Pooling") == 0) { - MaxPoolFixedSizeRT *r = new MaxPoolFixedSizeRT( readBUF(buf), //c + /* MaxPoolFixedSizeRT *r = new MaxPoolFixedSizeRT( readBUF(buf), //c readBUF(buf), //h readBUF(buf), //w readBUF(buf), //n readBUF(buf), //strideH readBUF(buf), //strideW readBUF(buf), //winSize - readBUF(buf)); //padding + readBUF(buf)); //padding*/ + + int cTemp = readBUF(buf); + int hTemp = readBUF(buf); + int wTemp = readBUF(buf); + int nTemp = readBUF(buf); + int strideHTemp = readBUF(buf); + int strideWTemp = readBUF(buf); + int winSizeTemp = readBUF(buf); + int paddingTemp = readBUF(buf); + + MaxPoolFixedSizeRT* r = new MaxPoolFixedSizeRT(cTemp, hTemp, wTemp, nTemp, strideHTemp, strideWTemp, winSizeTemp, paddingTemp); + assert(buf == bufCheck + serialLength); return r; } if(name.find("Resize") == 0) { - ResizeLayerRT *r = new ResizeLayerRT(readBUF(buf), //o_c + /*ResizeLayerRT *r = new ResizeLayerRT(readBUF(buf), //o_c readBUF(buf), //o_h - readBUF(buf)); //o_w + readBUF(buf)); //o_w*/ + int o_cTemp = readBUF(buf); + int o_hTemp = readBUF(buf); + int o_wTemp = readBUF(buf); + ResizeLayerRT* r = new ResizeLayerRT(o_cTemp, o_hTemp, o_wTemp); + r->i_c = readBUF(buf); r->i_h = readBUF(buf); r->i_w = readBUF(buf); + assert(buf == bufCheck + serialLength); return r; } @@ -726,6 +758,7 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa r->w = readBUF(buf); r->rows = readBUF(buf); r->cols = readBUF(buf); + assert(buf == bufCheck + serialLength); return r; } @@ -737,20 +770,33 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa new_dim.h = readBUF(buf); new_dim.w = readBUF(buf); ReshapeRT *r = new ReshapeRT(new_dim); + assert(buf == bufCheck + serialLength); return r; } if(name.find("Yolo") == 0) { - YoloRT *r = new YoloRT(readBUF(buf), //classes - readBUF(buf), //num - nullptr, //yolo - readBUF(buf), //n_masks - readBUF(buf), //scale_xy - readBUF(buf), //nms_thresh - readBUF(buf), //nms_kind - readBUF(buf) //new_coords - ); + + int classes_temp = readBUF(buf); + int num_temp = readBUF(buf); + int n_masks_temp = readBUF(buf); + float scale_xy_temp = readBUF(buf); + float nms_thresh_temp = readBUF(buf); + int nms_kind_temp = readBUF(buf); + int new_coords_temp = readBUF(buf); + std::cout << classes_temp << ":" << num_temp << ":" << ":" << n_masks_temp << ":" << nms_thresh_temp << ":" << nms_kind_temp << ":" << new_coords_temp << std::endl; + + YoloRT *r = new YoloRT(classes_temp,num_temp,nullptr,n_masks_temp,scale_xy_temp,nms_thresh_temp,nms_kind_temp,new_coords_temp); + + /* std::cout << "classes : " << r->classes; + std::cout << "num : " << r->num; + std::cout << "n_masks : " << r->n_masks; + std::cout << "scalexy : " << r->scaleXY; + std::cout << "nms_thresh : " << r->nms_thresh; + std::cout << "nms_kind : " << r->nms_kind; + std::cout << "new_coords : " << r->new_coords;*/ + + r->c = readBUF(buf); r->h = readBUF(buf); r->w = readBUF(buf); @@ -767,36 +813,62 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa tmp[j] = readBUF(buf); r->classesNames[i] = std::string(tmp); } + assert(buf == bufCheck + serialLength); yolos[n_yolos++] = r; return r; } if(name.find("Upsample") == 0) { - UpsampleRT *r = new UpsampleRT(readBUF(buf)); //stride + //UpsampleRT *r = new UpsampleRT(readBUF(buf)); //stride + int strideTemp = readBUF(buf); + UpsampleRT* r = new UpsampleRT(strideTemp); r->c = readBUF(buf); r->h = readBUF(buf); r->w = readBUF(buf); + assert(buf == bufCheck + serialLength); return r; } if(name.find("Route") == 0) { - RouteRT *r = new RouteRT(readBUF(buf),readBUF(buf)); + //RouteRT *r = new RouteRT(readBUF(buf),readBUF(buf)); + int groupsTemp = readBUF(buf); + int group_idTemp = readBUF(buf); + RouteRT* r = new RouteRT(groupsTemp, group_idTemp); r->in = readBUF(buf); for(int i=0; ic_in[i] = readBUF(buf); r->c = readBUF(buf); r->h = readBUF(buf); r->w = readBUF(buf); + assert(buf == bufCheck + serialLength); return r; } if(name.find("Deformable") == 0) { - DeformableConvRT *r = new DeformableConvRT(readBUF(buf), readBUF(buf), readBUF(buf), + /*DeformableConvRT *r = new DeformableConvRT(readBUF(buf), readBUF(buf), readBUF(buf), readBUF(buf), readBUF(buf), readBUF(buf), readBUF(buf), readBUF(buf), readBUF(buf),readBUF(buf),readBUF(buf),readBUF(buf), readBUF(buf),readBUF(buf),readBUF(buf),readBUF(buf), - nullptr); + nullptr); */ + int chuck_dimTemp = readBUF(buf); + int khTemp = readBUF(buf); + int kwTemp = readBUF(buf); + int shTemp = readBUF(buf); + int swTemp = readBUF(buf); + int phTemp = readBUF(buf); + int pwTemp = readBUF(buf); + int deformableGroupTemp = readBUF(buf); + int i_nTemp = readBUF(buf); + int i_cTemp = readBUF(buf); + int i_hTemp = readBUF(buf); + int i_wTemp = readBUF(buf); + int o_nTemp = readBUF(buf); + int o_cTemp = readBUF(buf); + int o_hTemp = readBUF(buf); + int o_wTemp = readBUF(buf); + + DeformableConvRT* r = new DeformableConvRT(chuck_dimTemp, khTemp, kwTemp, shTemp, swTemp, phTemp, pwTemp, deformableGroupTemp, i_nTemp, i_cTemp, i_hTemp, i_wTemp, o_nTemp, o_cTemp, o_hTemp, o_wTemp, nullptr); dnnType *aus = new dnnType[r->chunk_dim*2]; for(int i=0; ichunk_dim*2; i++) aus[i] = readBUF(buf); @@ -827,6 +899,7 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa aus[i] = readBUF(buf); checkCuda( cudaMemcpy(r->ones_d2, aus, sizeof(dnnType)*r->dim_ones, cudaMemcpyHostToDevice) ); free(aus); + assert(buf == bufCheck + serialLength); return r; } -- 2.52.0 From 78859fe19109f265489090911ee57629bf31ff67 Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Thu, 25 Mar 2021 16:38:53 +0530 Subject: [PATCH 047/162] timer fix --- .gitignore | 4 ++++ CMakeLists.txt | 2 +- include/tkDNN/utils.h | 2 +- 3 files changed, 6 insertions(+), 2 deletions(-) diff --git a/.gitignore b/.gitignore index 7e5c4ca..c1d362c 100644 --- a/.gitignore +++ b/.gitignore @@ -16,3 +16,7 @@ cmake-build-release/ demo/COCO_val2017 demo/BDD100K_val /.vs +cmake-build-minsizerel/* +scripts/COCO_val2017/* +scripts/COCO_val2017.zip +scripts/all_labels.txt \ No newline at end of file diff --git a/CMakeLists.txt b/CMakeLists.txt index f478cab..579a8bc 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -7,7 +7,7 @@ set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++14 -fPIC -Wno-deprecated-declara endif() if(WIN32) set(CMAKE_CXX_STANDARD 14) -set(CMAKE_CXX_FLAGS "/O2 /FS ") +set(CMAKE_CXX_FLAGS "/O1 /FS") set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON) endif(WIN32) include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN) diff --git a/include/tkDNN/utils.h b/include/tkDNN/utils.h index edc770e..ce4a617 100644 --- a/include/tkDNN/utils.h +++ b/include/tkDNN/utils.h @@ -60,7 +60,7 @@ #define TKDNN_TSTART auto start = std::chrono::high_resolution_clock::now(); #define TKDNN_TSTOP auto stop = std::chrono::high_resolution_clock::now(); \ std::chrono::duration duration = stop -start; \ -auto time_ms = std::chrono::duration_cast(duration);\ +auto time_ms = std::chrono::duration_cast(duration);\ double t_ns = time_ms.count(); #endif -- 2.52.0 From 44b71ae6f33acbaf09fd8a4064d15b8395768cfb Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Thu, 25 Mar 2021 20:44:43 +0530 Subject: [PATCH 048/162] ReadMe.md windows changes --- CMakeLists.txt | 2 +- README.md | 85 ++++++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 86 insertions(+), 1 deletion(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 579a8bc..1b7ed63 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -7,7 +7,7 @@ set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++14 -fPIC -Wno-deprecated-declara endif() if(WIN32) set(CMAKE_CXX_STANDARD 14) -set(CMAKE_CXX_FLAGS "/O1 /FS") +set(CMAKE_CXX_FLAGS "/O1 /FS /EHsc") set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON) endif(WIN32) include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN) diff --git a/README.md b/README.md index 84e0037..242220a 100644 --- a/README.md +++ b/README.md @@ -80,6 +80,13 @@ Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001 - [mAP demo](#map-demo) - [Existing tests and supported networks](#existing-tests-and-supported-networks) - [References](#references) + - [tkDNN on Windows 10 (experimental)](#tkdnn-on-windows) + - [Dependencies](#dependencies) + - [Compiling tkDNN on Windows](#tkdnn-windows-compile) + - [Run the demo on Windows](#run-the-demo-on-windows) + - [FP16 interference windows](#fp16-windows) + - [INT8 interference windows](#int8-windows) + @@ -355,6 +362,84 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing | yolo4tiny | Yolov4 tiny 9 | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) | | yolo4x | Yolov4x-mish 9 | [COCO 2017](http://cocodataset.org/) | 80 | 672x672 | [weights](https://cloud.hipert.unimore.it/s/BLPpiAigZJLorQD/download) | +##tkDNN on Windows 10 (experimental) + +### Dependencies +This branch should work on every NVIDIA GPU supported in windows with the following dependencies: + +* WINDOWS 10 1803 or HIGHER +* CUDA 10.0 (Recommended CUDA 11.0 +) +* CUDNN 7.6 (Recommended CUDNN 8.0.0 +) +* TENSORRT 6.0.1 (Recommended TENSORRT 7.1 +) +* OPENCV 3.4 (Recommended OPENCV 4.2.0 +) +* MSVC 16.7 (Recommended MSVC 16.8/16.9) +* YAML-CPP 0.5.2 +* EIGEN3 +* 7ZIP (ADD TO PATH) +* NINJA 1.10 + +All the above mentioned dependencies except 7ZIP can be installed using Microsoft's [VCPKG](https://github.com/microsoft/vcpkg.git) . +After bootstrapping VCPKG the dependencies can be built and installed using the following command : + +```vcpkg.exe install opencv4[tbb,jpeg,tiff,opengl,openmp,png,ffmpeg]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build``` + +After VCPKG finishes building and installing all the packages delete C:\temp_vcpkg_build and add C:\opt\x64-windows\bin and C:\opt\x64-windows\debug\bin to path + +### Compiling tkDNN on Windows + +tkDNN is built with cmake(3.15+) on windows along with ninja.Msbuild and NMake Makefiles are drastically slower when compiling the library compared to windows +``` +git clone https://git.hipert.unimore.it/research-cv-chandirasekar/tkdnn-windows.git +cd tkdnn-windows +mkdir build +cd build +cmake -DCMAKE_BUILD_TYPE=Release -G"Ninja" .. +ninja -j4 +``` + +### Run the demo on Windows + +This example uses yolo4_tiny.\ +To run the object detection file create .rt file bu running: +``` +.\test_yolo4tiny.exe +``` + +Once the rt file has been successfully create,run the demo using the following command: +``` +.\demo.exe yolo4tiny_fp32.rt ..\demo\yolo_test.mp4 y +``` + For general info on more demo paramters,check Run the demo section on top + +### FP16 interference windows + +This is an untested feature on windows.To run the object detection demo with FP16 interference follow the below steps(example with yolo4tiny): +``` +set TKDNN_MODE=FP16 +del /f yolo4tiny_fp16.rt +.\test_yolo4tiny.exe +.\demo.exe yolo4tiny_fp16.rt ..\demo\yolo_test.mp4 +``` + +### INT8 interference windows +To run object detection demo with INT8 (example with yolo4tiny): +``` +set TKDNN_MODE=INT8 +set TKDNN_CALIB_LABEL_PATH=..\demo\COCO_val2017\all_labels.txt +set TKDNN_CALIB_IMG_PATH=..\demo\COCO_val2017\all_images.txt +del /f yolo4tiny_int8.rt # be sure to delete(or move) old tensorRT files +.\test_yolo4tiny.exe # run the yolo test (is slow) +.\demo.exe yolo4tiny_int8.rt ..\demo\yolo_test.mp4 y + +``` + + + + + + + + ## References -- 2.52.0 From f3d159143025672d5cd1bc91a89370dac1d902c2 Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Thu, 25 Mar 2021 20:46:12 +0530 Subject: [PATCH 049/162] ReadMe.md windows changes --- README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 242220a..f685b58 100644 --- a/README.md +++ b/README.md @@ -81,7 +81,7 @@ Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001 - [Existing tests and supported networks](#existing-tests-and-supported-networks) - [References](#references) - [tkDNN on Windows 10 (experimental)](#tkdnn-on-windows) - - [Dependencies](#dependencies) + - [Dependencies-Windows](#dependencies-windows) - [Compiling tkDNN on Windows](#tkdnn-windows-compile) - [Run the demo on Windows](#run-the-demo-on-windows) - [FP16 interference windows](#fp16-windows) @@ -364,7 +364,7 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing ##tkDNN on Windows 10 (experimental) -### Dependencies +### Dependencies-Windows This branch should work on every NVIDIA GPU supported in windows with the following dependencies: * WINDOWS 10 1803 or HIGHER -- 2.52.0 From f12ec3c935ce4fdcb45c966e9b4f753454d630b9 Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Fri, 26 Mar 2021 12:18:46 +0530 Subject: [PATCH 050/162] added yolo4x and yolo4-csp from the github repo and download file corrections --- include/tkDNN/Layer.h | 7 +- include/tkDNN/NetworkRT.h | 1 + .../tkDNN/pluginsRT/ActivationLogisticRT.h | 60 + include/tkDNN/pluginsRT/YoloRT.h | 29 +- src/Activation.cpp | 6 +- src/DarknetParser.cpp | 1 + src/NetworkRT.cpp | 21 +- src/Yolo.cpp | 21 +- tests/darknet/cfg/yolo4-csp.cfg | 1279 +++++++++++++++++ tests/darknet/cfg/yolo4x.cfg | 21 +- tests/darknet/yolo4-csp.cpp | 36 + tests/darknet/yolo4x.cpp | 2 +- 12 files changed, 1444 insertions(+), 40 deletions(-) create mode 100644 include/tkDNN/pluginsRT/ActivationLogisticRT.h create mode 100644 tests/darknet/cfg/yolo4-csp.cfg create mode 100644 tests/darknet/yolo4-csp.cpp diff --git a/include/tkDNN/Layer.h b/include/tkDNN/Layer.h index 25c4565..e097372 100644 --- a/include/tkDNN/Layer.h +++ b/include/tkDNN/Layer.h @@ -19,6 +19,7 @@ enum layerType_t { LAYER_ACTIVATION_CRELU, LAYER_ACTIVATION_LEAKY, LAYER_ACTIVATION_MISH, + LAYER_ACTIVATION_LOGISTIC, LAYER_FLATTEN, LAYER_RESHAPE, LAYER_MULADD, @@ -68,6 +69,7 @@ public: case LAYER_ACTIVATION_CRELU: return "ActivationCReLU"; case LAYER_ACTIVATION_LEAKY: return "ActivationLeaky"; case LAYER_ACTIVATION_MISH: return "ActivationMish"; + case LAYER_ACTIVATION_LOGISTIC: return "ActivationLogistic"; case LAYER_FLATTEN: return "Flatten"; case LAYER_RESHAPE: return "Reshape"; case LAYER_MULADD: return "MulAdd"; @@ -212,7 +214,8 @@ public: typedef enum { ACTIVATION_ELU = 100, ACTIVATION_LEAKY = 101, - ACTIVATION_MISH = 102 + ACTIVATION_MISH = 102, + ACTIVATION_LOGISTIC = 103 } tkdnnActivationMode_t; /** @@ -233,6 +236,8 @@ public: return LAYER_ACTIVATION_LEAKY; else if (act_mode == ACTIVATION_MISH) return LAYER_ACTIVATION_MISH; + else if (act_mode == ACTIVATION_LOGISTIC) + return LAYER_ACTIVATION_LOGISTIC; else return LAYER_ACTIVATION; }; diff --git a/include/tkDNN/NetworkRT.h b/include/tkDNN/NetworkRT.h index a50cd8c..b39360c 100644 --- a/include/tkDNN/NetworkRT.h +++ b/include/tkDNN/NetworkRT.h @@ -27,6 +27,7 @@ using namespace nvinfer1; #include "pluginsRT/ActivationLeakyRT.h" #include "pluginsRT/ActivationReLUCeilingRT.h" #include "pluginsRT/ActivationMishRT.h" +#include "pluginsRT/ActivationLogisticRT.h" #include "pluginsRT/ReorgRT.h" #include "pluginsRT/RegionRT.h" #include "pluginsRT/RouteRT.h" diff --git a/include/tkDNN/pluginsRT/ActivationLogisticRT.h b/include/tkDNN/pluginsRT/ActivationLogisticRT.h new file mode 100644 index 0000000..83f62ff --- /dev/null +++ b/include/tkDNN/pluginsRT/ActivationLogisticRT.h @@ -0,0 +1,60 @@ +#include +#include "../kernels.h" + +class ActivationLogisticRT : public IPlugin { + +public: + ActivationLogisticRT() { + + + } + + ~ActivationLogisticRT(){ + + } + + int getNbOutputs() const override { + return 1; + } + + Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override { + return inputs[0]; + } + + void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override { + size = 1; + for(int i=0; i(inputs[0]), + reinterpret_cast(outputs[0]), batchSize*size, stream); + return 0; + } + + + virtual size_t getSerializationSize() override { + return 1*sizeof(int); + } + + virtual void serialize(void* buffer) override { + char *buf = reinterpret_cast(buffer); + tk::dnn::writeBUF(buf, size); + } + + int size; +}; \ No newline at end of file diff --git a/include/tkDNN/pluginsRT/YoloRT.h b/include/tkDNN/pluginsRT/YoloRT.h index 451d99f..0dd26e1 100644 --- a/include/tkDNN/pluginsRT/YoloRT.h +++ b/include/tkDNN/pluginsRT/YoloRT.h @@ -64,20 +64,23 @@ public: checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream)); - for (int b = 0; b < batchSize; ++b){ - for(int n = 0; n < n_masks; ++n){ - int index = entry_index(b, n*w*h, 0); - if (new_coords == 1) - activationLOGISTICForward(srcData + index, dstData + index, 4*w*h, stream); //x,y,w,h - else - activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); //x,y - if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); - - index = entry_index(b, n*w*h, 4); - activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream); - } - } + for (int b = 0; b < batchSize; ++b){ + for(int n = 0; n < n_masks; ++n){ + int index = entry_index(b, n*w*h, 0); + if (new_coords == 1){ + if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); + } + else{ + activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); //x,y + + if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); + + index = entry_index(b, n*w*h, 4); + activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream); + } + } + } //std::cout<<"YOLO END\n"; return 0; diff --git a/src/Activation.cpp b/src/Activation.cpp index 28c7624..4219271 100644 --- a/src/Activation.cpp +++ b/src/Activation.cpp @@ -52,7 +52,11 @@ dnnType* Activation::infer(dataDim_t &dim, dnnType* srcData) { else if(act_mode == ACTIVATION_MISH) { activationMishForward(srcData, dstData, dim.tot()); - } else { + } + else if(act_mode == ACTIVATION_LOGISTIC) { + activationLOGISTICForward(srcData, dstData, dim.tot()); + + }else { dnnType alpha = dnnType(1); dnnType beta = dnnType(0); checkCUDNN( cudnnActivationForward(net->cudnnHandle, diff --git a/src/DarknetParser.cpp b/src/DarknetParser.cpp index 7b5410c..69b6b29 100644 --- a/src/DarknetParser.cpp +++ b/src/DarknetParser.cpp @@ -187,6 +187,7 @@ namespace tk { namespace dnn { if(f.activation == "relu") act = tkdnnActivationMode_t(CUDNN_ACTIVATION_RELU); else if(f.activation == "leaky") act = tk::dnn::ACTIVATION_LEAKY; else if(f.activation == "mish") act = tk::dnn::ACTIVATION_MISH; + else if(f.activation == "logistic") act = tk::dnn::ACTIVATION_LOGISTIC; else { FatalError("activation not supported: " + f.activation); } netLayers[netLayers.size()-1] = new tk::dnn::Activation(net, act); }; diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index 29df411..1e0b062 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -229,7 +229,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Layer *l) { return convert_layer(input, (Conv2d*) l); if(type == LAYER_POOLING) return convert_layer(input, (Pooling*) l); - if(type == LAYER_ACTIVATION || type == LAYER_ACTIVATION_CRELU || type == LAYER_ACTIVATION_LEAKY || type == LAYER_ACTIVATION_MISH) + if(type == LAYER_ACTIVATION || type == LAYER_ACTIVATION_CRELU || type == LAYER_ACTIVATION_LEAKY || type == LAYER_ACTIVATION_MISH || type == LAYER_ACTIVATION_LOGISTIC) return convert_layer(input, (Activation*) l); if(type == LAYER_SOFTMAX) return convert_layer(input, (Softmax*) l); @@ -424,6 +424,12 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Activation *l) { checkNULL(lRT); return lRT; } + else if(l->act_mode == ACTIVATION_LOGISTIC) { + IPlugin *plugin = new ActivationLogisticRT(); + IPluginLayer *lRT = networkRT->addPlugin(&input, 1, *plugin); + checkNULL(lRT); + return lRT; + } else { FatalError("this Activation mode is not yet implemented"); return NULL; @@ -660,6 +666,11 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa assert(buf == bufCheck + serialLength); return a; } + if(name.find("ActivationLogistic") == 0) { + ActivationLogisticRT *a = new ActivationLogisticRT(); + a->size = readBUF(buf); + return a; + } if(name.find("ActivationCReLU") == 0) { float activationReluTemp = readBUF(buf); //ActivationReLUCeiling *a = new ActivationReLUCeiling(readBUF(buf)); @@ -784,17 +795,9 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa float nms_thresh_temp = readBUF(buf); int nms_kind_temp = readBUF(buf); int new_coords_temp = readBUF(buf); - std::cout << classes_temp << ":" << num_temp << ":" << ":" << n_masks_temp << ":" << nms_thresh_temp << ":" << nms_kind_temp << ":" << new_coords_temp << std::endl; YoloRT *r = new YoloRT(classes_temp,num_temp,nullptr,n_masks_temp,scale_xy_temp,nms_thresh_temp,nms_kind_temp,new_coords_temp); - /* std::cout << "classes : " << r->classes; - std::cout << "num : " << r->num; - std::cout << "n_masks : " << r->n_masks; - std::cout << "scalexy : " << r->scaleXY; - std::cout << "nms_thresh : " << r->nms_thresh; - std::cout << "nms_kind : " << r->nms_kind; - std::cout << "new_coords : " << r->new_coords;*/ r->c = readBUF(buf); diff --git a/src/Yolo.cpp b/src/Yolo.cpp index 6d0b546..1eb7843 100644 --- a/src/Yolo.cpp +++ b/src/Yolo.cpp @@ -73,8 +73,8 @@ Yolo::box get_yolo_box(float *x, float *biases, int n, int index, int i, int j, b.h = exp(x[index + 3*stride]) * biases[2*n+1] / h; } else{ - b.x = (i + x[index + 0 * stride] * 2 - 0.5) / lw; - b.y = (j + x[index + 1 * stride] * 2 - 0.5) / lh; + b.x = (i + x[index + 0 * stride] ) / lw; + b.y = (j + x[index + 1 * stride] ) / lh; b.w = x[index + 2 * stride] * x[index + 2 * stride] * 4 * biases[2 * n] / w; b.h = x[index + 3 * stride] * x[index + 3 * stride] * 4 * biases[2 * n + 1] / h; } @@ -88,15 +88,18 @@ dnnType* Yolo::infer(dataDim_t &dim, dnnType* srcData) { for (int b = 0; b < dim.n; ++b){ for(int n = 0; n < n_masks; ++n){ int index = entry_index(b, n*dim.w*dim.h, 0, classes, input_dim, output_dim); - if (new_coords == 1) - activationLOGISTICForward(srcData + index, dstData + index, 4*dim.w*dim.h); - else + std::cout<<"new_coords"<scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); + } + else{ activationLOGISTICForward(srcData + index, dstData + index, 2*dim.w*dim.h); - if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); - - index = entry_index(b, n*dim.w*dim.h, 4, classes, input_dim, output_dim); - activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*dim.w*dim.h); + if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); + + index = entry_index(b, n*dim.w*dim.h, 4, classes, input_dim, output_dim); + activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*dim.w*dim.h); + } } } diff --git a/tests/darknet/cfg/yolo4-csp.cfg b/tests/darknet/cfg/yolo4-csp.cfg new file mode 100644 index 0000000..887898e --- /dev/null +++ b/tests/darknet/cfg/yolo4-csp.cfg @@ -0,0 +1,1279 @@ +[net] +# Testing +#batch=1 +#subdivisions=1 +# Training +batch=64 +subdivisions=8 +width=512 +height=512 +channels=3 +momentum=0.949 +decay=0.0005 +angle=0 +saturation = 1.5 +exposure = 1.5 +hue=.1 + +learning_rate=0.001 +burn_in=1000 +max_batches = 500500 +policy=steps +steps=400000,450000 +scales=.1,.1 + +mosaic=1 + +letter_box=1 + +ema_alpha=0.9998 + +#optimized_memory=1 + +#23:104x104 54:52x52 85:26x26 104:13x13 for 416 + + + +[convolutional] +batch_normalize=1 +filters=32 +size=3 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=2 +pad=1 +activation=mish + +#[convolutional] +#batch_normalize=1 +#filters=64 +#size=1 +#stride=1 +#pad=1 +#activation=mish + +#[route] +#layers = -2 + +#[convolutional] +#batch_normalize=1 +#filters=64 +#size=1 +#stride=1 +#pad=1 +#activation=mish + +[convolutional] +batch_normalize=1 +filters=32 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +#[convolutional] +#batch_normalize=1 +#filters=64 +#size=1 +#stride=1 +#pad=1 +#activation=mish + +#[route] +#layers = -1,-7 + +#[convolutional] +#batch_normalize=1 +#filters=64 +#size=1 +#stride=1 +#pad=1 +#activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-10 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-28 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-28 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=1024 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-16 + +[convolutional] +batch_normalize=1 +filters=1024 +size=1 +stride=1 +pad=1 +activation=mish + +########################## + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +### SPP ### +[maxpool] +stride=1 +size=5 + +[route] +layers=-2 + +[maxpool] +stride=1 +size=9 + +[route] +layers=-4 + +[maxpool] +stride=1 +size=13 + +[route] +layers=-1,-3,-5,-6 +### End SPP ### + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[route] +layers = -1, -13 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[upsample] +stride=2 + +[route] +layers = 79 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[route] +layers = -1, -6 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[upsample] +stride=2 + +[route] +layers = 48 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=128 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=128 +activation=mish + +[route] +layers = -1, -6 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +########################## + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=logistic + + +[yolo] +mask = 0,1,2 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +scale_x_y = 2.0 +objectness_smooth=0 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=4.0 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 +max_delta=5 + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=256 +activation=mish + +[route] +layers = -1, -20 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=mish + +[route] +layers = -1,-6 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=logistic + + +[yolo] +mask = 3,4,5 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +scale_x_y = 2.0 +objectness_smooth=1 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=1.0 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 +max_delta=5 + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=512 +activation=mish + +[route] +layers = -1, -49 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=mish + +[route] +layers = -1,-6 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=logistic + + +[yolo] +mask = 6,7,8 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +scale_x_y = 2.0 +objectness_smooth=1 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=0.4 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 +max_delta=2 \ No newline at end of file diff --git a/tests/darknet/cfg/yolo4x.cfg b/tests/darknet/cfg/yolo4x.cfg index 89f2564..f6604f6 100644 --- a/tests/darknet/cfg/yolo4x.cfg +++ b/tests/darknet/cfg/yolo4x.cfg @@ -5,8 +5,8 @@ # Training batch=64 subdivisions=8 -width=672 -height=672 +width=640 +height=640 channels=3 momentum=0.949 decay=0.0005 @@ -15,7 +15,7 @@ saturation = 1.5 exposure = 1.5 hue=.1 -learning_rate=0.00261 +learning_rate=0.001 burn_in=1000 max_batches = 500500 policy=steps @@ -26,6 +26,8 @@ mosaic=1 letter_box=1 +#optimized_memory=1 + [convolutional] batch_normalize=1 filters=32 @@ -1131,6 +1133,7 @@ size=1 stride=1 pad=1 activation=mish +stopbackward=800 ########################## @@ -1147,7 +1150,7 @@ size=1 stride=1 pad=1 filters=255 -activation=linear +activation=logistic [yolo] @@ -1156,6 +1159,7 @@ anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 4 classes=80 num=9 jitter=.1 +scale_x_y = 2.0 objectness_smooth=0 ignore_thresh = .7 truth_thresh = 1 @@ -1169,6 +1173,7 @@ iou_loss=ciou nms_kind=diounms beta_nms=0.6 new_coords=1 +max_delta=5 [route] layers = -4 @@ -1275,7 +1280,7 @@ size=1 stride=1 pad=1 filters=255 -activation=linear +activation=logistic [yolo] @@ -1284,6 +1289,7 @@ anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 4 classes=80 num=9 jitter=.1 +scale_x_y = 2.0 objectness_smooth=1 ignore_thresh = .7 truth_thresh = 1 @@ -1297,6 +1303,7 @@ iou_loss=ciou nms_kind=diounms beta_nms=0.6 new_coords=1 +max_delta=5 [route] layers = -4 @@ -1403,7 +1410,7 @@ size=1 stride=1 pad=1 filters=255 -activation=linear +activation=logistic [yolo] @@ -1412,6 +1419,7 @@ anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 4 classes=80 num=9 jitter=.1 +scale_x_y = 2.0 objectness_smooth=1 ignore_thresh = .7 truth_thresh = 1 @@ -1425,3 +1433,4 @@ iou_loss=ciou nms_kind=diounms beta_nms=0.6 new_coords=1 +max_delta=2 \ No newline at end of file diff --git a/tests/darknet/yolo4-csp.cpp b/tests/darknet/yolo4-csp.cpp new file mode 100644 index 0000000..3802a9a --- /dev/null +++ b/tests/darknet/yolo4-csp.cpp @@ -0,0 +1,36 @@ +#include +#include +#include "tkdnn.h" +#include "test.h" +#include "DarknetParser.h" + +int main() { + std::string bin_path = "yolo4-csp"; + std::vector input_bins = { + bin_path + "/layers/input.bin" + }; + std::vector output_bins = { + bin_path + "/debug/layer144_out.bin", + bin_path + "/debug/layer159_out.bin", + bin_path + "/debug/layer174_out.bin" + }; + std::string wgs_path = bin_path + "/layers"; + std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4-csp.cfg"; + std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names"; + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download"); + + + + // parse darknet network + tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); + net->print(); + + //convert network to tensorRT + tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str())); + + int ret = testInference(input_bins, output_bins, net, netRT); + net->releaseLayers(); + delete net; + delete netRT; + return ret; +} \ No newline at end of file diff --git a/tests/darknet/yolo4x.cpp b/tests/darknet/yolo4x.cpp index b9ad003..8df1aef 100644 --- a/tests/darknet/yolo4x.cpp +++ b/tests/darknet/yolo4x.cpp @@ -17,7 +17,7 @@ int main() { std::string wgs_path = bin_path + "/layers"; std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4x.cfg"; std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names"; - downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/BLPpiAigZJLorQD/download"); + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download"); -- 2.52.0 From 7018d163ed2741bc1f095f9f129368beecdee567 Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Fri, 26 Mar 2021 19:37:44 +0530 Subject: [PATCH 051/162] python download file --- scripts/download_validation.py | 40 ++++++++++++++++++++++++++++++++++ 1 file changed, 40 insertions(+) create mode 100644 scripts/download_validation.py diff --git a/scripts/download_validation.py b/scripts/download_validation.py new file mode 100644 index 0000000..e49f2ab --- /dev/null +++ b/scripts/download_validation.py @@ -0,0 +1,40 @@ +import os +from pathlib import Path +import urllib.request as dowReq +import zipfile + +val = input("Enter BDD or COCO :") +if(val == "COCO"): + url = "https://cloud.hipert.unimore.it/s/LNxBDk4wzqXPL8c/download" + lib = "..\demo\COCO_val2017" + lib_zip = "COCO_val2017.zip" +elif(val == "BDD"): + url = "https://cloud.hipert.unimore.it/s/bikqk3FzCq2tg4D/download" + lib = "..\demo\BDD100k_val" + lib_zip = "BDD100k_val.zip" + +dowReq.urlretrieve(url,lib_zip) + +with zipfile.ZipFile(lib_zip,'r') as zip_ref: + zip_ref.extractall(lib) + +labelFolder = lib + "\labels" +imageFolder = lib + "\images" + +file1 = open(".\\..\\demo\\all_labels.txt","a") +path1 = os.path.realpath(labelFolder) +for file in os.listdir(labelFolder): + valTemp = path1 + "\\" + file + valTemp = valTemp + " \n" + file1.write(valTemp) +file1.close() + +file2 = open(".\\..\\demo\\all_images.txt","a") +path2 = os.path.realpath(imageFolder) +for file in os.listdir(imageFolder): + pathtemp = path2 + "\\" + file + pathtemp = pathtemp + " \n" + file2.write(pathtemp) +file2.close() + +print("Completed") \ No newline at end of file -- 2.52.0 From 5f3ab1472c8c02be43039abc273bcd6afc9f823d Mon Sep 17 00:00:00 2001 From: Harshvardhan Chandirasekar Date: Fri, 26 Mar 2021 15:09:28 +0100 Subject: [PATCH 052/162] Update download_validation.py --- scripts/download_validation.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/scripts/download_validation.py b/scripts/download_validation.py index e49f2ab..e531b4f 100644 --- a/scripts/download_validation.py +++ b/scripts/download_validation.py @@ -1,5 +1,4 @@ import os -from pathlib import Path import urllib.request as dowReq import zipfile @@ -37,4 +36,4 @@ for file in os.listdir(imageFolder): file2.write(pathtemp) file2.close() -print("Completed") \ No newline at end of file +print("Completed") -- 2.52.0 From 2e92944f1da276ec31eeb08bd99cf7e78679b23c Mon Sep 17 00:00:00 2001 From: hchandirasekar Date: Wed, 31 Mar 2021 03:44:51 -0700 Subject: [PATCH 053/162] Opencv cuda fix --- demo/demo/demo.cpp | 2 +- include/tkDNN/DetectionNN.h | 4 +--- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/demo/demo/demo.cpp b/demo/demo/demo.cpp index f97ead9..59bb1a2 100644 --- a/demo/demo/demo.cpp +++ b/demo/demo/demo.cpp @@ -25,7 +25,7 @@ int main(int argc, char *argv[]) { std::string net = "yolo4tiny_fp32.rt"; if(argc > 1) net = argv[1]; - std::string input = "../demo/yolo_test.mp4"; + std::string input = "..\..\..\demo\yolo_test.mp4"; if(argc > 2) input = argv[2]; char ntype = 'y'; diff --git a/include/tkDNN/DetectionNN.h b/include/tkDNN/DetectionNN.h index fb29be1..b1266e0 100644 --- a/include/tkDNN/DetectionNN.h +++ b/include/tkDNN/DetectionNN.h @@ -6,8 +6,6 @@ #include #ifdef __linux__ #include -#elif _WIN32 -#include #endif #include @@ -19,7 +17,7 @@ #include "tkdnn.h" -// #define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib. +#define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib. #ifdef OPENCV_CUDACONTRIB #include -- 2.52.0 From cc594f09efdbf0e7c5a33e4e4c40ea24ac2e7048 Mon Sep 17 00:00:00 2001 From: perseusdg Date: Thu, 8 Apr 2021 00:28:00 +0530 Subject: [PATCH 054/162] merge from gitlab --- include/tkDNN/NetworkRT.h | 10 ++++------ include/tkDNN/utils.h | 14 -------------- src/NetworkRT.cpp | 37 ++++--------------------------------- src/utils.cpp | 3 +-- 4 files changed, 9 insertions(+), 55 deletions(-) diff --git a/include/tkDNN/NetworkRT.h b/include/tkDNN/NetworkRT.h index b39360c..66892f5 100644 --- a/include/tkDNN/NetworkRT.h +++ b/include/tkDNN/NetworkRT.h @@ -55,16 +55,14 @@ class NetworkRT { public: nvinfer1::DataType dtRT; - //nvinfer1::IBuilder *builderRT; - std::unique_ptr builderRT; - //nvinfer1::IRuntime *runtimeRT; - std::unique_ptr runtimeRT; + nvinfer1::IBuilder *builderRT; + nvinfer1::IRuntime *runtimeRT; nvinfer1::INetworkDefinition *networkRT; #if NV_TENSORRT_MAJOR >= 6 nvinfer1::IBuilderConfig *configRT; #endif - std::shared_ptr engineRT; - //nvinfer1::ICudaEngine *engineRT; + + nvinfer1::ICudaEngine *engineRT; nvinfer1::IExecutionContext *contextRT; const static int MAX_BUFFERS_RT = 10; diff --git a/include/tkDNN/utils.h b/include/tkDNN/utils.h index ce4a617..eeef3c2 100644 --- a/include/tkDNN/utils.h +++ b/include/tkDNN/utils.h @@ -14,9 +14,6 @@ #ifdef __linux__ #include -#elif _WIN32 -#define NOMINMAX -#include #endif #include @@ -110,17 +107,6 @@ double t_ns = time_ms.count(); FatalError(_error.str()); \ } \ } -struct InferDeleter -{ - template - void operator()(T* obj) const - { - if (obj) - { - obj->destroy(); - } - } -}; typedef enum { ERROR_CUDNN = 2, diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index 1e0b062..b5005db 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -33,8 +33,7 @@ NetworkRT::NetworkRT(Network *net, const char *name) { float(NV_TENSORRT_PATCH)/100; std::cout<<"New NetworkRT (TensorRT v"<(createInferBuilder(loggerRT)); - //builderRT = createInferBuilder(loggerRT); + builderRT = createInferBuilder(loggerRT); std::cout<<"Float16 support: "<platformHasFastFp16()<<"\n"; std::cout<<"Int8 support: "<platformHasFastInt8()<<"\n"; #if NV_TENSORRT_MAJOR >= 5 @@ -138,8 +137,7 @@ NetworkRT::NetworkRT(Network *net, const char *name) { printCudaMemUsage(); std::cout<<"Building tensorRT cuda engine...\n"; #if NV_TENSORRT_MAJOR >= 6 - //engineRT = builderRT->buildEngineWithConfig(*networkRT, *configRT); - engineRT = std::shared_ptr(builderRT->buildEngineWithConfig(*networkRT,*configRT),InferDeleter()); + engineRT = builderRT->buildEngineWithConfig(*networkRT, *configRT); #else //engineRT = builderRT->buildCudaEngine(*networkRT); engineRT = std::shared_ptr(builderRT->buildCudaEngine(*networkRT)); @@ -637,10 +635,8 @@ bool NetworkRT::deserialize(const char *filename) { } pluginFactory = new PluginFactory(); - //runtimeRT = createInferRuntime(loggerRT); - runtimeRT = std::unique_ptr(createInferRuntime(loggerRT)); - //engineRT = runtimeRT->deserializeCudaEngine(gieModelStream, size, (IPluginFactory *) pluginFactory); - engineRT = std::shared_ptr(runtimeRT->deserializeCudaEngine(gieModelStream,size,(IPluginFactory*)pluginFactory),InferDeleter()); + runtimeRT = createInferRuntime(loggerRT); + engineRT = runtimeRT->deserializeCudaEngine(gieModelStream, size, (IPluginFactory *) pluginFactory); //if (gieModelStream) delete [] gieModelStream; return true; @@ -673,7 +669,6 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa } if(name.find("ActivationCReLU") == 0) { float activationReluTemp = readBUF(buf); - //ActivationReLUCeiling *a = new ActivationReLUCeiling(readBUF(buf)); ActivationReLUCeiling* a = new ActivationReLUCeiling(activationReluTemp); a->size = readBUF(buf); assert(buf == bufCheck + serialLength); @@ -684,9 +679,6 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa int classesTemp = readBUF(buf); int coordsTemp = readBUF(buf); int numTemp = readBUF(buf); - /*RegionRT *r = new RegionRT(readBUF(buf), //classes - readBUF(buf), //coords - readBUF(buf)); //num8*/ RegionRT* r = new RegionRT(classesTemp, coordsTemp, numTemp); r->c = readBUF(buf); @@ -698,7 +690,6 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa if(name.find("Reorg") == 0) { int strideTemp = readBUF(buf); - //ReorgRT *r = new ReorgRT(readBUF(buf)); //stride ReorgRT *r = new ReorgRT(strideTemp); r->c = readBUF(buf); r->h = readBUF(buf); @@ -723,15 +714,6 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa } if(name.find("Pooling") == 0) { - /* MaxPoolFixedSizeRT *r = new MaxPoolFixedSizeRT( readBUF(buf), //c - readBUF(buf), //h - readBUF(buf), //w - readBUF(buf), //n - readBUF(buf), //strideH - readBUF(buf), //strideW - readBUF(buf), //winSize - readBUF(buf)); //padding*/ - int cTemp = readBUF(buf); int hTemp = readBUF(buf); int wTemp = readBUF(buf); @@ -747,9 +729,6 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa } if(name.find("Resize") == 0) { - /*ResizeLayerRT *r = new ResizeLayerRT(readBUF(buf), //o_c - readBUF(buf), //o_h - readBUF(buf)); //o_w*/ int o_cTemp = readBUF(buf); int o_hTemp = readBUF(buf); int o_wTemp = readBUF(buf); @@ -822,7 +801,6 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa return r; } if(name.find("Upsample") == 0) { - //UpsampleRT *r = new UpsampleRT(readBUF(buf)); //stride int strideTemp = readBUF(buf); UpsampleRT* r = new UpsampleRT(strideTemp); r->c = readBUF(buf); @@ -833,7 +811,6 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa } if(name.find("Route") == 0) { - //RouteRT *r = new RouteRT(readBUF(buf),readBUF(buf)); int groupsTemp = readBUF(buf); int group_idTemp = readBUF(buf); RouteRT* r = new RouteRT(groupsTemp, group_idTemp); @@ -848,12 +825,6 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa } if(name.find("Deformable") == 0) { - /*DeformableConvRT *r = new DeformableConvRT(readBUF(buf), readBUF(buf), readBUF(buf), - readBUF(buf), readBUF(buf), readBUF(buf), - readBUF(buf), readBUF(buf), - readBUF(buf),readBUF(buf),readBUF(buf),readBUF(buf), - readBUF(buf),readBUF(buf),readBUF(buf),readBUF(buf), - nullptr); */ int chuck_dimTemp = readBUF(buf); int khTemp = readBUF(buf); int kwTemp = readBUF(buf); diff --git a/src/utils.cpp b/src/utils.cpp index 52775df..fa6458f 100644 --- a/src/utils.cpp +++ b/src/utils.cpp @@ -203,8 +203,7 @@ void getMemUsage(double& vm_usage_kb, double& resident_set_kb){ #ifdef __linux__ long page_size_kb = sysconf(_SC_PAGE_SIZE) / 1024; // in case x86-64 is configured to use 2MB pages #elif _WIN32 -SYSTEM_INFO sysInfo; - long page_size_kb = sysInfo.dwPageSize/1024; + long page_size_kb = 4096/1024; #endif vm_usage_kb = vsize / 1024.0; -- 2.52.0 From 37b2a5bd9856b8a6becb8ac696c95f70eaef20e9 Mon Sep 17 00:00:00 2001 From: perseusdg Date: Thu, 8 Apr 2021 00:57:58 +0530 Subject: [PATCH 055/162] minor modifications --- demo/demo/map.cpp | 2 -- include/tkDNN/ImuOdom.h | 1 - include/tkDNN/Int8BatchStream.h | 2 -- 3 files changed, 5 deletions(-) diff --git a/demo/demo/map.cpp b/demo/demo/map.cpp index 8e363ee..58490e5 100644 --- a/demo/demo/map.cpp +++ b/demo/demo/map.cpp @@ -4,8 +4,6 @@ #include /* srand, rand */ #ifdef __linux__ #include -#elif _WIN32 -#include #endif #include diff --git a/include/tkDNN/ImuOdom.h b/include/tkDNN/ImuOdom.h index fa870f3..d5429a8 100644 --- a/include/tkDNN/ImuOdom.h +++ b/include/tkDNN/ImuOdom.h @@ -7,7 +7,6 @@ #elif _WIN32 #define _USE_MATH_DEFINES #include -#include #endif #include diff --git a/include/tkDNN/Int8BatchStream.h b/include/tkDNN/Int8BatchStream.h index 7d2cef5..c39a11c 100644 --- a/include/tkDNN/Int8BatchStream.h +++ b/include/tkDNN/Int8BatchStream.h @@ -14,8 +14,6 @@ #include #ifdef __linux__ #include -#elif _WIN32 -#include #endif #include -- 2.52.0 From 1de804f98dd67e66e6893e0c64749befdb9b6371 Mon Sep 17 00:00:00 2001 From: perseusdg Date: Fri, 9 Apr 2021 13:17:41 +0530 Subject: [PATCH 056/162] Code cleanup and readme fixes --- CMakeLists.txt | 6 ++--- README.md | 39 +++++++++++++++++----------- demo/demo/demo.cpp | 7 ++++- include/tkDNN/DetectionNN.h | 2 +- include/tkDNN/pluginsRT/UpsampleRT.h | 2 -- include/tkDNN/pluginsRT/YoloRT.h | 1 - scripts/download_validation.py | 4 +-- src/NetworkRT.cpp | 2 +- 8 files changed, 37 insertions(+), 26 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 1b7ed63..c27e519 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -3,11 +3,11 @@ cmake_minimum_required(VERSION 3.5) project (tkDNN) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) if(UNIX) -set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++14 -fPIC -Wno-deprecated-declarations -Wno-unused-variable") +set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable -g ") endif() if(WIN32) -set(CMAKE_CXX_STANDARD 14) -set(CMAKE_CXX_FLAGS "/O1 /FS /EHsc") +set(CMAKE_CXX_STANDARD 11) +set(CMAKE_CXX_FLAGS "/O2 /FS /EHsc") set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON) endif(WIN32) include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN) diff --git a/README.md b/README.md index f685b58..cbc65b0 100644 --- a/README.md +++ b/README.md @@ -80,12 +80,13 @@ Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001 - [mAP demo](#map-demo) - [Existing tests and supported networks](#existing-tests-and-supported-networks) - [References](#references) - - [tkDNN on Windows 10 (experimental)](#tkdnn-on-windows) + - [tkDNN on Windows 10 (experimental)](#tkdnn-on-windows-10-experimental) - [Dependencies-Windows](#dependencies-windows) - - [Compiling tkDNN on Windows](#tkdnn-windows-compile) + - [Compiling tkDNN on Windows](#compiling-tkdnn-on-windows) - [Run the demo on Windows](#run-the-demo-on-windows) - - [FP16 interference windows](#fp16-windows) - - [INT8 interference windows](#int8-windows) + - [FP16 inference windows](#fp16-inference-windows) + - [INT8 inference windows](#int8-inference-windows) + - [Known issues with tkDNN on Windows](#known-issues-with-tkdnn-on-windows) @@ -362,26 +363,31 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing | yolo4tiny | Yolov4 tiny 9 | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) | | yolo4x | Yolov4x-mish 9 | [COCO 2017](http://cocodataset.org/) | 80 | 672x672 | [weights](https://cloud.hipert.unimore.it/s/BLPpiAigZJLorQD/download) | -##tkDNN on Windows 10 (experimental) +### tkDNN on Windows 10 (experimental) ### Dependencies-Windows This branch should work on every NVIDIA GPU supported in windows with the following dependencies: * WINDOWS 10 1803 or HIGHER -* CUDA 10.0 (Recommended CUDA 11.0 +) -* CUDNN 7.6 (Recommended CUDNN 8.0.0 +) -* TENSORRT 6.0.1 (Recommended TENSORRT 7.1 +) -* OPENCV 3.4 (Recommended OPENCV 4.2.0 +) -* MSVC 16.7 (Recommended MSVC 16.8/16.9) -* YAML-CPP 0.5.2 +* CUDA 10.0 (Recommended CUDA 11.2 ) +* CUDNN 7.6 (Recommended CUDNN 8.1.1 ) +* TENSORRT 6.0.1 (Recommended TENSORRT 7.2.3.4 ) +* OPENCV 3.4 (Recommended OPENCV 4.2.0 ) +* MSVC 16.7 +* YAML-CPP * EIGEN3 * 7ZIP (ADD TO PATH) * NINJA 1.10 + All the above mentioned dependencies except 7ZIP can be installed using Microsoft's [VCPKG](https://github.com/microsoft/vcpkg.git) . After bootstrapping VCPKG the dependencies can be built and installed using the following command : -```vcpkg.exe install opencv4[tbb,jpeg,tiff,opengl,openmp,png,ffmpeg]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build``` +``` +opencv4(normal) - vcpkg.exe install opencv4[tbb,jpeg,tiff,opengl,openmp,png,ffmpeg,eigen]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build + +opencv4(cuda) - vcpkg.exe install opencv4[cuda,nonfree,contrib,eigen,tbb,jpeg,tiff,opengl,openmp,png,ffmpeg]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build +``` After VCPKG finishes building and installing all the packages delete C:\temp_vcpkg_build and add C:\opt\x64-windows\bin and C:\opt\x64-windows\debug\bin to path @@ -411,7 +417,7 @@ Once the rt file has been successfully create,run the demo using the following c ``` For general info on more demo paramters,check Run the demo section on top -### FP16 interference windows +### FP16 inference windows This is an untested feature on windows.To run the object detection demo with FP16 interference follow the below steps(example with yolo4tiny): ``` @@ -421,7 +427,7 @@ del /f yolo4tiny_fp16.rt .\demo.exe yolo4tiny_fp16.rt ..\demo\yolo_test.mp4 ``` -### INT8 interference windows +### INT8 inference windows To run object detection demo with INT8 (example with yolo4tiny): ``` set TKDNN_MODE=INT8 @@ -433,10 +439,13 @@ del /f yolo4tiny_int8.rt # be sure to delete(or move) old tensorRT files ``` +### Known issues with tkDNN on Windows +Mobilenet and Centernet demos work properly only when built with msvc 16.7 in Release Mode,when built in debug mode for the mentioned networks one might encounter opencv assert errors +All Darknet models work properly with demo using MSVC version(16.7-16.9) - +It is recommended to use Nvidia Driver(465+),Cuda unknown errors have been observed when using older drivers on pascal(SM 61) devices. diff --git a/demo/demo/demo.cpp b/demo/demo/demo.cpp index 59bb1a2..317a574 100644 --- a/demo/demo/demo.cpp +++ b/demo/demo/demo.cpp @@ -25,7 +25,12 @@ int main(int argc, char *argv[]) { std::string net = "yolo4tiny_fp32.rt"; if(argc > 1) net = argv[1]; - std::string input = "..\..\..\demo\yolo_test.mp4"; + #ifdef __linux__ + std::string input = "../demo/yolo_test.mp4"; + #elif _WIN32 + std::string input = "..\\..\\..\\demo\\yolo_test.mp4"; + #endif + if(argc > 2) input = argv[2]; char ntype = 'y'; diff --git a/include/tkDNN/DetectionNN.h b/include/tkDNN/DetectionNN.h index b1266e0..a8c81f7 100644 --- a/include/tkDNN/DetectionNN.h +++ b/include/tkDNN/DetectionNN.h @@ -17,7 +17,7 @@ #include "tkdnn.h" -#define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib. +//#define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib. #ifdef OPENCV_CUDACONTRIB #include diff --git a/include/tkDNN/pluginsRT/UpsampleRT.h b/include/tkDNN/pluginsRT/UpsampleRT.h index 5350b7e..a11d7b4 100644 --- a/include/tkDNN/pluginsRT/UpsampleRT.h +++ b/include/tkDNN/pluginsRT/UpsampleRT.h @@ -59,8 +59,6 @@ public: tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); - std::cout << "Upsample Serialization SIze" << getSerializationSize() << std::endl; - assert(buf == a + getSerializationSize()); } diff --git a/include/tkDNN/pluginsRT/YoloRT.h b/include/tkDNN/pluginsRT/YoloRT.h index 0dd26e1..2911869 100644 --- a/include/tkDNN/pluginsRT/YoloRT.h +++ b/include/tkDNN/pluginsRT/YoloRT.h @@ -120,7 +120,6 @@ public: tk::dnn::writeBUF(buf, tmp[j]); } } - std::cout << getSerializationSize() << std::endl; assert(buf == a + getSerializationSize()); } diff --git a/scripts/download_validation.py b/scripts/download_validation.py index e531b4f..0e3b1d4 100644 --- a/scripts/download_validation.py +++ b/scripts/download_validation.py @@ -24,7 +24,7 @@ file1 = open(".\\..\\demo\\all_labels.txt","a") path1 = os.path.realpath(labelFolder) for file in os.listdir(labelFolder): valTemp = path1 + "\\" + file - valTemp = valTemp + " \n" + valTemp = valTemp + '\n' file1.write(valTemp) file1.close() @@ -32,7 +32,7 @@ file2 = open(".\\..\\demo\\all_images.txt","a") path2 = os.path.realpath(imageFolder) for file in os.listdir(imageFolder): pathtemp = path2 + "\\" + file - pathtemp = pathtemp + " \n" + pathtemp = pathtemp + '\n' file2.write(pathtemp) file2.close() diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index b5005db..5dc8ee0 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -648,7 +648,7 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa const char * buf = reinterpret_cast(serialData),*bufCheck = buf; std::string name(layerName); - std::cout< Date: Fri, 9 Apr 2021 13:20:44 +0530 Subject: [PATCH 057/162] Update CMakeLists.txt --- CMakeLists.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index c27e519..d3a89f5 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -3,7 +3,7 @@ cmake_minimum_required(VERSION 3.5) project (tkDNN) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) if(UNIX) -set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable -g ") +set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable ") endif() if(WIN32) set(CMAKE_CXX_STANDARD 11) -- 2.52.0 From 39323ca8d3ac02a8a1e74ede6c851e57546f18ab Mon Sep 17 00:00:00 2001 From: Harshvardhan Chandirasekar <43143075+perseusdg@users.noreply.github.com> Date: Wed, 14 Apr 2021 17:22:26 +0530 Subject: [PATCH 058/162] Update README.md --- README.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index cbc65b0..7fb389b 100644 --- a/README.md +++ b/README.md @@ -388,6 +388,7 @@ opencv4(normal) - vcpkg.exe install opencv4[tbb,jpeg,tiff,opengl,openmp,png,ffmp opencv4(cuda) - vcpkg.exe install opencv4[cuda,nonfree,contrib,eigen,tbb,jpeg,tiff,opengl,openmp,png,ffmpeg]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build ``` +To build opencv4 with cuda and cudnn version corresponding to your cuda version,vcpkg's cudnn portfile needs to be modified by adding ```$ENV{CUDA_PATH}``` at lines 16 and 17 in the portfile.cmake After VCPKG finishes building and installing all the packages delete C:\temp_vcpkg_build and add C:\opt\x64-windows\bin and C:\opt\x64-windows\debug\bin to path @@ -395,7 +396,7 @@ After VCPKG finishes building and installing all the packages delete C:\temp_vcp tkDNN is built with cmake(3.15+) on windows along with ninja.Msbuild and NMake Makefiles are drastically slower when compiling the library compared to windows ``` -git clone https://git.hipert.unimore.it/research-cv-chandirasekar/tkdnn-windows.git +git clone https://github.com/ceccocats/tkDNN.git cd tkdnn-windows mkdir build cd build @@ -416,6 +417,7 @@ Once the rt file has been successfully create,run the demo using the following c .\demo.exe yolo4tiny_fp32.rt ..\demo\yolo_test.mp4 y ``` For general info on more demo paramters,check Run the demo section on top + To run the test_all_tests.sh on windows,use git bash or msys2 ### FP16 inference windows -- 2.52.0 From be6ad27c11f85481576037658fda7e97005340c9 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Thu, 29 Apr 2021 11:13:24 +0200 Subject: [PATCH 059/162] Batch size > 1 for the 3D demo. This commit lets to use differtent batch size for 3D CenterNet and CenterTrack. Signed-off-by: Davide Sapienza --- demo/demo/demo3D.cpp | 71 ++++-- include/tkDNN/CenternetDetection3D.h | 12 +- include/tkDNN/CenternetDetection3DTrack.h | 13 +- include/tkDNN/DetectionNN3D.h | 83 +++--- src/CenternetDetection3D.cpp | 101 ++++---- src/CenternetDetection3DTrack.cpp | 291 +++++++++++----------- 6 files changed, 309 insertions(+), 262 deletions(-) diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp index 4286faa..c35ac51 100644 --- a/demo/demo/demo3D.cpp +++ b/demo/demo/demo3D.cpp @@ -1,7 +1,7 @@ #include #include #include /* srand, rand */ -#include +//#include #include #include "CenternetDetection3D.h" @@ -24,7 +24,12 @@ int main(int argc, char *argv[]) { std::string net = "dla34_cnet3d_fp32.rt"; if(argc > 1) net = argv[1]; - std::string input = "../demo/yolo_test.mp4"; + #ifdef __linux__ + std::string input = "../demo/yolo_test.mp4"; + #elif _WIN32 + std::string input = "..\\..\\..\\demo\\yolo_test.mp4"; + #endif + if(argc > 2) input = argv[2]; char ntype = 'c'; @@ -33,9 +38,18 @@ int main(int argc, char *argv[]) { int n_classes = 3; if(argc > 4) n_classes = atoi(argv[4]); - bool show = false; + int n_batch = 1; if(argc > 5) - show = atoi(argv[5]); + n_batch = atoi(argv[5]); + bool show = true; + if(argc > 6) + show = atoi(argv[6]); + float conf_thresh=0.3; + if(argc > 7) + conf_thresh = atof(argv[7]); + + if(n_batch < 1 || n_batch > 64) + FatalError("Batch dim not supported"); if(!show) SAVE_RESULT = true; @@ -57,7 +71,7 @@ int main(int argc, char *argv[]) { FatalError("Network type not allowed (3rd parameter)\n"); } - detNN->init(net, n_classes); + detNN->init(net, n_classes, n_batch, conf_thresh); gRun = true; @@ -75,30 +89,40 @@ int main(int argc, char *argv[]) { } cv::Mat frame; - cv::Mat dnn_input; if(show) - cv::namedWindow("detection", cv::WINDOW_NORMAL); + cv::namedWindow("detection", cv::WINDOW_NORMAL); - std::vector detected_bbox; + std::vector batch_frame; + std::vector batch_dnn_input; while(gRun) { - cap >> frame; - if(!frame.data) { - break; - } - - // this will be resized to the net format - dnn_input = frame.clone(); + batch_dnn_input.clear(); + batch_frame.clear(); + for(int bi=0; bi< n_batch; ++bi){ + cap >> frame; + if(!frame.data) + break; + + batch_frame.push_back(frame); + + // this will be resized to the net format + batch_dnn_input.push_back(frame.clone()); + } + if(!frame.data) + break; + //inference - detNN->update(dnn_input); - frame = detNN->draw(frame); - - if(show) { - cv::imshow("detection", frame); - cv::waitKey(1); - } - if(SAVE_RESULT) + detNN->update(batch_dnn_input, n_batch); + detNN->draw(batch_frame); + + if(show){ + for(int bi=0; bi< n_batch; ++bi){ + cv::imshow("detection", batch_frame[bi]); + cv::waitKey(1); + } + } + if(n_batch == 1 && SAVE_RESULT) resultVideo << frame; } @@ -124,7 +148,6 @@ int main(int argc, char *argv[]) { std::cout<<"Avg: "< detected3D; - std::vectorcls3D; std::vector> face_id; public: CenternetDetection3D() {}; ~CenternetDetection3D() {}; - bool init(const std::string& tensor_path, const int n_classes=3); - void preprocess(cv::Mat &frame); - void postprocess(); - cv::Mat draw(cv::Mat &frame); + bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3); + void preprocess(cv::Mat &frame, const int bi=0); + void postprocess(const int bi=0,const bool mAP=false); + void draw(std::vector& frames); }; diff --git a/include/tkDNN/CenternetDetection3DTrack.h b/include/tkDNN/CenternetDetection3DTrack.h index 5149d70..809d0c9 100644 --- a/include/tkDNN/CenternetDetection3DTrack.h +++ b/include/tkDNN/CenternetDetection3DTrack.h @@ -145,6 +145,7 @@ private: int count_det; //tracks std::vector tr_res; + std::vector> batchTracked; int count_tr; int track_id=0; @@ -153,7 +154,7 @@ private: bool init_pre_inf(); bool init_postprocessing(); bool init_visualization(const int n_classes); - void pre_inf(); + void pre_inf(const int bi); void _get_additional_inputs(); cv::Mat transform_preds_with_trans(float x1, float x2); void tracking(); @@ -161,11 +162,11 @@ private: public: tk::dnn::Network *pre_phase_net = nullptr; CenternetDetection3DTrack() {}; - ~CenternetDetection3DTrack() {}; - bool init(const std::string& tensor_path, const int n_classes=3); - void preprocess(cv::Mat &frame); - void postprocess(); - cv::Mat draw(cv::Mat &frame); + ~CenternetDetection3DTrack() {}; + bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3); + void preprocess(cv::Mat &frame, const int bi=0); + void postprocess(const int bi=0,const bool mAP=false); + void draw(std::vector& frames); }; diff --git a/include/tkDNN/DetectionNN3D.h b/include/tkDNN/DetectionNN3D.h index 2870aa0..65cd728 100644 --- a/include/tkDNN/DetectionNN3D.h +++ b/include/tkDNN/DetectionNN3D.h @@ -3,8 +3,11 @@ #include #include -#include +#include +#ifdef __linux__ #include +#endif + #include #include "utils.h" @@ -30,10 +33,12 @@ class DetectionNN3D { tk::dnn::NetworkRT *netRT = nullptr; dnnType *input_d; - cv::Size originalSize; + std::vector originalSize; cv::Scalar colors[256]; + int nBatches = 1; + #ifdef OPENCV_CUDACONTRIB cv::cuda::GpuMat bgr[3]; cv::cuda::GpuMat imagePreproc; @@ -47,21 +52,26 @@ class DetectionNN3D { * This method preprocess the image, before feeding it to the NN. * * @param frame original frame to adapt for inference. + * @param bi batch index */ - virtual void preprocess(cv::Mat &frame) = 0; + virtual void preprocess(cv::Mat &frame, const int bi=0) = 0; /** * This method postprocess the output of the NN to obtain the correct * boundig boxes. * + * @param bi batch index + * @param mAP set to true only if all the probabilities for a bounding + * box are needed, as in some cases for the mAP calculation */ - virtual void postprocess() = 0; + virtual void postprocess(const int bi=0,const bool mAP=false) = 0; public: int classes = 0; float confThreshold = 0.3; /*threshold on the confidence of the boxes*/ - - std::vector detected; /*bounding boxes in output*/ + + std::vector detected3D; /*bounding boxes in output*/ + std::vector> batchDetected; /*bounding boxes in output*/ std::vector pre_stats, stats, post_stats, visual_stats; /*keeps track of inference times (ms)*/ std::vector classesNames; @@ -69,68 +79,79 @@ class DetectionNN3D { ~DetectionNN3D(){}; /** - * Method used to inialize the class, allocate memory and compute + * Method used to initialize the class, allocate memory and compute * needed data. * - * @param tensor_path path to the rt file og the NN. + * @param tensor_path path to the rt file of the NN. * @param n_classes number of classes for the given dataset. + * @param n_batches maximum number of batches to use in inference. * @return true if everything is correct, false otherwise. */ - virtual bool init(const std::string& tensor_path, const int n_classes=3) = 0; - - /** - * Method to draw boundixg boxes and labels on a frame. - * - * @param frame orginal frame to draw bounding box on. - * @return frame with boundig boxes. - */ - virtual cv::Mat draw(cv::Mat &frame){}; + virtual bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3) = 0; /** * This method performs the whole detection of the NN. * - * @param frame frame to run detection on. + * @param frames frames to run detection on. + * @param cur_batches number of batches to use in inference. * @param save_times if set to true, preprocess, inference and postprocess times * are saved on a csv file, otherwise not. - * @param times pointer to the output stream where to write times + * @param times pointer to the output stream where to write times. + * @param mAP set to true only if all the probabilities for a bounding + * box are needed, as in some cases for the mAP calculation. */ - void update(cv::Mat &frame, bool save_times=false, std::ofstream *times=nullptr){ - if(!frame.data) - FatalError("No image data feed to detection"); - + void update(std::vector& frames, const int cur_batches=1, bool save_times=false, std::ofstream *times=nullptr, const bool mAP=false){ if(save_times && times==nullptr) FatalError("save_times set to true, but no valid ofstream given"); + if(cur_batches > nBatches) + FatalError("A batch size greater than nBatches cannot be used"); - originalSize = frame.size(); - printCenteredTitle(" TENSORRT detection ", '=', 30); + originalSize.clear(); + if(TKDNN_VERBOSE) printCenteredTitle(" TENSORRT detection ", '=', 30); { TKDNN_TSTART - preprocess(frame); + for(int bi=0; biinput_dim; + dim.n = cur_batches; { - dim.print(); + if(TKDNN_VERBOSE) dim.print(); TKDNN_TSTART netRT->infer(dim, input_d); TKDNN_TSTOP - dim.print(); + if(TKDNN_VERBOSE) dim.print(); stats.push_back(t_ns); if(save_times) *times<& frames){}; + }; }} diff --git a/src/CenternetDetection3D.cpp b/src/CenternetDetection3D.cpp index 1803381..53b3cf7 100644 --- a/src/CenternetDetection3D.cpp +++ b/src/CenternetDetection3D.cpp @@ -3,10 +3,12 @@ namespace tk { namespace dnn { -bool CenternetDetection3D::init(const std::string& tensor_path, const int n_classes){ +bool CenternetDetection3D::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh) { std::cout<<(tensor_path).c_str()<<"\n"; netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); classes = n_classes; + nBatches = n_batches; + confThreshold = conf_thresh; dim = netRT->input_dim; @@ -28,7 +30,7 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas trans = cv::Mat(cv::Size(3,2), CV_32F); trans2 = cv::Mat(cv::Size(3,2), CV_32F); - checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot())); + checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches)); dim_hm = tk::dnn::dataDim_t(1, 3, 128, 128, 1); dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1); @@ -91,7 +93,7 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas checkCuda(cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice)); checkCuda(cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice)); #else - checkCuda(cudaMallocHost(&input, sizeof(dnnType)*netRT->input_dim.tot())); + checkCuda(cudaMallocHost(&input, sizeof(dnnType)*netRT->input_dim.tot() * nBatches)); mean << 0.485, 0.456, 0.406; stddev << 0.229, 0.224, 0.225; #endif @@ -154,13 +156,13 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); } -void CenternetDetection3D::preprocess(cv::Mat &frame){ +void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi){ // -----------------------------------pre-process ------------------------------------------ // auto start_t = std::chrono::steady_clock::now(); // auto step_t = std::chrono::steady_clock::now(); // auto end_t = std::chrono::steady_clock::now(); - cv::Size sz = originalSize; + cv::Size sz = originalSize[bi]; // std::cout<<"image: "<(end_t - step_t).count() << " us" << std::endl; // step_t = end_t; - checkCuda(cudaMemcpy(input_d, d_ptrs, dim2.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice)); + checkCuda(cudaMemcpy(input_d+ netRT->input_dim.tot()*bi, d_ptrs, dim2.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice)); // end_t = std::chrono::steady_clock::now(); // std::cout << " TIME Memcpy to input_d: " << std::chrono::duration_cast(end_t - step_t).count() << " us" << std::endl; @@ -280,21 +282,21 @@ void CenternetDetection3D::preprocess(cv::Mat &frame){ int idx = i*imageF.rows*imageF.cols; int ch = dim2.c-3 +i; // std::cout<<"i: "<input_dim.tot()*bi], (void*)bgr[ch].data, imageF.rows*imageF.cols*sizeof(dnnType)); } - checkCuda(cudaMemcpyAsync(input_d, input, dim2.tot()*sizeof(dnnType), cudaMemcpyHostToDevice)); + checkCuda(cudaMemcpyAsync(input_d+ netRT->input_dim.tot()*bi, input+ netRT->input_dim.tot()*bi, dim2.tot()*sizeof(dnnType), cudaMemcpyHostToDevice)); #endif } -void CenternetDetection3D::postprocess(){ +void CenternetDetection3D::postprocess(const int bi, const bool mAP) { dnnType *rt_out[7]; - rt_out[0] = (dnnType *)netRT->buffersRT[1]; - rt_out[1] = (dnnType *)netRT->buffersRT[2]; - rt_out[2] = (dnnType *)netRT->buffersRT[3]; - rt_out[3] = (dnnType *)netRT->buffersRT[4]; - rt_out[4] = (dnnType *)netRT->buffersRT[5]; - rt_out[5] = (dnnType *)netRT->buffersRT[6]; - rt_out[6] = (dnnType *)netRT->buffersRT[7]; + rt_out[0] = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi; + rt_out[1] = (dnnType *)netRT->buffersRT[2]+ netRT->buffersDIM[2].tot()*bi; + rt_out[2] = (dnnType *)netRT->buffersRT[3]+ netRT->buffersDIM[3].tot()*bi; + rt_out[3] = (dnnType *)netRT->buffersRT[4]+ netRT->buffersDIM[4].tot()*bi; + rt_out[4] = (dnnType *)netRT->buffersRT[5]+ netRT->buffersDIM[5].tot()*bi; + rt_out[5] = (dnnType *)netRT->buffersRT[6]+ netRT->buffersDIM[6].tot()*bi; + rt_out[6] = (dnnType *)netRT->buffersRT[7]+ netRT->buffersDIM[7].tot()*bi; // ------------------------------------ process -------------------------------------------- activationSIGMOIDForward(rt_out[0], rt_out[0], dim_hm.tot()); @@ -404,8 +406,7 @@ void CenternetDetection3D::postprocess(){ if(rot_y peakThreshold) { - if(scores[j] > centerThreshold) { + if(scores[j] > confThreshold) { if(z>0) { // compute_box_3d r.at(0,0) = std::cos(rot_y); @@ -457,16 +458,17 @@ void CenternetDetection3D::postprocess(){ } res.cl = i; res.prob = scores[j]; - res.print(); + //res.print(); detected3D.push_back(res); } } } } } + batchDetected.push_back(detected3D); } -cv::Mat CenternetDetection3D::draw(cv::Mat &frame) { +void CenternetDetection3D::draw(std::vector& frames) { tk::dnn::box3D b; int x0, w, x1, y0, h, y1; int objClass; @@ -476,40 +478,41 @@ cv::Mat CenternetDetection3D::draw(cv::Mat &frame) { float font_scale = 0.5; int thickness = 2; - // draw dets - for(int i=0; i=0; ind_f--) { - for(int j=0; j<4; j++) { - cv::line(frame, cv::Point(b.corners.at(face_id.at(ind_f).at(j) * 2), - b.corners.at(face_id.at(ind_f).at(j) * 2 + 1)), - cv::Point(b.corners.at(face_id.at(ind_f).at((j+1)%4) * 2), - b.corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), - colors[b.cl], 2); - if(ind_f == 0) { - cv::line(frame, cv::Point(b.corners.at(face_id.at(ind_f).at(0) * 2), - b.corners.at(face_id.at(ind_f).at(0) * 2 + 1)), - cv::Point(b.corners.at(face_id.at(ind_f).at(2) * 2), - b.corners.at(face_id.at(ind_f).at(2) * 2 + 1)), colors[b.cl], 2); - cv::line(frame, cv::Point(b.corners.at(face_id.at(ind_f).at(1) * 2), - b.corners.at(face_id.at(ind_f).at(1) * 2 + 1)), - cv::Point(b.corners.at(face_id.at(ind_f).at(3) * 2), - b.corners.at(face_id.at(ind_f).at(3) * 2 + 1)), colors[b.cl], 2); + for(int ind_f = 3; ind_f>=0; ind_f--) { + for(int j=0; j<4; j++) { + cv::line(frames[bi], cv::Point(b.corners.at(face_id.at(ind_f).at(j) * 2), + b.corners.at(face_id.at(ind_f).at(j) * 2 + 1)), + cv::Point(b.corners.at(face_id.at(ind_f).at((j+1)%4) * 2), + b.corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), + colors[b.cl], 2); + if(ind_f == 0) { + cv::line(frames[bi], cv::Point(b.corners.at(face_id.at(ind_f).at(0) * 2), + b.corners.at(face_id.at(ind_f).at(0) * 2 + 1)), + cv::Point(b.corners.at(face_id.at(ind_f).at(2) * 2), + b.corners.at(face_id.at(ind_f).at(2) * 2 + 1)), colors[b.cl], 2); + cv::line(frames[bi], cv::Point(b.corners.at(face_id.at(ind_f).at(1) * 2), + b.corners.at(face_id.at(ind_f).at(1) * 2 + 1)), + cv::Point(b.corners.at(face_id.at(ind_f).at(3) * 2), + b.corners.at(face_id.at(ind_f).at(3) * 2 + 1)), colors[b.cl], 2); + } } } + // draw label + cv::Size text_size = getTextSize(classesNames[b.cl], cv::FONT_HERSHEY_SIMPLEX, font_scale, thickness, &baseline); + cv::rectangle(frames[bi], cv::Point(b.corners.at(face_id.at(0).at(0) * 2), + b.corners.at(face_id.at(0).at(0) * 2 + 1)), + cv::Point((b.corners.at(face_id.at(0).at(0) * 2) + text_size.width - 2), + (b.corners.at(face_id.at(0).at(0) * 2 + 1)) - text_size.height - 2), colors[b.cl], -1); + cv::putText(frames[bi], classesNames[b.cl], cv::Point(b.corners.at(face_id.at(0).at(0) * 2), + b.corners.at(face_id.at(0).at(0) * 2 + 1) - (baseline / 2)), + cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), thickness); } - // draw label - cv::Size text_size = getTextSize(classesNames[b.cl], cv::FONT_HERSHEY_SIMPLEX, font_scale, thickness, &baseline); - cv::rectangle(frame, cv::Point(b.corners.at(face_id.at(0).at(0) * 2), - b.corners.at(face_id.at(0).at(0) * 2 + 1)), - cv::Point((b.corners.at(face_id.at(0).at(0) * 2) + text_size.width - 2), - (b.corners.at(face_id.at(0).at(0) * 2 + 1)) - text_size.height - 2), colors[b.cl], -1); - cv::putText(frame, classesNames[b.cl], cv::Point(b.corners.at(face_id.at(0).at(0) * 2), - b.corners.at(face_id.at(0).at(0) * 2 + 1) - (baseline / 2)), - cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), thickness); } - return frame; } }} diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp index ef3e161..dfc38f9 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenternetDetection3DTrack.cpp @@ -3,12 +3,14 @@ namespace tk { namespace dnn { -bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n_classes){ - std::cout<<(tensor_path).c_str()<<"\n"; + +bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh) { netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); dim = netRT->input_dim; dim.c = 3; + nBatches = n_batches; + confThreshold = conf_thresh; init_preprocessing(); init_pre_inf(); @@ -46,13 +48,13 @@ bool CenternetDetection3DTrack::init_preprocessing(){ checkCuda(cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice)); checkCuda(cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice)); #else - checkCuda(cudaMallocHost(&input, sizeof(dnnType)*dim.tot())); + checkCuda(cudaMallocHost(&input, sizeof(dnnType)*dim.tot() * nBatches)); mean << 0.40789655, 0.44719303, 0.47026116; stddev << 0.2886383, 0.27408165, 0.27809834; #endif - checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot())); + checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches)); checkCuda(cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot())); checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) ); } @@ -276,20 +278,20 @@ void CenternetDetection3DTrack::_get_additional_inputs(){ //None no additional input } -void CenternetDetection3DTrack::pre_inf(){ +void CenternetDetection3DTrack::pre_inf(const int bi){ TKDNN_TSTART tk::dnn::dataDim_t dim_aus; pre_phase_net->infer(dim_aus, nullptr); TKDNN_TSTOP checkCuda( cudaDeviceSynchronize() ); - checkCuda( cudaMemcpy(input_d, pre_phase_net->layers[pre_phase_net->num_layers-1]->dstData, netRT->input_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice) ); + checkCuda( cudaMemcpy(input_d+ netRT->input_dim.tot()*bi, pre_phase_net->layers[pre_phase_net->num_layers-1]->dstData, netRT->input_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice) ); checkCuda( cudaDeviceSynchronize() ); } -void CenternetDetection3DTrack::preprocess(cv::Mat &frame){ - // -----------------------------------pre-process ------------------------------------------ - - cv::Size sz = originalSize; +void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi){ + // -----------------------------------pre-process ------------------------------------------ + batchTracked.clear(); + cv::Size sz = originalSize[bi]; cv::Size sz_old; float scale = 1.0; float new_height = sz.height * scale; @@ -302,7 +304,7 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame){ // float s = new_width >= new_height ? new_width : new_height; // ----------- get_affine_transform // rot_rad = pi * 0 / 100 --> 0 - dim.print(); + //dim.print(); src.at(0,0)=c[0]; src.at(0,1)=c[1]; src.at(1,0)=c[0]; @@ -389,7 +391,7 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame){ checkCuda( cudaDeviceSynchronize() ); iter0=false; } - pre_inf(); + pre_inf(bi); checkCuda( cudaMemcpy(img_d, input_pre_inf_d, dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice) ); checkCuda( cudaDeviceSynchronize() ); @@ -587,17 +589,17 @@ void CenternetDetection3DTrack::tracking(){ } -void CenternetDetection3DTrack::postprocess(){ +void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { dnnType *rt_out[9]; - rt_out[0] = (dnnType *)netRT->buffersRT[1]; - rt_out[1] = (dnnType *)netRT->buffersRT[2]; - rt_out[2] = (dnnType *)netRT->buffersRT[3]; - rt_out[3] = (dnnType *)netRT->buffersRT[4]; - rt_out[4] = (dnnType *)netRT->buffersRT[5]; - rt_out[5] = (dnnType *)netRT->buffersRT[6]; - rt_out[6] = (dnnType *)netRT->buffersRT[7]; - rt_out[7] = (dnnType *)netRT->buffersRT[8]; - rt_out[8] = (dnnType *)netRT->buffersRT[9]; + rt_out[0] = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi; + rt_out[1] = (dnnType *)netRT->buffersRT[2]+ netRT->buffersDIM[2].tot()*bi; + rt_out[2] = (dnnType *)netRT->buffersRT[3]+ netRT->buffersDIM[3].tot()*bi; + rt_out[3] = (dnnType *)netRT->buffersRT[4]+ netRT->buffersDIM[4].tot()*bi; + rt_out[4] = (dnnType *)netRT->buffersRT[5]+ netRT->buffersDIM[5].tot()*bi; + rt_out[5] = (dnnType *)netRT->buffersRT[6]+ netRT->buffersDIM[6].tot()*bi; + rt_out[6] = (dnnType *)netRT->buffersRT[7]+ netRT->buffersDIM[7].tot()*bi; + rt_out[7] = (dnnType *)netRT->buffersRT[8]+ netRT->buffersDIM[8].tot()*bi; + rt_out[8] = (dnnType *)netRT->buffersRT[9]+ netRT->buffersDIM[9].tot()*bi; // ------------------------------------ process -------------------------------------------- @@ -719,143 +721,144 @@ void CenternetDetection3DTrack::postprocess(){ } // track step tracking(); + batchTracked.push_back(tr_res); } -cv::Mat CenternetDetection3DTrack::draw(cv::Mat &frame) { - +void CenternetDetection3DTrack::draw(std::vector& frames) { + struct trackingRes t; float sc; int id; std::string txt; int baseline = 0; float font_scale = 0.8; - int thickness = 2; - for(int i=0; i vis_thresh){// && tr_res[i].active!=0) { - if(view2d) { - - - cv::rectangle(frame, cv::Point(tr_res[i].det_res.bb0.at(0,0), tr_res[i].det_res.bb0.at(0,1)), - cv::Point(tr_res[i].det_res.bb1.at(0,0), tr_res[i].det_res.bb1.at(0,1)), tr_colors[tr_res[i].color], thickness); - cv::rectangle(frame, cv::Point(tr_res[i].det_res.bb0.at(0,0), - tr_res[i].det_res.bb0.at(0,1) - text_size.height - thickness), - cv::Point(tr_res[i].det_res.bb0.at(0,0) + text_size.width, - tr_res[i].det_res.bb0.at(0,1)), tr_colors[tr_res[i].color], -1); - - cv::putText(frame, txt, cv::Point(tr_res[i].det_res.bb0.at(0,0), - tr_res[i].det_res.bb0.at(0,1) - thickness -1), - cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1); + int thickness = 2; + for(int bi=0; bi vis_thresh){// && t.active!=0) { + if(view2d) { + cv::rectangle(frames[bi], cv::Point(t.det_res.bb0.at(0,0), t.det_res.bb0.at(0,1)), + cv::Point(t.det_res.bb1.at(0,0), t.det_res.bb1.at(0,1)), tr_colors[t.color], thickness); + cv::rectangle(frames[bi], cv::Point(t.det_res.bb0.at(0,0), + t.det_res.bb0.at(0,1) - text_size.height - thickness), + cv::Point(t.det_res.bb0.at(0,0) + text_size.width, + t.det_res.bb0.at(0,1)), tr_colors[t.color], -1); + + cv::putText(frames[bi], txt, cv::Point(t.det_res.bb0.at(0,0), + t.det_res.bb0.at(0,1) - thickness -1), + cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1); - cv::arrowedLine(frame, cv::Point((int)tr_res[i].det_res.ct.at(0,0), - (int)tr_res[i].det_res.ct.at(0,1)), - cv::Point((int)(tr_res[i].det_res.ct.at(0,0) + tr_res[i].det_res.tr.at(0,0)), - (int)(tr_res[i].det_res.ct.at(0,1) + tr_res[i].det_res.tr.at(0,1))), - cv::Scalar(255, 0, 255), 2); - } - //3d - if(!view2d && tr_res[i].det_res.z > 1){ - r.at(0,0) = std::cos(tr_res[i].det_res.rot_y); - r.at(0,2) = std::sin(tr_res[i].det_res.rot_y); - r.at(2,0) = -std::sin(tr_res[i].det_res.rot_y); - r.at(2,2) = std::cos(tr_res[i].det_res.rot_y); - - corners.at(0,0) = tr_res[i].det_res.dim[2]/2; - corners.at(0,1) = tr_res[i].det_res.dim[2]/2; - corners.at(0,2) = -tr_res[i].det_res.dim[2]/2; - corners.at(0,3) = -tr_res[i].det_res.dim[2]/2; - corners.at(0,4) = tr_res[i].det_res.dim[2]/2; - corners.at(0,5) = tr_res[i].det_res.dim[2]/2; - corners.at(0,6) = -tr_res[i].det_res.dim[2]/2; - corners.at(0,7) = -tr_res[i].det_res.dim[2]/2; - - corners.at(1,4) = -tr_res[i].det_res.dim[0]; - corners.at(1,5) = -tr_res[i].det_res.dim[0]; - corners.at(1,6) = -tr_res[i].det_res.dim[0]; - corners.at(1,7) = -tr_res[i].det_res.dim[0]; - - corners.at(2,0) = tr_res[i].det_res.dim[1]/2; - corners.at(2,1) = -tr_res[i].det_res.dim[1]/2; - corners.at(2,2) = -tr_res[i].det_res.dim[1]/2; - corners.at(2,3) = tr_res[i].det_res.dim[1]/2; - corners.at(2,4) = tr_res[i].det_res.dim[1]/2; - corners.at(2,5) = -tr_res[i].det_res.dim[1]/2; - corners.at(2,6) = -tr_res[i].det_res.dim[1]/2; - corners.at(2,7) = tr_res[i].det_res.dim[1]/2; - - cv::Mat aus = r * corners; - - for(int k=0; k<8; k++) { - aus.at(0,k) += tr_res[i].det_res.x; - aus.at(1,k) += tr_res[i].det_res.y; - aus.at(2,k) += tr_res[i].det_res.z; + cv::arrowedLine(frames[bi], cv::Point((int)t.det_res.ct.at(0,0), + (int)t.det_res.ct.at(0,1)), + cv::Point((int)(t.det_res.ct.at(0,0) + t.det_res.tr.at(0,0)), + (int)(t.det_res.ct.at(0,1) + t.det_res.tr.at(0,1))), + cv::Scalar(255, 0, 255), 2); } - - // corners.copyTo(pts3DHomo(cv::Rect(0, 0, 8, 3))); - for(int k1=0; k1<3; k1++) { - for(int k2=0; k2<8; k2++) - pts3DHomo.at(k1,k2) = aus.at(k1,k2); - } - - aus.release(); - aus = calibs * pts3DHomo; - std::vector res_corners; - for(int k=0; k<8; k++) { - res_corners.push_back(aus.at(0,k) / aus.at(2,k)); - res_corners.push_back(aus.at(1,k) / aus.at(2,k)); - } - aus.release(); - for(int ind_f = 3; ind_f>=0; ind_f--) { - for(int j=0; j<4; j++) { - cv::line(frame, cv::Point((int)res_corners.at(face_id.at(ind_f).at(j) * 2), - (int)res_corners.at(face_id.at(ind_f).at(j) * 2 + 1)), - cv::Point((int)res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2), - (int)res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), - tr_colors[tr_res[i].color], 2); - if(ind_f == 0 && j==3) { - cv::line(frame, cv::Point((int)res_corners.at(face_id.at(ind_f).at(0) * 2), - (int)res_corners.at(face_id.at(ind_f).at(0) * 2 + 1)), - cv::Point((int)res_corners.at(face_id.at(ind_f).at(2) * 2), - (int)res_corners.at(face_id.at(ind_f).at(2) * 2 + 1)), tr_colors[tr_res[i].color], 2); - cv::line(frame, cv::Point((int)res_corners.at(face_id.at(ind_f).at(1) * 2), - (int)res_corners.at(face_id.at(ind_f).at(1) * 2 + 1)), - cv::Point((int)res_corners.at(face_id.at(ind_f).at(3) * 2), - (int)res_corners.at(face_id.at(ind_f).at(3) * 2 + 1)), tr_colors[tr_res[i].color], 2); + //3d + if(!view2d && t.det_res.z > 1){ + r.at(0,0) = std::cos(t.det_res.rot_y); + r.at(0,2) = std::sin(t.det_res.rot_y); + r.at(2,0) = -std::sin(t.det_res.rot_y); + r.at(2,2) = std::cos(t.det_res.rot_y); + + corners.at(0,0) = t.det_res.dim[2]/2; + corners.at(0,1) = t.det_res.dim[2]/2; + corners.at(0,2) = -t.det_res.dim[2]/2; + corners.at(0,3) = -t.det_res.dim[2]/2; + corners.at(0,4) = t.det_res.dim[2]/2; + corners.at(0,5) = t.det_res.dim[2]/2; + corners.at(0,6) = -t.det_res.dim[2]/2; + corners.at(0,7) = -t.det_res.dim[2]/2; + + corners.at(1,4) = -t.det_res.dim[0]; + corners.at(1,5) = -t.det_res.dim[0]; + corners.at(1,6) = -t.det_res.dim[0]; + corners.at(1,7) = -t.det_res.dim[0]; + + corners.at(2,0) = t.det_res.dim[1]/2; + corners.at(2,1) = -t.det_res.dim[1]/2; + corners.at(2,2) = -t.det_res.dim[1]/2; + corners.at(2,3) = t.det_res.dim[1]/2; + corners.at(2,4) = t.det_res.dim[1]/2; + corners.at(2,5) = -t.det_res.dim[1]/2; + corners.at(2,6) = -t.det_res.dim[1]/2; + corners.at(2,7) = t.det_res.dim[1]/2; + + cv::Mat aus = r * corners; + + for(int k=0; k<8; k++) { + aus.at(0,k) += t.det_res.x; + aus.at(1,k) += t.det_res.y; + aus.at(2,k) += t.det_res.z; + } + + // corners.copyTo(pts3DHomo(cv::Rect(0, 0, 8, 3))); + for(int k1=0; k1<3; k1++) { + for(int k2=0; k2<8; k2++) + pts3DHomo.at(k1,k2) = aus.at(k1,k2); + } + + aus.release(); + aus = calibs * pts3DHomo; + std::vector res_corners; + for(int k=0; k<8; k++) { + res_corners.push_back(aus.at(0,k) / aus.at(2,k)); + res_corners.push_back(aus.at(1,k) / aus.at(2,k)); + } + aus.release(); + for(int ind_f = 3; ind_f>=0; ind_f--) { + for(int j=0; j<4; j++) { + cv::line(frames[bi], cv::Point((int)res_corners.at(face_id.at(ind_f).at(j) * 2), + (int)res_corners.at(face_id.at(ind_f).at(j) * 2 + 1)), + cv::Point((int)res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2), + (int)res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), + tr_colors[t.color], 2); + if(ind_f == 0 && j==3) { + cv::line(frames[bi], cv::Point((int)res_corners.at(face_id.at(ind_f).at(0) * 2), + (int)res_corners.at(face_id.at(ind_f).at(0) * 2 + 1)), + cv::Point((int)res_corners.at(face_id.at(ind_f).at(2) * 2), + (int)res_corners.at(face_id.at(ind_f).at(2) * 2 + 1)), tr_colors[t.color], 2); + cv::line(frames[bi], cv::Point((int)res_corners.at(face_id.at(ind_f).at(1) * 2), + (int)res_corners.at(face_id.at(ind_f).at(1) * 2 + 1)), + cv::Point((int)res_corners.at(face_id.at(ind_f).at(3) * 2), + (int)res_corners.at(face_id.at(ind_f).at(3) * 2 + 1)), tr_colors[t.color], 2); + } } } - } - float bb0=(1 << 10), bb1=0, bb2=(1 << 10), bb3=0; - for(int k=0; k<8; k++) { - if(res_corners[2*k]bb1) - bb1=res_corners[2*k]; - if(res_corners[2*k+1]bb3) - bb3=res_corners[2*k+1]; - - } - // if(not no_bbox): - // cv::rectangle(frame, cv::Point(bb0, bb2), cv::Point(bb1, bb3), - // tr_colors[tr_res[i].color], thickness); - cv::rectangle(frame, cv::Point(bb0, bb2 - text_size.height - thickness), - cv::Point(bb0 + text_size.width, bb2), tr_colors[tr_res[i].color], -1); - - cv::putText(frame, txt, cv::Point(bb0, bb2 - thickness -1), cv::FONT_HERSHEY_SIMPLEX, - font_scale, cv::Scalar(255, 255, 255), 1); + float bb0=(1 << 10), bb1=0, bb2=(1 << 10), bb3=0; + for(int k=0; k<8; k++) { + if(res_corners[2*k]bb1) + bb1=res_corners[2*k]; + if(res_corners[2*k+1]bb3) + bb3=res_corners[2*k+1]; + + } + // if(not no_bbox): + // cv::rectangle(frame, cv::Point(bb0, bb2), cv::Point(bb1, bb3), + // tr_colors[t.color], thickness); + cv::rectangle(frames[bi], cv::Point(bb0, bb2 - text_size.height - thickness), + cv::Point(bb0 + text_size.width, bb2), tr_colors[t.color], -1); + + cv::putText(frames[bi], txt, cv::Point(bb0, bb2 - thickness -1), cv::FONT_HERSHEY_SIMPLEX, + font_scale, cv::Scalar(255, 255, 255), 1); - cv::arrowedLine(frame, cv::Point((int)((bb0 + bb1)/2), (int)((bb2 + bb3)/2)), - cv::Point((int)((bb0 + bb1)/2 + tr_res[i].det_res.tr.at(0,0)), - (int)((bb2 + bb3)/2 + tr_res[i].det_res.tr.at(0,1))), - cv::Scalar(255, 0, 255), 2); + cv::arrowedLine(frames[bi], cv::Point((int)((bb0 + bb1)/2), (int)((bb2 + bb3)/2)), + cv::Point((int)((bb0 + bb1)/2 + t.det_res.tr.at(0,0)), + (int)((bb2 + bb3)/2 + t.det_res.tr.at(0,1))), + cv::Scalar(255, 0, 255), 2); + } } } - } - return frame; } }} -- 2.52.0 From 2367519799ef3eb9806387dab94ec12d17e77649 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Fri, 30 Apr 2021 17:10:51 +0200 Subject: [PATCH 060/162] Add the calibration matrix reading for CenterTrack Signed-off-by: Davide Sapienza --- demo/demo/demo3D.cpp | 25 +++++++-- include/tkDNN/CenternetDetection3D.h | 4 +- include/tkDNN/CenternetDetection3DTrack.h | 10 ++-- include/tkDNN/DetectionNN3D.h | 10 ++-- src/CenternetDetection3D.cpp | 5 +- src/CenternetDetection3DTrack.cpp | 65 +++++++++++++---------- 6 files changed, 75 insertions(+), 44 deletions(-) diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp index c35ac51..558e6af 100644 --- a/demo/demo/demo3D.cpp +++ b/demo/demo/demo3D.cpp @@ -70,8 +70,17 @@ int main(int argc, char *argv[]) { default: FatalError("Network type not allowed (3rd parameter)\n"); } - - detNN->init(net, n_classes, n_batch, conf_thresh); + std::vector calibs; + // cv::Mat calib = cv::Mat::zeros(cv::Size(3,3), CV_32F); + // calib.at(0,0) = 864.1243196486207;// * 512.0;//884.081444212;//864.1243196486207 * 512.0;// 633.0; + // calib.at(0,2) = 726.7271690557819;// * 512.0;//0.0;//726.7271690557819 * 512.0;// 0.0; //w/2 + // calib.at(1,1) = 883.6552349216504;// * 512.0;//884.081444212;//883.6552349216504 * 512.0;// 633.0; + // calib.at(1,2) = 506.8548506986564;// * 512.0;//0.0;//506.8548506986564 * 512.0;// 0.0; //h/2 + // calibs.push_back(calib); + // calibs.push_back(calib); + // calibs.push_back(calib); + // calibs.push_back(calib); + detNN->init(net, n_classes, n_batch, conf_thresh, calibs); gRun = true; @@ -87,7 +96,8 @@ int main(int argc, char *argv[]) { int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT); resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h)); } - + cv::Size sz_resize = cv::Size(512,512); + std::vector sz_orig; cv::Mat frame; if(show) cv::namedWindow("detection", cv::WINDOW_NORMAL); @@ -98,12 +108,15 @@ int main(int argc, char *argv[]) { while(gRun) { batch_dnn_input.clear(); batch_frame.clear(); + sz_orig.clear(); for(int bi=0; bi< n_batch; ++bi){ cap >> frame; if(!frame.data) break; - + sz_orig.push_back(frame.size()); + if(calibs.size() != 0) + resize(frame, frame, sz_resize); batch_frame.push_back(frame); // this will be resized to the net format @@ -113,11 +126,13 @@ int main(int argc, char *argv[]) { break; //inference - detNN->update(batch_dnn_input, n_batch); + detNN->update(batch_dnn_input, n_batch, false, nullptr, false, sz_orig); detNN->draw(batch_frame); if(show){ for(int bi=0; bi< n_batch; ++bi){ + if(calibs.size() != 0) + resize(batch_frame[bi], batch_frame[bi], sz_orig[bi]); cv::imshow("detection", batch_frame[bi]); cv::waitKey(1); } diff --git a/include/tkDNN/CenternetDetection3D.h b/include/tkDNN/CenternetDetection3D.h index 668440c..cbffa22 100644 --- a/include/tkDNN/CenternetDetection3D.h +++ b/include/tkDNN/CenternetDetection3D.h @@ -82,8 +82,8 @@ public: CenternetDetection3D() {}; ~CenternetDetection3D() {}; - bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3); - void preprocess(cv::Mat &frame, const int bi=0); + bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3, const std::vector& k_calibs=std::vector()); + void preprocess(cv::Mat &frame, const int bi=0, const std::vector& stream_size=std::vector()); void postprocess(const int bi=0,const bool mAP=false); void draw(std::vector& frames); }; diff --git a/include/tkDNN/CenternetDetection3DTrack.h b/include/tkDNN/CenternetDetection3DTrack.h index 809d0c9..451bf6e 100644 --- a/include/tkDNN/CenternetDetection3DTrack.h +++ b/include/tkDNN/CenternetDetection3DTrack.h @@ -74,6 +74,10 @@ private: #endif float *d_ptrs; + std::vector inputCalibs; + + std::vector sz_old; + cv::Mat src; cv::Mat dst; cv::Mat dst2; @@ -124,7 +128,7 @@ private: /* visualization */ cv::Mat r; - cv::Mat calibs; + std::vector calibs; cv::Mat corners, pts3DHomo; std::vector> face_id; @@ -163,8 +167,8 @@ public: tk::dnn::Network *pre_phase_net = nullptr; CenternetDetection3DTrack() {}; ~CenternetDetection3DTrack() {}; - bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3); - void preprocess(cv::Mat &frame, const int bi=0); + bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3, const std::vector& k_calibs=std::vector()); + void preprocess(cv::Mat &frame, const int bi=0, const std::vector& stream_size=std::vector()); void postprocess(const int bi=0,const bool mAP=false); void draw(std::vector& frames); }; diff --git a/include/tkDNN/DetectionNN3D.h b/include/tkDNN/DetectionNN3D.h index 65cd728..7320bef 100644 --- a/include/tkDNN/DetectionNN3D.h +++ b/include/tkDNN/DetectionNN3D.h @@ -54,7 +54,7 @@ class DetectionNN3D { * @param frame original frame to adapt for inference. * @param bi batch index */ - virtual void preprocess(cv::Mat &frame, const int bi=0) = 0; + virtual void preprocess(cv::Mat &frame, const int bi=0 , const std::vector& stream_size=std::vector()) = 0; /** * This method postprocess the output of the NN to obtain the correct @@ -87,7 +87,8 @@ class DetectionNN3D { * @param n_batches maximum number of batches to use in inference. * @return true if everything is correct, false otherwise. */ - virtual bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3) = 0; + virtual bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, + const float conf_thresh=0.3, const std::vector& k_calibs=std::vector()) = 0; /** * This method performs the whole detection of the NN. @@ -100,7 +101,8 @@ class DetectionNN3D { * @param mAP set to true only if all the probabilities for a bounding * box are needed, as in some cases for the mAP calculation. */ - void update(std::vector& frames, const int cur_batches=1, bool save_times=false, std::ofstream *times=nullptr, const bool mAP=false){ + void update(std::vector& frames, const int cur_batches=1, bool save_times=false, + std::ofstream *times=nullptr, const bool mAP=false, const std::vector& stream_size=std::vector()){ if(save_times && times==nullptr) FatalError("save_times set to true, but no valid ofstream given"); if(cur_batches > nBatches) @@ -114,7 +116,7 @@ class DetectionNN3D { if(!frames[bi].data) FatalError("No image data feed to detection"); originalSize.push_back(frames[bi].size()); - preprocess(frames[bi], bi); + preprocess(frames[bi], bi, stream_size); } TKDNN_TSTOP pre_stats.push_back(t_ns); diff --git a/src/CenternetDetection3D.cpp b/src/CenternetDetection3D.cpp index 53b3cf7..73e4215 100644 --- a/src/CenternetDetection3D.cpp +++ b/src/CenternetDetection3D.cpp @@ -3,7 +3,8 @@ namespace tk { namespace dnn { -bool CenternetDetection3D::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh) { +bool CenternetDetection3D::init(const std::string& tensor_path, const int n_classes, const int n_batches, + const float conf_thresh, const std::vector& k_calibs) { std::cout<<(tensor_path).c_str()<<"\n"; netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); classes = n_classes; @@ -156,7 +157,7 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); } -void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi){ +void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi, const std::vector& stream_size){ // -----------------------------------pre-process ------------------------------------------ // auto start_t = std::chrono::steady_clock::now(); diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp index dfc38f9..02db674 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenternetDetection3DTrack.cpp @@ -4,14 +4,15 @@ namespace tk { namespace dnn { -bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh) { +bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n_classes, const int n_batches, + const float conf_thresh, const std::vector& k_calibs) { netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); dim = netRT->input_dim; dim.c = 3; nBatches = n_batches; confThreshold = conf_thresh; - + inputCalibs = k_calibs; init_preprocessing(); init_pre_inf(); init_postprocessing(); @@ -37,7 +38,10 @@ bool CenternetDetection3DTrack::init_preprocessing(){ dst2.at(2,0)=dst2.at(1,0) + (-dst2.at(0,1)+dst2.at(1,1) ); dst2.at(2,1)=dst2.at(1,1) + (dst2.at(0,0)-dst2.at(1,0) ); - + for(int bi=0; bi(0,0) = 633.0; - calibs.at(0,1) = 0.0; - calibs.at(0,2) = 0.0; //w/2 - calibs.at(0,3) = 0.0; - calibs.at(1,0) = 0.0; - calibs.at(1,1) = 633.0; - calibs.at(1,2) = 0.0; //h/2 - calibs.at(1,3) = 0.0; - calibs.at(2,0) = 0.0; - calibs.at(2,1) = 0.0; - calibs.at(2,2) = 1.0; - calibs.at(2,3) = 0.0; + for(int bi=0; bi(0,0) = 633.0; + calibs_.at(1,1) = 633.0; + calibs_.at(2,2) = 1.0; + } + calibs_.at(2,2) = 1.0; + calibs.push_back(calibs_); + } // Alloc array used in the kernel checkCuda( cudaMalloc(&src_out, K *sizeof(float)) ); @@ -288,17 +289,25 @@ void CenternetDetection3DTrack::pre_inf(const int bi){ checkCuda( cudaDeviceSynchronize() ); } -void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi){ +void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi, const std::vector& stream_size){ // -----------------------------------pre-process ------------------------------------------ batchTracked.clear(); cv::Size sz = originalSize[bi]; - cv::Size sz_old; float scale = 1.0; float new_height = sz.height * scale; float new_width = sz.width * scale; - if(sz.height != sz_old.height && sz.width != sz_old.width){ - calibs.at(0,2) = new_width / 2.0f; - calibs.at(1,2) = new_height /2.0f; + if(sz.height != sz_old[bi].height && sz.width != sz_old[bi].width){ + if(inputCalibs.size() == 0 || inputCalibs[bi].empty()) { + calibs[bi].at(0,2) = new_width / 2.0f; + calibs[bi].at(1,2) = new_height /2.0f; + } + else { + calibs[bi].at(0,0) = inputCalibs[bi].at(0,0) * dim.w / stream_size[bi].width; + calibs[bi].at(0,2) = inputCalibs[bi].at(0,2) * dim.w / stream_size[bi].width; + calibs[bi].at(1,1) = inputCalibs[bi].at(1,1) * dim.h / stream_size[bi].height; + calibs[bi].at(1,2) = inputCalibs[bi].at(1,2) * dim.h / stream_size[bi].height; + } + float c[] = {new_width / 2.0f, new_height /2.0f}; float s[] = {dim.w, dim.h}; // float s = new_width >= new_height ? new_width : new_height; @@ -324,7 +333,7 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi){ trans2 = cv::getAffineTransform( dst2, src ); trans2.convertTo(trans_out, CV_32F); } - sz_old = sz; + sz_old[bi] = sz; #ifdef OPENCV_CUDACONTRIB std::cout<<"OPENCV CPMTROB\n"; cv::cuda::GpuMat im_Orig; @@ -358,7 +367,7 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi){ #else std::cout<<"NO OPENCV CPMTROB\n"; cv::Mat imageF; - // resize(frame, imageF, cv::Size(new_width, new_height)); + //resize(frame, imageF, cv::Size(512, 512)); imageF = frame; sz = imageF.size(); @@ -701,9 +710,9 @@ void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { new_det_res.dim[2] = dim_[i+2*K]; // unproject_2d_to_3d - new_det_res.z = dep[i] - calibs.at(2,3); - new_det_res.x = ((float)new_det_res.ct.at(0,0) * dep[i] - calibs.at(0,3) - calibs.at(0,2) * new_det_res.z) / calibs.at(0,0); - new_det_res.y = ((float)new_det_res.ct.at(0,1) * dep[i] - calibs.at(1,3) - calibs.at(1,2) * new_det_res.z) / calibs.at(1,1) + (dim_[i] / 2); + new_det_res.z = dep[i] - calibs[bi].at(2,3); + new_det_res.x = ((float)new_det_res.ct.at(0,0) * dep[i] - calibs[bi].at(0,3) - calibs[bi].at(0,2) * new_det_res.z) / calibs[bi].at(0,0); + new_det_res.y = ((float)new_det_res.ct.at(0,1) * dep[i] - calibs[bi].at(1,3) - calibs[bi].at(1,2) * new_det_res.z) / calibs[bi].at(1,1) + (dim_[i] / 2); // alpha2rot_y // idx = rot[:, 1] > rot[:, 5] @@ -714,7 +723,7 @@ void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { new_det_res.alpha = std::atan2(rot[2*K + i], rot[3*K + i]) -0.5 * M_PI; else new_det_res.alpha = std::atan2(rot[6*K + i], rot[7*K + i]) +0.5 * M_PI; - new_det_res.rot_y = (new_det_res.alpha + std::atan2((float)new_det_res.ct.at(0,0) - calibs.at(0,2), calibs.at(0,0))); + new_det_res.rot_y = (new_det_res.alpha + std::atan2((float)new_det_res.ct.at(0,0) - calibs[bi].at(0,2), calibs[bi].at(0,0))); new_det_res.ct = new_det_res.ct + new_det_res.tr; //dest det_res.push_back(new_det_res); @@ -804,7 +813,7 @@ void CenternetDetection3DTrack::draw(std::vector& frames) { } aus.release(); - aus = calibs * pts3DHomo; + aus = calibs[bi] * pts3DHomo; std::vector res_corners; for(int k=0; k<8; k++) { res_corners.push_back(aus.at(0,k) / aus.at(2,k)); -- 2.52.0 From ff6e0e010adc120bd010c605f82355111758dd08 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Fri, 30 Apr 2021 22:26:33 +0200 Subject: [PATCH 061/162] Fix a bug with batch > 1 Signed-off-by: Davide Sapienza --- include/tkDNN/CenternetDetection3DTrack.h | 9 +- src/CenternetDetection3DTrack.cpp | 184 +++++++++++----------- 2 files changed, 99 insertions(+), 94 deletions(-) diff --git a/include/tkDNN/CenternetDetection3DTrack.h b/include/tkDNN/CenternetDetection3DTrack.h index 451bf6e..258fe9d 100644 --- a/include/tkDNN/CenternetDetection3DTrack.h +++ b/include/tkDNN/CenternetDetection3DTrack.h @@ -148,10 +148,9 @@ private: std::vector det_res; int count_det; //tracks - std::vector tr_res; - std::vector> batchTracked; - int count_tr; - int track_id=0; + std::vector> tr_res; + std::vector count_tr; + std::vector track_id; bool init_preprocessing(); @@ -161,7 +160,7 @@ private: void pre_inf(const int bi); void _get_additional_inputs(); cv::Mat transform_preds_with_trans(float x1, float x2); - void tracking(); + void tracking(const int bi); public: tk::dnn::Network *pre_phase_net = nullptr; diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp index 02db674..bdc1c59 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenternetDetection3DTrack.cpp @@ -17,8 +17,6 @@ bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n init_pre_inf(); init_postprocessing(); init_visualization(n_classes); - - count_tr = 0; } bool CenternetDetection3DTrack::init_preprocessing(){ @@ -201,6 +199,11 @@ bool CenternetDetection3DTrack::init_postprocessing(){ // Alloc array used in the kernel checkCuda( cudaMalloc(&src_out, K *sizeof(float)) ); checkCuda( cudaMalloc(&ids_out, K *sizeof(int)) ); + + for(int bi=0; bi& stream_size){ // -----------------------------------pre-process ------------------------------------------ - batchTracked.clear(); cv::Size sz = originalSize[bi]; float scale = 1.0; float new_height = sz.height * scale; @@ -414,8 +416,7 @@ cv::Mat CenternetDetection3DTrack::transform_preds_with_trans(float x1, float x2 return trans_out * target_coords; } -void CenternetDetection3DTrack::tracking(){ - +void CenternetDetection3DTrack::tracking(const int bi) { float item_size[count_det]; int item_cl[count_det]; float dets[2*count_det]; @@ -427,44 +428,44 @@ void CenternetDetection3DTrack::tracking(){ dets[i*2+1] = det_res[i].ct.at(0,1); } - float track_size[count_tr]; - int track_cl[count_tr]; - float tracks[2*count_tr]; - for(int i=0; i(0,0) - tr_res[i].det_res.bb0.at(0,0)) * - (tr_res[i].det_res.bb1.at(0,1) - tr_res[i].det_res.bb0.at(0,1)); - track_cl[i] = tr_res[i].det_res.cl; - tracks[i*2] = tr_res[i].det_res.ct.at(0,0); - tracks[i*2+1] = tr_res[i].det_res.ct.at(0,1); + float track_size[count_tr[bi]]; + int track_cl[count_tr[bi]]; + float tracks[2*count_tr[bi]]; + for(int i=0; i(0,0) - tr_res[bi][i].det_res.bb0.at(0,0)) * + (tr_res[bi][i].det_res.bb1.at(0,1) - tr_res[bi][i].det_res.bb0.at(0,1)); + track_cl[i] = tr_res[bi][i].det_res.cl; + tracks[i*2] = tr_res[bi][i].det_res.ct.at(0,0); + tracks[i*2+1] = tr_res[bi][i].det_res.ct.at(0,1); } - float dist[count_tr*count_det]; + float dist[count_tr[bi]*count_det]; bool invalid; - for(int i=0; i track_size[i] || dist[j*count_tr+i] > item_size[j] || item_cl[j] != track_cl[i]; - dist[j*count_tr+i] = dist[j*count_tr+i] + invalid * (1 << 18); + invalid = dist[j*count_tr[bi]+i] > track_size[i] || dist[j*count_tr[bi]+i] > item_size[j] || item_cl[j] != track_cl[i]; + dist[j*count_tr[bi]+i] = dist[j*count_tr[bi]+i] + invalid * (1 << 18); } } - int matched_indices[2*count_tr]; + int matched_indices[2*count_tr[bi]]; float min_tr; int min_idtr=-1; - for(int i=0; i new_tr_res; int id_new_tr=0; - for(int i=0; i new_thresh) { count_tr_ ++; @@ -583,17 +583,24 @@ void CenternetDetection3DTrack::tracking(){ new_tr_res_.det_res.y = det_res[i].y; new_tr_res_.det_res.z = det_res[i].z; new_tr_res_.det_res.rot_y = det_res[i].rot_y; - new_tr_res_.tracking_id = track_id++; + new_tr_res_.tracking_id = track_id[bi]++; new_tr_res_.age = 1; new_tr_res_.active = 1; new_tr_res_.color = rand() % 256; - tr_res.push_back(new_tr_res_); + if(tr_res.size() <= bi) { + std::vector v_new_tr_res_; + v_new_tr_res_.push_back(new_tr_res_); + tr_res.push_back(v_new_tr_res_); + } + else + tr_res[bi].push_back(new_tr_res_); } } - count_tr = count_tr_; + + count_tr[bi] = count_tr_; - if(track_id==1000) - track_id=0; + if(track_id[bi]==1000) + track_id[bi]=0; det_res.clear(); } @@ -729,8 +736,7 @@ void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { } // track step - tracking(); - batchTracked.push_back(tr_res); + tracking(bi); } void CenternetDetection3DTrack::draw(std::vector& frames) { @@ -743,8 +749,8 @@ void CenternetDetection3DTrack::draw(std::vector& frames) { int thickness = 2; for(int bi=0; bi Date: Mon, 3 May 2021 18:56:07 +0200 Subject: [PATCH 062/162] Fix a bug in the draw function of CenterTrack. Signed-off-by: Davide Sapienza --- src/CenternetDetection3DTrack.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp index bdc1c59..312bcdb 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenternetDetection3DTrack.cpp @@ -749,7 +749,7 @@ void CenternetDetection3DTrack::draw(std::vector& frames) { int thickness = 2; for(int bi=0; bi Date: Tue, 4 May 2021 11:21:05 +0200 Subject: [PATCH 063/162] Fix tracker for batch size > 1 --- include/tkDNN/CenternetDetection3DTrack.h | 12 +- src/CenternetDetection3DTrack.cpp | 168 +++++++++++----------- 2 files changed, 92 insertions(+), 88 deletions(-) diff --git a/include/tkDNN/CenternetDetection3DTrack.h b/include/tkDNN/CenternetDetection3DTrack.h index 451bf6e..dd092fa 100644 --- a/include/tkDNN/CenternetDetection3DTrack.h +++ b/include/tkDNN/CenternetDetection3DTrack.h @@ -1,11 +1,11 @@ #ifndef CENTERNETDETECTION3DTRACK_H #define CENTERNETDETECTION3DTRACK_H +#include +#include "opencv2/opencv.hpp" #include "kernels.h" #include "utils.h" #include "tkdnn.h" -#include -#include "opencv2/opencv.hpp" #include #include #include // std::iota @@ -51,7 +51,7 @@ struct trackingRes class CenternetDetection3DTrack : public DetectionNN3D { -private: +public: tk::dnn::dataDim_t dim; tk::dnn::dataDim_t dim2; tk::dnn::dataDim_t dim_hm; @@ -148,9 +148,9 @@ private: std::vector det_res; int count_det; //tracks - std::vector tr_res; + std::vector> tr_res; std::vector> batchTracked; - int count_tr; + std::vector count_tr; int track_id=0; @@ -161,7 +161,7 @@ private: void pre_inf(const int bi); void _get_additional_inputs(); cv::Mat transform_preds_with_trans(float x1, float x2); - void tracking(); + void tracking(int bi); public: tk::dnn::Network *pre_phase_net = nullptr; diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp index 02db674..25c4c0b 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenternetDetection3DTrack.cpp @@ -13,12 +13,13 @@ bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n nBatches = n_batches; confThreshold = conf_thresh; inputCalibs = k_calibs; + tr_res.resize(nBatches); init_preprocessing(); init_pre_inf(); init_postprocessing(); init_visualization(n_classes); - count_tr = 0; + count_tr.resize(nBatches, 0); } bool CenternetDetection3DTrack::init_preprocessing(){ @@ -30,6 +31,7 @@ bool CenternetDetection3DTrack::init_preprocessing(){ trans2 = cv::Mat(cv::Size(3,2), CV_32F); trans_out = cv::Mat(cv::Size(3,2), CV_32F); + dst2.at(0,0)=width * 0.5; dst2.at(0,1)=width * 0.5; dst2.at(1,0)=width * 0.5; @@ -372,6 +374,8 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi, const s sz = imageF.size(); cv::warpAffine(imageF, imageF, trans, cv::Size(dim.w, dim.h), cv::INTER_LINEAR ); + + cv::imshow("warp", imageF); sz = imageF.size(); imageF.convertTo(imageF, CV_32FC3, 1/255.0); @@ -414,7 +418,7 @@ cv::Mat CenternetDetection3DTrack::transform_preds_with_trans(float x1, float x2 return trans_out * target_coords; } -void CenternetDetection3DTrack::tracking(){ +void CenternetDetection3DTrack::tracking(int bi){ float item_size[count_det]; int item_cl[count_det]; @@ -427,44 +431,44 @@ void CenternetDetection3DTrack::tracking(){ dets[i*2+1] = det_res[i].ct.at(0,1); } - float track_size[count_tr]; - int track_cl[count_tr]; - float tracks[2*count_tr]; - for(int i=0; i(0,0) - tr_res[i].det_res.bb0.at(0,0)) * - (tr_res[i].det_res.bb1.at(0,1) - tr_res[i].det_res.bb0.at(0,1)); - track_cl[i] = tr_res[i].det_res.cl; - tracks[i*2] = tr_res[i].det_res.ct.at(0,0); - tracks[i*2+1] = tr_res[i].det_res.ct.at(0,1); + float track_size[count_tr[bi]]; + int track_cl[count_tr[bi]]; + float tracks[2*count_tr[bi]]; + for(int i=0; i(0,0) - tr_res[bi][i].det_res.bb0.at(0,0)) * + (tr_res[bi][i].det_res.bb1.at(0,1) - tr_res[bi][i].det_res.bb0.at(0,1)); + track_cl[i] = tr_res[bi][i].det_res.cl; + tracks[i*2] = tr_res[bi][i].det_res.ct.at(0,0); + tracks[i*2+1] = tr_res[bi][i].det_res.ct.at(0,1); } - float dist[count_tr*count_det]; + float dist[count_tr[bi]*count_det]; bool invalid; - for(int i=0; i track_size[i] || dist[j*count_tr+i] > item_size[j] || item_cl[j] != track_cl[i]; - dist[j*count_tr+i] = dist[j*count_tr+i] + invalid * (1 << 18); + invalid = dist[j*count_tr[bi]+i] > track_size[i] || dist[j*count_tr[bi]+i] > item_size[j] || item_cl[j] != track_cl[i]; + dist[j*count_tr[bi]+i] = dist[j*count_tr[bi]+i] + invalid * (1 << 18); } } - int matched_indices[2*count_tr]; + int matched_indices[2*count_tr[bi]]; float min_tr; int min_idtr=-1; - for(int i=0; i new_tr_res; int id_new_tr=0; - for(int i=0; i new_thresh) { count_tr_ ++; @@ -587,10 +591,10 @@ void CenternetDetection3DTrack::tracking(){ new_tr_res_.age = 1; new_tr_res_.active = 1; new_tr_res_.color = rand() % 256; - tr_res.push_back(new_tr_res_); + tr_res[bi].push_back(new_tr_res_); } } - count_tr = count_tr_; + count_tr[bi] = count_tr_; if(track_id==1000) track_id=0; @@ -729,8 +733,8 @@ void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { } // track step - tracking(); - batchTracked.push_back(tr_res); + tracking(bi); + batchTracked.push_back(tr_res[bi]); } void CenternetDetection3DTrack::draw(std::vector& frames) { -- 2.52.0 From 0dc96d2a9e0070f92d12588f9b9da4b931188a11 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Tue, 4 May 2021 17:59:20 +0200 Subject: [PATCH 064/162] Improve CenterTrack. Signed-off-by: Davide Sapienza --- demo/demo/demo3D.cpp | 10 +- include/tkDNN/CenternetDetection3D.h | 13 +- include/tkDNN/CenternetDetection3DTrack.h | 30 +- include/tkDNN/DetectionNN3D.h | 6 +- src/CenternetDetection3D.cpp | 107 ++-- src/CenternetDetection3DTrack.cpp | 627 +++++++++++----------- 6 files changed, 397 insertions(+), 396 deletions(-) diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp index 558e6af..620b0d4 100644 --- a/demo/demo/demo3D.cpp +++ b/demo/demo/demo3D.cpp @@ -96,8 +96,6 @@ int main(int argc, char *argv[]) { int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT); resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h)); } - cv::Size sz_resize = cv::Size(512,512); - std::vector sz_orig; cv::Mat frame; if(show) cv::namedWindow("detection", cv::WINDOW_NORMAL); @@ -108,15 +106,11 @@ int main(int argc, char *argv[]) { while(gRun) { batch_dnn_input.clear(); batch_frame.clear(); - sz_orig.clear(); for(int bi=0; bi< n_batch; ++bi){ cap >> frame; if(!frame.data) break; - sz_orig.push_back(frame.size()); - if(calibs.size() != 0) - resize(frame, frame, sz_resize); batch_frame.push_back(frame); // this will be resized to the net format @@ -126,13 +120,11 @@ int main(int argc, char *argv[]) { break; //inference - detNN->update(batch_dnn_input, n_batch, false, nullptr, false, sz_orig); + detNN->update(batch_dnn_input, n_batch, false, nullptr, false); detNN->draw(batch_frame); if(show){ for(int bi=0; bi< n_batch; ++bi){ - if(calibs.size() != 0) - resize(batch_frame[bi], batch_frame[bi], sz_orig[bi]); cv::imshow("detection", batch_frame[bi]); cv::waitKey(1); } diff --git a/include/tkDNN/CenternetDetection3D.h b/include/tkDNN/CenternetDetection3D.h index cbffa22..943fbf4 100644 --- a/include/tkDNN/CenternetDetection3D.h +++ b/include/tkDNN/CenternetDetection3D.h @@ -27,6 +27,8 @@ private: tk::dnn::dataDim_t dim_dep; tk::dnn::dataDim_t dim_rot; tk::dnn::dataDim_t dim_dim; + + std::vector inputCalibs; float *topk_scores; int *topk_inds_; float *topk_ys_; @@ -58,32 +60,33 @@ private: dnnType *input; #endif cv::Mat r; - cv::Mat calibs; float *d_ptrs; cv::Mat src; cv::Mat dst; cv::Mat dst2; cv::Mat trans, trans2; + std::vector calibs; + //processing int K = 100; int width = 128;//56; // TODO // pointer used in the kernels - float *src_out; - int *ids_out; + float *srcOut; + int *idsOut; struct threshold op; cv::Mat corners, pts3DHomo; - std::vector> face_id; + std::vector> faceId; public: CenternetDetection3D() {}; ~CenternetDetection3D() {}; bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3, const std::vector& k_calibs=std::vector()); - void preprocess(cv::Mat &frame, const int bi=0, const std::vector& stream_size=std::vector()); + void preprocess(cv::Mat &frame, const int bi=0); void postprocess(const int bi=0,const bool mAP=false); void draw(std::vector& frames); }; diff --git a/include/tkDNN/CenternetDetection3DTrack.h b/include/tkDNN/CenternetDetection3DTrack.h index 4c036f4..c500412 100644 --- a/include/tkDNN/CenternetDetection3DTrack.h +++ b/include/tkDNN/CenternetDetection3DTrack.h @@ -76,12 +76,12 @@ public: std::vector inputCalibs; - std::vector sz_old; + std::vector szOld; cv::Mat src; cv::Mat dst; cv::Mat dst2; - cv::Mat trans, trans2, trans_out; + cv::Mat trans, trans2, transOut; /* pre inf */ bool iter0; @@ -131,27 +131,25 @@ public: std::vector calibs; cv::Mat corners, pts3DHomo; - std::vector> face_id; - cv::Scalar tr_colors[256]; + std::vector> faceId; + cv::Scalar trColors[256]; bool view2d = false; //processing struct threshold op; - float out_thresh = 0.1; - float new_thresh = 0.3; - float vis_thresh = 0.3; - float peakThreshold = 0.2; - float centerThreshold = 0.3; //default 0.5 + float outThresh = 0.1; + float newThresh = 0.3; + // float peakThreshold = 0.2; + // float centerThreshold = 0.3; //default 0.5 //detections - std::vector det_res; - int count_det; + std::vector detRes; + int countDet; //tracks - std::vector> tr_res; - std::vector> batchTracked; - std::vector count_tr; - std::vector track_id; + std::vector> trRes; + std::vector countTr; + std::vector trackId; bool init_preprocessing(); @@ -168,7 +166,7 @@ public: CenternetDetection3DTrack() {}; ~CenternetDetection3DTrack() {}; bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3, const std::vector& k_calibs=std::vector()); - void preprocess(cv::Mat &frame, const int bi=0, const std::vector& stream_size=std::vector()); + void preprocess(cv::Mat &frame, const int bi=0); void postprocess(const int bi=0,const bool mAP=false); void draw(std::vector& frames); }; diff --git a/include/tkDNN/DetectionNN3D.h b/include/tkDNN/DetectionNN3D.h index 7320bef..af6eaf9 100644 --- a/include/tkDNN/DetectionNN3D.h +++ b/include/tkDNN/DetectionNN3D.h @@ -54,7 +54,7 @@ class DetectionNN3D { * @param frame original frame to adapt for inference. * @param bi batch index */ - virtual void preprocess(cv::Mat &frame, const int bi=0 , const std::vector& stream_size=std::vector()) = 0; + virtual void preprocess(cv::Mat &frame, const int bi=0) = 0; /** * This method postprocess the output of the NN to obtain the correct @@ -102,7 +102,7 @@ class DetectionNN3D { * box are needed, as in some cases for the mAP calculation. */ void update(std::vector& frames, const int cur_batches=1, bool save_times=false, - std::ofstream *times=nullptr, const bool mAP=false, const std::vector& stream_size=std::vector()){ + std::ofstream *times=nullptr, const bool mAP=false){ if(save_times && times==nullptr) FatalError("save_times set to true, but no valid ofstream given"); if(cur_batches > nBatches) @@ -116,7 +116,7 @@ class DetectionNN3D { if(!frames[bi].data) FatalError("No image data feed to detection"); originalSize.push_back(frames[bi].size()); - preprocess(frames[bi], bi, stream_size); + preprocess(frames[bi], bi); } TKDNN_TSTOP pre_stats.push_back(t_ns); diff --git a/src/CenternetDetection3D.cpp b/src/CenternetDetection3D.cpp index 73e4215..8f7d7c3 100644 --- a/src/CenternetDetection3D.cpp +++ b/src/CenternetDetection3D.cpp @@ -10,7 +10,7 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas classes = n_classes; nBatches = n_batches; confThreshold = conf_thresh; - + inputCalibs = k_calibs; dim = netRT->input_dim; const char *kitti_class_name[] = { @@ -99,19 +99,26 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas stddev << 0.229, 0.224, 0.225; #endif - calibs = cv::Mat(cv::Size(4,3), CV_32F); - calibs.at(0,0) = 707.0493; - calibs.at(0,1) = 0.0; - calibs.at(0,2) = 604.0814; - calibs.at(0,3) = 45.75831; - calibs.at(1,0) = 0.0; - calibs.at(1,1) = 707.0493; - calibs.at(1,2) = 180.5066; - calibs.at(1,3) = -0.3454157; - calibs.at(2,0) = 0.0; - calibs.at(2,1) = 0.0; - calibs.at(2,2) = 1.0; - calibs.at(2,3) = 0.004981016; + for(int bi=0; bi(0,0) = 707.0493; + calibs_.at(0,2) = 604.0814; + calibs_.at(1,1) = 707.0493; + calibs_.at(1,2) = 180.5066; + } + else { + calibs_.at(0,0) = inputCalibs[bi].at(0,0) * dim.w / 1440; + calibs_.at(0,2) = inputCalibs[bi].at(0,2) * dim.w / 1440; + calibs_.at(1,1) = inputCalibs[bi].at(1,1) * dim.h / 1080; + calibs_.at(1,2) = inputCalibs[bi].at(1,2) * dim.h / 1080; + } + calibs_.at(0,3) = 45.75831; + calibs_.at(1,3) = -0.3454157; + calibs_.at(2,2) = 1.0; + calibs_.at(2,3) = 0.004981016; + calibs.push_back(calibs_); + } r = cv::Mat(cv::Size(3,3), CV_32F); r.at(0,1) = 0.0; @@ -139,8 +146,8 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas checkCuda( cudaMalloc(&d_ptrs, dim.c * dim.h*dim.w * sizeof(float)) ); // Alloc array used in the kernel - checkCuda( cudaMalloc(&src_out, K *sizeof(float)) ); - checkCuda( cudaMalloc(&ids_out, K *sizeof(int)) ); + checkCuda( cudaMalloc(&srcOut, K *sizeof(float)) ); + checkCuda( cudaMalloc(&idsOut, K *sizeof(int)) ); dst2.at(0,0)=width * 0.5; dst2.at(0,1)=width * 0.5; @@ -150,16 +157,14 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas dst2.at(2,0)=dst2.at(1,0) + (-dst2.at(0,1)+dst2.at(1,1) ); dst2.at(2,1)=dst2.at(1,1) + (dst2.at(0,0)-dst2.at(1,0) ); - face_id.push_back({0,1,5,4}); - face_id.push_back({1,2,6, 5}); - face_id.push_back({2,3,7,6}); - face_id.push_back({3,0,4,7}); + faceId.push_back({0,1,5,4}); + faceId.push_back({1,2,6, 5}); + faceId.push_back({2,3,7,6}); + faceId.push_back({3,0,4,7}); // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); } -void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi, const std::vector& stream_size){ - // -----------------------------------pre-process ------------------------------------------ - +void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi){ // auto start_t = std::chrono::steady_clock::now(); // auto step_t = std::chrono::steady_clock::now(); // auto end_t = std::chrono::steady_clock::now(); @@ -338,20 +343,20 @@ void CenternetDetection3D::postprocess(const int bi, const bool mAP) { // ----------- topk end - topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3], src_out, ids_out); + topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3], srcOut, idsOut); // checkCuda( cudaDeviceSynchronize() ); - getRecordsFromTopKId(topk_inds_d, K, dim_dep.c, dim_dep.h * dim_dep.w, rt_out[4], dep_d, ids_out); + getRecordsFromTopKId(topk_inds_d, K, dim_dep.c, dim_dep.h * dim_dep.w, rt_out[4], dep_d, idsOut); checkCuda( cudaMemcpy(dep, dep_d, K * dim_dep.c * sizeof(float), cudaMemcpyDeviceToHost) ); - getRecordsFromTopKId(topk_inds_d, K, dim_rot.c, dim_rot.h * dim_rot.w, rt_out[5], rot_d, ids_out); + getRecordsFromTopKId(topk_inds_d, K, dim_rot.c, dim_rot.h * dim_rot.w, rt_out[5], rot_d, idsOut); checkCuda( cudaMemcpy(rot, rot_d, K * dim_rot.c * sizeof(float), cudaMemcpyDeviceToHost) ); - getRecordsFromTopKId(topk_inds_d, K, dim_dim.c, dim_dim.h * dim_dim.w, rt_out[6], dim_d, ids_out); + getRecordsFromTopKId(topk_inds_d, K, dim_dim.c, dim_dim.h * dim_dim.w, rt_out[6], dim_d, idsOut); checkCuda( cudaMemcpy(dim_, dim_d, K * dim_dim.c * sizeof(float), cudaMemcpyDeviceToHost) ); - getRecordsFromTopKId(topk_inds_d, K, dim_wh.c, dim_wh.h * dim_wh.w, rt_out[2], wh_d, ids_out); + getRecordsFromTopKId(topk_inds_d, K, dim_wh.c, dim_wh.h * dim_wh.w, rt_out[2], wh_d, idsOut); checkCuda( cudaMemcpy(wh, wh_d, K * dim_wh.c * sizeof(float), cudaMemcpyDeviceToHost) ); checkCuda( cudaMemcpy(xs, topk_xs_d, K * sizeof(float), cudaMemcpyDeviceToHost) ); @@ -397,11 +402,11 @@ void CenternetDetection3D::postprocess(const int bi, const bool mAP) { alpha = std::atan2(rot[6*K + j], rot[7*K + j]) +0.5 * M_PI; // unproject_2d_to_3d - z = dep[j] - calibs.at(2,3);// z = depth - P[2, 3] - x = (target_coords[j*4] * dep[j] - calibs.at(0,3) - calibs.at(0,2) * z) / calibs.at(0,0); - y = (target_coords[j*4+1] * dep[j] - calibs.at(1,3) - calibs.at(1,2) * z) / calibs.at(1,1) + (dim_[j] / 2); + z = dep[j] - calibs[bi].at(2,3);// z = depth - P[2, 3] + x = (target_coords[j*4] * dep[j] - calibs[bi].at(0,3) - calibs[bi].at(0,2) * z) / calibs[bi].at(0,0); + y = (target_coords[j*4+1] * dep[j] - calibs[bi].at(1,3) - calibs[bi].at(1,2) * z) / calibs[bi].at(1,1) + (dim_[j] / 2); // alpha2rot_y - rot_y = (alpha + std::atan2(target_coords[j*4] - calibs.at(0,2), calibs.at(0,0))); + rot_y = (alpha + std::atan2(target_coords[j*4] - calibs[bi].at(0,2), calibs[bi].at(0,0))); if(rot_y>M_PI) rot_y -= 2*M_PI; if(rot_y(k1,k2) = aus.at(k1,k2); } aus.release(); - aus = calibs * pts3DHomo; + aus = calibs[bi] * pts3DHomo; tk::dnn::box3D res; for(int k=0; k<8; k++) { @@ -486,31 +491,31 @@ void CenternetDetection3D::draw(std::vector& frames) { for(int ind_f = 3; ind_f>=0; ind_f--) { for(int j=0; j<4; j++) { - cv::line(frames[bi], cv::Point(b.corners.at(face_id.at(ind_f).at(j) * 2), - b.corners.at(face_id.at(ind_f).at(j) * 2 + 1)), - cv::Point(b.corners.at(face_id.at(ind_f).at((j+1)%4) * 2), - b.corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), + cv::line(frames[bi], cv::Point(b.corners.at(faceId.at(ind_f).at(j) * 2), + b.corners.at(faceId.at(ind_f).at(j) * 2 + 1)), + cv::Point(b.corners.at(faceId.at(ind_f).at((j+1)%4) * 2), + b.corners.at(faceId.at(ind_f).at((j+1)%4) * 2 + 1)), colors[b.cl], 2); if(ind_f == 0) { - cv::line(frames[bi], cv::Point(b.corners.at(face_id.at(ind_f).at(0) * 2), - b.corners.at(face_id.at(ind_f).at(0) * 2 + 1)), - cv::Point(b.corners.at(face_id.at(ind_f).at(2) * 2), - b.corners.at(face_id.at(ind_f).at(2) * 2 + 1)), colors[b.cl], 2); - cv::line(frames[bi], cv::Point(b.corners.at(face_id.at(ind_f).at(1) * 2), - b.corners.at(face_id.at(ind_f).at(1) * 2 + 1)), - cv::Point(b.corners.at(face_id.at(ind_f).at(3) * 2), - b.corners.at(face_id.at(ind_f).at(3) * 2 + 1)), colors[b.cl], 2); + cv::line(frames[bi], cv::Point(b.corners.at(faceId.at(ind_f).at(0) * 2), + b.corners.at(faceId.at(ind_f).at(0) * 2 + 1)), + cv::Point(b.corners.at(faceId.at(ind_f).at(2) * 2), + b.corners.at(faceId.at(ind_f).at(2) * 2 + 1)), colors[b.cl], 2); + cv::line(frames[bi], cv::Point(b.corners.at(faceId.at(ind_f).at(1) * 2), + b.corners.at(faceId.at(ind_f).at(1) * 2 + 1)), + cv::Point(b.corners.at(faceId.at(ind_f).at(3) * 2), + b.corners.at(faceId.at(ind_f).at(3) * 2 + 1)), colors[b.cl], 2); } } } // draw label cv::Size text_size = getTextSize(classesNames[b.cl], cv::FONT_HERSHEY_SIMPLEX, font_scale, thickness, &baseline); - cv::rectangle(frames[bi], cv::Point(b.corners.at(face_id.at(0).at(0) * 2), - b.corners.at(face_id.at(0).at(0) * 2 + 1)), - cv::Point((b.corners.at(face_id.at(0).at(0) * 2) + text_size.width - 2), - (b.corners.at(face_id.at(0).at(0) * 2 + 1)) - text_size.height - 2), colors[b.cl], -1); - cv::putText(frames[bi], classesNames[b.cl], cv::Point(b.corners.at(face_id.at(0).at(0) * 2), - b.corners.at(face_id.at(0).at(0) * 2 + 1) - (baseline / 2)), + cv::rectangle(frames[bi], cv::Point(b.corners.at(faceId.at(0).at(0) * 2), + b.corners.at(faceId.at(0).at(0) * 2 + 1)), + cv::Point((b.corners.at(faceId.at(0).at(0) * 2) + text_size.width - 2), + (b.corners.at(faceId.at(0).at(0) * 2 + 1)) - text_size.height - 2), colors[b.cl], -1); + cv::putText(frames[bi], classesNames[b.cl], cv::Point(b.corners.at(faceId.at(0).at(0) * 2), + b.corners.at(faceId.at(0).at(0) * 2 + 1) - (baseline / 2)), cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), thickness); } } diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenternetDetection3DTrack.cpp index 3653896..d119c1e 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenternetDetection3DTrack.cpp @@ -6,83 +6,77 @@ namespace tk { namespace dnn { bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh, const std::vector& k_calibs) { - netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); - - dim = netRT->input_dim; - dim.c = 3; - nBatches = n_batches; + netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); + dim = netRT->input_dim; + dim.c = 3; + nBatches = n_batches; confThreshold = conf_thresh; - inputCalibs = k_calibs; - tr_res.resize(nBatches); - count_tr.resize(nBatches, 0); + inputCalibs = k_calibs; init_preprocessing(); init_pre_inf(); init_postprocessing(); init_visualization(n_classes); - } bool CenternetDetection3DTrack::init_preprocessing(){ //image transformation - src = cv::Mat(cv::Size(2,3), CV_32F); - dst = cv::Mat(cv::Size(2,3), CV_32F); - dst2 = cv::Mat(cv::Size(2,3), CV_32F); - trans = cv::Mat(cv::Size(3,2), CV_32F); - trans2 = cv::Mat(cv::Size(3,2), CV_32F); - trans_out = cv::Mat(cv::Size(3,2), CV_32F); + src = cv::Mat(cv::Size(2,3), CV_32F); + dst = cv::Mat(cv::Size(2,3), CV_32F); + dst2 = cv::Mat(cv::Size(2,3), CV_32F); + trans = cv::Mat(cv::Size(3,2), CV_32F); + trans2 = cv::Mat(cv::Size(3,2), CV_32F); + transOut = cv::Mat(cv::Size(3,2), CV_32F); - - dst2.at(0,0)=width * 0.5; - dst2.at(0,1)=width * 0.5; - dst2.at(1,0)=width * 0.5; - dst2.at(1,1)=width * 0.5 + width * -0.5; - - dst2.at(2,0)=dst2.at(1,0) + (-dst2.at(0,1)+dst2.at(1,1) ); - dst2.at(2,1)=dst2.at(1,1) + (dst2.at(0,0)-dst2.at(1,0) ); + dst2.at(0,0) = width * 0.5; + dst2.at(0,1) = width * 0.5; + dst2.at(1,0) = width * 0.5; + dst2.at(1,1) = width * 0.5 + width * -0.5; + dst2.at(2,0) = dst2.at(1,0) + (-dst2.at(0,1)+dst2.at(1,1) ); + dst2.at(2,1) = dst2.at(1,1) + (dst2.at(0,0)-dst2.at(1,0) ); for(int bi=0; biinput_dim.tot() * nBatches)); - checkCuda(cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot())); + checkCuda( cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches)); + checkCuda( cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot())); checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) ); } bool CenternetDetection3DTrack::init_pre_inf(){ // initial steps: the first part of the network const char *pre_img_conv1_bin = "dla34_cnet3d_track/layers/base-pre_img_layer-0.bin"; - const char *pre_hm_conv1_bin = "dla34_cnet3d_track/layers/base-pre_hm_layer-0.bin"; - const char *conv1_bin = "dla34_cnet3d_track/layers/base-base_layer-0.bin"; - const char *conv2_bin = "dla34_cnet3d_track/layers/base-level0-0.bin"; + const char *pre_hm_conv1_bin = "dla34_cnet3d_track/layers/base-pre_hm_layer-0.bin"; + const char *conv1_bin = "dla34_cnet3d_track/layers/base-base_layer-0.bin"; + const char *conv2_bin = "dla34_cnet3d_track/layers/base-level0-0.bin"; dim_in0 = tk::dnn::dataDim_t(1, 3, 512, 512, 1); dim_in1 = tk::dnn::dataDim_t(1, 1, 512, 512, 1); - checkCuda( cudaMalloc(&out_d, netRT->input_dim.tot()*sizeof(dnnType)) ); checkCuda( cudaMalloc(&img_d, dim_in0.tot()*sizeof(dnnType)) ); checkCuda( cudaMalloc(&hm_d, dim_in1.tot()*sizeof(dnnType)) ); // init to zeros hm - dnnType *hm_h; + dnnType *hm_h; checkCuda( cudaMallocHost(&hm_h, 1 * dim.h * dim.w*sizeof(dnnType)) ); for(int i=0; i<1 * dim.h * dim.w; i++) - hm_h[i]=0.0f; + hm_h[i] = 0.0f; checkCuda( cudaMemcpy(hm_d, hm_h, 1 * dim.h * dim.w * sizeof(dnnType), cudaMemcpyHostToDevice) ); checkCuda( cudaFreeHost(hm_h) ); dnnType *i0_h, *i1_h, *i2_h; @@ -97,20 +91,20 @@ bool CenternetDetection3DTrack::init_pre_inf(){ pre_phase_net = new tk::dnn::Network(dim_in0); //pre-img - tk::dnn::Input *in_pre_img = new tk::dnn::Input(pre_phase_net, dim_in0, img_d); - tk::dnn::Conv2d *pre_img_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_img_conv1_bin, true); - tk::dnn::Activation *pre_img_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + tk::dnn::Input *in_pre_img = new tk::dnn::Input(pre_phase_net, dim_in0, img_d); + tk::dnn::Conv2d *pre_img_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_img_conv1_bin, true); + tk::dnn::Activation *pre_img_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); //pre-hm - tk::dnn::Input *in_pre_hm = new tk::dnn::Input(pre_phase_net, dim_in1, hm_d); - tk::dnn::Conv2d *pre_hm_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_hm_conv1_bin, true); - tk::dnn::Activation *pre_hm_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + tk::dnn::Input *in_pre_hm = new tk::dnn::Input(pre_phase_net, dim_in1, hm_d); + tk::dnn::Conv2d *pre_hm_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_hm_conv1_bin, true); + tk::dnn::Activation *pre_hm_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); // image input - tk::dnn::Input *input_image = new tk::dnn::Input(pre_phase_net, dim_in0, input_pre_inf_d); - tk::dnn::Conv2d *conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, conv1_bin, true); - tk::dnn::Activation *relu1 = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); + tk::dnn::Input *input_image = new tk::dnn::Input(pre_phase_net, dim_in0, input_pre_inf_d); + tk::dnn::Conv2d *conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, conv1_bin, true); + tk::dnn::Activation *relu1 = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU); - tk::dnn::Shortcut *s0_input = new tk::dnn::Shortcut(pre_phase_net, pre_img_relu); - tk::dnn::Shortcut *s1_input = new tk::dnn::Shortcut(pre_phase_net, pre_hm_relu); + tk::dnn::Shortcut *s0_input = new tk::dnn::Shortcut(pre_phase_net, pre_img_relu); + tk::dnn::Shortcut *s1_input = new tk::dnn::Shortcut(pre_phase_net, pre_hm_relu); // output data out_d = s1_input->dstData; //print network model @@ -123,14 +117,14 @@ bool CenternetDetection3DTrack::init_pre_inf(){ bool CenternetDetection3DTrack::init_postprocessing(){ srand(0); //seed = 0 for random colors - dim_hm = tk::dnn::dataDim_t(1, 10, 128, 128, 1); - dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1); - dim_reg = tk::dnn::dataDim_t(1, 2, 128, 128, 1); - dim_track = tk::dnn::dataDim_t(1, 2, 128, 128, 1); - dim_dep = tk::dnn::dataDim_t(1, 1, 128, 128, 1); - dim_rot = tk::dnn::dataDim_t(1, 8, 128, 128, 1); - dim_dim = tk::dnn::dataDim_t(1, 3, 128, 128, 1); - dim_amodel_offset = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_hm = tk::dnn::dataDim_t(1, 10, 128, 128, 1); + dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_reg = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_track = tk::dnn::dataDim_t(1, 2, 128, 128, 1); + dim_dep = tk::dnn::dataDim_t(1, 1, 128, 128, 1); + dim_rot = tk::dnn::dataDim_t(1, 8, 128, 128, 1); + dim_dim = tk::dnn::dataDim_t(1, 3, 128, 128, 1); + dim_amodel_offset = tk::dnn::dataDim_t(1, 2, 128, 128, 1); checkCuda( cudaMalloc(&topk_scores, dim_hm.c * K *sizeof(float)) ); checkCuda( cudaMalloc(&topk_inds_, dim_hm.c * K *sizeof(int)) ); @@ -138,7 +132,7 @@ bool CenternetDetection3DTrack::init_postprocessing(){ checkCuda( cudaMalloc(&topk_xs_, dim_hm.c * K *sizeof(float)) ); checkCuda( cudaMalloc(&ids_d, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) ); checkCuda( cudaMallocHost(&ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) ); - for(int i =0; i(coco_class_name, std::end( coco_class_name)); for(int c=0; c(3,6) = 1.0; pts3DHomo.at(3,7) = 1.0; - face_id.push_back({0,1,5,4}); - face_id.push_back({1,2,6, 5}); - face_id.push_back({3,0,4,7}); - face_id.push_back({2,3,7,6}); + faceId.push_back({0,1,5,4}); + faceId.push_back({1,2,6, 5}); + faceId.push_back({3,0,4,7}); + faceId.push_back({2,3,7,6}); // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); } @@ -296,22 +289,21 @@ void CenternetDetection3DTrack::pre_inf(const int bi){ checkCuda( cudaDeviceSynchronize() ); } -void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi, const std::vector& stream_size){ - // -----------------------------------pre-process ------------------------------------------ +void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi){ cv::Size sz = originalSize[bi]; - float scale = 1.0; - float new_height = sz.height * scale; - float new_width = sz.width * scale; - if(sz.height != sz_old[bi].height && sz.width != sz_old[bi].width){ + // float scale = 1.0; + float new_height = dim.h;//sz.height * scale; + float new_width = dim.w;//sz.width * scale; + if(sz.height != szOld[bi].height && sz.width != szOld[bi].width){ if(inputCalibs.size() == 0 || inputCalibs[bi].empty()) { calibs[bi].at(0,2) = new_width / 2.0f; calibs[bi].at(1,2) = new_height /2.0f; } else { - calibs[bi].at(0,0) = inputCalibs[bi].at(0,0) * dim.w / stream_size[bi].width; - calibs[bi].at(0,2) = inputCalibs[bi].at(0,2) * dim.w / stream_size[bi].width; - calibs[bi].at(1,1) = inputCalibs[bi].at(1,1) * dim.h / stream_size[bi].height; - calibs[bi].at(1,2) = inputCalibs[bi].at(1,2) * dim.h / stream_size[bi].height; + calibs[bi].at(0,0) = inputCalibs[bi].at(0,0) * dim.w / sz.width; + calibs[bi].at(0,2) = inputCalibs[bi].at(0,2) * dim.w / sz.width; + calibs[bi].at(1,1) = inputCalibs[bi].at(1,1) * dim.h / sz.height; + calibs[bi].at(1,2) = inputCalibs[bi].at(1,2) * dim.h / sz.height; } float c[] = {new_width / 2.0f, new_height /2.0f}; @@ -320,34 +312,33 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi, const s // ----------- get_affine_transform // rot_rad = pi * 0 / 100 --> 0 //dim.print(); - src.at(0,0)=c[0]; - src.at(0,1)=c[1]; - src.at(1,0)=c[0]; - src.at(1,1)=c[1] + s[0] * -0.5; - dst.at(0,0)=dim.w * 0.5; - dst.at(0,1)=dim.h * 0.5; - dst.at(1,0)=dim.w * 0.5; - dst.at(1,1)=dim.h * 0.5 + dim.w * -0.5; + src.at(0,0) = c[0]; + src.at(0,1) = c[1]; + src.at(1,0) = c[0]; + src.at(1,1) = c[1] + s[0] * -0.5; + dst.at(0,0) = dim.w * 0.5; + dst.at(0,1) = dim.h * 0.5; + dst.at(1,0) = dim.w * 0.5; + dst.at(1,1) = dim.h * 0.5 + dim.w * -0.5; - src.at(2,0)=src.at(1,0) + (-src.at(0,1)+src.at(1,1) ); - src.at(2,1)=src.at(1,1) + (src.at(0,0)-src.at(1,0) ); - dst.at(2,0)=dst.at(1,0) + (-dst.at(0,1)+dst.at(1,1) ); - dst.at(2,1)=dst.at(1,1) + (dst.at(0,0)-dst.at(1,0) ); + src.at(2,0) = src.at(1,0) + (-src.at(0,1)+src.at(1,1) ); + src.at(2,1) = src.at(1,1) + (src.at(0,0)-src.at(1,0) ); + dst.at(2,0) = dst.at(1,0) + (-dst.at(0,1)+dst.at(1,1) ); + dst.at(2,1) = dst.at(1,1) + (dst.at(0,0)-dst.at(1,0) ); trans = cv::getAffineTransform( src, dst ); trans2 = cv::getAffineTransform( dst2, src ); - trans2.convertTo(trans_out, CV_32F); + trans2.convertTo(transOut, CV_32F); } - sz_old[bi] = sz; + szOld[bi] = sz; #ifdef OPENCV_CUDACONTRIB - std::cout<<"OPENCV CPMTROB\n"; cv::cuda::GpuMat im_Orig; cv::cuda::GpuMat imageF1_d, imageF2_d; im_Orig = cv::cuda::GpuMat(frame); - // cv::cuda::resize (im_Orig, imageF1_d, cv::Size(new_width, new_height)); - imageF1_d = im_Orig; + cv::cuda::resize (im_Orig, imageF1_d, cv::Size(dim.w, dim.h)); + // imageF1_d = im_Orig; checkCuda( cudaDeviceSynchronize() ); sz = imageF1_d.size(); @@ -367,20 +358,18 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi, const s normalize(d_ptrs, dim.c, dim.h, dim.w, mean_d, stddev_d); - checkCuda(cudaMemcpy(input_pre_inf_d, d_ptrs, dim2.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice)); + checkCuda( cudaMemcpy(input_pre_inf_d, d_ptrs, dim2.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice)); checkCuda( cudaDeviceSynchronize() ); #else - std::cout<<"NO OPENCV CPMTROB\n"; cv::Mat imageF; - //resize(frame, imageF, cv::Size(512, 512)); - imageF = frame; + resize(frame, imageF, cv::Size(dim.w, dim.h)); + // imageF = frame; sz = imageF.size(); - cv::warpAffine(imageF, imageF, trans, cv::Size(dim.w, dim.h), cv::INTER_LINEAR ); - //cv::imshow("warp", imageF); - + // cv::imshow("warp", imageF); + sz = imageF.size(); imageF.convertTo(imageF, CV_32FC3, 1/255.0); @@ -394,11 +383,11 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi, const s bgr[i] = bgr[i] / stddev[i]; } for(int i=0; i(0,0) = x1; target_coords.at(0,1) = x2; target_coords.at(0,2) = 1.0; - return trans_out * target_coords; + return transOut * target_coords; } void CenternetDetection3DTrack::tracking(const int bi) { - float item_size[count_det]; - int item_cl[count_det]; - float dets[2*count_det]; - for(int i=0; i(0,0) - det_res[i].bb0.at(0,0)) * - (det_res[i].bb1.at(0,1) - det_res[i].bb0.at(0,1)); - item_cl[i] = det_res[i].cl; - dets[i*2] = det_res[i].ct.at(0,0); - dets[i*2+1] = det_res[i].ct.at(0,1); + float item_size[countDet]; + int item_cl[countDet]; + float dets[2*countDet]; + for(int i=0; i(0,0) - detRes[i].bb0.at(0,0)) * + (detRes[i].bb1.at(0,1) - detRes[i].bb0.at(0,1)); + item_cl[i] = detRes[i].cl; + dets[i*2] = detRes[i].ct.at(0,0); + dets[i*2+1] = detRes[i].ct.at(0,1); } - float track_size[count_tr[bi]]; - int track_cl[count_tr[bi]]; - float tracks[2*count_tr[bi]]; - for(int i=0; i(0,0) - tr_res[bi][i].det_res.bb0.at(0,0)) * - (tr_res[bi][i].det_res.bb1.at(0,1) - tr_res[bi][i].det_res.bb0.at(0,1)); - track_cl[i] = tr_res[bi][i].det_res.cl; - tracks[i*2] = tr_res[bi][i].det_res.ct.at(0,0); - tracks[i*2+1] = tr_res[bi][i].det_res.ct.at(0,1); + float track_size[countTr[bi]]; + int track_cl[countTr[bi]]; + float tracks[2*countTr[bi]]; + for(int i=0; i(0,0) - trRes[bi][i].det_res.bb0.at(0,0)) * + (trRes[bi][i].det_res.bb1.at(0,1) - trRes[bi][i].det_res.bb0.at(0,1)); + track_cl[i] = trRes[bi][i].det_res.cl; + tracks[i*2] = trRes[bi][i].det_res.ct.at(0,0); + tracks[i*2+1] = trRes[bi][i].det_res.ct.at(0,1); } - float dist[count_tr[bi]*count_det]; + float dist[countTr[bi]*countDet]; bool invalid; - for(int i=0; i track_size[i] || dist[j*count_tr[bi]+i] > item_size[j] || item_cl[j] != track_cl[i]; - dist[j*count_tr[bi]+i] = dist[j*count_tr[bi]+i] + invalid * (1 << 18); + for(int i=0; i track_size[i] || + dist[j*countTr[bi]+i] > item_size[j] || + item_cl[j] != track_cl[i]; + dist[j*countTr[bi]+i] = dist[j*countTr[bi]+i] + invalid * (1 << 18); } } - int matched_indices[2*count_tr[bi]]; + int matched_indices[2*countTr[bi]]; float min_tr; - int min_idtr=-1; - for(int i=0; i new_tr_res; int id_new_tr=0; - for(int i=0; i new_thresh) { + int count_tr_ = countTr[bi]; + for(int i=0; i newThresh) { count_tr_ ++; struct trackingRes new_tr_res_; - new_tr_res_.det_res.score = det_res[i].score; - new_tr_res_.det_res.cl = det_res[i].cl; - new_tr_res_.det_res.ct = det_res[i].ct; - new_tr_res_.det_res.tr = det_res[i].tr; - new_tr_res_.det_res.bb0 = det_res[i].bb0; - new_tr_res_.det_res.bb1 = det_res[i].bb1; - new_tr_res_.det_res.dep = det_res[i].dep; - new_tr_res_.det_res.dim[0] = det_res[i].dim[0]; - new_tr_res_.det_res.dim[1] = det_res[i].dim[1]; - new_tr_res_.det_res.dim[2] = det_res[i].dim[2]; - new_tr_res_.det_res.alpha = det_res[i].alpha; - new_tr_res_.det_res.x = det_res[i].x; - new_tr_res_.det_res.y = det_res[i].y; - new_tr_res_.det_res.z = det_res[i].z; - new_tr_res_.det_res.rot_y = det_res[i].rot_y; - new_tr_res_.tracking_id = track_id[bi]++; - new_tr_res_.age = 1; - new_tr_res_.active = 1; - new_tr_res_.color = rand() % 256; - if(tr_res.size() <= bi) { + new_tr_res_.det_res.score = detRes[i].score; + new_tr_res_.det_res.cl = detRes[i].cl; + new_tr_res_.det_res.ct = detRes[i].ct; + new_tr_res_.det_res.tr = detRes[i].tr; + new_tr_res_.det_res.bb0 = detRes[i].bb0; + new_tr_res_.det_res.bb1 = detRes[i].bb1; + new_tr_res_.det_res.dep = detRes[i].dep; + new_tr_res_.det_res.dim[0] = detRes[i].dim[0]; + new_tr_res_.det_res.dim[1] = detRes[i].dim[1]; + new_tr_res_.det_res.dim[2] = detRes[i].dim[2]; + new_tr_res_.det_res.alpha = detRes[i].alpha; + new_tr_res_.det_res.x = detRes[i].x; + new_tr_res_.det_res.y = detRes[i].y; + new_tr_res_.det_res.z = detRes[i].z; + new_tr_res_.det_res.rot_y = detRes[i].rot_y; + new_tr_res_.tracking_id = trackId[bi]++; + new_tr_res_.age = 1; + new_tr_res_.active = 1; + new_tr_res_.color = rand() % 256; + if(trRes.size() <= bi) { std::vector v_new_tr_res_; v_new_tr_res_.push_back(new_tr_res_); - tr_res.push_back(v_new_tr_res_); + trRes.push_back(v_new_tr_res_); } else - tr_res[bi].push_back(new_tr_res_); + trRes[bi].push_back(new_tr_res_); } } - count_tr[bi] = count_tr_; - - if(track_id[bi]==1000) - track_id[bi]=0; - det_res.clear(); + countTr[bi] = count_tr_; + //reset the tracker id + if(trackId[bi] == 1000) + trackId[bi] = 0; + detRes.clear(); } @@ -697,35 +686,36 @@ void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { // ---------------------------------- post-process ----------------------------------------- - count_det = 0; - det_res.clear(); - for(int i = 0; i(2,3); - new_det_res.x = ((float)new_det_res.ct.at(0,0) * dep[i] - calibs[bi].at(0,3) - calibs[bi].at(0,2) * new_det_res.z) / calibs[bi].at(0,0); - new_det_res.y = ((float)new_det_res.ct.at(0,1) * dep[i] - calibs[bi].at(1,3) - calibs[bi].at(1,2) * new_det_res.z) / calibs[bi].at(1,1) + (dim_[i] / 2); + new_det_res.x = ((float)new_det_res.ct.at(0,0) * dep[i] - calibs[bi].at(0,3) - + calibs[bi].at(0,2) * new_det_res.z) / calibs[bi].at(0,0); + new_det_res.y = ((float)new_det_res.ct.at(0,1) * dep[i] - calibs[bi].at(1,3) - + calibs[bi].at(1,2) * new_det_res.z) / calibs[bi].at(1,1) + (dim_[i] / 2); // alpha2rot_y // idx = rot[:, 1] > rot[:, 5] @@ -737,13 +727,11 @@ void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { else new_det_res.alpha = std::atan2(rot[6*K + i], rot[7*K + i]) +0.5 * M_PI; new_det_res.rot_y = (new_det_res.alpha + std::atan2((float)new_det_res.ct.at(0,0) - calibs[bi].at(0,2), calibs[bi].at(0,0))); - new_det_res.ct = new_det_res.ct + new_det_res.tr; //dest - det_res.push_back(new_det_res); - + new_det_res.ct = new_det_res.ct + new_det_res.tr; //dest + detRes.push_back(new_det_res); } // track step tracking(bi); - batchTracked.push_back(tr_res[bi]); } void CenternetDetection3DTrack::draw(std::vector& frames) { @@ -754,32 +742,38 @@ void CenternetDetection3DTrack::draw(std::vector& frames) { int baseline = 0; float font_scale = 0.8; int thickness = 2; + for(int bi=0; bi vis_thresh){// && t.active!=0) { + if(t.det_res.score > confThreshold){// && t.active!=0) { if(view2d) { - cv::rectangle(frames[bi], cv::Point(t.det_res.bb0.at(0,0), t.det_res.bb0.at(0,1)), - cv::Point(t.det_res.bb1.at(0,0), t.det_res.bb1.at(0,1)), tr_colors[t.color], thickness); - cv::rectangle(frames[bi], cv::Point(t.det_res.bb0.at(0,0), - t.det_res.bb0.at(0,1) - text_size.height - thickness), - cv::Point(t.det_res.bb0.at(0,0) + text_size.width, - t.det_res.bb0.at(0,1)), tr_colors[t.color], -1); + cv::rectangle(frames[bi], + cv::Point(t.det_res.bb0.at(0,0) * scale_x, t.det_res.bb0.at(0,1) * scale_y), + cv::Point(t.det_res.bb1.at(0,0) * scale_x, t.det_res.bb1.at(0,1) * scale_y), + trColors[t.color], thickness); + cv::rectangle(frames[bi], + cv::Point(t.det_res.bb0.at(0,0) * scale_x, t.det_res.bb0.at(0,1) * scale_y - text_size.height - thickness), + cv::Point(t.det_res.bb0.at(0,0) * scale_x + text_size.width, t.det_res.bb0.at(0,1) * scale_y), + trColors[t.color], -1); - cv::putText(frames[bi], txt, cv::Point(t.det_res.bb0.at(0,0), - t.det_res.bb0.at(0,1) - thickness -1), - cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1); + cv::putText(frames[bi], txt, + cv::Point(t.det_res.bb0.at(0,0) * scale_x, t.det_res.bb0.at(0,1) * scale_y - thickness -1), + cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1); - cv::arrowedLine(frames[bi], cv::Point((int)t.det_res.ct.at(0,0), - (int)t.det_res.ct.at(0,1)), - cv::Point((int)(t.det_res.ct.at(0,0) + t.det_res.tr.at(0,0)), - (int)(t.det_res.ct.at(0,1) + t.det_res.tr.at(0,1))), - cv::Scalar(255, 0, 255), 2); + cv::arrowedLine(frames[bi], + cv::Point((int)t.det_res.ct.at(0,0) * scale_x, (int)t.det_res.ct.at(0,1) * scale_y), + cv::Point((int)(t.det_res.ct.at(0,0) * scale_x + t.det_res.tr.at(0,0) * scale_x), + (int)(t.det_res.ct.at(0,1) * scale_y + t.det_res.tr.at(0,1) * scale_y)), + cv::Scalar(255, 0, 255), 2); } //3d if(!view2d && t.det_res.z > 1){ @@ -833,50 +827,59 @@ void CenternetDetection3DTrack::draw(std::vector& frames) { res_corners.push_back(aus.at(1,k) / aus.at(2,k)); } aus.release(); - for(int ind_f = 3; ind_f>=0; ind_f--) { + for(int ind_f=3; ind_f>=0; ind_f--) { for(int j=0; j<4; j++) { - cv::line(frames[bi], cv::Point((int)res_corners.at(face_id.at(ind_f).at(j) * 2), - (int)res_corners.at(face_id.at(ind_f).at(j) * 2 + 1)), - cv::Point((int)res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2), - (int)res_corners.at(face_id.at(ind_f).at((j+1)%4) * 2 + 1)), - tr_colors[t.color], 2); + cv::line(frames[bi], + cv::Point((int)res_corners.at(faceId.at(ind_f).at(j) * 2) * scale_x, + (int)res_corners.at(faceId.at(ind_f).at(j) * 2 + 1) * scale_y), + cv::Point((int)res_corners.at(faceId.at(ind_f).at((j+1)%4) * 2) * scale_x, + (int)res_corners.at(faceId.at(ind_f).at((j+1)%4) * 2 + 1) * scale_y), + trColors[t.color], 2); if(ind_f == 0 && j==3) { - cv::line(frames[bi], cv::Point((int)res_corners.at(face_id.at(ind_f).at(0) * 2), - (int)res_corners.at(face_id.at(ind_f).at(0) * 2 + 1)), - cv::Point((int)res_corners.at(face_id.at(ind_f).at(2) * 2), - (int)res_corners.at(face_id.at(ind_f).at(2) * 2 + 1)), tr_colors[t.color], 2); - cv::line(frames[bi], cv::Point((int)res_corners.at(face_id.at(ind_f).at(1) * 2), - (int)res_corners.at(face_id.at(ind_f).at(1) * 2 + 1)), - cv::Point((int)res_corners.at(face_id.at(ind_f).at(3) * 2), - (int)res_corners.at(face_id.at(ind_f).at(3) * 2 + 1)), tr_colors[t.color], 2); + cv::line(frames[bi], + cv::Point((int)res_corners.at(faceId.at(ind_f).at(0) * 2) * scale_x, + (int)res_corners.at(faceId.at(ind_f).at(0) * 2 + 1) * scale_y), + cv::Point((int)res_corners.at(faceId.at(ind_f).at(2) * 2) * scale_x, + (int)res_corners.at(faceId.at(ind_f).at(2) * 2 + 1) * scale_y), trColors[t.color], 2); + cv::line(frames[bi], + cv::Point((int)res_corners.at(faceId.at(ind_f).at(1) * 2) * scale_x, + (int)res_corners.at(faceId.at(ind_f).at(1) * 2 + 1) * scale_y), + cv::Point((int)res_corners.at(faceId.at(ind_f).at(3) * 2) * scale_x, + (int)res_corners.at(faceId.at(ind_f).at(3) * 2 + 1) * scale_y), trColors[t.color], 2); } } } float bb0=(1 << 10), bb1=0, bb2=(1 << 10), bb3=0; for(int k=0; k<8; k++) { - if(res_corners[2*k]bb1) - bb1=res_corners[2*k]; - if(res_corners[2*k+1]bb3) - bb3=res_corners[2*k+1]; + if(res_corners[2*k] < bb0) + bb0 = res_corners[2*k]; + if(res_corners[2*k] > bb1) + bb1 = res_corners[2*k]; + if(res_corners[2*k+1] < bb2) + bb2 = res_corners[2*k+1]; + if(res_corners[2*k+1] > bb3) + bb3 = res_corners[2*k+1]; } // if(not no_bbox): - // cv::rectangle(frame, cv::Point(bb0, bb2), cv::Point(bb1, bb3), - // tr_colors[t.color], thickness); - cv::rectangle(frames[bi], cv::Point(bb0, bb2 - text_size.height - thickness), - cv::Point(bb0 + text_size.width, bb2), tr_colors[t.color], -1); + // cv::rectangle(frame, + // cv::Point(bb0, bb2), + // cv::Point(bb1, bb3), + // trColors[t.color], thickness); + cv::rectangle(frames[bi], + cv::Point(bb0 * scale_x, bb2 * scale_y - text_size.height - thickness), + cv::Point(bb0 * scale_x + text_size.width, bb2 * scale_y), + trColors[t.color], -1); - cv::putText(frames[bi], txt, cv::Point(bb0, bb2 - thickness -1), cv::FONT_HERSHEY_SIMPLEX, - font_scale, cv::Scalar(255, 255, 255), 1); + cv::putText(frames[bi], txt, + cv::Point(bb0 * scale_x, bb2 * scale_y - thickness -1), + cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1); - cv::arrowedLine(frames[bi], cv::Point((int)((bb0 + bb1)/2), (int)((bb2 + bb3)/2)), - cv::Point((int)((bb0 + bb1)/2 + t.det_res.tr.at(0,0)), - (int)((bb2 + bb3)/2 + t.det_res.tr.at(0,1))), - cv::Scalar(255, 0, 255), 2); + cv::arrowedLine(frames[bi], + cv::Point((int)((bb0 + bb1)/2) * scale_x, (int)((bb2 + bb3)/2) * scale_y), + cv::Point((int)((bb0 + bb1)/2 + t.det_res.tr.at(0,0)) * scale_x, + (int)((bb2 + bb3)/2 + t.det_res.tr.at(0,1)) * scale_y), + cv::Scalar(255, 0, 255), 2); } } } -- 2.52.0 From 34c1c3d577cb55235f0c73a5eee201023fc0530b Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Tue, 11 May 2021 16:17:23 +0200 Subject: [PATCH 065/162] Update cnet branch. This commit splits the demo3D in two demo: one for the 3D object detection and one for the tracking. It renames the files related to CenterTrack. It adds a new parameter to select the tracker mode (2D or 3D). Signed-off-by: Davide Sapienza --- CMakeLists.txt | 13 +- demo/demo/demo3D.cpp | 5 - demo/demo/demoTracker.cpp | 157 +++++++++++++ ...ternetDetection3DTrack.h => CenterTrack.h} | 20 +- include/tkDNN/TrackingNN.h | 158 +++++++++++++ ...etDetection3DTrack.cpp => CenterTrack.cpp} | 47 ++-- .../dla34_ctrack/dla34_ctrack.cpp} | 222 +++++++++--------- 7 files changed, 469 insertions(+), 153 deletions(-) create mode 100644 demo/demo/demoTracker.cpp rename include/tkDNN/{CenternetDetection3DTrack.h => CenterTrack.h} (90%) create mode 100644 include/tkDNN/TrackingNN.h rename src/{CenternetDetection3DTrack.cpp => CenterTrack.cpp} (96%) rename tests/{centernet/dla34_cnet3d_track/dla34_cnet3d_track.cpp => centertrack/dla34_ctrack/dla34_ctrack.cpp} (73%) diff --git a/CMakeLists.txt b/CMakeLists.txt index cb2b1c6..197dced 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -46,9 +46,9 @@ include_directories(${EIGEN3_INCLUDE_DIR}) find_package(OpenCV REQUIRED) set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV") -if(OpenCV_CUDA_VERSION) - add_compile_definitions(OPENCV_CUDACONTRIB) -endif() +# if(OpenCV_CUDA_VERSION) +# add_compile_definitions(OPENCV_CUDACONTRIB) +# endif() # gives problems in cross-compiling, probably malformed cmake config find_package(yaml-cpp REQUIRED) @@ -120,8 +120,8 @@ target_link_libraries(test_resnet101_cnet3d tkDNN) add_executable(test_dla34_cnet3d tests/centernet/dla34_cnet3d/dla34_cnet3d.cpp) target_link_libraries(test_dla34_cnet3d tkDNN) -add_executable(test_dla34_cnet3d_track tests/centernet/dla34_cnet3d_track/dla34_cnet3d_track.cpp) -target_link_libraries(test_dla34_cnet3d_track tkDNN) +add_executable(test_dla34_ctrack tests/centertrack/dla34_ctrack/dla34_ctrack.cpp) +target_link_libraries(test_dla34_ctrack tkDNN) # DEMOS add_executable(test_rtinference tests/test_rtinference/rtinference.cpp) @@ -136,6 +136,9 @@ target_link_libraries(demo tkDNN) add_executable(demo3D demo/demo/demo3D.cpp) target_link_libraries(demo3D tkDNN) +add_executable(demoTracker demo/demo/demoTracker.cpp) +target_link_libraries(demoTracker tkDNN) + #------------------------------------------------------------------------------- # Install #------------------------------------------------------------------------------- diff --git a/demo/demo/demo3D.cpp b/demo/demo/demo3D.cpp index 620b0d4..90d4bfb 100644 --- a/demo/demo/demo3D.cpp +++ b/demo/demo/demo3D.cpp @@ -5,7 +5,6 @@ #include #include "CenternetDetection3D.h" -#include "CenternetDetection3DTrack.h" bool gRun; bool SAVE_RESULT = false; @@ -55,7 +54,6 @@ int main(int argc, char *argv[]) { SAVE_RESULT = true; tk::dnn::CenternetDetection3D cnet; - tk::dnn::CenternetDetection3DTrack ctrack; tk::dnn::DetectionNN3D *detNN; @@ -64,9 +62,6 @@ int main(int argc, char *argv[]) { case 'c': detNN = &cnet; break; - case 't': - detNN = &ctrack; - break; default: FatalError("Network type not allowed (3rd parameter)\n"); } diff --git a/demo/demo/demoTracker.cpp b/demo/demo/demoTracker.cpp new file mode 100644 index 0000000..ad6e204 --- /dev/null +++ b/demo/demo/demoTracker.cpp @@ -0,0 +1,157 @@ +#include +#include +#include /* srand, rand */ +//#include +#include + +#include "CenterTrack.h" + +bool gRun; +bool SAVE_RESULT = false; + +void sig_handler(int signo) { + std::cout<<"request gateway stop\n"; + gRun = false; +} + +int main(int argc, char *argv[]) { + + std::cout<<"detection\n"; + signal(SIGINT, sig_handler); + + + std::string net = "dla34_cnet3d_track_fp32.rt"; + if(argc > 1) + net = argv[1]; + #ifdef __linux__ + std::string input = "../demo/yolo_test.mp4"; + #elif _WIN32 + std::string input = "..\\..\\..\\demo\\yolo_test.mp4"; + #endif + + if(argc > 2) + input = argv[2]; + char ntype = 'c'; + if(argc > 3) + ntype = argv[3][0]; + int n_classes = 3; + if(argc > 4) + n_classes = atoi(argv[4]); + int n_batch = 1; + if(argc > 5) + n_batch = atoi(argv[5]); + bool show = true; + if(argc > 6) + show = atoi(argv[6]); + float conf_thresh=0.3; + if(argc > 7) + conf_thresh = atof(argv[7]); + bool t3d = true; + if(argc > 8) + t3d = atoi(argv[8]); + if(n_batch < 1 || n_batch > 64) + FatalError("Batch dim not supported"); + + if(!show) + SAVE_RESULT = true; + + tk::dnn::CenterTrack ctrack; + + tk::dnn::TrackingNN *trackNN; + + switch(ntype) + { + case 'c': + trackNN = &ctrack; + break; + default: + FatalError("Network type not allowed (3rd parameter)\n"); + } + std::vector calibs; + // cv::Mat calib = cv::Mat::zeros(cv::Size(3,3), CV_32F); + // calib.at(0,0) = 864.1243196486207;// * 512.0;//884.081444212;//864.1243196486207 * 512.0;// 633.0; + // calib.at(0,2) = 726.7271690557819;// * 512.0;//0.0;//726.7271690557819 * 512.0;// 0.0; //w/2 + // calib.at(1,1) = 883.6552349216504;// * 512.0;//884.081444212;//883.6552349216504 * 512.0;// 633.0; + // calib.at(1,2) = 506.8548506986564;// * 512.0;//0.0;//506.8548506986564 * 512.0;// 0.0; //h/2 + // calibs.push_back(calib); + // calibs.push_back(calib); + // calibs.push_back(calib); + // calibs.push_back(calib); + trackNN->init(net, n_classes, n_batch, conf_thresh, t3d, calibs); + + gRun = true; + + cv::VideoCapture cap(input); + if(!cap.isOpened()) + gRun = false; + else + std::cout<<"camera started\n"; + + cv::VideoWriter resultVideo; + if(SAVE_RESULT) { + int w = cap.get(cv::CAP_PROP_FRAME_WIDTH); + int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT); + resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h)); + } + cv::Mat frame; + if(show) + cv::namedWindow("detection", cv::WINDOW_NORMAL); + + std::vector batch_frame; + std::vector batch_dnn_input; + + while(gRun) { + batch_dnn_input.clear(); + batch_frame.clear(); + + for(int bi=0; bi< n_batch; ++bi){ + cap >> frame; + if(!frame.data) + break; + batch_frame.push_back(frame); + + // this will be resized to the net format + batch_dnn_input.push_back(frame.clone()); + } + if(!frame.data) + break; + + //inference + trackNN->update(batch_dnn_input, n_batch, false, nullptr, false); + trackNN->draw(batch_frame); + + if(show){ + for(int bi=0; bi< n_batch; ++bi){ + cv::imshow("detection", batch_frame[bi]); + cv::waitKey(1); + } + } + if(n_batch == 1 && SAVE_RESULT) + resultVideo << frame; + } + + std::cout<<"detection end\n"; + double mean = 0; + + std::cout<pre_stats.begin(), trackNN->pre_stats.end())<<" ms\n"; + std::cout<<"Max: "<<*std::max_element(trackNN->pre_stats.begin(), trackNN->pre_stats.end())<<" ms\n"; + for(int i=0; ipre_stats.size(); i++) mean += trackNN->pre_stats[i]; mean /= trackNN->pre_stats.size(); + std::cout<<"Avg: "<stats.begin(), trackNN->stats.end())<<" ms\n"; + std::cout<<"Max: "<<*std::max_element(trackNN->stats.begin(), trackNN->stats.end())<<" ms\n"; + for(int i=0; istats.size(); i++) mean += trackNN->stats[i]; mean /= trackNN->stats.size(); + std::cout<<"Avg: "<post_stats.begin(), trackNN->post_stats.end())<<" ms\n"; + std::cout<<"Max: "<<*std::max_element(trackNN->post_stats.begin(), trackNN->post_stats.end())<<" ms\n"; + for(int i=0; ipost_stats.size(); i++) mean += trackNN->post_stats[i]; mean /= trackNN->post_stats.size(); + std::cout<<"Avg: "< #include "opencv2/opencv.hpp" @@ -11,7 +11,7 @@ #include // std::iota #include // std::sort -#include "DetectionNN3D.h" +#include "TrackingNN.h" #include "kernelsThrust.h" @@ -49,7 +49,7 @@ struct trackingRes int color; }; -class CenternetDetection3DTrack : public DetectionNN3D +class CenterTrack : public TrackingNN { public: tk::dnn::dataDim_t dim; @@ -133,7 +133,7 @@ public: std::vector> faceId; cv::Scalar trColors[256]; - bool view2d = false; + bool mode3D; //processing struct threshold op; @@ -163,9 +163,11 @@ public: public: tk::dnn::Network *pre_phase_net = nullptr; - CenternetDetection3DTrack() {}; - ~CenternetDetection3DTrack() {}; - bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3, const std::vector& k_calibs=std::vector()); + CenterTrack() {}; + ~CenterTrack() {}; + bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, + const float conf_thresh=0.3, const bool mode_3d=true, + const std::vector& k_calibs=std::vector()); void preprocess(cv::Mat &frame, const int bi=0); void postprocess(const int bi=0,const bool mAP=false); void draw(std::vector& frames); @@ -176,4 +178,4 @@ public: } // namespace tk -#endif /*CENTERNETDETECTION3DTRACK_H*/ \ No newline at end of file +#endif /*CENTERTRACK_H*/ \ No newline at end of file diff --git a/include/tkDNN/TrackingNN.h b/include/tkDNN/TrackingNN.h new file mode 100644 index 0000000..476db53 --- /dev/null +++ b/include/tkDNN/TrackingNN.h @@ -0,0 +1,158 @@ +#ifndef TRACKINGNN_H +#define TRACKINGNN_H + +#include +#include +#include +#ifdef __linux__ +#include +#endif + +#include +#include "utils.h" + +#include +#include +#include + +#include "tkdnn.h" + +// #define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib. + +#ifdef OPENCV_CUDACONTRIB +#include +#include +#endif + + +namespace tk { namespace dnn { + +class TrackingNN { + + protected: + tk::dnn::NetworkRT *netRT = nullptr; + dnnType *input_d; + + std::vector originalSize; + + cv::Scalar colors[256]; + + int nBatches = 1; + +#ifdef OPENCV_CUDACONTRIB + cv::cuda::GpuMat bgr[3]; + cv::cuda::GpuMat imagePreproc; +#else + cv::Mat bgr[3]; + cv::Mat imagePreproc; + dnnType *input; +#endif + + /** + * This method preprocess the image, before feeding it to the NN. + * + * @param frame original frame to adapt for inference. + * @param bi batch index + */ + virtual void preprocess(cv::Mat &frame, const int bi=0) = 0; + + /** + * This method postprocess the output of the NN to obtain the correct + * boundig boxes. + * + * @param bi batch index + * @param mAP set to true only if all the probabilities for a bounding + * box are needed, as in some cases for the mAP calculation + */ + virtual void postprocess(const int bi=0,const bool mAP=false) = 0; + + public: + int classes = 0; + float confThreshold = 0.3; /*threshold on the confidence of the boxes*/ + + std::vector pre_stats, stats, post_stats, visual_stats; /*keeps track of inference times (ms)*/ + std::vector classesNames; + + TrackingNN() {}; + ~TrackingNN(){}; + + /** + * Method used to initialize the class, allocate memory and compute + * needed data. + * + * @param tensor_path path to the rt file of the NN. + * @param n_classes number of classes for the given dataset. + * @param n_batches maximum number of batches to use in inference. + * @return true if everything is correct, false otherwise. + */ + virtual bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, + const float conf_thresh=0.3, const bool mode_3d=true, const std::vector& k_calibs=std::vector()) = 0; + + /** + * This method performs the whole detection and tracking of the NN. + * + * @param frames frames to run detection and trcking on. + * @param cur_batches number of batches to use in inference. + * @param save_times if set to true, preprocess, inference and postprocess times + * are saved on a csv file, otherwise not. + * @param times pointer to the output stream where to write times. + * @param mAP set to true only if all the probabilities for a bounding + * box are needed, as in some cases for the mAP calculation. + */ + void update(std::vector& frames, const int cur_batches=1, bool save_times=false, + std::ofstream *times=nullptr, const bool mAP=false){ + if(save_times && times==nullptr) + FatalError("save_times set to true, but no valid ofstream given"); + if(cur_batches > nBatches) + FatalError("A batch size greater than nBatches cannot be used"); + + originalSize.clear(); + if(TKDNN_VERBOSE) printCenteredTitle(" TENSORRT detection ", '=', 30); + { + TKDNN_TSTART + for(int bi=0; biinput_dim; + dim.n = cur_batches; + { + if(TKDNN_VERBOSE) dim.print(); + TKDNN_TSTART + netRT->infer(dim, input_d); + TKDNN_TSTOP + if(TKDNN_VERBOSE) dim.print(); + stats.push_back(t_ns); + if(save_times) *times<& frames){}; + +}; + +}} + +#endif /* TRACKINGNN_H*/ diff --git a/src/CenternetDetection3DTrack.cpp b/src/CenterTrack.cpp similarity index 96% rename from src/CenternetDetection3DTrack.cpp rename to src/CenterTrack.cpp index d119c1e..dc15823 100644 --- a/src/CenternetDetection3DTrack.cpp +++ b/src/CenterTrack.cpp @@ -1,16 +1,17 @@ -#include "CenternetDetection3DTrack.h" +#include "CenterTrack.h" namespace tk { namespace dnn { -bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n_classes, const int n_batches, - const float conf_thresh, const std::vector& k_calibs) { +bool CenterTrack::init(const std::string& tensor_path, const int n_classes, const int n_batches, + const float conf_thresh, const bool mode_3d, const std::vector& k_calibs) { netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); dim = netRT->input_dim; dim.c = 3; nBatches = n_batches; confThreshold = conf_thresh; + mode3D = mode_3d; inputCalibs = k_calibs; init_preprocessing(); init_pre_inf(); @@ -18,7 +19,7 @@ bool CenternetDetection3DTrack::init(const std::string& tensor_path, const int n init_visualization(n_classes); } -bool CenternetDetection3DTrack::init_preprocessing(){ +bool CenterTrack::init_preprocessing(){ //image transformation src = cv::Mat(cv::Size(2,3), CV_32F); dst = cv::Mat(cv::Size(2,3), CV_32F); @@ -60,12 +61,12 @@ bool CenternetDetection3DTrack::init_preprocessing(){ checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) ); } -bool CenternetDetection3DTrack::init_pre_inf(){ +bool CenterTrack::init_pre_inf(){ // initial steps: the first part of the network - const char *pre_img_conv1_bin = "dla34_cnet3d_track/layers/base-pre_img_layer-0.bin"; - const char *pre_hm_conv1_bin = "dla34_cnet3d_track/layers/base-pre_hm_layer-0.bin"; - const char *conv1_bin = "dla34_cnet3d_track/layers/base-base_layer-0.bin"; - const char *conv2_bin = "dla34_cnet3d_track/layers/base-level0-0.bin"; + const char *pre_img_conv1_bin = "dla34_ctrack/layers/base-pre_img_layer-0.bin"; + const char *pre_hm_conv1_bin = "dla34_ctrack/layers/base-pre_hm_layer-0.bin"; + const char *conv1_bin = "dla34_ctrack/layers/base-base_layer-0.bin"; + const char *conv2_bin = "dla34_ctrack/layers/base-level0-0.bin"; dim_in0 = tk::dnn::dataDim_t(1, 3, 512, 512, 1); dim_in1 = tk::dnn::dataDim_t(1, 1, 512, 512, 1); @@ -82,9 +83,9 @@ bool CenternetDetection3DTrack::init_pre_inf(){ dnnType *i0_h, *i1_h, *i2_h; // dnnType *i0_d, *i1_d, *i2_d; - // const char *input_bin = "dla34_cnet3d_track/debug/input.bin"; - // const char *pre_img_bin = "dla34_cnet3d_track/debug/pre_imgages.bin"; - // const char *pre_hm_bin = "dla34_cnet3d_track/debug/pre_hms.bin"; + // const char *input_bin = "dla34_ctrack/debug/input.bin"; + // const char *pre_img_bin = "dla34_ctrack/debug/pre_imgages.bin"; + // const char *pre_hm_bin = "dla34_ctrack/debug/pre_hms.bin"; // readBinaryFile(pre_img_bin, dim_in0.tot(), &i0_h, &img_d); // readBinaryFile(pre_hm_bin, dim_in1.tot(), &i1_h, &hm_d); // readBinaryFile(input_bin, dim_in0.tot(), &i2_h, &input_pre_inf_d); @@ -114,7 +115,7 @@ bool CenternetDetection3DTrack::init_pre_inf(){ return true; } -bool CenternetDetection3DTrack::init_postprocessing(){ +bool CenterTrack::init_postprocessing(){ srand(0); //seed = 0 for random colors dim_hm = tk::dnn::dataDim_t(1, 10, 128, 128, 1); @@ -203,7 +204,7 @@ bool CenternetDetection3DTrack::init_postprocessing(){ trackId.resize(nBatches, 0); } -bool CenternetDetection3DTrack::init_visualization(const int n_classes){ +bool CenterTrack::init_visualization(const int n_classes){ classes = n_classes; // const char *kitti_class_name[] = { // "person", "car", "bicycle"}; @@ -275,11 +276,11 @@ bool CenternetDetection3DTrack::init_visualization(const int n_classes){ // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); } -void CenternetDetection3DTrack::_get_additional_inputs(){ +void CenterTrack::_get_additional_inputs(){ //None no additional input } -void CenternetDetection3DTrack::pre_inf(const int bi){ +void CenterTrack::pre_inf(const int bi){ TKDNN_TSTART tk::dnn::dataDim_t dim_aus; pre_phase_net->infer(dim_aus, nullptr); @@ -289,7 +290,7 @@ void CenternetDetection3DTrack::pre_inf(const int bi){ checkCuda( cudaDeviceSynchronize() ); } -void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi){ +void CenterTrack::preprocess(cv::Mat &frame, const int bi){ cv::Size sz = originalSize[bi]; // float scale = 1.0; float new_height = dim.h;//sz.height * scale; @@ -403,7 +404,7 @@ void CenternetDetection3DTrack::preprocess(cv::Mat &frame, const int bi){ checkCuda( cudaDeviceSynchronize() ); } -cv::Mat CenternetDetection3DTrack::transform_preds_with_trans(float x1, float x2){ +cv::Mat CenterTrack::transform_preds_with_trans(float x1, float x2){ cv::Mat target_coords(cv::Size(1,3), CV_32F); target_coords.at(0,0) = x1; target_coords.at(0,1) = x2; @@ -411,7 +412,7 @@ cv::Mat CenternetDetection3DTrack::transform_preds_with_trans(float x1, float x2 return transOut * target_coords; } -void CenternetDetection3DTrack::tracking(const int bi) { +void CenterTrack::tracking(const int bi) { float item_size[countDet]; int item_cl[countDet]; float dets[2*countDet]; @@ -600,7 +601,7 @@ void CenternetDetection3DTrack::tracking(const int bi) { } -void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { +void CenterTrack::postprocess(const int bi, const bool mAP) { dnnType *rt_out[9]; rt_out[0] = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi; rt_out[1] = (dnnType *)netRT->buffersRT[2]+ netRT->buffersDIM[2].tot()*bi; @@ -734,7 +735,7 @@ void CenternetDetection3DTrack::postprocess(const int bi, const bool mAP) { tracking(bi); } -void CenternetDetection3DTrack::draw(std::vector& frames) { +void CenterTrack::draw(std::vector& frames) { struct trackingRes t; float sc; int id; @@ -755,7 +756,7 @@ void CenternetDetection3DTrack::draw(std::vector& frames) { cv::Size text_size = getTextSize(txt, cv::FONT_HERSHEY_SIMPLEX, font_scale, thickness, &baseline); if(t.det_res.score > confThreshold){// && t.active!=0) { - if(view2d) { + if(!mode3D) { cv::rectangle(frames[bi], cv::Point(t.det_res.bb0.at(0,0) * scale_x, t.det_res.bb0.at(0,1) * scale_y), cv::Point(t.det_res.bb1.at(0,0) * scale_x, t.det_res.bb1.at(0,1) * scale_y), @@ -776,7 +777,7 @@ void CenternetDetection3DTrack::draw(std::vector& frames) { cv::Scalar(255, 0, 255), 2); } //3d - if(!view2d && t.det_res.z > 1){ + if(mode3D && t.det_res.z > 1){ r.at(0,0) = std::cos(t.det_res.rot_y); r.at(0,2) = std::sin(t.det_res.rot_y); r.at(2,0) = -std::sin(t.det_res.rot_y); diff --git a/tests/centernet/dla34_cnet3d_track/dla34_cnet3d_track.cpp b/tests/centertrack/dla34_ctrack/dla34_ctrack.cpp similarity index 73% rename from tests/centernet/dla34_cnet3d_track/dla34_cnet3d_track.cpp rename to tests/centertrack/dla34_ctrack/dla34_ctrack.cpp index 4829f16..eb3788c 100644 --- a/tests/centernet/dla34_cnet3d_track/dla34_cnet3d_track.cpp +++ b/tests/centertrack/dla34_ctrack/dla34_ctrack.cpp @@ -1,130 +1,130 @@ #include #include "tkdnn.h" -const char *input_bin = "dla34_cnet3d_track/debug/input_base-level0-0.bin"; -// const char *input_bin = "dla34_cnet3d_track/debug/input.bin"; -// const char *pre_img_bin = "dla34_cnet3d_track/debug/pre_imgages.bin"; -// const char *pre_hm_bin = "dla34_cnet3d_track/debug/pre_hms.bin"; +const char *input_bin = "dla34_ctrack/debug/input_base-level0-0.bin"; +// const char *input_bin = "dla34_ctrack/debug/input.bin"; +// const char *pre_img_bin = "dla34_ctrack/debug/pre_imgages.bin"; +// const char *pre_hm_bin = "dla34_ctrack/debug/pre_hms.bin"; // //pre -// const char *pre_img_conv1_bin = "dla34_cnet3d_track/layers/base-pre_img_layer-0.bin"; -// const char *pre_hm_conv1_bin = "dla34_cnet3d_track/layers/base-pre_hm_layer-0.bin"; -// const char *conv1_bin = "dla34_cnet3d_track/layers/base-base_layer-0.bin"; +// const char *pre_img_conv1_bin = "dla34_ctrack/layers/base-pre_img_layer-0.bin"; +// const char *pre_hm_conv1_bin = "dla34_ctrack/layers/base-pre_hm_layer-0.bin"; +// const char *conv1_bin = "dla34_ctrack/layers/base-base_layer-0.bin"; -const char *conv2_bin = "dla34_cnet3d_track/layers/base-level0-0.bin"; -const char *conv3_bin = "dla34_cnet3d_track/layers/base-level1-0.bin"; +const char *conv2_bin = "dla34_ctrack/layers/base-level0-0.bin"; +const char *conv3_bin = "dla34_ctrack/layers/base-level1-0.bin"; // s - stage, t - tree -const char *s1_t1_conv1_bin = "dla34_cnet3d_track/layers/base-level2-tree1-conv1.bin"; -const char *s1_t1_conv2_bin = "dla34_cnet3d_track/layers/base-level2-tree1-conv2.bin"; -const char *s1_t1_project = "dla34_cnet3d_track/layers/base-level2-project-0.bin"; -const char *s1_t2_conv1_bin = "dla34_cnet3d_track/layers/base-level2-tree2-conv1.bin"; -const char *s1_t2_conv2_bin = "dla34_cnet3d_track/layers/base-level2-tree2-conv2.bin"; -const char *s1_root_conv1_bin = "dla34_cnet3d_track/layers/base-level2-root-conv.bin"; -const char *s2_t1_t1_conv1_bin = "dla34_cnet3d_track/layers/base-level3-tree1-tree1-conv1.bin"; -const char *s2_t1_t1_conv2_bin = "dla34_cnet3d_track/layers/base-level3-tree1-tree1-conv2.bin"; -const char *s2_t1_t1_project = "dla34_cnet3d_track/layers/base-level3-tree1-project-0.bin"; -const char *s2_t1_t2_conv1_bin = "dla34_cnet3d_track/layers/base-level3-tree1-tree2-conv1.bin"; -const char *s2_t1_t2_conv2_bin = "dla34_cnet3d_track/layers/base-level3-tree1-tree2-conv2.bin"; -const char *s2_t1_root_conv1_bin = "dla34_cnet3d_track/layers/base-level3-tree1-root-conv.bin"; -const char *s2_t2_t1_conv1_bin = "dla34_cnet3d_track/layers/base-level3-tree2-tree1-conv1.bin"; -const char *s2_t2_t1_conv2_bin = "dla34_cnet3d_track/layers/base-level3-tree2-tree1-conv2.bin"; -const char *s2_t2_t2_conv1_bin = "dla34_cnet3d_track/layers/base-level3-tree2-tree2-conv1.bin"; -const char *s2_t2_t2_conv2_bin = "dla34_cnet3d_track/layers/base-level3-tree2-tree2-conv2.bin"; -const char *s2_t2_root_conv1_bin = "dla34_cnet3d_track/layers/base-level3-tree2-root-conv.bin"; -const char *s3_t1_t1_conv1_bin = "dla34_cnet3d_track/layers/base-level4-tree1-tree1-conv1.bin"; -const char *s3_t1_t1_conv2_bin = "dla34_cnet3d_track/layers/base-level4-tree1-tree1-conv2.bin"; -const char *s3_t1_t1_project = "dla34_cnet3d_track/layers/base-level4-tree1-project-0.bin"; -const char *s3_t1_t2_conv1_bin = "dla34_cnet3d_track/layers/base-level4-tree1-tree2-conv1.bin"; -const char *s3_t1_t2_conv2_bin = "dla34_cnet3d_track/layers/base-level4-tree1-tree2-conv2.bin"; -const char *s3_t1_root_conv1_bin = "dla34_cnet3d_track/layers/base-level4-tree1-root-conv.bin"; -const char *s3_t2_t1_conv1_bin = "dla34_cnet3d_track/layers/base-level4-tree2-tree1-conv1.bin"; -const char *s3_t2_t1_conv2_bin = "dla34_cnet3d_track/layers/base-level4-tree2-tree1-conv2.bin"; -const char *s3_t2_t2_conv1_bin = "dla34_cnet3d_track/layers/base-level4-tree2-tree2-conv1.bin"; -const char *s3_t2_t2_conv2_bin = "dla34_cnet3d_track/layers/base-level4-tree2-tree2-conv2.bin"; -const char *s3_t2_root_conv1_bin = "dla34_cnet3d_track/layers/base-level4-tree2-root-conv.bin"; -const char *s4_t1_conv1_bin = "dla34_cnet3d_track/layers/base-level5-tree1-conv1.bin"; -const char *s4_t1_conv2_bin = "dla34_cnet3d_track/layers/base-level5-tree1-conv2.bin"; -const char *s4_t1_project = "dla34_cnet3d_track/layers/base-level5-project-0.bin"; -const char *s4_t2_conv1_bin = "dla34_cnet3d_track/layers/base-level5-tree2-conv1.bin"; -const char *s4_t2_conv2_bin = "dla34_cnet3d_track/layers/base-level5-tree2-conv2.bin"; -const char *s4_root_conv1_bin = "dla34_cnet3d_track/layers/base-level5-root-conv.bin"; +const char *s1_t1_conv1_bin = "dla34_ctrack/layers/base-level2-tree1-conv1.bin"; +const char *s1_t1_conv2_bin = "dla34_ctrack/layers/base-level2-tree1-conv2.bin"; +const char *s1_t1_project = "dla34_ctrack/layers/base-level2-project-0.bin"; +const char *s1_t2_conv1_bin = "dla34_ctrack/layers/base-level2-tree2-conv1.bin"; +const char *s1_t2_conv2_bin = "dla34_ctrack/layers/base-level2-tree2-conv2.bin"; +const char *s1_root_conv1_bin = "dla34_ctrack/layers/base-level2-root-conv.bin"; +const char *s2_t1_t1_conv1_bin = "dla34_ctrack/layers/base-level3-tree1-tree1-conv1.bin"; +const char *s2_t1_t1_conv2_bin = "dla34_ctrack/layers/base-level3-tree1-tree1-conv2.bin"; +const char *s2_t1_t1_project = "dla34_ctrack/layers/base-level3-tree1-project-0.bin"; +const char *s2_t1_t2_conv1_bin = "dla34_ctrack/layers/base-level3-tree1-tree2-conv1.bin"; +const char *s2_t1_t2_conv2_bin = "dla34_ctrack/layers/base-level3-tree1-tree2-conv2.bin"; +const char *s2_t1_root_conv1_bin = "dla34_ctrack/layers/base-level3-tree1-root-conv.bin"; +const char *s2_t2_t1_conv1_bin = "dla34_ctrack/layers/base-level3-tree2-tree1-conv1.bin"; +const char *s2_t2_t1_conv2_bin = "dla34_ctrack/layers/base-level3-tree2-tree1-conv2.bin"; +const char *s2_t2_t2_conv1_bin = "dla34_ctrack/layers/base-level3-tree2-tree2-conv1.bin"; +const char *s2_t2_t2_conv2_bin = "dla34_ctrack/layers/base-level3-tree2-tree2-conv2.bin"; +const char *s2_t2_root_conv1_bin = "dla34_ctrack/layers/base-level3-tree2-root-conv.bin"; +const char *s3_t1_t1_conv1_bin = "dla34_ctrack/layers/base-level4-tree1-tree1-conv1.bin"; +const char *s3_t1_t1_conv2_bin = "dla34_ctrack/layers/base-level4-tree1-tree1-conv2.bin"; +const char *s3_t1_t1_project = "dla34_ctrack/layers/base-level4-tree1-project-0.bin"; +const char *s3_t1_t2_conv1_bin = "dla34_ctrack/layers/base-level4-tree1-tree2-conv1.bin"; +const char *s3_t1_t2_conv2_bin = "dla34_ctrack/layers/base-level4-tree1-tree2-conv2.bin"; +const char *s3_t1_root_conv1_bin = "dla34_ctrack/layers/base-level4-tree1-root-conv.bin"; +const char *s3_t2_t1_conv1_bin = "dla34_ctrack/layers/base-level4-tree2-tree1-conv1.bin"; +const char *s3_t2_t1_conv2_bin = "dla34_ctrack/layers/base-level4-tree2-tree1-conv2.bin"; +const char *s3_t2_t2_conv1_bin = "dla34_ctrack/layers/base-level4-tree2-tree2-conv1.bin"; +const char *s3_t2_t2_conv2_bin = "dla34_ctrack/layers/base-level4-tree2-tree2-conv2.bin"; +const char *s3_t2_root_conv1_bin = "dla34_ctrack/layers/base-level4-tree2-root-conv.bin"; +const char *s4_t1_conv1_bin = "dla34_ctrack/layers/base-level5-tree1-conv1.bin"; +const char *s4_t1_conv2_bin = "dla34_ctrack/layers/base-level5-tree1-conv2.bin"; +const char *s4_t1_project = "dla34_ctrack/layers/base-level5-project-0.bin"; +const char *s4_t2_conv1_bin = "dla34_ctrack/layers/base-level5-tree2-conv1.bin"; +const char *s4_t2_conv2_bin = "dla34_ctrack/layers/base-level5-tree2-conv2.bin"; +const char *s4_root_conv1_bin = "dla34_ctrack/layers/base-level5-root-conv.bin"; //final -// const char *fc_bin = "dla34_cnet3d_track/layers/output.bin"; +// const char *fc_bin = "dla34_ctrack/layers/output.bin"; -const char *ida_0_p_1_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_0-proj_1-conv.bin"; -const char *ida_0_p_1_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_0-proj_1-conv-conv_offset_mask.bin"; -const char *ida_0_up_1_deconv_bin = "dla34_cnet3d_track/layers/dla_up-ida_0-up_1.bin"; -const char *ida_0_n_1_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_0-node_1-conv.bin"; -const char *ida_0_n_1_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_0-node_1-conv-conv_offset_mask.bin"; +const char *ida_0_p_1_dcn_bin = "dla34_ctrack/layers/dla_up-ida_0-proj_1-conv.bin"; +const char *ida_0_p_1_conv_bin = "dla34_ctrack/layers/dla_up-ida_0-proj_1-conv-conv_offset_mask.bin"; +const char *ida_0_up_1_deconv_bin = "dla34_ctrack/layers/dla_up-ida_0-up_1.bin"; +const char *ida_0_n_1_dcn_bin = "dla34_ctrack/layers/dla_up-ida_0-node_1-conv.bin"; +const char *ida_0_n_1_conv_bin = "dla34_ctrack/layers/dla_up-ida_0-node_1-conv-conv_offset_mask.bin"; -const char *ida_1_p_1_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-proj_1-conv.bin"; -const char *ida_1_p_1_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-proj_1-conv-conv_offset_mask.bin"; -const char *ida_1_up_1_deconv_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-up_1.bin"; -const char *ida_1_n_1_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-node_1-conv.bin"; -const char *ida_1_n_1_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-node_1-conv-conv_offset_mask.bin"; -const char *ida_1_p_2_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-proj_2-conv.bin"; -const char *ida_1_p_2_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-proj_2-conv-conv_offset_mask.bin"; -const char *ida_1_up_2_deconv_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-up_2.bin"; -const char *ida_1_n_2_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-node_2-conv.bin"; -const char *ida_1_n_2_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_1-node_2-conv-conv_offset_mask.bin"; +const char *ida_1_p_1_dcn_bin = "dla34_ctrack/layers/dla_up-ida_1-proj_1-conv.bin"; +const char *ida_1_p_1_conv_bin = "dla34_ctrack/layers/dla_up-ida_1-proj_1-conv-conv_offset_mask.bin"; +const char *ida_1_up_1_deconv_bin = "dla34_ctrack/layers/dla_up-ida_1-up_1.bin"; +const char *ida_1_n_1_dcn_bin = "dla34_ctrack/layers/dla_up-ida_1-node_1-conv.bin"; +const char *ida_1_n_1_conv_bin = "dla34_ctrack/layers/dla_up-ida_1-node_1-conv-conv_offset_mask.bin"; +const char *ida_1_p_2_dcn_bin = "dla34_ctrack/layers/dla_up-ida_1-proj_2-conv.bin"; +const char *ida_1_p_2_conv_bin = "dla34_ctrack/layers/dla_up-ida_1-proj_2-conv-conv_offset_mask.bin"; +const char *ida_1_up_2_deconv_bin = "dla34_ctrack/layers/dla_up-ida_1-up_2.bin"; +const char *ida_1_n_2_dcn_bin = "dla34_ctrack/layers/dla_up-ida_1-node_2-conv.bin"; +const char *ida_1_n_2_conv_bin = "dla34_ctrack/layers/dla_up-ida_1-node_2-conv-conv_offset_mask.bin"; -const char *ida_2_p_1_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-proj_1-conv.bin"; -const char *ida_2_p_1_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-proj_1-conv-conv_offset_mask.bin"; -const char *ida_2_up_1_deconv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-up_1.bin"; -const char *ida_2_n_1_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-node_1-conv.bin"; -const char *ida_2_n_1_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-node_1-conv-conv_offset_mask.bin"; -const char *ida_2_p_2_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-proj_2-conv.bin"; -const char *ida_2_p_2_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-proj_2-conv-conv_offset_mask.bin"; -const char *ida_2_up_2_deconv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-up_2.bin"; -const char *ida_2_n_2_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-node_2-conv.bin"; -const char *ida_2_n_2_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-node_2-conv-conv_offset_mask.bin"; -const char *ida_2_p_3_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-proj_3-conv.bin"; -const char *ida_2_p_3_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-proj_3-conv-conv_offset_mask.bin"; -const char *ida_2_up_3_deconv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-up_3.bin"; -const char *ida_2_n_3_dcn_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-node_3-conv.bin"; -const char *ida_2_n_3_conv_bin = "dla34_cnet3d_track/layers/dla_up-ida_2-node_3-conv-conv_offset_mask.bin"; +const char *ida_2_p_1_dcn_bin = "dla34_ctrack/layers/dla_up-ida_2-proj_1-conv.bin"; +const char *ida_2_p_1_conv_bin = "dla34_ctrack/layers/dla_up-ida_2-proj_1-conv-conv_offset_mask.bin"; +const char *ida_2_up_1_deconv_bin = "dla34_ctrack/layers/dla_up-ida_2-up_1.bin"; +const char *ida_2_n_1_dcn_bin = "dla34_ctrack/layers/dla_up-ida_2-node_1-conv.bin"; +const char *ida_2_n_1_conv_bin = "dla34_ctrack/layers/dla_up-ida_2-node_1-conv-conv_offset_mask.bin"; +const char *ida_2_p_2_dcn_bin = "dla34_ctrack/layers/dla_up-ida_2-proj_2-conv.bin"; +const char *ida_2_p_2_conv_bin = "dla34_ctrack/layers/dla_up-ida_2-proj_2-conv-conv_offset_mask.bin"; +const char *ida_2_up_2_deconv_bin = "dla34_ctrack/layers/dla_up-ida_2-up_2.bin"; +const char *ida_2_n_2_dcn_bin = "dla34_ctrack/layers/dla_up-ida_2-node_2-conv.bin"; +const char *ida_2_n_2_conv_bin = "dla34_ctrack/layers/dla_up-ida_2-node_2-conv-conv_offset_mask.bin"; +const char *ida_2_p_3_dcn_bin = "dla34_ctrack/layers/dla_up-ida_2-proj_3-conv.bin"; +const char *ida_2_p_3_conv_bin = "dla34_ctrack/layers/dla_up-ida_2-proj_3-conv-conv_offset_mask.bin"; +const char *ida_2_up_3_deconv_bin = "dla34_ctrack/layers/dla_up-ida_2-up_3.bin"; +const char *ida_2_n_3_dcn_bin = "dla34_ctrack/layers/dla_up-ida_2-node_3-conv.bin"; +const char *ida_2_n_3_conv_bin = "dla34_ctrack/layers/dla_up-ida_2-node_3-conv-conv_offset_mask.bin"; -const char *ida_up_p_1_dcn_bin = "dla34_cnet3d_track/layers/ida_up-proj_1-conv.bin"; -const char *ida_up_p_1_conv_bin = "dla34_cnet3d_track/layers/ida_up-proj_1-conv-conv_offset_mask.bin"; -const char *ida_up_up_1_deconv_bin = "dla34_cnet3d_track/layers/ida_up-up_1.bin"; -const char *ida_up_n_1_dcn_bin = "dla34_cnet3d_track/layers/ida_up-node_1-conv.bin"; -const char *ida_up_n_1_conv_bin = "dla34_cnet3d_track/layers/ida_up-node_1-conv-conv_offset_mask.bin"; -const char *ida_up_p_2_dcn_bin = "dla34_cnet3d_track/layers/ida_up-proj_2-conv.bin"; -const char *ida_up_p_2_conv_bin = "dla34_cnet3d_track/layers/ida_up-proj_2-conv-conv_offset_mask.bin"; -const char *ida_up_up_2_deconv_bin = "dla34_cnet3d_track/layers/ida_up-up_2.bin"; -const char *ida_up_n_2_dcn_bin = "dla34_cnet3d_track/layers/ida_up-node_2-conv.bin"; -const char *ida_up_n_2_conv_bin = "dla34_cnet3d_track/layers/ida_up-node_2-conv-conv_offset_mask.bin"; +const char *ida_up_p_1_dcn_bin = "dla34_ctrack/layers/ida_up-proj_1-conv.bin"; +const char *ida_up_p_1_conv_bin = "dla34_ctrack/layers/ida_up-proj_1-conv-conv_offset_mask.bin"; +const char *ida_up_up_1_deconv_bin = "dla34_ctrack/layers/ida_up-up_1.bin"; +const char *ida_up_n_1_dcn_bin = "dla34_ctrack/layers/ida_up-node_1-conv.bin"; +const char *ida_up_n_1_conv_bin = "dla34_ctrack/layers/ida_up-node_1-conv-conv_offset_mask.bin"; +const char *ida_up_p_2_dcn_bin = "dla34_ctrack/layers/ida_up-proj_2-conv.bin"; +const char *ida_up_p_2_conv_bin = "dla34_ctrack/layers/ida_up-proj_2-conv-conv_offset_mask.bin"; +const char *ida_up_up_2_deconv_bin = "dla34_ctrack/layers/ida_up-up_2.bin"; +const char *ida_up_n_2_dcn_bin = "dla34_ctrack/layers/ida_up-node_2-conv.bin"; +const char *ida_up_n_2_conv_bin = "dla34_ctrack/layers/ida_up-node_2-conv-conv_offset_mask.bin"; -const char *hm_conv1_bin = "dla34_cnet3d_track/layers/hm-0.bin"; -const char *hm_conv2_bin = "dla34_cnet3d_track/layers/hm-2.bin"; -const char *wh_conv1_bin = "dla34_cnet3d_track/layers/wh-0.bin"; -const char *wh_conv2_bin = "dla34_cnet3d_track/layers/wh-2.bin"; -const char *reg_conv1_bin = "dla34_cnet3d_track/layers/reg-0.bin"; -const char *reg_conv2_bin = "dla34_cnet3d_track/layers/reg-2.bin"; -const char *track_conv1_bin = "dla34_cnet3d_track/layers/tracking-0.bin"; -const char *track_conv2_bin = "dla34_cnet3d_track/layers/tracking-2.bin"; -const char *dep_conv1_bin = "dla34_cnet3d_track/layers/dep-0.bin"; -const char *dep_conv2_bin = "dla34_cnet3d_track/layers/dep-2.bin"; -const char *rot_conv1_bin = "dla34_cnet3d_track/layers/rot-0.bin"; -const char *rot_conv2_bin = "dla34_cnet3d_track/layers/rot-2.bin"; -const char *dim_conv1_bin = "dla34_cnet3d_track/layers/dim-0.bin"; -const char *dim_conv2_bin = "dla34_cnet3d_track/layers/dim-2.bin"; -const char *a_off_conv1_bin = "dla34_cnet3d_track/layers/amodel_offset-0.bin"; -const char *a_off_conv2_bin = "dla34_cnet3d_track/layers/amodel_offset-2.bin"; +const char *hm_conv1_bin = "dla34_ctrack/layers/hm-0.bin"; +const char *hm_conv2_bin = "dla34_ctrack/layers/hm-2.bin"; +const char *wh_conv1_bin = "dla34_ctrack/layers/wh-0.bin"; +const char *wh_conv2_bin = "dla34_ctrack/layers/wh-2.bin"; +const char *reg_conv1_bin = "dla34_ctrack/layers/reg-0.bin"; +const char *reg_conv2_bin = "dla34_ctrack/layers/reg-2.bin"; +const char *track_conv1_bin = "dla34_ctrack/layers/tracking-0.bin"; +const char *track_conv2_bin = "dla34_ctrack/layers/tracking-2.bin"; +const char *dep_conv1_bin = "dla34_ctrack/layers/dep-0.bin"; +const char *dep_conv2_bin = "dla34_ctrack/layers/dep-2.bin"; +const char *rot_conv1_bin = "dla34_ctrack/layers/rot-0.bin"; +const char *rot_conv2_bin = "dla34_ctrack/layers/rot-2.bin"; +const char *dim_conv1_bin = "dla34_ctrack/layers/dim-0.bin"; +const char *dim_conv2_bin = "dla34_ctrack/layers/dim-2.bin"; +const char *a_off_conv1_bin = "dla34_ctrack/layers/amodel_offset-0.bin"; +const char *a_off_conv2_bin = "dla34_ctrack/layers/amodel_offset-2.bin"; const char *output_bin[]={ -"dla34_cnet3d_track/debug/hm.bin", -"dla34_cnet3d_track/debug/wh.bin", -"dla34_cnet3d_track/debug/reg.bin", -"dla34_cnet3d_track/debug/tracking.bin", -"dla34_cnet3d_track/debug/dep.bin", -"dla34_cnet3d_track/debug/rot.bin", -"dla34_cnet3d_track/debug/dim.bin", -"dla34_cnet3d_track/debug/amodel_offset.bin"}; -// const char *output_bin = "dla34_cnet3d_track/debug/base-level0-2.bin"; +"dla34_ctrack/debug/hm.bin", +"dla34_ctrack/debug/wh.bin", +"dla34_ctrack/debug/reg.bin", +"dla34_ctrack/debug/tracking.bin", +"dla34_ctrack/debug/dep.bin", +"dla34_ctrack/debug/rot.bin", +"dla34_ctrack/debug/dim.bin", +"dla34_ctrack/debug/amodel_offset.bin"}; +// const char *output_bin = "dla34_ctrack/debug/base-level0-2.bin"; int main() { - downloadWeightsifDoNotExist("dla34_cnet3d_track/debug/input.bin", "dla34_cnet3d_track", "https://cloud.hipert.unimore.it/s/rjNfgGL9FtAXLHp/download"); + downloadWeightsifDoNotExist("dla34_ctrack/debug/input.bin", "dla34_ctrack", "https://cloud.hipert.unimore.it/s/rjNfgGL9FtAXLHp/download"); // Network layout // tk::dnn::dataDim_t dim_in0(1, 3, 512, 512, 1); @@ -570,7 +570,7 @@ int main() net.print(); //convert network to tensorRT - tk::dnn::NetworkRT netRT(&net, net.getNetworkRTName("dla34_cnet3d_track")); + tk::dnn::NetworkRT netRT(&net, net.getNetworkRTName("dla34_ctrack")); tk::dnn::dataDim_t dim1 = dim_in0; //input dim printCenteredTitle(" CUDNN inference ", '=', 30); -- 2.52.0 From fd56e64938d2427de88854f519d628f0bc4ae899 Mon Sep 17 00:00:00 2001 From: Davide Sapienza Date: Tue, 11 May 2021 17:02:43 +0200 Subject: [PATCH 066/162] Update the README and split it into several files. Signed-off-by: Davide Sapienza --- README.md | 339 +------------------------------------- docs/demo.md | 213 ++++++++++++++++++++++++ docs/exporting_weights.md | 100 +++++++++++ docs/mAP_demo.md | 34 ++++ docs/windows.md | 95 +++++++++++ 5 files changed, 447 insertions(+), 334 deletions(-) create mode 100644 docs/demo.md create mode 100644 docs/exporting_weights.md create mode 100644 docs/mAP_demo.md create mode 100644 docs/windows.md diff --git a/README.md b/README.md index 9e4b811..12f2104 100644 --- a/README.md +++ b/README.md @@ -70,26 +70,11 @@ Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001 - [How to compile this repo](#how-to-compile-this-repo) - [Workflow](#workflow) - [How to export weights](#how-to-export-weights) - - [1)Export weights from darknet](#1export-weights-from-darknet) - - [2)Export weights for DLA34 and ResNet101](#2export-weights-for-dla34-and-resnet101) - - [3)Export weights for CenterNet](#3export-weights-for-centernet) - - [4)Export weights for MobileNetSSD](#4export-weights-for-mobilenetssd) - [Run the demo](#run-the-demo) - - [FP16 inference](#fp16-inference) - - [INT8 inference](#int8-inference) - [mAP demo](#map-demo) - [Existing tests and supported networks](#existing-tests-and-supported-networks) - [References](#references) - [tkDNN on Windows 10 (experimental)](#tkdnn-on-windows-10-experimental) - - [Dependencies-Windows](#dependencies-windows) - - [Compiling tkDNN on Windows](#compiling-tkdnn-on-windows) - - [Run the demo on Windows](#run-the-demo-on-windows) - - [FP16 inference windows](#fp16-inference-windows) - - [INT8 inference windows](#int8-inference-windows) - - [Known issues with tkDNN on Windows](#known-issues-with-tkdnn-on-windows) - - - ## Dependencies @@ -126,246 +111,17 @@ Steps needed to do inference on tkDNN with a custom neural network. * Create a new test and define the network, layer by layer using the weights extracted and the output to check the results. * Do inference. -## How to export weights +## Exporting weights -Weights are essential for any network to run inference. For each test a folder organized as follow is needed (in the build folder): -``` - test_nn - |---- layers/ (folder containing a binary file for each layer with the corresponding wieghts and bias) - |---- debug/ (folder containing a binary file for each layer with the corresponding outputs) -``` -Therefore, once the weights have been exported, the folders layers and debug should be placed in the corresponding test. - -### 1)Export weights from darknet -To export weights for NNs that are defined in darknet framework, use [this](https://git.hipert.unimore.it/fgatti/darknet.git) fork of darknet and follow these steps to obtain a correct debug and layers folder, ready for tkDNN. - -``` -git clone https://git.hipert.unimore.it/fgatti/darknet.git -cd darknet -make -mkdir layers debug -./darknet export layers -``` -N.b. Use compilation with CPU (leave GPU=0 in Makefile) if you also want debug. - -### 2)Export weights for DLA34 and ResNet101 -To get weights and outputs needed to run the tests dla34 and resnet101 use the Python script and the Anaconda environment included in the repository. - -Create Anaconda environment and activate it: -``` -conda env create -f file_name.yml -source activate env_name -python