13 Commits

Author SHA1 Message Date
Francesco Gatti 7c0620e391 test_all_test script save results in separate files 2022-03-30 15:47:41 +02:00
Micaela Verucchi 3bcc32ffdc Fix nms thresh
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2022-02-15 19:57:57 +01:00
Micaela Verucchi 04de9908a6 Add yolo4-csp for crowdhuman dataset, add shelfnet for coco-stuff dataset, fix minor in demo.cpp
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-11-29 17:33:57 +01:00
Micaela Verucchi 9cbac460bc Update README.md 2021-11-25 11:32:38 +01:00
Micaela Verucchi 24cdb4c4a7 Update Readme
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-11-24 18:20:26 +01:00
Micaela Verucchi be5864748a Use yaml config file for the demo instead of param list
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-11-23 16:29:40 +01:00
Micaela Verucchi 75c3cb0038 Add script to compare times_rtinference.csv files
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-11-23 13:01:58 +01:00
Micaela Verucchi 55df97afe1 Fix warnings, upgrade to C++14
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-11-23 13:01:07 +01:00
Micaela Verucchi d6fb6c6af4 Fix max elem (remove thrust) for segmentation
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-11-04 19:28:55 +01:00
Micaela Verucchi eca10ac0a8 Update weights
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-09-19 00:16:31 +02:00
Micaela Verucchi a992c9feb5 Update README.md 2021-07-23 16:50:15 +02:00
Micaela Verucchi 09080709a9 Remove Issues.md
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-07-23 15:23:47 +02:00
Micaela Verucchi 7521d10ba7 Update README with gifs
Signed-off-by: Micaela Verucchi <micaelaverucchi@gmail.com>
2021-07-23 15:07:52 +02:00
25 changed files with 1832 additions and 114 deletions
+5 -2
View File
@@ -3,10 +3,10 @@ cmake_minimum_required(VERSION 3.15)
project (tkDNN) project (tkDNN)
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake)
if(UNIX) if(UNIX)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable ") set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++14 -fPIC -Wno-deprecated-declarations")
endif() endif()
if(WIN32) if(WIN32)
set(CMAKE_CXX_STANDARD 11) set(CMAKE_CXX_STANDARD 14)
set(CMAKE_CXX_FLAGS "/O2 /FS /EHsc") set(CMAKE_CXX_FLAGS "/O2 /FS /EHsc")
set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON) set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON)
endif(WIN32) endif(WIN32)
@@ -140,6 +140,9 @@ target_link_libraries(test_shelfnet_berkeley tkDNN)
add_executable(test_shelfnet_mapillary tests/shelfnet/shelfnet_mapillary.cpp) add_executable(test_shelfnet_mapillary tests/shelfnet/shelfnet_mapillary.cpp)
target_link_libraries(test_shelfnet_mapillary tkDNN) target_link_libraries(test_shelfnet_mapillary tkDNN)
add_executable(test_shelfnet_coco tests/shelfnet/shelfnet_coco.cpp)
target_link_libraries(test_shelfnet_coco tkDNN)
# DEMOS # DEMOS
add_executable(test_rtinference tests/test_rtinference/rtinference.cpp) add_executable(test_rtinference tests/test_rtinference/rtinference.cpp)
target_link_libraries(test_rtinference tkDNN) target_link_libraries(test_rtinference tkDNN)
-1
View File
@@ -1 +0,0 @@
1)error C2131 @ Yolo3Detection.cpp(97) -> expression doesnt evaluate to a constant caused to read of variable outside its lifetime
+10 -6
View File
@@ -17,10 +17,15 @@ If you use tkDNN in your research, please cite the [following paper](https://iee
} }
``` ```
### What's new (20 July 2021) ### What's new
#### 20 July 2021
- [x] Support to sematic segmentation [README](docs/README_seg.md) - [x] Support to sematic segmentation [README](docs/README_seg.md)
- [x] Support 2D/3D Object Detection and Tracking [README](docs/README_2d3dtracking.md) - [x] Support 2D/3D Object Detection and Tracking [README](docs/README_2d3dtracking.md)
- [ ] Support to TensorRT8 (WIP) #### 24 November 2021
- [x] Support to sematic segmentation on cuda 11
- [x] Support to TensorRT8 (thanks to [Harshvardhan Chandirasekar](https://github.com/perseusdg)).
TensorRT8 (and therefore Jetpack 4.6) is currently supported only on the branch tensorrt8 due to [performance issue with TensorRT8](https://docs.nvidia.com/deeplearning/tensorrt/release-notes/tensorrt-8.html)). We will merge it to the master as soon as those issues are fixed (probably in future minor releases).
## FPS Results ## FPS Results
Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimension as the input size, on Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimension as the input size, on
@@ -82,7 +87,7 @@ Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001
## Dependencies ## Dependencies
This branch works on every NVIDIA GPU that supports the following (latest tested) dependencies: This branch works on every NVIDIA GPU that supports the following (latest tested) dependencies:
* CUDA 11.0 (or >= 10) * CUDA 11.0 (or >= 10) [the segmentation only works with CUDA 10 for now]
* cuDNN 8.0.4 (or >= 7.3) * cuDNN 8.0.4 (or >= 7.3)
* TensorRT 7.2.0 (or >=5) * TensorRT 7.2.0 (or >=5)
* OpenCV 4.5.2 (or >=4) * OpenCV 4.5.2 (or >=4)
@@ -168,11 +173,10 @@ For specific details on how to run tkDNN on Windows 10 see [HERE](./docs/windows
| yolo4_320 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 320x320 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) | | yolo4_320 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 320x320 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
| yolo4_512 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) | | yolo4_512 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
| yolo4_608 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 608x608 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) | | yolo4_608 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 608x608 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
| yolo4_berkeley | Yolov4 <sup>8</sup> | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 540x320 | [weights](https://cloud.hipert.unimore.it/s/nkWFa5fgb4NTdnB/download) | | yolo4_berkeley | Yolov4 <sup>8</sup> | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 544x320 | [weights](https://cloud.hipert.unimore.it/s/nkWFa5fgb4NTdnB/download) |
| yolo4tiny | Yolov4 tiny <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) | | yolo4tiny | Yolov4 tiny <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) |
| yolo4x | Yolov4x-mish <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | | yolo4x | Yolov4x-mish <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) |
| yolo4tiny_512 | Yolov4 tiny <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) | | yolo4tiny_512 | Yolov4 tiny <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) |
80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) |
| yolo4x-cps | Scaled Yolov4 <sup>10</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) | | yolo4x-cps | Scaled Yolov4 <sup>10</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) |
| shelfnet | ShelfNet18_realtime<sup>11</sup> | [Cityscapes](https://www.cityscapes-dataset.com/) | 19 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/mEDZMRJaGCFWSJF/download) | | shelfnet | ShelfNet18_realtime<sup>11</sup> | [Cityscapes](https://www.cityscapes-dataset.com/) | 19 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/mEDZMRJaGCFWSJF/download) |
| shelfnet_berkeley | ShelfNet18_realtime<sup>11</sup> | [DeepDrive](https://bdd-data.berkeley.edu/) | 20 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/m92e7QdD9gYMF7f/download) | | shelfnet_berkeley | ShelfNet18_realtime<sup>11</sup> | [DeepDrive](https://bdd-data.berkeley.edu/) | 20 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/m92e7QdD9gYMF7f/download) |
+45 -36
View File
@@ -9,7 +9,6 @@
#include "Yolo3Detection.h" #include "Yolo3Detection.h"
bool gRun; bool gRun;
bool SAVE_RESULT = false;
void sig_handler(int signo) { void sig_handler(int signo) {
std::cout<<"request gateway stop\n"; std::cout<<"request gateway stop\n";
@@ -18,43 +17,53 @@ void sig_handler(int signo) {
int main(int argc, char *argv[]) { int main(int argc, char *argv[]) {
std::cout<<"detection\n";
signal(SIGINT, sig_handler); signal(SIGINT, sig_handler);
// get config file path and read it
std::string net = "yolo4tiny_fp32.rt";
if(argc > 1)
net = argv[1];
#ifdef __linux__ #ifdef __linux__
std::string input = "../demo/yolo_test.mp4"; std::string config_file = "../demo/demoConfig.yaml";
#elif _WIN32 #elif _WIN32
std::string input = "..\\..\\..\\demo\\yolo_test.mp4"; std::string config_file = "..\\..\\..\\demo\\demoConfig.yaml";
#endif #endif
if(argc > 1)
config_file = argv[1];
YAML::Node conf = YAMLloadConf(config_file);
if(!conf)
FatalError("Problem with config file");
if(argc > 2) // read settings from config file
input = argv[2]; std::string net = YAMLgetConf<std::string>(conf, "net", "yolo4tiny_fp32.rt");
char ntype = 'y'; if(!fileExist(net.c_str()))
if(argc > 3) FatalError("The given network does not exist. Create the rt first.");
ntype = argv[3][0];
int n_classes = 80;
if(argc > 4)
n_classes = atoi(argv[4]);
int n_batch = 1;
if(argc > 5)
n_batch = atoi(argv[5]);
bool show = true;
if(argc > 6)
show = atoi(argv[6]);
float conf_thresh=0.3;
if(argc > 7)
conf_thresh = atof(argv[7]);
#ifdef __linux__
std::string input = YAMLgetConf<std::string>(conf, "input", "../demo/yolo_test.mp4");
#elif _WIN32
std::string input = YAMLgetConf(conf, "win_input", "..\\..\\..\\demo\\yolo_test.mp4");
#endif
if(!fileExist(input.c_str()))
FatalError("The given input video does not exist.");
char ntype = YAMLgetConf<char>(conf, "ntype", 'y');
int n_classes = YAMLgetConf<int>(conf, "n_classes", 80);
int n_batch = YAMLgetConf<int>(conf, "n_batch", 1);
if(n_batch < 1 || n_batch > 64) if(n_batch < 1 || n_batch > 64)
FatalError("Batch dim not supported"); FatalError("Batch dim not supported");
float conf_thresh = YAMLgetConf<float>(conf, "conf_thresh", 0.3);
bool show = YAMLgetConf<bool>(conf, "show", true);
bool save = YAMLgetConf<bool>(conf, "save", false);
if(!show) std::cout <<"Net settings - net: "<< net
SAVE_RESULT = true; <<", ntype: "<< ntype
<<", n_classes: "<< n_classes
<<", n_batch: "<< n_batch
<<", conf_thresh: "<< conf_thresh<<"\n";
std::cout <<"Demo settings - input: "<< input
<<", show: "<< show
<<", save: "<< save<<"\n\n";
// create detection network
tk::dnn::Yolo3Detection yolo; tk::dnn::Yolo3Detection yolo;
tk::dnn::CenternetDetection cnet; tk::dnn::CenternetDetection cnet;
tk::dnn::MobilenetDetection mbnet; tk::dnn::MobilenetDetection mbnet;
@@ -79,8 +88,7 @@ int main(int argc, char *argv[]) {
detNN->init(net, n_classes, n_batch, conf_thresh); detNN->init(net, n_classes, n_batch, conf_thresh);
gRun = true; // open video stream
cv::VideoCapture cap(input); cv::VideoCapture cap(input);
if(!cap.isOpened()) if(!cap.isOpened())
gRun = false; gRun = false;
@@ -88,19 +96,21 @@ int main(int argc, char *argv[]) {
std::cout<<"camera started\n"; std::cout<<"camera started\n";
cv::VideoWriter resultVideo; cv::VideoWriter resultVideo;
if(SAVE_RESULT) { if(save) {
int w = cap.get(cv::CAP_PROP_FRAME_WIDTH); int w = cap.get(cv::CAP_PROP_FRAME_WIDTH);
int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT); int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT);
resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h)); resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h));
} }
cv::Mat frame;
if(show) if(show)
cv::namedWindow("detection", cv::WINDOW_NORMAL); cv::namedWindow("detection", cv::WINDOW_NORMAL);
cv::Mat frame;
std::vector<cv::Mat> batch_frame; std::vector<cv::Mat> batch_frame;
std::vector<cv::Mat> batch_dnn_input; std::vector<cv::Mat> batch_dnn_input;
// start detection loop
gRun = true;
while(gRun) { while(gRun) {
batch_dnn_input.clear(); batch_dnn_input.clear();
batch_frame.clear(); batch_frame.clear();
@@ -128,19 +138,18 @@ int main(int argc, char *argv[]) {
cv::waitKey(1); cv::waitKey(1);
} }
} }
if(n_batch == 1 && SAVE_RESULT) if(n_batch == 1 && save)
resultVideo << frame; resultVideo << frame;
} }
std::cout<<"detection end\n"; std::cout<<"detection end\n";
double mean = 0;
double mean = 0;
std::cout<<COL_GREENB<<"\n\nTime stats:\n"; std::cout<<COL_GREENB<<"\n\nTime stats:\n";
std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n"; std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n"; std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
for(int i=0; i<detNN->stats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size(); for(int i=0; i<detNN->stats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size();
std::cout<<"Avg: "<<mean/n_batch<<" ms\t"<<1000/(mean/n_batch)<<" FPS\n"<<COL_END; std::cout<<"Avg: "<<mean/n_batch<<" ms\t"<<1000/(mean/n_batch)<<" FPS\n"<<COL_END;
return 0; return 0;
} }
+14
View File
@@ -0,0 +1,14 @@
# video input
input : "../demo/yolo_test.mp4"
win_input : "..\\..\\..\\demo\\yolo_test.mp4"
# network config
net : "yolo4tiny_fp32.rt"
ntype : 'y'
n_classes : 80
n_batch : 1
conf_thresh : 0.3
# demo config
show : true
save : true
+3
View File
@@ -18,6 +18,8 @@ where
* ```<calibration-file>``` is the camera calibration file (opencv format). It is important that the file contains entry "camera_matrix" with sub-entry "rows", "cols", "data". If you do not want to pass the calibration file, pass "NULL" instead. * ```<calibration-file>``` is the camera calibration file (opencv format). It is important that the file contains entry "camera_matrix" with sub-entry "rows", "cols", "data". If you do not want to pass the calibration file, pass "NULL" instead.
![demo](https://user-images.githubusercontent.com/11939259/126784875-c4285497-d369-424f-abda-58274cd747ac.gif)
## Object Detection and Tracking ## Object Detection and Tracking
To run the 3D object detection & tracking demo follow these steps (example with CenterTrack based on DLA34): To run the 3D object detection & tracking demo follow these steps (example with CenterTrack based on DLA34):
@@ -37,6 +39,7 @@ where
* ```<calibration-file>``` is the camera calibration file (opencv format). It is important that the file contains entry "camera_matrix" with sub-entry "rows", "cols", "data". If you do not want to pass the calibration file, pass "NULL" instead. * ```<calibration-file>``` is the camera calibration file (opencv format). It is important that the file contains entry "camera_matrix" with sub-entry "rows", "cols", "data". If you do not want to pass the calibration file, pass "NULL" instead.
* ```<2D/3D-flag>``` if set to 0 the demo will be in the 2D mode, while if set to 1 the demo will be in the 3D mode (Default is 1 - 3D mode). * ```<2D/3D-flag>``` if set to 0 the demo will be in the 2D mode, while if set to 1 the demo will be in the 3D mode (Default is 1 - 3D mode).
![demo](https://user-images.githubusercontent.com/11939259/126784878-513fa9e8-864a-4c24-b4bd-199737184708.gif)
## FPS Results ## FPS Results
+1 -1
View File
@@ -29,7 +29,7 @@ where
NB) By default it is used FP32 inference NB) By default it is used FP32 inference
NB) The batching is not used to work on more streams, rather to work on more tiles of the same image. Shelfnet never resized the input image, therefore for images greater than 1024x1024 tiles of 1024x1024 are given in input to the network in batch. NB) The batching is not used to work on more streams, rather to work on more tiles of the same image. Shelfnet never resized the input image, therefore for images greater than 1024x1024 tiles of 1024x1024 are given in input to the network in batch.
![gif](output.gif "Results on yolo_test.mp4") ![demo](https://user-images.githubusercontent.com/11939259/126784236-38d24fc3-02df-4514-81c4-497e87e40b65.gif "Results on yolo_test.mp4")
For other demo videos refer to [this playlist](https://www.youtube.com/playlist?list=PLv0nEQYDD45y5EdSiywwCGPBmJVUzIWwe). For other demo videos refer to [this playlist](https://www.youtube.com/playlist?list=PLv0nEQYDD45y5EdSiywwCGPBmJVUzIWwe).
+16 -15
View File
@@ -32,21 +32,20 @@ make
Once you have successfully created your rt file, run the demo: Once you have successfully created your rt file, run the demo:
``` ```
./demo yolo4_fp32.rt ../demo/yolo_test.mp4 y ./demo <path-to-config>
``` ```
In general the demo program takes 7 parameters: In general the demo program takes 1 parameter, the ```<path-to-config>``` that is the path to che configuration file. The parameter is optional and its default value is ```"../demo/demoConfig.yaml"```.
```
./demo <network-rt-file> <path-to-video> <kind-of-network> <number-of-classes> <n-batches> <show-flag> <conf-thresh>
```
where
* ```<network-rt-file>``` is the rt file generated by a test The config file is a yaml file with the following attributes:
* ```<<path-to-video>``` is the path to a video file or a camera input * ```net``` is the rt file generated by a test
* ```<kind-of-network>``` is the type of network. Thee types are currently supported: ```y``` (YOLO family), ```c``` (CenterNet family) and ```m``` (MobileNet-SSD family) * ```input``` is the path to a video file or a camera input (on Linux)
* ```<number-of-classes>```is the number of classes the network is trained on * ```win_input``` is the path to a video file or a camera input (on Windows)
* ```<n-batches>``` number of batches to use in inference (N.B. you should first export TKDNN_BATCHSIZE to the required n_batches and create again the rt file for the network). * ```ntype``` is the type of network. Thee types are currently supported: ```y``` (YOLO family), ```c``` (CenterNet family) and ```m``` (MobileNet-SSD family)
* ```<show-flag>``` if set to 0 the demo will not show the visualization but save the video into result.mp4 (if n-batches ==1) * ```n_classes``` is the number of classes the network is trained on
* ```<conf-thresh>``` confidence threshold for the detector. Only bounding boxes with threshold greater than conf-thresh will be displayed. * ```n_batch``` number of batches to use in inference (N.B. you should first export TKDNN_BATCHSIZE to the required n_batches and create again the rt file for the network).
* ```conf_thresh``` confidence threshold for the detector. Only bounding boxes with threshold greater than conf-thresh will be displayed.
* ```show``` if set to 0 the demo will not show the visualization (if n-batches ==1)
* ```save``` if set to 1 the demo will save the video of the demo into result.mp4 (if n-batches ==1)
N.B. By default it is used FP32 inference N.B. By default it is used FP32 inference
@@ -61,7 +60,8 @@ To run the demo with FP16 inference follow these steps (example with yolov3):
export TKDNN_MODE=FP16 # set the half floating point optimization export TKDNN_MODE=FP16 # set the half floating point optimization
rm yolo3_fp16.rt # be sure to delete(or move) old tensorRT files rm yolo3_fp16.rt # be sure to delete(or move) old tensorRT files
./test_yolo3 # run the yolo test (is slow) ./test_yolo3 # run the yolo test (is slow)
./demo yolo3_fp16.rt ../demo/yolo_test.mp4 y # set net: yolo3_fp16.rt in the config-file
./demo
``` ```
N.B. Using FP16 inference will lead to some errors in the results (first or second decimal). N.B. Using FP16 inference will lead to some errors in the results (first or second decimal).
@@ -86,7 +86,8 @@ export TKDNN_CALIB_LABEL_PATH=../demo/COCO_val2017/all_labels.txt
export TKDNN_CALIB_IMG_PATH=../demo/COCO_val2017/all_images.txt export TKDNN_CALIB_IMG_PATH=../demo/COCO_val2017/all_images.txt
rm yolo3_int8.rt # be sure to delete(or move) old tensorRT files rm yolo3_int8.rt # be sure to delete(or move) old tensorRT files
./test_yolo3 # run the yolo test (is slow) ./test_yolo3 # run the yolo test (is slow)
./demo yolo3_int8.rt ../demo/yolo_test.mp4 y # set net: yolo3_int8.rt in the config-file
./demo
``` ```
N.B. N.B.
BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 6.3 MiB

+1
View File
@@ -182,6 +182,7 @@ class SegmentationNN {
checkCuda(cudaMemcpyAsync(mean_d, mean.data(), mean.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream)); checkCuda(cudaMemcpyAsync(mean_d, mean.data(), mean.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream));
checkCuda(cudaMemcpyAsync(stddev_d, stddev.data(), stddev.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream)); checkCuda(cudaMemcpyAsync(stddev_d, stddev.data(), stddev.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream));
return true;
} }
/** /**
+16
View File
@@ -21,6 +21,7 @@
#include <ios> #include <ios>
#include <chrono> #include <chrono>
#include <yaml-cpp/yaml.h>
#define dnnType float #define dnnType float
@@ -137,4 +138,19 @@ static inline bool isCudaPointer(void *data) {
cudaPointerAttributes attr; cudaPointerAttributes attr;
return cudaPointerGetAttributes(&attr, data) == 0; return cudaPointerGetAttributes(&attr, data) == 0;
} }
inline YAML::Node YAMLloadConf(const std::string& conf_file) {
std::cerr<<"Loading YAML: "<<conf_file<<"\n";
return YAML::LoadFile(conf_file);
}
template<typename T>
inline T YAMLgetConf(YAML::Node conf, std::string key, T defaultVal) {
T val = defaultVal;
if(conf && conf[key]) {
val = conf[key].as<T>();
}
return val;
}
#endif //UTILS_H #endif //UTILS_H
+37
View File
@@ -0,0 +1,37 @@
import sys
import pandas as pd
if len(sys.argv) < 3:
print("Error: two csv files are needed, old first new second")
exit(1)
old_perf_file = str(sys.argv[1])
new_perf_file = str(sys.argv[2])
verbose = False
if len(sys.argv) == 4:
verbose = bool(sys.argv[3])
print("Comparing {} vs {}".format(old_perf_file, new_perf_file))
df_old = pd.read_csv (old_perf_file, sep=';', header=None, index_col=0)
df_new = pd.read_csv (new_perf_file, sep=';', header=None, index_col=0)
for index, row in df_new.iterrows():
if index in df_old.index:
if verbose:
print("New: ",row[1], row[2], row[3])
print("Old: ",df_old.loc[index][1], df_old.loc[index][2], df_old.loc[index][3])
print(index, end=': ')
if abs(row[1] - df_old.loc[index][1]) < df_old.loc[index][1]*0.1:
print("similar performance")
elif (row[1] < df_old.loc[index][1]):
print('\x1b[3;30;42m' + 'faster' + '\x1b[0m')
elif (row[1] > df_old.loc[index][1]):
if row[1] > df_old.loc[index][1] + df_old.loc[index][1] * 0.5 :
print('\x1b[3;30;41m' + 'WAY SLOWER' + '\x1b[0m')
else:
print('\x1b[3;30;41m' + 'slower' + '\x1b[0m')
+43 -39
View File
@@ -1,6 +1,6 @@
#!/bin/bash #!/bin/bash
cd build #cd build
RED='\033[1;31m' RED='\033[1;31m'
GREEN='\033[1;32m' GREEN='\033[1;32m'
@@ -29,24 +29,28 @@ function print_output {
} }
out_dir=results
out_file=results.log out_file=results.log
rm $out_file rm -rf $out_dir/
mkdir -p $out_dir
function test_net { function test_net {
./test_$1 &>> $out_file ./test_$1 &> $out_dir/$1_${TKDNN_MODE}_build_$out_file
print_output $? $1 print_output $? $1
./test_rtinference $1*.rt $TKDNN_BATCHSIZE &>> $out_file ./test_rtinference $1*.rt 1 &> $out_dir/$1_${TKDNN_MODE}_inference_batch1_$out_file
print_output $? "infer $1"
./test_rtinference $1*.rt $TKDNN_BATCHSIZE &> $out_dir/$1_${TKDNN_MODE}_inference_batch${TKDNN_BATCHSIZE}_$out_file
print_output $? "batched $1" print_output $? "batched $1"
} }
modes=( 1 ) # only FP32 # modes=( 1 ) # only FP32
# modes=( 1 2 ) # FP32 and FP16 modes=( 1 2 ) # FP32 and FP16
# modes=( 1 2 3 ) # FP32, FP16 and INT8 # modes=( 1 2 3 ) # FP32, FP16 and INT8
for i in "${modes[@]}" for i in "${modes[@]}"
do do
rm *rt rm -f *rt
if [ $i -eq 1 ] if [ $i -eq 1 ]
then then
export TKDNN_MODE=FP32 export TKDNN_MODE=FP32
@@ -73,37 +77,37 @@ do
# print_output $? imuodom # print_output $? imuodom
test_net yolo4 test_net yolo4
test_net yolo4_320 # test_net yolo4_320
test_net yolo4_320_coco2 # test_net yolo4_320_coco2
test_net yolo4_512 # test_net yolo4_512
test_net yolo4_608 # test_net yolo4_608
test_net yolo4-csp # test_net yolo4-csp
test_net yolo4x # test_net yolo4x
test_net yolo4_berkeley # test_net yolo4_berkeley
test_net yolo4_berkeley_f1 # test_net yolo4_berkeley_f1
test_net yolo4tiny # test_net yolo4tiny
test_net yolo4tiny_512 # test_net yolo4tiny_512
test_net yolo3 # test_net yolo3
test_net yolo3_berkeley # test_net yolo3_berkeley
test_net yolo3_coco4 # test_net yolo3_coco4
test_net yolo3_flir # test_net yolo3_flir
test_net yolo3_512 # test_net yolo3_512
test_net yolo3tiny # test_net yolo3tiny
test_net yolo3tiny_512 # test_net yolo3tiny_512
test_net yolo2 # test_net yolo2
test_net yolo2_voc # test_net yolo2_voc
#test_net yolo2tiny # test_net yolo2tiny
test_net csresnext50-panet-spp # test_net csresnext50-panet-spp
#test_net csresnext50-panet-spp_berkeley # test_net csresnext50-panet-spp_berkeley
test_net resnet101_cnet # test_net resnet101_cnet
test_net dla34_cnet # test_net dla34_cnet
test_net dla34_cnet3d # test_net dla34_cnet3d
test_net mobilenetv2ssd # test_net mobilenetv2ssd
test_net mobilenetv2ssd512 # test_net mobilenetv2ssd512
test_net bdd-mobilenetv2ssd # test_net bdd-mobilenetv2ssd
test_net dla34_ctrack # test_net dla34_ctrack
test_net shelfnet # test_net shelfnet
test_net shelfnet_berkeley # test_net shelfnet_berkeley
done done
echo "If errors occured, check logfile $out_file" echo "If errors occured, check logfiles in directory: $out_dir"
+9 -1
View File
@@ -17,6 +17,8 @@ bool CenterTrack::init(const std::string& tensor_path, const int n_classes, cons
init_pre_inf(); init_pre_inf();
init_postprocessing(); init_postprocessing();
init_visualization(n_classes); init_visualization(n_classes);
return true;
} }
bool CenterTrack::init_preprocessing(){ bool CenterTrack::init_preprocessing(){
@@ -59,6 +61,8 @@ bool CenterTrack::init_preprocessing(){
checkCuda( cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches)); checkCuda( cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches));
checkCuda( cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot())); checkCuda( cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot()));
checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) ); checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) );
return true;
} }
bool CenterTrack::init_pre_inf(){ bool CenterTrack::init_pre_inf(){
@@ -202,6 +206,8 @@ bool CenterTrack::init_postprocessing(){
trRes.resize(nBatches); trRes.resize(nBatches);
countTr.resize(nBatches, 0); countTr.resize(nBatches, 0);
trackId.resize(nBatches, 0); trackId.resize(nBatches, 0);
return true;
} }
bool CenterTrack::init_visualization(const int n_classes){ bool CenterTrack::init_visualization(const int n_classes){
@@ -274,6 +280,8 @@ bool CenterTrack::init_visualization(const int n_classes){
faceId.push_back({3,0,4,7}); faceId.push_back({3,0,4,7});
faceId.push_back({2,3,7,6}); faceId.push_back({2,3,7,6});
// ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]);
return true;
} }
void CenterTrack::_get_additional_inputs(){ void CenterTrack::_get_additional_inputs(){
@@ -308,7 +316,7 @@ void CenterTrack::preprocess(cv::Mat &frame, const int bi){
} }
float c[] = {new_width / 2.0f, new_height /2.0f}; float c[] = {new_width / 2.0f, new_height /2.0f};
float s[] = {dim.w, dim.h}; float s[] = {float(dim.w), float(dim.h)};
// float s = new_width >= new_height ? new_width : new_height; // float s = new_width >= new_height ? new_width : new_height;
// ----------- get_affine_transform // ----------- get_affine_transform
// rot_rad = pi * 0 / 100 --> 0 // rot_rad = pi * 0 / 100 --> 0
+1
View File
@@ -119,6 +119,7 @@ bool CenternetDetection::init(const std::string& tensor_path, const int n_classe
dst2.at<float>(2,0)=dst2.at<float>(1,0) + (-dst2.at<float>(0,1)+dst2.at<float>(1,1) ); dst2.at<float>(2,0)=dst2.at<float>(1,0) + (-dst2.at<float>(0,1)+dst2.at<float>(1,1) );
dst2.at<float>(2,1)=dst2.at<float>(1,1) + (dst2.at<float>(0,0)-dst2.at<float>(1,0) ); dst2.at<float>(2,1)=dst2.at<float>(1,1) + (dst2.at<float>(0,0)-dst2.at<float>(1,0) );
return true;
} }
+2
View File
@@ -167,6 +167,8 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas
faceId.push_back({2,3,7,6}); faceId.push_back({2,3,7,6});
faceId.push_back({3,0,4,7}); faceId.push_back({3,0,4,7});
// ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]); // ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]);
return true;
} }
void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi){ void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi){
+2 -2
View File
@@ -198,7 +198,7 @@ bool MobilenetDetection::init(const std::string& tensor_path, const int n_classe
"bottle" , "wine glass" , "cup" , "fork" , "knife" , "spoon" , "bowl" , "banana" , "bottle" , "wine glass" , "cup" , "fork" , "knife" , "spoon" , "bowl" , "banana" ,
"apple" , "sandwich" , "orange" , "broccoli" , "carrot" , "hot dog" , "pizza" , "apple" , "sandwich" , "orange" , "broccoli" , "carrot" , "hot dog" , "pizza" ,
"donut" , "cake" , "chair" , "sofa" , "pottedplant" , "bed" , "diningtable" , "donut" , "cake" , "chair" , "sofa" , "pottedplant" , "bed" , "diningtable" ,
"toilet" , "tvmonitor" , "laptop" , "mouse" , "remote" , "keyboard" , "toilet" , "tvmonitor" , "laptop" , "mouse" , "remote" , "keyboard" ,
"cell phone" , "microwave" , "oven" , "toaster" , "sink" , "refrigerator" , "cell phone" , "microwave" , "oven" , "toaster" , "sink" , "refrigerator" ,
"book" , "clock" , "vase" , "scissors" , "teddy bear" , "hair drier" , "toothbrush"}; "book" , "clock" , "vase" , "scissors" , "teddy bear" , "hair drier" , "toothbrush"};
classesNames = std::vector<std::string>(classes_names_, std::end(classes_names_)); classesNames = std::vector<std::string>(classes_names_, std::end(classes_names_));
@@ -207,7 +207,7 @@ bool MobilenetDetection::init(const std::string& tensor_path, const int n_classe
else{ else{
FatalError("Number of classes not supported for mobilenet"); FatalError("Number of classes not supported for mobilenet");
} }
return 1; return true;
} }
void MobilenetDetection::preprocess(cv::Mat &frame, const int bi){ void MobilenetDetection::preprocess(cv::Mat &frame, const int bi){
+3 -2
View File
@@ -278,6 +278,7 @@ void Yolo::mergeDetections(Yolo::detection *dets, int ndets, int classes, double
} }
total = k+1; total = k+1;
float thresh = 0.45f;
for(k = 0; k < classes; ++k){ for(k = 0; k < classes; ++k){
for(i = 0; i < total; ++i){ for(i = 0; i < total; ++i){
dets[i].sort_class = k; dets[i].sort_class = k;
@@ -288,9 +289,9 @@ void Yolo::mergeDetections(Yolo::detection *dets, int ndets, int classes, double
box a = dets[i].bbox; box a = dets[i].bbox;
for(j = i+1; j < total; ++j){ for(j = i+1; j < total; ++j){
box b = dets[j].bbox; box b = dets[j].bbox;
if (nsm_kind == GREEDY_NMS && yolo_box_iou(a, b) > nms_thresh) if (nsm_kind == GREEDY_NMS && yolo_box_iou(a, b) > thresh)
dets[j].prob[k] = 0; dets[j].prob[k] = 0;
else if (nsm_kind == DIOU_NMS && yolo_box_diou(a, b, nms_thresh) > nms_thresh) else if (nsm_kind == DIOU_NMS && yolo_box_diou(a, b, nms_thresh) > thresh)
dets[j].prob[k] = 0; dets[j].prob[k] = 0;
} }
} }
+9 -4
View File
@@ -46,11 +46,16 @@ void maxElem_kernel(float *src_begin, float *dst_begin, const int n_classes, con
if (i > size) if (i > size)
return; return;
thrust::device_ptr<float> dPbeg ( &src_begin[i*n_classes] ) ; float max = 0;
thrust::device_ptr<float> dPend = dPbeg + n_classes; int max_idx = 0;
thrust::device_ptr<float> result = thrust::max_element(thrust::device,dPbeg, dPend); for( int j = i*n_classes; j < i*n_classes + n_classes; ++j ){
if( src_begin[j] > max ){
max = src_begin[j];
max_idx = j;
}
}
dst_begin[i] = result - dPbeg; dst_begin[i] = max_idx - i*n_classes;
} }
void maxElem(dnnType *src_begin, dnnType *dst_begin, const int c, const int h, const int w){ void maxElem(dnnType *src_begin, dnnType *dst_begin, const int c, const int h, const int w){
File diff suppressed because it is too large Load Diff
+2
View File
@@ -0,0 +1,2 @@
person
head
+34
View File
@@ -0,0 +1,34 @@
#include<iostream>
#include<vector>
#include "tkdnn.h"
#include "test.h"
#include "DarknetParser.h"
int main() {
std::string bin_path = "yolo4-csp_crowd";
std::vector<std::string> input_bins = {
bin_path + "/layers/input.bin"
};
std::vector<std::string> output_bins = {
bin_path + "/debug/layer144_out.bin",
bin_path + "/debug/layer159_out.bin",
bin_path + "/debug/layer174_out.bin"
};
std::string wgs_path = bin_path + "/layers";
std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4-csp_crowd.cfg";
std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/crowdhuman.names";
downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/RKWfWNmWXfJigsK/download");
// parse darknet network
tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path);
net->print();
//convert network to tensorRT
tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str()));
int ret = testInference(input_bins, output_bins, net, netRT);
net->releaseLayers();
delete net;
delete netRT;
return ret;
}
+1 -1
View File
@@ -17,7 +17,7 @@ int main() {
std::string wgs_path = bin_path + "/layers"; std::string wgs_path = bin_path + "/layers";
std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4_berkeley.cfg"; std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4_berkeley.cfg";
std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/berkeley.names"; std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/berkeley.names";
downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/M7WJdGoGDaDACnN/download"); downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/q9dwoqQ5YQqEi7s/download");
// parse darknet network // parse darknet network
tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path);
+295
View File
@@ -0,0 +1,295 @@
#include <iostream>
#include <opencv2/highgui/highgui.hpp>
#include <opencv2/imgproc/imgproc.hpp>
#include "tkdnn.h"
#include "NetworkViz.h"
const char *input_bin = "shelfnet_coco/debug/input.bin";
const char *backbone[] = {
"shelfnet_coco/layers/backbone-conv1.bin",
"shelfnet_coco/layers/backbone-layer1-0-conv1.bin",
"shelfnet_coco/layers/backbone-layer1-0-conv2.bin",
"shelfnet_coco/layers/backbone-layer1-1-conv1.bin",
"shelfnet_coco/layers/backbone-layer1-1-conv2.bin",
"shelfnet_coco/layers/backbone-layer2-0-conv1.bin",
"shelfnet_coco/layers/backbone-layer2-0-conv2.bin",
"shelfnet_coco/layers/backbone-layer2-0-downsample-0.bin",
"shelfnet_coco/layers/backbone-layer2-1-conv1.bin",
"shelfnet_coco/layers/backbone-layer2-1-conv2.bin",
"shelfnet_coco/layers/backbone-layer3-0-conv1.bin",
"shelfnet_coco/layers/backbone-layer3-0-conv2.bin",
"shelfnet_coco/layers/backbone-layer3-0-downsample-0.bin",
"shelfnet_coco/layers/backbone-layer3-1-conv1.bin",
"shelfnet_coco/layers/backbone-layer3-1-conv2.bin",
"shelfnet_coco/layers/backbone-layer4-0-conv1.bin",
"shelfnet_coco/layers/backbone-layer4-0-conv2.bin",
"shelfnet_coco/layers/backbone-layer4-0-downsample-0.bin",
"shelfnet_coco/layers/backbone-layer4-1-conv1.bin",
"shelfnet_coco/layers/backbone-layer4-1-conv2.bin"};
const char *conv_out[] = {
"shelfnet_coco/layers/conv_out-conv-conv.bin",
"shelfnet_coco/layers/conv_out-conv_out.bin",
"shelfnet_coco/layers/conv_out16-conv-conv.bin",
"shelfnet_coco/layers/conv_out16-conv_out.bin",
"shelfnet_coco/layers/conv_out32-conv-conv.bin",
"shelfnet_coco/layers/conv_out32-conv_out.bin"
};
const char *decoder[] = {
"shelfnet_coco/layers/decoder-bottom-conv1.bin",
"shelfnet_coco/layers/decoder-bottom-conv12.bin",
"shelfnet_coco/layers/decoder-up_conv_list-0-conv-conv.bin",
"shelfnet_coco/layers/decoder-up_conv_list-0-conv_atten.bin",
"shelfnet_coco/layers/decoder-up_dense_list-0-conv.bin",
"shelfnet_coco/layers/decoder-up_conv_list-1-conv-conv.bin",
"shelfnet_coco/layers/decoder-up_conv_list-1-conv_atten.bin",
"shelfnet_coco/layers/decoder-up_dense_list-1-conv.bin"
};
const char *ladder[] = {
"shelfnet_coco/layers/ladder-inconv-conv1.bin",
"shelfnet_coco/layers/ladder-inconv-conv12.bin",
"shelfnet_coco/layers/ladder-down_module_list-0-conv1.bin",
"shelfnet_coco/layers/ladder-down_module_list-0-conv12.bin",
"shelfnet_coco/layers/ladder-down_conv_list-0.bin",
"shelfnet_coco/layers/ladder-down_module_list-1-conv1.bin",
"shelfnet_coco/layers/ladder-down_module_list-1-conv12.bin",
"shelfnet_coco/layers/ladder-down_conv_list-1.bin",
"shelfnet_coco/layers/ladder-bottom-conv1.bin",
"shelfnet_coco/layers/ladder-bottom-conv12.bin",
"shelfnet_coco/layers/ladder-up_conv_list-0-conv-conv.bin",
"shelfnet_coco/layers/ladder-up_conv_list-0-conv_atten.bin",
"shelfnet_coco/layers/ladder-up_dense_list-0-conv.bin",
"shelfnet_coco/layers/ladder-up_conv_list-1-conv-conv.bin",
"shelfnet_coco/layers/ladder-up_conv_list-1-conv_atten.bin",
"shelfnet_coco/layers/ladder-up_dense_list-1-conv.bin"};
const char *trans[] = {
"shelfnet_coco/layers/trans1-conv.bin",
"shelfnet_coco/layers/trans2-conv.bin",
"shelfnet_coco/layers/trans3-conv.bin"};
int main()
{
downloadWeightsifDoNotExist(input_bin, "shelfnet_coco", "https://cloud.hipert.unimore.it/s/KfQ9fGJQsgzNbiW/download");
int classes = 183;
// Network layout
tk::dnn::dataDim_t dim(1, 3, 1024, 1024, 1);
tk::dnn::Network net(dim);
int bi = 0, di = 0, li = 0, ci = 0;
new tk::dnn::Conv2d(&net, 64, 7, 7, 2, 2, 3, 3, backbone[bi++], true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
tk::dnn::Layer* last = new tk::dnn::Pooling (&net, 3, 3, 2, 2, 1, 1, tk::dnn::POOLING_MAX);
for(int i=0; i<2; ++i){
new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true);
new tk::dnn::Shortcut(&net, last);
last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
}
std::vector<tk::dnn::Layer*> features;
for(int i=0;i<3;++i){
int out_channel = pow(2,7+i);
std::cout<<out_channel<<std::endl;
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 2, 2, 1, 1, backbone[bi++], true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
tk::dnn::Layer* bn2 = new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, backbone[bi++], true);
new tk::dnn::Route(&net, &last, 1);
new tk::dnn::Conv2d (&net, out_channel, 1, 1, 2, 2, 0, 0, backbone[bi++], true);
new tk::dnn::Shortcut(&net, bn2);
last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, backbone[bi++], true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, backbone[bi++], true);
new tk::dnn::Shortcut(&net, last);
last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
features.push_back(last);
}
for(int i=0; i<features.size(); ++i){
new tk::dnn::Route(&net, &features[i], 1);
int out_channel = pow(2,6+i);
new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, trans[i], true);
features[i] = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
}
//DECODER
last = features[2];
std::vector<tk::dnn::Layer*> up_out;
//bottom
new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, decoder[di++], true, false, 1, true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, decoder[di++], true, false, 1, true);
new tk::dnn::Shortcut(&net, last);
last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
up_out.push_back(last);
for(int i=0; i<2; ++i){
int out_channel = pow(2,7-i);
//up-conv
std::cout<<out_channel<<std::endl;
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, decoder[di++], true);
last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Pooling(&net, last->output_dim.w, last->output_dim.h, last->output_dim.w, last->output_dim.h, 0, 0, tk::dnn::POOLING_AVERAGE);
new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, decoder[di++], true);
tk::dnn::Layer* act = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_SIGMOID);
new tk::dnn::Route(&net, &last, 1);
new tk::dnn::Shortcut(&net, act, true);
//interpolate
new tk::dnn::Resize(&net, 1,2,2);
new tk::dnn::Shortcut(&net, features[1-i]);
//up-dense
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, decoder[di++], true);
last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
up_out.push_back(last);
}
//LADDER
std::vector<tk::dnn::Layer*> down_out;
new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
new tk::dnn::Shortcut(&net, last);
new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
for(int i=0; i<2;++i){
int out_channel = pow(2,6+i);
tk::dnn::Layer* l_last = new tk::dnn::Shortcut(&net, up_out[2-i]);
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
new tk::dnn::Shortcut(&net, l_last);
l_last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
down_out.push_back(l_last);
new tk::dnn::Conv2d (&net, out_channel*2, 3, 3, 2, 2, 1, 1, ladder[li++], false);
last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.0f); //should be ReLU
}
new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
new tk::dnn::Shortcut(&net, last);
last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
up_out.clear();
up_out.push_back(last);
for(int i=0; i<2; ++i){
int out_channel = pow(2,7-i);
//up-conv
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true);
last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Pooling(&net, last->output_dim.w, last->output_dim.h, last->output_dim.w, last->output_dim.h, 0, 0, tk::dnn::POOLING_AVERAGE);
new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, ladder[li++], true);
tk::dnn::Layer* act = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_SIGMOID);
new tk::dnn::Route(&net, &last, 1);
new tk::dnn::Shortcut(&net, act, true);
//interpolate
new tk::dnn::Resize(&net, 1,2,2);
new tk::dnn::Shortcut(&net, down_out[1-i]);
// //up-dense
new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true);
last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
up_out.push_back(last);
}
// for(int i=2;i>=0;--i){
// new tk::dnn::Route(&net, &up_out[i], 1);
new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, conv_out[ci++], true);
new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
new tk::dnn::Conv2d (&net, classes, 3, 3, 1, 1, 1, 1, conv_out[ci++], false);
/*up_out[i] =*/ new tk::dnn::Resize(&net, classes, net.input_dim.h, net.input_dim.w, true, tk::dnn::ResizeMode_t::LINEAR);
// // }
new tk::dnn::Softmax(&net);
const char *output_bin = "shelfnet_coco/debug/softmax.bin";
// Load input
dnnType *data;
dnnType *input_h;
readBinaryFile(input_bin, dim.tot(), &input_h, &data);
std::cout<<"Input:"<<std::endl;
//print network model
net.print();
// // convert network to tensorRT
tk::dnn::NetworkRT netRT(&net, net.getNetworkRTName("shelfnet_coco"));
tk::dnn::dataDim_t dim1 = dim; //input dim
dnnType *cudnn_out = nullptr;
printCenteredTitle(" CUDNN inference ", '=', 30);
{
dim1.print();
TKDNN_TSTART
cudnn_out = net.infer(dim1, data);
TKDNN_TSTOP
dim1.print();
}
tk::dnn::dataDim_t dim2 = dim;
printCenteredTitle(" TENSORRT inference ", '=', 30);
{
dim2.print();
TKDNN_TSTART
netRT.infer(dim2, data);
TKDNN_TSTOP
dim2.print();
}
dnnType *rt_out1 = (dnnType *)netRT.buffersRT[1];
printCenteredTitle(std::string(" CHECK RESULTS ").c_str(), '=', 30);
dnnType *out1, *out1_h;
int odim1 = dim1.tot();
readBinaryFile(output_bin, odim1, &out1_h, &out1);
int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0;
// std::cout << "CUDNN vs correct" << std::endl;
// ret_cudnn |= checkResult(odim1, cudnn_out, out1, true, 20) == 0 ? 0 : ERROR_CUDNN;
std::cout << "TRT vs correct" << std::endl;
ret_tensorrt |=checkResult(odim1, rt_out1, out1) == 0 ? 0 : ERROR_TENSORRT;
std::cout << "CUDNN vs TRT " << std::endl;
ret_cudnn_tensorrt |= checkResult(odim1, cudnn_out, rt_out1) == 0 ? 0 : ERROR_CUDNNvsTENSORRT;
cv::Mat viz = vizLayer2Mat(&net, net.num_layers-1);
cv::imwrite("test.png", viz);
return ret_cudnn | ret_tensorrt | ret_cudnn_tensorrt;
}
+4 -4
View File
@@ -1,6 +1,6 @@
#include<iostream> #include<iostream>
#include<algorithm> #include<algorithm>
#include "tkdnn.h" #include "tkDNN/tkdnn.h"
#include <stdlib.h> /* srand, rand */ #include <stdlib.h> /* srand, rand */
@@ -66,11 +66,11 @@ int main(int argc, char *argv[]) {
} }
} }
double min = *std::min_element(stats.begin(), stats.end())/BATCH_SIZE; double min = *std::min_element(stats.begin(), stats.end()); ///BATCH_SIZE;
double max = *std::max_element(stats.begin(), stats.end())/BATCH_SIZE; double max = *std::max_element(stats.begin(), stats.end()); ///BATCH_SIZE;
double mean =0; double mean =0;
for(int i=0; i<stats.size(); i++) mean += stats[i]; mean /= stats.size(); for(int i=0; i<stats.size(); i++) mean += stats[i]; mean /= stats.size();
mean /=BATCH_SIZE; //mean /=BATCH_SIZE;
std::cout<<"Min: "<<min<<" ms\n"; std::cout<<"Min: "<<min<<" ms\n";
std::cout<<"Max: "<<max<<" ms\n"; std::cout<<"Max: "<<max<<" ms\n";