+1
-1
@@ -13,4 +13,4 @@ build/
|
|||||||
*.pk
|
*.pk
|
||||||
*.table
|
*.table
|
||||||
demo/COCO_val2017
|
demo/COCO_val2017
|
||||||
demo/BDD100k_val
|
demo/BDD100K_val
|
||||||
@@ -20,6 +20,8 @@ SET(CUDA_SEPARABLE_COMPILATION ON)
|
|||||||
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32)
|
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32)
|
||||||
|
|
||||||
find_package(CUDNN REQUIRED)
|
find_package(CUDNN REQUIRED)
|
||||||
|
include_directories(${CUDNN_INCLUDE_DIR})
|
||||||
|
|
||||||
|
|
||||||
# compile
|
# compile
|
||||||
file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu")
|
file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu")
|
||||||
|
|||||||
@@ -2,9 +2,44 @@
|
|||||||
tkDNN is a Deep Neural Network library built with cuDNN and tensorRT primitives, specifically thought to work on NVIDIA Jetson Boards. It has been tested on TK1(branch cudnn2), TX1, TX2, AGX Xavier and several discrete GPU.
|
tkDNN is a Deep Neural Network library built with cuDNN and tensorRT primitives, specifically thought to work on NVIDIA Jetson Boards. It has been tested on TK1(branch cudnn2), TX1, TX2, AGX Xavier and several discrete GPU.
|
||||||
The main goal of this project is to exploit NVIDIA boards as much as possible to obtain the best inference performance. It does not allow training.
|
The main goal of this project is to exploit NVIDIA boards as much as possible to obtain the best inference performance. It does not allow training.
|
||||||
|
|
||||||
Accepted paper @ IRC 2020, will soon been published.
|
|
||||||
|
If you use tkDNN in your research, please cite one of the following papers. For use in commercial solutions, write at gattifrancesco@hotmail.it or refer to https://hipert.unimore.it/ .
|
||||||
|
|
||||||
|
```
|
||||||
|
Accepted paper @ IRC 2020, will soon be published.
|
||||||
M. Verucchi, L. Bartoli, F. Bagni, F. Gatti, P. Burgio and M. Bertogna, "Real-Time clustering and LiDAR-camera fusion on embedded platforms for self-driving cars", in proceedings in IEEE Robotic Computing (2020)
|
M. Verucchi, L. Bartoli, F. Bagni, F. Gatti, P. Burgio and M. Bertogna, "Real-Time clustering and LiDAR-camera fusion on embedded platforms for self-driving cars", in proceedings in IEEE Robotic Computing (2020)
|
||||||
|
|
||||||
|
Accepted paper @ ETFA 2020, will soon be published.
|
||||||
|
M. Verucchi, G. Brilli, D. Sapienza, M. Verasani, M. Arena, F. Gatti, A. Capotondi, R. Cavicchioli, M. Bertogna, M. Solieri
|
||||||
|
"A Systematic Assessment of Embedded Neural Networks for Object Detection", in IEEE International Conference on Emerging Technologies and Factory Automation (2020)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Results
|
||||||
|
Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimesion as the input size, on
|
||||||
|
* RTX 2080Ti (CUDA 10.2, TensorRT 7.0.0, Cudnn 7.6.5);
|
||||||
|
* Xavier AGX, Jetpack 4.3 (CUDA 10.0, CUDNN 7.6.3, tensorrt 6.0.1 );
|
||||||
|
* Tx2, Jetpack 4.2 (CUDA 10.0, CUDNN 7.3.1, tensorrt 5.0.6 );
|
||||||
|
* Jetson Nano, Jetpack 4.4 (CUDA 10.2, CUDNN 8.0.0, tensorrt 7.1.0 ).
|
||||||
|
|
||||||
|
| Platform | Network | FP32, B=1 | FP32, B=4 | FP16, B=1 | FP16, B=4 | INT8, B=1 | INT8, B=4 |
|
||||||
|
| :------: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: |
|
||||||
|
| RTX 2080Ti | yolo4 320 | 118,59 |237,31 | 207,81 | 443,32 | 262,37 | 530,93 |
|
||||||
|
| RTX 2080Ti | yolo4 416 | 104,81 |162,86 | 169,06 | 293,78 | 206,93 | 353,26 |
|
||||||
|
| RTX 2080Ti | yolo4 512 | 92,98 |132,43 | 140,36 | 215,17 | 165,35 | 254,96 |
|
||||||
|
| RTX 2080Ti | yolo4 608 | 63,77 |81,53 | 111,39 | 152,89 | 127,79 | 184,72 |
|
||||||
|
| AGX Xavier | yolo4 320 | 26,78 |32,05 | 57,14 | 79,05 | 73,15 | 97,56 |
|
||||||
|
| AGX Xavier | yolo4 416 | 19,96 |21,52 | 41,01 | 49,00 | 50,81 | 60,61 |
|
||||||
|
| AGX Xavier | yolo4 512 | 16,58 |16,98 | 31,12 | 33,84 | 37,82 | 41,28 |
|
||||||
|
| AGX Xavier | yolo4 608 | 9,45 |10,13 | 21,92 | 23,36 | 27,05 | 28,93 |
|
||||||
|
| Tx2 | yolo4 320 | 11,18 | 12,07 | 15,32 | 16,31 | - | - |
|
||||||
|
| Tx2 | yolo4 416 | 7,30 | 7,58 | 9,45 | 9,90 | - | - |
|
||||||
|
| Tx2 | yolo4 512 | 5,96 | 5,95 | 7,22 | 7,23 | - | - |
|
||||||
|
| Tx2 | yolo4 608 | 3,63 | 3,65 | 4,67 | 4,70 | - | - |
|
||||||
|
| Nano | yolo4 320 | 4,23 | 4,55 | 6,14 | 6,53 | - | - |
|
||||||
|
| Nano | yolo4 416 | 2,88 | 3,00 | 3,90 | 4,04 | - | - |
|
||||||
|
| Nano | yolo4 512 | 2,32 | 2,34 | 3,02 | 3,04 | - | - |
|
||||||
|
| Nano | yolo4 608 | 1,40 | 1,41 | 1,92 | 1,93 | - | - |
|
||||||
|
|
||||||
## Index
|
## Index
|
||||||
- [tkDNN](#tkdnn)
|
- [tkDNN](#tkdnn)
|
||||||
- [Index](#index)
|
- [Index](#index)
|
||||||
|
|||||||
+61
-28
@@ -1,33 +1,66 @@
|
|||||||
# Find the header files
|
# find the library
|
||||||
|
if(CUDA_FOUND)
|
||||||
|
find_cuda_helper_libs(cudnn)
|
||||||
|
set(CUDNN_LIBRARY ${CUDA_cudnn_LIBRARY} CACHE FILEPATH "location of the cuDNN library")
|
||||||
|
unset(CUDA_cudnn_LIBRARY CACHE)
|
||||||
|
|
||||||
find_path(CUDNN_INCLUDE_DIR
|
find_cuda_helper_libs(nvinfer)
|
||||||
${CMAKE_SYSROOT}/usr/local/include
|
set(NVINFER_LIBRARY ${CUDA_nvinfer_LIBRARY} CACHE FILEPATH "location of the nvinfer library")
|
||||||
${CMAKE_SYSROOT}/usr/include
|
unset(CUDA_nvinfer_LIBRARY CACHE)
|
||||||
/usr/local/nvidia/tensorrt/include/
|
endif()
|
||||||
NO_DEFAULT_PATH
|
|
||||||
)
|
|
||||||
|
|
||||||
set(OLD_ROOT ${CMAKE_FIND_ROOT_PATH})
|
# find the include
|
||||||
list(APPEND CMAKE_FIND_ROOT_PATH /)
|
if(CUDNN_LIBRARY)
|
||||||
list(APPEND CMAKE_FIND_LIBRARY_SUFFIXES .so.7)
|
find_path(CUDNN_INCLUDE_DIR
|
||||||
list(APPEND CMAKE_FIND_LIBRARY_SUFFIXES .so.5)
|
cudnn.h
|
||||||
find_library(CUDNN_LIB
|
PATHS ${CUDA_TOOLKIT_INCLUDE}
|
||||||
NAMES cudnn
|
DOC "location of cudnn.h"
|
||||||
PATHS
|
|
||||||
/usr/local/driveworks/targets/${CMAKE_SYSTEM_PROCESSOR}-Linux/lib
|
|
||||||
/usr/lib/${CMAKE_SYSTEM_PROCESSOR}-linux-gnu/
|
|
||||||
NO_DEFAULT_PATH
|
NO_DEFAULT_PATH
|
||||||
)
|
)
|
||||||
find_library(CUDNN_NVLIB
|
|
||||||
NAMES "nvinfer"
|
if(NOT CUDNN_INCLUDE_DIR)
|
||||||
PATHS
|
find_path(CUDNN_INCLUDE_DIR
|
||||||
/usr/local/driveworks/targets/${CMAKE_SYSTEM_PROCESSOR}-Linux/lib
|
cudnn.h
|
||||||
/usr/lib/${CMAKE_SYSTEM_PROCESSOR}-linux-gnu/
|
DOC "location of cudnn.h"
|
||||||
NO_DEFAULT_PATH
|
)
|
||||||
)
|
endif()
|
||||||
set(CMAKE_FIND_ROOT_PATH ${OLD_ROOT})
|
|
||||||
|
message("-- Found CUDNN: " ${CUDNN_LIBRARY})
|
||||||
|
message("-- Found CUDNN include: " ${CUDNN_INCLUDE_DIR})
|
||||||
|
endif()
|
||||||
|
|
||||||
|
if(NVINFER_LIBRARY)
|
||||||
|
find_path(NVINFER_INCLUDE_DIR
|
||||||
|
NvInfer.h
|
||||||
|
PATHS ${CUDA_TOOLKIT_INCLUDE}
|
||||||
|
DOC "location of NvInfer.h"
|
||||||
|
NO_DEFAULT_PATH
|
||||||
|
)
|
||||||
|
|
||||||
|
if(NOT NVINFER_INCLUDE_DIR)
|
||||||
|
find_path(NVINFER_INCLUDE_DIR
|
||||||
|
NvInfer.h
|
||||||
|
DOC "location of NvInfer.h"
|
||||||
|
)
|
||||||
|
endif()
|
||||||
|
|
||||||
|
message("-- Found NVINFER: " ${NVINFER_LIBRARY})
|
||||||
|
message("-- Found NVINFER include: " ${NVINFER_INCLUDE_DIR})
|
||||||
|
endif()
|
||||||
|
|
||||||
|
|
||||||
|
include(FindPackageHandleStandardArgs)
|
||||||
|
find_package_handle_standard_args(CUDNN
|
||||||
|
FOUND_VAR CUDNN_FOUND
|
||||||
|
REQUIRED_VARS
|
||||||
|
CUDNN_LIBRARY
|
||||||
|
CUDNN_INCLUDE_DIR
|
||||||
|
VERSION_VAR CUDNN_VERSION
|
||||||
|
)
|
||||||
|
|
||||||
|
if(CUDNN_FOUND)
|
||||||
|
set(CUDNN_LIBRARIES ${CUDNN_LIBRARY} ${NVINFER_LIBRARY})
|
||||||
|
set(CUDNN_INCLUDE_DIRS ${CUDNN_INCLUDE_DIR} ${NVINFER_INCLUDE_DIR})
|
||||||
|
endif()
|
||||||
|
|
||||||
set(CUDNN_LIBRARIES ${CUDNN_LIB} ${CUDNN_NVLIB})
|
|
||||||
message("-- Found CUDNN: " ${CUDNN_LIB})
|
|
||||||
message("-- Found NVINFER: " ${CUDNN_NVLIB})
|
|
||||||
set(CUDNN_FOUND true)
|
set(CUDNN_FOUND true)
|
||||||
+2
-2
@@ -153,7 +153,7 @@ int main(int argc, char *argv[])
|
|||||||
|
|
||||||
std::ofstream myfile;
|
std::ofstream myfile;
|
||||||
if(write_dets)
|
if(write_dets)
|
||||||
myfile.open ("det/"+f.lFilename.substr(f.lFilename.find("000")));
|
myfile.open ("det/"+f.lFilename.substr(f.lFilename.find("labels/") + 7));
|
||||||
|
|
||||||
// save detections labels
|
// save detections labels
|
||||||
for(auto d:detected_bbox){
|
for(auto d:detected_bbox){
|
||||||
@@ -169,7 +169,7 @@ int main(int argc, char *argv[])
|
|||||||
f.det.push_back(b);
|
f.det.push_back(b);
|
||||||
|
|
||||||
if(write_dets)
|
if(write_dets)
|
||||||
myfile << d.cl << " "<< d.prob << " "<< d.x << " "<< d.y << " "<< d.w << " "<< d.h <<"\n";
|
myfile << d.cl << " "<< d.prob << " "<< b.x << " "<< b.y << " "<< b.w << " "<< b.h <<"\n";
|
||||||
|
|
||||||
if(show)// draw rectangle for detection
|
if(show)// draw rectangle for detection
|
||||||
cv::rectangle(batch_frames[0], cv::Point(d.x, d.y), cv::Point(d.x + d.w, d.y + d.h), cv::Scalar(0, 0, 255), 2);
|
cv::rectangle(batch_frames[0], cv::Point(d.x, d.y), cv::Point(d.x + d.w, d.y + d.h), cv::Scalar(0, 0, 255), 2);
|
||||||
|
|||||||
@@ -62,5 +62,5 @@ make -j4
|
|||||||
sudo make install
|
sudo make install
|
||||||
sudo ldconfig
|
sudo ldconfig
|
||||||
|
|
||||||
cd '~/Downloads/opencv4/lib/python3.6/site-packages'
|
cd ~/Downloads/opencv4/lib/python3.6/site-packages
|
||||||
ln -s /usr/local/lib/python3.6/site-packages/cv2.cpython-36m-aarch64-linux-gnu.so cv2.so
|
ln -s /usr/local/lib/python3.6/site-packages/cv2.cpython-36m-aarch64-linux-gnu.so cv2.so
|
||||||
|
|||||||
@@ -86,7 +86,11 @@ LSTM::LSTM( Network *net, int hiddensize, bool returnSeq, std::string fname_weig
|
|||||||
// RNN descriptors
|
// RNN descriptors
|
||||||
checkCUDNN(cudnnCreateRNNDescriptor(&rnnDesc));
|
checkCUDNN(cudnnCreateRNNDescriptor(&rnnDesc));
|
||||||
|
|
||||||
|
#if CUDNN_MAJOR > 7
|
||||||
|
checkCUDNN(cudnnSetRNNDescriptor_v6(net->cudnnHandle,
|
||||||
|
#else
|
||||||
checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle,
|
checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle,
|
||||||
|
#endif
|
||||||
rnnDesc, stateSize, numLayers, dropoutDesc,
|
rnnDesc, stateSize, numLayers, dropoutDesc,
|
||||||
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
|
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
|
||||||
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
|
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
|
||||||
|
|||||||
+1
-1
@@ -595,7 +595,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, DeformConv2d *l) {
|
|||||||
|
|
||||||
bool NetworkRT::serialize(const char *filename) {
|
bool NetworkRT::serialize(const char *filename) {
|
||||||
|
|
||||||
std::ofstream p(filename);
|
std::ofstream p(filename, std::ios::binary);
|
||||||
if (!p) {
|
if (!p) {
|
||||||
FatalError("could not open plan output file");
|
FatalError("could not open plan output file");
|
||||||
return false;
|
return false;
|
||||||
|
|||||||
Reference in New Issue
Block a user