Merge branch 'perseusdg-master'

This commit is contained in:
Francesco Gatti
2021-04-24 18:55:44 +02:00
36 changed files with 421 additions and 149 deletions
+7 -1
View File
@@ -12,5 +12,11 @@ build/
*.hdf5 *.hdf5
*.pk *.pk
*.table *.table
cmake-build-release/
demo/COCO_val2017 demo/COCO_val2017
demo/BDD100K_val demo/BDD100K_val
/.vs
cmake-build-minsizerel/*
scripts/COCO_val2017/*
scripts/COCO_val2017.zip
scripts/all_labels.txt
+12 -4
View File
@@ -2,7 +2,14 @@ cmake_minimum_required(VERSION 3.5)
project (tkDNN) project (tkDNN)
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable") if(UNIX)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable ")
endif()
if(WIN32)
set(CMAKE_CXX_STANDARD 11)
set(CMAKE_CXX_FLAGS "/O2 /FS /EHsc")
set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON)
endif(WIN32)
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN) include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN)
# project specific flags # project specific flags
@@ -18,7 +25,7 @@ add_definitions(-DTKDNN_PATH="${CMAKE_CURRENT_SOURCE_DIR}")
find_package(CUDA 9.0 REQUIRED) find_package(CUDA 9.0 REQUIRED)
SET(CUDA_SEPARABLE_COMPILATION ON) SET(CUDA_SEPARABLE_COMPILATION ON)
#set(CUDA_NVCC_FLAGS "${CUDA_NVCC_FLAGS} -arch=sm_30 --compiler-options '-fPIC'") #set(CUDA_NVCC_FLAGS "${CUDA_NVCC_FLAGS} -arch=sm_30 --compiler-options '-fPIC'")
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32) set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32 -arch=sm_61 )
find_package(CUDNN REQUIRED) find_package(CUDNN REQUIRED)
include_directories(${CUDNN_INCLUDE_DIR}) include_directories(${CUDNN_INCLUDE_DIR})
@@ -28,6 +35,7 @@ include_directories(${CUDNN_INCLUDE_DIR})
file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu") file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu")
cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${CUDNN_INCLUDE_DIRS}) cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${CUDNN_INCLUDE_DIRS})
cuda_add_library(kernels SHARED ${tkdnn_CUSRC}) cuda_add_library(kernels SHARED ${tkdnn_CUSRC})
target_link_libraries(kernels ${CUDA_CUBLAS_LIBRARIES})
#------------------------------------------------------------------------------- #-------------------------------------------------------------------------------
@@ -40,7 +48,7 @@ find_package(OpenCV REQUIRED)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV") set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV")
# gives problems in cross-compiling, probably malformed cmake config # gives problems in cross-compiling, probably malformed cmake config
#find_package(yaml-cpp REQUIRED) find_package(yaml-cpp REQUIRED)
#------------------------------------------------------------------------------- #-------------------------------------------------------------------------------
# Build Libraries # Build Libraries
@@ -48,7 +56,7 @@ set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV")
file(GLOB tkdnn_SRC "src/*.cpp") file(GLOB tkdnn_SRC "src/*.cpp")
set(tkdnn_LIBS kernels ${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDNN_LIBRARIES} ${OpenCV_LIBS} yaml-cpp) set(tkdnn_LIBS kernels ${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDNN_LIBRARIES} ${OpenCV_LIBS} yaml-cpp)
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11") set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS}")
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${OPENCV_INCLUDE_DIRS} ${NVINFER_INCLUDES}) include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${OPENCV_INCLUDE_DIRS} ${NVINFER_INCLUDES})
add_library(tkDNN SHARED ${tkdnn_SRC}) add_library(tkDNN SHARED ${tkdnn_SRC})
target_link_libraries(tkDNN ${tkdnn_LIBS}) target_link_libraries(tkDNN ${tkdnn_LIBS})
+1
View File
@@ -0,0 +1 @@
1)error C2131 @ Yolo3Detection.cpp(97) -> expression doesnt evaluate to a constant caused to read of variable outside its lifetime
+96
View File
@@ -80,6 +80,14 @@ Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001
- [mAP demo](#map-demo) - [mAP demo](#map-demo)
- [Existing tests and supported networks](#existing-tests-and-supported-networks) - [Existing tests and supported networks](#existing-tests-and-supported-networks)
- [References](#references) - [References](#references)
- [tkDNN on Windows 10 (experimental)](#tkdnn-on-windows-10-experimental)
- [Dependencies-Windows](#dependencies-windows)
- [Compiling tkDNN on Windows](#compiling-tkdnn-on-windows)
- [Run the demo on Windows](#run-the-demo-on-windows)
- [FP16 inference windows](#fp16-inference-windows)
- [INT8 inference windows](#int8-inference-windows)
- [Known issues with tkDNN on Windows](#known-issues-with-tkdnn-on-windows)
@@ -356,6 +364,94 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing
| yolo4x | Yolov4x-mish <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) | | yolo4x | Yolov4x-mish <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) |
| yolo4x-cps | Scaled Yolov4 <sup>10</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) | | yolo4x-cps | Scaled Yolov4 <sup>10</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) |
### tkDNN on Windows 10 (experimental)
### Dependencies-Windows
This branch should work on every NVIDIA GPU supported in windows with the following dependencies:
* WINDOWS 10 1803 or HIGHER
* CUDA 10.0 (Recommended CUDA 11.2 )
* CUDNN 7.6 (Recommended CUDNN 8.1.1 )
* TENSORRT 6.0.1 (Recommended TENSORRT 7.2.3.4 )
* OPENCV 3.4 (Recommended OPENCV 4.2.0 )
* MSVC 16.7
* YAML-CPP
* EIGEN3
* 7ZIP (ADD TO PATH)
* NINJA 1.10
All the above mentioned dependencies except 7ZIP can be installed using Microsoft's [VCPKG](https://github.com/microsoft/vcpkg.git) .
After bootstrapping VCPKG the dependencies can be built and installed using the following command :
```
opencv4(normal) - vcpkg.exe install opencv4[tbb,jpeg,tiff,opengl,openmp,png,ffmpeg,eigen]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build
opencv4(cuda) - vcpkg.exe install opencv4[cuda,nonfree,contrib,eigen,tbb,jpeg,tiff,opengl,openmp,png,ffmpeg]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build
```
To build opencv4 with cuda and cudnn version corresponding to your cuda version,vcpkg's cudnn portfile needs to be modified by adding ```$ENV{CUDA_PATH}``` at lines 16 and 17 in the portfile.cmake
After VCPKG finishes building and installing all the packages delete C:\temp_vcpkg_build and add C:\opt\x64-windows\bin and C:\opt\x64-windows\debug\bin to path
### Compiling tkDNN on Windows
tkDNN is built with cmake(3.15+) on windows along with ninja.Msbuild and NMake Makefiles are drastically slower when compiling the library compared to windows
```
git clone https://github.com/ceccocats/tkDNN.git
cd tkdnn-windows
mkdir build
cd build
cmake -DCMAKE_BUILD_TYPE=Release -G"Ninja" ..
ninja -j4
```
### Run the demo on Windows
This example uses yolo4_tiny.\
To run the object detection file create .rt file bu running:
```
.\test_yolo4tiny.exe
```
Once the rt file has been successfully create,run the demo using the following command:
```
.\demo.exe yolo4tiny_fp32.rt ..\demo\yolo_test.mp4 y
```
For general info on more demo paramters,check Run the demo section on top
To run the test_all_tests.sh on windows,use git bash or msys2
### FP16 inference windows
This is an untested feature on windows.To run the object detection demo with FP16 interference follow the below steps(example with yolo4tiny):
```
set TKDNN_MODE=FP16
del /f yolo4tiny_fp16.rt
.\test_yolo4tiny.exe
.\demo.exe yolo4tiny_fp16.rt ..\demo\yolo_test.mp4
```
### INT8 inference windows
To run object detection demo with INT8 (example with yolo4tiny):
```
set TKDNN_MODE=INT8
set TKDNN_CALIB_LABEL_PATH=..\demo\COCO_val2017\all_labels.txt
set TKDNN_CALIB_IMG_PATH=..\demo\COCO_val2017\all_images.txt
del /f yolo4tiny_int8.rt # be sure to delete(or move) old tensorRT files
.\test_yolo4tiny.exe # run the yolo test (is slow)
.\demo.exe yolo4tiny_int8.rt ..\demo\yolo_test.mp4 y
```
### Known issues with tkDNN on Windows
Mobilenet and Centernet demos work properly only when built with msvc 16.7 in Release Mode,when built in debug mode for the mentioned networks one might encounter opencv assert errors
All Darknet models work properly with demo using MSVC version(16.7-16.9)
It is recommended to use Nvidia Driver(465+),Cuda unknown errors have been observed when using older drivers on pascal(SM 61) devices.
## References ## References
+9 -4
View File
@@ -1,7 +1,7 @@
#include <iostream> #include <iostream>
#include <signal.h> #include <signal.h>
#include <stdlib.h> /* srand, rand */ #include <stdlib.h> /* srand, rand */
#include <unistd.h> //#include <unistd.h>
#include <mutex> #include <mutex>
#include "CenternetDetection.h" #include "CenternetDetection.h"
@@ -22,10 +22,15 @@ int main(int argc, char *argv[]) {
signal(SIGINT, sig_handler); signal(SIGINT, sig_handler);
std::string net = "yolo3_berkeley.rt"; std::string net = "yolo4tiny_fp32.rt";
if(argc > 1) if(argc > 1)
net = argv[1]; net = argv[1];
std::string input = "../demo/yolo_test.mp4"; #ifdef __linux__
std::string input = "../demo/yolo_test.mp4";
#elif _WIN32
std::string input = "..\\..\\..\\demo\\yolo_test.mp4";
#endif
if(argc > 2) if(argc > 2)
input = argv[2]; input = argv[2];
char ntype = 'y'; char ntype = 'y';
@@ -131,7 +136,7 @@ int main(int argc, char *argv[]) {
double mean = 0; double mean = 0;
std::cout<<COL_GREENB<<"\n\nTime stats:\n"; std::cout<<COL_GREENB<<"\n\nTime stats:\n";
std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n"; std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n"; std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
for(int i=0; i<detNN->stats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size(); for(int i=0; i<detNN->stats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size();
std::cout<<"Avg: "<<mean/n_batch<<" ms\t"<<1000/(mean/n_batch)<<" FPS\n"<<COL_END; std::cout<<"Avg: "<<mean/n_batch<<" ms\t"<<1000/(mean/n_batch)<<" FPS\n"<<COL_END;
+3
View File
@@ -2,7 +2,10 @@
#include <iostream> #include <iostream>
#include <signal.h> #include <signal.h>
#include <stdlib.h> /* srand, rand */ #include <stdlib.h> /* srand, rand */
#ifdef __linux__
#include <unistd.h> #include <unistd.h>
#endif
#include <mutex> #include <mutex>
#include "utils.h" #include "utils.h"
+4 -2
View File
@@ -4,7 +4,10 @@
#include <iostream> #include <iostream>
#include <signal.h> #include <signal.h>
#include <stdlib.h> #include <stdlib.h>
#ifdef __linux__
#include <unistd.h> #include <unistd.h>
#endif
#include <mutex> #include <mutex>
#include "utils.h" #include "utils.h"
@@ -14,7 +17,7 @@
#include "tkdnn.h" #include "tkdnn.h"
// #define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib. //#define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib.
#ifdef OPENCV_CUDACONTRIB #ifdef OPENCV_CUDACONTRIB
#include <opencv2/cudawarping.hpp> #include <opencv2/cudawarping.hpp>
@@ -150,7 +153,6 @@ class DetectionNN {
int x0, w, x1, y0, h, y1; int x0, w, x1, y0, h, y1;
int objClass; int objClass;
std::string det_class; std::string det_class;
int baseline = 0; int baseline = 0;
float font_scale = 0.5; float font_scale = 0.5;
int thickness = 2; int thickness = 2;
+7
View File
@@ -1,7 +1,14 @@
#include <iostream> #include <iostream>
#include <signal.h> #include <signal.h>
#include <stdlib.h> /* srand, rand */ #include <stdlib.h> /* srand, rand */
#ifdef __linux__
#include <unistd.h> #include <unistd.h>
#elif _WIN32
#define _USE_MATH_DEFINES
#include <math.h>
#endif
#include <mutex> #include <mutex>
#include <Eigen/Dense> #include <Eigen/Dense>
#include "utils.h" #include "utils.h"
+4 -1
View File
@@ -11,8 +11,11 @@
#include <fstream> #include <fstream>
#include <iomanip> #include <iomanip>
#include <signal.h> #include <signal.h>
#include <stdlib.h> #include <stdlib.h>
#ifdef __linux__
#include <unistd.h> #include <unistd.h>
#endif
#include <mutex> #include <mutex>
#include "NvInfer.h" #include "NvInfer.h"
+5
View File
@@ -6,6 +6,7 @@
#include "Network.h" #include "Network.h"
#include "Layer.h" #include "Layer.h"
#include "NvInfer.h" #include "NvInfer.h"
#include <memory>
namespace tk { namespace dnn { namespace tk { namespace dnn {
@@ -60,6 +61,7 @@ public:
#if NV_TENSORRT_MAJOR >= 6 #if NV_TENSORRT_MAJOR >= 6
nvinfer1::IBuilderConfig *configRT; nvinfer1::IBuilderConfig *configRT;
#endif #endif
nvinfer1::ICudaEngine *engineRT; nvinfer1::ICudaEngine *engineRT;
nvinfer1::IExecutionContext *contextRT; nvinfer1::IExecutionContext *contextRT;
@@ -115,6 +117,9 @@ public:
bool serialize(const char *filename); bool serialize(const char *filename);
bool deserialize(const char *filename); bool deserialize(const char *filename);
}; };
}} }}
+2 -1
View File
@@ -52,8 +52,9 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, size); tk::dnn::writeBUF(buf, size);
assert(buf == a + getSerializationSize());
} }
int size; int size;
+36 -36
View File
@@ -4,57 +4,57 @@
class ActivationLogisticRT : public IPlugin { class ActivationLogisticRT : public IPlugin {
public: public:
ActivationLogisticRT() { ActivationLogisticRT() {
} }
~ActivationLogisticRT(){ ~ActivationLogisticRT(){
} }
int getNbOutputs() const override { int getNbOutputs() const override {
return 1; return 1;
} }
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override { Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
return inputs[0]; return inputs[0];
} }
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override { void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
size = 1; size = 1;
for(int i=0; i<outputDims[0].nbDims; i++) for(int i=0; i<outputDims[0].nbDims; i++)
size *= outputDims[0].d[i]; size *= outputDims[0].d[i];
} }
int initialize() override { int initialize() override {
return 0; return 0;
} }
virtual void terminate() override { virtual void terminate() override {
} }
virtual size_t getWorkspaceSize(int maxBatchSize) const override { virtual size_t getWorkspaceSize(int maxBatchSize) const override {
return 0; return 0;
} }
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override { virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
activationLOGISTICForward((dnnType*)reinterpret_cast<const dnnType*>(inputs[0]), activationLOGISTICForward((dnnType*)reinterpret_cast<const dnnType*>(inputs[0]),
reinterpret_cast<dnnType*>(outputs[0]), batchSize*size, stream); reinterpret_cast<dnnType*>(outputs[0]), batchSize*size, stream);
return 0; return 0;
} }
virtual size_t getSerializationSize() override { virtual size_t getSerializationSize() override {
return 1*sizeof(int); return 1*sizeof(int);
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer);
tk::dnn::writeBUF(buf, size); tk::dnn::writeBUF(buf, size);
} }
int size; int size;
}; };
+2 -1
View File
@@ -52,8 +52,9 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, size); tk::dnn::writeBUF(buf, size);
assert(buf == a + getSerializationSize());
} }
int size; int size;
@@ -51,9 +51,10 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, ceiling); tk::dnn::writeBUF(buf, ceiling);
tk::dnn::writeBUF(buf, size); tk::dnn::writeBUF(buf, size);
assert(buf = a + getSerializationSize());
} }
@@ -52,8 +52,9 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, size); tk::dnn::writeBUF(buf, size);
assert(buf == a + getSerializationSize());
} }
int size; int size;
+2 -1
View File
@@ -116,7 +116,7 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, chunk_dim); tk::dnn::writeBUF(buf, chunk_dim);
tk::dnn::writeBUF(buf, kh); tk::dnn::writeBUF(buf, kh);
tk::dnn::writeBUF(buf, kw); tk::dnn::writeBUF(buf, kw);
@@ -163,6 +163,7 @@ public:
for(int i=0; i<dim_ones; i++) for(int i=0; i<dim_ones; i++)
tk::dnn::writeBUF(buf, aus[i]); tk::dnn::writeBUF(buf, aus[i]);
free(aus); free(aus);
assert(buf == a + getSerializationSize());
} }
cublasStatus_t stat; cublasStatus_t stat;
+2 -1
View File
@@ -65,12 +65,13 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a = buf;
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c);
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h);
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w);
tk::dnn::writeBUF(buf, rows); tk::dnn::writeBUF(buf, rows);
tk::dnn::writeBUF(buf, cols); tk::dnn::writeBUF(buf, cols);
assert(buf == a + getSerializationSize());
} }
int c, h, w; int c, h, w;
@@ -55,7 +55,7 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, this->c); tk::dnn::writeBUF(buf, this->c);
tk::dnn::writeBUF(buf, this->h); tk::dnn::writeBUF(buf, this->h);
@@ -65,6 +65,7 @@ public:
tk::dnn::writeBUF(buf, this->stride_W); tk::dnn::writeBUF(buf, this->stride_W);
tk::dnn::writeBUF(buf, this->winSize); tk::dnn::writeBUF(buf, this->winSize);
tk::dnn::writeBUF(buf, this->padding); tk::dnn::writeBUF(buf, this->padding);
assert(buf == a + getSerializationSize());
} }
int n, c, h, w; int n, c, h, w;
+2 -1
View File
@@ -73,13 +73,14 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, classes); tk::dnn::writeBUF(buf, classes);
tk::dnn::writeBUF(buf, coords); tk::dnn::writeBUF(buf, coords);
tk::dnn::writeBUF(buf, num); tk::dnn::writeBUF(buf, num);
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c);
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h);
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w);
assert(buf == a + getSerializationSize());
} }
int c, h, w; int c, h, w;
+2 -1
View File
@@ -52,11 +52,12 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, stride); tk::dnn::writeBUF(buf, stride);
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c);
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h);
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w);
assert(buf == a + getSerializationSize());
} }
int c, h, w, stride; int c, h, w, stride;
+2 -1
View File
@@ -50,11 +50,12 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a = buf;
tk::dnn::writeBUF(buf, n); tk::dnn::writeBUF(buf, n);
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c);
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h);
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w);
assert(buf == a + getSerializationSize());
} }
int n, c, h, w; int n, c, h, w;
+2 -1
View File
@@ -52,7 +52,7 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, o_c); tk::dnn::writeBUF(buf, o_c);
tk::dnn::writeBUF(buf, o_h); tk::dnn::writeBUF(buf, o_h);
@@ -61,6 +61,7 @@ public:
tk::dnn::writeBUF(buf, i_c); tk::dnn::writeBUF(buf, i_c);
tk::dnn::writeBUF(buf, i_h); tk::dnn::writeBUF(buf, i_h);
tk::dnn::writeBUF(buf, i_w); tk::dnn::writeBUF(buf, i_w);
assert(buf == a + getSerializationSize());
} }
int i_c, i_h, i_w, o_c, o_h, o_w; int i_c, i_h, i_w, o_c, o_h, o_w;
+2 -1
View File
@@ -75,7 +75,7 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, groups); tk::dnn::writeBUF(buf, groups);
tk::dnn::writeBUF(buf, group_id); tk::dnn::writeBUF(buf, group_id);
tk::dnn::writeBUF(buf, in); tk::dnn::writeBUF(buf, in);
@@ -85,6 +85,7 @@ public:
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c);
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h);
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w);
assert(buf == a + getSerializationSize());
} }
static const int MAX_INPUTS = 4; static const int MAX_INPUTS = 4;
+2 -1
View File
@@ -59,13 +59,14 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, bc); tk::dnn::writeBUF(buf, bc);
tk::dnn::writeBUF(buf, bh); tk::dnn::writeBUF(buf, bh);
tk::dnn::writeBUF(buf, bw); tk::dnn::writeBUF(buf, bw);
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c);
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h);
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w);
assert(buf == a + getSerializationSize());
} }
+2 -1
View File
@@ -54,11 +54,12 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, stride); tk::dnn::writeBUF(buf, stride);
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c);
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h);
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w);
assert(buf == a + getSerializationSize());
} }
int c, h, w, stride; int c, h, w, stride;
+36 -30
View File
@@ -64,22 +64,23 @@ public:
checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream)); checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream));
for (int b = 0; b < batchSize; ++b){
for(int n = 0; n < n_masks; ++n){
int index = entry_index(b, n*w*h, 0);
if (new_coords == 1){
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
}
else{
activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); //x,y
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); for (int b = 0; b < batchSize; ++b){
for(int n = 0; n < n_masks; ++n){
index = entry_index(b, n*w*h, 4); int index = entry_index(b, n*w*h, 0);
activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream); if (new_coords == 1){
} if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
} }
} else{
activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); //x,y
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
index = entry_index(b, n*w*h, 4);
activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream);
}
}
}
//std::cout<<"YOLO END\n"; //std::cout<<"YOLO END\n";
return 0; return 0;
@@ -91,21 +92,25 @@ public:
} }
virtual void serialize(void* buffer) override { virtual void serialize(void* buffer) override {
char *buf = reinterpret_cast<char*>(buffer); char *buf = reinterpret_cast<char*>(buffer),*a=buf;
tk::dnn::writeBUF(buf, classes); tk::dnn::writeBUF(buf, classes); std::cout << "Classes :" << classes << std::endl;
tk::dnn::writeBUF(buf, num); tk::dnn::writeBUF(buf, num); std::cout << "Num : " << num << std::endl;
tk::dnn::writeBUF(buf, n_masks); tk::dnn::writeBUF(buf, n_masks); std::cout << "N_Masks" << n_masks << std::endl;
tk::dnn::writeBUF(buf, scaleXY); tk::dnn::writeBUF(buf, scaleXY); std::cout << "ScaleXY :" << scaleXY << std::endl;
tk::dnn::writeBUF(buf, nms_thresh); tk::dnn::writeBUF(buf, nms_thresh); std::cout << "nms_thresh :" << nms_thresh << std::endl;
tk::dnn::writeBUF(buf, nms_kind); tk::dnn::writeBUF(buf, nms_kind); std::cout << "nms_kind : " << nms_kind << std::endl;
tk::dnn::writeBUF(buf, new_coords); tk::dnn::writeBUF(buf, new_coords); std::cout << "new_coords : " << new_coords << std::endl;
tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, c); std::cout << "C : " << c << std::endl;
tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, h); std::cout << "H : " << h << std::endl;
tk::dnn::writeBUF(buf, w); tk::dnn::writeBUF(buf, w); std::cout << "C : " << c << std::endl;
for(int i=0; i<n_masks; i++) for (int i = 0; i < n_masks; i++)
tk::dnn::writeBUF(buf, mask[i]); {
for(int i=0; i<n_masks*2*num; i++) tk::dnn::writeBUF(buf, mask[i]); std::cout << "mask[i] : " << mask[i] << std::endl;
tk::dnn::writeBUF(buf, bias[i]); }
for (int i = 0; i < n_masks * 2 * num; i++)
{
tk::dnn::writeBUF(buf, bias[i]); std::cout << "bias[i] : " << bias[i] << std::endl;
}
// save classes names // save classes names
for(int i=0; i<classes; i++) { for(int i=0; i<classes; i++) {
@@ -115,6 +120,7 @@ public:
tk::dnn::writeBUF(buf, tmp[j]); tk::dnn::writeBUF(buf, tmp[j]);
} }
} }
assert(buf == a + getSerializationSize());
} }
int c, h, w; int c, h, w;
+4 -3
View File
@@ -29,7 +29,8 @@ int testInference(std::vector<std::string> input_bins, std::vector<std::string>
readBinaryFile(input_bins[0], net->input_dim.tot(), &input_h, &data); readBinaryFile(input_bins[0], net->input_dim.tot(), &input_h, &data);
// outputs // outputs
dnnType *cudnn_out[outputs.size()], *rt_out[outputs.size()]; //dnnType *cudnn_out[outputs.size()], *rt_out[outputs.size()];
std::vector<dnnType *> cudnn_out,rt_out;
tk::dnn::dataDim_t dim1 = net->input_dim; //input dim tk::dnn::dataDim_t dim1 = net->input_dim; //input dim
printCenteredTitle(" CUDNN inference ", '=', 30); { printCenteredTitle(" CUDNN inference ", '=', 30); {
@@ -39,7 +40,7 @@ int testInference(std::vector<std::string> input_bins, std::vector<std::string>
TKDNN_TSTOP TKDNN_TSTOP
dim1.print(); dim1.print();
} }
for(int i=0; i<outputs.size(); i++) cudnn_out[i] = outputs[i]->dstData; for(int i=0; i<outputs.size(); i++) cudnn_out.push_back(outputs[i]->dstData);
if(netRT != nullptr) { if(netRT != nullptr) {
tk::dnn::dataDim_t dim2 = net->input_dim; tk::dnn::dataDim_t dim2 = net->input_dim;
@@ -50,7 +51,7 @@ int testInference(std::vector<std::string> input_bins, std::vector<std::string>
TKDNN_TSTOP TKDNN_TSTOP
dim2.print(); dim2.print();
} }
for(int i=0; i<outputs.size(); i++) rt_out[i] = (dnnType*)netRT->buffersRT[i+1]; for(int i=0; i<outputs.size(); i++) rt_out.push_back((dnnType*)netRT->buffersRT[i+1]);
} }
int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0; int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0;
+13
View File
@@ -12,8 +12,12 @@
#include <cublas_v2.h> #include <cublas_v2.h>
#include <cudnn.h> #include <cudnn.h>
#ifdef __linux__
#include <unistd.h> #include <unistd.h>
#endif
#include <ios> #include <ios>
#include <chrono>
#define dnnType float #define dnnType float
@@ -39,6 +43,7 @@
#define TKDNN_VERBOSE 0 #define TKDNN_VERBOSE 0
// Simple Timer // Simple Timer
#ifdef __linux__
#define TKDNN_TSTART timespec start, end; \ #define TKDNN_TSTART timespec start, end; \
clock_gettime(CLOCK_MONOTONIC, &start); clock_gettime(CLOCK_MONOTONIC, &start);
@@ -48,6 +53,14 @@
if(show) std::cout<<col<<"Time:"<<std::setw(16)<<t_ns<<" ms\n"<<COL_END; if(show) std::cout<<col<<"Time:"<<std::setw(16)<<t_ns<<" ms\n"<<COL_END;
#define TKDNN_TSTOP TKDNN_TSTOP_C(COL_CYANB, TKDNN_VERBOSE) #define TKDNN_TSTOP TKDNN_TSTOP_C(COL_CYANB, TKDNN_VERBOSE)
#elif _WIN32
#define TKDNN_TSTART auto start = std::chrono::high_resolution_clock::now();
#define TKDNN_TSTOP auto stop = std::chrono::high_resolution_clock::now(); \
std::chrono::duration<double> duration = stop -start; \
auto time_ms = std::chrono::duration_cast<std::chrono::milliseconds>(duration);\
double t_ns = time_ms.count();
#endif
/******************************************************** /********************************************************
* Prints the error message, and exits * Prints the error message, and exits
+39
View File
@@ -0,0 +1,39 @@
import os
import urllib.request as dowReq
import zipfile
val = input("Enter BDD or COCO :")
if(val == "COCO"):
url = "https://cloud.hipert.unimore.it/s/LNxBDk4wzqXPL8c/download"
lib = "..\demo\COCO_val2017"
lib_zip = "COCO_val2017.zip"
elif(val == "BDD"):
url = "https://cloud.hipert.unimore.it/s/bikqk3FzCq2tg4D/download"
lib = "..\demo\BDD100k_val"
lib_zip = "BDD100k_val.zip"
dowReq.urlretrieve(url,lib_zip)
with zipfile.ZipFile(lib_zip,'r') as zip_ref:
zip_ref.extractall(lib)
labelFolder = lib + "\labels"
imageFolder = lib + "\images"
file1 = open(".\\..\\demo\\all_labels.txt","a")
path1 = os.path.realpath(labelFolder)
for file in os.listdir(labelFolder):
valTemp = path1 + "\\" + file
valTemp = valTemp + '\n'
file1.write(valTemp)
file1.close()
file2 = open(".\\..\\demo\\all_images.txt","a")
path2 = os.path.realpath(imageFolder)
for file in os.listdir(imageFolder):
pathtemp = path2 + "\\" + file
pathtemp = pathtemp + '\n'
file2.write(pathtemp)
file2.close()
print("Completed")
+9 -4
View File
@@ -87,17 +87,22 @@ LSTM::LSTM( Network *net, int hiddensize, bool returnSeq, std::string fname_weig
checkCUDNN(cudnnCreateRNNDescriptor(&rnnDesc)); checkCUDNN(cudnnCreateRNNDescriptor(&rnnDesc));
#if CUDNN_MAJOR > 7 #if CUDNN_MAJOR > 7
checkCUDNN(cudnnSetRNNDescriptor_v6(net->cudnnHandle, checkCUDNN(cudnnSetRNNDescriptor_v6(net->cudnnHandle,rnnDesc, stateSize, numLayers, dropoutDesc,
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL,
cudnnRNNMode_t::CUDNN_LSTM,
cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD,
net->dataType));
#else #else
checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle, checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle,rnnDesc, stateSize, numLayers, dropoutDesc,
#endif
rnnDesc, stateSize, numLayers, dropoutDesc,
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT, cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL), //(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL, cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL,
cudnnRNNMode_t::CUDNN_LSTM, cudnnRNNMode_t::CUDNN_LSTM,
cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD, cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD,
net->dataType)); net->dataType));
#endif
// Get temp space sizes // Get temp space sizes
+82 -36
View File
@@ -139,7 +139,8 @@ NetworkRT::NetworkRT(Network *net, const char *name) {
#if NV_TENSORRT_MAJOR >= 6 #if NV_TENSORRT_MAJOR >= 6
engineRT = builderRT->buildEngineWithConfig(*networkRT, *configRT); engineRT = builderRT->buildEngineWithConfig(*networkRT, *configRT);
#else #else
engineRT = builderRT->buildCudaEngine(*networkRT); //engineRT = builderRT->buildCudaEngine(*networkRT);
engineRT = std::shared_ptr<nvinfer1::ICudaEngine>(builderRT->buildCudaEngine(*networkRT));
#endif #endif
if(engineRT == nullptr) if(engineRT == nullptr)
FatalError("cloud not build cuda engine") FatalError("cloud not build cuda engine")
@@ -567,7 +568,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, DeformConv2d *l) {
IPluginLayer *lRT = networkRT->addPlugin(inputs, 2, *plugin); IPluginLayer *lRT = networkRT->addPlugin(inputs, 2, *plugin);
checkNULL(lRT); checkNULL(lRT);
lRT->setName( ("Deformable" + std::to_string(l->id)).c_str() ); lRT->setName( ("Deformable" + std::to_string(l->id)).c_str() );
delete(inputs); delete[](inputs);
// batchnorm // batchnorm
void *bias_b, *power_b, *mean_b, *variance_b, *scales_b; void *bias_b, *power_b, *mean_b, *variance_b, *scales_b;
if(dtRT == DataType::kHALF) { if(dtRT == DataType::kHALF) {
@@ -644,7 +645,7 @@ bool NetworkRT::deserialize(const char *filename) {
IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialData, size_t serialLength) { IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialData, size_t serialLength) {
const char * buf = reinterpret_cast<const char*>(serialData); const char * buf = reinterpret_cast<const char*>(serialData),*bufCheck = buf;
std::string name(layerName); std::string name(layerName);
//std::cout<<name<<std::endl; //std::cout<<name<<std::endl;
@@ -652,11 +653,18 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
if(name.find("ActivationLeaky") == 0) { if(name.find("ActivationLeaky") == 0) {
ActivationLeakyRT *a = new ActivationLeakyRT(); ActivationLeakyRT *a = new ActivationLeakyRT();
a->size = readBUF<int>(buf); a->size = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return a; return a;
} }
if(name.find("ActivationMish") == 0) { if(name.find("ActivationMish") == 0) {
ActivationMishRT *a = new ActivationMishRT(); ActivationMishRT *a = new ActivationMishRT();
a->size = readBUF<int>(buf); a->size = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return a;
}
if(name.find("ActivationLogistic") == 0) {
ActivationLogisticRT *a = new ActivationLogisticRT();
a->size = readBUF<int>(buf);
return a; return a;
} }
if(name.find("ActivationLogistic") == 0) { if(name.find("ActivationLogistic") == 0) {
@@ -665,27 +673,33 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
return a; return a;
} }
if(name.find("ActivationCReLU") == 0) { if(name.find("ActivationCReLU") == 0) {
ActivationReLUCeiling *a = new ActivationReLUCeiling(readBUF<float>(buf)); float activationReluTemp = readBUF<float>(buf);
ActivationReLUCeiling* a = new ActivationReLUCeiling(activationReluTemp);
a->size = readBUF<int>(buf); a->size = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return a; return a;
} }
if(name.find("Region") == 0) { if(name.find("Region") == 0) {
RegionRT *r = new RegionRT(readBUF<int>(buf), //classes int classesTemp = readBUF<int>(buf);
readBUF<int>(buf), //coords int coordsTemp = readBUF<int>(buf);
readBUF<int>(buf)); //num int numTemp = readBUF<int>(buf);
RegionRT* r = new RegionRT(classesTemp, coordsTemp, numTemp);
r->c = readBUF<int>(buf); r->c = readBUF<int>(buf);
r->h = readBUF<int>(buf); r->h = readBUF<int>(buf);
r->w = readBUF<int>(buf); r->w = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
if(name.find("Reorg") == 0) { if(name.find("Reorg") == 0) {
ReorgRT *r = new ReorgRT(readBUF<int>(buf)); //stride int strideTemp = readBUF<int>(buf);
ReorgRT *r = new ReorgRT(strideTemp);
r->c = readBUF<int>(buf); r->c = readBUF<int>(buf);
r->h = readBUF<int>(buf); r->h = readBUF<int>(buf);
r->w = readBUF<int>(buf); r->w = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
@@ -701,27 +715,34 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
r->h = readBUF<int>(buf); r->h = readBUF<int>(buf);
r->w = readBUF<int>(buf); r->w = readBUF<int>(buf);
return r; return r;
assert(buf == bufCheck + serialLength);
} }
if(name.find("Pooling") == 0) { if(name.find("Pooling") == 0) {
MaxPoolFixedSizeRT *r = new MaxPoolFixedSizeRT( readBUF<int>(buf), //c int cTemp = readBUF<int>(buf);
readBUF<int>(buf), //h int hTemp = readBUF<int>(buf);
readBUF<int>(buf), //w int wTemp = readBUF<int>(buf);
readBUF<int>(buf), //n int nTemp = readBUF<int>(buf);
readBUF<int>(buf), //strideH int strideHTemp = readBUF<int>(buf);
readBUF<int>(buf), //strideW int strideWTemp = readBUF<int>(buf);
readBUF<int>(buf), //winSize int winSizeTemp = readBUF<int>(buf);
readBUF<int>(buf)); //padding int paddingTemp = readBUF<int>(buf);
MaxPoolFixedSizeRT* r = new MaxPoolFixedSizeRT(cTemp, hTemp, wTemp, nTemp, strideHTemp, strideWTemp, winSizeTemp, paddingTemp);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
if(name.find("Resize") == 0) { if(name.find("Resize") == 0) {
ResizeLayerRT *r = new ResizeLayerRT(readBUF<int>(buf), //o_c int o_cTemp = readBUF<int>(buf);
readBUF<int>(buf), //o_h int o_hTemp = readBUF<int>(buf);
readBUF<int>(buf)); //o_w int o_wTemp = readBUF<int>(buf);
ResizeLayerRT* r = new ResizeLayerRT(o_cTemp, o_hTemp, o_wTemp);
r->i_c = readBUF<int>(buf); r->i_c = readBUF<int>(buf);
r->i_h = readBUF<int>(buf); r->i_h = readBUF<int>(buf);
r->i_w = readBUF<int>(buf); r->i_w = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
@@ -732,6 +753,7 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
r->w = readBUF<int>(buf); r->w = readBUF<int>(buf);
r->rows = readBUF<int>(buf); r->rows = readBUF<int>(buf);
r->cols = readBUF<int>(buf); r->cols = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
@@ -743,20 +765,25 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
new_dim.h = readBUF<int>(buf); new_dim.h = readBUF<int>(buf);
new_dim.w = readBUF<int>(buf); new_dim.w = readBUF<int>(buf);
ReshapeRT *r = new ReshapeRT(new_dim); ReshapeRT *r = new ReshapeRT(new_dim);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
if(name.find("Yolo") == 0) { if(name.find("Yolo") == 0) {
YoloRT *r = new YoloRT(readBUF<int>(buf), //classes
readBUF<int>(buf), //num int classes_temp = readBUF<int>(buf);
nullptr, //yolo int num_temp = readBUF<int>(buf);
readBUF<int>(buf), //n_masks int n_masks_temp = readBUF<int>(buf);
readBUF<float>(buf), //scale_xy float scale_xy_temp = readBUF<float>(buf);
readBUF<float>(buf), //nms_thresh float nms_thresh_temp = readBUF<float>(buf);
readBUF<int>(buf), //nms_kind int nms_kind_temp = readBUF<int>(buf);
readBUF<int>(buf) //new_coords int new_coords_temp = readBUF<int>(buf);
);
YoloRT *r = new YoloRT(classes_temp,num_temp,nullptr,n_masks_temp,scale_xy_temp,nms_thresh_temp,nms_kind_temp,new_coords_temp);
r->c = readBUF<int>(buf); r->c = readBUF<int>(buf);
r->h = readBUF<int>(buf); r->h = readBUF<int>(buf);
r->w = readBUF<int>(buf); r->w = readBUF<int>(buf);
@@ -773,36 +800,54 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
tmp[j] = readBUF<char>(buf); tmp[j] = readBUF<char>(buf);
r->classesNames[i] = std::string(tmp); r->classesNames[i] = std::string(tmp);
} }
assert(buf == bufCheck + serialLength);
yolos[n_yolos++] = r; yolos[n_yolos++] = r;
return r; return r;
} }
if(name.find("Upsample") == 0) { if(name.find("Upsample") == 0) {
UpsampleRT *r = new UpsampleRT(readBUF<int>(buf)); //stride int strideTemp = readBUF<int>(buf);
UpsampleRT* r = new UpsampleRT(strideTemp);
r->c = readBUF<int>(buf); r->c = readBUF<int>(buf);
r->h = readBUF<int>(buf); r->h = readBUF<int>(buf);
r->w = readBUF<int>(buf); r->w = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
if(name.find("Route") == 0) { if(name.find("Route") == 0) {
RouteRT *r = new RouteRT(readBUF<int>(buf),readBUF<int>(buf)); int groupsTemp = readBUF<int>(buf);
int group_idTemp = readBUF<int>(buf);
RouteRT* r = new RouteRT(groupsTemp, group_idTemp);
r->in = readBUF<int>(buf); r->in = readBUF<int>(buf);
for(int i=0; i<RouteRT::MAX_INPUTS; i++) for(int i=0; i<RouteRT::MAX_INPUTS; i++)
r->c_in[i] = readBUF<int>(buf); r->c_in[i] = readBUF<int>(buf);
r->c = readBUF<int>(buf); r->c = readBUF<int>(buf);
r->h = readBUF<int>(buf); r->h = readBUF<int>(buf);
r->w = readBUF<int>(buf); r->w = readBUF<int>(buf);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
if(name.find("Deformable") == 0) { if(name.find("Deformable") == 0) {
DeformableConvRT *r = new DeformableConvRT(readBUF<int>(buf), readBUF<int>(buf), readBUF<int>(buf), int chuck_dimTemp = readBUF<int>(buf);
readBUF<int>(buf), readBUF<int>(buf), readBUF<int>(buf), int khTemp = readBUF<int>(buf);
readBUF<int>(buf), readBUF<int>(buf), int kwTemp = readBUF<int>(buf);
readBUF<int>(buf),readBUF<int>(buf),readBUF<int>(buf),readBUF<int>(buf), int shTemp = readBUF<int>(buf);
readBUF<int>(buf),readBUF<int>(buf),readBUF<int>(buf),readBUF<int>(buf), int swTemp = readBUF<int>(buf);
nullptr); int phTemp = readBUF<int>(buf);
int pwTemp = readBUF<int>(buf);
int deformableGroupTemp = readBUF<int>(buf);
int i_nTemp = readBUF<int>(buf);
int i_cTemp = readBUF<int>(buf);
int i_hTemp = readBUF<int>(buf);
int i_wTemp = readBUF<int>(buf);
int o_nTemp = readBUF<int>(buf);
int o_cTemp = readBUF<int>(buf);
int o_hTemp = readBUF<int>(buf);
int o_wTemp = readBUF<int>(buf);
DeformableConvRT* r = new DeformableConvRT(chuck_dimTemp, khTemp, kwTemp, shTemp, swTemp, phTemp, pwTemp, deformableGroupTemp, i_nTemp, i_cTemp, i_hTemp, i_wTemp, o_nTemp, o_cTemp, o_hTemp, o_wTemp, nullptr);
dnnType *aus = new dnnType[r->chunk_dim*2]; dnnType *aus = new dnnType[r->chunk_dim*2];
for(int i=0; i<r->chunk_dim*2; i++) for(int i=0; i<r->chunk_dim*2; i++)
aus[i] = readBUF<dnnType>(buf); aus[i] = readBUF<dnnType>(buf);
@@ -833,6 +878,7 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
aus[i] = readBUF<dnnType>(buf); aus[i] = readBUF<dnnType>(buf);
checkCuda( cudaMemcpy(r->ones_d2, aus, sizeof(dnnType)*r->dim_ones, cudaMemcpyHostToDevice) ); checkCuda( cudaMemcpy(r->ones_d2, aus, sizeof(dnnType)*r->dim_ones, cudaMemcpyHostToDevice) );
free(aus); free(aus);
assert(buf == bufCheck + serialLength);
return r; return r;
} }
+5 -5
View File
@@ -9,6 +9,7 @@
#include "Layer.h" #include "Layer.h"
#include "kernels.h" #include "kernels.h"
namespace tk { namespace dnn { namespace tk { namespace dnn {
Yolo::Yolo(Network *net, int classes, int num, std::string fname_weights, int n_masks, float scale_xy, double nms_thresh, nmsKind_t nsm_kind, int new_coords) : Yolo::Yolo(Network *net, int classes, int num, std::string fname_weights, int n_masks, float scale_xy, double nms_thresh, nmsKind_t nsm_kind, int new_coords) :
@@ -95,7 +96,6 @@ dnnType* Yolo::infer(dataDim_t &dim, dnnType* srcData) {
activationLOGISTICForward(srcData + index, dstData + index, 2*dim.w*dim.h); activationLOGISTICForward(srcData + index, dstData + index, 2*dim.w*dim.h);
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
index = entry_index(b, n*dim.w*dim.h, 4, classes, input_dim, output_dim); index = entry_index(b, n*dim.w*dim.h, 4, classes, input_dim, output_dim);
activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*dim.w*dim.h); activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*dim.w*dim.h);
} }
@@ -212,10 +212,10 @@ float yolo_box_iou(Yolo::box a, Yolo::box b)
} }
void box_c(const Yolo::box a, const Yolo::box b, float& top, float& bot, float& left, float& right) { void box_c(const Yolo::box a, const Yolo::box b, float& top, float& bot, float& left, float& right) {
top = std::min(a.y - a.h / 2, b.y - b.h / 2); top = (std::min)(a.y - a.h / 2, b.y - b.h / 2);
bot = std::max(a.y + a.h / 2, b.y + b.h / 2); bot = (std::max)(a.y + a.h / 2, b.y + b.h / 2);
left = std::min(a.x - a.w / 2, b.x - b.w / 2); left = (std::min)(a.x - a.w / 2, b.x - b.w / 2);
right = std::max(a.x + a.w / 2, b.x + b.w / 2); right = (std::max)(a.x + a.w / 2, b.x + b.w / 2);
} }
// https://github.com/Zzh-tju/DIoU-darknet // https://github.com/Zzh-tju/DIoU-darknet
+4 -3
View File
@@ -94,9 +94,10 @@ void Yolo3Detection::preprocess(cv::Mat &frame, const int bi){
void Yolo3Detection::postprocess(const int bi, const bool mAP){ void Yolo3Detection::postprocess(const int bi, const bool mAP){
//get yolo outputs //get yolo outputs
dnnType *rt_out[netRT->pluginFactory->n_yolos]; std::vector<float *> rt_out;
for(int i=0; i<netRT->pluginFactory->n_yolos; i++) //dnnType *rt_out[netRT->pluginFactory->n_yolos];
rt_out[i] = (dnnType*)netRT->buffersRT[i+1] + netRT->buffersDIM[i+1].tot()*bi; for(int i=0; i<netRT->pluginFactory->n_yolos; i++)
rt_out.push_back((dnnType*)netRT->buffersRT[i+1] + netRT->buffersDIM[i+1].tot()*bi);
float x_ratio = float(originalSize[bi].width) / float(netRT->input_dim.w); float x_ratio = float(originalSize[bi].width) / float(netRT->input_dim.w);
float y_ratio = float(originalSize[bi].height) / float(netRT->input_dim.h); float y_ratio = float(originalSize[bi].height) / float(netRT->input_dim.h);
+1 -1
View File
@@ -18,7 +18,7 @@ inline int GET_BLOCKS(const int N)
} }
__device__ float dmcn_im2col_bilinear(const float *bottom_data, const int data_width, __device__ __host__ float dmcn_im2col_bilinear(const float *bottom_data, const int data_width,
const int height, const int width, float h, float w) { const int height, const int width, float h, float w) {
int h_low = floor(h); int h_low = floor(h);
int w_low = floor(w); int w_low = floor(w);
+15 -2
View File
@@ -23,14 +23,23 @@ bool fileExist(const char *fname) {
void downloadWeightsifDoNotExist(const std::string& input_bin, const std::string& test_folder, const std::string& weights_url){ void downloadWeightsifDoNotExist(const std::string& input_bin, const std::string& test_folder, const std::string& weights_url){
if(!fileExist(input_bin.c_str())){ if(!fileExist(input_bin.c_str())){
std::string mkdir_cmd = "mkdir " + test_folder; std::string mkdir_cmd = "mkdir " + test_folder;
std::string wget_cmd = "wget " + weights_url + " -O " + test_folder + "/weights.zip"; std::string wget_cmd = "curl " + weights_url + " --output " + test_folder + "/weights.zip";
#ifdef __linux__
std::string unzip_cmd = "unzip " + test_folder + "/weights.zip -d" + test_folder; std::string unzip_cmd = "unzip " + test_folder + "/weights.zip -d" + test_folder;
std::string rm_cmd = "rm " + test_folder + "/weights.zip"; std::string rm_cmd = "rm " + test_folder + "/weights.zip";
#elif _WIN32
std::string unzip_cmd = "7z x " + test_folder + "/weights.zip -o" + test_folder;
#endif
int err = 0; int err = 0;
err = system(mkdir_cmd.c_str()); err = system(mkdir_cmd.c_str());
err = system(wget_cmd.c_str()); err = system(wget_cmd.c_str());
err = system(unzip_cmd.c_str()); err = system(unzip_cmd.c_str());
#ifdef __linux__
err = system(rm_cmd.c_str()); err = system(rm_cmd.c_str());
#endif
} }
} }
@@ -191,8 +200,12 @@ void getMemUsage(double& vm_usage_kb, double& resident_set_kb){
>> O >> itrealvalue >> starttime >> vsize >> rss; >> O >> itrealvalue >> starttime >> vsize >> rss;
stat_stream.close(); stat_stream.close();
#ifdef __linux__
long page_size_kb = sysconf(_SC_PAGE_SIZE) / 1024; // in case x86-64 is configured to use 2MB pages long page_size_kb = sysconf(_SC_PAGE_SIZE) / 1024; // in case x86-64 is configured to use 2MB pages
#elif _WIN32
long page_size_kb = 4096/1024;
#endif
vm_usage_kb = vsize / 1024.0; vm_usage_kb = vsize / 1024.0;
resident_set_kb = rss * page_size_kb; resident_set_kb = rss * page_size_kb;
} }
+1 -3
View File
@@ -19,8 +19,6 @@ int main() {
std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names"; std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names";
downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download"); downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download");
// parse darknet network // parse darknet network
tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path);
net->print(); net->print();
@@ -33,4 +31,4 @@ int main() {
delete net; delete net;
delete netRT; delete netRT;
return ret; return ret;
} }