Merge branch 'perseusdg-master'
This commit is contained in:
+7
-1
@@ -12,5 +12,11 @@ build/
|
|||||||
*.hdf5
|
*.hdf5
|
||||||
*.pk
|
*.pk
|
||||||
*.table
|
*.table
|
||||||
|
cmake-build-release/
|
||||||
demo/COCO_val2017
|
demo/COCO_val2017
|
||||||
demo/BDD100K_val
|
demo/BDD100K_val
|
||||||
|
/.vs
|
||||||
|
cmake-build-minsizerel/*
|
||||||
|
scripts/COCO_val2017/*
|
||||||
|
scripts/COCO_val2017.zip
|
||||||
|
scripts/all_labels.txt
|
||||||
+12
-4
@@ -2,7 +2,14 @@ cmake_minimum_required(VERSION 3.5)
|
|||||||
|
|
||||||
project (tkDNN)
|
project (tkDNN)
|
||||||
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake)
|
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake)
|
||||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable")
|
if(UNIX)
|
||||||
|
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable ")
|
||||||
|
endif()
|
||||||
|
if(WIN32)
|
||||||
|
set(CMAKE_CXX_STANDARD 11)
|
||||||
|
set(CMAKE_CXX_FLAGS "/O2 /FS /EHsc")
|
||||||
|
set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON)
|
||||||
|
endif(WIN32)
|
||||||
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN)
|
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN)
|
||||||
|
|
||||||
# project specific flags
|
# project specific flags
|
||||||
@@ -18,7 +25,7 @@ add_definitions(-DTKDNN_PATH="${CMAKE_CURRENT_SOURCE_DIR}")
|
|||||||
find_package(CUDA 9.0 REQUIRED)
|
find_package(CUDA 9.0 REQUIRED)
|
||||||
SET(CUDA_SEPARABLE_COMPILATION ON)
|
SET(CUDA_SEPARABLE_COMPILATION ON)
|
||||||
#set(CUDA_NVCC_FLAGS "${CUDA_NVCC_FLAGS} -arch=sm_30 --compiler-options '-fPIC'")
|
#set(CUDA_NVCC_FLAGS "${CUDA_NVCC_FLAGS} -arch=sm_30 --compiler-options '-fPIC'")
|
||||||
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32)
|
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32 -arch=sm_61 )
|
||||||
|
|
||||||
find_package(CUDNN REQUIRED)
|
find_package(CUDNN REQUIRED)
|
||||||
include_directories(${CUDNN_INCLUDE_DIR})
|
include_directories(${CUDNN_INCLUDE_DIR})
|
||||||
@@ -28,6 +35,7 @@ include_directories(${CUDNN_INCLUDE_DIR})
|
|||||||
file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu")
|
file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu")
|
||||||
cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${CUDNN_INCLUDE_DIRS})
|
cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${CUDNN_INCLUDE_DIRS})
|
||||||
cuda_add_library(kernels SHARED ${tkdnn_CUSRC})
|
cuda_add_library(kernels SHARED ${tkdnn_CUSRC})
|
||||||
|
target_link_libraries(kernels ${CUDA_CUBLAS_LIBRARIES})
|
||||||
|
|
||||||
|
|
||||||
#-------------------------------------------------------------------------------
|
#-------------------------------------------------------------------------------
|
||||||
@@ -40,7 +48,7 @@ find_package(OpenCV REQUIRED)
|
|||||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV")
|
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV")
|
||||||
|
|
||||||
# gives problems in cross-compiling, probably malformed cmake config
|
# gives problems in cross-compiling, probably malformed cmake config
|
||||||
#find_package(yaml-cpp REQUIRED)
|
find_package(yaml-cpp REQUIRED)
|
||||||
|
|
||||||
#-------------------------------------------------------------------------------
|
#-------------------------------------------------------------------------------
|
||||||
# Build Libraries
|
# Build Libraries
|
||||||
@@ -48,7 +56,7 @@ set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV")
|
|||||||
file(GLOB tkdnn_SRC "src/*.cpp")
|
file(GLOB tkdnn_SRC "src/*.cpp")
|
||||||
set(tkdnn_LIBS kernels ${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDNN_LIBRARIES} ${OpenCV_LIBS} yaml-cpp)
|
set(tkdnn_LIBS kernels ${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDNN_LIBRARIES} ${OpenCV_LIBS} yaml-cpp)
|
||||||
|
|
||||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11")
|
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS}")
|
||||||
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${OPENCV_INCLUDE_DIRS} ${NVINFER_INCLUDES})
|
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${OPENCV_INCLUDE_DIRS} ${NVINFER_INCLUDES})
|
||||||
add_library(tkDNN SHARED ${tkdnn_SRC})
|
add_library(tkDNN SHARED ${tkdnn_SRC})
|
||||||
target_link_libraries(tkDNN ${tkdnn_LIBS})
|
target_link_libraries(tkDNN ${tkdnn_LIBS})
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
1)error C2131 @ Yolo3Detection.cpp(97) -> expression doesnt evaluate to a constant caused to read of variable outside its lifetime
|
||||||
@@ -80,6 +80,14 @@ Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001
|
|||||||
- [mAP demo](#map-demo)
|
- [mAP demo](#map-demo)
|
||||||
- [Existing tests and supported networks](#existing-tests-and-supported-networks)
|
- [Existing tests and supported networks](#existing-tests-and-supported-networks)
|
||||||
- [References](#references)
|
- [References](#references)
|
||||||
|
- [tkDNN on Windows 10 (experimental)](#tkdnn-on-windows-10-experimental)
|
||||||
|
- [Dependencies-Windows](#dependencies-windows)
|
||||||
|
- [Compiling tkDNN on Windows](#compiling-tkdnn-on-windows)
|
||||||
|
- [Run the demo on Windows](#run-the-demo-on-windows)
|
||||||
|
- [FP16 inference windows](#fp16-inference-windows)
|
||||||
|
- [INT8 inference windows](#int8-inference-windows)
|
||||||
|
- [Known issues with tkDNN on Windows](#known-issues-with-tkdnn-on-windows)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@@ -356,6 +364,94 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing
|
|||||||
| yolo4x | Yolov4x-mish <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) |
|
| yolo4x | Yolov4x-mish <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) |
|
||||||
| yolo4x-cps | Scaled Yolov4 <sup>10</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) |
|
| yolo4x-cps | Scaled Yolov4 <sup>10</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) |
|
||||||
|
|
||||||
|
### tkDNN on Windows 10 (experimental)
|
||||||
|
|
||||||
|
### Dependencies-Windows
|
||||||
|
This branch should work on every NVIDIA GPU supported in windows with the following dependencies:
|
||||||
|
|
||||||
|
* WINDOWS 10 1803 or HIGHER
|
||||||
|
* CUDA 10.0 (Recommended CUDA 11.2 )
|
||||||
|
* CUDNN 7.6 (Recommended CUDNN 8.1.1 )
|
||||||
|
* TENSORRT 6.0.1 (Recommended TENSORRT 7.2.3.4 )
|
||||||
|
* OPENCV 3.4 (Recommended OPENCV 4.2.0 )
|
||||||
|
* MSVC 16.7
|
||||||
|
* YAML-CPP
|
||||||
|
* EIGEN3
|
||||||
|
* 7ZIP (ADD TO PATH)
|
||||||
|
* NINJA 1.10
|
||||||
|
|
||||||
|
|
||||||
|
All the above mentioned dependencies except 7ZIP can be installed using Microsoft's [VCPKG](https://github.com/microsoft/vcpkg.git) .
|
||||||
|
After bootstrapping VCPKG the dependencies can be built and installed using the following command :
|
||||||
|
|
||||||
|
```
|
||||||
|
opencv4(normal) - vcpkg.exe install opencv4[tbb,jpeg,tiff,opengl,openmp,png,ffmpeg,eigen]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build
|
||||||
|
|
||||||
|
opencv4(cuda) - vcpkg.exe install opencv4[cuda,nonfree,contrib,eigen,tbb,jpeg,tiff,opengl,openmp,png,ffmpeg]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build
|
||||||
|
```
|
||||||
|
To build opencv4 with cuda and cudnn version corresponding to your cuda version,vcpkg's cudnn portfile needs to be modified by adding ```$ENV{CUDA_PATH}``` at lines 16 and 17 in the portfile.cmake
|
||||||
|
|
||||||
|
After VCPKG finishes building and installing all the packages delete C:\temp_vcpkg_build and add C:\opt\x64-windows\bin and C:\opt\x64-windows\debug\bin to path
|
||||||
|
|
||||||
|
### Compiling tkDNN on Windows
|
||||||
|
|
||||||
|
tkDNN is built with cmake(3.15+) on windows along with ninja.Msbuild and NMake Makefiles are drastically slower when compiling the library compared to windows
|
||||||
|
```
|
||||||
|
git clone https://github.com/ceccocats/tkDNN.git
|
||||||
|
cd tkdnn-windows
|
||||||
|
mkdir build
|
||||||
|
cd build
|
||||||
|
cmake -DCMAKE_BUILD_TYPE=Release -G"Ninja" ..
|
||||||
|
ninja -j4
|
||||||
|
```
|
||||||
|
|
||||||
|
### Run the demo on Windows
|
||||||
|
|
||||||
|
This example uses yolo4_tiny.\
|
||||||
|
To run the object detection file create .rt file bu running:
|
||||||
|
```
|
||||||
|
.\test_yolo4tiny.exe
|
||||||
|
```
|
||||||
|
|
||||||
|
Once the rt file has been successfully create,run the demo using the following command:
|
||||||
|
```
|
||||||
|
.\demo.exe yolo4tiny_fp32.rt ..\demo\yolo_test.mp4 y
|
||||||
|
```
|
||||||
|
For general info on more demo paramters,check Run the demo section on top
|
||||||
|
To run the test_all_tests.sh on windows,use git bash or msys2
|
||||||
|
|
||||||
|
### FP16 inference windows
|
||||||
|
|
||||||
|
This is an untested feature on windows.To run the object detection demo with FP16 interference follow the below steps(example with yolo4tiny):
|
||||||
|
```
|
||||||
|
set TKDNN_MODE=FP16
|
||||||
|
del /f yolo4tiny_fp16.rt
|
||||||
|
.\test_yolo4tiny.exe
|
||||||
|
.\demo.exe yolo4tiny_fp16.rt ..\demo\yolo_test.mp4
|
||||||
|
```
|
||||||
|
|
||||||
|
### INT8 inference windows
|
||||||
|
To run object detection demo with INT8 (example with yolo4tiny):
|
||||||
|
```
|
||||||
|
set TKDNN_MODE=INT8
|
||||||
|
set TKDNN_CALIB_LABEL_PATH=..\demo\COCO_val2017\all_labels.txt
|
||||||
|
set TKDNN_CALIB_IMG_PATH=..\demo\COCO_val2017\all_images.txt
|
||||||
|
del /f yolo4tiny_int8.rt # be sure to delete(or move) old tensorRT files
|
||||||
|
.\test_yolo4tiny.exe # run the yolo test (is slow)
|
||||||
|
.\demo.exe yolo4tiny_int8.rt ..\demo\yolo_test.mp4 y
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
### Known issues with tkDNN on Windows
|
||||||
|
|
||||||
|
Mobilenet and Centernet demos work properly only when built with msvc 16.7 in Release Mode,when built in debug mode for the mentioned networks one might encounter opencv assert errors
|
||||||
|
|
||||||
|
All Darknet models work properly with demo using MSVC version(16.7-16.9)
|
||||||
|
|
||||||
|
It is recommended to use Nvidia Driver(465+),Cuda unknown errors have been observed when using older drivers on pascal(SM 61) devices.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
## References
|
## References
|
||||||
|
|
||||||
|
|||||||
+9
-4
@@ -1,7 +1,7 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
#include <signal.h>
|
#include <signal.h>
|
||||||
#include <stdlib.h> /* srand, rand */
|
#include <stdlib.h> /* srand, rand */
|
||||||
#include <unistd.h>
|
//#include <unistd.h>
|
||||||
#include <mutex>
|
#include <mutex>
|
||||||
|
|
||||||
#include "CenternetDetection.h"
|
#include "CenternetDetection.h"
|
||||||
@@ -22,10 +22,15 @@ int main(int argc, char *argv[]) {
|
|||||||
signal(SIGINT, sig_handler);
|
signal(SIGINT, sig_handler);
|
||||||
|
|
||||||
|
|
||||||
std::string net = "yolo3_berkeley.rt";
|
std::string net = "yolo4tiny_fp32.rt";
|
||||||
if(argc > 1)
|
if(argc > 1)
|
||||||
net = argv[1];
|
net = argv[1];
|
||||||
std::string input = "../demo/yolo_test.mp4";
|
#ifdef __linux__
|
||||||
|
std::string input = "../demo/yolo_test.mp4";
|
||||||
|
#elif _WIN32
|
||||||
|
std::string input = "..\\..\\..\\demo\\yolo_test.mp4";
|
||||||
|
#endif
|
||||||
|
|
||||||
if(argc > 2)
|
if(argc > 2)
|
||||||
input = argv[2];
|
input = argv[2];
|
||||||
char ntype = 'y';
|
char ntype = 'y';
|
||||||
@@ -131,7 +136,7 @@ int main(int argc, char *argv[]) {
|
|||||||
double mean = 0;
|
double mean = 0;
|
||||||
|
|
||||||
std::cout<<COL_GREENB<<"\n\nTime stats:\n";
|
std::cout<<COL_GREENB<<"\n\nTime stats:\n";
|
||||||
std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
|
std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
|
||||||
std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
|
std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
|
||||||
for(int i=0; i<detNN->stats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size();
|
for(int i=0; i<detNN->stats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size();
|
||||||
std::cout<<"Avg: "<<mean/n_batch<<" ms\t"<<1000/(mean/n_batch)<<" FPS\n"<<COL_END;
|
std::cout<<"Avg: "<<mean/n_batch<<" ms\t"<<1000/(mean/n_batch)<<" FPS\n"<<COL_END;
|
||||||
|
|||||||
@@ -2,7 +2,10 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
#include <signal.h>
|
#include <signal.h>
|
||||||
#include <stdlib.h> /* srand, rand */
|
#include <stdlib.h> /* srand, rand */
|
||||||
|
#ifdef __linux__
|
||||||
#include <unistd.h>
|
#include <unistd.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
#include <mutex>
|
#include <mutex>
|
||||||
#include "utils.h"
|
#include "utils.h"
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,10 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
#include <signal.h>
|
#include <signal.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
|
#ifdef __linux__
|
||||||
#include <unistd.h>
|
#include <unistd.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
#include <mutex>
|
#include <mutex>
|
||||||
#include "utils.h"
|
#include "utils.h"
|
||||||
|
|
||||||
@@ -14,7 +17,7 @@
|
|||||||
|
|
||||||
#include "tkdnn.h"
|
#include "tkdnn.h"
|
||||||
|
|
||||||
// #define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib.
|
//#define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib.
|
||||||
|
|
||||||
#ifdef OPENCV_CUDACONTRIB
|
#ifdef OPENCV_CUDACONTRIB
|
||||||
#include <opencv2/cudawarping.hpp>
|
#include <opencv2/cudawarping.hpp>
|
||||||
@@ -150,7 +153,6 @@ class DetectionNN {
|
|||||||
int x0, w, x1, y0, h, y1;
|
int x0, w, x1, y0, h, y1;
|
||||||
int objClass;
|
int objClass;
|
||||||
std::string det_class;
|
std::string det_class;
|
||||||
|
|
||||||
int baseline = 0;
|
int baseline = 0;
|
||||||
float font_scale = 0.5;
|
float font_scale = 0.5;
|
||||||
int thickness = 2;
|
int thickness = 2;
|
||||||
|
|||||||
@@ -1,7 +1,14 @@
|
|||||||
#include <iostream>
|
#include <iostream>
|
||||||
#include <signal.h>
|
#include <signal.h>
|
||||||
#include <stdlib.h> /* srand, rand */
|
#include <stdlib.h> /* srand, rand */
|
||||||
|
|
||||||
|
#ifdef __linux__
|
||||||
#include <unistd.h>
|
#include <unistd.h>
|
||||||
|
#elif _WIN32
|
||||||
|
#define _USE_MATH_DEFINES
|
||||||
|
#include <math.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
#include <mutex>
|
#include <mutex>
|
||||||
#include <Eigen/Dense>
|
#include <Eigen/Dense>
|
||||||
#include "utils.h"
|
#include "utils.h"
|
||||||
|
|||||||
@@ -11,8 +11,11 @@
|
|||||||
#include <fstream>
|
#include <fstream>
|
||||||
#include <iomanip>
|
#include <iomanip>
|
||||||
#include <signal.h>
|
#include <signal.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
|
#ifdef __linux__
|
||||||
#include <unistd.h>
|
#include <unistd.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
#include <mutex>
|
#include <mutex>
|
||||||
|
|
||||||
#include "NvInfer.h"
|
#include "NvInfer.h"
|
||||||
|
|||||||
@@ -6,6 +6,7 @@
|
|||||||
#include "Network.h"
|
#include "Network.h"
|
||||||
#include "Layer.h"
|
#include "Layer.h"
|
||||||
#include "NvInfer.h"
|
#include "NvInfer.h"
|
||||||
|
#include <memory>
|
||||||
|
|
||||||
namespace tk { namespace dnn {
|
namespace tk { namespace dnn {
|
||||||
|
|
||||||
@@ -60,6 +61,7 @@ public:
|
|||||||
#if NV_TENSORRT_MAJOR >= 6
|
#if NV_TENSORRT_MAJOR >= 6
|
||||||
nvinfer1::IBuilderConfig *configRT;
|
nvinfer1::IBuilderConfig *configRT;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
nvinfer1::ICudaEngine *engineRT;
|
nvinfer1::ICudaEngine *engineRT;
|
||||||
nvinfer1::IExecutionContext *contextRT;
|
nvinfer1::IExecutionContext *contextRT;
|
||||||
|
|
||||||
@@ -115,6 +117,9 @@ public:
|
|||||||
|
|
||||||
bool serialize(const char *filename);
|
bool serialize(const char *filename);
|
||||||
bool deserialize(const char *filename);
|
bool deserialize(const char *filename);
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
};
|
};
|
||||||
|
|
||||||
}}
|
}}
|
||||||
|
|||||||
@@ -52,8 +52,9 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, size);
|
tk::dnn::writeBUF(buf, size);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int size;
|
int size;
|
||||||
|
|||||||
@@ -4,57 +4,57 @@
|
|||||||
class ActivationLogisticRT : public IPlugin {
|
class ActivationLogisticRT : public IPlugin {
|
||||||
|
|
||||||
public:
|
public:
|
||||||
ActivationLogisticRT() {
|
ActivationLogisticRT() {
|
||||||
|
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
~ActivationLogisticRT(){
|
~ActivationLogisticRT(){
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
int getNbOutputs() const override {
|
int getNbOutputs() const override {
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||||
return inputs[0];
|
return inputs[0];
|
||||||
}
|
}
|
||||||
|
|
||||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||||
size = 1;
|
size = 1;
|
||||||
for(int i=0; i<outputDims[0].nbDims; i++)
|
for(int i=0; i<outputDims[0].nbDims; i++)
|
||||||
size *= outputDims[0].d[i];
|
size *= outputDims[0].d[i];
|
||||||
}
|
}
|
||||||
|
|
||||||
int initialize() override {
|
int initialize() override {
|
||||||
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
virtual void terminate() override {
|
virtual void terminate() override {
|
||||||
}
|
}
|
||||||
|
|
||||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||||
|
|
||||||
activationLOGISTICForward((dnnType*)reinterpret_cast<const dnnType*>(inputs[0]),
|
activationLOGISTICForward((dnnType*)reinterpret_cast<const dnnType*>(inputs[0]),
|
||||||
reinterpret_cast<dnnType*>(outputs[0]), batchSize*size, stream);
|
reinterpret_cast<dnnType*>(outputs[0]), batchSize*size, stream);
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
virtual size_t getSerializationSize() override {
|
virtual size_t getSerializationSize() override {
|
||||||
return 1*sizeof(int);
|
return 1*sizeof(int);
|
||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer);
|
||||||
tk::dnn::writeBUF(buf, size);
|
tk::dnn::writeBUF(buf, size);
|
||||||
}
|
}
|
||||||
|
|
||||||
int size;
|
int size;
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -52,8 +52,9 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, size);
|
tk::dnn::writeBUF(buf, size);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int size;
|
int size;
|
||||||
|
|||||||
@@ -51,9 +51,10 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, ceiling);
|
tk::dnn::writeBUF(buf, ceiling);
|
||||||
tk::dnn::writeBUF(buf, size);
|
tk::dnn::writeBUF(buf, size);
|
||||||
|
assert(buf = a + getSerializationSize());
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -52,8 +52,9 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, size);
|
tk::dnn::writeBUF(buf, size);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int size;
|
int size;
|
||||||
|
|||||||
@@ -116,7 +116,7 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, chunk_dim);
|
tk::dnn::writeBUF(buf, chunk_dim);
|
||||||
tk::dnn::writeBUF(buf, kh);
|
tk::dnn::writeBUF(buf, kh);
|
||||||
tk::dnn::writeBUF(buf, kw);
|
tk::dnn::writeBUF(buf, kw);
|
||||||
@@ -163,6 +163,7 @@ public:
|
|||||||
for(int i=0; i<dim_ones; i++)
|
for(int i=0; i<dim_ones; i++)
|
||||||
tk::dnn::writeBUF(buf, aus[i]);
|
tk::dnn::writeBUF(buf, aus[i]);
|
||||||
free(aus);
|
free(aus);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
cublasStatus_t stat;
|
cublasStatus_t stat;
|
||||||
|
|||||||
@@ -65,12 +65,13 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a = buf;
|
||||||
tk::dnn::writeBUF(buf, c);
|
tk::dnn::writeBUF(buf, c);
|
||||||
tk::dnn::writeBUF(buf, h);
|
tk::dnn::writeBUF(buf, h);
|
||||||
tk::dnn::writeBUF(buf, w);
|
tk::dnn::writeBUF(buf, w);
|
||||||
tk::dnn::writeBUF(buf, rows);
|
tk::dnn::writeBUF(buf, rows);
|
||||||
tk::dnn::writeBUF(buf, cols);
|
tk::dnn::writeBUF(buf, cols);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int c, h, w;
|
int c, h, w;
|
||||||
|
|||||||
@@ -55,7 +55,7 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
|
|
||||||
tk::dnn::writeBUF(buf, this->c);
|
tk::dnn::writeBUF(buf, this->c);
|
||||||
tk::dnn::writeBUF(buf, this->h);
|
tk::dnn::writeBUF(buf, this->h);
|
||||||
@@ -65,6 +65,7 @@ public:
|
|||||||
tk::dnn::writeBUF(buf, this->stride_W);
|
tk::dnn::writeBUF(buf, this->stride_W);
|
||||||
tk::dnn::writeBUF(buf, this->winSize);
|
tk::dnn::writeBUF(buf, this->winSize);
|
||||||
tk::dnn::writeBUF(buf, this->padding);
|
tk::dnn::writeBUF(buf, this->padding);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int n, c, h, w;
|
int n, c, h, w;
|
||||||
|
|||||||
@@ -73,13 +73,14 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, classes);
|
tk::dnn::writeBUF(buf, classes);
|
||||||
tk::dnn::writeBUF(buf, coords);
|
tk::dnn::writeBUF(buf, coords);
|
||||||
tk::dnn::writeBUF(buf, num);
|
tk::dnn::writeBUF(buf, num);
|
||||||
tk::dnn::writeBUF(buf, c);
|
tk::dnn::writeBUF(buf, c);
|
||||||
tk::dnn::writeBUF(buf, h);
|
tk::dnn::writeBUF(buf, h);
|
||||||
tk::dnn::writeBUF(buf, w);
|
tk::dnn::writeBUF(buf, w);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int c, h, w;
|
int c, h, w;
|
||||||
|
|||||||
@@ -52,11 +52,12 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, stride);
|
tk::dnn::writeBUF(buf, stride);
|
||||||
tk::dnn::writeBUF(buf, c);
|
tk::dnn::writeBUF(buf, c);
|
||||||
tk::dnn::writeBUF(buf, h);
|
tk::dnn::writeBUF(buf, h);
|
||||||
tk::dnn::writeBUF(buf, w);
|
tk::dnn::writeBUF(buf, w);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int c, h, w, stride;
|
int c, h, w, stride;
|
||||||
|
|||||||
@@ -50,11 +50,12 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a = buf;
|
||||||
tk::dnn::writeBUF(buf, n);
|
tk::dnn::writeBUF(buf, n);
|
||||||
tk::dnn::writeBUF(buf, c);
|
tk::dnn::writeBUF(buf, c);
|
||||||
tk::dnn::writeBUF(buf, h);
|
tk::dnn::writeBUF(buf, h);
|
||||||
tk::dnn::writeBUF(buf, w);
|
tk::dnn::writeBUF(buf, w);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int n, c, h, w;
|
int n, c, h, w;
|
||||||
|
|||||||
@@ -52,7 +52,7 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
|
|
||||||
tk::dnn::writeBUF(buf, o_c);
|
tk::dnn::writeBUF(buf, o_c);
|
||||||
tk::dnn::writeBUF(buf, o_h);
|
tk::dnn::writeBUF(buf, o_h);
|
||||||
@@ -61,6 +61,7 @@ public:
|
|||||||
tk::dnn::writeBUF(buf, i_c);
|
tk::dnn::writeBUF(buf, i_c);
|
||||||
tk::dnn::writeBUF(buf, i_h);
|
tk::dnn::writeBUF(buf, i_h);
|
||||||
tk::dnn::writeBUF(buf, i_w);
|
tk::dnn::writeBUF(buf, i_w);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int i_c, i_h, i_w, o_c, o_h, o_w;
|
int i_c, i_h, i_w, o_c, o_h, o_w;
|
||||||
|
|||||||
@@ -75,7 +75,7 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, groups);
|
tk::dnn::writeBUF(buf, groups);
|
||||||
tk::dnn::writeBUF(buf, group_id);
|
tk::dnn::writeBUF(buf, group_id);
|
||||||
tk::dnn::writeBUF(buf, in);
|
tk::dnn::writeBUF(buf, in);
|
||||||
@@ -85,6 +85,7 @@ public:
|
|||||||
tk::dnn::writeBUF(buf, c);
|
tk::dnn::writeBUF(buf, c);
|
||||||
tk::dnn::writeBUF(buf, h);
|
tk::dnn::writeBUF(buf, h);
|
||||||
tk::dnn::writeBUF(buf, w);
|
tk::dnn::writeBUF(buf, w);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
static const int MAX_INPUTS = 4;
|
static const int MAX_INPUTS = 4;
|
||||||
|
|||||||
@@ -59,13 +59,14 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, bc);
|
tk::dnn::writeBUF(buf, bc);
|
||||||
tk::dnn::writeBUF(buf, bh);
|
tk::dnn::writeBUF(buf, bh);
|
||||||
tk::dnn::writeBUF(buf, bw);
|
tk::dnn::writeBUF(buf, bw);
|
||||||
tk::dnn::writeBUF(buf, c);
|
tk::dnn::writeBUF(buf, c);
|
||||||
tk::dnn::writeBUF(buf, h);
|
tk::dnn::writeBUF(buf, h);
|
||||||
tk::dnn::writeBUF(buf, w);
|
tk::dnn::writeBUF(buf, w);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -54,11 +54,12 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, stride);
|
tk::dnn::writeBUF(buf, stride);
|
||||||
tk::dnn::writeBUF(buf, c);
|
tk::dnn::writeBUF(buf, c);
|
||||||
tk::dnn::writeBUF(buf, h);
|
tk::dnn::writeBUF(buf, h);
|
||||||
tk::dnn::writeBUF(buf, w);
|
tk::dnn::writeBUF(buf, w);
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int c, h, w, stride;
|
int c, h, w, stride;
|
||||||
|
|||||||
@@ -64,22 +64,23 @@ public:
|
|||||||
|
|
||||||
checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream));
|
checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream));
|
||||||
|
|
||||||
for (int b = 0; b < batchSize; ++b){
|
|
||||||
for(int n = 0; n < n_masks; ++n){
|
|
||||||
int index = entry_index(b, n*w*h, 0);
|
|
||||||
if (new_coords == 1){
|
|
||||||
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
|
|
||||||
}
|
|
||||||
else{
|
|
||||||
activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); //x,y
|
|
||||||
|
|
||||||
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
|
for (int b = 0; b < batchSize; ++b){
|
||||||
|
for(int n = 0; n < n_masks; ++n){
|
||||||
index = entry_index(b, n*w*h, 4);
|
int index = entry_index(b, n*w*h, 0);
|
||||||
activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream);
|
if (new_coords == 1){
|
||||||
}
|
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
|
||||||
}
|
}
|
||||||
}
|
else{
|
||||||
|
activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); //x,y
|
||||||
|
|
||||||
|
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
|
||||||
|
|
||||||
|
index = entry_index(b, n*w*h, 4);
|
||||||
|
activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
//std::cout<<"YOLO END\n";
|
//std::cout<<"YOLO END\n";
|
||||||
return 0;
|
return 0;
|
||||||
@@ -91,21 +92,25 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
virtual void serialize(void* buffer) override {
|
virtual void serialize(void* buffer) override {
|
||||||
char *buf = reinterpret_cast<char*>(buffer);
|
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||||
tk::dnn::writeBUF(buf, classes);
|
tk::dnn::writeBUF(buf, classes); std::cout << "Classes :" << classes << std::endl;
|
||||||
tk::dnn::writeBUF(buf, num);
|
tk::dnn::writeBUF(buf, num); std::cout << "Num : " << num << std::endl;
|
||||||
tk::dnn::writeBUF(buf, n_masks);
|
tk::dnn::writeBUF(buf, n_masks); std::cout << "N_Masks" << n_masks << std::endl;
|
||||||
tk::dnn::writeBUF(buf, scaleXY);
|
tk::dnn::writeBUF(buf, scaleXY); std::cout << "ScaleXY :" << scaleXY << std::endl;
|
||||||
tk::dnn::writeBUF(buf, nms_thresh);
|
tk::dnn::writeBUF(buf, nms_thresh); std::cout << "nms_thresh :" << nms_thresh << std::endl;
|
||||||
tk::dnn::writeBUF(buf, nms_kind);
|
tk::dnn::writeBUF(buf, nms_kind); std::cout << "nms_kind : " << nms_kind << std::endl;
|
||||||
tk::dnn::writeBUF(buf, new_coords);
|
tk::dnn::writeBUF(buf, new_coords); std::cout << "new_coords : " << new_coords << std::endl;
|
||||||
tk::dnn::writeBUF(buf, c);
|
tk::dnn::writeBUF(buf, c); std::cout << "C : " << c << std::endl;
|
||||||
tk::dnn::writeBUF(buf, h);
|
tk::dnn::writeBUF(buf, h); std::cout << "H : " << h << std::endl;
|
||||||
tk::dnn::writeBUF(buf, w);
|
tk::dnn::writeBUF(buf, w); std::cout << "C : " << c << std::endl;
|
||||||
for(int i=0; i<n_masks; i++)
|
for (int i = 0; i < n_masks; i++)
|
||||||
tk::dnn::writeBUF(buf, mask[i]);
|
{
|
||||||
for(int i=0; i<n_masks*2*num; i++)
|
tk::dnn::writeBUF(buf, mask[i]); std::cout << "mask[i] : " << mask[i] << std::endl;
|
||||||
tk::dnn::writeBUF(buf, bias[i]);
|
}
|
||||||
|
for (int i = 0; i < n_masks * 2 * num; i++)
|
||||||
|
{
|
||||||
|
tk::dnn::writeBUF(buf, bias[i]); std::cout << "bias[i] : " << bias[i] << std::endl;
|
||||||
|
}
|
||||||
|
|
||||||
// save classes names
|
// save classes names
|
||||||
for(int i=0; i<classes; i++) {
|
for(int i=0; i<classes; i++) {
|
||||||
@@ -115,6 +120,7 @@ public:
|
|||||||
tk::dnn::writeBUF(buf, tmp[j]);
|
tk::dnn::writeBUF(buf, tmp[j]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
assert(buf == a + getSerializationSize());
|
||||||
}
|
}
|
||||||
|
|
||||||
int c, h, w;
|
int c, h, w;
|
||||||
|
|||||||
@@ -29,7 +29,8 @@ int testInference(std::vector<std::string> input_bins, std::vector<std::string>
|
|||||||
readBinaryFile(input_bins[0], net->input_dim.tot(), &input_h, &data);
|
readBinaryFile(input_bins[0], net->input_dim.tot(), &input_h, &data);
|
||||||
|
|
||||||
// outputs
|
// outputs
|
||||||
dnnType *cudnn_out[outputs.size()], *rt_out[outputs.size()];
|
//dnnType *cudnn_out[outputs.size()], *rt_out[outputs.size()];
|
||||||
|
std::vector<dnnType *> cudnn_out,rt_out;
|
||||||
|
|
||||||
tk::dnn::dataDim_t dim1 = net->input_dim; //input dim
|
tk::dnn::dataDim_t dim1 = net->input_dim; //input dim
|
||||||
printCenteredTitle(" CUDNN inference ", '=', 30); {
|
printCenteredTitle(" CUDNN inference ", '=', 30); {
|
||||||
@@ -39,7 +40,7 @@ int testInference(std::vector<std::string> input_bins, std::vector<std::string>
|
|||||||
TKDNN_TSTOP
|
TKDNN_TSTOP
|
||||||
dim1.print();
|
dim1.print();
|
||||||
}
|
}
|
||||||
for(int i=0; i<outputs.size(); i++) cudnn_out[i] = outputs[i]->dstData;
|
for(int i=0; i<outputs.size(); i++) cudnn_out.push_back(outputs[i]->dstData);
|
||||||
|
|
||||||
if(netRT != nullptr) {
|
if(netRT != nullptr) {
|
||||||
tk::dnn::dataDim_t dim2 = net->input_dim;
|
tk::dnn::dataDim_t dim2 = net->input_dim;
|
||||||
@@ -50,7 +51,7 @@ int testInference(std::vector<std::string> input_bins, std::vector<std::string>
|
|||||||
TKDNN_TSTOP
|
TKDNN_TSTOP
|
||||||
dim2.print();
|
dim2.print();
|
||||||
}
|
}
|
||||||
for(int i=0; i<outputs.size(); i++) rt_out[i] = (dnnType*)netRT->buffersRT[i+1];
|
for(int i=0; i<outputs.size(); i++) rt_out.push_back((dnnType*)netRT->buffersRT[i+1]);
|
||||||
}
|
}
|
||||||
|
|
||||||
int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0;
|
int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0;
|
||||||
|
|||||||
@@ -12,8 +12,12 @@
|
|||||||
#include <cublas_v2.h>
|
#include <cublas_v2.h>
|
||||||
#include <cudnn.h>
|
#include <cudnn.h>
|
||||||
|
|
||||||
|
#ifdef __linux__
|
||||||
#include <unistd.h>
|
#include <unistd.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
#include <ios>
|
#include <ios>
|
||||||
|
#include <chrono>
|
||||||
|
|
||||||
|
|
||||||
#define dnnType float
|
#define dnnType float
|
||||||
@@ -39,6 +43,7 @@
|
|||||||
#define TKDNN_VERBOSE 0
|
#define TKDNN_VERBOSE 0
|
||||||
|
|
||||||
// Simple Timer
|
// Simple Timer
|
||||||
|
#ifdef __linux__
|
||||||
#define TKDNN_TSTART timespec start, end; \
|
#define TKDNN_TSTART timespec start, end; \
|
||||||
clock_gettime(CLOCK_MONOTONIC, &start);
|
clock_gettime(CLOCK_MONOTONIC, &start);
|
||||||
|
|
||||||
@@ -48,6 +53,14 @@
|
|||||||
if(show) std::cout<<col<<"Time:"<<std::setw(16)<<t_ns<<" ms\n"<<COL_END;
|
if(show) std::cout<<col<<"Time:"<<std::setw(16)<<t_ns<<" ms\n"<<COL_END;
|
||||||
|
|
||||||
#define TKDNN_TSTOP TKDNN_TSTOP_C(COL_CYANB, TKDNN_VERBOSE)
|
#define TKDNN_TSTOP TKDNN_TSTOP_C(COL_CYANB, TKDNN_VERBOSE)
|
||||||
|
#elif _WIN32
|
||||||
|
#define TKDNN_TSTART auto start = std::chrono::high_resolution_clock::now();
|
||||||
|
#define TKDNN_TSTOP auto stop = std::chrono::high_resolution_clock::now(); \
|
||||||
|
std::chrono::duration<double> duration = stop -start; \
|
||||||
|
auto time_ms = std::chrono::duration_cast<std::chrono::milliseconds>(duration);\
|
||||||
|
double t_ns = time_ms.count();
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
/********************************************************
|
/********************************************************
|
||||||
* Prints the error message, and exits
|
* Prints the error message, and exits
|
||||||
|
|||||||
@@ -0,0 +1,39 @@
|
|||||||
|
import os
|
||||||
|
import urllib.request as dowReq
|
||||||
|
import zipfile
|
||||||
|
|
||||||
|
val = input("Enter BDD or COCO :")
|
||||||
|
if(val == "COCO"):
|
||||||
|
url = "https://cloud.hipert.unimore.it/s/LNxBDk4wzqXPL8c/download"
|
||||||
|
lib = "..\demo\COCO_val2017"
|
||||||
|
lib_zip = "COCO_val2017.zip"
|
||||||
|
elif(val == "BDD"):
|
||||||
|
url = "https://cloud.hipert.unimore.it/s/bikqk3FzCq2tg4D/download"
|
||||||
|
lib = "..\demo\BDD100k_val"
|
||||||
|
lib_zip = "BDD100k_val.zip"
|
||||||
|
|
||||||
|
dowReq.urlretrieve(url,lib_zip)
|
||||||
|
|
||||||
|
with zipfile.ZipFile(lib_zip,'r') as zip_ref:
|
||||||
|
zip_ref.extractall(lib)
|
||||||
|
|
||||||
|
labelFolder = lib + "\labels"
|
||||||
|
imageFolder = lib + "\images"
|
||||||
|
|
||||||
|
file1 = open(".\\..\\demo\\all_labels.txt","a")
|
||||||
|
path1 = os.path.realpath(labelFolder)
|
||||||
|
for file in os.listdir(labelFolder):
|
||||||
|
valTemp = path1 + "\\" + file
|
||||||
|
valTemp = valTemp + '\n'
|
||||||
|
file1.write(valTemp)
|
||||||
|
file1.close()
|
||||||
|
|
||||||
|
file2 = open(".\\..\\demo\\all_images.txt","a")
|
||||||
|
path2 = os.path.realpath(imageFolder)
|
||||||
|
for file in os.listdir(imageFolder):
|
||||||
|
pathtemp = path2 + "\\" + file
|
||||||
|
pathtemp = pathtemp + '\n'
|
||||||
|
file2.write(pathtemp)
|
||||||
|
file2.close()
|
||||||
|
|
||||||
|
print("Completed")
|
||||||
+9
-4
@@ -87,17 +87,22 @@ LSTM::LSTM( Network *net, int hiddensize, bool returnSeq, std::string fname_weig
|
|||||||
checkCUDNN(cudnnCreateRNNDescriptor(&rnnDesc));
|
checkCUDNN(cudnnCreateRNNDescriptor(&rnnDesc));
|
||||||
|
|
||||||
#if CUDNN_MAJOR > 7
|
#if CUDNN_MAJOR > 7
|
||||||
checkCUDNN(cudnnSetRNNDescriptor_v6(net->cudnnHandle,
|
checkCUDNN(cudnnSetRNNDescriptor_v6(net->cudnnHandle,rnnDesc, stateSize, numLayers, dropoutDesc,
|
||||||
|
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
|
||||||
|
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
|
||||||
|
cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL,
|
||||||
|
cudnnRNNMode_t::CUDNN_LSTM,
|
||||||
|
cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD,
|
||||||
|
net->dataType));
|
||||||
#else
|
#else
|
||||||
checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle,
|
checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle,rnnDesc, stateSize, numLayers, dropoutDesc,
|
||||||
#endif
|
|
||||||
rnnDesc, stateSize, numLayers, dropoutDesc,
|
|
||||||
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
|
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
|
||||||
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
|
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
|
||||||
cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL,
|
cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL,
|
||||||
cudnnRNNMode_t::CUDNN_LSTM,
|
cudnnRNNMode_t::CUDNN_LSTM,
|
||||||
cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD,
|
cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD,
|
||||||
net->dataType));
|
net->dataType));
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
// Get temp space sizes
|
// Get temp space sizes
|
||||||
|
|||||||
+82
-36
@@ -139,7 +139,8 @@ NetworkRT::NetworkRT(Network *net, const char *name) {
|
|||||||
#if NV_TENSORRT_MAJOR >= 6
|
#if NV_TENSORRT_MAJOR >= 6
|
||||||
engineRT = builderRT->buildEngineWithConfig(*networkRT, *configRT);
|
engineRT = builderRT->buildEngineWithConfig(*networkRT, *configRT);
|
||||||
#else
|
#else
|
||||||
engineRT = builderRT->buildCudaEngine(*networkRT);
|
//engineRT = builderRT->buildCudaEngine(*networkRT);
|
||||||
|
engineRT = std::shared_ptr<nvinfer1::ICudaEngine>(builderRT->buildCudaEngine(*networkRT));
|
||||||
#endif
|
#endif
|
||||||
if(engineRT == nullptr)
|
if(engineRT == nullptr)
|
||||||
FatalError("cloud not build cuda engine")
|
FatalError("cloud not build cuda engine")
|
||||||
@@ -567,7 +568,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, DeformConv2d *l) {
|
|||||||
IPluginLayer *lRT = networkRT->addPlugin(inputs, 2, *plugin);
|
IPluginLayer *lRT = networkRT->addPlugin(inputs, 2, *plugin);
|
||||||
checkNULL(lRT);
|
checkNULL(lRT);
|
||||||
lRT->setName( ("Deformable" + std::to_string(l->id)).c_str() );
|
lRT->setName( ("Deformable" + std::to_string(l->id)).c_str() );
|
||||||
delete(inputs);
|
delete[](inputs);
|
||||||
// batchnorm
|
// batchnorm
|
||||||
void *bias_b, *power_b, *mean_b, *variance_b, *scales_b;
|
void *bias_b, *power_b, *mean_b, *variance_b, *scales_b;
|
||||||
if(dtRT == DataType::kHALF) {
|
if(dtRT == DataType::kHALF) {
|
||||||
@@ -644,7 +645,7 @@ bool NetworkRT::deserialize(const char *filename) {
|
|||||||
|
|
||||||
|
|
||||||
IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialData, size_t serialLength) {
|
IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialData, size_t serialLength) {
|
||||||
const char * buf = reinterpret_cast<const char*>(serialData);
|
const char * buf = reinterpret_cast<const char*>(serialData),*bufCheck = buf;
|
||||||
|
|
||||||
std::string name(layerName);
|
std::string name(layerName);
|
||||||
//std::cout<<name<<std::endl;
|
//std::cout<<name<<std::endl;
|
||||||
@@ -652,11 +653,18 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
|
|||||||
if(name.find("ActivationLeaky") == 0) {
|
if(name.find("ActivationLeaky") == 0) {
|
||||||
ActivationLeakyRT *a = new ActivationLeakyRT();
|
ActivationLeakyRT *a = new ActivationLeakyRT();
|
||||||
a->size = readBUF<int>(buf);
|
a->size = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return a;
|
return a;
|
||||||
}
|
}
|
||||||
if(name.find("ActivationMish") == 0) {
|
if(name.find("ActivationMish") == 0) {
|
||||||
ActivationMishRT *a = new ActivationMishRT();
|
ActivationMishRT *a = new ActivationMishRT();
|
||||||
a->size = readBUF<int>(buf);
|
a->size = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
|
return a;
|
||||||
|
}
|
||||||
|
if(name.find("ActivationLogistic") == 0) {
|
||||||
|
ActivationLogisticRT *a = new ActivationLogisticRT();
|
||||||
|
a->size = readBUF<int>(buf);
|
||||||
return a;
|
return a;
|
||||||
}
|
}
|
||||||
if(name.find("ActivationLogistic") == 0) {
|
if(name.find("ActivationLogistic") == 0) {
|
||||||
@@ -665,27 +673,33 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
|
|||||||
return a;
|
return a;
|
||||||
}
|
}
|
||||||
if(name.find("ActivationCReLU") == 0) {
|
if(name.find("ActivationCReLU") == 0) {
|
||||||
ActivationReLUCeiling *a = new ActivationReLUCeiling(readBUF<float>(buf));
|
float activationReluTemp = readBUF<float>(buf);
|
||||||
|
ActivationReLUCeiling* a = new ActivationReLUCeiling(activationReluTemp);
|
||||||
a->size = readBUF<int>(buf);
|
a->size = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return a;
|
return a;
|
||||||
}
|
}
|
||||||
|
|
||||||
if(name.find("Region") == 0) {
|
if(name.find("Region") == 0) {
|
||||||
RegionRT *r = new RegionRT(readBUF<int>(buf), //classes
|
int classesTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //coords
|
int coordsTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf)); //num
|
int numTemp = readBUF<int>(buf);
|
||||||
|
RegionRT* r = new RegionRT(classesTemp, coordsTemp, numTemp);
|
||||||
|
|
||||||
r->c = readBUF<int>(buf);
|
r->c = readBUF<int>(buf);
|
||||||
r->h = readBUF<int>(buf);
|
r->h = readBUF<int>(buf);
|
||||||
r->w = readBUF<int>(buf);
|
r->w = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
if(name.find("Reorg") == 0) {
|
if(name.find("Reorg") == 0) {
|
||||||
ReorgRT *r = new ReorgRT(readBUF<int>(buf)); //stride
|
int strideTemp = readBUF<int>(buf);
|
||||||
|
ReorgRT *r = new ReorgRT(strideTemp);
|
||||||
r->c = readBUF<int>(buf);
|
r->c = readBUF<int>(buf);
|
||||||
r->h = readBUF<int>(buf);
|
r->h = readBUF<int>(buf);
|
||||||
r->w = readBUF<int>(buf);
|
r->w = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -701,27 +715,34 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
|
|||||||
r->h = readBUF<int>(buf);
|
r->h = readBUF<int>(buf);
|
||||||
r->w = readBUF<int>(buf);
|
r->w = readBUF<int>(buf);
|
||||||
return r;
|
return r;
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
}
|
}
|
||||||
|
|
||||||
if(name.find("Pooling") == 0) {
|
if(name.find("Pooling") == 0) {
|
||||||
MaxPoolFixedSizeRT *r = new MaxPoolFixedSizeRT( readBUF<int>(buf), //c
|
int cTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //h
|
int hTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //w
|
int wTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //n
|
int nTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //strideH
|
int strideHTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //strideW
|
int strideWTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //winSize
|
int winSizeTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf)); //padding
|
int paddingTemp = readBUF<int>(buf);
|
||||||
|
|
||||||
|
MaxPoolFixedSizeRT* r = new MaxPoolFixedSizeRT(cTemp, hTemp, wTemp, nTemp, strideHTemp, strideWTemp, winSizeTemp, paddingTemp);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
if(name.find("Resize") == 0) {
|
if(name.find("Resize") == 0) {
|
||||||
ResizeLayerRT *r = new ResizeLayerRT(readBUF<int>(buf), //o_c
|
int o_cTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //o_h
|
int o_hTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf)); //o_w
|
int o_wTemp = readBUF<int>(buf);
|
||||||
|
ResizeLayerRT* r = new ResizeLayerRT(o_cTemp, o_hTemp, o_wTemp);
|
||||||
|
|
||||||
r->i_c = readBUF<int>(buf);
|
r->i_c = readBUF<int>(buf);
|
||||||
r->i_h = readBUF<int>(buf);
|
r->i_h = readBUF<int>(buf);
|
||||||
r->i_w = readBUF<int>(buf);
|
r->i_w = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -732,6 +753,7 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
|
|||||||
r->w = readBUF<int>(buf);
|
r->w = readBUF<int>(buf);
|
||||||
r->rows = readBUF<int>(buf);
|
r->rows = readBUF<int>(buf);
|
||||||
r->cols = readBUF<int>(buf);
|
r->cols = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -743,20 +765,25 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
|
|||||||
new_dim.h = readBUF<int>(buf);
|
new_dim.h = readBUF<int>(buf);
|
||||||
new_dim.w = readBUF<int>(buf);
|
new_dim.w = readBUF<int>(buf);
|
||||||
ReshapeRT *r = new ReshapeRT(new_dim);
|
ReshapeRT *r = new ReshapeRT(new_dim);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
|
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
if(name.find("Yolo") == 0) {
|
if(name.find("Yolo") == 0) {
|
||||||
YoloRT *r = new YoloRT(readBUF<int>(buf), //classes
|
|
||||||
readBUF<int>(buf), //num
|
int classes_temp = readBUF<int>(buf);
|
||||||
nullptr, //yolo
|
int num_temp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), //n_masks
|
int n_masks_temp = readBUF<int>(buf);
|
||||||
readBUF<float>(buf), //scale_xy
|
float scale_xy_temp = readBUF<float>(buf);
|
||||||
readBUF<float>(buf), //nms_thresh
|
float nms_thresh_temp = readBUF<float>(buf);
|
||||||
readBUF<int>(buf), //nms_kind
|
int nms_kind_temp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf) //new_coords
|
int new_coords_temp = readBUF<int>(buf);
|
||||||
);
|
|
||||||
|
YoloRT *r = new YoloRT(classes_temp,num_temp,nullptr,n_masks_temp,scale_xy_temp,nms_thresh_temp,nms_kind_temp,new_coords_temp);
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
r->c = readBUF<int>(buf);
|
r->c = readBUF<int>(buf);
|
||||||
r->h = readBUF<int>(buf);
|
r->h = readBUF<int>(buf);
|
||||||
r->w = readBUF<int>(buf);
|
r->w = readBUF<int>(buf);
|
||||||
@@ -773,36 +800,54 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
|
|||||||
tmp[j] = readBUF<char>(buf);
|
tmp[j] = readBUF<char>(buf);
|
||||||
r->classesNames[i] = std::string(tmp);
|
r->classesNames[i] = std::string(tmp);
|
||||||
}
|
}
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
|
|
||||||
yolos[n_yolos++] = r;
|
yolos[n_yolos++] = r;
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
if(name.find("Upsample") == 0) {
|
if(name.find("Upsample") == 0) {
|
||||||
UpsampleRT *r = new UpsampleRT(readBUF<int>(buf)); //stride
|
int strideTemp = readBUF<int>(buf);
|
||||||
|
UpsampleRT* r = new UpsampleRT(strideTemp);
|
||||||
r->c = readBUF<int>(buf);
|
r->c = readBUF<int>(buf);
|
||||||
r->h = readBUF<int>(buf);
|
r->h = readBUF<int>(buf);
|
||||||
r->w = readBUF<int>(buf);
|
r->w = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
if(name.find("Route") == 0) {
|
if(name.find("Route") == 0) {
|
||||||
RouteRT *r = new RouteRT(readBUF<int>(buf),readBUF<int>(buf));
|
int groupsTemp = readBUF<int>(buf);
|
||||||
|
int group_idTemp = readBUF<int>(buf);
|
||||||
|
RouteRT* r = new RouteRT(groupsTemp, group_idTemp);
|
||||||
r->in = readBUF<int>(buf);
|
r->in = readBUF<int>(buf);
|
||||||
for(int i=0; i<RouteRT::MAX_INPUTS; i++)
|
for(int i=0; i<RouteRT::MAX_INPUTS; i++)
|
||||||
r->c_in[i] = readBUF<int>(buf);
|
r->c_in[i] = readBUF<int>(buf);
|
||||||
r->c = readBUF<int>(buf);
|
r->c = readBUF<int>(buf);
|
||||||
r->h = readBUF<int>(buf);
|
r->h = readBUF<int>(buf);
|
||||||
r->w = readBUF<int>(buf);
|
r->w = readBUF<int>(buf);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
if(name.find("Deformable") == 0) {
|
if(name.find("Deformable") == 0) {
|
||||||
DeformableConvRT *r = new DeformableConvRT(readBUF<int>(buf), readBUF<int>(buf), readBUF<int>(buf),
|
int chuck_dimTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), readBUF<int>(buf), readBUF<int>(buf),
|
int khTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf), readBUF<int>(buf),
|
int kwTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf),readBUF<int>(buf),readBUF<int>(buf),readBUF<int>(buf),
|
int shTemp = readBUF<int>(buf);
|
||||||
readBUF<int>(buf),readBUF<int>(buf),readBUF<int>(buf),readBUF<int>(buf),
|
int swTemp = readBUF<int>(buf);
|
||||||
nullptr);
|
int phTemp = readBUF<int>(buf);
|
||||||
|
int pwTemp = readBUF<int>(buf);
|
||||||
|
int deformableGroupTemp = readBUF<int>(buf);
|
||||||
|
int i_nTemp = readBUF<int>(buf);
|
||||||
|
int i_cTemp = readBUF<int>(buf);
|
||||||
|
int i_hTemp = readBUF<int>(buf);
|
||||||
|
int i_wTemp = readBUF<int>(buf);
|
||||||
|
int o_nTemp = readBUF<int>(buf);
|
||||||
|
int o_cTemp = readBUF<int>(buf);
|
||||||
|
int o_hTemp = readBUF<int>(buf);
|
||||||
|
int o_wTemp = readBUF<int>(buf);
|
||||||
|
|
||||||
|
DeformableConvRT* r = new DeformableConvRT(chuck_dimTemp, khTemp, kwTemp, shTemp, swTemp, phTemp, pwTemp, deformableGroupTemp, i_nTemp, i_cTemp, i_hTemp, i_wTemp, o_nTemp, o_cTemp, o_hTemp, o_wTemp, nullptr);
|
||||||
dnnType *aus = new dnnType[r->chunk_dim*2];
|
dnnType *aus = new dnnType[r->chunk_dim*2];
|
||||||
for(int i=0; i<r->chunk_dim*2; i++)
|
for(int i=0; i<r->chunk_dim*2; i++)
|
||||||
aus[i] = readBUF<dnnType>(buf);
|
aus[i] = readBUF<dnnType>(buf);
|
||||||
@@ -833,6 +878,7 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa
|
|||||||
aus[i] = readBUF<dnnType>(buf);
|
aus[i] = readBUF<dnnType>(buf);
|
||||||
checkCuda( cudaMemcpy(r->ones_d2, aus, sizeof(dnnType)*r->dim_ones, cudaMemcpyHostToDevice) );
|
checkCuda( cudaMemcpy(r->ones_d2, aus, sizeof(dnnType)*r->dim_ones, cudaMemcpyHostToDevice) );
|
||||||
free(aus);
|
free(aus);
|
||||||
|
assert(buf == bufCheck + serialLength);
|
||||||
return r;
|
return r;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+5
-5
@@ -9,6 +9,7 @@
|
|||||||
#include "Layer.h"
|
#include "Layer.h"
|
||||||
#include "kernels.h"
|
#include "kernels.h"
|
||||||
|
|
||||||
|
|
||||||
namespace tk { namespace dnn {
|
namespace tk { namespace dnn {
|
||||||
|
|
||||||
Yolo::Yolo(Network *net, int classes, int num, std::string fname_weights, int n_masks, float scale_xy, double nms_thresh, nmsKind_t nsm_kind, int new_coords) :
|
Yolo::Yolo(Network *net, int classes, int num, std::string fname_weights, int n_masks, float scale_xy, double nms_thresh, nmsKind_t nsm_kind, int new_coords) :
|
||||||
@@ -95,7 +96,6 @@ dnnType* Yolo::infer(dataDim_t &dim, dnnType* srcData) {
|
|||||||
activationLOGISTICForward(srcData + index, dstData + index, 2*dim.w*dim.h);
|
activationLOGISTICForward(srcData + index, dstData + index, 2*dim.w*dim.h);
|
||||||
|
|
||||||
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
|
if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1);
|
||||||
|
|
||||||
index = entry_index(b, n*dim.w*dim.h, 4, classes, input_dim, output_dim);
|
index = entry_index(b, n*dim.w*dim.h, 4, classes, input_dim, output_dim);
|
||||||
activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*dim.w*dim.h);
|
activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*dim.w*dim.h);
|
||||||
}
|
}
|
||||||
@@ -212,10 +212,10 @@ float yolo_box_iou(Yolo::box a, Yolo::box b)
|
|||||||
}
|
}
|
||||||
|
|
||||||
void box_c(const Yolo::box a, const Yolo::box b, float& top, float& bot, float& left, float& right) {
|
void box_c(const Yolo::box a, const Yolo::box b, float& top, float& bot, float& left, float& right) {
|
||||||
top = std::min(a.y - a.h / 2, b.y - b.h / 2);
|
top = (std::min)(a.y - a.h / 2, b.y - b.h / 2);
|
||||||
bot = std::max(a.y + a.h / 2, b.y + b.h / 2);
|
bot = (std::max)(a.y + a.h / 2, b.y + b.h / 2);
|
||||||
left = std::min(a.x - a.w / 2, b.x - b.w / 2);
|
left = (std::min)(a.x - a.w / 2, b.x - b.w / 2);
|
||||||
right = std::max(a.x + a.w / 2, b.x + b.w / 2);
|
right = (std::max)(a.x + a.w / 2, b.x + b.w / 2);
|
||||||
}
|
}
|
||||||
|
|
||||||
// https://github.com/Zzh-tju/DIoU-darknet
|
// https://github.com/Zzh-tju/DIoU-darknet
|
||||||
|
|||||||
@@ -94,9 +94,10 @@ void Yolo3Detection::preprocess(cv::Mat &frame, const int bi){
|
|||||||
void Yolo3Detection::postprocess(const int bi, const bool mAP){
|
void Yolo3Detection::postprocess(const int bi, const bool mAP){
|
||||||
|
|
||||||
//get yolo outputs
|
//get yolo outputs
|
||||||
dnnType *rt_out[netRT->pluginFactory->n_yolos];
|
std::vector<float *> rt_out;
|
||||||
for(int i=0; i<netRT->pluginFactory->n_yolos; i++)
|
//dnnType *rt_out[netRT->pluginFactory->n_yolos];
|
||||||
rt_out[i] = (dnnType*)netRT->buffersRT[i+1] + netRT->buffersDIM[i+1].tot()*bi;
|
for(int i=0; i<netRT->pluginFactory->n_yolos; i++)
|
||||||
|
rt_out.push_back((dnnType*)netRT->buffersRT[i+1] + netRT->buffersDIM[i+1].tot()*bi);
|
||||||
|
|
||||||
float x_ratio = float(originalSize[bi].width) / float(netRT->input_dim.w);
|
float x_ratio = float(originalSize[bi].width) / float(netRT->input_dim.w);
|
||||||
float y_ratio = float(originalSize[bi].height) / float(netRT->input_dim.h);
|
float y_ratio = float(originalSize[bi].height) / float(netRT->input_dim.h);
|
||||||
|
|||||||
@@ -18,7 +18,7 @@ inline int GET_BLOCKS(const int N)
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
__device__ float dmcn_im2col_bilinear(const float *bottom_data, const int data_width,
|
__device__ __host__ float dmcn_im2col_bilinear(const float *bottom_data, const int data_width,
|
||||||
const int height, const int width, float h, float w) {
|
const int height, const int width, float h, float w) {
|
||||||
int h_low = floor(h);
|
int h_low = floor(h);
|
||||||
int w_low = floor(w);
|
int w_low = floor(w);
|
||||||
|
|||||||
+15
-2
@@ -23,14 +23,23 @@ bool fileExist(const char *fname) {
|
|||||||
void downloadWeightsifDoNotExist(const std::string& input_bin, const std::string& test_folder, const std::string& weights_url){
|
void downloadWeightsifDoNotExist(const std::string& input_bin, const std::string& test_folder, const std::string& weights_url){
|
||||||
if(!fileExist(input_bin.c_str())){
|
if(!fileExist(input_bin.c_str())){
|
||||||
std::string mkdir_cmd = "mkdir " + test_folder;
|
std::string mkdir_cmd = "mkdir " + test_folder;
|
||||||
std::string wget_cmd = "wget " + weights_url + " -O " + test_folder + "/weights.zip";
|
std::string wget_cmd = "curl " + weights_url + " --output " + test_folder + "/weights.zip";
|
||||||
|
#ifdef __linux__
|
||||||
std::string unzip_cmd = "unzip " + test_folder + "/weights.zip -d" + test_folder;
|
std::string unzip_cmd = "unzip " + test_folder + "/weights.zip -d" + test_folder;
|
||||||
std::string rm_cmd = "rm " + test_folder + "/weights.zip";
|
std::string rm_cmd = "rm " + test_folder + "/weights.zip";
|
||||||
|
|
||||||
|
#elif _WIN32
|
||||||
|
|
||||||
|
std::string unzip_cmd = "7z x " + test_folder + "/weights.zip -o" + test_folder;
|
||||||
|
#endif
|
||||||
int err = 0;
|
int err = 0;
|
||||||
err = system(mkdir_cmd.c_str());
|
err = system(mkdir_cmd.c_str());
|
||||||
err = system(wget_cmd.c_str());
|
err = system(wget_cmd.c_str());
|
||||||
err = system(unzip_cmd.c_str());
|
err = system(unzip_cmd.c_str());
|
||||||
|
#ifdef __linux__
|
||||||
err = system(rm_cmd.c_str());
|
err = system(rm_cmd.c_str());
|
||||||
|
#endif
|
||||||
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -191,8 +200,12 @@ void getMemUsage(double& vm_usage_kb, double& resident_set_kb){
|
|||||||
>> O >> itrealvalue >> starttime >> vsize >> rss;
|
>> O >> itrealvalue >> starttime >> vsize >> rss;
|
||||||
|
|
||||||
stat_stream.close();
|
stat_stream.close();
|
||||||
|
#ifdef __linux__
|
||||||
long page_size_kb = sysconf(_SC_PAGE_SIZE) / 1024; // in case x86-64 is configured to use 2MB pages
|
long page_size_kb = sysconf(_SC_PAGE_SIZE) / 1024; // in case x86-64 is configured to use 2MB pages
|
||||||
|
#elif _WIN32
|
||||||
|
long page_size_kb = 4096/1024;
|
||||||
|
#endif
|
||||||
|
|
||||||
vm_usage_kb = vsize / 1024.0;
|
vm_usage_kb = vsize / 1024.0;
|
||||||
resident_set_kb = rss * page_size_kb;
|
resident_set_kb = rss * page_size_kb;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -19,8 +19,6 @@ int main() {
|
|||||||
std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names";
|
std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names";
|
||||||
downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download");
|
downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download");
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
// parse darknet network
|
// parse darknet network
|
||||||
tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path);
|
tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path);
|
||||||
net->print();
|
net->print();
|
||||||
@@ -33,4 +31,4 @@ int main() {
|
|||||||
delete net;
|
delete net;
|
||||||
delete netRT;
|
delete netRT;
|
||||||
return ret;
|
return ret;
|
||||||
}
|
}
|
||||||
Reference in New Issue
Block a user