Compare commits
458 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c95c2dbfbc | |||
| d4f7b4ad8b | |||
| 5e71b99265 | |||
| fa9db167b8 | |||
| 30098ca6b4 | |||
| df31375b67 | |||
| 69bb7370a5 | |||
| 7c0620e391 | |||
| c690d537f1 | |||
| 40266a6c32 | |||
| 480b5a9c5a | |||
| 3bcc32ffdc | |||
| b3bc93693f | |||
| ef564f134d | |||
| f996659845 | |||
| afdad8e661 | |||
| 71366befc2 | |||
| 3e86671c50 | |||
| 00f06f7bcc | |||
| 53725c88a9 | |||
| 6a133d8dec | |||
| decd73d298 | |||
| 907df27e07 | |||
| 19e41a8b99 | |||
| 061bc79a69 | |||
| bcf0c4eab3 | |||
| e1eac2d42a | |||
| 3023926695 | |||
| 4d8f99b441 | |||
| cabebc95a2 | |||
| 3b58fbcb82 | |||
| f189efcbb1 | |||
| 7298dcfb2f | |||
| b75cecb105 | |||
| ba022663f1 | |||
| 6837644eb2 | |||
| dcf4054bc6 | |||
| 04de9908a6 | |||
| 9cbac460bc | |||
| 24cdb4c4a7 | |||
| a4781244f4 | |||
| bbae618118 | |||
| 0707c26bbd | |||
| 9cecc5051a | |||
| be5864748a | |||
| 75c3cb0038 | |||
| 55df97afe1 | |||
| a8c98e3c31 | |||
| 19f12d9c15 | |||
| 6f936096ae | |||
| 367061fea2 | |||
| d488c3bd17 | |||
| ee5000ccca | |||
| 9e328c0daa | |||
| 7dd33cd118 | |||
| 936b680f2f | |||
| c0e2097397 | |||
| 744396fb0e | |||
| d2d44e9e92 | |||
| d6fb6c6af4 | |||
| 802c01bd8f | |||
| 7a89a4a573 | |||
| c5e66c6bf6 | |||
| 8c36dd0431 | |||
| 54e7af11ed | |||
| eca10ac0a8 | |||
| 18cc6abbb6 | |||
| 03473743c4 | |||
| 9fa116ce4a | |||
| 504dee5016 | |||
| 65ba5c9844 | |||
| de83ae5d25 | |||
| 2ffe07057e | |||
| 3d8b1ac494 | |||
| eba78e7e78 | |||
| d2ce966c46 | |||
| bb902f5b65 | |||
| 1412aa66c4 | |||
| e58ddadb7d | |||
| a992c9feb5 | |||
| 09080709a9 | |||
| 7521d10ba7 | |||
| ab349083bf | |||
| 4f9f27152a | |||
| 0292e21c33 | |||
| 2bb70859da | |||
| b9f82510d3 | |||
| 1216e8bb74 | |||
| be2d361ac6 | |||
| f78f7bfddc | |||
| f4e71a5a28 | |||
| 548e87fc0e | |||
| 89b1bb7bee | |||
| 318dfe4b55 | |||
| 4ed247df58 | |||
| 946b1fe5f5 | |||
| 69256992a3 | |||
| 84ff978ccb | |||
| 4ca69836e9 | |||
| b3c44a86a8 | |||
| b86a93e85d | |||
| 7f9b10ca71 | |||
| c306b36860 | |||
| 9b78f143cb | |||
| d847a4b852 | |||
| 8df7d5fd1b | |||
| 6d7c456f72 | |||
| b0fdeb4127 | |||
| 710cb54db3 | |||
| 23fbcd1850 | |||
| ba8199a030 | |||
| 6b8ae1e27c | |||
| 25d02ef3ea | |||
| fd56e64938 | |||
| 34c1c3d577 | |||
| 0dc96d2a9e | |||
| f8327e2dac | |||
| 10f39d1055 | |||
| d1ae1791d9 | |||
| ff6e0e010a | |||
| 2367519799 | |||
| be6ad27c11 | |||
| 28fab9c3e1 | |||
| 9e3977d9ea | |||
| a638592fc7 | |||
| a473a02a44 | |||
| 6611a91201 | |||
| 39323ca8d3 | |||
| f90ee8ab7d | |||
| 1de804f98d | |||
| 37b2a5bd98 | |||
| cc594f09ef | |||
| 2e92944f1d | |||
| 5f3ab1472c | |||
| 7018d163ed | |||
| f12ec3c935 | |||
| f3d1591430 | |||
| 44b71ae6f3 | |||
| 78859fe191 | |||
| e94e1f7622 | |||
| 06787a931f | |||
| 304ab49897 | |||
| 6aa8666be5 | |||
| 4b3731928c | |||
| f055341af6 | |||
| 4a90314333 | |||
| 2d4dececb6 | |||
| fb52444cdc | |||
| adac8576b0 | |||
| 512acd8cba | |||
| 56feb54377 | |||
| 59b0f434a7 | |||
| 1cfa199ee6 | |||
| 9e1d7b3bb4 | |||
| dbc052865c | |||
| 4543df8533 | |||
| 48ecebe6dd | |||
| 9f10c3f6e2 | |||
| 7f65ee0b2a | |||
| 8fb5772ad9 | |||
| 43d213ba36 | |||
| f137bcb694 | |||
| a52e18b6e6 | |||
| a17e7800b9 | |||
| b8855b9599 | |||
| 702791e41a | |||
| 86478f9384 | |||
| a0e7f05a50 | |||
| d3372aad31 | |||
| be818d5e3a | |||
| 04b4a69107 | |||
| fa2b3d26cb | |||
| 38106a9495 | |||
| df5443e017 | |||
| 8bef544bae | |||
| f778e1aa99 | |||
| a5d2d4792a | |||
| 3a0802d70c | |||
| 286e777300 | |||
| f4970d1e6f | |||
| 6a68f19b2c | |||
| c4aad7fe95 | |||
| b2df9fc110 | |||
| b12cf0d7c2 | |||
| 594947f301 | |||
| e7779ad773 | |||
| 65e2074dda | |||
| a68f45cb4e | |||
| b4c8c2bbad | |||
| 79cd96de6f | |||
| 7c2155decf | |||
| a5cc4e3eda | |||
| a4dca23111 | |||
| 04602f3952 | |||
| fe2e4eae92 | |||
| 61aa24c6b7 | |||
| 3bf9547502 | |||
| 6d9beb1ec5 | |||
| d25803d438 | |||
| 082920f3f5 | |||
| 94e558003d | |||
| 6fd261f628 | |||
| 2817ade782 | |||
| 9f1e30eaa9 | |||
| 1dfc69ba89 | |||
| 07193fc343 | |||
| 1b8f45703f | |||
| 3d3a2427c9 | |||
| 285c77d6dd | |||
| 6dff675db7 | |||
| cbfc8ea4f2 | |||
| 567dc0f75d | |||
| ab6d2d1766 | |||
| 8d08f5aade | |||
| e094a3e0fc | |||
| c4e955eab5 | |||
| be9e327aef | |||
| 211eff8ad0 | |||
| 20303ac32e | |||
| 3ea23815a4 | |||
| 62e4a3f779 | |||
| 0458f361b1 | |||
| 2f243f26e5 | |||
| c8ed6d782a | |||
| d4e0d07e09 | |||
| 0548a662f5 | |||
| a0e4e9f209 | |||
| 18794d52c2 | |||
| d2e2669b6d | |||
| d8fbee58d8 | |||
| a6eef498fa | |||
| 826fcc97c8 | |||
| 6d81473b2a | |||
| 0621ed223b | |||
| 0ae96a6bc4 | |||
| d45947eec5 | |||
| e5e6654b1d | |||
| 298487d5c5 | |||
| 6b6dbabba3 | |||
| e1d5f58c3f | |||
| 041a5cf65d | |||
| 8935d85e91 | |||
| 64d22c51f1 | |||
| eec8de3efa | |||
| 548a3dd33c | |||
| 15105e90d3 | |||
| 4e1c7a70b1 | |||
| d936e5f740 | |||
| 3d2405323b | |||
| 7a677d5c10 | |||
| 90dd1d95f3 | |||
| f07f333ae5 | |||
| 6bdf47bae6 | |||
| ba8c282384 | |||
| 64098ad244 | |||
| 854e1d909a | |||
| 377310af50 | |||
| 0a9e01db15 | |||
| 23b40de508 | |||
| 5d01a3f629 | |||
| 40456592fc | |||
| 2e3cb52cff | |||
| f684a2126e | |||
| 533bb48789 | |||
| ca7631250c | |||
| 7f0d9a930d | |||
| 85bbbf42a1 | |||
| df76766890 | |||
| 2594e491d4 | |||
| 5a9ed44b6a | |||
| 9b3752d77a | |||
| 5ab2e63de4 | |||
| 98537624cf | |||
| 4fd84b1876 | |||
| d8f034a7ad | |||
| d6c28c5ba2 | |||
| 3e2d0630b7 | |||
| 986ec5d00c | |||
| adb5a693cd | |||
| db0f8a4d99 | |||
| e6d435a463 | |||
| a874fad2bd | |||
| 04bd5d7ff4 | |||
| 89f91f568e | |||
| c975a467b1 | |||
| 6e5ab031c4 | |||
| f418a1f9b2 | |||
| 3ba276a236 | |||
| 13c9dc6620 | |||
| 1ea6097a97 | |||
| 8cff886ee5 | |||
| 488887992c | |||
| 8caff5f598 | |||
| 3d940a9fa2 | |||
| 7c81c5a43c | |||
| b1818b81d9 | |||
| 2fbac7705d | |||
| cbea75a8f6 | |||
| 4a308467e6 | |||
| 29b99f4e61 | |||
| e3e5a133f5 | |||
| f637aa0ea8 | |||
| c654610569 | |||
| 85588f7343 | |||
| f9e3f17c0a | |||
| cb5b7a1d98 | |||
| 106cbb5c73 | |||
| fabad7bace | |||
| 4361d5fec0 | |||
| 3502c5b676 | |||
| ed0596a52c | |||
| 2635d24855 | |||
| 1bb9f73a0b | |||
| c1247930c9 | |||
| bddb0110ca | |||
| 6d1fda0c21 | |||
| 35e86ca5cb | |||
| 0debd01ba7 | |||
| 0951f4644f | |||
| 6ec417d89f | |||
| ba04328d48 | |||
| ef78d7f176 | |||
| f089d59d10 | |||
| da79128271 | |||
| 5f6011b206 | |||
| ae1d8cd9e6 | |||
| 0e5c90634f | |||
| 5562f599a6 | |||
| 326c7e0940 | |||
| c36befaf2b | |||
| cf3fbeddbd | |||
| ae876e22ee | |||
| 26dcb7b8c9 | |||
| 455291b6dd | |||
| 8575059666 | |||
| 9ec0913d7f | |||
| c8308963df | |||
| e3442e4764 | |||
| c0a978a480 | |||
| 674c61d280 | |||
| dfdeb8b36e | |||
| e4900120db | |||
| 79a05a8159 | |||
| af13e7c954 | |||
| c06c9fcf23 | |||
| bb157be82c | |||
| ce20868bad | |||
| 573924cf2c | |||
| f2a4125d9f | |||
| d7276c720d | |||
| 482e122655 | |||
| 08e1801c60 | |||
| 7c55dcb708 | |||
| 3eda9b9219 | |||
| df37e11709 | |||
| bbcc33c0cf | |||
| c7d9c38ea0 | |||
| 58fe723b6a | |||
| f44f377771 | |||
| 41b1135fb3 | |||
| 43fd92f701 | |||
| 1d388ca51a | |||
| 9c25d15ff2 | |||
| 62c8f528a4 | |||
| 82d907971c | |||
| 555b32b5dc | |||
| cf7fbadcd2 | |||
| 5e0c1879bc | |||
| 32dd807ab0 | |||
| 8aa792925d | |||
| 9d17cb42f4 | |||
| db10ebeb34 | |||
| 62e7882744 | |||
| 13b4dffd71 | |||
| f6527f51e3 | |||
| 66e3b98b04 | |||
| 0cc2666e26 | |||
| d0d462c015 | |||
| 2d46f16b70 | |||
| e6aa73d7bc | |||
| 4b85a2238a | |||
| 8944778dcf | |||
| 406f8cc9b3 | |||
| 7ea31d123a | |||
| 7dbbbe4d9a | |||
| 01a42ffe27 | |||
| 296a6cbc87 | |||
| fe206ea24c | |||
| 3eb079dd13 | |||
| 110bf56dc4 | |||
| 209a3e8492 | |||
| 38a1b9dcb2 | |||
| 443691414a | |||
| 1a1c54f364 | |||
| 10b7160677 | |||
| 4746121d43 | |||
| 2c1df5619f | |||
| 4fa5d2c231 | |||
| c1c2173e4d | |||
| 03d39d991c | |||
| a9c0db0bf6 | |||
| 97b88ef52d | |||
| 4c2d2a7965 | |||
| 636d899f9f | |||
| 7e21b10aee | |||
| 6e2ff405f4 | |||
| 1936e54870 | |||
| b5c0baa99d | |||
| 801b8b5641 | |||
| 51ffcb1f50 | |||
| 10831ab450 | |||
| d601e980f6 | |||
| b06dc286e9 | |||
| c02238ddc8 | |||
| e72aa348a0 | |||
| 62fe82ce9e | |||
| 9007e25a00 | |||
| 289a97d06c | |||
| f32d8a859b | |||
| 5c501f529b | |||
| 2503eba173 | |||
| c21a0687ce | |||
| 70eb5214cc | |||
| 37019597af | |||
| d8eeb36d4b | |||
| c695d8c5d7 | |||
| 4616be0738 | |||
| fe85c26888 | |||
| 7f239efdc0 | |||
| c68d6f318e | |||
| 8a4d1cac17 | |||
| b23cf9fe70 | |||
| 7838cb4922 | |||
| 146e144249 | |||
| 2f57ba1222 | |||
| 0f0c8c28e1 | |||
| 23a1365dc4 | |||
| 7e7f480e2b | |||
| 5272f1cde6 | |||
| f0ea2c027f | |||
| 854a4c316a | |||
| d889ed385d | |||
| e99b353d8b | |||
| df888a3457 | |||
| 44b2bce3ff | |||
| a2db98670a | |||
| 2594f59d0d | |||
| f247300469 | |||
| 6a1707f65e | |||
| 8b64d1876f | |||
| d6d93a74f8 | |||
| 33f36ab204 | |||
| f9afee2f3b | |||
| 02936fa928 | |||
| e9ec582223 | |||
| 058f9b795f | |||
| 4ac2c87d1c | |||
| 42a1ea02b9 |
+14
-1
@@ -8,4 +8,17 @@ build/
|
||||
*.h5
|
||||
*.tar.gz
|
||||
*.weights
|
||||
.idea/
|
||||
.idea/
|
||||
*.hdf5
|
||||
*.pk
|
||||
*.table
|
||||
cmake-build-release/
|
||||
demo/COCO_val2017
|
||||
demo/BDD100K_val
|
||||
/.vs
|
||||
cmake-build-minsizerel/*
|
||||
scripts/COCO_val2017/*
|
||||
scripts/COCO_val2017.zip
|
||||
scripts/all_labels.txt
|
||||
/cmake/cuda_script
|
||||
/cmake-build-debug/
|
||||
|
||||
+188
-56
@@ -1,8 +1,69 @@
|
||||
cmake_minimum_required(VERSION 3.5)
|
||||
|
||||
project (tkDNN)
|
||||
cmake_minimum_required(VERSION 3.15)
|
||||
project(tkDNN)
|
||||
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC")
|
||||
set(CMAKE_CXX_STANDARD 14)
|
||||
|
||||
option(ENABLE_OPENCV_CUDA_CONTRIB "Enable OpenCV CUDA Contrib" OFF )
|
||||
|
||||
if(NOT CMAKE_BUILD_TYPE)
|
||||
set(CMAKE_BUILD_TYPE "Release" CACHE STRING "default build" FORCE)
|
||||
endif(NOT CMAKE_BUILD_TYPE)
|
||||
|
||||
find_package(CUDA 9.0 REQUIRED)
|
||||
if (CUDA_FOUND)
|
||||
set(OUTPUTFILE ${CMAKE_CURRENT_SOURCE_DIR}/cmake/cuda_script) # No suffix required
|
||||
execute_process(COMMAND "rm ${OUTPUTFILE}")
|
||||
set(CUDAFILE ${CMAKE_CURRENT_SOURCE_DIR}/cmake/getCudaArch.cu)
|
||||
execute_process(COMMAND ${CUDA_NVCC_EXECUTABLE} -lcuda ${CUDAFILE} -o ${OUTPUTFILE})
|
||||
execute_process(COMMAND ${OUTPUTFILE}
|
||||
RESULT_VARIABLE CUDA_RETURN_CODE
|
||||
OUTPUT_VARIABLE ARCH)
|
||||
|
||||
if(${CUDA_RETURN_CODE} EQUAL 0)
|
||||
set(CUDA_SUCCESS "TRUE")
|
||||
else()
|
||||
set(CUDA_SUCCESS "FALSE")
|
||||
endif()
|
||||
|
||||
if (${CUDA_SUCCESS})
|
||||
message(STATUS "CUDA Architecture: ${ARCH}")
|
||||
message(STATUS "CUDA Version: ${CUDA_VERSION_STRING}")
|
||||
message(STATUS "CUDA Path: ${CUDA_TOOLKIT_ROOT_DIR}")
|
||||
message(STATUS "CUDA Libararies: ${CUDA_LIBRARIES}")
|
||||
message(STATUS "CUDA Performance Primitives: ${CUDA_npp_LIBRARY}")
|
||||
set(CUDA_NVCC_FLAGS "${ARCH}")
|
||||
else()
|
||||
message(WARNING ${ARCH})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
SET(CUDA_SEPARABLE_COMPILATION ON)
|
||||
|
||||
if(UNIX)
|
||||
if(CMAKE_BUILD_TYPE MATCHES Release)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -fPIC -Wno-deprecated-declarations -Wno-unused-variable -O3")
|
||||
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32)
|
||||
endif()
|
||||
|
||||
if(CMAKE_BUILD_TYPE MATCHES Debug)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -fPIC -Wno-deprecated-declarations -Wno-unused-variable -g3")
|
||||
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32 -G -g)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(WIN32)
|
||||
if(CMAKE_BUILD_TYPE MATCHES Release)
|
||||
set(CMAKE_CXX_FLAGS "/O2 /FS /EHsc /MD")
|
||||
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32)
|
||||
endif()
|
||||
|
||||
if(CMAKE_BUILD_TYPE MATCHES Debug)
|
||||
set(CMAKE_CXX_FLAGS "/Od /FS /EHsc /MDd")
|
||||
set(CUDA_NVCC_FLAGS ${CUDA_NVCC_FLAGS} --maxrregcount=32 -G -g)
|
||||
endif()
|
||||
set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON)
|
||||
endif(WIN32)
|
||||
|
||||
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include/tkDNN)
|
||||
|
||||
# project specific flags
|
||||
@@ -10,44 +71,77 @@ if(DEBUG)
|
||||
add_definitions(-DDEBUG)
|
||||
endif()
|
||||
|
||||
if(TKDNN_PATH)
|
||||
message("SET TKDNN_PATH:" ${TKDNN_PATH})
|
||||
add_definitions(-DTKDNN_PATH="${TKDNN_PATH}")
|
||||
else()
|
||||
add_definitions(-DTKDNN_PATH="${CMAKE_CURRENT_SOURCE_DIR}")
|
||||
endif()
|
||||
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# CUDA
|
||||
#-------------------------------------------------------------------------------
|
||||
find_package(CUDA 9.0 REQUIRED)
|
||||
SET(CUDA_SEPARABLE_COMPILATION ON)
|
||||
#set(CUDA_NVCC_FLAGS "${CUDA_NVCC_FLAGS} -arch=sm_30 --compiler-options '-fPIC'")
|
||||
set(CUDA_NVCC_FLAGS "${CUDA_NVCC_FLAGS}" --compiler-options '-fPIC')
|
||||
|
||||
|
||||
find_package(CUDNN REQUIRED)
|
||||
include_directories(${CUDNN_INCLUDE_DIR})
|
||||
|
||||
find_package(yaml-cpp REQUIRED)
|
||||
|
||||
|
||||
# compile
|
||||
file(GLOB tkdnn_CUSRC "src/kernels/*.cu")
|
||||
file(GLOB tkdnn_CUSRC "src/kernels/*.cu" "src/sorting.cu" "src/pluginsRT/*.cpp")
|
||||
cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${CUDNN_INCLUDE_DIRS})
|
||||
cuda_add_library(kernels SHARED ${tkdnn_CUSRC})
|
||||
target_link_libraries(kernels ${CUDA_CUBLAS_LIBRARIES} ${CUDA_LIBRARIES} ${CUDNN_LIBRARIES} yaml-cpp)
|
||||
|
||||
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# External Libraries
|
||||
#-------------------------------------------------------------------------------
|
||||
find_package(Eigen3 REQUIRED)
|
||||
message("Eigen DIR: " ${EIGEN3_INCLUDE_DIR})
|
||||
include_directories(${EIGEN3_INCLUDE_DIR})
|
||||
|
||||
find_package(OpenCV REQUIRED)
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -DOPENCV")
|
||||
if(ENABLE_OPENCV_CUDA_CONTRIB)
|
||||
if (OpenCV_FOUND)
|
||||
find_package(OpenCV COMPONENTS cudawarping cudaarithm)
|
||||
if(OpenCV_cudawarping_FOUND AND OpenCV_cudaarithm_FOUND)
|
||||
add_compile_definitions(OPENCV_CUDACONTRIB)
|
||||
message("OpenCV Cuda Contrib modules found")
|
||||
else()
|
||||
message("OpenCV Cuda Contrib modules not found")
|
||||
set(ENABLE_OPENCV_CUDA_CONTRIB OFF)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
# if(OpenCV_CUDA_VERSION)
|
||||
# add_compile_definitions(OPENCV_CUDACONTRIB)
|
||||
# endif()
|
||||
|
||||
# gives problems in cross-compiling, probably malformed cmake config
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Build Libraries
|
||||
#-------------------------------------------------------------------------------
|
||||
file(GLOB tkdnn_SRC "src/*.cpp")
|
||||
set(tkdnn_LIBS kernels ${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDNN_LIBRARIES} ${OpenCV_LIBS})
|
||||
set(tkdnn_LIBS kernels ${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDNN_LIBRARIES} ${OpenCV_LIBS} yaml-cpp)
|
||||
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Wall -std=c++11")
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS}")
|
||||
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS} ${OPENCV_INCLUDE_DIRS} ${NVINFER_INCLUDES})
|
||||
add_library(tkDNN SHARED ${tkdnn_SRC})
|
||||
target_link_libraries(tkDNN ${tkdnn_LIBS})
|
||||
target_link_libraries(tkDNN ${tkdnn_LIBS} ${CUDA_CUBLAS_LIBRARIES})
|
||||
|
||||
#static
|
||||
#add_library(tkDNN_static STATIC ${tkdnn_SRC})
|
||||
#target_link_libraries(tkDNN_static ${tkdnn_LIBS})
|
||||
|
||||
# SMALL NETS
|
||||
add_executable(test_simple tests/simple/test_simple.cpp)
|
||||
target_link_libraries(test_simple tkDNN)
|
||||
|
||||
@@ -57,43 +151,93 @@ target_link_libraries(test_mnist tkDNN)
|
||||
add_executable(test_mnistRT tests/mnist/test_mnistRT.cpp)
|
||||
target_link_libraries(test_mnistRT tkDNN)
|
||||
|
||||
## YOLO NETS
|
||||
add_executable(test_yolo tests/yolo/yolo.cpp)
|
||||
target_link_libraries(test_yolo tkDNN)
|
||||
add_executable(test_imuodom tests/imuodom/imuodom.cpp)
|
||||
target_link_libraries(test_imuodom tkDNN)
|
||||
|
||||
add_executable(test_yolo_voc tests/yolo_voc/yolo_voc.cpp)
|
||||
target_link_libraries(test_yolo_voc tkDNN)
|
||||
# DARKNET
|
||||
file(GLOB darknet_SRC "tests/darknet/*.cpp")
|
||||
foreach(test_SRC ${darknet_SRC})
|
||||
get_filename_component(test_NAME "${test_SRC}" NAME_WE)
|
||||
set(test_NAME test_${test_NAME})
|
||||
add_executable(${test_NAME} ${test_SRC})
|
||||
target_link_libraries(${test_NAME} tkDNN)
|
||||
install(TARGETS ${test_NAME} DESTINATION bin)
|
||||
endforeach()
|
||||
|
||||
add_executable(test_yolo_tiny tests/yolo_tiny/yolo_tiny.cpp)
|
||||
target_link_libraries(test_yolo_tiny tkDNN)
|
||||
# MOBILENET
|
||||
add_executable(test_mobilenetv2ssd tests/mobilenet/mobilenetv2ssd/mobilenetv2ssd.cpp)
|
||||
target_link_libraries(test_mobilenetv2ssd tkDNN)
|
||||
|
||||
add_executable(test_yolo_relu tests/yolo_relu/yolo_relu.cpp)
|
||||
target_link_libraries(test_yolo_relu tkDNN)
|
||||
|
||||
|
||||
add_executable(test_yolo_224 tests/yolo_224/yolo_224.cpp)
|
||||
target_link_libraries(test_yolo_224 tkDNN)
|
||||
|
||||
add_executable(test_yolo_berkeley tests/yolo_berkeley/yolo_berkeley.cpp)
|
||||
target_link_libraries(test_yolo_berkeley tkDNN)
|
||||
|
||||
add_executable(test_yolo3_coco4 tests/yolo3_coco4/yolo3_coco4.cpp)
|
||||
target_link_libraries(test_yolo3_coco4 tkDNN)
|
||||
|
||||
add_executable(test_yolo3_berkeley tests/yolo3_berkeley/yolo3_berkeley.cpp)
|
||||
target_link_libraries(test_yolo3_berkeley tkDNN)
|
||||
|
||||
add_executable(test_yolo3_flir tests/yolo3_flir/yolo3_flir.cpp)
|
||||
target_link_libraries(test_yolo3_flir tkDNN)
|
||||
################################################################################
|
||||
add_executable(test_bdd-mobilenetv2ssd tests/mobilenet/bdd-mobilenetv2ssd/bdd-mobilenetv2ssd.cpp)
|
||||
target_link_libraries(test_bdd-mobilenetv2ssd tkDNN)
|
||||
|
||||
add_executable(test_mobilenetv2ssd512 tests/mobilenet/mobilenetv2ssd512/mobilenetv2ssd512.cpp)
|
||||
target_link_libraries(test_mobilenetv2ssd512 tkDNN)
|
||||
|
||||
# BACKBONES
|
||||
add_executable(test_resnet101 tests/backbones/resnet101/resnet101.cpp)
|
||||
target_link_libraries(test_resnet101 tkDNN)
|
||||
|
||||
add_executable(test_dla34 tests/backbones/dla34/dla34.cpp)
|
||||
target_link_libraries(test_dla34 tkDNN)
|
||||
|
||||
# CENTERNET
|
||||
add_executable(test_resnet101_cnet tests/centernet/resnet101_cnet/resnet101_cnet.cpp)
|
||||
target_link_libraries(test_resnet101_cnet tkDNN)
|
||||
|
||||
add_executable(test_dla34_cnet tests/centernet/dla34_cnet/dla34_cnet.cpp)
|
||||
target_link_libraries(test_dla34_cnet tkDNN)
|
||||
|
||||
add_executable(test_dla34_cnet3d tests/centernet/dla34_cnet3d/dla34_cnet3d.cpp)
|
||||
target_link_libraries(test_dla34_cnet3d tkDNN)
|
||||
|
||||
# CENTERTRACK
|
||||
|
||||
add_executable(test_dla34_ctrack tests/centertrack/dla34_ctrack/dla34_ctrack.cpp)
|
||||
target_link_libraries(test_dla34_ctrack tkDNN)
|
||||
|
||||
# SHELFNET
|
||||
add_executable(test_shelfnet tests/shelfnet/shelfnet.cpp)
|
||||
target_link_libraries(test_shelfnet tkDNN)
|
||||
|
||||
add_executable(test_shelfnet_berkeley tests/shelfnet/shelfnet_berkeley.cpp)
|
||||
target_link_libraries(test_shelfnet_berkeley tkDNN)
|
||||
|
||||
add_executable(test_shelfnet_mapillary tests/shelfnet/shelfnet_mapillary.cpp)
|
||||
target_link_libraries(test_shelfnet_mapillary tkDNN)
|
||||
|
||||
add_executable(test_shelfnet_coco tests/shelfnet/shelfnet_coco.cpp)
|
||||
target_link_libraries(test_shelfnet_coco tkDNN)
|
||||
|
||||
# MONODEPTH2
|
||||
add_executable(test_monodepth2_640 tests/monodepth2/monodepth2_640.cpp)
|
||||
target_link_libraries(test_monodepth2_640 tkDNN)
|
||||
|
||||
add_executable(test_monodepth2_1024 tests/monodepth2/monodepth2_1024.cpp)
|
||||
target_link_libraries(test_monodepth2_1024 tkDNN)
|
||||
|
||||
|
||||
# DEMOS
|
||||
add_executable(test_rtinference tests/test_rtinference/rtinference.cpp)
|
||||
target_link_libraries(test_rtinference tkDNN)
|
||||
|
||||
add_executable(yolo3_demo demo/demo/demo.cpp)
|
||||
target_link_libraries(yolo3_demo tkDNN)
|
||||
add_executable(map_demo demo/demo/map.cpp)
|
||||
target_link_libraries(map_demo tkDNN)
|
||||
|
||||
add_executable(demo demo/demo/demo.cpp)
|
||||
target_link_libraries(demo tkDNN)
|
||||
|
||||
add_executable(demo3D demo/demo/demo3D.cpp)
|
||||
target_link_libraries(demo3D tkDNN)
|
||||
|
||||
add_executable(demoTracker demo/demo/demoTracker.cpp)
|
||||
target_link_libraries(demoTracker tkDNN)
|
||||
|
||||
add_executable(seg_demo demo/demo/seg_demo.cpp)
|
||||
target_link_libraries(seg_demo tkDNN)
|
||||
|
||||
add_executable(demoDepth demo/demo/demoDepth.cpp)
|
||||
target_link_libraries(demoDepth tkDNN)
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Install
|
||||
@@ -104,23 +248,11 @@ target_link_libraries(yolo3_demo tkDNN)
|
||||
#endif()
|
||||
message("install dir:" ${CMAKE_INSTALL_PREFIX})
|
||||
install(DIRECTORY include/ DESTINATION include/)
|
||||
install(TARGETS tkDNN kernels DESTINATION lib)
|
||||
install(TARGETS tkDNN DESTINATION lib)
|
||||
install(TARGETS test_simple test_mnist test_mnistRT test_rtinference demo map_demo DESTINATION bin)
|
||||
install(DIRECTORY "${CMAKE_CURRENT_SOURCE_DIR}/cmake/" # source directory
|
||||
DESTINATION "share/tkDNN/cmake/" # target directory
|
||||
)
|
||||
|
||||
|
||||
#-------------------------------------------------------------------------------
|
||||
# Prepare for test
|
||||
#-------------------------------------------------------------------------------
|
||||
set(TEST_DATA true CACHE BOOL "If true download deps")
|
||||
if( ${TEST_DATA} )
|
||||
message("Launching pre-build dependency installer script...")
|
||||
|
||||
execute_process (COMMAND bash -c "bash build_models.sh download"
|
||||
WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/tests)
|
||||
|
||||
set(TEST_DATA false CACHE BOOL "If true download deps" FORCE)
|
||||
message("Finished dowloading test weights")
|
||||
endif()
|
||||
|
||||
install(DIRECTORY "${CMAKE_CURRENT_SOURCE_DIR}/tests/" # source directory
|
||||
DESTINATION "share/tkDNN/tests" # target directory
|
||||
)
|
||||
|
||||
@@ -0,0 +1,339 @@
|
||||
GNU GENERAL PUBLIC LICENSE
|
||||
Version 2, June 1991
|
||||
|
||||
Copyright (C) 1989, 1991 Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
||||
Everyone is permitted to copy and distribute verbatim copies
|
||||
of this license document, but changing it is not allowed.
|
||||
|
||||
Preamble
|
||||
|
||||
The licenses for most software are designed to take away your
|
||||
freedom to share and change it. By contrast, the GNU General Public
|
||||
License is intended to guarantee your freedom to share and change free
|
||||
software--to make sure the software is free for all its users. This
|
||||
General Public License applies to most of the Free Software
|
||||
Foundation's software and to any other program whose authors commit to
|
||||
using it. (Some other Free Software Foundation software is covered by
|
||||
the GNU Lesser General Public License instead.) You can apply it to
|
||||
your programs, too.
|
||||
|
||||
When we speak of free software, we are referring to freedom, not
|
||||
price. Our General Public Licenses are designed to make sure that you
|
||||
have the freedom to distribute copies of free software (and charge for
|
||||
this service if you wish), that you receive source code or can get it
|
||||
if you want it, that you can change the software or use pieces of it
|
||||
in new free programs; and that you know you can do these things.
|
||||
|
||||
To protect your rights, we need to make restrictions that forbid
|
||||
anyone to deny you these rights or to ask you to surrender the rights.
|
||||
These restrictions translate to certain responsibilities for you if you
|
||||
distribute copies of the software, or if you modify it.
|
||||
|
||||
For example, if you distribute copies of such a program, whether
|
||||
gratis or for a fee, you must give the recipients all the rights that
|
||||
you have. You must make sure that they, too, receive or can get the
|
||||
source code. And you must show them these terms so they know their
|
||||
rights.
|
||||
|
||||
We protect your rights with two steps: (1) copyright the software, and
|
||||
(2) offer you this license which gives you legal permission to copy,
|
||||
distribute and/or modify the software.
|
||||
|
||||
Also, for each author's protection and ours, we want to make certain
|
||||
that everyone understands that there is no warranty for this free
|
||||
software. If the software is modified by someone else and passed on, we
|
||||
want its recipients to know that what they have is not the original, so
|
||||
that any problems introduced by others will not reflect on the original
|
||||
authors' reputations.
|
||||
|
||||
Finally, any free program is threatened constantly by software
|
||||
patents. We wish to avoid the danger that redistributors of a free
|
||||
program will individually obtain patent licenses, in effect making the
|
||||
program proprietary. To prevent this, we have made it clear that any
|
||||
patent must be licensed for everyone's free use or not licensed at all.
|
||||
|
||||
The precise terms and conditions for copying, distribution and
|
||||
modification follow.
|
||||
|
||||
GNU GENERAL PUBLIC LICENSE
|
||||
TERMS AND CONDITIONS FOR COPYING, DISTRIBUTION AND MODIFICATION
|
||||
|
||||
0. This License applies to any program or other work which contains
|
||||
a notice placed by the copyright holder saying it may be distributed
|
||||
under the terms of this General Public License. The "Program", below,
|
||||
refers to any such program or work, and a "work based on the Program"
|
||||
means either the Program or any derivative work under copyright law:
|
||||
that is to say, a work containing the Program or a portion of it,
|
||||
either verbatim or with modifications and/or translated into another
|
||||
language. (Hereinafter, translation is included without limitation in
|
||||
the term "modification".) Each licensee is addressed as "you".
|
||||
|
||||
Activities other than copying, distribution and modification are not
|
||||
covered by this License; they are outside its scope. The act of
|
||||
running the Program is not restricted, and the output from the Program
|
||||
is covered only if its contents constitute a work based on the
|
||||
Program (independent of having been made by running the Program).
|
||||
Whether that is true depends on what the Program does.
|
||||
|
||||
1. You may copy and distribute verbatim copies of the Program's
|
||||
source code as you receive it, in any medium, provided that you
|
||||
conspicuously and appropriately publish on each copy an appropriate
|
||||
copyright notice and disclaimer of warranty; keep intact all the
|
||||
notices that refer to this License and to the absence of any warranty;
|
||||
and give any other recipients of the Program a copy of this License
|
||||
along with the Program.
|
||||
|
||||
You may charge a fee for the physical act of transferring a copy, and
|
||||
you may at your option offer warranty protection in exchange for a fee.
|
||||
|
||||
2. You may modify your copy or copies of the Program or any portion
|
||||
of it, thus forming a work based on the Program, and copy and
|
||||
distribute such modifications or work under the terms of Section 1
|
||||
above, provided that you also meet all of these conditions:
|
||||
|
||||
a) You must cause the modified files to carry prominent notices
|
||||
stating that you changed the files and the date of any change.
|
||||
|
||||
b) You must cause any work that you distribute or publish, that in
|
||||
whole or in part contains or is derived from the Program or any
|
||||
part thereof, to be licensed as a whole at no charge to all third
|
||||
parties under the terms of this License.
|
||||
|
||||
c) If the modified program normally reads commands interactively
|
||||
when run, you must cause it, when started running for such
|
||||
interactive use in the most ordinary way, to print or display an
|
||||
announcement including an appropriate copyright notice and a
|
||||
notice that there is no warranty (or else, saying that you provide
|
||||
a warranty) and that users may redistribute the program under
|
||||
these conditions, and telling the user how to view a copy of this
|
||||
License. (Exception: if the Program itself is interactive but
|
||||
does not normally print such an announcement, your work based on
|
||||
the Program is not required to print an announcement.)
|
||||
|
||||
These requirements apply to the modified work as a whole. If
|
||||
identifiable sections of that work are not derived from the Program,
|
||||
and can be reasonably considered independent and separate works in
|
||||
themselves, then this License, and its terms, do not apply to those
|
||||
sections when you distribute them as separate works. But when you
|
||||
distribute the same sections as part of a whole which is a work based
|
||||
on the Program, the distribution of the whole must be on the terms of
|
||||
this License, whose permissions for other licensees extend to the
|
||||
entire whole, and thus to each and every part regardless of who wrote it.
|
||||
|
||||
Thus, it is not the intent of this section to claim rights or contest
|
||||
your rights to work written entirely by you; rather, the intent is to
|
||||
exercise the right to control the distribution of derivative or
|
||||
collective works based on the Program.
|
||||
|
||||
In addition, mere aggregation of another work not based on the Program
|
||||
with the Program (or with a work based on the Program) on a volume of
|
||||
a storage or distribution medium does not bring the other work under
|
||||
the scope of this License.
|
||||
|
||||
3. You may copy and distribute the Program (or a work based on it,
|
||||
under Section 2) in object code or executable form under the terms of
|
||||
Sections 1 and 2 above provided that you also do one of the following:
|
||||
|
||||
a) Accompany it with the complete corresponding machine-readable
|
||||
source code, which must be distributed under the terms of Sections
|
||||
1 and 2 above on a medium customarily used for software interchange; or,
|
||||
|
||||
b) Accompany it with a written offer, valid for at least three
|
||||
years, to give any third party, for a charge no more than your
|
||||
cost of physically performing source distribution, a complete
|
||||
machine-readable copy of the corresponding source code, to be
|
||||
distributed under the terms of Sections 1 and 2 above on a medium
|
||||
customarily used for software interchange; or,
|
||||
|
||||
c) Accompany it with the information you received as to the offer
|
||||
to distribute corresponding source code. (This alternative is
|
||||
allowed only for noncommercial distribution and only if you
|
||||
received the program in object code or executable form with such
|
||||
an offer, in accord with Subsection b above.)
|
||||
|
||||
The source code for a work means the preferred form of the work for
|
||||
making modifications to it. For an executable work, complete source
|
||||
code means all the source code for all modules it contains, plus any
|
||||
associated interface definition files, plus the scripts used to
|
||||
control compilation and installation of the executable. However, as a
|
||||
special exception, the source code distributed need not include
|
||||
anything that is normally distributed (in either source or binary
|
||||
form) with the major components (compiler, kernel, and so on) of the
|
||||
operating system on which the executable runs, unless that component
|
||||
itself accompanies the executable.
|
||||
|
||||
If distribution of executable or object code is made by offering
|
||||
access to copy from a designated place, then offering equivalent
|
||||
access to copy the source code from the same place counts as
|
||||
distribution of the source code, even though third parties are not
|
||||
compelled to copy the source along with the object code.
|
||||
|
||||
4. You may not copy, modify, sublicense, or distribute the Program
|
||||
except as expressly provided under this License. Any attempt
|
||||
otherwise to copy, modify, sublicense or distribute the Program is
|
||||
void, and will automatically terminate your rights under this License.
|
||||
However, parties who have received copies, or rights, from you under
|
||||
this License will not have their licenses terminated so long as such
|
||||
parties remain in full compliance.
|
||||
|
||||
5. You are not required to accept this License, since you have not
|
||||
signed it. However, nothing else grants you permission to modify or
|
||||
distribute the Program or its derivative works. These actions are
|
||||
prohibited by law if you do not accept this License. Therefore, by
|
||||
modifying or distributing the Program (or any work based on the
|
||||
Program), you indicate your acceptance of this License to do so, and
|
||||
all its terms and conditions for copying, distributing or modifying
|
||||
the Program or works based on it.
|
||||
|
||||
6. Each time you redistribute the Program (or any work based on the
|
||||
Program), the recipient automatically receives a license from the
|
||||
original licensor to copy, distribute or modify the Program subject to
|
||||
these terms and conditions. You may not impose any further
|
||||
restrictions on the recipients' exercise of the rights granted herein.
|
||||
You are not responsible for enforcing compliance by third parties to
|
||||
this License.
|
||||
|
||||
7. If, as a consequence of a court judgment or allegation of patent
|
||||
infringement or for any other reason (not limited to patent issues),
|
||||
conditions are imposed on you (whether by court order, agreement or
|
||||
otherwise) that contradict the conditions of this License, they do not
|
||||
excuse you from the conditions of this License. If you cannot
|
||||
distribute so as to satisfy simultaneously your obligations under this
|
||||
License and any other pertinent obligations, then as a consequence you
|
||||
may not distribute the Program at all. For example, if a patent
|
||||
license would not permit royalty-free redistribution of the Program by
|
||||
all those who receive copies directly or indirectly through you, then
|
||||
the only way you could satisfy both it and this License would be to
|
||||
refrain entirely from distribution of the Program.
|
||||
|
||||
If any portion of this section is held invalid or unenforceable under
|
||||
any particular circumstance, the balance of the section is intended to
|
||||
apply and the section as a whole is intended to apply in other
|
||||
circumstances.
|
||||
|
||||
It is not the purpose of this section to induce you to infringe any
|
||||
patents or other property right claims or to contest validity of any
|
||||
such claims; this section has the sole purpose of protecting the
|
||||
integrity of the free software distribution system, which is
|
||||
implemented by public license practices. Many people have made
|
||||
generous contributions to the wide range of software distributed
|
||||
through that system in reliance on consistent application of that
|
||||
system; it is up to the author/donor to decide if he or she is willing
|
||||
to distribute software through any other system and a licensee cannot
|
||||
impose that choice.
|
||||
|
||||
This section is intended to make thoroughly clear what is believed to
|
||||
be a consequence of the rest of this License.
|
||||
|
||||
8. If the distribution and/or use of the Program is restricted in
|
||||
certain countries either by patents or by copyrighted interfaces, the
|
||||
original copyright holder who places the Program under this License
|
||||
may add an explicit geographical distribution limitation excluding
|
||||
those countries, so that distribution is permitted only in or among
|
||||
countries not thus excluded. In such case, this License incorporates
|
||||
the limitation as if written in the body of this License.
|
||||
|
||||
9. The Free Software Foundation may publish revised and/or new versions
|
||||
of the General Public License from time to time. Such new versions will
|
||||
be similar in spirit to the present version, but may differ in detail to
|
||||
address new problems or concerns.
|
||||
|
||||
Each version is given a distinguishing version number. If the Program
|
||||
specifies a version number of this License which applies to it and "any
|
||||
later version", you have the option of following the terms and conditions
|
||||
either of that version or of any later version published by the Free
|
||||
Software Foundation. If the Program does not specify a version number of
|
||||
this License, you may choose any version ever published by the Free Software
|
||||
Foundation.
|
||||
|
||||
10. If you wish to incorporate parts of the Program into other free
|
||||
programs whose distribution conditions are different, write to the author
|
||||
to ask for permission. For software which is copyrighted by the Free
|
||||
Software Foundation, write to the Free Software Foundation; we sometimes
|
||||
make exceptions for this. Our decision will be guided by the two goals
|
||||
of preserving the free status of all derivatives of our free software and
|
||||
of promoting the sharing and reuse of software generally.
|
||||
|
||||
NO WARRANTY
|
||||
|
||||
11. BECAUSE THE PROGRAM IS LICENSED FREE OF CHARGE, THERE IS NO WARRANTY
|
||||
FOR THE PROGRAM, TO THE EXTENT PERMITTED BY APPLICABLE LAW. EXCEPT WHEN
|
||||
OTHERWISE STATED IN WRITING THE COPYRIGHT HOLDERS AND/OR OTHER PARTIES
|
||||
PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY OF ANY KIND, EITHER EXPRESSED
|
||||
OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF
|
||||
MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE. THE ENTIRE RISK AS
|
||||
TO THE QUALITY AND PERFORMANCE OF THE PROGRAM IS WITH YOU. SHOULD THE
|
||||
PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF ALL NECESSARY SERVICING,
|
||||
REPAIR OR CORRECTION.
|
||||
|
||||
12. IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING
|
||||
WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MAY MODIFY AND/OR
|
||||
REDISTRIBUTE THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES,
|
||||
INCLUDING ANY GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING
|
||||
OUT OF THE USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED
|
||||
TO LOSS OF DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY
|
||||
YOU OR THIRD PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER
|
||||
PROGRAMS), EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE
|
||||
POSSIBILITY OF SUCH DAMAGES.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
How to Apply These Terms to Your New Programs
|
||||
|
||||
If you develop a new program, and you want it to be of the greatest
|
||||
possible use to the public, the best way to achieve this is to make it
|
||||
free software which everyone can redistribute and change under these terms.
|
||||
|
||||
To do so, attach the following notices to the program. It is safest
|
||||
to attach them to the start of each source file to most effectively
|
||||
convey the exclusion of warranty; and each file should have at least
|
||||
the "copyright" line and a pointer to where the full notice is found.
|
||||
|
||||
tkDNN
|
||||
Copyright (C) 2017 Francesco Gatti
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation; either version 2 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License along
|
||||
with this program; if not, write to the Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
||||
|
||||
Also add information on how to contact you by electronic and paper mail.
|
||||
|
||||
If the program is interactive, make it output a short notice like this
|
||||
when it starts in an interactive mode:
|
||||
|
||||
Gnomovision version 69, Copyright (C) year name of author
|
||||
Gnomovision comes with ABSOLUTELY NO WARRANTY; for details type `show w'.
|
||||
This is free software, and you are welcome to redistribute it
|
||||
under certain conditions; type `show c' for details.
|
||||
|
||||
The hypothetical commands `show w' and `show c' should show the appropriate
|
||||
parts of the General Public License. Of course, the commands you use may
|
||||
be called something other than `show w' and `show c'; they could even be
|
||||
mouse-clicks or menu items--whatever suits your program.
|
||||
|
||||
You should also get your employer (if you work as a programmer) or your
|
||||
school, if any, to sign a "copyright disclaimer" for the program, if
|
||||
necessary. Here is a sample; alter the names:
|
||||
|
||||
Yoyodyne, Inc., hereby disclaims all copyright interest in the program
|
||||
`Gnomovision' (which makes passes at compilers) written by James Hacker.
|
||||
|
||||
<signature of Ty Coon>, 1 April 1989
|
||||
Ty Coon, President of Vice
|
||||
|
||||
This General Public License does not permit incorporating your program into
|
||||
proprietary programs. If your program is a subroutine library, you may
|
||||
consider it more useful to permit linking proprietary applications with the
|
||||
library. If this is what you want to do, use the GNU Lesser General
|
||||
Public License instead of this License.
|
||||
@@ -1,51 +1,215 @@
|
||||
# tkDNN
|
||||
tkDNN is a Deep Neural Network library built with cuDNN primitives specifically thought to work on NVIDIA TK1(and all successive) board.<br>
|
||||
The main scope is to do high performance inference on already trained models.
|
||||
tkDNN is a Deep Neural Network library built with cuDNN and tensorRT primitives, specifically thought to work on NVIDIA Jetson Boards. It has been tested on TK1(branch cudnn2), TX1, TX2, AGX Xavier, Nano and several discrete GPUs.
|
||||
The main goal of this project is to exploit NVIDIA boards as much as possible to obtain the best inference performance. It does not allow training.
|
||||
|
||||
this branch actually work on every NVIDIA GPU that support the dependencies:
|
||||
* CUDA 10.0
|
||||
* CUDNN 7.603
|
||||
* TENSORRT 6.01
|
||||
* OPENCV 4.1
|
||||
|
||||
## Workflow
|
||||
The recommended workflow follow these step:
|
||||
* Build and train a model in Keras (on any PC)
|
||||
* Export weights and bias
|
||||
* Define the model on tkDNN
|
||||
* Do inference (on TK1)
|
||||
If you use tkDNN in your research, please cite the [following paper](https://ieeexplore.ieee.org/stamp/stamp.jsp?arnumber=9212130&casa_token=sQTJXi7tJNoAAAAA:BguH9xCIY48MxbtDS3LXzIXzO-9sWArm7Hd7y7BwaLmqRuM_Gx8bOYizFPNMNtpo5K0kB-P-). For use in commercial solutions, write at gattifrancesco@hotmail.it and micaela.verucchi@unimore.it or refer to https://hipert.unimore.it/ .
|
||||
|
||||
## Compile the library
|
||||
Build with cmake
|
||||
```
|
||||
@inproceedings{verucchi2020systematic,
|
||||
title={A Systematic Assessment of Embedded Neural Networks for Object Detection},
|
||||
author={Verucchi, Micaela and Brilli, Gianluca and Sapienza, Davide and Verasani, Mattia and Arena, Marco and Gatti, Francesco and Capotondi, Alessandro and Cavicchioli, Roberto and Bertogna, Marko and Solieri, Marco},
|
||||
booktitle={2020 25th IEEE International Conference on Emerging Technologies and Factory Automation (ETFA)},
|
||||
volume={1},
|
||||
pages={937--944},
|
||||
year={2020},
|
||||
organization={IEEE}
|
||||
}
|
||||
```
|
||||
|
||||
### What's new
|
||||
#### 20 July 2021
|
||||
- [x] Support to sematic segmentation [README](docs/README_seg.md)
|
||||
- [x] Support 2D/3D Object Detection and Tracking [README](docs/README_2d3dtracking.md)
|
||||
#### 24 November 2021
|
||||
- [x] Support to sematic segmentation on cuda 11
|
||||
- [x] Support to TensorRT8. (thanks to [Harshvardhan Chandirasekar](https://github.com/perseusdg))
|
||||
#### 30 March 2022
|
||||
- [x] Support to monocular depth esitmation [README](docs/README_depth.md) (thanks to [Harshvardhan Chandirasekar](https://github.com/perseusdg))
|
||||
|
||||
|
||||
|
||||
## FPS Results
|
||||
Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimension as the input size, on
|
||||
* RTX 2080Ti (CUDA 10.2, TensorRT 7.0.0, Cudnn 7.6.5);
|
||||
* Xavier AGX, Jetpack 4.3 (CUDA 10.0, CUDNN 7.6.3, tensorrt 6.0.1 );
|
||||
* Xavier NX, Jetpack 4.4 (CUDA 10.2, CUDNN 8.0.0, tensorrt 7.1.0 ).
|
||||
* Tx2, Jetpack 4.2 (CUDA 10.0, CUDNN 7.3.1, tensorrt 5.0.6 );
|
||||
* Jetson Nano, Jetpack 4.4 (CUDA 10.2, CUDNN 8.0.0, tensorrt 7.1.0 ).
|
||||
|
||||
| Platform | Network | FP32, B=1 | FP32, B=4 | FP16, B=1 | FP16, B=4 | INT8, B=1 | INT8, B=4 |
|
||||
| :------: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: |
|
||||
| RTX 2080Ti | yolo4 320 | 118.59 | 237.31 | 207.81 | 443.32 | 262.37 | 530.93 |
|
||||
| RTX 2080Ti | yolo4 416 | 104.81 | 162.86 | 169.06 | 293.78 | 206.93 | 353.26 |
|
||||
| RTX 2080Ti | yolo4 512 | 92.98 | 132.43 | 140.36 | 215.17 | 165.35 | 254.96 |
|
||||
| RTX 2080Ti | yolo4 608 | 63.77 | 81.53 | 111.39 | 152.89 | 127.79 | 184.72 |
|
||||
| AGX Xavier | yolo4 320 | 26.78 | 32.05 | 57.14 | 79.05 | 73.15 | 97.56 |
|
||||
| AGX Xavier | yolo4 416 | 19.96 | 21.52 | 41.01 | 49.00 | 50.81 | 60.61 |
|
||||
| AGX Xavier | yolo4 512 | 16.58 | 16.98 | 31.12 | 33.84 | 37.82 | 41.28 |
|
||||
| AGX Xavier | yolo4 608 | 9.45 | 10.13 | 21.92 | 23.36 | 27.05 | 28.93 |
|
||||
| Xavier NX | yolo4 320 | 14.56 | 16.25 | 30.14 | 41.15 | 42.13 | 53.42 |
|
||||
| Xavier NX | yolo4 416 | 10.02 | 10.60 | 22.43 | 25.59 | 29.08 | 32.94 |
|
||||
| Xavier NX | yolo4 512 | 8.10 | 8.32 | 15.78 | 17.13 | 20.51 | 22.46 |
|
||||
| Xavier NX | yolo4 608 | 5.26 | 5.18 | 11.54 | 12.06 | 15.09 | 15.82 |
|
||||
| Tx2 | yolo4 320 | 11.18 | 12.07 | 15.32 | 16.31 | - | - |
|
||||
| Tx2 | yolo4 416 | 7.30 | 7.58 | 9.45 | 9.90 | - | - |
|
||||
| Tx2 | yolo4 512 | 5.96 | 5.95 | 7.22 | 7.23 | - | - |
|
||||
| Tx2 | yolo4 608 | 3.63 | 3.65 | 4.67 | 4.70 | - | - |
|
||||
| Nano | yolo4 320 | 4.23 | 4.55 | 6.14 | 6.53 | - | - |
|
||||
| Nano | yolo4 416 | 2.88 | 3.00 | 3.90 | 4.04 | - | - |
|
||||
| Nano | yolo4 512 | 2.32 | 2.34 | 3.02 | 3.04 | - | - |
|
||||
| Nano | yolo4 608 | 1.40 | 1.41 | 1.92 | 1.93 | - | - |
|
||||
|
||||
## MAP Results
|
||||
Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001
|
||||
|
||||
| | CodaLab | CodaLab | CodaLab | CodaLab | tkDNN map | tkDNN map |
|
||||
| -------------------- | :-----------: | :-------: | :-----------: | :---------: | :-----------: | :-------: |
|
||||
| | **tkDNN** | **tkDNN** | **darknet** | **darknet** | **tkDNN** | **tkDNN** |
|
||||
| | MAP(0.5:0.95) | AP50 | MAP(0.5:0.95) | AP50 | MAP(0.5:0.95) | AP50 |
|
||||
| Yolov3 (416x416) | 0.381 | 0.675 | 0.380 | 0.675 | 0.372 | 0.663 |
|
||||
| yolov4 (416x416) | 0.468 | 0.705 | 0.471 | 0.710 | 0.459 | 0.695 |
|
||||
| yolov3tiny (416x416) | 0.096 | 0.202 | 0.096 | 0.201 | 0.093 | 0.198 |
|
||||
| yolov4tiny (416x416) | 0.202 | 0.400 | 0.201 | 0.400 | 0.197 | 0.395 |
|
||||
| Cnet-dla34 (512x512) | 0.366 | 0.543 | \- | \- | 0.361 | 0.535 |
|
||||
| mv2SSD (512x512) | 0.226 | 0.381 | \- | \- | 0.223 | 0.378 |
|
||||
|
||||
## Index
|
||||
- [tkDNN](#tkdnn)
|
||||
- [Index](#index)
|
||||
- [Dependencies](#dependencies)
|
||||
- [How to compile this repo](#how-to-compile-this-repo)
|
||||
- [Workflow](#workflow)
|
||||
- [Exporting weights](#exporting-weights)
|
||||
- [Run the demos](#run-the-demos)
|
||||
- [tkDNN on Windows 10 or Windows 11](#tkdnn-on-windows-10-or-windows-11)
|
||||
- [Existing tests and supported networks](#existing-tests-and-supported-networks)
|
||||
- [References](#references)
|
||||
|
||||
|
||||
## Dependencies
|
||||
This branch works on every NVIDIA GPU that supports the following (latest tested) dependencies:
|
||||
* CUDA 11.3 (or >= 10.2)
|
||||
* cuDNN 8.2.1 (or >= 8.0.4)
|
||||
* TensorRT 8.0.3 (or >=7.2)
|
||||
* OpenCV 4.5.4 (or >=4)
|
||||
* cmake 3.21 (or >= 3.15)
|
||||
* yaml-cpp 0.5.2
|
||||
* eigen3 3.3.4
|
||||
* curl 7.58
|
||||
|
||||
```
|
||||
sudo apt install libyaml-cpp-dev curl libeigen3-dev
|
||||
|
||||
```
|
||||
|
||||
#### About OpenCV
|
||||
To compile and install OpenCV4 with contrib us the script ```install_OpenCV4.sh```. It will download and compile OpenCV in Download folder.
|
||||
```
|
||||
bash scripts/install_OpenCV4.sh
|
||||
```
|
||||
If you have OpenCV compiled with cuda and contrib and want to use it with tkDNN pass ```ENABLE_OPENCV_CUDA_CONTRIB=ON``` flag when compiling tkDBB
|
||||
. If the flag is not passed,the preprocessing of the networks is computed on the CPU, otherwise on the GPU. In the latter case some milliseconds are saved in the end-to-end latency.
|
||||
|
||||
## How to compile this repo
|
||||
Build with cmake. If using Ubuntu 18.04 a new version of cmake is needed (3.15 or above).
|
||||
On both linux and windows ,the ```CMAKE_BUILD_TYPE``` variable needs to be defined as either ```Release``` or ```Debug```.
|
||||
```
|
||||
git clone https://github.com/ceccocats/tkDNN
|
||||
cd tkDNN
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
# use -DTEST_DATA=False to skip dataset download
|
||||
cmake -DCMAKE_BUILD_TYPE=Release ..
|
||||
make
|
||||
```
|
||||
during the cmake configuration it will be dowloaded the weights needed for running
|
||||
the tests
|
||||
|
||||
## Test
|
||||
Assumiung you have correctly builded the library these are the test ready to exec:
|
||||
* test_simple: a simple convolutional and dense network (CUDNN only)
|
||||
* test_mnist: the famous mnist netwok (CUDNN and TENSORRT)
|
||||
* test_mnistRT: the mnist network hardcoded in using tensorRT apis (TENSORRT only)
|
||||
* test_yolo: YOLO detection network (CUDNN and TENSORRT)
|
||||
* test_yolo_tiny: smaller version of YOLO (CUDNN and TENSRRT)
|
||||
* test_yolo3_berkeley: our yolo3 version trained with BDD100K dateset
|
||||
## Workflow
|
||||
Steps needed to do inference on tkDNN with a custom neural network.
|
||||
* Build and train a NN model with your favorite framework.
|
||||
* Export weights and bias for each layer and save them in a binary file (one for layer).
|
||||
* Export outputs for each layer and save them in a binary file (one for layer).
|
||||
* Create a new test and define the network, layer by layer using the weights extracted and the output to check the results.
|
||||
* Do inference.
|
||||
|
||||
## yolo3 berkeley demo detection
|
||||
For the live detection you need to precompile the tensorRT file by luncing the desidered network test, this is the recommended process:
|
||||
```
|
||||
export TKDNN_MODE=FP16 # set the half floating point optimization
|
||||
rm yolo3_berkeley.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_yolo3_berkeley # run the yolo test (is slow)
|
||||
# with f16 inference the result will be a bit incorrect
|
||||
```
|
||||
this will genereate a yolo3_berkeley.rt file that can be used for live detection:
|
||||
```
|
||||
./yolo3_demo # launch detection on a demo video
|
||||
./yolo3_demo yolo3_berkeley.rt /dev/video0 # launch detection on device 0
|
||||
```
|
||||
## Exporting weights
|
||||
|
||||
For specific details on how to export weights see [HERE](./docs/exporting_weights.md).
|
||||
|
||||
## Run the demos
|
||||
|
||||
For specific details on how to run:
|
||||
- 2D object detection demos, details on FP16, INT8 and batching see [HERE](./docs/demo.md).
|
||||
- segmentation demos see [HERE](./docs/README_seg.md).
|
||||
- monocular depth estimation see [HERE](./docs/README_depth.md).
|
||||
- 2D/3D object detection and tracking demos see [HERE](./docs/README_2d3dtracking.md).
|
||||
- mAP demo to evaluate 2D object detectors see [HERE](./docs/mAP_demo.md).
|
||||
|
||||

|
||||
|
||||
## tkDNN on Windows 10 or Windows 11
|
||||
|
||||
For specific details on how to run tkDNN on Windows 10/11 see [HERE](./docs/windows.md).
|
||||
|
||||
## Existing tests and supported networks
|
||||
|
||||
| Test Name | Network | Dataset | N Classes | Input size | Weights |
|
||||
| :---------------- | :-------------------------------------------- | :-----------------------------------------------------------: | :-------: | :-----------: | :------------------------------------------------------------------------ |
|
||||
| yolo | YOLO v2<sup>1</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 608x608 | [weights](https://cloud.hipert.unimore.it/s/nf4PJ3k8bxBETwL/download) |
|
||||
| yolo_224 | YOLO v2<sup>1</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 224x224 | weights |
|
||||
| yolo_berkeley | YOLO v2<sup>1</sup> | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 416x736 | weights |
|
||||
| yolo_relu | YOLO v2 (with ReLU, not Leaky)<sup>1</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 416x416 | weights |
|
||||
| yolo_tiny | YOLO v2 tiny<sup>1</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/m3orfJr8pGrN5mQ/download) |
|
||||
| yolo_voc | YOLO v2<sup>1</sup> | [VOC ](http://host.robots.ox.ac.uk/pascal/VOC/) | 21 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/DJC5Fi2pEjfNDP9/download) |
|
||||
| yolo3 | YOLO v3<sup>2</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/jPXmHyptpLoNdNR/download) |
|
||||
| yolo3_512 | YOLO v3<sup>2</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/RGecMeGLD4cXEWL/download) |
|
||||
| yolo3_berkeley | YOLO v3<sup>2</sup> | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 320x544 | [weights](https://cloud.hipert.unimore.it/s/o5cHa4AjTKS64oD/download) |
|
||||
| yolo3_coco4 | YOLO v3<sup>2</sup> | [COCO 2014](http://cocodataset.org/) | 4 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/o27NDzSAartbyc4/download) |
|
||||
| yolo3_flir | YOLO v3<sup>2</sup> | [FREE FLIR](https://www.flir.com/oem/adas/adas-dataset-form/) | 3 | 320x544 | [weights](https://cloud.hipert.unimore.it/s/62DECncmF6bMMiH/download) |
|
||||
| yolo3_tiny | YOLO v3 tiny<sup>2</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/LMcSHtWaLeps8yN/download) |
|
||||
| yolo3_tiny512 | YOLO v3 tiny<sup>2</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/8Zt6bHwHADqP4JC/download) |
|
||||
| dla34 | Deep Leayer Aggreagtion (DLA) 34<sup>3</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 224x224 | weights |
|
||||
| dla34_cnet | Centernet (DLA34 backend)<sup>4</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/KRZBbCQsKAtQwpZ/download) |
|
||||
| mobilenetv2ssd | Mobilnet v2 SSD Lite<sup>5</sup> | [VOC ](http://host.robots.ox.ac.uk/pascal/VOC/) | 21 | 300x300 | [weights](https://cloud.hipert.unimore.it/s/x4ZfxBKN23zAJQp/download) |
|
||||
| mobilenetv2ssd512 | Mobilnet v2 SSD Lite<sup>5</sup> | [COCO 2017](http://cocodataset.org/) | 81 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/pdCw2dYyHMJrcEM/download) |
|
||||
| resnet101 | Resnet 101<sup>6</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 224x224 | weights |
|
||||
| resnet101_cnet | Centernet (Resnet101 backend)<sup>4</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/5BTjHMWBcJk8g3i/download) |
|
||||
| csresnext50-panet-spp | Cross Stage Partial Network <sup>7</sup> | [COCO 2014](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/Kcs4xBozwY4wFx8/download) |
|
||||
| yolo4 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
|
||||
| yolo4_320 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 320x320 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
|
||||
| yolo4_512 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
|
||||
| yolo4_608 | Yolov4 <sup>8</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 608x608 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
|
||||
| yolo4_berkeley | Yolov4 <sup>8</sup> | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 544x320 | [weights](https://cloud.hipert.unimore.it/s/nkWFa5fgb4NTdnB/download) |
|
||||
| yolo4tiny | Yolov4 tiny <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) |
|
||||
| yolo4x | Yolov4x-mish <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) |
|
||||
| yolo4tiny_512 | Yolov4 tiny <sup>9</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) |
|
||||
| yolo4x-cps | Scaled Yolov4 <sup>10</sup> | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) |
|
||||
| shelfnet | ShelfNet18_realtime<sup>11</sup> | [Cityscapes](https://www.cityscapes-dataset.com/) | 19 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/mEDZMRJaGCFWSJF/download) |
|
||||
| shelfnet_berkeley | ShelfNet18_realtime<sup>11</sup> | [DeepDrive](https://bdd-data.berkeley.edu/) | 20 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/m92e7QdD9gYMF7f/download) |
|
||||
| dla34_cnet3d | Centernet3D (DLA34 backend)<sup>4</sup> | [KITTI 2017](http://www.cvlibs.net/datasets/kitti/eval_object.php?obj_benchmark=3d) | 1 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/2MDyWGzQsTKMjmR/download) |
|
||||
| dla34_ctrack | CenterTrack (DLA34 backend)<sup>12</sup> | [NuScenes 3D](https://www.nuscenes.org/) | 7 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/rjNfgGL9FtAXLHp/download) |
|
||||
| monodepth2 | Monodepth2 <sup>13</sup> | [KITTI DEPTH](http://www.cvlibs.net/datasets/kitti/raw_data.php) | - | 640x192 | [weights-mono](https://cloud.hipert.unimore.it/s/iYw9QwgP6CsqxLR/download) |
|
||||
| monodepth2 | Monodepth2 <sup>13</sup> | [KITTI DEPTH](http://www.cvlibs.net/datasets/kitti/raw_data.php) | - | 640x192 | [weights-stereo](https://cloud.hipert.unimore.it/s/XmwbWNXDfqyQ4EL/download) |
|
||||
|
||||
|
||||
## References
|
||||
|
||||
1. Redmon, Joseph, and Ali Farhadi. "YOLO9000: better, faster, stronger." Proceedings of the IEEE conference on computer vision and pattern recognition. 2017.
|
||||
2. Redmon, Joseph, and Ali Farhadi. "Yolov3: An incremental improvement." arXiv preprint arXiv:1804.02767 (2018).
|
||||
3. Yu, Fisher, et al. "Deep layer aggregation." Proceedings of the IEEE conference on computer vision and pattern recognition. 2018.
|
||||
4. Zhou, Xingyi, Dequan Wang, and Philipp Krähenbühl. "Objects as points." arXiv preprint arXiv:1904.07850 (2019).
|
||||
5. Sandler, Mark, et al. "Mobilenetv2: Inverted residuals and linear bottlenecks." Proceedings of the IEEE conference on computer vision and pattern recognition. 2018.
|
||||
6. He, Kaiming, et al. "Deep residual learning for image recognition." Proceedings of the IEEE conference on computer vision and pattern recognition. 2016.
|
||||
7. Wang, Chien-Yao, et al. "CSPNet: A New Backbone that can Enhance Learning Capability of CNN." arXiv preprint arXiv:1911.11929 (2019).
|
||||
8. Bochkovskiy, Alexey, Chien-Yao Wang, and Hong-Yuan Mark Liao. "YOLOv4: Optimal Speed and Accuracy of Object Detection." arXiv preprint arXiv:2004.10934 (2020).
|
||||
9. Bochkovskiy, Alexey, "Yolo v4, v3 and v2 for Windows and Linux" (https://github.com/AlexeyAB/darknet)
|
||||
10. Wang, Chien-Yao, Alexey Bochkovskiy, and Hong-Yuan Mark Liao. "Scaled-YOLOv4: Scaling Cross Stage Partial Network." arXiv preprint arXiv:2011.08036 (2020).
|
||||
11. Zhuang, Juntang, et al. "ShelfNet for fast semantic segmentation." Proceedings of the IEEE International Conference on Computer Vision Workshops. 2019.
|
||||
12. Zhou, Xingyi, Vladlen Koltun, and Philipp Krähenbühl. "Tracking objects as points." European Conference on Computer Vision. Springer, Cham, 2020.
|
||||
13. Godard, Clément, et al. "Digging into self-supervised monocular depth estimation." Proceedings of the IEEE/CVF International Conference on Computer Vision. 2019.
|
||||
|
||||
## Contributors
|
||||
The main contibutors, in chronological order, are:
|
||||
- [Francesco Gatti](https://github.com/ceccocats), francesco.gatti@hipert.it
|
||||
- [Micaela Verucchi](https://github.com/mive93), micaela.verucchi@unimore.it
|
||||
- [Davide Sapienza](https://github.com/sapienzadavide), davide.sapienza@unimore.it
|
||||
- [Harshvardhan Chandirasekar](https://github.com/perseusdg), f20180523@goa.bits-pilani.ac.in
|
||||
|
||||
+62
-29
@@ -1,33 +1,66 @@
|
||||
# Find the header files
|
||||
# find the library
|
||||
if(CUDA_FOUND)
|
||||
find_cuda_helper_libs(cudnn)
|
||||
set(CUDNN_LIBRARY ${CUDA_cudnn_LIBRARY} CACHE FILEPATH "location of the cuDNN library")
|
||||
unset(CUDA_cudnn_LIBRARY CACHE)
|
||||
|
||||
find_path(CUDNN_INCLUDE_DIR
|
||||
${CMAKE_SYSROOT}/usr/local/include
|
||||
${CMAKE_SYSROOT}/usr/include
|
||||
/usr/local/nvidia/tensorrt/include/
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
find_cuda_helper_libs(nvinfer)
|
||||
set(NVINFER_LIBRARY ${CUDA_nvinfer_LIBRARY} CACHE FILEPATH "location of the nvinfer library")
|
||||
unset(CUDA_nvinfer_LIBRARY CACHE)
|
||||
endif()
|
||||
|
||||
set(OLD_ROOT ${CMAKE_FIND_ROOT_PATH})
|
||||
list(APPEND CMAKE_FIND_ROOT_PATH /)
|
||||
list(APPEND CMAKE_FIND_LIBRARY_SUFFIXES .so.7)
|
||||
list(APPEND CMAKE_FIND_LIBRARY_SUFFIXES .so.5)
|
||||
find_library(CUDNN_LIB
|
||||
NAMES cudnn
|
||||
PATHS
|
||||
/usr/local/driveworks/targets/${CMAKE_SYSTEM_PROCESSOR}-Linux/lib
|
||||
/usr/lib/${CMAKE_SYSTEM_PROCESSOR}-linux-gnu/
|
||||
# find the include
|
||||
if(CUDNN_LIBRARY)
|
||||
find_path(CUDNN_INCLUDE_DIR
|
||||
cudnn.h
|
||||
PATHS ${CUDA_TOOLKIT_INCLUDE}
|
||||
DOC "location of cudnn.h"
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
find_library(CUDNN_NVLIB
|
||||
NAMES "nvinfer"
|
||||
PATHS
|
||||
/usr/local/driveworks/targets/${CMAKE_SYSTEM_PROCESSOR}-Linux/lib
|
||||
/usr/lib/${CMAKE_SYSTEM_PROCESSOR}-linux-gnu/
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
set(CMAKE_FIND_ROOT_PATH ${OLD_ROOT})
|
||||
)
|
||||
|
||||
set(CUDNN_LIBRARIES ${CUDNN_LIB} ${CUDNN_NVLIB})
|
||||
message("-- Found CUDNN: " ${CUDNN_LIB})
|
||||
message("-- Found NVINFER: " ${CUDNN_NVLIB})
|
||||
set(CUDNN_FOUND true)
|
||||
if(NOT CUDNN_INCLUDE_DIR)
|
||||
find_path(CUDNN_INCLUDE_DIR
|
||||
cudnn.h
|
||||
DOC "location of cudnn.h"
|
||||
)
|
||||
endif()
|
||||
|
||||
message("-- Found CUDNN: " ${CUDNN_LIBRARY})
|
||||
message("-- Found CUDNN include: " ${CUDNN_INCLUDE_DIR})
|
||||
endif()
|
||||
|
||||
if(NVINFER_LIBRARY)
|
||||
find_path(NVINFER_INCLUDE_DIR
|
||||
NvInfer.h
|
||||
PATHS ${CUDA_TOOLKIT_INCLUDE}
|
||||
DOC "location of NvInfer.h"
|
||||
NO_DEFAULT_PATH
|
||||
)
|
||||
|
||||
if(NOT NVINFER_INCLUDE_DIR)
|
||||
find_path(NVINFER_INCLUDE_DIR
|
||||
NvInfer.h
|
||||
DOC "location of NvInfer.h"
|
||||
)
|
||||
endif()
|
||||
|
||||
message("-- Found NVINFER: " ${NVINFER_LIBRARY})
|
||||
message("-- Found NVINFER include: " ${NVINFER_INCLUDE_DIR})
|
||||
endif()
|
||||
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
find_package_handle_standard_args(CUDNN
|
||||
FOUND_VAR CUDNN_FOUND
|
||||
REQUIRED_VARS
|
||||
CUDNN_LIBRARY
|
||||
CUDNN_INCLUDE_DIR
|
||||
VERSION_VAR CUDNN_VERSION
|
||||
)
|
||||
|
||||
if(CUDNN_FOUND)
|
||||
set(CUDNN_LIBRARIES ${CUDNN_LIBRARY} ${NVINFER_LIBRARY})
|
||||
set(CUDNN_INCLUDE_DIRS ${CUDNN_INCLUDE_DIR} ${NVINFER_INCLUDE_DIR})
|
||||
endif()
|
||||
|
||||
set(CUDNN_FOUND true)
|
||||
@@ -0,0 +1,20 @@
|
||||
#include <stdio.h>
|
||||
|
||||
int main(int argc, char **argv){
|
||||
cudaDeviceProp dP;
|
||||
float min_cc = 5.0;
|
||||
|
||||
int rc = cudaGetDeviceProperties(&dP, 0);
|
||||
if(rc != cudaSuccess) {
|
||||
cudaError_t error = cudaGetLastError();
|
||||
printf("CUDA error: %s", cudaGetErrorString(error));
|
||||
return rc; /* Failure */
|
||||
}
|
||||
if((dP.major+(dP.minor/10)) < min_cc) {
|
||||
printf("Min Compute Capability of %2.1f required: %d.%d found\n Not Building CUDA Code", min_cc, dP.major, dP.minor);
|
||||
return 1; /* Failure */
|
||||
} else {
|
||||
printf("-arch=sm_%d%d", dP.major, dP.minor);
|
||||
return 0; /* Success */
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
classes : 80 #number of classes
|
||||
map_points : 101 #number of recall points (0 for all, 101 for COCO, 11 PascalVOC)
|
||||
map_levels : 10 #number of IoU step for the AP
|
||||
map_step : 0.05 #step of IoU
|
||||
IoU_thresh : 0.5 #starting IoU threshold
|
||||
conf_thresh : 0.001 #threshold on the condifence of the bbox
|
||||
verbose : false #print on screen information
|
||||
@@ -0,0 +1,7 @@
|
||||
classes : 3 #number of classes
|
||||
map_points : 101 #number of recall points (0 for all, 101 for COCO, 11 PascalVOC)
|
||||
map_levels : 10 #number of IoU step for the AP
|
||||
map_step : 0.05 #step of IoU
|
||||
IoU_thresh : 0.5 #starting IoU threshold
|
||||
conf_thresh : 0.0 #threshold on the condifence of the bbox
|
||||
verbose : false #print on screen information
|
||||
+110
-63
@@ -1,19 +1,14 @@
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h> /* srand, rand */
|
||||
#include <unistd.h>
|
||||
//#include <unistd.h>
|
||||
#include <mutex>
|
||||
#include "utils.h"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/videoio.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
#include "CenternetDetection.h"
|
||||
#include "MobilenetDetection.h"
|
||||
#include "Yolo3Detection.h"
|
||||
|
||||
bool gRun;
|
||||
bool SAVE_RESULT = false;
|
||||
bool gRun;
|
||||
|
||||
void sig_handler(int signo) {
|
||||
std::cout<<"request gateway stop\n";
|
||||
@@ -22,88 +17,140 @@ void sig_handler(int signo) {
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
|
||||
std::cout<<"detection\n";
|
||||
signal(SIGINT, sig_handler);
|
||||
|
||||
|
||||
char *net = "yolo3_berkeley.rt";
|
||||
// get config file path and read it
|
||||
#ifdef __linux__
|
||||
std::string config_file = "../demo/demoConfig.yaml";
|
||||
#elif _WIN32
|
||||
std::string config_file = "..\\..\\..\\demo\\demoConfig.yaml";
|
||||
#endif
|
||||
if(argc > 1)
|
||||
net = argv[1];
|
||||
char *input = "../demo/yolo_test.mp4";
|
||||
if(argc > 2)
|
||||
input = argv[2];
|
||||
config_file = argv[1];
|
||||
|
||||
YAML::Node conf = YAMLloadConf(config_file);
|
||||
if(!conf)
|
||||
FatalError("Problem with config file");
|
||||
|
||||
// read settings from config file
|
||||
std::string net = YAMLgetConf<std::string>(conf, "net", "yolo4tiny_fp32.rt");
|
||||
if(!fileExist(net.c_str()))
|
||||
FatalError("The given network does not exist. Create the rt first.");
|
||||
|
||||
#ifdef __linux__
|
||||
std::string input = YAMLgetConf<std::string>(conf, "input", "../demo/yolo_test.mp4");
|
||||
#elif _WIN32
|
||||
std::string input = YAMLgetConf<std::string>(conf, "win_input", "..\\..\\..\\demo\\yolo_test.mp4");
|
||||
#endif
|
||||
if(!fileExist(input.c_str()))
|
||||
FatalError("The given input video does not exist.");
|
||||
|
||||
char ntype = YAMLgetConf<char>(conf, "ntype", 'y');
|
||||
int n_classes = YAMLgetConf<int>(conf, "n_classes", 80);
|
||||
int n_batch = YAMLgetConf<int>(conf, "n_batch", 1);
|
||||
if(n_batch < 1 || n_batch > 64)
|
||||
FatalError("Batch dim not supported");
|
||||
float conf_thresh = YAMLgetConf<float>(conf, "conf_thresh", 0.3);
|
||||
bool show = YAMLgetConf<bool>(conf, "show", true);
|
||||
bool save = YAMLgetConf<bool>(conf, "save", false);
|
||||
|
||||
std::cout <<"Net settings - net: "<< net
|
||||
<<", ntype: "<< ntype
|
||||
<<", n_classes: "<< n_classes
|
||||
<<", n_batch: "<< n_batch
|
||||
<<", conf_thresh: "<< conf_thresh<<"\n";
|
||||
std::cout <<"Demo settings - input: "<< input
|
||||
<<", show: "<< show
|
||||
<<", save: "<< save<<"\n\n";
|
||||
|
||||
// create detection network
|
||||
tk::dnn::Yolo3Detection yolo;
|
||||
yolo.init(net);
|
||||
tk::dnn::CenternetDetection cnet;
|
||||
tk::dnn::MobilenetDetection mbnet;
|
||||
|
||||
gRun = true;
|
||||
tk::dnn::DetectionNN *detNN;
|
||||
|
||||
switch(ntype)
|
||||
{
|
||||
case 'y':
|
||||
detNN = &yolo;
|
||||
break;
|
||||
case 'c':
|
||||
detNN = &cnet;
|
||||
break;
|
||||
case 'm':
|
||||
detNN = &mbnet;
|
||||
n_classes++;
|
||||
break;
|
||||
default:
|
||||
FatalError("Network type not allowed (3rd parameter)\n");
|
||||
}
|
||||
|
||||
detNN->init(net,n_classes,n_batch,conf_thresh);
|
||||
|
||||
// open video stream
|
||||
cv::VideoCapture cap(input);
|
||||
if(!cap.isOpened())
|
||||
gRun = false;
|
||||
else
|
||||
std::cout<<"camera started\n";
|
||||
|
||||
|
||||
cv::VideoWriter resultVideo;
|
||||
if(SAVE_RESULT) {
|
||||
if(save) {
|
||||
int w = cap.get(cv::CAP_PROP_FRAME_WIDTH);
|
||||
int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT);
|
||||
resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h));
|
||||
}
|
||||
|
||||
if(show)
|
||||
cv::namedWindow("detection", cv::WINDOW_NORMAL);
|
||||
|
||||
cv::Mat frame;
|
||||
cv::Mat dnn_input;
|
||||
cv::namedWindow("detection", cv::WINDOW_NORMAL);
|
||||
|
||||
std::vector<cv::Mat> batch_frame;
|
||||
std::vector<cv::Mat> batch_dnn_input;
|
||||
|
||||
// start detection loop
|
||||
gRun = true;
|
||||
while(gRun) {
|
||||
cap >> frame;
|
||||
if(!frame.data) {
|
||||
batch_dnn_input.clear();
|
||||
batch_frame.clear();
|
||||
|
||||
for(int bi=0; bi< n_batch; ++bi){
|
||||
cap >> frame;
|
||||
if(!frame.data)
|
||||
break;
|
||||
|
||||
batch_frame.push_back(frame);
|
||||
|
||||
// this will be resized to the net format
|
||||
batch_dnn_input.push_back(frame.clone());
|
||||
}
|
||||
if(!frame.data)
|
||||
break;
|
||||
}
|
||||
|
||||
// this will be resized to the net format
|
||||
dnn_input = frame.clone();
|
||||
// TODO: async infer
|
||||
yolo.update(dnn_input);
|
||||
|
||||
// draw dets
|
||||
for(int i=0; i<yolo.detected.size(); i++) {
|
||||
tk::dnn::box b = yolo.detected[i];
|
||||
int x0 = b.x;
|
||||
int x1 = b.x + b.w;
|
||||
int y0 = b.y;
|
||||
int y1 = b.y + b.h;
|
||||
std::string det_class = yolo.getYoloLayer()->classesNames[b.cl];
|
||||
float prob = b.prob;
|
||||
|
||||
std::cout<<det_class<<" ("<<prob<<"): "<<x0<<" "<<y0<<" "<<x1<<" "<<y1<<"\n";
|
||||
// draw rectangle
|
||||
cv::rectangle(frame, cv::Point(x0, y0), cv::Point(x1, y1), yolo.colors[b.cl], 2);
|
||||
|
||||
// draw label
|
||||
int baseline = 0;
|
||||
float fontScale = 0.5;
|
||||
int thickness = 2;
|
||||
cv::Size textSize = getTextSize(det_class, cv::FONT_HERSHEY_SIMPLEX, fontScale, thickness, &baseline);
|
||||
cv::rectangle(frame, cv::Point(x0, y0), cv::Point((x0 + textSize.width - 2), (y0 - textSize.height - 2)), yolo.colors[b.cl], -1);
|
||||
cv::putText(frame, det_class, cv::Point(x0, (y0 - (baseline / 2))), cv::FONT_HERSHEY_SIMPLEX, fontScale, cv::Scalar(255, 255, 255), thickness);
|
||||
}
|
||||
|
||||
cv::imshow("detection", frame);
|
||||
cv::waitKey(1);
|
||||
if(SAVE_RESULT)
|
||||
//inference
|
||||
detNN->update(batch_dnn_input, n_batch);
|
||||
detNN->draw(batch_frame);
|
||||
|
||||
if(show){
|
||||
for(int bi=0; bi< n_batch; ++bi){
|
||||
cv::imshow("detection", batch_frame[bi]);
|
||||
cv::waitKey(1);
|
||||
}
|
||||
}
|
||||
if(n_batch == 1 && save)
|
||||
resultVideo << frame;
|
||||
}
|
||||
|
||||
std::cout<<"detection end\n";
|
||||
|
||||
|
||||
|
||||
double mean = 0;
|
||||
std::cout<<COL_GREENB<<"\n\nTime stats:\n";
|
||||
std::cout<<"Min: "<<*std::min_element(yolo.stats.begin(), yolo.stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(yolo.stats.begin(), yolo.stats.end())<<" ms\n";
|
||||
double mean = 0; for(int i=0; i<yolo.stats.size(); i++) mean += yolo.stats[i]; mean /= yolo.stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\n"<<COL_END;
|
||||
std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())<<" ms\n";
|
||||
for(int i=0; i<detNN->stats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\t"<<1000/(mean)<<" FPS\n"<<COL_END;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h> /* srand, rand */
|
||||
//#include <unistd.h>
|
||||
#include <mutex>
|
||||
|
||||
#include "demo_utils.h"
|
||||
#include "CenternetDetection3D.h"
|
||||
|
||||
bool gRun;
|
||||
bool SAVE_RESULT = false;
|
||||
|
||||
void sig_handler(int signo) {
|
||||
std::cout<<"request gateway stop\n";
|
||||
gRun = false;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
|
||||
std::cout<<"detection\n";
|
||||
signal(SIGINT, sig_handler);
|
||||
|
||||
|
||||
std::string net = "dla34_cnet3d_fp32.rt";
|
||||
if(argc > 1)
|
||||
net = argv[1];
|
||||
#ifdef __linux__
|
||||
std::string input = "../demo/yolo_test.mp4";
|
||||
#elif _WIN32
|
||||
std::string input = "..\\..\\..\\demo\\yolo_test.mp4";
|
||||
#endif
|
||||
|
||||
if(argc > 2)
|
||||
input = argv[2];
|
||||
std::string calib_params = "";
|
||||
if(argc > 3)
|
||||
calib_params = argv[3];
|
||||
char ntype = 'c';
|
||||
if(argc > 4)
|
||||
ntype = argv[4][0];
|
||||
int n_classes = 3;
|
||||
if(argc > 5)
|
||||
n_classes = atoi(argv[5]);
|
||||
int n_batch = 1;
|
||||
if(argc > 6)
|
||||
n_batch = atoi(argv[6]);
|
||||
|
||||
bool show = true;
|
||||
if(argc > 7)
|
||||
show = atoi(argv[7]);
|
||||
float conf_thresh=0.3;
|
||||
if(argc > 8)
|
||||
conf_thresh = atof(argv[8]);
|
||||
|
||||
if(n_batch < 1 || n_batch > 64)
|
||||
FatalError("Batch dim not supported");
|
||||
|
||||
if(!show)
|
||||
SAVE_RESULT = true;
|
||||
|
||||
tk::dnn::CenternetDetection3D cnet;
|
||||
|
||||
tk::dnn::DetectionNN3D *detNN;
|
||||
|
||||
switch(ntype)
|
||||
{
|
||||
case 'c':
|
||||
detNN = &cnet;
|
||||
break;
|
||||
default:
|
||||
FatalError("Network type not allowed (3rd parameter)\n");
|
||||
}
|
||||
std::vector<cv::Mat> calibs;
|
||||
if(!calib_params.empty() && calib_params!="NULL") {
|
||||
std::cout<<"calib_params: "<<calib_params<<std::endl;
|
||||
cv::Mat calib;
|
||||
// the calibration matrix must be a 3x3 matrix
|
||||
readCalibrationMatrix(calib_params, calib);
|
||||
for(int bi=0; bi< n_batch; ++bi)
|
||||
calibs.push_back(calib);
|
||||
}
|
||||
detNN->init(net, n_classes, n_batch, conf_thresh, calibs);
|
||||
|
||||
gRun = true;
|
||||
|
||||
cv::VideoCapture cap(input);
|
||||
if(!cap.isOpened())
|
||||
gRun = false;
|
||||
else
|
||||
std::cout<<"camera started\n";
|
||||
|
||||
cv::VideoWriter resultVideo;
|
||||
if(SAVE_RESULT) {
|
||||
int w = cap.get(cv::CAP_PROP_FRAME_WIDTH);
|
||||
int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT);
|
||||
resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h));
|
||||
}
|
||||
cv::Mat frame;
|
||||
if(show)
|
||||
cv::namedWindow("detection", cv::WINDOW_NORMAL);
|
||||
|
||||
std::vector<cv::Mat> batch_frame;
|
||||
std::vector<cv::Mat> batch_dnn_input;
|
||||
|
||||
while(gRun) {
|
||||
batch_dnn_input.clear();
|
||||
batch_frame.clear();
|
||||
|
||||
for(int bi=0; bi< n_batch; ++bi){
|
||||
cap >> frame;
|
||||
if(!frame.data)
|
||||
break;
|
||||
batch_frame.push_back(frame);
|
||||
|
||||
// this will be resized to the net format
|
||||
batch_dnn_input.push_back(frame.clone());
|
||||
}
|
||||
if(!frame.data)
|
||||
break;
|
||||
|
||||
//inference
|
||||
detNN->update(batch_dnn_input, n_batch, false, nullptr, false);
|
||||
detNN->draw(batch_frame);
|
||||
|
||||
if(show){
|
||||
for(int bi=0; bi< n_batch; ++bi){
|
||||
cv::imshow("detection", batch_frame[bi]);
|
||||
cv::waitKey(1);
|
||||
}
|
||||
}
|
||||
if(n_batch == 1 && SAVE_RESULT)
|
||||
resultVideo << frame;
|
||||
}
|
||||
|
||||
std::cout<<"detection end\n";
|
||||
double mean = 0;
|
||||
|
||||
std::cout<<COL_GREENB<<"\n\nTime preprocessing stats:\n";
|
||||
std::cout<<"Min: "<<*std::min_element(detNN->pre_stats.begin(), detNN->pre_stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(detNN->pre_stats.begin(), detNN->pre_stats.end())<<" ms\n";
|
||||
for(int i=0; i<detNN->pre_stats.size(); i++) mean += detNN->pre_stats[i]; mean /= detNN->pre_stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\n"<<COL_END;
|
||||
mean=0;
|
||||
std::cout<<COL_GREENB<<"\n\nTime stats:\n";
|
||||
std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())<<" ms\n";
|
||||
for(int i=0; i<detNN->stats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\n"<<COL_END;
|
||||
mean=0;
|
||||
std::cout<<COL_GREENB<<"\n\nTime postprocessing stats:\n";
|
||||
std::cout<<"Min: "<<*std::min_element(detNN->post_stats.begin(), detNN->post_stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(detNN->post_stats.begin(), detNN->post_stats.end())<<" ms\n";
|
||||
for(int i=0; i<detNN->post_stats.size(); i++) mean += detNN->post_stats[i]; mean /= detNN->post_stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\n"<<COL_END;
|
||||
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h> /* srand, rand */
|
||||
//#include <unistd.h>
|
||||
#include <mutex>
|
||||
|
||||
#include "tkDNN/DepthNN.h"
|
||||
|
||||
bool gRun;
|
||||
|
||||
void sig_handler(int signo) {
|
||||
std::cout<<"request gateway stop\n";
|
||||
gRun = false;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
|
||||
signal(SIGINT, sig_handler);
|
||||
|
||||
std::string net = "monodepth2_fp32.rt";
|
||||
if(argc > 1)
|
||||
net = argv[1];
|
||||
#ifdef __linux__
|
||||
std::string input = "../demo/yolo_test.mp4";
|
||||
#elif _WIN32
|
||||
std::string input = "..\\..\\..\\demo\\yolo_test.mp4";
|
||||
#endif
|
||||
if(argc > 2)
|
||||
input = argv[2];
|
||||
bool show = true;
|
||||
if(argc > 3)
|
||||
show = atoi(argv[3]);
|
||||
bool save = true;
|
||||
if(argc > 4)
|
||||
save = atoi(argv[4]);
|
||||
|
||||
std::cout <<"Net settings - net: "<< net
|
||||
<<"\n";
|
||||
std::cout <<"Demo settings - input: "<< input
|
||||
<<", show: "<< show
|
||||
<<", save: "<< save<<"\n\n";
|
||||
|
||||
tk::dnn::DepthNN depthNN;
|
||||
|
||||
// create depth network
|
||||
int n_batch = 1;
|
||||
depthNN.init(net, n_batch);
|
||||
|
||||
// open video stream
|
||||
cv::VideoCapture cap(input);
|
||||
if(!cap.isOpened())
|
||||
gRun = false;
|
||||
else
|
||||
std::cout<<"camera started\n";
|
||||
|
||||
cv::VideoWriter resultVideo;
|
||||
if(save) {
|
||||
int w = depthNN.output_w;
|
||||
int h = depthNN.output_h;
|
||||
resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h));
|
||||
}
|
||||
|
||||
if(show)
|
||||
cv::namedWindow("depth", cv::WINDOW_NORMAL);
|
||||
|
||||
cv::Mat frame;
|
||||
std::vector<cv::Mat> batch_frame;
|
||||
std::vector<cv::Mat> batch_dnn_input;
|
||||
|
||||
// start detection loop
|
||||
gRun = true;
|
||||
while(gRun) {
|
||||
batch_dnn_input.clear();
|
||||
batch_frame.clear();
|
||||
|
||||
//read frame
|
||||
cap >> frame;
|
||||
if(!frame.data)
|
||||
break;
|
||||
batch_frame.push_back(frame);
|
||||
batch_dnn_input.push_back(frame.clone());
|
||||
|
||||
//inference
|
||||
depthNN.update(batch_dnn_input, 1);
|
||||
if(show){
|
||||
cv::imshow("depth", depthNN.depthMats[0]);
|
||||
cv::waitKey(1);
|
||||
|
||||
}
|
||||
|
||||
if(save)
|
||||
resultVideo << depthNN.depthMats[0];
|
||||
}
|
||||
|
||||
std::cout<<"detection end\n";
|
||||
|
||||
double mean = 0;
|
||||
std::cout<<COL_GREENB<<"\n\nTime stats depth:\n";
|
||||
std::cout<<"Min: "<<*std::min_element(depthNN.stats.begin(), depthNN.stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(depthNN.stats.begin(), depthNN.stats.end())<<" ms\n";
|
||||
for(int i=0; i<depthNN.stats.size(); i++) mean += depthNN.stats[i]; mean /= depthNN.stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\t"<<1000/(mean)<<" FPS\n";
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,160 @@
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h> /* srand, rand */
|
||||
//#include <unistd.h>
|
||||
#include <mutex>
|
||||
|
||||
#include "demo_utils.h"
|
||||
#include "CenterTrack.h"
|
||||
|
||||
bool gRun;
|
||||
bool SAVE_RESULT = false;
|
||||
|
||||
void sig_handler(int signo) {
|
||||
std::cout<<"request gateway stop\n";
|
||||
gRun = false;
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
|
||||
std::cout<<"detection\n";
|
||||
signal(SIGINT, sig_handler);
|
||||
|
||||
|
||||
std::string net = "dla34_cnet3d_track_fp32.rt";
|
||||
if(argc > 1)
|
||||
net = argv[1];
|
||||
#ifdef __linux__
|
||||
std::string input = "../demo/yolo_test.mp4";
|
||||
#elif _WIN32
|
||||
std::string input = "..\\..\\..\\demo\\yolo_test.mp4";
|
||||
#endif
|
||||
|
||||
if(argc > 2)
|
||||
input = argv[2];
|
||||
std::string calib_params = "";
|
||||
if(argc > 3)
|
||||
calib_params = argv[3];
|
||||
char ntype = 'c';
|
||||
if(argc > 4)
|
||||
ntype = argv[4][0];
|
||||
int n_classes = 3;
|
||||
if(argc > 5)
|
||||
n_classes = atoi(argv[5]);
|
||||
int n_batch = 1;
|
||||
if(argc > 6)
|
||||
n_batch = atoi(argv[6]);
|
||||
bool show = true;
|
||||
if(argc > 7)
|
||||
show = atoi(argv[7]);
|
||||
float conf_thresh=0.3;
|
||||
if(argc > 8)
|
||||
conf_thresh = atof(argv[8]);
|
||||
bool t3d = true;
|
||||
if(argc > 9)
|
||||
t3d = atoi(argv[9]);
|
||||
if(n_batch < 1 || n_batch > 64)
|
||||
FatalError("Batch dim not supported");
|
||||
|
||||
if(!show)
|
||||
SAVE_RESULT = true;
|
||||
|
||||
tk::dnn::CenterTrack ctrack;
|
||||
|
||||
tk::dnn::TrackingNN *trackNN;
|
||||
|
||||
switch(ntype)
|
||||
{
|
||||
case 'c':
|
||||
trackNN = &ctrack;
|
||||
break;
|
||||
default:
|
||||
FatalError("Network type not allowed (3rd parameter)\n");
|
||||
}
|
||||
std::vector<cv::Mat> calibs;
|
||||
if(!calib_params.empty() && calib_params!="NULL") {
|
||||
std::cout<<"calib_params: "<<calib_params<<std::endl;
|
||||
cv::Mat calib;
|
||||
// the calibration matrix must be a 3x3 matrix
|
||||
readCalibrationMatrix(calib_params, calib);
|
||||
for(int bi=0; bi< n_batch; ++bi)
|
||||
calibs.push_back(calib);
|
||||
}
|
||||
trackNN->init(net, n_classes, n_batch, conf_thresh, t3d, calibs);
|
||||
|
||||
gRun = true;
|
||||
|
||||
cv::VideoCapture cap(input);
|
||||
if(!cap.isOpened())
|
||||
gRun = false;
|
||||
else
|
||||
std::cout<<"camera started\n";
|
||||
|
||||
cv::VideoWriter resultVideo;
|
||||
if(SAVE_RESULT) {
|
||||
int w = cap.get(cv::CAP_PROP_FRAME_WIDTH);
|
||||
int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT);
|
||||
resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h));
|
||||
}
|
||||
cv::Mat frame;
|
||||
if(show)
|
||||
cv::namedWindow("detection", cv::WINDOW_NORMAL);
|
||||
|
||||
std::vector<cv::Mat> batch_frame;
|
||||
std::vector<cv::Mat> batch_dnn_input;
|
||||
|
||||
while(gRun) {
|
||||
batch_dnn_input.clear();
|
||||
batch_frame.clear();
|
||||
|
||||
for(int bi=0; bi< n_batch; ++bi){
|
||||
cap >> frame;
|
||||
if(!frame.data)
|
||||
break;
|
||||
batch_frame.push_back(frame);
|
||||
|
||||
// this will be resized to the net format
|
||||
batch_dnn_input.push_back(frame.clone());
|
||||
}
|
||||
if(!frame.data)
|
||||
break;
|
||||
|
||||
//inference
|
||||
trackNN->update(batch_dnn_input, n_batch, false, nullptr, false);
|
||||
trackNN->draw(batch_frame);
|
||||
|
||||
if(show){
|
||||
for(int bi=0; bi< n_batch; ++bi){
|
||||
cv::imshow("detection", batch_frame[bi]);
|
||||
cv::waitKey(1);
|
||||
}
|
||||
}
|
||||
if(n_batch == 1 && SAVE_RESULT)
|
||||
resultVideo << frame;
|
||||
}
|
||||
|
||||
std::cout<<"detection end\n";
|
||||
double mean = 0;
|
||||
|
||||
std::cout<<COL_GREENB<<"\n\nTime preprocessing stats:\n";
|
||||
std::cout<<"Min: "<<*std::min_element(trackNN->pre_stats.begin(), trackNN->pre_stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(trackNN->pre_stats.begin(), trackNN->pre_stats.end())<<" ms\n";
|
||||
for(int i=0; i<trackNN->pre_stats.size(); i++) mean += trackNN->pre_stats[i]; mean /= trackNN->pre_stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\n"<<COL_END;
|
||||
mean=0;
|
||||
std::cout<<COL_GREENB<<"\n\nTime stats:\n";
|
||||
std::cout<<"Min: "<<*std::min_element(trackNN->stats.begin(), trackNN->stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(trackNN->stats.begin(), trackNN->stats.end())<<" ms\n";
|
||||
for(int i=0; i<trackNN->stats.size(); i++) mean += trackNN->stats[i]; mean /= trackNN->stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\n"<<COL_END;
|
||||
mean=0;
|
||||
std::cout<<COL_GREENB<<"\n\nTime postprocessing stats:\n";
|
||||
std::cout<<"Min: "<<*std::min_element(trackNN->post_stats.begin(), trackNN->post_stats.end())<<" ms\n";
|
||||
std::cout<<"Max: "<<*std::max_element(trackNN->post_stats.begin(), trackNN->post_stats.end())<<" ms\n";
|
||||
for(int i=0; i<trackNN->post_stats.size(); i++) mean += trackNN->post_stats[i]; mean /= trackNN->post_stats.size();
|
||||
std::cout<<"Avg: "<<mean<<" ms\n"<<COL_END;
|
||||
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,264 @@
|
||||
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h> /* srand, rand */
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <mutex>
|
||||
#include "utils.h"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/videoio.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
#include "Yolo3Detection.h"
|
||||
#include "CenternetDetection.h"
|
||||
#include "MobilenetDetection.h"
|
||||
|
||||
#include "evaluation.h"
|
||||
|
||||
#include <map>
|
||||
|
||||
void convertFilename(std::string &filename,const std::string l_folder, const std::string i_folder, const std::string l_ext,const std::string i_ext)
|
||||
{
|
||||
filename.replace(filename.find(l_folder),l_folder.length(),i_folder);
|
||||
filename.replace(filename.find(l_ext),l_ext.length(),i_ext);
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
char ntype = 'y';
|
||||
const char *config_filename = "../demo/config.yaml";
|
||||
const char * net = "yolo4tiny_fp32.rt";
|
||||
const char * labels_path = "../demo/COCO_val2017/all_labels.txt";
|
||||
int n_batches = 1;
|
||||
float confidence_thresh = 0.3;
|
||||
bool show = false;
|
||||
bool write_dets = false;
|
||||
bool write_res_on_file = true;
|
||||
bool write_coco_json = false;
|
||||
int n_images = 5000;
|
||||
|
||||
bool verbose;
|
||||
int classes, map_points, map_levels;
|
||||
float map_step, IoU_thresh, conf_thresh;
|
||||
double vm_total = 0, rss_total = 0;
|
||||
double vm, rss;
|
||||
|
||||
//read args
|
||||
if(argc > 1)
|
||||
net = argv[1];
|
||||
if(argc > 2)
|
||||
ntype = argv[2][0];
|
||||
if(argc > 3)
|
||||
labels_path = argv[3];
|
||||
if(argc > 4)
|
||||
config_filename = argv[4];
|
||||
if(argc > 5)
|
||||
n_batches = atoi(argv[5]);
|
||||
if(argc > 6)
|
||||
confidence_thresh = atof(argv[6]);
|
||||
|
||||
std::cout<<"conf t: "<<confidence_thresh<<std::endl;
|
||||
|
||||
//check if files needed exist
|
||||
if(!fileExist(config_filename))
|
||||
FatalError("Wrong config file path.");
|
||||
if(!fileExist(net))
|
||||
FatalError("Wrong net file path.");
|
||||
if(!fileExist(labels_path))
|
||||
FatalError("Wrong labels file path.");
|
||||
|
||||
//read mAP parameters
|
||||
tk::dnn::readmAPParams( config_filename, classes, map_points, map_levels, map_step,
|
||||
IoU_thresh, conf_thresh, verbose);
|
||||
|
||||
//extract network name from rt path
|
||||
std::string net_name;
|
||||
removePathAndExtension(net, net_name);
|
||||
std::cout<<"Network: "<<net_name<<std::endl;
|
||||
|
||||
//open files (if needed)
|
||||
std::ofstream times, memory, coco_json;
|
||||
|
||||
if(write_coco_json){
|
||||
coco_json.open(net_name+"_COCO_res.json");
|
||||
coco_json << "[\n";
|
||||
}
|
||||
|
||||
if(write_res_on_file){
|
||||
times.open("times_"+net_name+"_"+ std::to_string(n_batches)+"_"+std::to_string(confidence_thresh)+".csv");
|
||||
memory.open("memory.csv", std::ios_base::app);
|
||||
memory<<net_name+"_"+ std::to_string(n_batches)+"_"+std::to_string(confidence_thresh)<<";";
|
||||
}
|
||||
|
||||
// instantiate detector
|
||||
tk::dnn::Yolo3Detection yolo;
|
||||
tk::dnn::CenternetDetection cnet;
|
||||
tk::dnn::MobilenetDetection mbnet;
|
||||
tk::dnn::DetectionNN *detNN;
|
||||
int n_classes = classes;
|
||||
switch(ntype){
|
||||
case 'y':
|
||||
detNN = &yolo;
|
||||
break;
|
||||
case 'c':
|
||||
detNN = &cnet;
|
||||
break;
|
||||
case 'm':
|
||||
detNN = &mbnet;
|
||||
n_classes++;
|
||||
break;
|
||||
default:
|
||||
FatalError("Network type not allowed (3rd parameter)\n");
|
||||
}
|
||||
detNN->init(net,n_classes, 1, conf_thresh);
|
||||
|
||||
//read images
|
||||
std::ifstream all_labels(labels_path);
|
||||
std::string l_filename;
|
||||
std::vector<tk::dnn::Frame> images;
|
||||
std::vector<tk::dnn::box> detected_bbox;
|
||||
|
||||
std::cout<<"Reading groundtruth and generating detections"<<std::endl;
|
||||
|
||||
if(show)
|
||||
cv::namedWindow("detection", cv::WINDOW_NORMAL);
|
||||
|
||||
bool file_ok = false;
|
||||
|
||||
int images_done;
|
||||
for (images_done=0 ; images_done < n_images ;) {
|
||||
|
||||
|
||||
int cur_batches = 0;
|
||||
std::vector<cv::Mat> batch_frames;
|
||||
std::vector<cv::Mat> batch_dnn_input;
|
||||
|
||||
std::vector<tk::dnn::Frame> cur_frames;
|
||||
for(;cur_batches<n_batches && images_done < n_images;cur_batches++, ++images_done){
|
||||
|
||||
std::getline(all_labels, l_filename);
|
||||
file_ok = all_labels ? true : false ;
|
||||
if (!file_ok)
|
||||
break;
|
||||
|
||||
tk::dnn::Frame f;
|
||||
f.lFilename = l_filename;
|
||||
f.iFilename = l_filename;
|
||||
convertFilename(f.iFilename, "labels", "images", ".txt", ".jpg");
|
||||
|
||||
// read frame
|
||||
if(!fileExist(f.iFilename.c_str()))
|
||||
FatalError("Wrong image file path.");
|
||||
|
||||
cv::Mat frame = cv::imread(f.iFilename.c_str(), cv::IMREAD_COLOR);
|
||||
batch_frames.push_back(frame);
|
||||
f.height = frame.rows;
|
||||
f.width = frame.cols;
|
||||
|
||||
if(!frame.data)
|
||||
break;
|
||||
batch_dnn_input.push_back(frame.clone());
|
||||
|
||||
// read and save groundtruth labels
|
||||
if(fileExist(f.lFilename.c_str()))
|
||||
{
|
||||
std::ifstream labels(f.lFilename);
|
||||
for(std::string line; std::getline(labels, line); ){
|
||||
std::istringstream in(line);
|
||||
tk::dnn::BoundingBox b;
|
||||
in >> b.cl >> b.x >> b.y >> b.w >> b.h;
|
||||
b.prob = 1;
|
||||
b.truthFlag = 1;
|
||||
f.gt.push_back(b);
|
||||
|
||||
if(show)// draw rectangle for groundtruth
|
||||
cv::rectangle(batch_frames[cur_batches], cv::Point((b.x-b.w/2)*f.width, (b.y-b.h/2)*f.height), cv::Point((b.x+b.w/2)*f.width,(b.y+b.h/2)*f.height), cv::Scalar(0, 255, 0), 2);
|
||||
}
|
||||
}
|
||||
|
||||
cur_frames.push_back(f);
|
||||
}
|
||||
if (!file_ok)
|
||||
break;
|
||||
|
||||
//inference
|
||||
detNN->update(batch_dnn_input,cur_batches,write_res_on_file, ×, write_coco_json);
|
||||
detNN->draw(batch_frames);
|
||||
|
||||
for(int j=0;j<cur_frames.size(); ++j){
|
||||
if(write_coco_json)
|
||||
printJsonCOCOFormat(&coco_json, cur_frames[j].iFilename.c_str(), detNN->batchDetected[j], classes, cur_frames[j].width, cur_frames[j].height);
|
||||
|
||||
std::ofstream myfile;
|
||||
if(write_dets)
|
||||
myfile.open ("det/"+cur_frames[j].lFilename.substr(cur_frames[j].lFilename.find("labels/") + 7));
|
||||
|
||||
// save detections labels
|
||||
for(auto d:detNN->batchDetected[j]){
|
||||
//convert detected bb in the same format as label
|
||||
//<x_center>/<image_width> <y_center>/<image_width> <width>/<image_width> <height>/<image_width>
|
||||
tk::dnn::BoundingBox b;
|
||||
b.x = (d.x + d.w/2) / cur_frames[j].width;
|
||||
b.y = (d.y + d.h/2) / cur_frames[j].height;
|
||||
b.w = d.w / cur_frames[j].width;
|
||||
b.h = d.h / cur_frames[j].height;
|
||||
b.prob = d.prob;
|
||||
b.cl = d.cl;
|
||||
cur_frames[j].det.push_back(b);
|
||||
|
||||
if(write_dets)
|
||||
myfile << d.cl << " "<< d.prob << " "<< b.x << " "<< b.y << " "<< b.w << " "<< b.h <<"\n";
|
||||
|
||||
if(show)// draw rectangle for detection
|
||||
cv::rectangle(batch_frames[j], cv::Point(d.x, d.y), cv::Point(d.x + d.w, d.y + d.h), cv::Scalar(0, 0, 255), 2);
|
||||
}
|
||||
|
||||
if(write_dets)
|
||||
myfile.close();
|
||||
|
||||
images.push_back(cur_frames[j]);
|
||||
|
||||
if(show){
|
||||
cv::imshow("detection", batch_frames[j]);
|
||||
cv::waitKey(0);
|
||||
}
|
||||
|
||||
}
|
||||
std::cout <<COL_ORANGEB<< "Images done:\t" << images_done<< "\tcur batch:\t"<<cur_batches<< "\n"<<COL_END;
|
||||
|
||||
getMemUsage(vm, rss);
|
||||
vm_total += vm;
|
||||
rss_total += rss;
|
||||
|
||||
|
||||
}
|
||||
|
||||
if(write_coco_json){
|
||||
coco_json.seekp (coco_json.tellp() - std::streampos(2));
|
||||
coco_json << "\n]\n";
|
||||
coco_json.close();
|
||||
}
|
||||
|
||||
std::cout << "Avg VM[MB]: " << vm_total/images_done/1024.0 << ";Avg RSS[MB]: " << rss_total/images_done/1024.0 << std::endl;
|
||||
|
||||
//compute mAP
|
||||
double AP = tk::dnn::computeMapNIoULevels(images,classes,IoU_thresh,confidence_thresh, map_points, map_step, map_levels, verbose, write_res_on_file, net_name+"_"+ std::to_string(n_batches)+"_"+std::to_string(confidence_thresh));
|
||||
std::cout<<"mAP "<<IoU_thresh<<":"<<IoU_thresh+map_step*(map_levels-1)<<" = "<<AP<<std::endl;
|
||||
|
||||
//compute average precision, recall and f1score
|
||||
tk::dnn::computeTPFPFN(images,classes,IoU_thresh,confidence_thresh, verbose, write_res_on_file, net_name +"_"+ std::to_string(n_batches)+"_"+std::to_string(confidence_thresh));
|
||||
|
||||
if(write_res_on_file){
|
||||
memory<<vm_total/images_done/1024.0<<";"<<rss_total/images_done/1024.0<<"\n";
|
||||
times.close();
|
||||
memory.close();
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,150 @@
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h> /* srand, rand */
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
#include <mutex>
|
||||
|
||||
#include "SegmentationNN.h"
|
||||
|
||||
bool gRun;
|
||||
bool SAVE_RESULT = true;
|
||||
|
||||
void sig_handler(int signo) {
|
||||
std::cout<<"request gateway stop\n";
|
||||
gRun = false;
|
||||
}
|
||||
|
||||
void writePred(const std::string& images_names, const std::string& gt_folder, const std::string& out_folder, tk::dnn::SegmentationNN& segNN, int& width, int& height, bool show=false){
|
||||
std::ifstream all_gt(images_names);
|
||||
std::string filename;
|
||||
cv::Mat frame;
|
||||
for (; std::getline(all_gt, filename); ) {
|
||||
std::cout<<filename<<std::endl;
|
||||
frame = cv::imread(gt_folder + filename);
|
||||
height = frame.rows;
|
||||
width = frame.cols;
|
||||
segNN.updateOriginal(frame, false);
|
||||
if(show)
|
||||
segNN.draw();
|
||||
cv::imwrite(out_folder + filename, segNN.segmented[0]);
|
||||
}
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
|
||||
std::cout<<"detection\n";
|
||||
signal(SIGINT, sig_handler);
|
||||
|
||||
|
||||
std::string net = "shelfnet_fp32.rt";
|
||||
if(argc > 1)
|
||||
net = argv[1];
|
||||
std::string input = "../demo/yolo_test.mp4";
|
||||
if(argc > 2)
|
||||
input = argv[2];
|
||||
int n_batch = 1;
|
||||
if(argc > 3)
|
||||
n_batch = atoi(argv[3]);
|
||||
int n_classes = 19;
|
||||
if(argc > 4)
|
||||
n_classes = atoi(argv[4]);
|
||||
bool resize = false;
|
||||
if(argc > 5)
|
||||
resize = atoi(argv[5]);
|
||||
int baseline_resize = 1024;
|
||||
if(argc > 6)
|
||||
baseline_resize = atoi(argv[6]);
|
||||
bool show = true;
|
||||
if(argc > 7)
|
||||
show = atoi(argv[7]);
|
||||
bool write_pred = false;
|
||||
if(argc > 8)
|
||||
write_pred = atoi(argv[8]);
|
||||
|
||||
if(resize && (baseline_resize < 0 || baseline_resize > 5000))
|
||||
FatalError("Problem with baseline resize")
|
||||
if(n_batch < 1 || n_batch > 64)
|
||||
FatalError("Batch dim not supported");
|
||||
|
||||
//net initialization
|
||||
tk::dnn::SegmentationNN segNN;
|
||||
segNN.init(net, n_classes, n_batch);
|
||||
|
||||
int height = 0, width = 0;
|
||||
int basewidth=baseline_resize, hsize;
|
||||
|
||||
if(write_pred){
|
||||
std::string gt_folder = "../demo/CityScapes_val/images/";
|
||||
std::string images_names = "../demo/CityScapes_val/all_images.txt";
|
||||
std::string out_folder = "seg/";
|
||||
|
||||
writePred(images_names, gt_folder, out_folder, segNN, width, height, show);
|
||||
}
|
||||
else{
|
||||
if(!show)
|
||||
SAVE_RESULT = true;
|
||||
|
||||
gRun = true;
|
||||
|
||||
cv::VideoCapture cap(input);
|
||||
if(!cap.isOpened())
|
||||
gRun = false;
|
||||
else
|
||||
std::cout<<"camera started\n";
|
||||
|
||||
cv::VideoWriter resultVideo;
|
||||
if(SAVE_RESULT) {
|
||||
int w,h;
|
||||
if(resize){
|
||||
w = basewidth;
|
||||
h = int((float(cap.get(cv::CAP_PROP_FRAME_HEIGHT))*float(basewidth/float(cap.get(cv::CAP_PROP_FRAME_WIDTH)))));
|
||||
}
|
||||
else{
|
||||
w = cap.get(cv::CAP_PROP_FRAME_WIDTH);
|
||||
h = cap.get(cv::CAP_PROP_FRAME_HEIGHT);
|
||||
}
|
||||
resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h));
|
||||
}
|
||||
|
||||
cv::Mat frame;
|
||||
while(gRun) {
|
||||
cap >> frame;
|
||||
if(!frame.data)
|
||||
break;
|
||||
|
||||
if(resize){
|
||||
hsize = int((float(frame.rows)*float(basewidth/float(frame.cols))));
|
||||
cv::resize(frame, frame, cv::Size(basewidth, hsize));
|
||||
}
|
||||
|
||||
height = frame.rows;
|
||||
width = frame.cols;
|
||||
|
||||
//inference
|
||||
segNN.updateOriginal(frame, true);
|
||||
if(show)
|
||||
segNN.draw();
|
||||
|
||||
if(SAVE_RESULT)
|
||||
resultVideo << segNN.segmented[0];
|
||||
}
|
||||
}
|
||||
|
||||
std::cout<<"segmentation end\n";
|
||||
double mean = 0, mean_pre = 0, mean_post = 0;
|
||||
|
||||
std::cout<<COL_GREENB<<"\n\nTime stats for size ["<<width<<","<<height<<"] :\n";
|
||||
|
||||
for(int i=0; i<segNN.stats.size(); i++) mean += segNN.stats[i]; mean /= segNN.stats.size();
|
||||
for(int i=0; i<segNN.stats_pre.size(); i++) mean_pre += segNN.stats_pre[i]; mean_pre /= segNN.stats_pre.size();
|
||||
for(int i=0; i<segNN.stats_post.size(); i++) mean_post += segNN.stats_post[i]; mean_post /= segNN.stats_post.size();
|
||||
std::cout<<"Avg pre:\t"<<mean_pre<<" ms\t"<<1000/(mean_pre)<<" FPS\n";
|
||||
std::cout<<"Avg inf:\t"<<mean<<" ms\t"<<1000/(mean)<<" FPS\n";
|
||||
std::cout<<"Avg post:\t"<<mean_post<<" ms\t"<<1000/(mean_post)<<" FPS\n\n";
|
||||
std::cout<<"Avg tot:\t"<<(mean_pre + mean_post + mean) <<" ms\t"<<1000/((mean_pre + mean_post + mean))<<" FPS\n"<<COL_END;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
# video input
|
||||
input : "../demo/yolo_test.mp4"
|
||||
win_input : "..\\..\\..\\demo\\yolo_test.mp4"
|
||||
|
||||
# network config
|
||||
net : "yolo4_berkeley_fp32.rt"
|
||||
ntype : 'y'
|
||||
n_classes : 80
|
||||
n_batch : 1
|
||||
conf_thresh : 0.3
|
||||
|
||||
# demo config
|
||||
show : true
|
||||
save : false
|
||||
@@ -0,0 +1,7 @@
|
||||
FROM ceccocats/tkdnn:latest
|
||||
LABEL maintainer "Francesco Gatti"
|
||||
|
||||
RUN cd && git clone https://github.com/ceccocats/tkDNN.git && cd tkDNN && mkdir build && cd build \
|
||||
&& cmake .. && make -j12
|
||||
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
FROM nvidia/cudagl:11.3.1-devel-ubuntu20.04
|
||||
|
||||
LABEL maintainer "TKDNN AUTHORS"
|
||||
LABEL Description="tkDNN+cudagl"
|
||||
LABEL com.tkdnn.nvidia.version="11.3.1"
|
||||
|
||||
ENV DEBIAN_FRONTEND noninteractive
|
||||
ENV CC gcc
|
||||
ENV CXX g++
|
||||
|
||||
RUN apt-get update && apt-get install -y \
|
||||
libblkid-dev && apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN apt-get update && apt-get install -y \
|
||||
libcudnn8-dev=8.2.1.32-1+cuda11.3 \
|
||||
libcudnn8=8.2.1.32-1+cuda11.3 \
|
||||
libnvinfer-dev=8.0.3-1+cuda11.3 \
|
||||
libnvinfer8=8.0.3-1+cuda11.3 && apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libblkid-dev \
|
||||
locales \
|
||||
lsb-release \
|
||||
mesa-utils \
|
||||
git \
|
||||
nano \
|
||||
terminator \
|
||||
wget \
|
||||
curl \
|
||||
libssl-dev \
|
||||
htop \
|
||||
dbus-x11 \
|
||||
libqt5opengl5-dev \
|
||||
libgtk-3-dev \
|
||||
libvtk7-dev \
|
||||
libv4l-dev \
|
||||
tar \
|
||||
libgoogle-glog-dev \
|
||||
libgflags-dev \
|
||||
gfortran-9 \
|
||||
libtbb-dev \
|
||||
libgstreamer1.0-dev \
|
||||
libgstreamer-plugins-base1.0-dev \
|
||||
libdc1394-22-dev \
|
||||
libavresample-dev \
|
||||
libatlas-cpp-0.6-dev \
|
||||
python3-dev \
|
||||
gdb \
|
||||
python3-pip \
|
||||
unzip libtbb-dev && \
|
||||
apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
software-properties-common && apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN apt-add-repository universe
|
||||
RUN apt-get update && apt-get install -y python3-pip python3 openssh-server ssh pyqt5-dev sip-dev && apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||
RUN pip3 install --upgrade pip
|
||||
RUN pip3 install --upgrade virtualenv
|
||||
RUN pip3 install --upgrade paramiko
|
||||
RUN pip3 install --ignore-installed --upgrade numpy protobuf
|
||||
|
||||
|
||||
RUN cd ~ && mkdir build
|
||||
RUN cd ~/build && wget https://github.com/Kitware/CMake/releases/download/v3.21.4/cmake-3.21.4.tar.gz && \
|
||||
tar -xvf cmake-3.21.4.tar.gz && cd cmake-3.21.4 && ./configure --prefix=/usr/local --qt-gui --parallel=12 && \
|
||||
make -j8 && make install
|
||||
|
||||
RUN apt-get update && apt-get install -y automake autoconf pkg-config libevent-dev libncurses5-dev bison && \
|
||||
apt-get clean && rm -rf /var/lib/apt/lists/
|
||||
|
||||
RUN git clone https://github.com/tmux/tmux.git && \
|
||||
cd tmux && git checkout tags/3.2 && ls -la && sh autogen.sh && ./configure && make -j8 && make install
|
||||
|
||||
RUN apt-get update && apt-get install -y zsh && apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||
RUN wget https://github.com/robbyrussell/oh-my-zsh/raw/master/tools/install.sh -O - | zsh || true
|
||||
RUN chsh -s /usr/bin/zsh root
|
||||
RUN git clone https://github.com/sindresorhus/pure /root/.oh-my-zsh/custom/pure
|
||||
RUN ln -s /root/.oh-my-zsh/custom/pure/pure.zsh-theme /root/.oh-my-zsh/custom/
|
||||
RUN ln -s /root/.oh-my-zsh/custom/pure/async.zsh /root/.oh-my-zsh/custom/
|
||||
RUN sed -i -e 's/robbyrussell/refined/g' /root/.zshrc
|
||||
RUN sed -i '/plugins=(/c\plugins=(git git-flow adb pyenv tmux)' /root/.zshrc
|
||||
|
||||
RUN mkdir -p /root/.config/terminator/
|
||||
COPY assets/terminator_config /root/.config/terminator/config
|
||||
|
||||
RUN echo "/usr/local/nvidia/lib" >> /etc/ld.so.conf.d/nvidia.conf && \
|
||||
echo "/usr/local/nvidia/lib64" >> /etc/ld.so.conf.d/nvidia.conf && \
|
||||
echo "/usr/local/cuda/lib64" >> /etc/ld.so.conf.d/nvidia.conf
|
||||
|
||||
|
||||
ENV PATH /usr/local/nvidia/bin:/usr/local/cuda/bin:${PATH}
|
||||
ENV LD_LIBRARY_PATH /usr/local/nvidia/lib:/usr/local/nvidia/lib64:/usr/local/cuda/lib64:/usr/lib:/usr/lib/x86_64-linux-gnu:/usr/local/lib:${LD_LIBRARY_PATH}
|
||||
ENV NVIDIA_VISIBLE_DEVICES all
|
||||
ENV NVIDIA_DRIVER_CAPABILITIES compute,utility,graphics
|
||||
|
||||
|
||||
|
||||
RUN cd ~/build && wget https://github.com/opencv/opencv/archive/4.5.4.tar.gz && tar -xf 4.5.4.tar.gz && rm 4.5.4.tar.gz
|
||||
RUN cd ~/build && wget https://github.com/opencv/opencv_contrib/archive/4.5.4.tar.gz && tar -xf 4.5.4.tar.gz && rm 4.5.4.tar.gz
|
||||
RUN cd ~/build && \
|
||||
cd opencv-4.5.4 && mkdir build && cd build && \
|
||||
cmake -D CMAKE_BUILD_TYPE=RELEASE \
|
||||
-D CMAKE_INSTALL_PREFIX=/usr/local \
|
||||
-D INSTALL_PYTHON_EXAMPLES=OFF \
|
||||
-D INSTALL_C_EXAMPLES=OFF \
|
||||
-D OPENCV_EXTRA_MODULES_PATH='~/build/opencv_contrib-4.5.4/modules' \
|
||||
-D BUILD_EXAMPLES=OFF \
|
||||
-D BUILD_TESTS=OFF \
|
||||
-D BUILD_PERF_TESTS=OFF \
|
||||
-D BUILD_DOCS=OFF \
|
||||
-D WITH_CUDA=ON \
|
||||
-D WITH_OPENGL=ON \
|
||||
-D WITH_NVCUVID=ON \
|
||||
-D CUDA_ARCH_BIN=7.2 \
|
||||
-D CUDA_ARCH_PTX=7.2 \
|
||||
-D ENABLE_FAST_MATH=ON \
|
||||
-D CUDA_FAST_MATH=ON \
|
||||
-D WITH_CUBLAS=ON \
|
||||
-D WITH_CUDNN=ON \
|
||||
-D WITH_OPENMP=ON \
|
||||
-D WITH_NONFREE=ON \
|
||||
-D WITH_LIBV4L=ON \
|
||||
-D WITH_GSTREAMER=ON \
|
||||
-D WITH_GSTREAMER_0_10=OFF \
|
||||
-D WITH_TBB=ON \
|
||||
../ && make -j12 && make install && ldconfig
|
||||
|
||||
RUN cd ~ && rm -rf build
|
||||
|
||||
RUN cd ~ && mkdir Development && cd Development && \
|
||||
git clone https://github.com/ceccocats/tkDNN.git && cd tkDNN && \
|
||||
mkdir build && cd build && \
|
||||
cmake -DCMAKE_BUILD_TYPE=Release .. && \
|
||||
make -j6
|
||||
|
||||
RUN apt-get clean && rm -rf /var/lib/apt/lists/*
|
||||
COPY assets/entrypoint_setup.sh /
|
||||
ENTRYPOINT ["/entrypoint_setup.sh"]
|
||||
CMD ["terminator"]
|
||||
@@ -0,0 +1,18 @@
|
||||
# Use the prebuilt image
|
||||
```
|
||||
# build image
|
||||
docker build -t tkdnn:build -f Dockerfile .
|
||||
```
|
||||
|
||||
# Build Base Docker image
|
||||
```
|
||||
# make nvidia docker working
|
||||
# follow this guide: https://github.com/NVIDIA/nvidia-docker
|
||||
|
||||
# build image
|
||||
docker build -t ceccocats/tkdnn:latest -f Dockerfile.base .
|
||||
|
||||
# run image
|
||||
./docker_launch.sh
|
||||
```
|
||||
|
||||
Executable
+123
@@ -0,0 +1,123 @@
|
||||
#! /bin/bash
|
||||
|
||||
CMD=
|
||||
|
||||
# Functions
|
||||
# TOOD: Check if we can use: getent passwd $USER to extract all variables
|
||||
# TODO: Check for valid inputs, cause now it will go through even with bad inputs
|
||||
check_envs () {
|
||||
DOCKER_CUSTOM_USER_OK=true;
|
||||
if [ -z ${DOCKER_USER_NAME+x} ]; then
|
||||
DOCKER_CUSTOM_USER_OK=false;
|
||||
return;
|
||||
fi
|
||||
|
||||
if [ -z ${DOCKER_USER_ID+x} ]; then
|
||||
DOCKER_CUSTOM_USER_OK=false;
|
||||
return;
|
||||
else
|
||||
if ! [ -z "${DOCKER_USER_ID##[0-9]*}" ]; then
|
||||
echo -e "\033[1;33mWarning: User-ID should be a number. Falling back to defaults.\033[0m"
|
||||
DOCKER_CUSTOM_USER_OK=false;
|
||||
return;
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ -z ${DOCKER_USER_GROUP_NAME+x} ]; then
|
||||
DOCKER_CUSTOM_USER_OK=false;
|
||||
return;
|
||||
fi
|
||||
|
||||
if [ -z ${DOCKER_USER_GROUP_ID+x} ]; then
|
||||
DOCKER_CUSTOM_USER_OK=false;
|
||||
return;
|
||||
else
|
||||
if ! [ -z "${DOCKER_USER_GROUP_ID##[0-9]*}" ]; then
|
||||
echo -e "\033[1;33mWarning: Group-ID should be a number. Falling back to defaults.\033[0m"
|
||||
DOCKER_CUSTOM_USER_OK=false;
|
||||
return;
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
setup_env_user () {
|
||||
USER=$1
|
||||
USER_ID=$2
|
||||
GROUP=$3
|
||||
GROUP_ID=$4
|
||||
|
||||
## Create user
|
||||
useradd -m $USER
|
||||
|
||||
## Copy zsh/sh configs
|
||||
cp /root/.profile /home/$USER/
|
||||
cp /root/.bashrc /home/$USER/
|
||||
cp /root/.zshrc /home/$USER/
|
||||
## Copy terminator configs
|
||||
mkdir -p /home/$USER/.config/terminator
|
||||
cp /root/.config/terminator/config /home/$USER/.config/terminator/config
|
||||
cp /root/.config/terminator/background.png /home/$USER/.config/terminator/background.png
|
||||
cp -rf /root/.oh-my-zsh /home/$USER/
|
||||
cp -rf /root/tkDNN /home/$USER/
|
||||
rm -rf /home/$USER/.oh-my-zsh/custom/pure.zsh-theme /home/$USER/.oh-my-zsh/custom/async.zsh
|
||||
ln -s /home/$USER/.oh-my-zsh/custom/pure/pure.zsh-theme /home/$USER/.oh-my-zsh/custom/
|
||||
ln -s /home/$USER/.oh-my-zsh/custom/pure/async.zsh /home/$USER/.oh-my-zsh/custom/
|
||||
sed -i -e 's@ZSH=\"/root@ZSH=\"/home/$USER@g' /home/$USER/.zshrc
|
||||
# Copy SSH keys & fix owner
|
||||
if [ -d "/root/.ssh" ]; then
|
||||
cp -rf /root/.ssh /home/$USER/
|
||||
chown -R $USER:$GROUP /home/$USER/.ssh
|
||||
fi
|
||||
|
||||
## Fix owner
|
||||
chown $USER:$GROUP /home/$USER
|
||||
chown -R $USER:$GROUP /home/$USER/.config
|
||||
chown $USER:$GROUP /home/$USER/.profile
|
||||
chown $USER:$GROUP /home/$USER/.bashrc
|
||||
chown $USER:$GROUP /home/$USER/.zshrc
|
||||
chown -R $USER:$GROUP /home/$USER/.oh-my-zsh
|
||||
chown -R $USER:$GROUP /home/$USER/tkDNN
|
||||
|
||||
## This a trick to keep the evnironmental variables of root which is important!
|
||||
echo "if ! [ \"$DOCKER_USER_NAME\" = \"$(id -un)\" ]; then" >> /root/.bashrc
|
||||
echo " cd /home/$DOCKER_USER_NAME" >> /root/.bashrc
|
||||
echo " su $DOCKER_USER_NAME" >> /root/.bashrc
|
||||
echo "fi" >> /root/.bashrc
|
||||
|
||||
echo "if ! [ \"$DOCKER_USER_NAME\" = \"$(id -un)\" ]; then" >> /root/.zshrc
|
||||
echo " cd /home/$DOCKER_USER_NAME" >> /root/.zshrc
|
||||
echo " su $DOCKER_USER_NAME" >> /root/.zshrc
|
||||
echo "fi" >> /root/.zshrc
|
||||
|
||||
## Setup Password-file
|
||||
PASSWDCONTENTS=$(grep -v "^${USER}:" /etc/passwd)
|
||||
GROUPCONTENTS=$(grep -v -e "^${GROUP}:" -e "^docker:" /etc/group)
|
||||
|
||||
(echo "${PASSWDCONTENTS}" && echo "${USER}:x:$USER_ID:$GROUP_ID::/home/$USER:/bin/bash") > /etc/passwd
|
||||
(echo "${GROUPCONTENTS}" && echo "${GROUP}:x:${GROUP_ID}:") > /etc/group
|
||||
(if test -f /etc/sudoers ; then echo "${USER} ALL=(ALL) NOPASSWD: ALL" >> /etc/sudoers ; fi)
|
||||
}
|
||||
|
||||
|
||||
# ---Main---
|
||||
|
||||
# Create new user
|
||||
## Check Inputs
|
||||
check_envs
|
||||
|
||||
## Determine user & Setup Environment
|
||||
if [ $DOCKER_CUSTOM_USER_OK == true ]; then
|
||||
echo " -->DOCKER_USER Input is set to '$DOCKER_USER_NAME:$DOCKER_USER_ID:$DOCKER_USER_GROUP_NAME:$DOCKER_USER_GROUP_ID'";
|
||||
echo -e "\033[0;32mSetting up environment for user=$DOCKER_USER_NAME\033[0m"
|
||||
setup_env_user $DOCKER_USER_NAME $DOCKER_USER_ID $DOCKER_USER_GROUP_NAME $DOCKER_USER_GROUP_ID
|
||||
else
|
||||
echo " -->DOCKER_USER* variables not set. Using 'root'.";
|
||||
echo -e "\033[0;32mSetting up environment for user=root\033[0m"
|
||||
DOCKER_USER_NAME="root"
|
||||
fi
|
||||
|
||||
# Change shell to zsh
|
||||
chsh -s /usr/bin/zsh $DOCKER_USER_NAME
|
||||
|
||||
# Run CMD from Docker
|
||||
"$@"
|
||||
@@ -0,0 +1,18 @@
|
||||
[global_config]
|
||||
title_transmit_bg_color = "#2e3436"
|
||||
[keybindings]
|
||||
[layouts]
|
||||
[[default]]
|
||||
[[[child1]]]
|
||||
parent = window0
|
||||
type = Terminal
|
||||
[[[window0]]]
|
||||
parent = ""
|
||||
type = Window
|
||||
[plugins]
|
||||
[profiles]
|
||||
[[default]]
|
||||
background_color = "#282828"
|
||||
cursor_color = "#aaaaaa"
|
||||
foreground_color = "#f3f3f3"
|
||||
palette = "#000000:#aa0000:#00aa00:#c4a000:#3465a4:#75507b:#06989a:#d3d7cf:#88807c:#f15d22:#73c48f:#ffce51:#48b9c7:#ad7fa8:#34e2e2:#eeeeec"
|
||||
Executable
+9
@@ -0,0 +1,9 @@
|
||||
xhost local:root
|
||||
docker run --rm -it --runtime=nvidia --privileged --net=host --cap-add sys_ptrace -d --ipc=host \
|
||||
-v /tmp/.X11-unix:/tmp/.X11-unix -e DISPLAY=$DISPLAY \
|
||||
-v $HOME/.Xauthority:/home/$(id -un)/.Xauthority -e XAUTHORITY=/home/$(id -un)/.Xauthority \
|
||||
-e DOCKER_USER_NAME=$(id -un) \
|
||||
-e DOCKER_USER_ID=$(id -u) \
|
||||
-e DOCKER_USER_GROUP_NAME=$(id -gn) \
|
||||
-e DOCKER_USER_GROUP_ID=$(id -g) \
|
||||
-v $HOME/.ssh:/home/$(id -un)/.ssh ceccocats/tkdnn
|
||||
@@ -0,0 +1,92 @@
|
||||
# 2D/3D Object Detection and Tracking
|
||||
|
||||
Currently tkDNN supports only CenterTrack as 3DOD & 2D/3D Tracker network.
|
||||
|
||||
## 3D Object Detection
|
||||
|
||||
To run the 3D object detection demo follow these steps (example with CenterNet based on DLA34):
|
||||
```
|
||||
rm dla34_cnet3d_fp32.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_dla34_cnet3d # run the yolo test (is slow)
|
||||
./demo3D dla34_cnet3d_fp32.rt ../demo/yolo_test.mp4 NULL c
|
||||
```
|
||||
The demo3D program takes the same parameters of the demo program:
|
||||
```
|
||||
./demo3D <network-rt-file> <path-to-video> <calibration-file> <kind-of-network> <number-of-classes> <n-batches> <show-flag> <conf-thresh>
|
||||
```
|
||||
where
|
||||
|
||||
* ```<calibration-file>``` is the camera calibration file (opencv format). It is important that the file contains entry "camera_matrix" with sub-entry "rows", "cols", "data". If you do not want to pass the calibration file, pass "NULL" instead.
|
||||
|
||||

|
||||
|
||||
## Object Detection and Tracking
|
||||
|
||||
To run the 3D object detection & tracking demo follow these steps (example with CenterTrack based on DLA34):
|
||||
```
|
||||
rm dla34_ctrack_fp32.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_dla34_ctrack # run the yolo test (is slow)
|
||||
./demoTracker dla34_ctrack_fp32.rt ../demo/yolo_test.mp4 NULL c
|
||||
```
|
||||
|
||||
The demoTracker program takes the same parameters of the demo program:
|
||||
```
|
||||
./demoTracker <network-rt-file> <path-to-video> <calibration-file> <kind-of-network> <number-of-classes> <n-batches> <show-flag> <conf-thresh> <2D/3D-flag>
|
||||
```
|
||||
|
||||
where
|
||||
|
||||
* ```<calibration-file>``` is the camera calibration file (opencv format). It is important that the file contains entry "camera_matrix" with sub-entry "rows", "cols", "data". If you do not want to pass the calibration file, pass "NULL" instead.
|
||||
* ```<2D/3D-flag>``` if set to 0 the demo will be in the 2D mode, while if set to 1 the demo will be in the 3D mode (Default is 1 - 3D mode).
|
||||
|
||||

|
||||
|
||||
## FPS Results
|
||||
|
||||
Inference FPS of shelfnet with tkDNN, average of 1200 images on:
|
||||
* RTX 2080Ti (CUDA 10.2, TensorRT 7.0.0, Cudnn 7.6.5);
|
||||
* Xavier AGX, Jetpack 4.3 (CUDA 10.0, CUDNN 7.6.3, tensorrt 6.0.1 );
|
||||
|
||||
### 3D OD and Tracking
|
||||
|
||||
| Platform | Test | Phase | FP32, ms | FP32, FPS | FP16, ms | FP16, FPS | INT8, ms | INT8, FPS |
|
||||
| :------: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: |
|
||||
| RTX 2080Ti | CenterTrack3D 512x512 (B=1) | pre | 4.43883 | 225.285 | 4.42951 | 225.759 | 4.44278 | 225.084 |
|
||||
| RTX 2080Ti | CenterTrack3D 512x512 (B=1) | inf | 9.03454 | 110.686 | 6.02013 | 166.109 | 5.31611 | 188.108 |
|
||||
| RTX 2080Ti | CenterTrack3D 512x512 (B=1) | post | 0.96631 | 1034.87 | 0.96824 | 1032.80 | 0.95066 | 1051.90 |
|
||||
| RTX 2080Ti | CenterTrack3D 512x512 (B=1) | tot | 14.4397 | 69.2535 | 11.4179 | 87.5818 | 10.7095 | 93.3750 |
|
||||
| RTX 2080Ti | CenterTrack3D 512x512 (B=4) | pre | 4.60075 | 217.356 | 4.28658 | 233.286 | 4.29473 | 232.844 |
|
||||
| RTX 2080Ti | CenterTrack3D 512x512 (B=4) | inf | 8.48365 | 117.874 | 5.25150 | 190.422 | 4.58463 | 218.120 |
|
||||
| RTX 2080Ti | CenterTrack3D 512x512 (B=4) | post | 0.99484 | 1005.19 | 0.91776 | 1089.61 | 0.89853 | 1112.93 |
|
||||
| RTX 2080Ti | CenterTrack3D 512x512 (B=4) | tot | 14.0792 | 71.0266 | 10.4558 | 95.6405 | 9.77788 | 102.272 |
|
||||
| AGX Xavier | CenterTrack3D 512x512 (B=1) | pre | 34.9915 | 28.5784 | 33.5976 | 29.7440 | 34.4425 | 29.0339 |
|
||||
| AGX Xavier | CenterTrack3D 512x512 (B=1) | inf | 76.3579 | 13.0962 | 52.4759 | 19.0564 | 51.4610 | 19.4322 |
|
||||
| AGX Xavier | CenterTrack3D 512x512 (B=1) | post | 3.38576 | 295.355 | 3.26010 | 306.739 | 3.19770 | 312.725 |
|
||||
| AGX Xavier | CenterTrack3D 512x512 (B=1) | tot | 114.735 | 8.71574 | 89.3336 | 11.1940 | 89.1012 | 11.2232 |
|
||||
| AGX Xavier | CenterTrack3D 512x512 (B=4) | pre | 32.8933 | 30.4014 | 32.7950 | 30.4925 | 32.9603 | 30.3396 |
|
||||
| AGX Xavier | CenterTrack3D 512x512 (B=4) | inf | 74.2840 | 13.4618 | 50.3858 | 19.8469 | 49.2030 | 20.3240 |
|
||||
| AGX Xavier | CenterTrack3D 512x512 (B=4) | post | 3.14888 | 317.574 | 3.13615 | 318.862 | 3.02550 | 330.524 |
|
||||
| AGX Xavier | CenterTrack3D 512x512 (B=4) | tot | 110.326 | 9.06404 | 86.3169 | 11.5852 | 85.1888 | 11.7386 |
|
||||
|
||||
|
||||
### 2D OD and Tracking
|
||||
|
||||
| Platform | Test | Phase | FP32, ms | FP32, FPS | FP16, ms | FP16, FPS | INT8, ms | INT8, FPS |
|
||||
| :------: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: |
|
||||
| RTX 2080Ti | CenterTrack2D 512x512 (B=1) | pre | 4.44386 | 225.030 | 4.43828 | 225.313 | 4.47747 | 223.340 |
|
||||
| RTX 2080Ti | CenterTrack2D 512x512 (B=1) | inf | 9.08365 | 110.088 | 6.04842 | 165.332 | 5.34787 | 186.990 |
|
||||
| RTX 2080Ti | CenterTrack2D 512x512 (B=1) | post | 0.98593 | 1014.27 | 0.97745 | 1023.07 | 0.96595 | 1035.25 |
|
||||
| RTX 2080Ti | CenterTrack2D 512x512 (B=1) | tot | 14.5134 | 68.9018 | 11.4642 | 87.2281 | 10.7913 | 92.6672 |
|
||||
| RTX 2080Ti | CenterTrack2D 512x512 (B=4) | pre | 4.41188 | 226.661 | 4.50800 | 221.828 | 4.29238 | 232.971 |
|
||||
| RTX 2080Ti | CenterTrack2D 512x512 (B=4) | inf | 8.29015 | 120.625 | 5.38630 | 185.656 | 4.58500 | 218.103 |
|
||||
| RTX 2080Ti | CenterTrack2D 512x512 (B=4) | post | 0.96847 | 1032.55 | 0.97997 | 1020.44 | 0.91791 | 1089.43 |
|
||||
| RTX 2080Ti | CenterTrack2D 512x512 (B=4) | tot | 13.6705 | 73.1502 | 10.8743 | 91.9602 | 9.79528 | 102.090 |
|
||||
| AGX Xavier | CenterTrack2D 512x512 (B=1) | pre | 33.4745 | 29.8735 | 33.4847 | 29.8643 | 33.5022 | 29.8488 |
|
||||
| AGX Xavier | CenterTrack2D 512x512 (B=1) | inf | 76.2077 | 13.1220 | 52.5111 | 19.0436 | 51.6057 | 19.3777 |
|
||||
| AGX Xavier | CenterTrack2D 512x512 (B=1) | post | 3.26055 | 306.697 | 3.26806 | 305.992 | 3.21988 | 310.571 |
|
||||
| AGX Xavier | CenterTrack2D 512x512 (B=1) | tot | 111.943 | 8.93312 | 89.2639 | 11.2027 | 88.3278 | 11.3215 |
|
||||
| AGX Xavier | CenterTrack2D 512x512 (B=4) | pre | 32.8323 | 30.4579 | 32.8595 | 30.4326 | 32.8195 | 30.4697 |
|
||||
| AGX Xavier | CenterTrack2D 512x512 (B=4) | inf | 74.3075 | 13.4576 | 50.3555 | 19.8588 | 49.1805 | 20.3333 |
|
||||
| AGX Xavier | CenterTrack2D 512x512 (B=4) | post | 3.12360 | 320.143 | 3.13570 | 318.908 | 3.04943 | 327.931 |
|
||||
| AGX Xavier | CenterTrack2D 512x512 (B=4) | tot | 110.263 | 9.06920 | 86.3507 | 11.5807 | 85.0494 | 11.7579 |
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
# Monocular depth estimation with tkDNN
|
||||
|
||||
Currently tkDNN supports only Monodepth2 as monocular depth esitmation network.
|
||||
|
||||
|
||||
## Run the demo
|
||||
|
||||
To run the depth estimation demo follow these steps (example with monodepth2):
|
||||
```
|
||||
rm monodepth2_fp32.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_monodepth2 # run the yolo test (is slow)
|
||||
./demoDepth monodepth2_fp32.rt ../demo/yolo_test.mp4
|
||||
```
|
||||
In general the demo program takes the following parameters:
|
||||
```
|
||||
./demoDepth <network-rt-file> <path-to-video> <show-flag> <save-flag>
|
||||
```
|
||||
where
|
||||
* ```<network-rt-file>``` is the rt file generated by a test
|
||||
* ```<<path-to-video>``` is the path to a video file or a camera input
|
||||
* ```<show-flag>``` if set to 0 the demo will not show the visualization, it will otherwise (default=1)
|
||||
* ```<save-flag>``` if set to 1 the demo will save the video into result.mp4, it won't otherwise (default=1)
|
||||
|
||||
NB) By default it is used FP32 inference
|
||||
|
||||
|
||||

|
||||
|
||||
|
||||
<!-- ## FPS Results
|
||||
|
||||
Inference FPS of shelfnet with tkDNN, average of 1200 images on:
|
||||
* RTX 2080Ti (CUDA 10.2, TensorRT 7.0.0, Cudnn 7.6.5);
|
||||
* Xavier AGX, Jetpack 4.3 (CUDA 10.0, CUDNN 7.6.3, tensorrt 6.0.1 );
|
||||
|
||||
| Platform | Test | Phase | FP32, ms | FP32, FPS | FP16, ms | FP16, FPS | INT8, ms | INT8, FPS |
|
||||
| :------: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: |
|
||||
| RTX 2080Ti | shelfnet 1024x1024 (B=1) | pre | 6.11863 | 163.435 | 5.81465 | 171.979 | 5.88699 | 169.866 |
|
||||
| RTX 2080Ti | shelfnet 1024x1024 (B=1) | inf | 11.5464 | 86.6074 | 7.35396 | 135.981 | 6.37623 | 156.832 |
|
||||
| RTX 2080Ti | shelfnet 1024x1024 (B=1) | post | 4.09058 | 244.464 | 3.91961 | 255.128 | 4.07343 | 245.493 |
|
||||
| RTX 2080Ti | shelfnet 1024x1024 (B=1) | tot | 21.7556 | 45.9652 | 17.0882 | 58.5199 | 16.3366 | 61.2121 |
|
||||
| RTX 2080Ti | shelfnet 2048x2048 (B=4) | pre | 25.435 | 39.3158 | 25.2953 | 39.5331 | 25.9303 | 38.565 |
|
||||
| RTX 2080Ti | shelfnet 2048x2048 (B=4) | inf | 36.5015 | 27.3961 | 17.0534 | 58.6395 | 15.6061 | 64.0773 |
|
||||
| RTX 2080Ti | shelfnet 2048x2048 (B=4) | post | 17.3917 | 57.4985 | 17.1649 | 58.2583 | 17.5539 | 56.9675 |
|
||||
| RTX 2080Ti | shelfnet 2048x2048 (B=4) | tot | 79.3283 | 12.6058 | 59.5136 | 16.8029 | 59.0903 | 16.9233 |
|
||||
| AGX Xavier | shelfnet 1024x1024 (B=1) | pre | 8.0174 | 124.729 | 7.5117 | 133.126 | 7.47333 | 133.809 |
|
||||
| AGX Xavier | shelfnet 1024x1024 (B=1) | inf | 72.4173 | 13.8089 | 37.505 | 26.6631 | 31.3286 | 31.9197 |
|
||||
| AGX Xavier | shelfnet 1024x1024 (B=1) | post | 8.89958 | 112.365 | 8.83576 | 113.176 | 9.42655 | 106.083 |
|
||||
| AGX Xavier | shelfnet 1024x1024 (B=1) | tot | 89.3342 | 11.1939 | 53.8525 | 18.5692 | 48.2285 | 20.7346 |
|
||||
| AGX Xavier | shelfnet 2048x2048 (B=4) | pre | 47.1454 | 21.211 | 21.6475 | 46.1947 | 21.4201 | 46.6851 |
|
||||
| AGX Xavier | shelfnet 2048x2048 (B=4) | inf | 266.537 | 3.75183 | 128.321 | 7.79293 | 107.621 | 9.29185 |
|
||||
| AGX Xavier | shelfnet 2048x2048 (B=4) | post | 44.0711 | 22.6906 | 40.1732 | 24.8922 | 39.873 | 25.0796 |
|
||||
| AGX Xavier | shelfnet 2048x2048 (B=4) | tot | 357.753 | 2.79522 | 190.142 | 5.25922 | 168.914 | 5.92016 | -->
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
# Semantic Segmentation with tkDNN
|
||||
|
||||
Currently tkDNN supports only ShelfNet as semantic segmentation network.
|
||||
|
||||
|
||||
## Run the demo
|
||||
|
||||
To run the semantic segmentation demo follow these steps (example with shelfnet):
|
||||
```
|
||||
rm shelfnet_fp32.rt # be sure to delete(or move) old tensorRT files
|
||||
export TKDNN_BATCHSIZE=4 # be sure you have batch size > than 1 if you want to run inference on images bigger than 1024
|
||||
./test_shelfnet # run the yolo test (is slow)
|
||||
./demo shelfnet_fp32.rt ../demo/yolo_test.mp4 1 19
|
||||
```
|
||||
In general the demo program takes the following parameters:
|
||||
```
|
||||
./seg_demo <network-rt-file> <path-to-video> <n-batches> <number-of-classes> <resize-flag> <baseline-resize> <show-flag> <write-pred>
|
||||
```
|
||||
where
|
||||
* ```<network-rt-file>``` is the rt file generated by a test
|
||||
* ```<<path-to-video>``` is the path to a video file or a camera input
|
||||
* ```<n-batches>``` number of batches to use in inference (N.B. you should first export TKDNN_BATCHSIZE to the required n_batches and create again the rt file for the network).
|
||||
* ```<number-of-classes>```is the number of classes the network is trained on
|
||||
* ```<resize-flag>``` if set to 0 the demo will not resize the input frames, but use it as it is, otherwise it will resize it.
|
||||
* ```<baseline-resize>``` is ```<resize-flag>``` is set to 1, then the input frames will be proportionally resized using ```<baseline-resize>``` as width baseline.
|
||||
* ```<show-flag>``` if set to 0 the demo will not show the visualization but save the video into result.mp4 (if n-batches ==1)
|
||||
* ```<write-pred>``` if set to 0 (default) the demo will run, otherwise the evaluation of a dataset will run and the output of the segmentation will be saved. Attention: this is under development and paths are embedded, so change them in the code in advance.
|
||||
|
||||
NB) By default it is used FP32 inference
|
||||
NB) The batching is not used to work on more streams, rather to work on more tiles of the same image. Shelfnet never resized the input image, therefore for images greater than 1024x1024 tiles of 1024x1024 are given in input to the network in batch.
|
||||
|
||||

|
||||
|
||||
For other demo videos refer to [this playlist](https://www.youtube.com/playlist?list=PLv0nEQYDD45y5EdSiywwCGPBmJVUzIWwe).
|
||||
|
||||
NB) The gif and the videos are obtained with Mapillary Vistas weights, that we cannot publicly share due to its license restrictions. However, you can train Shelfnet using Mapillary and [this](https://git.hipert.unimore.it/mverucchi/shelfnet) fork of the original repo.
|
||||
|
||||
|
||||
## FPS Results
|
||||
|
||||
Inference FPS of shelfnet with tkDNN, average of 1200 images on:
|
||||
* RTX 2080Ti (CUDA 10.2, TensorRT 7.0.0, Cudnn 7.6.5);
|
||||
* Xavier AGX, Jetpack 4.3 (CUDA 10.0, CUDNN 7.6.3, tensorrt 6.0.1 );
|
||||
|
||||
| Platform | Test | Phase | FP32, ms | FP32, FPS | FP16, ms | FP16, FPS | INT8, ms | INT8, FPS |
|
||||
| :------: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: |
|
||||
| RTX 2080Ti | shelfnet 1024x1024 (B=1) | pre | 6.11863 | 163.435 | 5.81465 | 171.979 | 5.88699 | 169.866 |
|
||||
| RTX 2080Ti | shelfnet 1024x1024 (B=1) | inf | 11.5464 | 86.6074 | 7.35396 | 135.981 | 6.37623 | 156.832 |
|
||||
| RTX 2080Ti | shelfnet 1024x1024 (B=1) | post | 4.09058 | 244.464 | 3.91961 | 255.128 | 4.07343 | 245.493 |
|
||||
| RTX 2080Ti | shelfnet 1024x1024 (B=1) | tot | 21.7556 | 45.9652 | 17.0882 | 58.5199 | 16.3366 | 61.2121 |
|
||||
| RTX 2080Ti | shelfnet 2048x2048 (B=4) | pre | 25.435 | 39.3158 | 25.2953 | 39.5331 | 25.9303 | 38.565 |
|
||||
| RTX 2080Ti | shelfnet 2048x2048 (B=4) | inf | 36.5015 | 27.3961 | 17.0534 | 58.6395 | 15.6061 | 64.0773 |
|
||||
| RTX 2080Ti | shelfnet 2048x2048 (B=4) | post | 17.3917 | 57.4985 | 17.1649 | 58.2583 | 17.5539 | 56.9675 |
|
||||
| RTX 2080Ti | shelfnet 2048x2048 (B=4) | tot | 79.3283 | 12.6058 | 59.5136 | 16.8029 | 59.0903 | 16.9233 |
|
||||
| AGX Xavier | shelfnet 1024x1024 (B=1) | pre | 8.0174 | 124.729 | 7.5117 | 133.126 | 7.47333 | 133.809 |
|
||||
| AGX Xavier | shelfnet 1024x1024 (B=1) | inf | 72.4173 | 13.8089 | 37.505 | 26.6631 | 31.3286 | 31.9197 |
|
||||
| AGX Xavier | shelfnet 1024x1024 (B=1) | post | 8.89958 | 112.365 | 8.83576 | 113.176 | 9.42655 | 106.083 |
|
||||
| AGX Xavier | shelfnet 1024x1024 (B=1) | tot | 89.3342 | 11.1939 | 53.8525 | 18.5692 | 48.2285 | 20.7346 |
|
||||
| AGX Xavier | shelfnet 2048x2048 (B=4) | pre | 47.1454 | 21.211 | 21.6475 | 46.1947 | 21.4201 | 46.6851 |
|
||||
| AGX Xavier | shelfnet 2048x2048 (B=4) | inf | 266.537 | 3.75183 | 128.321 | 7.79293 | 107.621 | 9.29185 |
|
||||
| AGX Xavier | shelfnet 2048x2048 (B=4) | post | 44.0711 | 22.6906 | 40.1732 | 24.8922 | 39.873 | 25.0796 |
|
||||
| AGX Xavier | shelfnet 2048x2048 (B=4) | tot | 357.753 | 2.79522 | 190.142 | 5.25922 | 168.914 | 5.92016 |
|
||||
|
||||
|
||||
## Known issues
|
||||
|
||||
When creating the rt file all the checks returns errors. It is due to a different resize function and handling of the original ShelfNet outputs.
|
||||
However, the network is supposed to work.
|
||||
+120
@@ -0,0 +1,120 @@
|
||||
# 2D Object Detection with tkDNN
|
||||
|
||||
## Supported Networks
|
||||
|
||||
* Yolo4, Yolo4-csp, Yolo4x, Yolo4_berkeley, Yolo4tiny
|
||||
* Yolo3, Yolo3_berkeley, Yolo3_coco4, Yolo3_flir, Yolo3_512, Yolo3tiny, Yolo3tiny_512
|
||||
* Yolo2, Yolo2_voc, Yolo2tiny
|
||||
* Csresnext50-panet-spp, Csresnext50-panet-spp_berkeley
|
||||
* Resnet101_cnet, Dla34_cnet
|
||||
* Mobilenetv2ssd, Mobilenetv2ssd512, Bdd-mobilenetv2ssd
|
||||
|
||||
## Index
|
||||
|
||||
- [2D Object Detection](#2d-object-detection)
|
||||
- [FP16 inference](#fp16-inference)
|
||||
- [INT8 inference](#int8-inference)
|
||||
- [Batching](#batching)
|
||||
|
||||
### 2D Object Detection
|
||||
This is an example using yolov4.
|
||||
|
||||
To run the an object detection first create the .rt file by running:
|
||||
```
|
||||
rm yolo4_fp32.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_yolo4 # run the yolo test (is slow)
|
||||
```
|
||||
If you get problems in the creation, try to check the error activating the debug of TensorRT in this way:
|
||||
```
|
||||
cmake .. -DCMAKE_BUILD_TYPE=Debug -DDEBUG=True
|
||||
make
|
||||
```
|
||||
|
||||
Once you have successfully created your rt file, run the demo:
|
||||
```
|
||||
./ demo <path-to-config>
|
||||
```
|
||||
In general the demo program takes 1 parameter, the ```<path-to-config>``` that is the path to che configuration file. The parameter is optional and its default value is ```"../demo/demoConfig.yaml"```.
|
||||
|
||||
The config file is a yaml file with the following attributes:
|
||||
* ```net``` is the rt file generated by a test
|
||||
* ```input``` is the path to a video file or a camera input (on Linux)
|
||||
* ```win_input``` is the path to a video file or a camera input (on Windows)
|
||||
* ```ntype``` is the type of network. Thee types are currently supported: ```y``` (YOLO family), ```c``` (CenterNet family) and ```m``` (MobileNet-SSD family)
|
||||
* ```n_classes``` is the number of classes the network is trained on
|
||||
* ```n_batch``` number of batches to use in inference (N.B. you should first export TKDNN_BATCHSIZE to the required n_batches and create again the rt file for the network).
|
||||
* ```conf_thresh``` confidence threshold for the detector. Only bounding boxes with threshold greater than conf-thresh will be displayed.
|
||||
* ```show``` if set to 0 the demo will not show the visualization (if n-batches ==1)
|
||||
* ```save``` if set to 1 the demo will save the video of the demo into result.mp4 (if n-batches ==1)
|
||||
|
||||
N.B. By default it is used FP32 inference
|
||||
|
||||
|
||||

|
||||
|
||||
|
||||
### FP16 inference
|
||||
|
||||
To run the demo with FP16 inference follow these steps (example with yolov3):
|
||||
```
|
||||
export TKDNN_MODE=FP16 # set the half floating point optimization
|
||||
rm yolo4_fp16.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_yolo4 # run the yolo test (is slow)
|
||||
#set net: yolo4_fp16.rt in the config file
|
||||
./demo
|
||||
```
|
||||
N.B. Using FP16 inference will lead to some errors in the results (first or second decimal).
|
||||
|
||||
### INT8 inference
|
||||
|
||||
To run the demo with INT8 inference three environment variables need to be set:
|
||||
|
||||
* ```export TKDNN_MODE=INT8```: set the 8-bit integer optimization
|
||||
* ```export TKDNN_CALIB_IMG_PATH=/path/to/calibration/image_list.txt``` : image_list.txt has in each line the absolute path to a calibration image
|
||||
* ```export TKDNN_CALIB_LABEL_PATH=/path/to/calibration/label_list.txt```: label_list.txt has in each line the absolute path to a calibration label
|
||||
|
||||
You should provide image_list.txt and label_list.txt, using training images. However, if you want to quickly test the INT8 inference you can run (from this repo root folder)
|
||||
```
|
||||
bash scripts/download_validation.sh COCO
|
||||
```
|
||||
to automatically download COCO2017 validation (inside demo folder) and create those needed file. Use BDD instead of COCO to download BDD validation.
|
||||
|
||||
Then a complete example using yolo3 and COCO dataset would be:
|
||||
```
|
||||
export TKDNN_MODE=INT8
|
||||
export TKDNN_CALIB_LABEL_PATH=../demo/COCO_val2017/all_labels.txt
|
||||
export TKDNN_CALIB_IMG_PATH=../demo/COCO_val2017/all_images.txt
|
||||
rm yolo4_int8.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_yolo4 # run the yolo test (is slow)
|
||||
#set net: yolo4_int8.rt in the config file
|
||||
./demo
|
||||
```
|
||||
N.B.
|
||||
|
||||
* Using INT8 inference will lead to some errors in the results.
|
||||
* The test will be slower: this is due to the INT8 calibration, which may take some time to complete.
|
||||
* INT8 calibration requires TensorRT version greater than or equal to 6.0
|
||||
* Only 100 images are used to create the calibration table by default (set in the code).
|
||||
|
||||
### Batching
|
||||
|
||||
#### BatchSize bigger than 1
|
||||
```
|
||||
export TKDNN_BATCHSIZE=2
|
||||
# build tensorRT files
|
||||
```
|
||||
This will create a TensorRT file with the desired **max** batch size.
|
||||
The test will still run with a batch of 1, but the created tensorRT can manage the desired batch size.
|
||||
|
||||
#### Test batch Inference
|
||||
This will test the network with random input and check if the output of each batch is the same.
|
||||
```
|
||||
./test_rtinference <network-rt-file> <number-of-batches>
|
||||
# <number-of-batches> should be less or equal to the max batch size of the <network-rt-file>
|
||||
|
||||
# example
|
||||
export TKDNN_BATCHSIZE=4 # set max batch size
|
||||
rm yolo3_fp32.rt # be sure to delete(or move) old tensorRT files
|
||||
./test_yolo3 # build RT file
|
||||
./test_rtinference yolo3_fp32.rt 4 # test with a batch size of 4
|
||||
```
|
||||
@@ -0,0 +1,127 @@
|
||||
# tkDNN export weights
|
||||
|
||||
## Index
|
||||
|
||||
- [How to export weights](#how-to-export-weights)
|
||||
- [1)Export weights from darknet](#1export-weights-from-darknet)
|
||||
- [2)Export weights for DLA34 and ResNet101](#2export-weights-for-dla34-and-resnet101)
|
||||
- [3)Export weights for CenterNet](#3export-weights-for-centernet)
|
||||
- [4)Export weights for MobileNetSSD](#4export-weights-for-mobilenetssd)
|
||||
- [5)Export weights for CenterTrack](#5export-weights-for-centertrack)
|
||||
- [6)Export weights for ShelfNet](#6export-weights-for-shelfnet)
|
||||
- [Darknet Parser](#darknet-parser)
|
||||
|
||||
## How to export weights
|
||||
|
||||
Weights are essential for any network to run inference. For each test a folder organized as follow is needed (in the build folder):
|
||||
```
|
||||
test_nn
|
||||
|---- layers/ (folder containing a binary file for each layer with the corresponding wieghts and bias)
|
||||
|---- debug/ (folder containing a binary file for each layer with the corresponding outputs)
|
||||
```
|
||||
Therefore, once the weights have been exported, the folders layers and debug should be placed in the corresponding test.
|
||||
|
||||
### 1)Export weights from darknet
|
||||
To export weights for NNs that are defined in darknet framework, use [this](https://git.hipert.unimore.it/fgatti/darknet.git) fork of darknet and follow these steps to obtain a correct debug and layers folder, ready for tkDNN.
|
||||
|
||||
```
|
||||
git clone https://git.hipert.unimore.it/fgatti/darknet.git
|
||||
cd darknet
|
||||
make
|
||||
mkdir layers debug
|
||||
./darknet export <path-to-cfg-file> <path-to-weights> layers
|
||||
```
|
||||
N.B. Use compilation with CPU (leave GPU=0 in Makefile) if you also want debug.
|
||||
|
||||
### 2)Export weights for DLA34 and ResNet101
|
||||
To get weights and outputs needed to run the tests dla34 and resnet101 use the Python script and the Anaconda environment included in the repository.
|
||||
|
||||
Create Anaconda environment and activate it:
|
||||
```
|
||||
conda env create -f file_name.yml
|
||||
source activate env_name
|
||||
python <script name>
|
||||
```
|
||||
### 3)Export weights for CenterNet
|
||||
To get the weights needed to run Centernet tests use [this](https://github.com/sapienzadavide/CenterNet.git) fork of the original Centernet.
|
||||
```
|
||||
git clone https://github.com/sapienzadavide/CenterNet.git
|
||||
```
|
||||
* follow the instruction in the README.md and INSTALL.md
|
||||
|
||||
```
|
||||
python demo.py --input_res 512 --arch resdcn_101 ctdet --demo /path/to/image/or/folder/or/video/or/webcam --load_model ../models/ctdet_coco_resdcn101.pth --exp_wo --exp_wo_dim 512
|
||||
python demo.py --input_res 512 --arch dla_34 ctdet --demo /path/to/image/or/folder/or/video/or/webcam --load_model ../models/ctdet_coco_dla_2x.pth --exp_wo --exp_wo_dim 512
|
||||
```
|
||||
### 4)Export weights for MobileNetSSD
|
||||
To get the weights needed to run Mobilenet tests use [this](https://github.com/mive93/pytorch-ssd) fork of a Pytorch implementation of SSD network.
|
||||
|
||||
```
|
||||
git clone https://github.com/mive93/pytorch-ssd
|
||||
cd pytorch-ssd
|
||||
conda env create -f env_mobv2ssd.yml
|
||||
python run_ssd_live_demo.py mb2-ssd-lite <pth-model-fil> <labels-file>
|
||||
```
|
||||
### 5)Export weights for CenterTrack
|
||||
To get the weights needed to run CenterTrack tests use [this](https://github.com/sapienzadavide/CenterTrack.git) fork of the original CenterTrack.
|
||||
```
|
||||
git clone https://github.com/sapienzadavide/CenterTrack.git
|
||||
```
|
||||
* follow the instruction in the README.md and INSTALL.md
|
||||
|
||||
```
|
||||
python demo.py tracking,ddd --load_model ../models/nuScenes_3Dtracking.pth --dataset nuscenes --pre_hm --track_thresh 0.1 --demo /path/to/image/or/folder/or/video/or/webcam --test_focal_length 633 --exp_wo --exp_wo_dim 512 --input_h 512 --input_w 512
|
||||
```
|
||||
|
||||
### 6)Export weights for ShelfNet
|
||||
To get the weights needed to run Shelfnet tests use [this](https://git.hipert.unimore.it/mverucchi/shelfnet) fork of a Pytorch implementation of Shelfnet network.
|
||||
|
||||
```
|
||||
git clone https://git.hipert.unimore.it/mverucchi/shelfnet
|
||||
cd shelfnet
|
||||
cd ShelfNet18_realtime
|
||||
conda env create --file shelfnet_env.yml
|
||||
conda activate shelfnet
|
||||
mkdir layer debug
|
||||
python export.py
|
||||
```
|
||||
|
||||
### 6)Export weights for monodepth2
|
||||
To get the weights needed to run Shelfnet tests use [this](https://github.com/perseusdg/monodepth2) fork of a Pytorch implementation of monodepth2 network.
|
||||
|
||||
```
|
||||
git clone https://github.com/perseusdg/monodepth2
|
||||
cd monodepth2
|
||||
mkdir models # Download the official weights and put depth.pth and encorder.pth inside this new folder
|
||||
conda env create --file monodepth.yaml
|
||||
conda activate monodepth2
|
||||
python exporter.py # you will find the weights inside the tkDNN_bin folder
|
||||
```
|
||||
|
||||
## Darknet Parser
|
||||
tkDNN implement and easy parser for darknet cfg files, a network can be converted with *tk::dnn::darknetParser*:
|
||||
```
|
||||
// example of parsing yolo4
|
||||
tk::dnn::Network *net = tk::dnn::darknetParser("yolov4.cfg", "yolov4/layers", "coco.names");
|
||||
net->print();
|
||||
```
|
||||
All models from darknet are now parsed directly from cfg, you still need to export the weights with the described tools in the previous section.
|
||||
<details>
|
||||
<summary>Supported layers</summary>
|
||||
convolutional
|
||||
maxpool
|
||||
avgpool
|
||||
shortcut
|
||||
upsample
|
||||
route
|
||||
reorg
|
||||
region
|
||||
yolo
|
||||
</details>
|
||||
<details>
|
||||
<summary>Supported activations</summary>
|
||||
relu
|
||||
leaky
|
||||
mish
|
||||
logistic
|
||||
</details>
|
||||
@@ -0,0 +1,32 @@
|
||||
# Run the mAP demo
|
||||
|
||||
To compute mAP, precision, recall and f1score to evaluate 2D object detectors, run the map_demo.
|
||||
|
||||
A validation set is needed.
|
||||
To download COCO_val2017 (80 classes) run (form the root folder):
|
||||
```
|
||||
bash scripts/download_validation.sh COCO
|
||||
```
|
||||
To download Berkeley_val (10 classes) run (form the root folder):
|
||||
```
|
||||
bash scripts/download_validation.sh BDD
|
||||
```
|
||||
|
||||
To compute the map, the following parameters are needed:
|
||||
```
|
||||
./map_demo <network rt> <network type [y|c|m]> <labels file path> <config file path>
|
||||
```
|
||||
where
|
||||
* ```<network rt>```: rt file of a chosen network on which compute the mAP.
|
||||
* ```<network type [y|c|m]>```: type of network. Right now only y(yolo), c(centernet) and m(mobilenet) are allowed
|
||||
* ```<labels file path>```: path to a text file containing all the paths of the ground-truth labels. It is important that all the labels of the ground-truth are in a folder called 'labels'. In the folder containing the folder 'labels' there should be also a folder 'images', containing all the ground-truth images having the same same as the labels. To better understand, if there is a label path/to/labels/000001.txt there should be a corresponding image path/to/images/000001.jpg.
|
||||
* ```<config file path>```: path to a yaml file with the parameters needed for the mAP computation, similar to demo/config.yaml
|
||||
|
||||
Example:
|
||||
|
||||
```
|
||||
cd build
|
||||
./map_demo dla34_cnet_FP32.rt c ../demo/COCO_val2017/all_labels.txt ../demo/config.yaml
|
||||
```
|
||||
|
||||
This demo also creates a json file named ```net_name_COCO_res.json``` containing all the detections computed. The detections are in COCO format, the correct format to submit the results to [CodaLab COCO detection challenge](https://competitions.codalab.org/competitions/20794#participate).
|
||||
+102
@@ -0,0 +1,102 @@
|
||||
# tkDNN on Windows
|
||||
|
||||
## Index
|
||||
|
||||
- [Dependencies-Windows](#dependencies-windows)
|
||||
- [Compiling tkDNN on Windows](#compiling-tkdnn-on-windows)
|
||||
- [Run the demo on Windows](#run-the-demo-on-windows)
|
||||
- [FP16 inference windows](#fp16-inference-windows)
|
||||
- [INT8 inference windows](#int8-inference-windows)
|
||||
- [Run tkDNN on WSL2 with cuda](#tkdnn-on-cuda-wsl)
|
||||
- [Known issues with tkDNN on Windows](#known-issues-with-tkdnn-on-windows)
|
||||
|
||||
### Dependencies-Windows
|
||||
This branch should work on every NVIDIA GPU supported in windows with the following dependencies:
|
||||
|
||||
* WINDOWS 10 1803/WINDOWS 11 or HIGHER
|
||||
* CUDA 11.2
|
||||
* CUDNN 8.1.1
|
||||
* TENSORRT 7.2.3
|
||||
* OPENCV 4.2
|
||||
* MSVC 16.9+
|
||||
* YAML-CPP
|
||||
* EIGEN3
|
||||
* 7ZIP (ADD TO PATH)
|
||||
* NINJA 1.10
|
||||
|
||||
|
||||
All the above mentioned dependencies except 7ZIP can be installed using Microsoft's [VCPKG](https://github.com/microsoft/vcpkg.git) .
|
||||
After bootstrapping VCPKG the dependencies can be built and installed using the following command :
|
||||
|
||||
```
|
||||
opencv4(normal) - vcpkg.exe install opencv4[tbb,jpeg,tiff,opengl,openmp,png,ffmpeg,eigen]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build
|
||||
|
||||
opencv4(cuda) - vcpkg.exe install opencv4[cuda,nonfree,contrib,eigen,tbb,jpeg,tiff,opengl,openmp,png,ffmpeg]:x64-windows yaml-cpp:x64-windows eigen3:x64-windows --x-install-root=C:\opt --x-buildtrees-root=C:\temp_vcpkg_build
|
||||
```
|
||||
To build opencv4 with cuda and cudnn version corresponding to your cuda version,vcpkg's cudnn portfile needs to be modified by adding ```$ENV{CUDA_PATH}``` at lines 16 and 17 in the portfile.cmake
|
||||
|
||||
After VCPKG finishes building and installing all the packages delete C:\temp_vcpkg_build and add C:\opt\x64-windows\bin and C:\opt\x64-windows\debug\bin to path
|
||||
|
||||
### Compiling tkDNN on Windows
|
||||
|
||||
tkDNN is built with cmake(3.15+) on windows along with ninja.Msbuild and NMake Makefiles are drastically slower when compiling the library compared to windows
|
||||
```
|
||||
git clone https://github.com/ceccocats/tkDNN.git
|
||||
cd tkdnn-windows
|
||||
mkdir build
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=Release -G"Ninja" ..
|
||||
ninja -j4
|
||||
```
|
||||
|
||||
### Run the demo on Windows
|
||||
|
||||
This example uses yolo4_tiny.\
|
||||
To run the object detection file create .rt file bu running:
|
||||
```
|
||||
.\test_yolo4tiny.exe
|
||||
```
|
||||
|
||||
Once the rt file has been successfully create,run the demo using the following command:
|
||||
```
|
||||
.\demo.exe yolo4_fp32.rt ..\demo\yolo_test.mp4 y 80 ..\tests\darknet\cfg\yolo4.cfg ..\tests\darknet\names\cococ.names
|
||||
```
|
||||
For general info on more demo paramters,check Run the demo section on top
|
||||
To run the test_all_tests.sh on windows,use git bash or msys2
|
||||
|
||||
### FP16 inference windows
|
||||
|
||||
This is an untested feature on windows.To run the object detection demo with FP16 interference follow the below steps(example with yolo4tiny):
|
||||
```
|
||||
set TKDNN_MODE=FP16
|
||||
del /f yolo4tiny_fp16.rt
|
||||
.\test_yolo4tiny.exe
|
||||
.\demo.exe yolo4tiny_fp16.rt ..\demo\yolo_test.mp4
|
||||
```
|
||||
|
||||
### INT8 inference windows
|
||||
To run object detection demo with INT8 (example with yolo4tiny):
|
||||
```
|
||||
set TKDNN_MODE=INT8
|
||||
set TKDNN_CALIB_LABEL_PATH=..\demo\COCO_val2017\all_labels.txt
|
||||
set TKDNN_CALIB_IMG_PATH=..\demo\COCO_val2017\all_images.txt
|
||||
del /f yolo4tiny_int8.rt # be sure to delete(or move) old tensorRT files
|
||||
.\test_yolo4tiny.exe # run the yolo test (is slow)
|
||||
.\demo.exe yolo4tiny_int8.rt ..\demo\yolo_test.mp4 y
|
||||
|
||||
```
|
||||
|
||||
### Run tkDNN on WSL2 with cuda
|
||||
tkDNN works on wsl2 with cuda,although not all networks (centernet,mobilenet) work properly.
|
||||
If you encounter issues with running the network as a result of driver not found or cuda launch error,running the following command should solve the issue
|
||||
```cp /usr/lib/wsl/lib/lib* /usr/lib/x86_64-linux-gnu/ ```
|
||||
|
||||
|
||||
|
||||
### Known issues with tkDNN on Windows
|
||||
|
||||
In theory all models (centernet,mobilenet,darknet,centertrack,cnet3d and shelfnet) should work on Windows.
|
||||
|
||||
On pascal cards(sm 6x) ,nvidia cuda wsl driver 510.06 don't work well with tkDNN both on windows and cuda wsl , Nvidia drivers >465+ and < 500 are completely supported .
|
||||
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
#ifndef BOUNDINGBOX_H
|
||||
#define BOUNDINGBOX_H
|
||||
|
||||
#include <iostream>
|
||||
#include "tkdnn.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
class BoundingBox : public tk::dnn::box
|
||||
{
|
||||
float overlap(const float p1, const float l1, const float p2, const float l22);
|
||||
float boxesIntersection(const BoundingBox &b);
|
||||
float boxesUnion(const BoundingBox &b);
|
||||
|
||||
public:
|
||||
|
||||
int uniqueTruthIndex = -1;
|
||||
int truthFlag = 0;
|
||||
float maxIoU = 0;
|
||||
|
||||
float IoU(const BoundingBox &b);
|
||||
void clear();
|
||||
|
||||
friend std::ostream& operator<<(std::ostream& os, const BoundingBox& bb);
|
||||
};
|
||||
|
||||
std::ostream& operator<<(std::ostream& os, const BoundingBox& bb);
|
||||
bool boxComparison (const BoundingBox& a,const BoundingBox& b) ;
|
||||
|
||||
}}
|
||||
#endif /*BOUNDINGBOX_H*/
|
||||
|
||||
@@ -0,0 +1,186 @@
|
||||
#ifndef CENTERTRACK_H
|
||||
#define CENTERTRACK_H
|
||||
|
||||
#include <opencv2/videoio.hpp>
|
||||
#include "opencv2/opencv.hpp"
|
||||
#include "kernels.h"
|
||||
#include "utils.h"
|
||||
#include "tkdnn.h"
|
||||
#include <time.h>
|
||||
#include <vector>
|
||||
#include <numeric> // std::iota
|
||||
#include <algorithm> // std::sort
|
||||
|
||||
#include "TrackingNN.h"
|
||||
|
||||
#ifdef _WIN32
|
||||
#define _USE_MATH_DEFINES
|
||||
#include <math.h>
|
||||
#endif
|
||||
|
||||
#include "kernelsThrust.h"
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
struct detectionRes
|
||||
{
|
||||
float score;
|
||||
int cl;
|
||||
cv::Mat ct, tr, bb0, bb1;
|
||||
float dep;
|
||||
float dim[3];
|
||||
float alpha;
|
||||
float x,y,z;
|
||||
float rot_y;
|
||||
detectionRes() : ct(cv::Mat(cv::Size(1,2), CV_32F)),
|
||||
tr(cv::Mat(cv::Size(1,2), CV_32F)),
|
||||
bb0(cv::Mat(cv::Size(1,2), CV_32F)),
|
||||
bb1(cv::Mat(cv::Size(1,2), CV_32F)) { }
|
||||
~detectionRes() {
|
||||
ct.release();
|
||||
tr.release();
|
||||
bb0.release();
|
||||
bb1.release();
|
||||
}
|
||||
};
|
||||
|
||||
struct trackingRes
|
||||
{
|
||||
struct detectionRes det_res;
|
||||
int tracking_id;
|
||||
int age;
|
||||
int active;
|
||||
int color;
|
||||
};
|
||||
|
||||
class CenterTrack : public TrackingNN
|
||||
{
|
||||
public:
|
||||
tk::dnn::dataDim_t dim;
|
||||
tk::dnn::dataDim_t dim2;
|
||||
tk::dnn::dataDim_t dim_hm;
|
||||
tk::dnn::dataDim_t dim_wh;
|
||||
tk::dnn::dataDim_t dim_reg;
|
||||
tk::dnn::dataDim_t dim_track;
|
||||
tk::dnn::dataDim_t dim_dep;
|
||||
tk::dnn::dataDim_t dim_rot;
|
||||
tk::dnn::dataDim_t dim_dim;
|
||||
tk::dnn::dataDim_t dim_amodel_offset;
|
||||
|
||||
/* preprocessing */
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
float *mean_d;
|
||||
float *stddev_d;
|
||||
#else
|
||||
cv::Vec<float, 3> mean;
|
||||
cv::Vec<float, 3> stddev;
|
||||
dnnType *input;
|
||||
#endif
|
||||
float *d_ptrs;
|
||||
|
||||
std::vector<cv::Mat> inputCalibs;
|
||||
|
||||
std::vector<cv::Size> szOld;
|
||||
|
||||
cv::Mat src;
|
||||
cv::Mat dst;
|
||||
cv::Mat dst2;
|
||||
cv::Mat trans, trans2, transOut;
|
||||
|
||||
/* pre inf */
|
||||
bool iter0;
|
||||
dnnType *input_pre_inf_d;
|
||||
bool test_pre_inf = true;
|
||||
dnnType *img_d, *hm_d;
|
||||
tk::dnn::dataDim_t dim_in0;
|
||||
tk::dnn::dataDim_t dim_in1;
|
||||
dnnType *out_d;
|
||||
|
||||
|
||||
/* postprocessing */
|
||||
int K = 100;
|
||||
int width = 128;//56; // TODO
|
||||
|
||||
// pointer used in the kernels
|
||||
float *src_out;
|
||||
int *ids_out;
|
||||
|
||||
float *topk_scores;
|
||||
int *topk_inds_;
|
||||
float *topk_ys_;
|
||||
float *topk_xs_;
|
||||
int *ids_d, *ids_;
|
||||
|
||||
float *ones;
|
||||
|
||||
float *scores, *scores_d;
|
||||
int *clses, *clses_d;
|
||||
int *topk_inds_d;
|
||||
float *topk_ys_d;
|
||||
float *topk_xs_d;
|
||||
int *inttopk_xs_d, *inttopk_ys_d;
|
||||
|
||||
float *bbx0, *bby0, *bbx1, *bby1;
|
||||
float *bbx0_d, *bby0_d, *bbx1_d, *bby1_d;
|
||||
|
||||
int *intxs, *intys;
|
||||
|
||||
float *track, *dep, *rot, *dim_, *wh, *amodel_offset;
|
||||
float *track_d, *dep_d, *rot_d, *dim_d, *wh_d, *amodel_offset_d;
|
||||
|
||||
float *target_coords;
|
||||
|
||||
/* visualization */
|
||||
cv::Mat r;
|
||||
std::vector<cv::Mat> calibs;
|
||||
cv::Mat corners, pts3DHomo;
|
||||
|
||||
std::vector<std::vector<int>> faceId;
|
||||
cv::Scalar trColors[256];
|
||||
bool mode3D;
|
||||
|
||||
//processing
|
||||
struct threshold op;
|
||||
float outThresh = 0.1;
|
||||
float newThresh = 0.3;
|
||||
// float peakThreshold = 0.2;
|
||||
// float centerThreshold = 0.3; //default 0.5
|
||||
|
||||
|
||||
//detections
|
||||
std::vector<struct detectionRes> detRes;
|
||||
int countDet;
|
||||
//tracks
|
||||
std::vector<std::vector<struct trackingRes>> trRes;
|
||||
std::vector<int> countTr;
|
||||
std::vector<int> trackId;
|
||||
|
||||
|
||||
bool init_preprocessing();
|
||||
bool init_pre_inf();
|
||||
bool init_postprocessing();
|
||||
bool init_visualization(const int n_classes);
|
||||
void pre_inf(const int bi);
|
||||
void _get_additional_inputs();
|
||||
cv::Mat transform_preds_with_trans(float x1, float x2);
|
||||
void tracking(const int bi);
|
||||
|
||||
public:
|
||||
tk::dnn::Network *pre_phase_net = nullptr;
|
||||
CenterTrack() {};
|
||||
~CenterTrack() {};
|
||||
bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1,
|
||||
const float conf_thresh=0.3, const bool mode_3d=true,
|
||||
const std::vector<cv::Mat>& k_calibs=std::vector<cv::Mat>());
|
||||
void preprocess(cv::Mat &frame, const int bi=0);
|
||||
void postprocess(const int bi=0,const bool mAP=false);
|
||||
void draw(std::vector<cv::Mat>& frames);
|
||||
};
|
||||
|
||||
|
||||
} // namespace dnn
|
||||
} // namespace tk
|
||||
|
||||
|
||||
#endif /*CENTERTRACK_H*/
|
||||
@@ -0,0 +1,86 @@
|
||||
#ifndef CENTERNETDETECTION_H
|
||||
#define CENTERNETDETECTION_H
|
||||
|
||||
#include "kernels.h"
|
||||
#include <opencv2/videoio.hpp>
|
||||
#include "opencv2/opencv.hpp"
|
||||
#include <time.h>
|
||||
#include <vector>
|
||||
#include <numeric> // std::iota
|
||||
#include <algorithm> // std::sort
|
||||
|
||||
#include "DetectionNN.h"
|
||||
|
||||
#include "kernelsThrust.h"
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
class CenternetDetection : public DetectionNN
|
||||
{
|
||||
private:
|
||||
tk::dnn::dataDim_t dim;
|
||||
tk::dnn::dataDim_t dim2;
|
||||
tk::dnn::dataDim_t dim_hm;
|
||||
tk::dnn::dataDim_t dim_wh;
|
||||
tk::dnn::dataDim_t dim_reg;
|
||||
float *topk_scores;
|
||||
int *topk_inds_;
|
||||
float *topk_ys_;
|
||||
float *topk_xs_;
|
||||
int *ids_d, *ids_, *ids_2, *ids_2d;
|
||||
|
||||
float *scores, *scores_d;
|
||||
int *clses, *clses_d;
|
||||
int *topk_inds_d;
|
||||
float *topk_ys_d;
|
||||
float *topk_xs_d;
|
||||
int *inttopk_xs_d, *inttopk_ys_d;
|
||||
|
||||
|
||||
float *bbx0, *bby0, *bbx1, *bby1;
|
||||
float *bbx0_d, *bby0_d, *bbx1_d, *bby1_d;
|
||||
|
||||
float *target_coords;
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
float *mean_d;
|
||||
float *stddev_d;
|
||||
#else
|
||||
cv::Vec<float, 3> mean;
|
||||
cv::Vec<float, 3> stddev;
|
||||
dnnType *input;
|
||||
#endif
|
||||
|
||||
float *d_ptrs;
|
||||
|
||||
cv::Mat src;
|
||||
cv::Mat dst;
|
||||
cv::Mat dst2;
|
||||
cv::Mat trans, trans2;
|
||||
//processing
|
||||
float toll = 0.000001;
|
||||
int K = 100;
|
||||
int width = 128;//56; // TODO
|
||||
|
||||
// pointer used in the kernels
|
||||
float *src_out;
|
||||
int *ids_out;
|
||||
|
||||
struct threshold op;
|
||||
|
||||
public:
|
||||
CenternetDetection() {};
|
||||
~CenternetDetection() {};
|
||||
|
||||
bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1, const float conf_thresh=0.3);
|
||||
void preprocess(cv::Mat &frame, const int bi=0);
|
||||
void postprocess(const int bi=0,const bool mAP=false);
|
||||
};
|
||||
|
||||
|
||||
} // namespace dnn
|
||||
} // namespace tk
|
||||
|
||||
|
||||
#endif /*CENTERNETDETECTION_H*/
|
||||
@@ -0,0 +1,106 @@
|
||||
#ifndef CENTERNETDETECTION3D_H
|
||||
#define CENTERNETDETECTION3D_H
|
||||
|
||||
#include "kernels.h"
|
||||
#include <opencv2/videoio.hpp>
|
||||
#include "opencv2/opencv.hpp"
|
||||
#include <time.h>
|
||||
#include <vector>
|
||||
#include <numeric> // std::iota
|
||||
#include <algorithm> // std::sort
|
||||
|
||||
#ifdef _WIN32
|
||||
#define _USE_MATH_DEFINES
|
||||
#include <math.h>
|
||||
#endif
|
||||
|
||||
#include "DetectionNN3D.h"
|
||||
|
||||
#include "kernelsThrust.h"
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
class CenternetDetection3D : public DetectionNN3D
|
||||
{
|
||||
private:
|
||||
tk::dnn::dataDim_t dim;
|
||||
tk::dnn::dataDim_t dim2;
|
||||
tk::dnn::dataDim_t dim_hm;
|
||||
tk::dnn::dataDim_t dim_wh;
|
||||
tk::dnn::dataDim_t dim_reg;
|
||||
tk::dnn::dataDim_t dim_dep;
|
||||
tk::dnn::dataDim_t dim_rot;
|
||||
tk::dnn::dataDim_t dim_dim;
|
||||
|
||||
std::vector<cv::Mat> inputCalibs;
|
||||
float *topk_scores;
|
||||
int *topk_inds_;
|
||||
float *topk_ys_;
|
||||
float *topk_xs_;
|
||||
int *ids_d, *ids_;
|
||||
|
||||
float *ones;
|
||||
|
||||
float *scores, *scores_d;
|
||||
int *clses, *clses_d;
|
||||
int *topk_inds_d;
|
||||
float *topk_ys_d;
|
||||
float *topk_xs_d;
|
||||
int *inttopk_xs_d, *inttopk_ys_d;
|
||||
|
||||
float *xs, *ys;
|
||||
|
||||
float *dep, *rot, *dim_, *wh;
|
||||
float *dep_d, *rot_d, *dim_d, *wh_d;
|
||||
|
||||
float *target_coords;
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
float *mean_d;
|
||||
float *stddev_d;
|
||||
#else
|
||||
cv::Vec<float, 3> mean;
|
||||
cv::Vec<float, 3> stddev;
|
||||
dnnType *input;
|
||||
#endif
|
||||
cv::Mat r;
|
||||
float *d_ptrs;
|
||||
|
||||
cv::Size sz_old;
|
||||
|
||||
cv::Mat src;
|
||||
cv::Mat dst;
|
||||
cv::Mat dst2;
|
||||
cv::Mat trans, trans2;
|
||||
std::vector<cv::Mat> calibs;
|
||||
|
||||
//processing
|
||||
int K = 100;
|
||||
int width = 128;//56; // TODO
|
||||
|
||||
// pointer used in the kernels
|
||||
float *srcOut;
|
||||
int *idsOut;
|
||||
|
||||
struct threshold op;
|
||||
cv::Mat corners, pts3DHomo;
|
||||
|
||||
std::vector<std::vector<int>> faceId;
|
||||
|
||||
public:
|
||||
CenternetDetection3D() {};
|
||||
~CenternetDetection3D() {};
|
||||
|
||||
bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1, const float conf_thresh=0.3, const std::vector<cv::Mat>& k_calibs=std::vector<cv::Mat>());
|
||||
void preprocess(cv::Mat &frame, const int bi=0);
|
||||
void postprocess(const int bi=0,const bool mAP=false);
|
||||
void draw(std::vector<cv::Mat>& frames);
|
||||
};
|
||||
|
||||
|
||||
} // namespace dnn
|
||||
} // namespace tk
|
||||
|
||||
|
||||
#endif /*CENTERNETDETECTION_H*/
|
||||
@@ -0,0 +1,54 @@
|
||||
#pragma once
|
||||
#include <iostream>
|
||||
#include "tkDNN/tkdnn.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
struct darknetFields_t{
|
||||
std::string type = "";
|
||||
int width = 0;
|
||||
int height = 0;
|
||||
int channels = 3;
|
||||
int batch_normalize=0;
|
||||
int groups = 1;
|
||||
int group_id = 0;
|
||||
int filters=1;
|
||||
int size_x=1;
|
||||
int size_y=1;
|
||||
int stride_x=1;
|
||||
int stride_y=1;
|
||||
int padding_x = 0;
|
||||
int padding_y = 0;
|
||||
int n_mask = 0;
|
||||
int classes = 20;
|
||||
int num = 1;
|
||||
int pad = 0;
|
||||
int coords = 4;
|
||||
int nms_kind = 0;
|
||||
int new_coords= 0;
|
||||
float scale_xy = 1;
|
||||
float nms_thresh = 0.45;
|
||||
std::vector<int> layers;
|
||||
std::string activation = "linear";
|
||||
|
||||
friend std::ostream& operator<<(std::ostream& os, const darknetFields_t& f){
|
||||
os << f.width << " " << f.height << " " << f.channels << " " << f.batch_normalize<< " " << f.filters << " " << f.activation<< " " << f.scale_xy;
|
||||
return os;
|
||||
}
|
||||
};
|
||||
|
||||
std::string darknetParseType(const std::string& line);
|
||||
bool divideNameAndValue(const std::string& line, std::string&name, std::string& value);
|
||||
std::vector<int> fromStringToIntVec(const std::string& line, const char delimiter);
|
||||
|
||||
bool darknetParseFields(const std::string& line, darknetFields_t& fields);
|
||||
tk::dnn::Network *darknetAddNet(darknetFields_t &fields);
|
||||
void darknetAddLayer(tk::dnn::Network *net, darknetFields_t &f, std::string wgs_path,
|
||||
std::vector<tk::dnn::Layer*> &netLayers, const std::vector<std::string>& names);
|
||||
std::vector<std::string> darknetReadNames(const std::string& names_file);
|
||||
tk::dnn::Network* darknetParser(const std::string& cfg_file, const std::string& wgs_path, const std::string& names_file);
|
||||
void loadYoloInfo(const std::string &cfg_file,int lineNo,std::vector<float> &mask,std::vector<float> &anchors,int &num,int &classes,float &nms_thresh,int &nms_kind,int &coords);
|
||||
void loadYoloInitInfo(int &channels,int &width,int &height,const std::string &cfg_file);
|
||||
std::vector<int> noYolosLine(const std::string &cfg_file);
|
||||
|
||||
}}
|
||||
@@ -0,0 +1,180 @@
|
||||
#ifndef DEPTHNN_H
|
||||
#define DEPTHNN_H
|
||||
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h>
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <mutex>
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
#include "tkDNN/utils.h"
|
||||
#include "tkDNN/tkdnn.h"
|
||||
|
||||
#include "NetworkViz.h"
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
class DepthNN {
|
||||
|
||||
public:
|
||||
tk::dnn::NetworkRT *netRT = nullptr;
|
||||
dnnType *input_h;
|
||||
dnnType *input_d;
|
||||
float* depth_h;
|
||||
|
||||
int output_w;
|
||||
int output_h;
|
||||
|
||||
int nBatches = 1;
|
||||
|
||||
cv::Mat bgr[3];
|
||||
cv::Mat imagePreproc;
|
||||
|
||||
std::vector<double> stats; /*keeps track of inference times (ms)*/
|
||||
std::vector<std::vector<float>> depths;
|
||||
std::vector<cv::Mat> depthMats;
|
||||
|
||||
DepthNN() {};
|
||||
~DepthNN(){};
|
||||
|
||||
/**
|
||||
* Method used to initialize the class, allocate memory and compute
|
||||
* needed data.
|
||||
*
|
||||
* @param tensor_path path to the rt file of the NN.
|
||||
* @param n_batches maximum number of batches to use in inference
|
||||
* @return true if everything is correct, false otherwise.
|
||||
*/
|
||||
void init(const std::string& tensor_path, const int n_batches=1){
|
||||
//create net
|
||||
|
||||
std::cout<<(tensor_path).c_str()<<"\n";
|
||||
nBatches = n_batches;
|
||||
netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str());
|
||||
|
||||
//allocate memory for NN input
|
||||
checkCuda(cudaMallocHost(&input_h, sizeof(dnnType) * netRT->input_dim.tot() * nBatches));
|
||||
checkCuda(cudaMalloc(&input_d, sizeof(dnnType) * netRT->input_dim.tot() * nBatches));
|
||||
|
||||
//allocate memory for NN output
|
||||
depthMats.resize(nBatches);
|
||||
depths.resize(nBatches);
|
||||
for(int i=0; i< depths.size();++i)
|
||||
depths[i].resize(netRT->buffersDIM[1].tot());
|
||||
|
||||
depth_h = (float *)malloc(netRT->buffersDIM[1].tot() * sizeof(float));
|
||||
|
||||
output_h = netRT->buffersDIM[1].h;
|
||||
output_w = netRT->buffersDIM[1].w;
|
||||
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* This method preprocess the image, before feeding it to the NN.
|
||||
*
|
||||
* @param frame original frame to adapt for inference.
|
||||
* @param bi batch index
|
||||
*/
|
||||
void preprocess(cv::Mat &frame, const int bi=0) {
|
||||
//resize image, remove mean, divide by std
|
||||
cv::Mat frame_nomean;
|
||||
resize(frame, frame, cv::Size(netRT->input_dim.w, netRT->input_dim.h));
|
||||
frame.convertTo(frame_nomean, CV_32FC3);
|
||||
frame_nomean.convertTo(imagePreproc, CV_32FC3, 1 / 255.0, 0);
|
||||
|
||||
//copy image into tensor and copy it into GPU
|
||||
cv::split(imagePreproc, bgr);
|
||||
for (int i = 0; i < netRT->input_dim.c; i++){
|
||||
int idx = i * imagePreproc.rows * imagePreproc.cols;
|
||||
int ch = netRT->input_dim.c-1 -i;
|
||||
memcpy((void *)&input_h[idx + netRT->input_dim.tot()*bi], (void *)bgr[ch].data, imagePreproc.rows * imagePreproc.cols * sizeof(dnnType));
|
||||
}
|
||||
checkCuda(cudaMemcpyAsync(input_d+ netRT->input_dim.tot()*bi, input_h + netRT->input_dim.tot()*bi, netRT->input_dim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream));
|
||||
}
|
||||
|
||||
/**
|
||||
* This method postprocess the output of the NN to obtain the correct
|
||||
* boundig boxes.
|
||||
*
|
||||
* @param bi batch index
|
||||
* @param mAP set to true only if all the probabilities for a bounding
|
||||
* box are needed, as in some cases for the mAP calculation
|
||||
*/
|
||||
void postprocess(const int bi=0) {
|
||||
|
||||
dnnType *rt_out[1];
|
||||
rt_out[0] = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi;
|
||||
checkCuda(cudaMemcpy(depth_h, rt_out[0], netRT->buffersDIM[1].tot()* sizeof(float), cudaMemcpyDeviceToHost));
|
||||
memcpy(&depths[bi][0], &depth_h[0], netRT->buffersDIM[1].tot()* sizeof(float));
|
||||
|
||||
// cv::Mat d(netRT->buffersDIM[1].h, netRT->buffersDIM[1].w, CV_8UC1, depth_h);
|
||||
// depthMats[bi] = d.clone();
|
||||
|
||||
cv::Mat depth_mat = vizData2Mat(rt_out[0], netRT->buffersDIM[1], netRT->buffersDIM[1].h, netRT->buffersDIM[1].w);
|
||||
// cv::Mat depth_mat = vizData2Mat((dnnType *)netRT->buffersRT[0], netRT->buffersDIM[0], netRT->buffersDIM[0].h, netRT->buffersDIM[0].w);
|
||||
depthMats[bi] = depth_mat.clone();
|
||||
|
||||
}
|
||||
|
||||
/**
|
||||
* This method performs the inference of the NN.
|
||||
*
|
||||
* @param frames frames to build the embedding from.
|
||||
* @param cur_batches number of batches to use in inference
|
||||
*/
|
||||
void update(std::vector<cv::Mat>& frames, const int cur_batches=1){
|
||||
if(cur_batches > nBatches)
|
||||
FatalError("A batch size greater than nBatches cannot be used");
|
||||
|
||||
if(TKDNN_VERBOSE) printCenteredTitle(" TENSORRT feature extraction ", '=', 30);
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi){
|
||||
if(!frames[bi].data)
|
||||
FatalError("No image data feed to extract features");
|
||||
preprocess(frames[bi], bi);
|
||||
}
|
||||
TKDNN_TSTOP
|
||||
}
|
||||
|
||||
//do inference
|
||||
tk::dnn::dataDim_t dim = netRT->input_dim;
|
||||
dim.n = cur_batches;
|
||||
{
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
TKDNN_TSTART
|
||||
netRT->infer(dim, input_d);
|
||||
TKDNN_TSTOP
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
stats.push_back(t_ns);
|
||||
}
|
||||
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi)
|
||||
postprocess(bi);
|
||||
TKDNN_TSTOP
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Method to draw the result.
|
||||
*
|
||||
*/
|
||||
void draw() { }
|
||||
|
||||
};
|
||||
|
||||
|
||||
}}
|
||||
|
||||
#endif /* DEPTHNN_H*/
|
||||
@@ -0,0 +1,185 @@
|
||||
#ifndef DETECTIONNN_H
|
||||
#define DETECTIONNN_H
|
||||
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h>
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <mutex>
|
||||
#include "utils.h"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
#include "tkdnn.h"
|
||||
|
||||
//#define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib.
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
#include <opencv2/cudawarping.hpp>
|
||||
#include <opencv2/cudaarithm.hpp>
|
||||
#endif
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
class DetectionNN {
|
||||
|
||||
protected:
|
||||
tk::dnn::NetworkRT *netRT = nullptr;
|
||||
dnnType *input_d;
|
||||
|
||||
std::vector<cv::Size> originalSize;
|
||||
|
||||
cv::Scalar colors[256];
|
||||
|
||||
int nBatches = 1;
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
cv::cuda::GpuMat bgr[3];
|
||||
cv::cuda::GpuMat imagePreproc;
|
||||
#else
|
||||
cv::Mat bgr[3];
|
||||
cv::Mat imagePreproc;
|
||||
dnnType *input;
|
||||
#endif
|
||||
|
||||
/**
|
||||
* This method preprocess the image, before feeding it to the NN.
|
||||
*
|
||||
* @param frame original frame to adapt for inference.
|
||||
* @param bi batch index
|
||||
*/
|
||||
virtual void preprocess(cv::Mat &frame, const int bi=0) = 0;
|
||||
|
||||
/**
|
||||
* This method postprocess the output of the NN to obtain the correct
|
||||
* boundig boxes.
|
||||
*
|
||||
* @param bi batch index
|
||||
* @param mAP set to true only if all the probabilities for a bounding
|
||||
* box are needed, as in some cases for the mAP calculation
|
||||
*/
|
||||
virtual void postprocess(const int bi=0,const bool mAP=false) = 0;
|
||||
|
||||
public:
|
||||
int classes = 0;
|
||||
float confThreshold = 0.3; /*threshold on the confidence of the boxes*/
|
||||
|
||||
std::vector<tk::dnn::box> detected; /*bounding boxes in output*/
|
||||
std::vector<std::vector<tk::dnn::box>> batchDetected; /*bounding boxes in output*/
|
||||
std::vector<double> stats; /*keeps track of inference times (ms)*/
|
||||
std::vector<std::string> classesNames;
|
||||
|
||||
DetectionNN() {};
|
||||
~DetectionNN(){};
|
||||
|
||||
/**
|
||||
* Method used to initialize the class, allocate memory and compute
|
||||
* needed data.
|
||||
*
|
||||
* @param tensor_path path to the rt file of the NN.
|
||||
* @param n_classes number of classes for the given dataset.
|
||||
* @param n_batches maximum number of batches to use in inference
|
||||
* @return true if everything is correct, false otherwise.
|
||||
*/
|
||||
virtual bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1, const float conf_thresh=0.3) = 0;
|
||||
|
||||
/**
|
||||
* This method performs the whole detection of the NN.
|
||||
*
|
||||
* @param frames frames to run detection on.
|
||||
* @param cur_batches number of batches to use in inference
|
||||
* @param save_times if set to true, preprocess, inference and postprocess times
|
||||
* are saved on a csv file, otherwise not.
|
||||
* @param times pointer to the output stream where to write times
|
||||
* @param mAP set to true only if all the probabilities for a bounding
|
||||
* box are needed, as in some cases for the mAP calculation
|
||||
*/
|
||||
void update(std::vector<cv::Mat>& frames, const int cur_batches=1, bool save_times=false, std::ofstream *times=nullptr, const bool mAP=false){
|
||||
if(save_times && times==nullptr)
|
||||
FatalError("save_times set to true, but no valid ofstream given");
|
||||
if(cur_batches > nBatches)
|
||||
FatalError("A batch size greater than nBatches cannot be used");
|
||||
|
||||
originalSize.clear();
|
||||
if(TKDNN_VERBOSE) printCenteredTitle(" TENSORRT detection ", '=', 30);
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi){
|
||||
if(!frames[bi].data)
|
||||
FatalError("No image data feed to detection");
|
||||
originalSize.push_back(frames[bi].size());
|
||||
preprocess(frames[bi], bi);
|
||||
}
|
||||
TKDNN_TSTOP
|
||||
if(save_times) *times<<t_ns<<";";
|
||||
}
|
||||
|
||||
//do inference
|
||||
tk::dnn::dataDim_t dim = netRT->input_dim;
|
||||
dim.n = cur_batches;
|
||||
{
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
TKDNN_TSTART
|
||||
netRT->infer(dim, input_d);
|
||||
TKDNN_TSTOP
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
stats.push_back(t_ns);
|
||||
if(save_times) *times<<t_ns<<";";
|
||||
}
|
||||
|
||||
batchDetected.clear();
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi)
|
||||
postprocess(bi, mAP);
|
||||
TKDNN_TSTOP
|
||||
if(save_times) *times<<t_ns<<"\n";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Method to draw bounding boxes and labels on a frame.
|
||||
*
|
||||
* @param frames original frame to draw bounding box on.
|
||||
*/
|
||||
void draw(std::vector<cv::Mat>& frames) {
|
||||
tk::dnn::box b;
|
||||
int x0, w, x1, y0, h, y1;
|
||||
int objClass;
|
||||
std::string det_class;
|
||||
int baseline = 0;
|
||||
float font_scale = 0.5;
|
||||
int thickness = 2;
|
||||
|
||||
for(int bi=0; bi<frames.size(); ++bi){
|
||||
// draw dets
|
||||
for(int i=0; i<batchDetected[bi].size(); i++) {
|
||||
b = batchDetected[bi][i];
|
||||
x0 = b.x;
|
||||
x1 = b.x + b.w;
|
||||
y0 = b.y;
|
||||
y1 = b.y + b.h;
|
||||
det_class = classesNames[b.cl];
|
||||
|
||||
// draw rectangle
|
||||
cv::rectangle(frames[bi], cv::Point(x0, y0), cv::Point(x1, y1), colors[b.cl], 2);
|
||||
|
||||
// draw label
|
||||
cv::Size text_size = getTextSize(det_class, cv::FONT_HERSHEY_SIMPLEX, font_scale, thickness, &baseline);
|
||||
cv::rectangle(frames[bi], cv::Point(x0, y0), cv::Point((x0 + text_size.width - 2), (y0 - text_size.height - 2)), colors[b.cl], -1);
|
||||
cv::putText(frames[bi], det_class, cv::Point(x0, (y0 - (baseline / 2))), cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), thickness);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
}}
|
||||
|
||||
#endif /* DETECTIONNN_H*/
|
||||
@@ -0,0 +1,161 @@
|
||||
#ifndef DETECTIONNN3D_H
|
||||
#define DETECTIONNN3D_H
|
||||
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h>
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <mutex>
|
||||
#include "utils.h"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
#include "tkdnn.h"
|
||||
|
||||
// #define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib.
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
#include <opencv2/cudawarping.hpp>
|
||||
#include <opencv2/cudaarithm.hpp>
|
||||
#endif
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
class DetectionNN3D {
|
||||
|
||||
protected:
|
||||
tk::dnn::NetworkRT *netRT = nullptr;
|
||||
dnnType *input_d;
|
||||
|
||||
std::vector<cv::Size> originalSize;
|
||||
|
||||
cv::Scalar colors[256];
|
||||
|
||||
int nBatches = 1;
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
cv::cuda::GpuMat bgr[3];
|
||||
cv::cuda::GpuMat imagePreproc;
|
||||
#else
|
||||
cv::Mat bgr[3];
|
||||
cv::Mat imagePreproc;
|
||||
dnnType *input;
|
||||
#endif
|
||||
|
||||
/**
|
||||
* This method preprocess the image, before feeding it to the NN.
|
||||
*
|
||||
* @param frame original frame to adapt for inference.
|
||||
* @param bi batch index
|
||||
*/
|
||||
virtual void preprocess(cv::Mat &frame, const int bi=0) = 0;
|
||||
|
||||
/**
|
||||
* This method postprocess the output of the NN to obtain the correct
|
||||
* boundig boxes.
|
||||
*
|
||||
* @param bi batch index
|
||||
* @param mAP set to true only if all the probabilities for a bounding
|
||||
* box are needed, as in some cases for the mAP calculation
|
||||
*/
|
||||
virtual void postprocess(const int bi=0,const bool mAP=false) = 0;
|
||||
|
||||
public:
|
||||
int classes = 0;
|
||||
float confThreshold = 0.3; /*threshold on the confidence of the boxes*/
|
||||
|
||||
std::vector<tk::dnn::box3D> detected3D; /*bounding boxes in output*/
|
||||
std::vector<std::vector<tk::dnn::box3D>> batchDetected; /*bounding boxes in output*/
|
||||
std::vector<double> pre_stats, stats, post_stats, visual_stats; /*keeps track of inference times (ms)*/
|
||||
std::vector<std::string> classesNames;
|
||||
|
||||
DetectionNN3D() {};
|
||||
~DetectionNN3D(){};
|
||||
|
||||
/**
|
||||
* Method used to initialize the class, allocate memory and compute
|
||||
* needed data.
|
||||
*
|
||||
* @param tensor_path path to the rt file of the NN.
|
||||
* @param n_classes number of classes for the given dataset.
|
||||
* @param n_batches maximum number of batches to use in inference.
|
||||
* @return true if everything is correct, false otherwise.
|
||||
*/
|
||||
virtual bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1,
|
||||
const float conf_thresh=0.3, const std::vector<cv::Mat>& k_calibs=std::vector<cv::Mat>()) = 0;
|
||||
|
||||
/**
|
||||
* This method performs the whole detection of the NN.
|
||||
*
|
||||
* @param frames frames to run detection on.
|
||||
* @param cur_batches number of batches to use in inference.
|
||||
* @param save_times if set to true, preprocess, inference and postprocess times
|
||||
* are saved on a csv file, otherwise not.
|
||||
* @param times pointer to the output stream where to write times.
|
||||
* @param mAP set to true only if all the probabilities for a bounding
|
||||
* box are needed, as in some cases for the mAP calculation.
|
||||
*/
|
||||
void update(std::vector<cv::Mat>& frames, const int cur_batches=1, bool save_times=false,
|
||||
std::ofstream *times=nullptr, const bool mAP=false){
|
||||
if(save_times && times==nullptr)
|
||||
FatalError("save_times set to true, but no valid ofstream given");
|
||||
if(cur_batches > nBatches)
|
||||
FatalError("A batch size greater than nBatches cannot be used");
|
||||
|
||||
originalSize.clear();
|
||||
if(TKDNN_VERBOSE) printCenteredTitle(" TENSORRT detection ", '=', 30);
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi){
|
||||
if(!frames[bi].data)
|
||||
FatalError("No image data feed to detection");
|
||||
originalSize.push_back(frames[bi].size());
|
||||
preprocess(frames[bi], bi);
|
||||
}
|
||||
TKDNN_TSTOP
|
||||
pre_stats.push_back(t_ns);
|
||||
if(save_times) *times<<t_ns<<";";
|
||||
}
|
||||
|
||||
//do inference
|
||||
tk::dnn::dataDim_t dim = netRT->input_dim;
|
||||
dim.n = cur_batches;
|
||||
{
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
TKDNN_TSTART
|
||||
netRT->infer(dim, input_d);
|
||||
TKDNN_TSTOP
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
stats.push_back(t_ns);
|
||||
if(save_times) *times<<t_ns<<";";
|
||||
}
|
||||
|
||||
batchDetected.clear();
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi)
|
||||
postprocess(bi, mAP);
|
||||
TKDNN_TSTOP
|
||||
post_stats.push_back(t_ns);
|
||||
if(save_times) *times<<t_ns<<"\n";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Method to draw bounding boxes and labels on a frame.
|
||||
*
|
||||
* @param frames original frame to draw bounding box on.
|
||||
*/
|
||||
virtual void draw(std::vector<cv::Mat>& frames){};
|
||||
|
||||
};
|
||||
|
||||
}}
|
||||
|
||||
#endif /* DETECTIONNN3D_H*/
|
||||
@@ -0,0 +1,168 @@
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h> /* srand, rand */
|
||||
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#elif _WIN32
|
||||
#define _USE_MATH_DEFINES
|
||||
#include <math.h>
|
||||
#endif
|
||||
|
||||
#include <mutex>
|
||||
#include <Eigen/Dense>
|
||||
#include "utils.h"
|
||||
#include "tkdnn.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
/**
|
||||
*
|
||||
* @author Francesco Gatti
|
||||
*/
|
||||
class ImuOdom {
|
||||
|
||||
public:
|
||||
tk::dnn::Network *net = nullptr;
|
||||
|
||||
// Network input dim
|
||||
tk::dnn::dataDim_t dim0;
|
||||
tk::dnn::dataDim_t dim1;
|
||||
tk::dnn::dataDim_t dim2;
|
||||
|
||||
// Network output dim
|
||||
tk::dnn::dataDim_t odim0;
|
||||
tk::dnn::dataDim_t odim1;
|
||||
|
||||
// input pointers
|
||||
dnnType *i0_d, *i1_d, *i2_d;
|
||||
// output pointers
|
||||
dnnType *o0_d, *o1_d;
|
||||
|
||||
// output eigen CPU
|
||||
Eigen::MatrixXf deltaP, deltaQ;
|
||||
|
||||
Eigen::MatrixXd odomPOS, odomEULER;
|
||||
Eigen::Matrix3d odomROT;
|
||||
Eigen::Isometry3f tf = Eigen::Isometry3f::Identity();
|
||||
|
||||
ImuOdom() {}
|
||||
|
||||
virtual ~ImuOdom() {}
|
||||
|
||||
/**
|
||||
* Method used for initialize the class
|
||||
*
|
||||
* @return Success of the initialization
|
||||
*/
|
||||
bool init(std::string layers_path) {
|
||||
|
||||
dim0 = tk::dnn::dataDim_t(1, 4, 1, 100);
|
||||
dim1 = tk::dnn::dataDim_t(1, 3, 1, 100);
|
||||
dim2 = tk::dnn::dataDim_t(1, 3, 1, 100);
|
||||
|
||||
checkCuda( cudaMalloc(&i0_d, dim0.tot()*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&i1_d, dim1.tot()*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&i2_d, dim2.tot()*sizeof(dnnType)) );
|
||||
|
||||
std::string c0_bin = layers_path + "/conv1d_7.bin";
|
||||
std::string c1_bin = layers_path + "/conv1d_8.bin";
|
||||
std::string c2_bin = layers_path + "/conv1d_9.bin";
|
||||
std::string c3_bin = layers_path + "/conv1d_10.bin";
|
||||
std::string c4_bin = layers_path + "/conv1d_11.bin";
|
||||
std::string c5_bin = layers_path + "/conv1d_12.bin";
|
||||
std::string l0_bin = layers_path + "/bidirectional_3.bin";
|
||||
std::string l1_bin = layers_path + "/bidirectional_4.bin";
|
||||
std::string d0_bin = layers_path + "/dense_3.bin";
|
||||
std::string d1_bin = layers_path + "/dense_4.bin";
|
||||
|
||||
net = new tk::dnn::Network(dim0);
|
||||
tk::dnn::Input *x0 = new tk::dnn::Input (net, dim0, i0_d);
|
||||
tk::dnn::Conv2d *x0_0 = new tk::dnn::Conv2d (net, 128, 1, 11, 1, 1, 0, 0, c0_bin);
|
||||
tk::dnn::Conv2d *x0_1 = new tk::dnn::Conv2d (net, 128, 1, 11, 1, 1, 0, 0, c1_bin);
|
||||
tk::dnn::Pooling *x0_2 = new tk::dnn::Pooling(net, 1, 3, 1, 3 ,0, 0, tk::dnn::tkdnnPoolingMode_t::POOLING_MAX);
|
||||
|
||||
tk::dnn::Input *x1 = new tk::dnn::Input (net, dim1, i1_d);
|
||||
tk::dnn::Conv2d *x1_0 = new tk::dnn::Conv2d (net, 128, 1, 11, 1, 1, 0, 0, c2_bin);
|
||||
tk::dnn::Conv2d *x1_1 = new tk::dnn::Conv2d (net, 128, 1, 11, 1, 1, 0, 0, c3_bin);
|
||||
tk::dnn::Pooling *x1_2 = new tk::dnn::Pooling(net, 1, 3, 1, 3, 0, 0, tk::dnn::tkdnnPoolingMode_t::POOLING_MAX);
|
||||
|
||||
tk::dnn::Input *x2 = new tk::dnn::Input (net, dim2, i2_d);
|
||||
tk::dnn::Conv2d *x2_0 = new tk::dnn::Conv2d (net, 128, 1, 11, 1, 1, 0, 0, c4_bin);
|
||||
tk::dnn::Conv2d *x2_1 = new tk::dnn::Conv2d (net, 128, 1, 11, 1, 1, 0, 0, c5_bin);
|
||||
tk::dnn::Pooling *x2_2 = new tk::dnn::Pooling(net, 1, 3, 1, 3, 0, 0, tk::dnn::tkdnnPoolingMode_t::POOLING_MAX);
|
||||
|
||||
tk::dnn::Layer *concat_l[3] = { x0_2, x1_2, x2_2 };
|
||||
tk::dnn::Route *concat = new tk::dnn::Route(net, concat_l, 3);
|
||||
|
||||
tk::dnn::LSTM *lstm0 = new tk::dnn::LSTM(net, 128, true, l0_bin);
|
||||
tk::dnn::LSTM *lstm1 = new tk::dnn::LSTM(net, 128, false, l1_bin);
|
||||
|
||||
tk::dnn::Dense *d0 = new tk::dnn::Dense(net, 3, d0_bin);
|
||||
|
||||
tk::dnn::Layer *lstm1_l[1] = { lstm1 };
|
||||
tk::dnn::Route *lstm1_link = new tk::dnn::Route(net, lstm1_l, 1);
|
||||
tk::dnn::Dense *d1 = new tk::dnn::Dense(net, 4, d1_bin);
|
||||
|
||||
net->print();
|
||||
|
||||
// output data
|
||||
o0_d = d0->dstData;
|
||||
o1_d = d1->dstData;
|
||||
odim0 = d0->output_dim;
|
||||
odim1 = d1->output_dim;
|
||||
|
||||
deltaP.resize(odim0.tot(), 1);
|
||||
deltaQ.resize(odim1.tot(), 1);
|
||||
|
||||
odomPOS = Eigen::MatrixXd::Zero(3, 1);
|
||||
odomROT = Eigen::MatrixXd::Identity(3, 3);
|
||||
odomEULER = Eigen::MatrixXd::Zero(3, 1);
|
||||
return true;
|
||||
}
|
||||
|
||||
void close() {
|
||||
// TODO: dealloc :)
|
||||
}
|
||||
|
||||
void update(dnnType *x0, dnnType *x1, dnnType *x2) {
|
||||
|
||||
checkCuda( cudaMemcpy(i0_d, x0, dim0.tot()*sizeof(dnnType), cudaMemcpyHostToDevice) );
|
||||
checkCuda( cudaMemcpy(i1_d, x1, dim1.tot()*sizeof(dnnType), cudaMemcpyHostToDevice) );
|
||||
checkCuda( cudaMemcpy(i2_d, x2, dim2.tot()*sizeof(dnnType), cudaMemcpyHostToDevice) );
|
||||
|
||||
// Inference
|
||||
tk::dnn::dataDim_t dim;
|
||||
net->infer(dim, nullptr);
|
||||
|
||||
checkCuda( cudaMemcpy(deltaP.data(), o0_d, odim0.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(deltaQ.data(), o1_d, odim1.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) );
|
||||
|
||||
// compute odom
|
||||
Eigen::Quaterniond q;
|
||||
q.w() = deltaQ(0);
|
||||
q.x() = deltaQ(1);
|
||||
q.y() = deltaQ(2);
|
||||
q.z() = deltaQ(3);
|
||||
odomPOS = odomPOS + odomROT*deltaP.cast<double>(); // V1
|
||||
//odomPOS = odomPOS + deltaP.cast<double>(); // V2
|
||||
odomROT = odomROT * q.normalized().toRotationMatrix();
|
||||
|
||||
// compute Euler
|
||||
auto newEULER = odomROT.eulerAngles(0, 1, 2);
|
||||
for(int i=0; i<3; i++) {
|
||||
while( fabs(newEULER(i) - odomEULER(i)) > M_PI_2 ) {
|
||||
newEULER(i) += newEULER(i) - odomEULER(i) > 0 ? -M_PI : +M_PI;
|
||||
//std::cout<<newEULER(i)<<" "<<odomEULER(i)<<"\n";
|
||||
}
|
||||
}
|
||||
odomEULER = newEULER;
|
||||
|
||||
// compose tf
|
||||
tf.matrix().block(0, 0, 3, 3) = odomROT.cast<float>();
|
||||
tf.matrix().block(0, 3, 3, 1) = odomPOS.cast<float>();
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
}}
|
||||
@@ -0,0 +1,72 @@
|
||||
#ifndef INT8BATCHSTREAM_H
|
||||
#define INT8BATCHSTREAM_H
|
||||
|
||||
#include <vector>
|
||||
#include <assert.h>
|
||||
#include <algorithm>
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
#include <iostream>
|
||||
#include <string>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h>
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <mutex>
|
||||
|
||||
#include "NvInfer.h"
|
||||
#include "utils.h"
|
||||
#include "tkdnn.h"
|
||||
|
||||
/*
|
||||
* BatchStream implements the stream for the INT8 calibrator.
|
||||
* It reads the two files .txt with the list of image file names
|
||||
* and the list of label file names.
|
||||
* It then iterates on images and labels.
|
||||
*/
|
||||
class BatchStream {
|
||||
public:
|
||||
BatchStream(tk::dnn::dataDim_t dim, int batchSize, int maxBatches, const std::string& fileimglist, const std::string& filelabellist);
|
||||
virtual ~BatchStream() { }
|
||||
void reset(int firstBatch);
|
||||
bool next();
|
||||
void skip(int skipCount);
|
||||
float *getBatch() { return mBatch.data(); }
|
||||
float *getLabels() { return mLabels.data(); }
|
||||
int getBatchesRead() const { return mBatchCount; }
|
||||
int getBatchSize() const { return mBatchSize; }
|
||||
nvinfer1::Dims4 getDims() const { return mDims; }
|
||||
float* getFileBatch() { return &mFileBatch[0]; }
|
||||
float* getFileLabels() { return &mFileLabels[0]; }
|
||||
void readInListFile(const std::string& dataFilePath, std::vector<std::string>& mListIn);
|
||||
void readCVimage(std::string inputFileName, std::vector<float>& res, bool fixshape = true);
|
||||
void readLabels(std::string inputFileName ,std::vector<float>& ris);
|
||||
bool update();
|
||||
|
||||
private:
|
||||
int mBatchSize{ 0 };
|
||||
int mMaxBatches{ 0 };
|
||||
int mBatchCount{ 0 };
|
||||
int mFileCount{ 0 };
|
||||
int mFileBatchPos{ 0 };
|
||||
int mImageSize{ 0 };
|
||||
|
||||
nvinfer1::Dims4 mDims;
|
||||
std::vector<float> mBatch;
|
||||
std::vector<float> mLabels;
|
||||
std::vector<float> mFileBatch;
|
||||
std::vector<float> mFileLabels;
|
||||
|
||||
int mHeight;
|
||||
int mWidth;
|
||||
std::string mFileImgList;
|
||||
std::vector<std::string> mListImg;
|
||||
std::string mFileLabelList;
|
||||
std::vector<std::string> mListLabel;
|
||||
};
|
||||
|
||||
#endif //INT8BATCHSTREAM
|
||||
@@ -0,0 +1,49 @@
|
||||
#ifndef INT8CALIBRATOR_H
|
||||
#define INT8CALIBRATOR_H
|
||||
|
||||
#include <vector>
|
||||
#include <assert.h>
|
||||
#include <algorithm>
|
||||
#include <iterator>
|
||||
#include <stdint.h>
|
||||
#include <iostream>
|
||||
#include <string>
|
||||
#include "NvInfer.h"
|
||||
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
|
||||
#include "Int8BatchStream.h"
|
||||
|
||||
#include "tkdnn.h"
|
||||
#include "utils.h"
|
||||
|
||||
/*
|
||||
* Int8EntropyCalibrator implements the INT8 calibrator to achieve the
|
||||
* INT8 quantization. It uses a BatchStream stream to scroll through
|
||||
* images data. It also implements the calibration cache, a way to
|
||||
* save the calibration process results to reduce the running time:
|
||||
* the calibration process takes a long time.
|
||||
*/
|
||||
class Int8EntropyCalibrator : public nvinfer1::IInt8EntropyCalibrator {
|
||||
public:
|
||||
Int8EntropyCalibrator(BatchStream& stream, int firstBatch, const std::string& calibTableFilePath,
|
||||
const std::string& inputBlobName, bool readCache = true);
|
||||
virtual ~Int8EntropyCalibrator() { checkCuda(cudaFree(mDeviceInput)); }
|
||||
int getBatchSize() const NOEXCEPT override { return mStream.getBatchSize(); }
|
||||
bool getBatch(void* bindings[], const char* names[], int nbBindings) NOEXCEPT override;
|
||||
const void* readCalibrationCache(size_t& length) NOEXCEPT override;
|
||||
void writeCalibrationCache(const void* cache, size_t length) NOEXCEPT override;
|
||||
|
||||
private:
|
||||
BatchStream mStream;
|
||||
const std::string mCalibTableFilePath{ nullptr };
|
||||
const std::string mInputBlobName;
|
||||
bool mReadCache{ true };
|
||||
|
||||
size_t mInputCount;
|
||||
void* mDeviceInput{ nullptr };
|
||||
std::vector<char> mCalibrationCache;
|
||||
};
|
||||
|
||||
#endif //INT8CALIBRATOR_H
|
||||
+389
-54
@@ -9,10 +9,20 @@
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
enum layerType_t {
|
||||
LAYER_INPUT,
|
||||
LAYER_DENSE,
|
||||
LAYER_CONV2D,
|
||||
LAYER_DECONV2D,
|
||||
LAYER_DEFORMCONV2D,
|
||||
LAYER_LSTM,
|
||||
LAYER_ACTIVATION,
|
||||
LAYER_ACTIVATION_CRELU,
|
||||
LAYER_ACTIVATION_LEAKY,
|
||||
LAYER_ACTIVATION_MISH,
|
||||
LAYER_ACTIVATION_LOGISTIC,
|
||||
LAYER_FLATTEN,
|
||||
LAYER_RESHAPE,
|
||||
LAYER_RESIZE,
|
||||
LAYER_MULADD,
|
||||
LAYER_POOLING,
|
||||
LAYER_SOFTMAX,
|
||||
@@ -21,7 +31,8 @@ enum layerType_t {
|
||||
LAYER_SHORTCUT,
|
||||
LAYER_UPSAMPLE,
|
||||
LAYER_REGION,
|
||||
LAYER_YOLO
|
||||
LAYER_YOLO,
|
||||
LAYER_PADDING,
|
||||
};
|
||||
|
||||
#define TKDNN_BN_MIN_EPSILON 1e-5
|
||||
@@ -40,27 +51,45 @@ public:
|
||||
std::cout<<"No infer action for this layer\n";
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void setFinal() { this->final = true; }
|
||||
dataDim_t input_dim, output_dim;
|
||||
dnnType *dstData; //where results will be putted
|
||||
dnnType *dstData = nullptr; //where results will be putted
|
||||
|
||||
int id = 0;
|
||||
bool final; //if the layer is the final one
|
||||
unsigned int n_params = 0;
|
||||
unsigned int feature_map_size = 0;
|
||||
long unsigned MACC = 0;
|
||||
|
||||
|
||||
std::string getLayerName() {
|
||||
layerType_t type = getLayerType();
|
||||
switch(type) {
|
||||
case LAYER_DENSE: return "Dense";
|
||||
case LAYER_CONV2D: return "Conv2d";
|
||||
case LAYER_ACTIVATION: return "Activation";
|
||||
case LAYER_FLATTEN: return "Flatten";
|
||||
case LAYER_MULADD: return "MulAdd";
|
||||
case LAYER_POOLING: return "Pooling";
|
||||
case LAYER_SOFTMAX: return "Softmax";
|
||||
case LAYER_ROUTE: return "Route";
|
||||
case LAYER_REORG: return "Reorg";
|
||||
case LAYER_SHORTCUT: return "Shortcut";
|
||||
case LAYER_UPSAMPLE: return "Upsample";
|
||||
case LAYER_REGION: return "Region";
|
||||
case LAYER_YOLO: return "Yolo";
|
||||
default: return "unknown";
|
||||
case LAYER_INPUT: return "Input";
|
||||
case LAYER_DENSE: return "Dense";
|
||||
case LAYER_CONV2D: return "Conv2d";
|
||||
case LAYER_DECONV2D: return "DeConv2d";
|
||||
case LAYER_DEFORMCONV2D: return "DeformConv2d";
|
||||
case LAYER_LSTM: return "LSTM";
|
||||
case LAYER_ACTIVATION: return "Activation";
|
||||
case LAYER_ACTIVATION_CRELU: return "ActivationCReLU";
|
||||
case LAYER_ACTIVATION_LEAKY: return "ActivationLeaky";
|
||||
case LAYER_ACTIVATION_MISH: return "ActivationMish";
|
||||
case LAYER_ACTIVATION_LOGISTIC: return "ActivationLogistic";
|
||||
case LAYER_FLATTEN: return "Flatten";
|
||||
case LAYER_RESHAPE: return "Reshape";
|
||||
case LAYER_RESIZE: return "Resize";
|
||||
case LAYER_MULADD: return "MulAdd";
|
||||
case LAYER_POOLING: return "Pooling";
|
||||
case LAYER_SOFTMAX: return "Softmax";
|
||||
case LAYER_ROUTE: return "Route";
|
||||
case LAYER_REORG: return "Reorg";
|
||||
case LAYER_SHORTCUT: return "Shortcut";
|
||||
case LAYER_UPSAMPLE: return "Upsample";
|
||||
case LAYER_REGION: return "Region";
|
||||
case LAYER_YOLO: return "Yolo";
|
||||
case LAYER_PADDING: return "Padding";
|
||||
default: return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
@@ -77,8 +106,8 @@ protected:
|
||||
class LayerWgs : public Layer {
|
||||
|
||||
public:
|
||||
LayerWgs(Network *net, int inputs, int outputs, int kh, int kw, int kt,
|
||||
std::string fname_weights, bool batchnorm = false);
|
||||
LayerWgs(Network *net, int inputs, int outputs, int kh, int kw, int kt,
|
||||
std::string fname_weights, bool batchnorm = false, bool additional_bias = false, bool deConv = false, int groups = 1);
|
||||
virtual ~LayerWgs();
|
||||
|
||||
int inputs, outputs;
|
||||
@@ -87,21 +116,89 @@ public:
|
||||
dnnType *data_h, *data_d;
|
||||
dnnType *bias_h, *bias_d;
|
||||
|
||||
// additional bias for DCN
|
||||
bool additional_bias;
|
||||
dnnType *bias2_h = nullptr, *bias2_d = nullptr;
|
||||
|
||||
//batchnorm
|
||||
bool batchnorm;
|
||||
dnnType *power_h;
|
||||
dnnType *scales_h, *scales_d;
|
||||
dnnType *mean_h, *mean_d;
|
||||
dnnType *variance_h, *variance_d;
|
||||
dnnType *power_h = nullptr;
|
||||
dnnType *scales_h = nullptr, *scales_d = nullptr;
|
||||
dnnType *mean_h = nullptr, *mean_d = nullptr;
|
||||
dnnType *variance_h = nullptr, *variance_d = nullptr;
|
||||
|
||||
//fp16
|
||||
__half *data16_h, *bias16_h;
|
||||
__half *data16_d, *bias16_d;
|
||||
__half *data16_h = nullptr, *bias16_h = nullptr;
|
||||
__half *data16_d = nullptr, *bias16_d = nullptr;
|
||||
__half *bias216_h = nullptr, *bias216_d = nullptr;
|
||||
|
||||
__half *power16_h, *power16_d;
|
||||
__half *scales16_h, *scales16_d;
|
||||
__half *mean16_h, *mean16_d;
|
||||
__half *variance16_h, *variance16_d;
|
||||
__half *power16_h = nullptr, *power16_d = nullptr;
|
||||
__half *scales16_h = nullptr, *scales16_d = nullptr;
|
||||
__half *mean16_h = nullptr, *mean16_d = nullptr;
|
||||
__half *variance16_h = nullptr, *variance16_d = nullptr;
|
||||
|
||||
void releaseHost(bool release32 = true, bool release16 = true) {
|
||||
if(release32) {
|
||||
if( data_h != nullptr) { delete [] data_h; data_h = nullptr; }
|
||||
if( bias_h != nullptr) { delete [] bias_h; bias_h = nullptr; }
|
||||
if( bias2_h != nullptr) { delete [] bias2_h; bias2_h = nullptr; }
|
||||
if( scales_h != nullptr) { delete [] scales_h; scales_h = nullptr; }
|
||||
if( mean_h != nullptr) { delete [] mean_h; mean_h = nullptr; }
|
||||
if(variance_h != nullptr) { delete [] variance_h; variance_h = nullptr; }
|
||||
if( power_h != nullptr) { delete [] power_h; power_h = nullptr; }
|
||||
}
|
||||
if(net->fp16 && release16) {
|
||||
if( data16_h != nullptr) { delete [] data16_h; data16_h = nullptr; }
|
||||
if( bias16_h != nullptr) { delete [] bias16_h; bias16_h = nullptr; }
|
||||
if( bias216_h != nullptr) { delete [] bias216_h; bias216_h = nullptr; }
|
||||
if( scales16_h != nullptr) { delete [] scales16_h; scales16_h = nullptr; }
|
||||
if( mean16_h != nullptr) { delete [] mean16_h; mean16_h = nullptr; }
|
||||
if(variance16_h != nullptr) { delete [] variance16_h; variance16_h = nullptr; }
|
||||
if( power16_h != nullptr) { delete [] power16_h; power16_h = nullptr; }
|
||||
|
||||
}
|
||||
}
|
||||
void releaseDevice(bool release32 = true, bool release16 = true) {
|
||||
if(release32) {
|
||||
if( data_d != nullptr) { cudaFree( data_d); data_d = nullptr; }
|
||||
if( bias_d != nullptr) { cudaFree( bias_d); bias_d = nullptr; }
|
||||
if( bias2_d != nullptr) { cudaFree( bias2_d); bias2_d = nullptr; }
|
||||
if( scales_d != nullptr) { cudaFree( scales_d); scales_d = nullptr; }
|
||||
if( mean_d != nullptr) { cudaFree( mean_d); mean_d = nullptr; }
|
||||
if(variance_d != nullptr) { cudaFree(variance_d); variance_d = nullptr; }
|
||||
}
|
||||
if(net->fp16 && release16) {
|
||||
if( data16_d != nullptr) { cudaFree( data16_d); data16_d = nullptr; }
|
||||
if( bias16_d != nullptr) { cudaFree( bias16_d); bias16_d = nullptr; }
|
||||
if( bias216_d != nullptr) { cudaFree( bias216_d); bias216_d = nullptr; }
|
||||
if( scales16_d != nullptr) { cudaFree( scales16_d); scales16_d = nullptr; }
|
||||
if( mean16_d != nullptr) { cudaFree( mean16_d); mean16_d = nullptr; }
|
||||
if(variance16_d != nullptr) { cudaFree(variance16_d); variance16_d = nullptr; }
|
||||
if( power16_d != nullptr) { cudaFree( power16_d); power16_d = nullptr; }
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
Input layer (it doesn't need weights)
|
||||
*/
|
||||
class Input : public Layer {
|
||||
|
||||
public:
|
||||
|
||||
Input(Network *net, dataDim_t &dim, dnnType* srcData) : Layer(net) {
|
||||
input_dim = dim;
|
||||
output_dim = dim;
|
||||
dstData = srcData;
|
||||
}
|
||||
virtual ~Input() {}
|
||||
virtual layerType_t getLayerType() { return LAYER_INPUT; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData) {
|
||||
dim = output_dim;
|
||||
return dstData;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -120,24 +217,39 @@ public:
|
||||
|
||||
|
||||
/**
|
||||
Avaible activation functions
|
||||
Available activation functions
|
||||
*/
|
||||
typedef enum {
|
||||
ACTIVATION_ELU = 100,
|
||||
ACTIVATION_LEAKY = 101
|
||||
ACTIVATION_LEAKY = 101,
|
||||
ACTIVATION_MISH = 102,
|
||||
ACTIVATION_LOGISTIC = 103
|
||||
} tkdnnActivationMode_t;
|
||||
|
||||
/**
|
||||
Activation layer (it doesnt need weigths)
|
||||
Activation layer (it doesn't need weights)
|
||||
*/
|
||||
class Activation : public Layer {
|
||||
|
||||
public:
|
||||
int act_mode;
|
||||
float ceiling;
|
||||
float slope;
|
||||
|
||||
Activation(Network *net, int act_mode);
|
||||
Activation(Network *net, int act_mode, const float ceiling=0.0, const float slope=0.1);
|
||||
virtual ~Activation();
|
||||
virtual layerType_t getLayerType() { return LAYER_ACTIVATION; };
|
||||
virtual layerType_t getLayerType() {
|
||||
if(act_mode == CUDNN_ACTIVATION_CLIPPED_RELU)
|
||||
return LAYER_ACTIVATION_CRELU;
|
||||
else if (act_mode == ACTIVATION_LEAKY)
|
||||
return LAYER_ACTIVATION_LEAKY;
|
||||
else if (act_mode == ACTIVATION_MISH)
|
||||
return LAYER_ACTIVATION_MISH;
|
||||
else if (act_mode == ACTIVATION_LOGISTIC)
|
||||
return LAYER_ACTIVATION_LOGISTIC;
|
||||
else
|
||||
return LAYER_ACTIVATION;
|
||||
};
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
|
||||
@@ -148,30 +260,158 @@ protected:
|
||||
|
||||
/**
|
||||
Convolutional 2D layer
|
||||
|
||||
WEIGHTS shape: OUTCH, INCH, KH, KW ...
|
||||
BIAS shape: OUTCH
|
||||
|
||||
with BATCHNORM:
|
||||
scales: OUTCH
|
||||
means: OUTCH
|
||||
variance: OUTCH
|
||||
*/
|
||||
class Conv2d : public LayerWgs {
|
||||
|
||||
public:
|
||||
Conv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
||||
int strideH, int strideW, int paddingH, int paddingW,
|
||||
std::string fname_weights, bool batchnorm = false);
|
||||
std::string fname_weights, bool batchnorm = false, bool deConv = false, int groups = 1, bool additional_bias=false);
|
||||
virtual ~Conv2d();
|
||||
virtual layerType_t getLayerType() { return LAYER_CONV2D; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
|
||||
int kernelH, kernelW, strideH, strideW, paddingH, paddingW;
|
||||
bool deConv, additional_bias;
|
||||
int groups;
|
||||
|
||||
protected:
|
||||
cudnnFilterDescriptor_t filterDesc;
|
||||
cudnnConvolutionDescriptor_t convDesc;
|
||||
cudnnConvolutionFwdAlgo_t algo;
|
||||
cudnnConvolutionFwdAlgoPerf_t algo;
|
||||
cudnnConvolutionBwdDataAlgoPerf_t bwAlgo;
|
||||
cudnnTensorDescriptor_t biasTensorDesc;
|
||||
|
||||
void initCUDNN(bool back = false);
|
||||
void inferCUDNN(dnnType* srcData, bool back = false);
|
||||
void* workSpace;
|
||||
size_t ws_sizeInBytes;
|
||||
};
|
||||
|
||||
/**
|
||||
Bidirectional LSTM layer
|
||||
ONLY BIDIRECTIONAL (TODO: more configurable)
|
||||
currently implemented as 2 inferences: forward and backward (TODO: only 1 cudnn inference)
|
||||
|
||||
implementation info:
|
||||
https://github.com/jiangnanhugo/seq2seq_cuda/blob/e4dbdcfa0517c972bfd4beea9f11a5233954093c/src/rnn.cpp
|
||||
https://github.com/Jeffery-Song/mxnet-test/blob/aab666faad44011f7a67b527b5f6c960367d0422/src/operator/cudnn_rnn-inl.h
|
||||
https://stackoverflow.com/a/38737941
|
||||
https://colah.github.io/posts/2015-08-Understanding-LSTMs/
|
||||
|
||||
PARAMS (numlayers*2):
|
||||
layer0:
|
||||
( INCH, ? ) ???
|
||||
( HIDDEN, ? ) ???
|
||||
( HIDDEN * 8 ) ???
|
||||
layer2:
|
||||
( INCH, ? ) ???
|
||||
( HIDDEN, ? ) ???
|
||||
( HIDDEN * 8 ) ???
|
||||
|
||||
OUTPUT shape:
|
||||
(N, C, 1, W) ---> LSTM(HIDDEN, returnSeq=True) ---> (N, 2*HIDDEN, 1, W) # W is seqLength
|
||||
(N, C, 1, W) ---> LSTM(HIDDEN, returnSeq=False) ---> (N, 2*HIDDEN, 1, 1)
|
||||
*/
|
||||
class LSTM : public Layer {
|
||||
|
||||
public:
|
||||
LSTM(Network *net, int hiddensize, bool returnSeq, std::string fname_weights);
|
||||
virtual ~LSTM();
|
||||
virtual layerType_t getLayerType() { return LAYER_LSTM; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
|
||||
const bool bidirectional = true; /**> is the net bidir */
|
||||
bool returnSeq = false; /**> if false return only the result of last timestamp */
|
||||
int stateSize = 0; /**> number of hidden states */
|
||||
int seqLen = 0; /**> number of timestamp */
|
||||
int numLayers = 1; /**> number of internal layers */
|
||||
|
||||
protected:
|
||||
cudnnRNNDescriptor_t rnnDesc;
|
||||
cudnnDropoutDescriptor_t dropoutDesc;
|
||||
dnnType *dropout_states_, *work_space_;
|
||||
|
||||
size_t workspace_byte_, dropout_byte_;
|
||||
int workspace_size_, dropout_size_;
|
||||
|
||||
std::vector<cudnnTensorDescriptor_t> x_desc_vec_, y_desc_vec_;
|
||||
cudnnTensorDescriptor_t hx_desc_, cx_desc_;
|
||||
cudnnTensorDescriptor_t hy_desc_, cy_desc_;
|
||||
dnnType *hx_ptr, *cx_ptr, *hy_ptr, *cy_ptr;
|
||||
int stateDataDim;
|
||||
|
||||
cudnnFilterDescriptor_t w_desc_;
|
||||
dnnType *w_ptr;
|
||||
dnnType *w_h;
|
||||
dnnType *wf_ptr, *wb_ptr; // params pointer forward and backward layer
|
||||
|
||||
// used during inference
|
||||
dataDim_t one_output_dim; // output dim of as single inference
|
||||
dnnType *srcF, *srcB; // input of single inference
|
||||
dnnType *dstF, *dstB_NR, *dstB; // output of single inference, dstB_NR = dstB not reversed
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
Convolutional 2D layer
|
||||
*/
|
||||
class DeConv2d : public Conv2d {
|
||||
|
||||
public:
|
||||
DeConv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
||||
int strideH, int strideW, int paddingH, int paddingW,
|
||||
std::string fname_weights, bool batchnorm = false, int groups = 1) :
|
||||
Conv2d(net, out_ch, kernelH, kernelW, strideH, strideW, paddingH, paddingW, fname_weights, batchnorm, true, groups) {}
|
||||
virtual ~DeConv2d() {}
|
||||
virtual layerType_t getLayerType() { return LAYER_DECONV2D; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
Deformable Convolutional 2d layer
|
||||
*/
|
||||
class DeformConv2d : public LayerWgs {
|
||||
|
||||
public:
|
||||
DeformConv2d( Network *net, int out_ch, int deformable_group, int kernelH, int kernelW,
|
||||
int strideH, int strideW, int paddingH, int paddingW,
|
||||
std::string d_fname_weights, std::string fname_weights, bool batchnorm);
|
||||
virtual ~DeformConv2d();
|
||||
virtual layerType_t getLayerType() { return LAYER_DEFORMCONV2D; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
tk::dnn::Conv2d *preconv;
|
||||
int out_ch;
|
||||
int deformableGroup;
|
||||
int kernelH, kernelW, strideH, strideW, paddingH, paddingW;
|
||||
dnnType *ones_d1;
|
||||
dnnType *ones_d2;
|
||||
int chunk_dim;
|
||||
dnnType *offset, *mask;
|
||||
dnnType *output_conv;
|
||||
|
||||
cublasStatus_t stat;
|
||||
cublasHandle_t handle;
|
||||
|
||||
protected:
|
||||
|
||||
cudnnTensorDescriptor_t biasTensorDesc;
|
||||
void initCUDNN();
|
||||
|
||||
};
|
||||
|
||||
/**
|
||||
Flatten layer
|
||||
@@ -185,8 +425,42 @@ public:
|
||||
virtual layerType_t getLayerType() { return LAYER_FLATTEN; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
|
||||
int c, h, w, rows, cols;
|
||||
};
|
||||
|
||||
/**
|
||||
Reshape layer
|
||||
*/
|
||||
class Reshape : public Layer {
|
||||
|
||||
public:
|
||||
Reshape(Network *net, dataDim_t new_dim);
|
||||
virtual ~Reshape();
|
||||
virtual layerType_t getLayerType() { return LAYER_RESHAPE; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
int n,c,h,w;
|
||||
|
||||
};
|
||||
|
||||
enum ResizeMode_t { NEAREST= 0,
|
||||
LINEAR= 1};
|
||||
|
||||
/**
|
||||
Resize layer
|
||||
*/
|
||||
class Resize : public Layer {
|
||||
|
||||
public:
|
||||
Resize(Network *net, int scale_c, int scale_h, int scale_w, bool fixed=false, ResizeMode_t mode=NEAREST);
|
||||
virtual ~Resize();
|
||||
virtual layerType_t getLayerType() { return LAYER_RESIZE; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
|
||||
ResizeMode_t mode;
|
||||
};
|
||||
|
||||
/**
|
||||
MulAdd layer
|
||||
@@ -201,7 +475,6 @@ public:
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
|
||||
protected:
|
||||
dnnType mul, add;
|
||||
dnnType *add_vector;
|
||||
};
|
||||
@@ -209,17 +482,18 @@ protected:
|
||||
|
||||
|
||||
/**
|
||||
Avaible pooling functions (padding on tkDNN is not supported)
|
||||
Available pooling functions (padding on tkDNN is not supported)
|
||||
*/
|
||||
typedef enum {
|
||||
POOLING_MAX = 0,
|
||||
POOLING_AVERAGE = 1, // count for average includes padded values
|
||||
POOLING_AVERAGE_EXCLUDE_PADDING = 2 // count for average does not include padded values
|
||||
POOLING_AVERAGE = 1, // count for average includes padded values
|
||||
POOLING_AVERAGE_EXCLUDE_PADDING = 2, // count for average does not include padded values
|
||||
POOLING_MAX_FIXEDSIZE = 100 // max pool darknet fashion
|
||||
} tkdnnPoolingMode_t;
|
||||
|
||||
/**
|
||||
Pooling layer
|
||||
currenty supported only 2d pooing (also on 3d input)
|
||||
currently supported only 2d pooing (also on 3d input)
|
||||
*/
|
||||
class Pooling : public Layer {
|
||||
|
||||
@@ -227,9 +501,14 @@ public:
|
||||
int winH, winW;
|
||||
int strideH, strideW;
|
||||
int paddingH, paddingW;
|
||||
int padding;
|
||||
bool size;
|
||||
tkdnnPoolingMode_t pool_mode;
|
||||
|
||||
Pooling(Network *net, int winH, int winW,
|
||||
int strideH, int strideW, tkdnnPoolingMode_t pool_mode);
|
||||
int strideH, int strideW,
|
||||
int paddingH, int paddingW,
|
||||
tkdnnPoolingMode_t pool_mode);
|
||||
virtual ~Pooling();
|
||||
virtual layerType_t getLayerType() { return LAYER_POOLING; };
|
||||
|
||||
@@ -238,22 +517,49 @@ public:
|
||||
protected:
|
||||
|
||||
cudnnPoolingDescriptor_t poolingDesc;
|
||||
tkdnnPoolingMode_t pool_mode;
|
||||
dnnType *tmpInputData, *tmpOutputData;
|
||||
bool poolOn3d;
|
||||
};
|
||||
|
||||
/**
|
||||
* Padding Layers
|
||||
* tkDNN supports reflection,constant and zero padding
|
||||
*/
|
||||
|
||||
typedef enum {
|
||||
PADDING_MODE_CONSTANT = 0,
|
||||
PADDING_MODE_ZERO = 1,
|
||||
PADDING_MODE_REFLECTION = 2
|
||||
} tkdnnPaddingMode_t;
|
||||
|
||||
class Padding : public Layer {
|
||||
public:
|
||||
Padding(Network *net,int32_t pad_h,int32_t pad_w,tkdnnPaddingMode_t padding_mode,float constant = 0.0);
|
||||
virtual ~Padding();
|
||||
virtual layerType_t getLayerType(){return LAYER_PADDING ;};
|
||||
virtual dnnType* infer(dataDim_t& dim,dnnType* srcData);
|
||||
int32_t paddingH,paddingW;
|
||||
tkdnnPaddingMode_t padding_mode;
|
||||
float constant;
|
||||
|
||||
};
|
||||
|
||||
|
||||
|
||||
/**
|
||||
Softmax layer
|
||||
*/
|
||||
|
||||
class Softmax : public Layer {
|
||||
|
||||
public:
|
||||
Softmax(Network *net);
|
||||
Softmax(Network *net, const tk::dnn::dataDim_t* dim=nullptr, const cudnnSoftmaxMode_t mode=CUDNN_SOFTMAX_MODE_CHANNEL);
|
||||
virtual ~Softmax();
|
||||
virtual layerType_t getLayerType() { return LAYER_SOFTMAX; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
dataDim_t dim;
|
||||
cudnnSoftmaxMode_t mode;
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -263,21 +569,24 @@ public:
|
||||
class Route : public Layer {
|
||||
|
||||
public:
|
||||
Route(Network *net, Layer **layers, int layers_n);
|
||||
Route(Network *net, Layer **layers, int layers_n, int groups = 1, int group_id = 0);
|
||||
virtual ~Route();
|
||||
virtual layerType_t getLayerType() { return LAYER_ROUTE; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
|
||||
public:
|
||||
Layer **layers; //ids of layers to be merged
|
||||
static const int MAX_LAYERS = 32;
|
||||
Layer *layers[MAX_LAYERS]; //ids of layers to be merged
|
||||
int layers_n; //number of layers
|
||||
int groups;
|
||||
int group_id;
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
Reorg layer
|
||||
Mantain same dimension but change C*H*W distribution
|
||||
Maintains same dimension but change C*H*W distribution
|
||||
*/
|
||||
class Reorg : public Layer {
|
||||
|
||||
@@ -298,19 +607,22 @@ public:
|
||||
class Shortcut : public Layer {
|
||||
|
||||
public:
|
||||
Shortcut(Network *net, Layer *backLayer);
|
||||
Shortcut(Network *net, Layer *backLayer, bool mul=false);
|
||||
virtual ~Shortcut();
|
||||
virtual layerType_t getLayerType() { return LAYER_SHORTCUT; };
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
|
||||
int c,h,w;
|
||||
|
||||
public:
|
||||
Layer *backLayer;
|
||||
bool mul = false;
|
||||
};
|
||||
|
||||
/**
|
||||
Upsample layer
|
||||
Mantain same dimension but change C*H*W distribution
|
||||
Maintains same dimension but change C*H*W distribution
|
||||
*/
|
||||
class Upsample : public Layer {
|
||||
|
||||
@@ -323,18 +635,35 @@ public:
|
||||
|
||||
int stride;
|
||||
bool reverse;
|
||||
int c,h,w;
|
||||
};
|
||||
|
||||
struct box {
|
||||
int cl;
|
||||
float x, y, w, h;
|
||||
float prob;
|
||||
std::vector<float> probs;
|
||||
|
||||
void print()
|
||||
{
|
||||
std::cout<<"x: "<<x<<"\ty: "<<y<<"\tw: "<<w<<"\th: "<<h<<"\tcl: "<<cl<<"\tprob: "<<prob<<std::endl;
|
||||
}
|
||||
};
|
||||
struct sortable_bbox {
|
||||
int index;
|
||||
int cl;
|
||||
float **probs;
|
||||
};
|
||||
struct box3D {
|
||||
int cl;
|
||||
std::vector<float> corners;
|
||||
float prob;
|
||||
|
||||
void print()
|
||||
{
|
||||
std::cout<<"\tcl: "<<cl<<"\tprob: "<<prob<<"\tshape corners: "<<corners.size()<<std::endl;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
Yolo3 layer
|
||||
@@ -355,23 +684,28 @@ public:
|
||||
int sort_class;
|
||||
};
|
||||
|
||||
Yolo(Network *net, int classes, int num, std::string fname_weights);
|
||||
enum nmsKind_t {GREEDY_NMS=0, DIOU_NMS=1};
|
||||
|
||||
Yolo(Network *net, int classes, int num, std::string fname_weights,int n_masks=3, float scale_xy=1, double nms_thresh=0.45, nmsKind_t nsm_kind=GREEDY_NMS, int new_coords=0);
|
||||
virtual ~Yolo();
|
||||
virtual layerType_t getLayerType() { return LAYER_YOLO; };
|
||||
|
||||
int classes, num;
|
||||
int classes, num, n_masks, new_coords;
|
||||
dnnType *mask_h, *mask_d; //anchors
|
||||
dnnType *bias_h, *bias_d; //anchors
|
||||
float scaleXY;
|
||||
double nms_thresh;
|
||||
nmsKind_t nsm_kind;
|
||||
std::vector<std::string> classesNames;
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
int computeDetections(Yolo::detection *dets, int &ndets, int netw, int neth, float thresh);
|
||||
int computeDetections(Yolo::detection *dets, int &ndets, int netw, int neth, float thresh, int new_coords=0);
|
||||
|
||||
dnnType *predictions;
|
||||
|
||||
static const int MAX_DETECTIONS = 256;
|
||||
static const int MAX_DETECTIONS = 8192*2;
|
||||
static Yolo::detection *allocateDetections(int nboxes, int classes);
|
||||
static void mergeDetections(Yolo::detection *dets, int ndets, int classes);
|
||||
static void mergeDetections(Yolo::detection *dets, int ndets, int classes, double nms_thresh=0.45, nmsKind_t nsm_kind=GREEDY_NMS);
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -385,6 +719,7 @@ public:
|
||||
virtual layerType_t getLayerType() { return LAYER_REGION; };
|
||||
|
||||
int classes, coords, num;
|
||||
int c,h,w;
|
||||
|
||||
virtual dnnType* infer(dataDim_t &dim, dnnType* srcData);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
#ifndef MOBILENETDETECTION_H
|
||||
#define MOBILENETDETECTION_H
|
||||
|
||||
#include <opencv2/videoio.hpp>
|
||||
#include "opencv2/opencv.hpp"
|
||||
|
||||
#include "DetectionNN.h"
|
||||
|
||||
#define N_COORDS 4
|
||||
#define N_SSDSPEC 6
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
struct SSDSpec
|
||||
{
|
||||
int featureSize = 0;
|
||||
int shrinkage = 0;
|
||||
int boxWidth = 0;
|
||||
int boxHeight = 0;
|
||||
int ratio1 = 0;
|
||||
int ratio2 = 0;
|
||||
|
||||
SSDSpec() {}
|
||||
SSDSpec(int feature_size, int shrinkage, int box_width, int box_height, int ratio1, int ratio2) :
|
||||
featureSize(feature_size), shrinkage(shrinkage), boxWidth(box_width),
|
||||
boxHeight(box_height), ratio1(ratio1), ratio2(ratio2) {}
|
||||
void setAll(int feature_size, int shrinkage, int box_width, int box_height, int ratio1, int ratio2)
|
||||
{
|
||||
this->featureSize = feature_size;
|
||||
this->shrinkage = shrinkage;
|
||||
this->boxWidth = box_width;
|
||||
this->boxHeight = box_height;
|
||||
this->ratio1 = ratio1;
|
||||
this->ratio2 = ratio2;
|
||||
}
|
||||
void print()
|
||||
{
|
||||
std::cout << "fsize: " << featureSize << "\tshrinkage: " << shrinkage <<
|
||||
"\t box W:" << boxWidth << "\tbox H: " << boxHeight <<
|
||||
"\t x ratio:" << ratio1 << "\t y ratio:" << ratio2 << std::endl;
|
||||
}
|
||||
};
|
||||
|
||||
class MobilenetDetection : public DetectionNN
|
||||
{
|
||||
private:
|
||||
float IoUThreshold = 0.45;
|
||||
float centerVariance = 0.1;
|
||||
float sizeVariance = 0.2;
|
||||
int imageSize;
|
||||
|
||||
float *priors = nullptr;
|
||||
int nPriors = 0;
|
||||
float *locations_h, *confidences_h;
|
||||
|
||||
|
||||
|
||||
void generate_ssd_priors(const SSDSpec *specs, const int n_specs, bool clamp = true);
|
||||
void convert_locatios_to_boxes_and_center();
|
||||
float iou(const tk::dnn::box &a, const tk::dnn::box &b);
|
||||
|
||||
|
||||
|
||||
public:
|
||||
MobilenetDetection() {};
|
||||
~MobilenetDetection() {};
|
||||
|
||||
bool init(const std::string& tensor_path,const int n_classes, const int n_batches=1, const float conf_thresh=0.3);
|
||||
void preprocess(cv::Mat &frame, const int bi=0);
|
||||
void postprocess(const int bi=0,const bool mAP=false);
|
||||
};
|
||||
|
||||
|
||||
} // namespace dnn
|
||||
} // namespace tk
|
||||
|
||||
#endif /*MOBILENETDETECTION_H*/
|
||||
+17
-6
@@ -1,17 +1,18 @@
|
||||
#ifndef NETWORK_H
|
||||
#define NETWORK_H
|
||||
|
||||
#include <string>
|
||||
#include "utils.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
/**
|
||||
Data rapresentation beetween layers
|
||||
Data representation between layers
|
||||
n = batch size
|
||||
c = channels
|
||||
h = heigth (lines)
|
||||
h = height (lines)
|
||||
w = width (rows)
|
||||
l = lenght (3rd dimension)
|
||||
l = length (3rd dimension)
|
||||
*/
|
||||
struct dataDim_t {
|
||||
|
||||
@@ -32,21 +33,24 @@ struct dataDim_t {
|
||||
};
|
||||
|
||||
class Layer;
|
||||
const int MAX_LAYERS = 256;
|
||||
const int MAX_LAYERS = 512;
|
||||
|
||||
class Network {
|
||||
|
||||
public:
|
||||
Network(dataDim_t input_dim);
|
||||
virtual ~Network();
|
||||
void releaseLayers();
|
||||
|
||||
/**
|
||||
Do inferece for every added layer
|
||||
Do inference for every added layer
|
||||
*/
|
||||
dnnType* infer(dataDim_t &dim, dnnType* data);
|
||||
|
||||
bool addLayer(Layer *l);
|
||||
void print();
|
||||
const char *getNetworkRTName(const char *network_name);
|
||||
void adjustFeatureMapSizeWithShortcuts();
|
||||
|
||||
cudnnDataType_t dataType;
|
||||
cudnnTensorFormat_t tensorFormat;
|
||||
@@ -59,7 +63,14 @@ public:
|
||||
dataDim_t input_dim;
|
||||
dataDim_t getOutputDim();
|
||||
|
||||
bool fp16, dla;
|
||||
bool fp16, dla, int8;
|
||||
int maxBatchSize;
|
||||
bool dontLoadWeights;
|
||||
std::string fileImgList;
|
||||
std::string fileLabelList;
|
||||
std::string networkName;
|
||||
std::string networkNameRT;
|
||||
|
||||
};
|
||||
|
||||
}}
|
||||
|
||||
+61
-41
@@ -6,43 +6,30 @@
|
||||
#include "Network.h"
|
||||
#include "Layer.h"
|
||||
#include "NvInfer.h"
|
||||
#include <memory>
|
||||
#include <tkDNN/kernels.h>
|
||||
#include <pluginsRT/ActivationLeakyRT.h>
|
||||
#include <pluginsRT/ActivationLogisticRT.h>
|
||||
#include <pluginsRT/ActivationMishRT.h>
|
||||
#include <pluginsRT/ActivationReLUCeilingRT.h>
|
||||
#include <pluginsRT/DeformableConvRT.h>
|
||||
#include <pluginsRT/FlattenConcatRT.h>
|
||||
#include <pluginsRT/MaxPoolingFixedSizeRT.h>
|
||||
#include <pluginsRT/RegionRT.h>
|
||||
#include <pluginsRT/ReorgRT.h>
|
||||
#include <pluginsRT/ReshapeRT.h>
|
||||
#include <pluginsRT/ResizeLayerRT.h>
|
||||
#include <pluginsRT/RouteRT.h>
|
||||
#include <pluginsRT/ShortcutRT.h>
|
||||
#include <pluginsRT/UpsampleRT.h>
|
||||
#include <pluginsRT/YoloRT.h>
|
||||
#include <pluginsRT/ConstantPaddingRT.h>
|
||||
#include <pluginsRT/ReflectionPadding.h>
|
||||
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
template<typename T> void writeBUF(char*& buffer, const T& val)
|
||||
{
|
||||
*reinterpret_cast<T*>(buffer) = val;
|
||||
buffer += sizeof(T);
|
||||
}
|
||||
|
||||
template<typename T> T readBUF(const char*& buffer)
|
||||
{
|
||||
T val = *reinterpret_cast<const T*>(buffer);
|
||||
buffer += sizeof(T);
|
||||
return val;
|
||||
}
|
||||
|
||||
using namespace nvinfer1;
|
||||
#include "pluginsRT/ActivationLeakyRT.h"
|
||||
#include "pluginsRT/ReorgRT.h"
|
||||
#include "pluginsRT/RegionRT.h"
|
||||
//#include "pluginsRT/RouteRT.h"
|
||||
#include "pluginsRT/ShortcutRT.h"
|
||||
#include "pluginsRT/YoloRT.h"
|
||||
#include "pluginsRT/UpsampleRT.h"
|
||||
//#include "pluginsRT/Int8Calibrator.h"
|
||||
|
||||
class PluginFactory : IPluginFactory
|
||||
{
|
||||
public:
|
||||
YoloRT *yolos[16];
|
||||
int n_yolos;
|
||||
|
||||
virtual IPlugin* createPlugin(const char* layerName, const void* serialData, size_t serialLength);
|
||||
};
|
||||
|
||||
|
||||
|
||||
class NetworkRT {
|
||||
|
||||
public:
|
||||
@@ -50,28 +37,46 @@ public:
|
||||
nvinfer1::IBuilder *builderRT;
|
||||
nvinfer1::IRuntime *runtimeRT;
|
||||
nvinfer1::INetworkDefinition *networkRT;
|
||||
#if NV_TENSORRT_MAJOR >= 6
|
||||
nvinfer1::IBuilderConfig *configRT;
|
||||
#endif
|
||||
|
||||
nvinfer1::ICudaEngine *engineRT;
|
||||
nvinfer1::IExecutionContext *contextRT;
|
||||
|
||||
const static int MAX_BUFFERS_RT = 10;
|
||||
void* buffersRT[MAX_BUFFERS_RT];
|
||||
dataDim_t buffersDIM[MAX_BUFFERS_RT];
|
||||
int buf_input_idx, buf_output_idx;
|
||||
|
||||
bool builderActive = false;
|
||||
dataDim_t input_dim, output_dim;
|
||||
dnnType *output;
|
||||
cudaStream_t stream;
|
||||
|
||||
PluginFactory *pluginFactory;
|
||||
std::vector<nvinfer1::YoloRT*> yolo_plugins; // yolo layers in network
|
||||
|
||||
NetworkRT(Network *net, const char *name);
|
||||
virtual ~NetworkRT();
|
||||
|
||||
int getMaxBatchSize() {
|
||||
if(engineRT != nullptr)
|
||||
return engineRT->getMaxBatchSize();
|
||||
else
|
||||
return 0;
|
||||
}
|
||||
|
||||
int getBuffersN() {
|
||||
if(engineRT != nullptr)
|
||||
return engineRT->getNbBindings();
|
||||
else
|
||||
return 0;
|
||||
}
|
||||
|
||||
/**
|
||||
Do inferece
|
||||
Do inference
|
||||
*/
|
||||
dnnType* infer(dataDim_t &dim, dnnType* data);
|
||||
void enqueue();
|
||||
void enqueue(int batchSize = 1);
|
||||
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Layer *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Conv2d *l);
|
||||
@@ -80,14 +85,29 @@ public:
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Pooling *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Softmax *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Route *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Reorg *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Region *l);
|
||||
nvinfer1::IPluginV2Layer* convert_layer(nvinfer1::ITensor *input, Flatten *l);
|
||||
nvinfer1::IPluginV2Layer* convert_layer(nvinfer1::ITensor *input, Reshape *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Resize *l);
|
||||
nvinfer1::IPluginV2Layer* convert_layer(nvinfer1::ITensor *input, Reorg *l);
|
||||
nvinfer1::IPluginV2Layer* convert_layer(nvinfer1::ITensor *input, Region *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Shortcut *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Yolo *l);
|
||||
nvinfer1::IPluginV2Layer* convert_layer(nvinfer1::ITensor *input, Yolo *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, Upsample *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input, DeformConv2d *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor *input,Padding *l);
|
||||
nvinfer1::ILayer* convert_layer(nvinfer1::ITensor* input,MulAdd *l);
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 5 && NV_TENSORRT_MAJOR < 8
|
||||
bool serialize(const char *filename);
|
||||
#else
|
||||
bool serialize(const char *filename,nvinfer1::IHostMemory *ptr);
|
||||
#endif
|
||||
|
||||
bool deserialize(const char *filename);
|
||||
void destroy();
|
||||
|
||||
|
||||
|
||||
};
|
||||
|
||||
}}
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
#pragma once
|
||||
#include <iostream>
|
||||
#include <opencv2/core/types.hpp>
|
||||
#include "tkdnn.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
cv::Mat vizFloat2colorMap(cv::Mat map, double min=0, double max=0, int classes=19);
|
||||
cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int img_h, int img_w, double min=0, double max=0, int classes=0);
|
||||
cv::Mat vizLayer2Mat(tk::dnn::Network *net, int layer, int imgdim = 1000);
|
||||
|
||||
}}
|
||||
@@ -0,0 +1,407 @@
|
||||
#ifndef SEGMENTATIONNN_H
|
||||
#define SEGMENTATIONNN_H
|
||||
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h>
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
#include <mutex>
|
||||
#include "utils.h"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
#include <opencv2/core/hal/interface.h>
|
||||
|
||||
#include "tkdnn.h"
|
||||
#include "NetworkViz.h"
|
||||
#include "kernelsThrust.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
class SegmentationNN {
|
||||
|
||||
protected:
|
||||
tk::dnn::NetworkRT *netRT = nullptr;
|
||||
int nBatches = 1;
|
||||
|
||||
std::vector<cv::Size> originalSize;
|
||||
cv::Mat bgr[3];
|
||||
dnnType *input;
|
||||
dnnType *input_d;
|
||||
float* confidences_h;
|
||||
|
||||
float * tmpInputData_d;
|
||||
float *tmpOutData_d;
|
||||
float *tmpOutData_h;
|
||||
|
||||
float *mean_d, *stddev_d;
|
||||
|
||||
cublasHandle_t cublasHandle;
|
||||
|
||||
void computeBorders(const int or_width, const int or_height, int& top, int& bottom, int& left, int&right){
|
||||
top = 0;
|
||||
bottom = 0;
|
||||
left = 0;
|
||||
right = 0;
|
||||
|
||||
if(or_height != or_width){
|
||||
if(or_height < or_width){
|
||||
top = (or_width - or_height)/2;
|
||||
bottom = or_width - top - or_height;
|
||||
}
|
||||
else{
|
||||
left = (or_height - or_width)/2;
|
||||
right = or_height - left - or_width;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* This method preprocess the image, before feeding it to the NN.
|
||||
*
|
||||
* @param frame original frame to adapt for inference.
|
||||
* @param bi batch index
|
||||
*/
|
||||
void preprocess(cv::Mat &frame, const int bi=0) {
|
||||
originalSize[bi] = frame.size();
|
||||
|
||||
frame.convertTo(frame, CV_32FC3, 1 / 255.0, 0);
|
||||
int H = frame.rows;
|
||||
int W = frame.cols;
|
||||
cv::Mat frame_cropped;
|
||||
|
||||
int top, bottom, left, right;
|
||||
computeBorders(W, H, top, bottom, left, right);
|
||||
cv::copyMakeBorder(frame, frame_cropped, top, bottom, left, right, cv::BORDER_CONSTANT, cv::Scalar(0,0,0) );
|
||||
|
||||
tk::dnn::dataDim_t idim = netRT->input_dim;
|
||||
|
||||
resize(frame_cropped, frame_cropped, cv::Size(idim.w, idim.h));
|
||||
|
||||
cv::split(frame_cropped, bgr);
|
||||
for (int i = 0; i < idim.c; i++){
|
||||
int idx = i * frame_cropped.rows * frame_cropped.cols;
|
||||
int ch = idim.c-1 -i;
|
||||
memcpy((void *)&input[idx + idim.tot()*bi], (void *)bgr[ch].data, frame_cropped.rows * frame_cropped.cols * sizeof(dnnType));
|
||||
}
|
||||
|
||||
checkCuda(cudaMemcpyAsync(input_d+ idim.tot()*bi, input + idim.tot()*bi, idim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream));
|
||||
|
||||
normalize(input_d + idim.tot()*bi, idim.c, idim.h, idim.w, mean_d, stddev_d);
|
||||
}
|
||||
|
||||
/**
|
||||
* This method postprocess the output of the NN to obtain the correct
|
||||
* boundig boxes.
|
||||
*
|
||||
* @param bi batch index
|
||||
*/
|
||||
void postprocess(const int bi=0, bool appy_colormap = true) {
|
||||
dnnType *rt_out = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi;
|
||||
|
||||
dataDim_t odim = netRT->output_dim;
|
||||
|
||||
matrixTranspose(cublasHandle, rt_out, tmpInputData_d, odim.c, odim.w*odim.h);
|
||||
maxElem(tmpInputData_d, tmpOutData_d, odim.c, odim.h, odim.w);
|
||||
checkCuda(cudaMemcpy(tmpOutData_h, tmpOutData_d, odim.w*odim.h * sizeof(float), cudaMemcpyDeviceToHost));
|
||||
|
||||
dataDim_t vdim = odim;
|
||||
vdim.c = 1;
|
||||
|
||||
cv::Mat colored;
|
||||
|
||||
if(appy_colormap)
|
||||
colored = vizData2Mat(tmpOutData_h, vdim, netRT->input_dim.h, netRT->input_dim.w, 0, classes, classes);
|
||||
else{
|
||||
cv::Mat colored_fp32 (cv::Size(odim.w, odim.h),CV_32FC1, tmpOutData_h);
|
||||
colored_fp32.convertTo(colored, CV_8UC1);
|
||||
}
|
||||
|
||||
int max_dim = (originalSize[bi].width > originalSize[bi].height) ? originalSize[bi].width : originalSize[bi].height;
|
||||
resize(colored, colored, cv::Size(max_dim, max_dim));
|
||||
int top, bottom, left, right;
|
||||
computeBorders(originalSize[bi].width, originalSize[bi].height, top, bottom, left, right);
|
||||
cv::Rect roi(left,top,originalSize[bi].width, originalSize[bi].height);
|
||||
cv::Mat or_size (colored, roi);
|
||||
segmented[bi] = or_size;
|
||||
};
|
||||
|
||||
public:
|
||||
int classes = 0;
|
||||
std::vector<double> stats; /*keeps track of inference times (ms)*/
|
||||
std::vector<double> stats_pre;
|
||||
std::vector<double> stats_post;
|
||||
std::vector<std::string> classesNames;
|
||||
std::vector<cv::Mat> segmented;
|
||||
|
||||
SegmentationNN() {
|
||||
checkERROR( cublasCreate(&cublasHandle) );
|
||||
};
|
||||
~SegmentationNN(){
|
||||
checkERROR( cublasDestroy(cublasHandle) );
|
||||
};
|
||||
|
||||
/**
|
||||
* Method used to inialize the class, allocate memory and compute
|
||||
* needed data.
|
||||
*
|
||||
* @param tensor_path path to the rt file og the NN.
|
||||
* @param n_classes number of classes for the given dataset.
|
||||
* @param n_batches maximum number of batches to use in inference
|
||||
* @return true if everything is correct, false otherwise.
|
||||
*/
|
||||
bool init(const std::string& tensor_path, const int n_classes=19, const int n_batches=1){
|
||||
std::cout<<(tensor_path).c_str()<<"\n";
|
||||
if(!fileExist(tensor_path.c_str()))
|
||||
FatalError("This file do not exists" + tensor_path );
|
||||
|
||||
netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str());
|
||||
classes = n_classes;
|
||||
nBatches = n_batches;
|
||||
|
||||
checkCuda(cudaMallocHost(&input, sizeof(dnnType) * netRT->input_dim.tot() * nBatches));
|
||||
checkCuda(cudaMalloc(&input_d, sizeof(dnnType) * netRT->input_dim.tot() * nBatches));
|
||||
|
||||
dataDim_t odim = netRT->output_dim;
|
||||
|
||||
checkCuda(cudaMallocHost(&confidences_h, sizeof(float) * odim.tot()));
|
||||
checkCuda(cudaMalloc(&tmpInputData_d, sizeof(float) * odim.tot()));
|
||||
checkCuda(cudaMalloc(&tmpOutData_d, sizeof(float) * odim.w*odim.h));
|
||||
checkCuda(cudaMallocHost(&tmpOutData_h, sizeof(float) * odim.w*odim.h));
|
||||
|
||||
segmented.resize(nBatches);
|
||||
originalSize.resize(nBatches);
|
||||
|
||||
std::vector<float> mean = {0.485, 0.456, 0.406};
|
||||
std::vector<float> stddev = {0.229, 0.224, 0.225};
|
||||
|
||||
checkCuda(cudaMalloc(&mean_d, sizeof(float) * mean.size()));
|
||||
checkCuda(cudaMalloc(&stddev_d, sizeof(float) * stddev.size()));
|
||||
|
||||
checkCuda(cudaMemcpyAsync(mean_d, mean.data(), mean.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream));
|
||||
checkCuda(cudaMemcpyAsync(stddev_d, stddev.data(), stddev.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream));
|
||||
return true;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* This method performs the whole detection of the NN.
|
||||
*
|
||||
* @param frames frames to run detection on.
|
||||
* @param cur_batches number of batches to use in inference
|
||||
* @param save_times if set to true, preprocess, inference and postprocess times
|
||||
* are saved on a csv file, otherwise not.
|
||||
* @param times pointer to the output stream where to write times
|
||||
* @param mAP set to true only if all the probabilities for a bounding
|
||||
* box are needed, as in some cases for the mAP calculation
|
||||
*/
|
||||
void update(std::vector<cv::Mat>& frames, const int cur_batches=1, bool apply_colormap=true){
|
||||
if(cur_batches > nBatches)
|
||||
FatalError("A batch size greater than nBatches cannot be used");
|
||||
|
||||
originalSize.clear();
|
||||
if(TKDNN_VERBOSE) printCenteredTitle(" TENSORRT detection ", '=', 30);
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi){
|
||||
if(!frames[bi].data)
|
||||
FatalError("No image data feed to detection");
|
||||
originalSize.push_back(frames[bi].size());
|
||||
preprocess(frames[bi], bi);
|
||||
}
|
||||
TKDNN_TSTOP
|
||||
stats_pre.push_back(t_ns);
|
||||
}
|
||||
|
||||
//do inference
|
||||
tk::dnn::dataDim_t dim = netRT->input_dim;
|
||||
dim.n = cur_batches;
|
||||
{
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
TKDNN_TSTART
|
||||
netRT->infer(dim, input_d);
|
||||
TKDNN_TSTOP
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
stats.push_back(t_ns);
|
||||
}
|
||||
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi)
|
||||
postprocess(bi, apply_colormap);
|
||||
TKDNN_TSTOP
|
||||
stats_post.push_back(t_ns);
|
||||
}
|
||||
}
|
||||
|
||||
void updateOriginal(cv::Mat frame, bool apply_colormap=true){
|
||||
|
||||
std::vector<cv::Mat> splitted_frames;
|
||||
int H, W, net_H, net_W;
|
||||
int top = 0, bottom = 0, left = 0, right = 0;
|
||||
std::vector<std::pair<int,int>> pos;
|
||||
|
||||
{
|
||||
TKDNN_TSTART
|
||||
cv::Size original_size = frame.size();
|
||||
|
||||
frame.convertTo(frame, CV_32FC3, 1 / 255.0, 0);
|
||||
H = frame.rows;
|
||||
W = frame.cols;
|
||||
net_H = netRT->input_dim.h;
|
||||
net_W = netRT->input_dim.w;
|
||||
|
||||
cv::Mat frame_cropped;
|
||||
|
||||
if( H <= net_H && W <= net_W ){ // smaller size wrt network
|
||||
top = (net_H - H)/2;
|
||||
bottom = net_H - H - top ;
|
||||
left = (net_W - W)/2;
|
||||
right = net_W - W - left ;
|
||||
cv::copyMakeBorder(frame, frame_cropped, top, bottom, left, right, cv::BORDER_CONSTANT, cv::Scalar(0,0,0) );
|
||||
splitted_frames.push_back(frame_cropped);
|
||||
}
|
||||
else{ //bigger size wrt network
|
||||
|
||||
|
||||
if(H < net_H || W < net_W){
|
||||
if(H < net_H){
|
||||
top = (net_H - H)/2;
|
||||
bottom = net_H - H - top ;
|
||||
}
|
||||
else{
|
||||
left = (net_W - W)/2;
|
||||
right = net_W - W - left ;
|
||||
}
|
||||
cv::copyMakeBorder(frame, frame_cropped, top, bottom, left, right, cv::BORDER_CONSTANT, cv::Scalar(0,0,0));
|
||||
}
|
||||
|
||||
for(int x=0; x+net_W<=W ;){
|
||||
for(int y=0; y+net_H <=H ; ){
|
||||
cv::Rect roi(x, y, net_W, net_H);
|
||||
cv::Mat image_roi = frame(roi);
|
||||
splitted_frames.push_back(image_roi);
|
||||
pos.push_back(std::make_pair(x,y));
|
||||
|
||||
y += net_H;
|
||||
if(y == H)
|
||||
break;
|
||||
if(y + net_H > H) y = H - net_H;
|
||||
}
|
||||
x += net_W;
|
||||
if(x == W)
|
||||
break;
|
||||
if(x + net_W > W) x = W - net_W;
|
||||
}
|
||||
}
|
||||
|
||||
tk::dnn::dataDim_t idim = netRT->input_dim;
|
||||
|
||||
if(splitted_frames.size()> nBatches)
|
||||
FatalError(std::to_string(splitted_frames.size()) + " min batches required");
|
||||
|
||||
for(int bi=0; bi<splitted_frames.size();++bi){
|
||||
cv::split(splitted_frames[bi], bgr);
|
||||
for (int i = 0; i < idim.c; i++){
|
||||
int idx = i * splitted_frames[bi].rows * splitted_frames[bi].cols;
|
||||
int ch = idim.c-1 -i;
|
||||
memcpy((void *)&input[idx + idim.tot()*bi], (void *)bgr[ch].data, splitted_frames[bi].rows * splitted_frames[bi].cols * sizeof(dnnType));
|
||||
}
|
||||
|
||||
checkCuda(cudaMemcpyAsync(input_d+ idim.tot()*bi, input + idim.tot()*bi, idim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream));
|
||||
normalize(input_d + idim.tot()*bi, idim.c, idim.h, idim.w, mean_d, stddev_d);
|
||||
}
|
||||
TKDNN_TSTOP
|
||||
stats_pre.push_back(t_ns);
|
||||
}
|
||||
|
||||
tk::dnn::dataDim_t dim = netRT->input_dim;
|
||||
dim.n = splitted_frames.size();
|
||||
{
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
TKDNN_TSTART
|
||||
netRT->infer(dim, input_d);
|
||||
TKDNN_TSTOP
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
stats.push_back(t_ns);
|
||||
}
|
||||
|
||||
dataDim_t odim = netRT->output_dim;
|
||||
|
||||
std::vector<cv::Mat> out_img;
|
||||
|
||||
{
|
||||
TKDNN_TSTART
|
||||
|
||||
for(int bi=0; bi<splitted_frames.size();++bi){
|
||||
|
||||
dnnType *rt_out = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi;
|
||||
|
||||
matrixTranspose(cublasHandle, rt_out, tmpInputData_d, odim.c, odim.w*odim.h);
|
||||
maxElem(tmpInputData_d, tmpOutData_d, odim.c, odim.h, odim.w);
|
||||
checkCuda(cudaMemcpy(tmpOutData_h, tmpOutData_d, odim.w*odim.h * sizeof(float), cudaMemcpyDeviceToHost));
|
||||
|
||||
dataDim_t vdim = odim;
|
||||
vdim.c = 1;
|
||||
|
||||
cv::Mat colored;
|
||||
|
||||
if(apply_colormap)
|
||||
colored = vizData2Mat(tmpOutData_h, vdim, netRT->input_dim.h, netRT->input_dim.w, 0, classes, classes);
|
||||
else{
|
||||
cv::Mat colored_fp32 (cv::Size(odim.w, odim.h),CV_32FC1, tmpOutData_h);
|
||||
colored_fp32.convertTo(colored, CV_8UC1);
|
||||
}
|
||||
out_img.push_back(colored);
|
||||
}
|
||||
|
||||
|
||||
cv::Mat seg(frame.size(), out_img[0].type());
|
||||
if(out_img.size() == 1)
|
||||
{
|
||||
cv::Rect roi(left, top, W, H);
|
||||
seg = out_img[0](roi);
|
||||
}
|
||||
else{
|
||||
int bi=0;
|
||||
|
||||
if(top == 0 && left == 0){
|
||||
|
||||
for(int i=0; i<out_img.size(); ++i){
|
||||
cv::Mat roi_collage = seg(cv::Rect( pos[i].first ,pos[i].second,out_img[i].cols,out_img[i].rows));
|
||||
out_img[i].copyTo(roi_collage);
|
||||
}
|
||||
}
|
||||
else{
|
||||
FatalError("Not handled case")
|
||||
}
|
||||
}
|
||||
segmented[0] = seg;
|
||||
|
||||
TKDNN_TSTOP
|
||||
stats_post.push_back(t_ns);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Method to draw boundixg boxes and labels on a frame.
|
||||
*/
|
||||
cv::Mat draw(const int cur_batches=1) {
|
||||
for(int i=0; i<cur_batches; ++i){
|
||||
|
||||
cv::imshow("segmented", segmented[i]);
|
||||
cv::resizeWindow("segmented", cv::Size(512,288));
|
||||
cv::waitKey(1);
|
||||
}
|
||||
return segmented[0];
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
}}
|
||||
|
||||
#endif /* SEGMENTATIONNN_H*/
|
||||
@@ -0,0 +1,158 @@
|
||||
#ifndef TRACKINGNN_H
|
||||
#define TRACKINGNN_H
|
||||
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h>
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <mutex>
|
||||
#include "utils.h"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
#include "tkdnn.h"
|
||||
|
||||
// #define OPENCV_CUDACONTRIB //if OPENCV has been compiled with CUDA and contrib.
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
#include <opencv2/cudawarping.hpp>
|
||||
#include <opencv2/cudaarithm.hpp>
|
||||
#endif
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
class TrackingNN {
|
||||
|
||||
protected:
|
||||
tk::dnn::NetworkRT *netRT = nullptr;
|
||||
dnnType *input_d;
|
||||
|
||||
std::vector<cv::Size> originalSize;
|
||||
|
||||
cv::Scalar colors[256];
|
||||
|
||||
int nBatches = 1;
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
cv::cuda::GpuMat bgr[3];
|
||||
cv::cuda::GpuMat imagePreproc;
|
||||
#else
|
||||
cv::Mat bgr[3];
|
||||
cv::Mat imagePreproc;
|
||||
dnnType *input;
|
||||
#endif
|
||||
|
||||
/**
|
||||
* This method preprocess the image, before feeding it to the NN.
|
||||
*
|
||||
* @param frame original frame to adapt for inference.
|
||||
* @param bi batch index
|
||||
*/
|
||||
virtual void preprocess(cv::Mat &frame, const int bi=0) = 0;
|
||||
|
||||
/**
|
||||
* This method postprocess the output of the NN to obtain the correct
|
||||
* boundig boxes.
|
||||
*
|
||||
* @param bi batch index
|
||||
* @param mAP set to true only if all the probabilities for a bounding
|
||||
* box are needed, as in some cases for the mAP calculation
|
||||
*/
|
||||
virtual void postprocess(const int bi=0,const bool mAP=false) = 0;
|
||||
|
||||
public:
|
||||
int classes = 0;
|
||||
float confThreshold = 0.3; /*threshold on the confidence of the boxes*/
|
||||
|
||||
std::vector<double> pre_stats, stats, post_stats, visual_stats; /*keeps track of inference times (ms)*/
|
||||
std::vector<std::string> classesNames;
|
||||
|
||||
TrackingNN() {};
|
||||
~TrackingNN(){};
|
||||
|
||||
/**
|
||||
* Method used to initialize the class, allocate memory and compute
|
||||
* needed data.
|
||||
*
|
||||
* @param tensor_path path to the rt file of the NN.
|
||||
* @param n_classes number of classes for the given dataset.
|
||||
* @param n_batches maximum number of batches to use in inference.
|
||||
* @return true if everything is correct, false otherwise.
|
||||
*/
|
||||
virtual bool init(const std::string& tensor_path, const int n_classes=3, const int n_batches=1,
|
||||
const float conf_thresh=0.3, const bool mode_3d=true, const std::vector<cv::Mat>& k_calibs=std::vector<cv::Mat>()) = 0;
|
||||
|
||||
/**
|
||||
* This method performs the whole detection and tracking of the NN.
|
||||
*
|
||||
* @param frames frames to run detection and trcking on.
|
||||
* @param cur_batches number of batches to use in inference.
|
||||
* @param save_times if set to true, preprocess, inference and postprocess times
|
||||
* are saved on a csv file, otherwise not.
|
||||
* @param times pointer to the output stream where to write times.
|
||||
* @param mAP set to true only if all the probabilities for a bounding
|
||||
* box are needed, as in some cases for the mAP calculation.
|
||||
*/
|
||||
void update(std::vector<cv::Mat>& frames, const int cur_batches=1, bool save_times=false,
|
||||
std::ofstream *times=nullptr, const bool mAP=false){
|
||||
if(save_times && times==nullptr)
|
||||
FatalError("save_times set to true, but no valid ofstream given");
|
||||
if(cur_batches > nBatches)
|
||||
FatalError("A batch size greater than nBatches cannot be used");
|
||||
|
||||
originalSize.clear();
|
||||
if(TKDNN_VERBOSE) printCenteredTitle(" TENSORRT detection ", '=', 30);
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi){
|
||||
if(!frames[bi].data)
|
||||
FatalError("No image data feed to detection");
|
||||
originalSize.push_back(frames[bi].size());
|
||||
preprocess(frames[bi], bi);
|
||||
}
|
||||
TKDNN_TSTOP
|
||||
pre_stats.push_back(t_ns);
|
||||
if(save_times) *times<<t_ns<<";";
|
||||
}
|
||||
|
||||
//do inference
|
||||
tk::dnn::dataDim_t dim = netRT->input_dim;
|
||||
dim.n = cur_batches;
|
||||
{
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
TKDNN_TSTART
|
||||
netRT->infer(dim, input_d);
|
||||
TKDNN_TSTOP
|
||||
if(TKDNN_VERBOSE) dim.print();
|
||||
stats.push_back(t_ns);
|
||||
if(save_times) *times<<t_ns<<";";
|
||||
}
|
||||
|
||||
{
|
||||
TKDNN_TSTART
|
||||
for(int bi=0; bi<cur_batches;++bi)
|
||||
postprocess(bi, mAP);
|
||||
TKDNN_TSTOP
|
||||
post_stats.push_back(t_ns);
|
||||
if(save_times) *times<<t_ns<<"\n";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Method to draw bounding boxes and labels on a frame.
|
||||
*
|
||||
* @param frames original frame to draw bounding box on.
|
||||
*/
|
||||
virtual void draw(std::vector<cv::Mat>& frames){};
|
||||
|
||||
};
|
||||
|
||||
}}
|
||||
|
||||
#endif /* TRACKINGNN_H*/
|
||||
@@ -1,67 +1,36 @@
|
||||
#include <iostream>
|
||||
#include <signal.h>
|
||||
#include <stdlib.h> /* srand, rand */
|
||||
#include <unistd.h>
|
||||
#include <mutex>
|
||||
#include "utils.h"
|
||||
#ifndef Yolo3Detection_H
|
||||
#define Yolo3Detection_H
|
||||
#include <opencv2/videoio.hpp>
|
||||
#include "opencv2/opencv.hpp"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
#include "tkdnn.h"
|
||||
#include "DetectionNN.h"
|
||||
#include "DarknetParser.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
class Yolo3Detection : public DetectionNN
|
||||
{
|
||||
private:
|
||||
int num = 0;
|
||||
int nMasks = 0;
|
||||
int nDets = 0;
|
||||
tk::dnn::Yolo::detection *dets = nullptr;
|
||||
tk::dnn::Yolo* yolo[3];
|
||||
|
||||
/**
|
||||
*
|
||||
* @author Francesco Gatti
|
||||
*/
|
||||
class Yolo3Detection {
|
||||
tk::dnn::Yolo* getYoloLayer(int n=0);
|
||||
|
||||
private:
|
||||
tk::dnn::NetworkRT *netRT = nullptr;
|
||||
tk::dnn::Yolo* yolo[3];
|
||||
dnnType *input, *input_d;
|
||||
|
||||
int ndets = 0;
|
||||
tk::dnn::Yolo::detection *dets = nullptr;
|
||||
|
||||
cv::Mat imageF;
|
||||
cv::Mat bgr[3];
|
||||
|
||||
public:
|
||||
int classes = 0;
|
||||
int num = 0;
|
||||
float thresh = 0.3;
|
||||
cv::Scalar colors[256];
|
||||
|
||||
// this is filled with results
|
||||
std::vector<tk::dnn::box> detected;
|
||||
|
||||
// keep track of inference times (ms)
|
||||
std::vector<double> stats;
|
||||
|
||||
Yolo3Detection() {}
|
||||
|
||||
virtual ~Yolo3Detection() {}
|
||||
|
||||
/**
|
||||
* Method used for inizialize the class
|
||||
*
|
||||
* @return Success of the initialization
|
||||
*/
|
||||
bool init(std::string tensor_path);
|
||||
|
||||
void update(cv::Mat &frame);
|
||||
|
||||
tk::dnn::Yolo* getYoloLayer(int n=0) {
|
||||
if(n<3)
|
||||
return yolo[n];
|
||||
else
|
||||
return nullptr;
|
||||
}
|
||||
cv::Mat bgr_h;
|
||||
|
||||
public:
|
||||
Yolo3Detection() {};
|
||||
~Yolo3Detection() {};
|
||||
|
||||
bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1, const float conf_thresh=0.3);
|
||||
void preprocess(cv::Mat &frame, const int bi=0);
|
||||
void postprocess(const int bi=0,const bool mAP=false);
|
||||
};
|
||||
|
||||
}}
|
||||
|
||||
} // namespace dnn
|
||||
} // namespace tk
|
||||
|
||||
#endif /* Yolo3Detection_H*/
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
#ifndef DEMO_UTILS_H
|
||||
#define DEMO_UTILS_H
|
||||
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
#include <stdlib.h>
|
||||
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
#include <yaml-cpp/yaml.h>
|
||||
|
||||
|
||||
void readCalibrationMatrix(const std::string& path, cv::Mat& calib_mat);
|
||||
|
||||
#endif //DEMO_UTILS_H
|
||||
@@ -0,0 +1,118 @@
|
||||
#ifndef EVALUATION_H
|
||||
#define EVALUATION_H
|
||||
|
||||
#include <iostream>
|
||||
#include <vector>
|
||||
#include <algorithm>
|
||||
|
||||
#include <yaml-cpp/yaml.h>
|
||||
|
||||
#include "tkdnn.h"
|
||||
#include "BoundingBox.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
struct Frame
|
||||
{
|
||||
std::string lFilename;
|
||||
std::string iFilename;
|
||||
std::vector<BoundingBox> gt;
|
||||
std::vector<BoundingBox> det;
|
||||
int width;
|
||||
int height;
|
||||
|
||||
void print() const;
|
||||
};
|
||||
|
||||
struct PR
|
||||
{
|
||||
double precision = 0;
|
||||
double recall = 0;
|
||||
int tp = 0, fp = 0, fn = 0;
|
||||
|
||||
void print();
|
||||
};
|
||||
|
||||
void readmAPParams( const char* config_filename, int& classes, int& map_points,
|
||||
int& map_levels, float& map_step, float& IoU_thresh,
|
||||
float& conf_thresh, bool& verbose);
|
||||
|
||||
/**
|
||||
* This method computes the mean Average Precision for a set of detections and
|
||||
* groundtruths. It returns the mAP for a given IoU threshold, and a given
|
||||
* confidence threshold over all the classes.
|
||||
*
|
||||
* @param images collection of frames on which to compute the metrics
|
||||
* @param classes number of classes of the considered dataset
|
||||
* @param IoU_thresh threshold used to compute Intersection over Union
|
||||
* @param conf_thresh threshold used to filter bounding boxes based on their
|
||||
* confidence (or probability)
|
||||
* @param map_points number of point used to compute the mAP. if 0 is given,
|
||||
* all the recall levels are evaluated, otherwise only
|
||||
* map_point recall levels are used. For COCO evaluation
|
||||
* 101 points are used.
|
||||
* @param verbose is set to true, prints on screen additional info
|
||||
*
|
||||
* @return mAP computed
|
||||
*/
|
||||
double computeMap( std::vector<Frame> &images,const int classes,
|
||||
const float IoU_thresh, const float conf_thresh=0.3,
|
||||
const int map_points=101, const bool verbose=false);
|
||||
|
||||
|
||||
/**
|
||||
* This method computes the mean Average Precision for a set of detections and
|
||||
* groundtruths on several IoU thresholds. It is used to compute, for example,
|
||||
* the most used metric in Object Detection, namely the mAP 0.5:0.95, which is
|
||||
* the average among the mAP for IoU level from 0.5 to 0.95 with a step of 0.05.
|
||||
*
|
||||
* @param images collection of frames on which to compute the metrics
|
||||
* @param classes number of classes of the considered dataset
|
||||
* @param IoU_thresh starting threshold used to compute Intersection over Union
|
||||
* @param conf_thresh threshold used to filter bounding boxes based on their
|
||||
* confidence (or probability)
|
||||
* @param map_points number of point used to compute the mAP. if 0 is given,
|
||||
* all the recall levels are evaluated, otherwise only
|
||||
* map_point recall levels are used. For COCO evaluation
|
||||
* 101 points are used.
|
||||
* @param map_step step used to increment IoU threshold
|
||||
* @param map_levels number of IoU step to perform
|
||||
* @param verbose is set to true, prints on screen additional info
|
||||
* @param write_on_file if set to true, the results produced by this function
|
||||
* are written on file
|
||||
* @param net name of the considered neural network
|
||||
*
|
||||
* @return mAP IoU_tresh:IoU_tresh+map_step*map_levels (e.g. mAP 0.5:0.95 when
|
||||
* map_step=0.05 and map_levels=10)
|
||||
*/
|
||||
double computeMapNIoULevels(std::vector<Frame> &images,const int classes,
|
||||
const float i_IoU_thresh=0.5, const float conf_thresh=0.3,
|
||||
const int map_points=101, const float map_step=0.05,
|
||||
const int map_levels=10, const bool verbose=false,
|
||||
const bool write_on_file = false, std::string net = "");
|
||||
/**
|
||||
* This method computes the number of True Positive (TP), False Positive (FP),
|
||||
* False Negative (FN), precision, recall and f1-score.
|
||||
* Those values are computer over all the detections, over all the classes.
|
||||
*
|
||||
* @param images collection of frames on which to compute the metrics
|
||||
* @param classes number of classes of the considered dataset
|
||||
* @param IoU_thresh threshold used to compute Intersection over Union
|
||||
* @param conf_thresh threshold used to filter bounding boxes based on their
|
||||
* confidence (or probability)
|
||||
* @param verbose is set to true, prints on screen additional info
|
||||
* @param write_on_file if set to true, the results produced by this function
|
||||
* are written on file
|
||||
* @param net name of the considered neural network
|
||||
*/
|
||||
void computeTPFPFN( std::vector<Frame> &images,const int classes,
|
||||
const float IoU_thresh=0.5, const float conf_thresh=0.3,
|
||||
bool verbose=false, const bool write_on_file=false,
|
||||
std::string net="");
|
||||
|
||||
|
||||
void printJsonCOCOFormat(std::ofstream *out_file, const std::string image_path, std::vector<tk::dnn::box> bbox, const int classes, const int w, const int h);
|
||||
|
||||
}}
|
||||
#endif /*EVALUATION_H*/
|
||||
|
||||
+44
-13
@@ -3,25 +3,56 @@
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
void activationELUForward(dnnType* srcData, dnnType* dstData, int size, cudaStream_t stream = cudaStream_t(0));
|
||||
void activationLEAKYForward(dnnType* srcData, dnnType* dstData, int size, cudaStream_t stream = cudaStream_t(0));
|
||||
void activationLOGISTICForward(dnnType* srcData, dnnType* dstData, int size, cudaStream_t stream = cudaStream_t(0));
|
||||
void activationELUForward(dnnType *srcData, dnnType *dstData, int size, cudaStream_t stream = cudaStream_t(0));
|
||||
void activationLEAKYForward(dnnType *srcData, dnnType *dstData, int size, float slope, cudaStream_t stream = cudaStream_t(0));
|
||||
void activationReLUCeilingForward(dnnType *srcData, dnnType *dstData, int size, const float ceiling, cudaStream_t stream = cudaStream_t(0));
|
||||
void activationLOGISTICForward(dnnType *srcData, dnnType *dstData, int size, cudaStream_t stream = cudaStream_t(0));
|
||||
void activationSIGMOIDForward(dnnType *srcData, dnnType *dstData, int size, cudaStream_t stream = cudaStream_t(0));
|
||||
void activationMishForward(dnnType* srcData, dnnType* dstData, int size, cudaStream_t stream= cudaStream_t(0));
|
||||
|
||||
void fill(dnnType* data, int size, dnnType val, cudaStream_t stream = cudaStream_t(0));
|
||||
void fill(dnnType *data, int size, dnnType val, cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void reorgForward( dnnType* srcData, dnnType* dstData,
|
||||
int n, int c, int h, int w, int stride, cudaStream_t stream = cudaStream_t(0));
|
||||
void softmaxForward(float *input, int n, int batch, int batch_offset,
|
||||
void resizeForward(dnnType *srcData, dnnType *dstData, int n, int i_c, int i_h, int i_w,
|
||||
int o_c, int o_h, int o_w, cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void reorgForward(dnnType *srcData, dnnType *dstData,
|
||||
int n, int c, int h, int w, int stride, cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void MaxPoolingForward(dnnType *srcData, dnnType *dstData, int n, int c, int h, int w, int stride_x, int stride_y, int size, int padding, cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void softmaxForward(float *input, int n, int batch, int batch_offset,
|
||||
int groups, int group_offset, int stride, float temp, float *output, cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
|
||||
void shortcutForward(dnnType* srcData, dnnType* dstData, int n1, int c1, int h1, int w1, int s1,
|
||||
int n2, int c2, int h2, int w2, int s2,
|
||||
void shortcutForward(dnnType *srcData, dnnType *dstData, int n1, int c1, int h1, int w1, int s1,
|
||||
int n2, int c2, int h2, int w2, int s2, bool mul,
|
||||
cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void upsampleForward(dnnType* srcData, dnnType* dstData,
|
||||
int n, int c, int h, int w, int s, int forward, float scale,
|
||||
void upsampleForward(dnnType *srcData, dnnType *dstData,
|
||||
int n, int c, int h, int w, int s, int forward, float scale,
|
||||
cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void float2half(float* srcData, __half* dstData, int size, const cudaStream_t stream = cudaStream_t(0));
|
||||
void float2half(float *srcData, __half *dstData, int size, const cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void dcnV2CudaForward(cublasStatus_t stat, cublasHandle_t handle,
|
||||
float *input, float *weight,
|
||||
float *bias, float *ones,
|
||||
float *offset, float *mask,
|
||||
float *output, float *columns,
|
||||
int kernel_h, int kernel_w,
|
||||
const int stride_h, const int stride_w,
|
||||
const int pad_h, const int pad_w,
|
||||
const int dilation_h, const int dilation_w,
|
||||
const int deformable_group, const int batch_id,
|
||||
const int in_n, const int in_c, const int in_h, const int in_w,
|
||||
const int out_n, const int out_c, const int out_h, const int out_w,
|
||||
const int dst_dim, cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void scalAdd(dnnType* dstData, int size, float alpha, float beta, int inc, cudaStream_t stream = cudaStream_t(0));
|
||||
|
||||
void reflection_pad2d_out_forward(int32_t pad_h,int32_t pad_w,float *srcData,float *dstData,int32_t input_h,int32_t input_w,int32_t plane_dim,int32_t n_batch,cudaStream_t cudaStream = cudaStream_t(0));
|
||||
|
||||
void constant_pad2d_forward(dnnType *srcData,dnnType *dstData,int32_t input_h,int32_t input_w,int32_t output_h,
|
||||
int32_t output_w,int32_t c,int32_t n,int32_t padT,int32_t padL,dnnType constant,cudaStream_t cudaStream = cudaStream_t(0));
|
||||
|
||||
|
||||
#endif //KERNELS_H
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
#ifndef KERNELSTHRUST_H
|
||||
#define KERNELSTHRUST_H
|
||||
|
||||
|
||||
#include <thrust/extrema.h>
|
||||
#include <thrust/sort.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
#include <thrust/functional.h>
|
||||
#include <thrust/transform.h>
|
||||
#include <thrust/iterator/constant_iterator.h>
|
||||
#include <thrust/gather.h>
|
||||
#include <thrust/copy.h>
|
||||
#include <thrust/device_ptr.h>
|
||||
|
||||
|
||||
#include "tkdnn.h"
|
||||
|
||||
struct threshold : public thrust::binary_function<float,float,float>
|
||||
{
|
||||
__host__ __device__
|
||||
float operator()(float x, float y) {
|
||||
double toll = 1e-6;
|
||||
if(fabsf(x-y)>toll)
|
||||
return 0.0f;
|
||||
else
|
||||
return x;
|
||||
}
|
||||
};
|
||||
|
||||
void sort(dnnType *src_begin, dnnType *src_end, int *idsrc);
|
||||
void topk(dnnType *src_begin, int *idsrc, int K, float *topk_scores,
|
||||
int *topk_inds, float *topk_ys, float *topk_xs);
|
||||
// void sortAndTopKonDevice(dnnType *src_begin, int *idsrc, float *topk_scores, int *topk_inds, float *topk_ys, float *topk_xs, const int size, const int K, const int n_classes);
|
||||
void normalize(float *bgr, const int ch, const int h, const int w, const float *mean, const float *stddev);
|
||||
void transformDep(float *src_begin, float *src_end, float *dst_begin, float *dst_end);
|
||||
void subtractWithThreshold(dnnType *src_begin, dnnType *src_end, dnnType *src2_begin, dnnType *src_out, struct threshold op);
|
||||
void topKxyclasses(int *ids_begin, int *ids_end, const int K, const int size, const int wh, int *clses, int *xs, int *ys);
|
||||
void topKxyAddOffset(int * ids_begin, const int K, const int size, int *intxs_begin, int *intys_begin,
|
||||
float *xs_begin, float *ys_begin, dnnType *src_begin, float *src_out, int *ids_out);
|
||||
void bboxes(int * ids_begin, const int K, const int size, float *xs_begin, float *ys_begin,
|
||||
dnnType *src_begin, float *bbx0, float *bbx1, float *bby0, float *bby1, float *src_out, int *ids_out);
|
||||
void getRecordsFromTopKId(int * ids_begin, const int K, const int ch, const int size, dnnType *src_begin, float *src_out, int *ids_out);
|
||||
|
||||
void maxElem(dnnType *src_begin, dnnType *dst_begin, const int c, const int h, const int w);
|
||||
|
||||
#endif //KERNELSTHRUST_H
|
||||
@@ -1,289 +0,0 @@
|
||||
int preYoloFilters = (classes+5)*3;
|
||||
|
||||
std::string input_bin = bin_path + "/layers/input.bin";
|
||||
std::vector<std::string> output_bins = {
|
||||
bin_path + "/debug/layer82_out.bin",
|
||||
bin_path + "/debug/layer94_out.bin",
|
||||
bin_path + "/debug/layer106_out.bin"
|
||||
};
|
||||
std::string c0_bin = bin_path + "/layers/c0.bin";
|
||||
std::string c1_bin = bin_path + "/layers/c1.bin";
|
||||
std::string c2_bin = bin_path + "/layers/c2.bin";
|
||||
std::string c3_bin = bin_path + "/layers/c3.bin";
|
||||
std::string c5_bin = bin_path + "/layers/c5.bin";
|
||||
std::string c6_bin = bin_path + "/layers/c6.bin";
|
||||
std::string c7_bin = bin_path + "/layers/c7.bin";
|
||||
std::string c9_bin = bin_path + "/layers/c9.bin";
|
||||
std::string c10_bin = bin_path + "/layers/c10.bin";
|
||||
std::string c12_bin = bin_path + "/layers/c12.bin";
|
||||
std::string c13_bin = bin_path + "/layers/c13.bin";
|
||||
std::string c14_bin = bin_path + "/layers/c14.bin";
|
||||
std::string c16_bin = bin_path + "/layers/c16.bin";
|
||||
std::string c17_bin = bin_path + "/layers/c17.bin";
|
||||
std::string c19_bin = bin_path + "/layers/c19.bin";
|
||||
std::string c20_bin = bin_path + "/layers/c20.bin";
|
||||
std::string c22_bin = bin_path + "/layers/c22.bin";
|
||||
std::string c23_bin = bin_path + "/layers/c23.bin";
|
||||
std::string c25_bin = bin_path + "/layers/c25.bin";
|
||||
std::string c26_bin = bin_path + "/layers/c26.bin";
|
||||
std::string c28_bin = bin_path + "/layers/c28.bin";
|
||||
std::string c29_bin = bin_path + "/layers/c29.bin";
|
||||
std::string c31_bin = bin_path + "/layers/c31.bin";
|
||||
std::string c32_bin = bin_path + "/layers/c32.bin";
|
||||
std::string c34_bin = bin_path + "/layers/c34.bin";
|
||||
std::string c35_bin = bin_path + "/layers/c35.bin";
|
||||
std::string c37_bin = bin_path + "/layers/c37.bin";
|
||||
std::string c38_bin = bin_path + "/layers/c38.bin";
|
||||
std::string c39_bin = bin_path + "/layers/c39.bin";
|
||||
std::string c41_bin = bin_path + "/layers/c41.bin";
|
||||
std::string c42_bin = bin_path + "/layers/c42.bin";
|
||||
std::string c44_bin = bin_path + "/layers/c44.bin";
|
||||
std::string c45_bin = bin_path + "/layers/c45.bin";
|
||||
std::string c47_bin = bin_path + "/layers/c47.bin";
|
||||
std::string c48_bin = bin_path + "/layers/c48.bin";
|
||||
std::string c50_bin = bin_path + "/layers/c50.bin";
|
||||
std::string c51_bin = bin_path + "/layers/c51.bin";
|
||||
std::string c53_bin = bin_path + "/layers/c53.bin";
|
||||
std::string c54_bin = bin_path + "/layers/c54.bin";
|
||||
std::string c56_bin = bin_path + "/layers/c56.bin";
|
||||
std::string c57_bin = bin_path + "/layers/c57.bin";
|
||||
std::string c59_bin = bin_path + "/layers/c59.bin";
|
||||
std::string c60_bin = bin_path + "/layers/c60.bin";
|
||||
std::string c62_bin = bin_path + "/layers/c62.bin";
|
||||
std::string c63_bin = bin_path + "/layers/c63.bin";
|
||||
std::string c64_bin = bin_path + "/layers/c64.bin";
|
||||
std::string c66_bin = bin_path + "/layers/c66.bin";
|
||||
std::string c67_bin = bin_path + "/layers/c67.bin";
|
||||
std::string c69_bin = bin_path + "/layers/c69.bin";
|
||||
std::string c70_bin = bin_path + "/layers/c70.bin";
|
||||
std::string c72_bin = bin_path + "/layers/c72.bin";
|
||||
std::string c73_bin = bin_path + "/layers/c73.bin";
|
||||
std::string c75_bin = bin_path + "/layers/c75.bin";
|
||||
std::string c76_bin = bin_path + "/layers/c76.bin";
|
||||
std::string c77_bin = bin_path + "/layers/c77.bin";
|
||||
std::string c78_bin = bin_path + "/layers/c78.bin";
|
||||
std::string c79_bin = bin_path + "/layers/c79.bin";
|
||||
std::string c80_bin = bin_path + "/layers/c80.bin";
|
||||
std::string c81_bin = bin_path + "/layers/c81.bin";
|
||||
std::string g82_bin = bin_path + "/layers/g82.bin";
|
||||
std::string c84_bin = bin_path + "/layers/c84.bin";
|
||||
std::string c87_bin = bin_path + "/layers/c87.bin";
|
||||
std::string c88_bin = bin_path + "/layers/c88.bin";
|
||||
std::string c89_bin = bin_path + "/layers/c89.bin";
|
||||
std::string c90_bin = bin_path + "/layers/c90.bin";
|
||||
std::string c91_bin = bin_path + "/layers/c91.bin";
|
||||
std::string c92_bin = bin_path + "/layers/c92.bin";
|
||||
std::string c93_bin = bin_path + "/layers/c93.bin";
|
||||
std::string g94_bin = bin_path + "/layers/g94.bin";
|
||||
std::string c96_bin = bin_path + "/layers/c96.bin";
|
||||
std::string c99_bin = bin_path + "/layers/c99.bin";
|
||||
std::string c100_bin = bin_path + "/layers/c100.bin";
|
||||
std::string c101_bin = bin_path + "/layers/c101.bin";
|
||||
std::string c102_bin = bin_path + "/layers/c102.bin";
|
||||
std::string c103_bin = bin_path + "/layers/c103.bin";
|
||||
std::string c104_bin = bin_path + "/layers/c104.bin";
|
||||
std::string c105_bin = bin_path + "/layers/c105.bin";
|
||||
std::string g106_bin = bin_path + "/layers/g106.bin";
|
||||
|
||||
tk::dnn::Conv2d c0 (&net, 32, 3, 3, 1, 1, 1, 1, c0_bin, true);
|
||||
tk::dnn::Activation a0 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c1 (&net, 64, 3, 3, 2, 2, 1, 1, c1_bin, true);
|
||||
tk::dnn::Activation a1 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c2 (&net, 32, 1, 1, 1, 1, 0, 0, c2_bin, true);
|
||||
tk::dnn::Activation a2 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c3 (&net, 64, 3, 3, 1, 1, 1, 1, c3_bin, true);
|
||||
tk::dnn::Activation a3 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s4 (&net, &a1);
|
||||
tk::dnn::Conv2d c5 (&net, 128, 3, 3, 2, 2, 1, 1, c5_bin, true);
|
||||
tk::dnn::Activation a5 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c6 (&net, 64, 1, 1, 1, 1, 0, 0, c6_bin, true);
|
||||
tk::dnn::Activation a6 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c7 (&net, 128, 3, 3, 1, 1, 1, 1, c7_bin, true);
|
||||
tk::dnn::Activation a7 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s8 (&net, &a5);
|
||||
tk::dnn::Conv2d c9 (&net, 64, 1, 1, 1, 1, 0, 0, c9_bin, true);
|
||||
tk::dnn::Activation a9 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c10 (&net, 128, 3, 3, 1, 1, 1, 1, c10_bin, true);
|
||||
tk::dnn::Activation a10 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s11 (&net, &s8);
|
||||
|
||||
tk::dnn::Conv2d c12 (&net, 256, 3, 3, 2, 2, 1, 1, c12_bin, true);
|
||||
tk::dnn::Activation a12 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c13 (&net, 128, 1, 1, 1, 1, 0, 0, c13_bin, true);
|
||||
tk::dnn::Activation a13 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c14 (&net, 256, 3, 3, 1, 1, 1, 1, c14_bin, true);
|
||||
tk::dnn::Activation a14 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s15 (&net, &a12);
|
||||
|
||||
tk::dnn::Conv2d c16 (&net, 128, 1, 1, 1, 1, 0, 0, c16_bin, true);
|
||||
tk::dnn::Activation a16 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c17 (&net, 256, 3, 3, 1, 1, 1, 1, c17_bin, true);
|
||||
tk::dnn::Activation a17 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s18 (&net, &s15);
|
||||
tk::dnn::Conv2d c19 (&net, 128, 1, 1, 1, 1, 0, 0, c19_bin, true);
|
||||
tk::dnn::Activation a19 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c20 (&net, 256, 3, 3, 1, 1, 1, 1, c20_bin, true);
|
||||
tk::dnn::Activation a20 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s21 (&net, &s18);
|
||||
tk::dnn::Conv2d c22 (&net, 128, 1, 1, 1, 1, 0, 0, c22_bin, true);
|
||||
tk::dnn::Activation a22 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c23 (&net, 256, 3, 3, 1, 1, 1, 1, c23_bin, true);
|
||||
tk::dnn::Activation a23 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s24 (&net, &s21);
|
||||
tk::dnn::Conv2d c25 (&net, 128, 1, 1, 1, 1, 0, 0, c25_bin, true);
|
||||
tk::dnn::Activation a25 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c26 (&net, 256, 3, 3, 1, 1, 1, 1, c26_bin, true);
|
||||
tk::dnn::Activation a26 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s27 (&net, &s24);
|
||||
tk::dnn::Conv2d c28 (&net, 128, 1, 1, 1, 1, 0, 0, c28_bin, true);
|
||||
tk::dnn::Activation a28 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c29 (&net, 256, 3, 3, 1, 1, 1, 1, c29_bin, true);
|
||||
tk::dnn::Activation a29 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s30 (&net, &s27);
|
||||
tk::dnn::Conv2d c31 (&net, 128, 1, 1, 1, 1, 0, 0, c31_bin, true);
|
||||
tk::dnn::Activation a31 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c32 (&net, 256, 3, 3, 1, 1, 1, 1, c32_bin, true);
|
||||
tk::dnn::Activation a32 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s33 (&net, &s30);
|
||||
tk::dnn::Conv2d c34 (&net, 128, 1, 1, 1, 1, 0, 0, c34_bin, true);
|
||||
tk::dnn::Activation a34 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c35 (&net, 256, 3, 3, 1, 1, 1, 1, c35_bin, true);
|
||||
tk::dnn::Activation a35 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s36 (&net, &s33);
|
||||
|
||||
tk::dnn::Conv2d c37 (&net, 512, 3, 3, 2, 2, 1, 1, c37_bin, true);
|
||||
tk::dnn::Activation a37 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c38 (&net, 256, 1, 1, 1, 1, 0, 0, c38_bin, true);
|
||||
tk::dnn::Activation a38 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c39 (&net, 512, 3, 3, 1, 1, 1, 1, c39_bin, true);
|
||||
tk::dnn::Activation a39 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s40 (&net, &a37);
|
||||
|
||||
tk::dnn::Conv2d c41 (&net, 256, 1, 1, 1, 1, 0, 0, c41_bin, true);
|
||||
tk::dnn::Activation a41 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c42 (&net, 512, 3, 3, 1, 1, 1, 1, c42_bin, true);
|
||||
tk::dnn::Activation a42 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s43 (&net, &s40);
|
||||
tk::dnn::Conv2d c44 (&net, 256, 1, 1, 1, 1, 0, 0, c44_bin, true);
|
||||
tk::dnn::Activation a44 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c45 (&net, 512, 3, 3, 1, 1, 1, 1, c45_bin, true);
|
||||
tk::dnn::Activation a45 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s46 (&net, &s43);
|
||||
tk::dnn::Conv2d c47 (&net, 256, 1, 1, 1, 1, 0, 0, c47_bin, true);
|
||||
tk::dnn::Activation a47 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c48 (&net, 512, 3, 3, 1, 1, 1, 1, c48_bin, true);
|
||||
tk::dnn::Activation a48 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s49 (&net, &s46);
|
||||
tk::dnn::Conv2d c50 (&net, 256, 1, 1, 1, 1, 0, 0, c50_bin, true);
|
||||
tk::dnn::Activation a50 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c51 (&net, 512, 3, 3, 1, 1, 1, 1, c51_bin, true);
|
||||
tk::dnn::Activation a51 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s52 (&net, &s49);
|
||||
tk::dnn::Conv2d c53 (&net, 256, 1, 1, 1, 1, 0, 0, c53_bin, true);
|
||||
tk::dnn::Activation a53 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c54 (&net, 512, 3, 3, 1, 1, 1, 1, c54_bin, true);
|
||||
tk::dnn::Activation a54 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s55 (&net, &s52);
|
||||
tk::dnn::Conv2d c56 (&net, 256, 1, 1, 1, 1, 0, 0, c56_bin, true);
|
||||
tk::dnn::Activation a56 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c57 (&net, 512, 3, 3, 1, 1, 1, 1, c57_bin, true);
|
||||
tk::dnn::Activation a57 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s58 (&net, &s55);
|
||||
tk::dnn::Conv2d c59 (&net, 256, 1, 1, 1, 1, 0, 0, c59_bin, true);
|
||||
tk::dnn::Activation a59 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c60 (&net, 512, 3, 3, 1, 1, 1, 1, c60_bin, true);
|
||||
tk::dnn::Activation a60 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s61 (&net, &s58);
|
||||
|
||||
tk::dnn::Conv2d c62 (&net,1024, 3, 3, 2, 2, 1, 1, c62_bin, true);
|
||||
tk::dnn::Activation a62 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c63 (&net, 512, 1, 1, 1, 1, 0, 0, c63_bin, true);
|
||||
tk::dnn::Activation a63 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c64 (&net,1024, 3, 3, 1, 1, 1, 1, c64_bin, true);
|
||||
tk::dnn::Activation a64 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s65 (&net, &a62);
|
||||
|
||||
tk::dnn::Conv2d c66 (&net, 512, 1, 1, 1, 1, 0, 0, c66_bin, true);
|
||||
tk::dnn::Activation a66 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c67 (&net,1024, 3, 3, 1, 1, 1, 1, c67_bin, true);
|
||||
tk::dnn::Activation a67 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s68 (&net, &s65);
|
||||
|
||||
tk::dnn::Conv2d c69 (&net, 512, 1, 1, 1, 1, 0, 0, c69_bin, true);
|
||||
tk::dnn::Activation a69 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c70 (&net,1024, 3, 3, 1, 1, 1, 1, c70_bin, true);
|
||||
tk::dnn::Activation a70 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s71 (&net, &s68);
|
||||
|
||||
tk::dnn::Conv2d c72 (&net, 512, 1, 1, 1, 1, 0, 0, c72_bin, true);
|
||||
tk::dnn::Activation a72 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c73 (&net,1024, 3, 3, 1, 1, 1, 1, c73_bin, true);
|
||||
tk::dnn::Activation a73 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Shortcut s74 (&net, &s71);
|
||||
|
||||
tk::dnn::Conv2d c75 (&net, 512, 1, 1, 1, 1, 0, 0, c75_bin, true);
|
||||
tk::dnn::Activation a75 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c76 (&net,1024, 3, 3, 1, 1, 1, 1, c76_bin, true);
|
||||
tk::dnn::Activation a76 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c77 (&net, 512, 1, 1, 1, 1, 0, 0, c77_bin, true);
|
||||
tk::dnn::Activation a77 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c78 (&net,1024, 3, 3, 1, 1, 1, 1, c78_bin, true);
|
||||
tk::dnn::Activation a78 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c79 (&net, 512, 1, 1, 1, 1, 0, 0, c79_bin, true);
|
||||
tk::dnn::Activation a79 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c80 (&net,1024, 3, 3, 1, 1, 1, 1, c80_bin, true);
|
||||
tk::dnn::Activation a80 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c81 (&net, preYoloFilters, 1, 1, 1, 1, 0, 0, c81_bin, false);
|
||||
tk::dnn::Yolo yolo0 (&net, classes, 3, g82_bin);
|
||||
|
||||
tk::dnn::Layer *m83_layers[1] = { &a79 };
|
||||
tk::dnn::Route m83 (&net, m83_layers, 1);
|
||||
tk::dnn::Conv2d c84 (&net, 256, 1, 1, 1, 1, 0, 0, c84_bin, true);
|
||||
tk::dnn::Activation a84 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Upsample u85 (&net, 2);
|
||||
|
||||
tk::dnn::Layer *m86_layers[2] = { &u85, &s61 };
|
||||
tk::dnn::Route m86 (&net, m86_layers, 2);
|
||||
tk::dnn::Conv2d c87 (&net, 256, 1, 1, 1, 1, 0, 0, c87_bin, true);
|
||||
tk::dnn::Activation a87 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c88 (&net, 512, 3, 3, 1, 1, 1, 1, c88_bin, true);
|
||||
tk::dnn::Activation a88 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c89 (&net, 256, 1, 1, 1, 1, 0, 0, c89_bin, true);
|
||||
tk::dnn::Activation a89 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c90 (&net, 512, 3, 3, 1, 1, 1, 1, c90_bin, true);
|
||||
tk::dnn::Activation a90 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c91 (&net, 256, 1, 1, 1, 1, 0, 0, c91_bin, true);
|
||||
tk::dnn::Activation a91 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
|
||||
tk::dnn::Conv2d c92 (&net, 512, 3, 3, 1, 1, 1, 1, c92_bin, true);
|
||||
tk::dnn::Activation a92 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c93 (&net, preYoloFilters, 1, 1, 1, 1, 0, 0, c93_bin, false);
|
||||
tk::dnn::Yolo yolo1 (&net, classes, 3, g94_bin);
|
||||
|
||||
tk::dnn::Layer *m95_layers[1] = { &a91 };
|
||||
tk::dnn::Route m95 (&net, m95_layers, 1);
|
||||
tk::dnn::Conv2d c96 (&net, 128, 1, 1, 1, 1, 0, 0, c96_bin, true);
|
||||
tk::dnn::Activation a96 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Upsample u97 (&net, 2);
|
||||
|
||||
tk::dnn::Layer *m98_layers[2] = { &u97, &s36 };
|
||||
tk::dnn::Route m98 (&net, m98_layers, 2);
|
||||
tk::dnn::Conv2d c99 (&net, 128, 1, 1, 1, 1, 0, 0, c99_bin, true);
|
||||
tk::dnn::Activation a99 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c100 (&net, 256, 3, 3, 1, 1, 1, 1, c100_bin, true);
|
||||
tk::dnn::Activation a100 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c101 (&net, 128, 1, 1, 1, 1, 0, 0, c101_bin, true);
|
||||
tk::dnn::Activation a101 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c102 (&net, 256, 3, 3, 1, 1, 1, 1, c102_bin, true);
|
||||
tk::dnn::Activation a102 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c103 (&net, 128, 1, 1, 1, 1, 0, 0, c103_bin, true);
|
||||
tk::dnn::Activation a103 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
|
||||
tk::dnn::Conv2d c104 (&net, 256, 3, 3, 1, 1, 1, 1, c104_bin, true);
|
||||
tk::dnn::Activation a104 (&net, tk::dnn::ACTIVATION_LEAKY);
|
||||
tk::dnn::Conv2d c105 (&net, preYoloFilters, 1, 1, 1, 1, 0, 0, c105_bin, false);
|
||||
tk::dnn::Yolo yolo2 (&net, classes, 3, g106_bin);
|
||||
|
||||
yolo[0] = &yolo0;
|
||||
yolo[1] = &yolo1;
|
||||
yolo[2] = &yolo2;
|
||||
@@ -1,60 +1,88 @@
|
||||
#include<cassert>
|
||||
#include "NvInfer.h"
|
||||
#include "../kernels.h"
|
||||
#include <cassert>
|
||||
#include <vector>
|
||||
|
||||
class ActivationLeakyRT : public IPlugin {
|
||||
namespace nvinfer1 {
|
||||
class ActivationLeakyRT : public IPluginV2 {
|
||||
|
||||
public:
|
||||
ActivationLeakyRT() {
|
||||
public:
|
||||
explicit ActivationLeakyRT(float s);
|
||||
|
||||
ActivationLeakyRT(const void *data, size_t length);
|
||||
|
||||
~ActivationLeakyRT();
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override;
|
||||
|
||||
void
|
||||
configureWithFormat(const Dims *inputDims, int nbInputs, const Dims *outputDims, int nbOutputs, DataType type,
|
||||
PluginFormat format, int maxBatchSize) NOEXCEPT override;
|
||||
|
||||
int initialize() NOEXCEPT override;
|
||||
|
||||
void terminate() NOEXCEPT override {}
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override;
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, void const *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
void destroy() NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
IPluginV2 *clone() const NOEXCEPT override;
|
||||
|
||||
int size;
|
||||
float slope;
|
||||
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class ActivationLeakyRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ActivationLeakyRTPluginCreator();
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
IPluginV2 *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
|
||||
}
|
||||
|
||||
~ActivationLeakyRT(){
|
||||
|
||||
}
|
||||
|
||||
int getNbOutputs() const override {
|
||||
return 1;
|
||||
}
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||
return inputs[0];
|
||||
}
|
||||
|
||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||
size = 1;
|
||||
for(int i=0; i<outputDims[0].nbDims; i++)
|
||||
size *= outputDims[0].d[i];
|
||||
}
|
||||
|
||||
int initialize() override {
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual void terminate() override {
|
||||
}
|
||||
|
||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||
|
||||
activationLEAKYForward((dnnType*)reinterpret_cast<const dnnType*>(inputs[0]),
|
||||
reinterpret_cast<dnnType*>(outputs[0]), size, stream);
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
virtual size_t getSerializationSize() override {
|
||||
return 1*sizeof(int);
|
||||
}
|
||||
|
||||
virtual void serialize(void* buffer) override {
|
||||
char *buf = reinterpret_cast<char*>(buffer);
|
||||
tk::dnn::writeBUF(buf, size);
|
||||
}
|
||||
|
||||
int size;
|
||||
};
|
||||
REGISTER_TENSORRT_PLUGIN(ActivationLeakyRTPluginCreator);
|
||||
};
|
||||
@@ -0,0 +1,88 @@
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <utils.h>
|
||||
|
||||
namespace nvinfer1 {
|
||||
|
||||
class ActivationLogisticRT : public IPluginV2 {
|
||||
|
||||
public:
|
||||
ActivationLogisticRT() ;
|
||||
|
||||
ActivationLogisticRT(const void *data, size_t length) ;
|
||||
|
||||
~ActivationLogisticRT() ;
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
void configureWithFormat(const Dims *inputDims, int nbInputs, const Dims *outputDims, int nbOutputs, DataType type,
|
||||
PluginFormat format, int maxBatchSize) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override;
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *clone() const NOEXCEPT override ;
|
||||
|
||||
int size;
|
||||
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class ActivationLogisticRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ActivationLogisticRTPluginCreator() ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ActivationLogisticRTPluginCreator);
|
||||
};
|
||||
@@ -0,0 +1,82 @@
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
|
||||
namespace nvinfer1 {
|
||||
class ActivationMishRT : public IPluginV2 {
|
||||
|
||||
public:
|
||||
ActivationMishRT() ;
|
||||
|
||||
~ActivationMishRT() ;
|
||||
|
||||
ActivationMishRT(const void *data, size_t length) ;
|
||||
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
void configureWithFormat(const Dims *inputDims, int nbInputs, const Dims *outputDims, int nbOutputs, DataType type,
|
||||
PluginFormat format, int maxBatchSize) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override { delete this; }
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *plguinNamespace) NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *clone() const NOEXCEPT override ;
|
||||
|
||||
int size;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class ActivationMishRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ActivationMishRTPluginCreator() ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ActivationMishRTPluginCreator);
|
||||
};
|
||||
@@ -0,0 +1,81 @@
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <utils.h>
|
||||
|
||||
namespace nvinfer1 {
|
||||
class ActivationReLUCeiling : public IPluginV2 {
|
||||
|
||||
public:
|
||||
explicit ActivationReLUCeiling(const float ceiling) ;
|
||||
|
||||
~ActivationReLUCeiling() ;
|
||||
|
||||
ActivationReLUCeiling(const void *data, size_t length) ;
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
void configureWithFormat(const Dims *inputDims, int nbInputs, const Dims *outputDims, int nbOutputs, DataType type,PluginFormat format, int maxBatchSize) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *clone() const NOEXCEPT override ;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
int size;
|
||||
float ceiling;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class ActivationReLUCeilingPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ActivationReLUCeilingPluginCreator() ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
public:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ActivationReLUCeilingPluginCreator);
|
||||
};
|
||||
@@ -0,0 +1,61 @@
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
|
||||
class ActivationSigmoidRT : public IPlugin {
|
||||
|
||||
public:
|
||||
ActivationSigmoidRT() {
|
||||
|
||||
|
||||
}
|
||||
|
||||
~ActivationSigmoidRT(){
|
||||
|
||||
}
|
||||
|
||||
int getNbOutputs() const override {
|
||||
return 1;
|
||||
}
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||
return inputs[0];
|
||||
}
|
||||
|
||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||
size = 1;
|
||||
for(int i=0; i<outputDims[0].nbDims; i++)
|
||||
size *= outputDims[0].d[i];
|
||||
}
|
||||
|
||||
int initialize() override {
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual void terminate() override {
|
||||
}
|
||||
|
||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||
|
||||
activationSIGMOIDForward((dnnType*)reinterpret_cast<const dnnType*>(inputs[0]),
|
||||
reinterpret_cast<dnnType*>(outputs[0]), batchSize*size, stream);
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
virtual size_t getSerializationSize() override {
|
||||
return 1*sizeof(int);
|
||||
}
|
||||
|
||||
virtual void serialize(void* buffer) override {
|
||||
char *buf = reinterpret_cast<char*>(buffer),*a=buf;
|
||||
tk::dnn::writeBUF(buf, size);
|
||||
assert(buf == a + getSerializationSize());
|
||||
}
|
||||
|
||||
int size;
|
||||
};
|
||||
@@ -0,0 +1,109 @@
|
||||
//
|
||||
// Created by perseusdg on 1/7/22.
|
||||
//
|
||||
|
||||
#ifndef _CONSTANTPADDINGRT_PLUGIN_H
|
||||
#define _CONSTANTPADDINGRT_PLUGIN_H
|
||||
|
||||
#include<cassert>
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <utils.h>
|
||||
#include <kernels.h>
|
||||
|
||||
namespace nvinfer1{
|
||||
class ConstantPaddingRT : public IPluginV2Ext {
|
||||
public:
|
||||
ConstantPaddingRT(int32_t padH,int32_t padW,int32_t n,int32_t c,int32_t i_h,int32_t i_w,int32_t o_h,int32_t o_w,float constant);
|
||||
|
||||
ConstantPaddingRT(const void *data,size_t length);
|
||||
|
||||
~ConstantPaddingRT();
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace, cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR <= 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
bool supportsFormat (DataType type, PluginFormat format) const NOEXCEPT override;
|
||||
|
||||
int32_t i_h,i_w,o_h,o_w,n,c,padH,padW;
|
||||
float constant;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
|
||||
};
|
||||
|
||||
class ConstantPaddingRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ConstantPaddingRTPluginCreator();
|
||||
|
||||
void setPluginNamespace(const char* pluginNamespace) NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ConstantPaddingRTPluginCreator);
|
||||
};
|
||||
|
||||
|
||||
#endif //TKDNN_CONSTANTPADDINGRT_H
|
||||
@@ -0,0 +1,137 @@
|
||||
#ifndef _DEFORMABLECONVRT_PLUGIN_H
|
||||
#define _DEFORMABLECONVRT_PLUGIN_H
|
||||
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <tkdnn.h>
|
||||
|
||||
namespace nvinfer1 {
|
||||
class DeformableConvRT : public IPluginV2Ext {
|
||||
|
||||
|
||||
public:
|
||||
DeformableConvRT(int chunk_dim, int kh, int kw, int sh, int sw, int ph, int pw,
|
||||
int deformableGroup, int i_n, int i_c, int i_h, int i_w,
|
||||
int o_n, int o_c, int o_h, int o_w,std::vector<dnnType> data_H,std::vector<dnnType> bias2_H,
|
||||
std::vector<dnnType> ones_d1_h,std::vector<dnnType> ones_d2_h,std::vector<dnnType> offsetH,std::vector<dnnType> maskH,int height_ones,
|
||||
int width_ones,int dim_ones);
|
||||
|
||||
~DeformableConvRT();
|
||||
|
||||
DeformableConvRT(const void *data, size_t length) ;
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
|
||||
cublasStatus_t stat;
|
||||
cublasHandle_t handle{nullptr};
|
||||
int i_n, i_c, i_h, i_w;
|
||||
int o_n, o_c, o_h, o_w;
|
||||
int size;
|
||||
int chunk_dim;
|
||||
int kh, kw;
|
||||
int sh, sw;
|
||||
int ph, pw;
|
||||
int deformableGroup;
|
||||
int height_ones;
|
||||
int width_ones;
|
||||
int dim_ones;
|
||||
|
||||
std::vector<dnnType> data_d_v;
|
||||
std::vector<dnnType> bias2_d_v;
|
||||
std::vector<dnnType> ones_d1_v;
|
||||
std::vector<dnnType> offset_v;
|
||||
std::vector<dnnType> mask_v;
|
||||
std::vector<dnnType> ones_d2_v;
|
||||
dnnType* data_d;
|
||||
dnnType* bias2_d;
|
||||
dnnType* ones_d1;
|
||||
dnnType* offset;
|
||||
dnnType* mask;
|
||||
dnnType* ones_d2;
|
||||
// dnnType *input_n;
|
||||
// dnnType *offset_n;
|
||||
// dnnType *mask_n;
|
||||
// dnnType *output_n;
|
||||
|
||||
|
||||
tk::dnn::DeformConv2d *defRT;
|
||||
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class DeformableConvRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
DeformableConvRTPluginCreator();
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
REGISTER_TENSORRT_PLUGIN(DeformableConvRTPluginCreator);
|
||||
};
|
||||
#endif
|
||||
@@ -0,0 +1,100 @@
|
||||
#ifndef _FLATTENCONCATRT_PLUGIN_H
|
||||
#define _FLATTENCONCATRT_PLUGIN_H
|
||||
|
||||
#include<cassert>
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <utils.h>
|
||||
namespace nvinfer1 {
|
||||
class FlattenConcatRT : public IPluginV2Ext {
|
||||
|
||||
public:
|
||||
FlattenConcatRT(int c,int h,int w,int rows,int cols) ;
|
||||
|
||||
FlattenConcatRT(const void *data, size_t length) ;
|
||||
|
||||
~FlattenConcatRT() ;
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace, cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR <= 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
bool supportsFormat (DataType type, PluginFormat format) const NOEXCEPT override;
|
||||
|
||||
int c, h, w;
|
||||
int rows, cols;
|
||||
cublasHandle_t handle{nullptr};
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class FlattenConcatRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
FlattenConcatRTPluginCreator() ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(FlattenConcatRTPluginCreator);
|
||||
};
|
||||
#endif
|
||||
@@ -0,0 +1,105 @@
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <utils.h>
|
||||
|
||||
|
||||
namespace nvinfer1 {
|
||||
class MaxPoolFixedSizeRT : public IPluginV2Ext {
|
||||
|
||||
public:
|
||||
MaxPoolFixedSizeRT(int c, int h, int w, int n, int strideH, int strideW, int winSize, int padding) ;
|
||||
|
||||
MaxPoolFixedSizeRT(const void *data, size_t length) ;
|
||||
|
||||
~MaxPoolFixedSizeRT() ;
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR <= 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
|
||||
int n, c, h, w;
|
||||
int stride_H, stride_W;
|
||||
int winSize;
|
||||
int padding;
|
||||
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class MaxPoolFixedSizeRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
MaxPoolFixedSizeRTPluginCreator() ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(MaxPoolFixedSizeRTPluginCreator);
|
||||
};
|
||||
@@ -0,0 +1,101 @@
|
||||
#ifndef _REFLECTIONPADDINGRT_PLUGIN_H
|
||||
#define _REFLECTIONPADDINGRT_PLUGIN_H
|
||||
|
||||
#include<cassert>
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <utils.h>
|
||||
#include <kernels.h>
|
||||
|
||||
namespace nvinfer1{
|
||||
class ReflectionPaddingRT : public IPluginV2Ext {
|
||||
public:
|
||||
ReflectionPaddingRT(int32_t padH,int32_t padW,int32_t input_h,int32_t input_w,int32_t output_h,int32_t output_w,int32_t c,int32_t n);
|
||||
|
||||
ReflectionPaddingRT(const void *data,size_t length);
|
||||
|
||||
~ReflectionPaddingRT();
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace, cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR <= 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
bool supportsFormat (DataType type, PluginFormat format) const NOEXCEPT override;
|
||||
|
||||
int32_t padH,padW,input_h,input_w,output_h,output_w,n,c;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
|
||||
};
|
||||
|
||||
class ReflectionPaddingRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ReflectionPaddingRTPluginCreator();
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ReflectionPaddingRTPluginCreator);
|
||||
};
|
||||
#endif
|
||||
|
||||
@@ -1,94 +1,110 @@
|
||||
#ifndef _REGIONRT_PLUGIN_H
|
||||
#define _REGIONRT_PLUGIN_H
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <utils.h>
|
||||
|
||||
class RegionRT : public IPlugin {
|
||||
namespace nvinfer1 {
|
||||
class RegionRT : public IPluginV2Ext {
|
||||
|
||||
public:
|
||||
RegionRT(int classes, int coords, int num) {
|
||||
public:
|
||||
RegionRT(int classes, int coords, int num,int c,int h,int w);
|
||||
|
||||
this->classes = classes;
|
||||
this->coords = coords;
|
||||
this->num = num;
|
||||
}
|
||||
~RegionRT() ;
|
||||
|
||||
~RegionRT(){
|
||||
RegionRT(const void *data, size_t length) ;
|
||||
|
||||
}
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
int getNbOutputs() const override {
|
||||
return 1;
|
||||
}
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||
return inputs[0];
|
||||
}
|
||||
|
||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||
c = inputDims[0].d[0];
|
||||
h = inputDims[0].d[1];
|
||||
w = inputDims[0].d[2];
|
||||
}
|
||||
|
||||
int initialize() override {
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual void terminate() override {
|
||||
}
|
||||
|
||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||
|
||||
dnnType *srcData = (dnnType*)reinterpret_cast<const dnnType*>(inputs[0]);
|
||||
dnnType *dstData = reinterpret_cast<dnnType*>(outputs[0]);
|
||||
|
||||
checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream));
|
||||
|
||||
for (int b = 0; b < batchSize; ++b){
|
||||
for(int n = 0; n < num; ++n){
|
||||
int index = entry_index(b, n*w*h, 0, batchSize);
|
||||
activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream);
|
||||
|
||||
index = entry_index(b, n*w*h, coords, batchSize);
|
||||
activationLOGISTICForward(srcData + index, dstData + index, w*h, stream);
|
||||
}
|
||||
}
|
||||
|
||||
//softmax start
|
||||
int index = entry_index(0, 0, coords + 1, batchSize);
|
||||
softmaxForward( srcData + index, classes, batchSize*num,
|
||||
(batchSize*c*h*w)/num,
|
||||
w*h, 1, w*h, 1, dstData + index, stream);
|
||||
|
||||
return 0;
|
||||
}
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
|
||||
virtual size_t getSerializationSize() override {
|
||||
return 6*sizeof(int);
|
||||
}
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
virtual void serialize(void* buffer) override {
|
||||
char *buf = reinterpret_cast<char*>(buffer);
|
||||
tk::dnn::writeBUF(buf, classes);
|
||||
tk::dnn::writeBUF(buf, coords);
|
||||
tk::dnn::writeBUF(buf, num);
|
||||
tk::dnn::writeBUF(buf, c);
|
||||
tk::dnn::writeBUF(buf, h);
|
||||
tk::dnn::writeBUF(buf, w);
|
||||
}
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
|
||||
int c, h, w;
|
||||
int classes, coords, num;
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
int entry_index(int batch, int location, int entry, int batchSize) {
|
||||
int n = location / (w*h);
|
||||
int loc = location % (w*h);
|
||||
return batch*c*h*w*batchSize + n*w*h*(coords+classes+1) + entry*w*h + loc;
|
||||
}
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
int c, h, w;
|
||||
int classes, coords, num;
|
||||
|
||||
int entry_index(int batch, int location, int entry) {
|
||||
int n = location / (w * h);
|
||||
int loc = location % (w * h);
|
||||
return batch * c * h * w + n * w * h * (coords + classes + 1) + entry * w * h + loc;
|
||||
}
|
||||
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class RegionRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
RegionRTPluginCreator();
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(RegionRTPluginCreator);
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,63 +1,98 @@
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
|
||||
class ReorgRT : public IPlugin {
|
||||
namespace nvinfer1 {
|
||||
class ReorgRT : public IPluginV2Ext {
|
||||
|
||||
public:
|
||||
ReorgRT(int stride) {
|
||||
this->stride = stride;
|
||||
}
|
||||
public:
|
||||
ReorgRT(int stride,int c,int h,int w);
|
||||
|
||||
~ReorgRT(){
|
||||
~ReorgRT();
|
||||
|
||||
}
|
||||
ReorgRT(const void *data, size_t length);
|
||||
|
||||
int getNbOutputs() const override {
|
||||
return 1;
|
||||
}
|
||||
int getNbOutputs() const NOEXCEPT override;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||
return DimsCHW{inputs[0].d[0]*stride*stride, inputs[0].d[1]/stride, inputs[0].d[2]/stride};
|
||||
}
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override;
|
||||
|
||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||
c = inputDims[0].d[0];
|
||||
h = inputDims[0].d[1];
|
||||
w = inputDims[0].d[2];
|
||||
}
|
||||
int initialize() NOEXCEPT override;
|
||||
|
||||
int initialize() override {
|
||||
void terminate() NOEXCEPT override;
|
||||
|
||||
return 0;
|
||||
}
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override;
|
||||
|
||||
virtual void terminate() override {
|
||||
}
|
||||
|
||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||
|
||||
reorgForward((dnnType*)reinterpret_cast<const dnnType*>(inputs[0]),
|
||||
reinterpret_cast<dnnType*>(outputs[0]),
|
||||
batchSize, c, h, w, stride, stream);
|
||||
return 0;
|
||||
}
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
|
||||
virtual size_t getSerializationSize() override {
|
||||
return 4*sizeof(int);
|
||||
}
|
||||
size_t getSerializationSize() const NOEXCEPT override;
|
||||
|
||||
virtual void serialize(void* buffer) override {
|
||||
char *buf = reinterpret_cast<char*>(buffer);
|
||||
tk::dnn::writeBUF(buf, stride);
|
||||
tk::dnn::writeBUF(buf, c);
|
||||
tk::dnn::writeBUF(buf, h);
|
||||
tk::dnn::writeBUF(buf, w);
|
||||
}
|
||||
void serialize(void *buffer) const NOEXCEPT override;
|
||||
|
||||
int c, h, w, stride;
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
void destroy() NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
int c, h, w, stride;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class ReorgRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ReorgRTPluginCreator();
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ReorgRTPluginCreator);
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,101 @@
|
||||
#ifndef _RESHAPERT_PLUGIN_H
|
||||
#define _RESHAPERT_PLUGIN_H
|
||||
|
||||
#include<cassert>
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <tkdnn.h>
|
||||
|
||||
|
||||
namespace nvinfer1 {
|
||||
class ReshapeRT : public IPluginV2Ext {
|
||||
|
||||
public:
|
||||
ReshapeRT(int n,int c,int h,int w) ;
|
||||
|
||||
ReshapeRT(const void *data, size_t length) ;
|
||||
|
||||
~ReshapeRT() ;
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace, cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
int n, c, h, w;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class ReshapeRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ReshapeRTPluginCreator() ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ReshapeRTPluginCreator);
|
||||
};
|
||||
#endif
|
||||
@@ -0,0 +1,104 @@
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <utils.h>
|
||||
|
||||
namespace nvinfer1 {
|
||||
|
||||
class ResizeLayerRT : public IPluginV2Ext {
|
||||
|
||||
public:
|
||||
ResizeLayerRT(int oc, int oh, int ow,int ic,int ih,int iw) ;
|
||||
|
||||
ResizeLayerRT(const void *data, size_t length) ;
|
||||
|
||||
~ResizeLayerRT() ;
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR <= 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
int i_c, i_h, i_w, o_c, o_h, o_w;
|
||||
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class ResizeLayerRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ResizeLayerRTPluginCreator() ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
|
||||
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ResizeLayerRTPluginCreator);
|
||||
};
|
||||
|
||||
@@ -1,82 +1,90 @@
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <vector>
|
||||
#include <NvInfer.h>
|
||||
|
||||
class RouteRT : public IPlugin {
|
||||
namespace nvinfer1 {
|
||||
class RouteRT : public IPluginV2 {
|
||||
|
||||
public:
|
||||
RouteRT() {
|
||||
}
|
||||
/**
|
||||
THIS IS NOT USED ANYMORE
|
||||
*/
|
||||
|
||||
~RouteRT(){
|
||||
public:
|
||||
RouteRT(int groups, int group_id) ;
|
||||
|
||||
}
|
||||
~RouteRT() ;
|
||||
|
||||
int getNbOutputs() const override {
|
||||
return 1;
|
||||
}
|
||||
RouteRT(const void *data, size_t length) ;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||
int out_c = 0;
|
||||
for(int i=0; i<nbInputDims; i++) out_c += inputs[i].d[0];
|
||||
return DimsCHW{out_c, inputs[0].d[1], inputs[0].d[2]};
|
||||
}
|
||||
int getNbOutputs() const NOEXCEPT override ;
|
||||
|
||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||
in = nbInputs;
|
||||
c = 0;
|
||||
for(int i=0; i<nbInputs; i++) {
|
||||
c_in[i] = inputDims[i].d[0];
|
||||
c += inputDims[i].d[0];
|
||||
}
|
||||
h = inputDims[0].d[1];
|
||||
w = inputDims[0].d[2];
|
||||
}
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override ;
|
||||
|
||||
int initialize() override {
|
||||
void configureWithFormat(const Dims *inputDims, int nbInputs, const Dims *outputDims, int nbOutputs, DataType type,PluginFormat format, int maxBatchSize) NOEXCEPT override ;
|
||||
|
||||
return 0;
|
||||
}
|
||||
int initialize() NOEXCEPT override ;
|
||||
|
||||
virtual void terminate() override {
|
||||
}
|
||||
void terminate() NOEXCEPT override ;
|
||||
|
||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||
return 0;
|
||||
}
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override ;
|
||||
|
||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,cudaStream_t stream) NOEXCEPT override ;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
dnnType *dstData = reinterpret_cast<dnnType*>(outputs[0]);
|
||||
size_t getSerializationSize() const NOEXCEPT override ;
|
||||
|
||||
int offset = 0;
|
||||
for(int i=0; i<in; i++) {
|
||||
dnnType *input = (dnnType*)reinterpret_cast<const dnnType*>(inputs[i]);
|
||||
int in_dim = c_in[i]*h*w;
|
||||
checkCuda( cudaMemcpyAsync(dstData + offset, input, in_dim*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream) );
|
||||
offset += in_dim;
|
||||
}
|
||||
void serialize(void *buffer) const NOEXCEPT override ;
|
||||
|
||||
return 0;
|
||||
}
|
||||
const char *getPluginType() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
virtual size_t getSerializationSize() override {
|
||||
return (4+MAX_INPUTS)*sizeof(int);
|
||||
}
|
||||
void destroy() NOEXCEPT override ;
|
||||
|
||||
virtual void serialize(void* buffer) override {
|
||||
char *buf = reinterpret_cast<char*>(buffer);
|
||||
tk::dnn::writeBUF(buf, in);
|
||||
for(int i=0; i<MAX_INPUTS; i++)
|
||||
tk::dnn::writeBUF(buf, c_in[i]);
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
tk::dnn::writeBUF(buf, c);
|
||||
tk::dnn::writeBUF(buf, h);
|
||||
tk::dnn::writeBUF(buf, w);
|
||||
}
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
static const int MAX_INPUTS = 4;
|
||||
int in;
|
||||
int c_in[MAX_INPUTS];
|
||||
int c, h, w;
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *clone() const NOEXCEPT override ;
|
||||
|
||||
static const int MAX_INPUTS = 4;
|
||||
int in;
|
||||
int c_in[MAX_INPUTS];
|
||||
int c, h, w;
|
||||
int groups, group_id;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class RouteRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
RouteRTPluginCreator() ;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override ;
|
||||
|
||||
IPluginV2 *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override ;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override ;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override ;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override ;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(RouteRTPluginCreator);
|
||||
};
|
||||
|
||||
@@ -1,65 +1,109 @@
|
||||
#ifndef _SHORTCUTRT_PLUGIN_H
|
||||
#define _SHORTCUTRT_PLUGIN_H
|
||||
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
|
||||
class ShortcutRT : public IPlugin {
|
||||
|
||||
public:
|
||||
ShortcutRT() {
|
||||
}
|
||||
|
||||
~ShortcutRT(){
|
||||
|
||||
}
|
||||
|
||||
int getNbOutputs() const override {
|
||||
return 1;
|
||||
}
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||
return DimsCHW{inputs[0].d[0], inputs[0].d[1], inputs[0].d[2]};
|
||||
}
|
||||
|
||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||
c = inputDims[0].d[0];
|
||||
h = inputDims[0].d[1];
|
||||
w = inputDims[0].d[2];
|
||||
}
|
||||
|
||||
int initialize() override {
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual void terminate() override {
|
||||
}
|
||||
|
||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||
|
||||
dnnType *srcData = (dnnType*)reinterpret_cast<const dnnType*>(inputs[0]);
|
||||
dnnType *srcDataBack = (dnnType*)reinterpret_cast<const dnnType*>(inputs[1]);
|
||||
dnnType *dstData = reinterpret_cast<dnnType*>(outputs[0]);
|
||||
|
||||
checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream));
|
||||
shortcutForward(srcDataBack, dstData, batchSize, c, h, w, 1, batchSize, c, h, w, 1, stream);
|
||||
|
||||
return 0;
|
||||
}
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
#include <tkdnn.h>
|
||||
|
||||
|
||||
virtual size_t getSerializationSize() override {
|
||||
return 3*sizeof(int);
|
||||
}
|
||||
namespace nvinfer1 {
|
||||
|
||||
virtual void serialize(void* buffer) override {
|
||||
char *buf = reinterpret_cast<char*>(buffer);
|
||||
tk::dnn::writeBUF(buf, c);
|
||||
tk::dnn::writeBUF(buf, h);
|
||||
tk::dnn::writeBUF(buf, w);
|
||||
}
|
||||
class ShortcutRT : public IPluginV2Ext {
|
||||
|
||||
public:
|
||||
ShortcutRT(int bc,int bh,int bw,int c,int h,int w ,bool mul);
|
||||
|
||||
~ShortcutRT();
|
||||
|
||||
ShortcutRT(const void *data, size_t length);
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override;
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims, int32_t nbOutputs,
|
||||
DataType const *inputTypes, DataType const *outputTypes, bool const *inputIsBroadcast,
|
||||
bool const *outputIsBroadcast, PluginFormat floatFormat, int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch (int32_t outputIndex, bool const *inputIsBroadcasted, int32_t nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch (int32_t inputIndex) const NOEXCEPT override;
|
||||
|
||||
void attachToContext (cudnnContext *, cublasContext *, IGpuAllocator *) NOEXCEPT override;
|
||||
|
||||
void detachFromContext () NOEXCEPT override;
|
||||
|
||||
DataType getOutputDataType(int32_t index, nvinfer1::DataType const *inputTypes, int32_t nbInputs) const NOEXCEPT override;
|
||||
|
||||
int initialize() NOEXCEPT override;
|
||||
|
||||
void terminate() NOEXCEPT override;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override;
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
void destroy() NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override;
|
||||
|
||||
int c, h, w;
|
||||
int bc, bh, bw,bl;
|
||||
bool mul;
|
||||
tk::dnn::dataDim_t bDim;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
|
||||
class ShortcutRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
ShortcutRTPluginCreator();
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override;
|
||||
|
||||
public:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(ShortcutRTPluginCreator);
|
||||
|
||||
int c, h, w;
|
||||
};
|
||||
|
||||
#endif
|
||||
@@ -1,65 +1,103 @@
|
||||
#ifndef _UPSAMPLERT_PLUGIN_H
|
||||
#define _UPSAMPLERT_PLUGIN_H
|
||||
|
||||
#include<cassert>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
#include <vector>
|
||||
|
||||
class UpsampleRT : public IPlugin {
|
||||
namespace nvinfer1 {
|
||||
|
||||
public:
|
||||
UpsampleRT(int stride) {
|
||||
this->stride = stride;
|
||||
}
|
||||
class UpsampleRT : public IPluginV2Ext {
|
||||
|
||||
~UpsampleRT(){
|
||||
public:
|
||||
UpsampleRT(int stride,int c,int h,int w);
|
||||
|
||||
}
|
||||
UpsampleRT(const void *data, size_t length);
|
||||
|
||||
int getNbOutputs() const override {
|
||||
return 1;
|
||||
}
|
||||
~UpsampleRT();
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||
return DimsCHW(inputs[0].d[0], inputs[0].d[1]*stride, inputs[0].d[2]*stride);
|
||||
}
|
||||
int getNbOutputs() const NOEXCEPT override;
|
||||
|
||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||
c = inputDims[0].d[0];
|
||||
h = inputDims[0].d[1];
|
||||
w = inputDims[0].d[2];
|
||||
}
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override;
|
||||
|
||||
int initialize() override {
|
||||
int initialize() NOEXCEPT override;
|
||||
|
||||
return 0;
|
||||
}
|
||||
void terminate() NOEXCEPT override;
|
||||
|
||||
virtual void terminate() override {
|
||||
}
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override;
|
||||
|
||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||
return 0;
|
||||
}
|
||||
|
||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||
|
||||
dnnType *srcData = (dnnType*)reinterpret_cast<const dnnType*>(inputs[0]);
|
||||
dnnType *dstData = reinterpret_cast<dnnType*>(outputs[0]);
|
||||
|
||||
fill(dstData, batchSize*c*h*w*stride*stride, 0.0, stream);
|
||||
upsampleForward(srcData, dstData, batchSize, c, h, w, stride, 1, 1, stream);
|
||||
return 0;
|
||||
}
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
|
||||
virtual size_t getSerializationSize() override {
|
||||
return 4*sizeof(int);
|
||||
}
|
||||
size_t getSerializationSize() const NOEXCEPT override;
|
||||
|
||||
virtual void serialize(void* buffer) override {
|
||||
char *buf = reinterpret_cast<char*>(buffer);
|
||||
tk::dnn::writeBUF(buf, stride);
|
||||
tk::dnn::writeBUF(buf, c);
|
||||
tk::dnn::writeBUF(buf, h);
|
||||
tk::dnn::writeBUF(buf, w);
|
||||
}
|
||||
void serialize(void *buffer) const NOEXCEPT override;
|
||||
|
||||
int c, h, w, stride;
|
||||
};
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
void destroy() NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override ;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch (int32_t outputIndex, bool const *inputIsBroadcasted, int32_t nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch (int32_t inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims, int32_t nbOutputs,
|
||||
DataType const *inputTypes, DataType const *outputTypes, bool const *inputIsBroadcast,
|
||||
bool const *outputIsBroadcast, PluginFormat floatFormat, int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void attachToContext (cudnnContext *, cublasContext *, IGpuAllocator *) NOEXCEPT override;
|
||||
|
||||
void detachFromContext () NOEXCEPT override;
|
||||
|
||||
DataType getOutputDataType (int32_t index, nvinfer1::DataType const *inputTypes, int32_t nbInputs) const NOEXCEPT override;
|
||||
|
||||
|
||||
int c, h, w, stride;
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
class UpsampleRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
UpsampleRTPluginCreator();
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override;
|
||||
|
||||
const char *getPluginName() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override;
|
||||
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(UpsampleRTPluginCreator);
|
||||
};
|
||||
|
||||
#endif
|
||||
@@ -1,116 +1,125 @@
|
||||
#ifndef _YOLORT_PLUGIN_H
|
||||
#define _YOLORT_PLUGIN_H
|
||||
|
||||
#include<cassert>
|
||||
#include <vector>
|
||||
#include "../kernels.h"
|
||||
#include <NvInfer.h>
|
||||
|
||||
#define YOLORT_CLASSNAME_W 256
|
||||
|
||||
class YoloRT : public IPlugin {
|
||||
namespace nvinfer1 {
|
||||
class YoloRT : public IPluginV2Ext {
|
||||
|
||||
public:
|
||||
YoloRT(int classes, int num,int c,int h,int w, int n_masks = 3, float scale_xy = 1,
|
||||
float nms_thresh = 0.45, int nms_kind = 0, int new_coords = 0);
|
||||
|
||||
YoloRT(const void *data, size_t length);
|
||||
|
||||
~YoloRT();
|
||||
|
||||
|
||||
int getNbOutputs() const NOEXCEPT override;
|
||||
|
||||
public:
|
||||
YoloRT(int classes, int num, tk::dnn::Yolo *yolo = nullptr) {
|
||||
Dims getOutputDimensions(int index, const Dims *inputs, int nbInputDims) NOEXCEPT override;
|
||||
|
||||
this->classes = classes;
|
||||
this->num = num;
|
||||
int initialize() NOEXCEPT override;
|
||||
|
||||
mask = new dnnType[num];
|
||||
bias = new dnnType[num*3*2];
|
||||
if(yolo != nullptr) {
|
||||
memcpy(mask, yolo->mask_h, sizeof(dnnType)*num);
|
||||
memcpy(bias, yolo->bias_h, sizeof(dnnType)*num*3*2);
|
||||
classesNames = yolo->classesNames;
|
||||
void terminate() NOEXCEPT override;
|
||||
|
||||
size_t getWorkspaceSize(int maxBatchSize) const NOEXCEPT override;
|
||||
|
||||
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
int enqueue(int batchSize, const void *const *inputs, void *const *outputs, void *workspace,
|
||||
cudaStream_t stream) NOEXCEPT override;
|
||||
#elif NV_TENSORRT_MAJOR == 7
|
||||
int32_t enqueue (int32_t batchSize, const void *const *inputs, void **outputs, void *workspace, cudaStream_t stream) override;
|
||||
#endif
|
||||
|
||||
|
||||
size_t getSerializationSize() const NOEXCEPT override;
|
||||
|
||||
bool supportsFormat(DataType type, PluginFormat format) const NOEXCEPT override;
|
||||
|
||||
void serialize(void *buffer) const NOEXCEPT override;
|
||||
|
||||
const char *getPluginType() const NOEXCEPT override;
|
||||
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
void destroy() NOEXCEPT override;
|
||||
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
IPluginV2Ext *clone() const NOEXCEPT override;
|
||||
|
||||
DataType getOutputDataType(int index, const nvinfer1::DataType* inputTypes, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
void attachToContext(cudnnContext* cudnnContext, cublasContext* cublasContext, IGpuAllocator* gpuAllocator) NOEXCEPT override;
|
||||
|
||||
bool isOutputBroadcastAcrossBatch(int outputIndex, const bool* inputIsBroadcasted, int nbInputs) const NOEXCEPT override;
|
||||
|
||||
bool canBroadcastInputAcrossBatch(int inputIndex) const NOEXCEPT override;
|
||||
|
||||
void configurePlugin (Dims const *inputDims, int32_t nbInputs, Dims const *outputDims,
|
||||
int32_t nbOutputs, DataType const *inputTypes, DataType const *outputTypes,
|
||||
bool const *inputIsBroadcast, bool const *outputIsBroadcast, PluginFormat floatFormat,
|
||||
int32_t maxBatchSize) NOEXCEPT override;
|
||||
|
||||
void detachFromContext() NOEXCEPT override;
|
||||
|
||||
|
||||
int c, h, w;
|
||||
int classes, num, n_masks;
|
||||
float scaleXY;
|
||||
float nms_thresh;
|
||||
int nms_kind;
|
||||
int new_coords;
|
||||
|
||||
std::vector<std::string> classesNames;
|
||||
std::vector<dnnType> mask;
|
||||
std::vector<dnnType> bias;
|
||||
|
||||
|
||||
int entry_index(int batch, int location, int entry) {
|
||||
int n = location / (w * h);
|
||||
int loc = location % (w * h);
|
||||
return batch * c * h * w + n * w * h * (4 + classes + 1) + entry * w * h + loc;
|
||||
}
|
||||
}
|
||||
|
||||
~YoloRT(){
|
||||
private:
|
||||
std::string mPluginNamespace;
|
||||
|
||||
}
|
||||
};
|
||||
|
||||
int getNbOutputs() const override {
|
||||
return 1;
|
||||
}
|
||||
class YoloRTPluginCreator : public IPluginCreator {
|
||||
public:
|
||||
YoloRTPluginCreator();
|
||||
|
||||
Dims getOutputDimensions(int index, const Dims* inputs, int nbInputDims) override {
|
||||
return inputs[0];
|
||||
}
|
||||
void setPluginNamespace(const char *pluginNamespace) NOEXCEPT override;
|
||||
|
||||
void configure(const Dims* inputDims, int nbInputs, const Dims* outputDims, int nbOutputs, int maxBatchSize) override {
|
||||
c = inputDims[0].d[0];
|
||||
h = inputDims[0].d[1];
|
||||
w = inputDims[0].d[2];
|
||||
}
|
||||
const char *getPluginNamespace() const NOEXCEPT override;
|
||||
|
||||
int initialize() override {
|
||||
IPluginV2Ext *deserializePlugin(const char *name, const void *serialData, size_t serialLength) NOEXCEPT override;
|
||||
|
||||
return 0;
|
||||
}
|
||||
IPluginV2Ext *createPlugin(const char *name, const PluginFieldCollection *fc) NOEXCEPT override;
|
||||
|
||||
virtual void terminate() override {
|
||||
}
|
||||
const char *getPluginName() const NOEXCEPT override;
|
||||
|
||||
virtual size_t getWorkspaceSize(int maxBatchSize) const override {
|
||||
return 0;
|
||||
}
|
||||
const char *getPluginVersion() const NOEXCEPT override;
|
||||
|
||||
virtual int enqueue(int batchSize, const void*const * inputs, void** outputs, void* workspace, cudaStream_t stream) override {
|
||||
const PluginFieldCollection *getFieldNames() NOEXCEPT override;
|
||||
|
||||
dnnType *srcData = (dnnType*)reinterpret_cast<const dnnType*>(inputs[0]);
|
||||
dnnType *dstData = reinterpret_cast<dnnType*>(outputs[0]);
|
||||
|
||||
checkCuda( cudaMemcpyAsync(dstData, srcData, batchSize*c*h*w*sizeof(dnnType), cudaMemcpyDeviceToDevice, stream));
|
||||
|
||||
for (int b = 0; b < batchSize; ++b){
|
||||
for(int n = 0; n < num; ++n){
|
||||
int index = entry_index(b, n*w*h, 0, batchSize);
|
||||
activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream);
|
||||
|
||||
index = entry_index(b, n*w*h, 4, batchSize);
|
||||
activationLOGISTICForward(srcData + index, dstData + index, (1+classes)*w*h, stream);
|
||||
}
|
||||
}
|
||||
|
||||
//std::cout<<"YOLO END\n";
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
virtual size_t getSerializationSize() override {
|
||||
return 5*sizeof(int) + num*sizeof(dnnType) + num*3*2*sizeof(dnnType) + YOLORT_CLASSNAME_W*classes*sizeof(char);
|
||||
}
|
||||
|
||||
virtual void serialize(void* buffer) override {
|
||||
char *buf = reinterpret_cast<char*>(buffer);
|
||||
tk::dnn::writeBUF(buf, classes);
|
||||
tk::dnn::writeBUF(buf, num);
|
||||
tk::dnn::writeBUF(buf, c);
|
||||
tk::dnn::writeBUF(buf, h);
|
||||
tk::dnn::writeBUF(buf, w);
|
||||
for(int i=0; i<num; i++)
|
||||
tk::dnn::writeBUF(buf, mask[i]);
|
||||
for(int i=0; i<3*2*num; i++)
|
||||
tk::dnn::writeBUF(buf, bias[i]);
|
||||
|
||||
// save classes names
|
||||
for(int i=0; i<classes; i++) {
|
||||
char tmp[YOLORT_CLASSNAME_W];
|
||||
strcpy(tmp, classesNames[i].c_str());
|
||||
for(int j=0; j<YOLORT_CLASSNAME_W; j++) {
|
||||
tk::dnn::writeBUF(buf, tmp[j]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int c, h, w;
|
||||
int classes, num;
|
||||
std::vector<std::string> classesNames;
|
||||
|
||||
dnnType *mask;
|
||||
dnnType *bias;
|
||||
|
||||
int entry_index(int batch, int location, int entry, int batchSize) {
|
||||
int n = location / (w*h);
|
||||
int loc = location % (w*h);
|
||||
return batch*c*h*w*batchSize + n*w*h*(4+classes+1) + entry*w*h + loc;
|
||||
}
|
||||
private:
|
||||
static PluginFieldCollection mFC;
|
||||
static std::vector<PluginField> mPluginAttributes;
|
||||
std::string mPluginNamespace;
|
||||
};
|
||||
|
||||
REGISTER_TENSORRT_PLUGIN(YoloRTPluginCreator);
|
||||
};
|
||||
#endif
|
||||
@@ -0,0 +1,78 @@
|
||||
|
||||
#include <tkdnn.h>
|
||||
int testInference(std::vector<std::string> input_bins, std::vector<std::string> output_bins,
|
||||
tk::dnn::Network *net, tk::dnn::NetworkRT *netRT = nullptr) {
|
||||
|
||||
std::vector<tk::dnn::Layer*> outputs;
|
||||
for(int i=0; i<net->num_layers; i++) {
|
||||
if(net->layers[i]->final)
|
||||
outputs.push_back(net->layers[i]);
|
||||
}
|
||||
// no final layers, set last as output
|
||||
if(outputs.size() == 0) {
|
||||
outputs.push_back(net->layers[net->num_layers-1]);
|
||||
}
|
||||
|
||||
|
||||
// check input
|
||||
if(input_bins.size() != 1) {
|
||||
FatalError("currently support only 1 input");
|
||||
}
|
||||
if(output_bins.size() != outputs.size()) {
|
||||
std::cout<<output_bins.size()<<" "<<outputs.size()<<"\n";
|
||||
FatalError("outputs size mismatch");
|
||||
}
|
||||
|
||||
// Load input
|
||||
dnnType *data;
|
||||
dnnType *input_h;
|
||||
readBinaryFile(input_bins[0], net->input_dim.tot(), &input_h, &data);
|
||||
|
||||
// outputs
|
||||
//dnnType *cudnn_out[outputs.size()], *rt_out[outputs.size()];
|
||||
std::vector<dnnType *> cudnn_out,rt_out;
|
||||
|
||||
tk::dnn::dataDim_t dim1 = net->input_dim; //input dim
|
||||
printCenteredTitle(" CUDNN inference ", '=', 30); {
|
||||
dim1.print();
|
||||
TKDNN_TSTART
|
||||
net->infer(dim1, data);
|
||||
TKDNN_TSTOP
|
||||
dim1.print();
|
||||
}
|
||||
for(int i=0; i<outputs.size(); i++) cudnn_out.push_back(outputs[i]->dstData);
|
||||
|
||||
if(netRT != nullptr) {
|
||||
tk::dnn::dataDim_t dim2 = net->input_dim;
|
||||
printCenteredTitle(" TENSORRT inference ", '=', 30); {
|
||||
dim2.print();
|
||||
TKDNN_TSTART
|
||||
netRT->infer(dim2, data);
|
||||
TKDNN_TSTOP
|
||||
dim2.print();
|
||||
}
|
||||
for(int i=0; i<outputs.size(); i++) rt_out.push_back((dnnType*)netRT->buffersRT[i+1]);
|
||||
}
|
||||
|
||||
int ret_cudnn = 0, ret_tensorrt = 0, ret_cudnn_tensorrt = 0;
|
||||
for(int i=0; i<outputs.size(); i++) {
|
||||
printCenteredTitle((std::string(" OUTPUT ") + std::to_string(i) + " CHECK RESULTS ").c_str(), '=', 30);
|
||||
dnnType *out, *out_h;
|
||||
int odim = outputs[i]->output_dim.tot();
|
||||
readBinaryFile(output_bins[i], odim, &out_h, &out);
|
||||
std::cout<<"CUDNN vs correct";
|
||||
ret_cudnn |= checkResult(odim, cudnn_out[i], out) == 0 ? 0: ERROR_CUDNN;
|
||||
if(netRT != nullptr) {
|
||||
std::cout<<"TRT vs correct";
|
||||
ret_tensorrt |= checkResult(odim, rt_out[i], out) == 0 ? 0 : ERROR_TENSORRT;
|
||||
std::cout<<"CUDNN vs TRT ";
|
||||
ret_cudnn_tensorrt |= checkResult(odim, cudnn_out[i], rt_out[i]) == 0 ? 0 : ERROR_CUDNNvsTENSORRT;
|
||||
}
|
||||
|
||||
delete [] out_h;
|
||||
checkCuda( cudaFree(out) );
|
||||
}
|
||||
delete [] input_h;
|
||||
checkCuda( cudaFree(data) );
|
||||
return ret_cudnn | ret_tensorrt | ret_cudnn_tensorrt;
|
||||
}
|
||||
@@ -5,4 +5,4 @@
|
||||
#include "Layer.h"
|
||||
#include "NetworkRT.h"
|
||||
|
||||
#define TKDNN_VERSION 400
|
||||
#define TKDNN_VERSION 700
|
||||
|
||||
+86
-6
@@ -6,14 +6,52 @@
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
#include <stdlib.h>
|
||||
#include <yaml-cpp/yaml.h>
|
||||
|
||||
|
||||
#include "cuda.h"
|
||||
#include "cuda_runtime_api.h"
|
||||
#include <cublas_v2.h>
|
||||
#include <cudnn.h>
|
||||
#include <NvInferVersion.h>
|
||||
|
||||
|
||||
#ifdef __linux__
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <ios>
|
||||
#include <chrono>
|
||||
|
||||
#include <yaml-cpp/yaml.h>
|
||||
|
||||
|
||||
|
||||
#ifndef NOEXCEPT
|
||||
#if NV_TENSORRT_MAJOR > 7
|
||||
#define NOEXCEPT noexcept
|
||||
#else
|
||||
#define NOEXCEPT
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
#define dnnType float
|
||||
|
||||
template<typename T> void writeBUF(char*& buffer, const T& val)
|
||||
{
|
||||
*reinterpret_cast<T*>(buffer) = val;
|
||||
buffer += sizeof(T);
|
||||
}
|
||||
|
||||
template<typename T> T readBUF(const char*& buffer)
|
||||
{
|
||||
T val = *reinterpret_cast<const T*>(buffer);
|
||||
buffer += sizeof(T);
|
||||
return val;
|
||||
}
|
||||
|
||||
|
||||
// Colored output
|
||||
#define COL_END "\033[0m"
|
||||
|
||||
@@ -31,16 +69,27 @@
|
||||
#define COL_PURPLEB "\033[1;35m"
|
||||
#define COL_CYANB "\033[1;36m"
|
||||
|
||||
#define TKDNN_VERBOSE 0
|
||||
|
||||
// Simple Timer
|
||||
#define TIMER_START timespec start, end; \
|
||||
#ifdef __linux__
|
||||
#define TKDNN_TSTART timespec start, end; \
|
||||
clock_gettime(CLOCK_MONOTONIC, &start);
|
||||
|
||||
#define TIMER_STOP_C(col) clock_gettime(CLOCK_MONOTONIC, &end); \
|
||||
#define TKDNN_TSTOP_C(col, show) clock_gettime(CLOCK_MONOTONIC, &end); \
|
||||
double t_ns = ((double)(end.tv_sec - start.tv_sec) * 1.0e9 + \
|
||||
(double)(end.tv_nsec - start.tv_nsec))/1.0e6; \
|
||||
std::cout<<col<<"Time:"<<std::setw(16)<<t_ns<<" ms\n"<<COL_END;
|
||||
if(show) std::cout<<col<<"Time:"<<std::setw(16)<<t_ns<<" ms\n"<<COL_END;
|
||||
|
||||
#define TKDNN_TSTOP TKDNN_TSTOP_C(COL_CYANB, TKDNN_VERBOSE)
|
||||
#elif _WIN32
|
||||
#define TKDNN_TSTART auto start = std::chrono::high_resolution_clock::now();
|
||||
#define TKDNN_TSTOP auto stop = std::chrono::high_resolution_clock::now(); \
|
||||
std::chrono::duration<double> duration = stop -start; \
|
||||
auto time_ms = std::chrono::duration_cast<std::chrono::milliseconds>(duration);\
|
||||
double t_ns = time_ms.count();
|
||||
#endif
|
||||
|
||||
#define TIMER_STOP TIMER_STOP_C(COL_CYANB)
|
||||
|
||||
/********************************************************
|
||||
* Prints the error message, and exits
|
||||
@@ -88,15 +137,46 @@
|
||||
} \
|
||||
}
|
||||
|
||||
void printCenteredTitle(const char *title, char fill, int dim);
|
||||
typedef enum {
|
||||
ERROR_CUDNN = 2,
|
||||
ERROR_TENSORRT = 4,
|
||||
ERROR_CUDNNvsTENSORRT = 8
|
||||
} resultError_t;
|
||||
|
||||
void printCenteredTitle(const char *title, char fill, int dim = 30);
|
||||
bool fileExist(const char *fname);
|
||||
void downloadWeightsifDoNotExist(const std::string& input_bin, const std::string& test_folder, const std::string& weights_url);
|
||||
void readBinaryFile(std::string fname, int size, dnnType** data_h, dnnType** data_d, int seek = 0);
|
||||
int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device = true);
|
||||
int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device = true, int limit = 10, bool verbose=true);
|
||||
void printDeviceVector(int size, dnnType* vec_d, bool device = true);
|
||||
float getColor(const int c, const int x, const int max);
|
||||
void resize(int size, dnnType **data);
|
||||
|
||||
void matrixTranspose(cublasHandle_t handle, dnnType* srcData, dnnType* dstData, int rows, int cols);
|
||||
|
||||
void matrixMulAdd( cublasHandle_t handle, dnnType* srcData, dnnType* dstData,
|
||||
dnnType* add_vector, int dim, dnnType mul);
|
||||
|
||||
void getMemUsage(double& vm_usage_kb, double& resident_set_kb);
|
||||
void printCudaMemUsage();
|
||||
void removePathAndExtension(const std::string &full_string, std::string &name);
|
||||
static inline bool isCudaPointer(void *data) {
|
||||
cudaPointerAttributes attr;
|
||||
return cudaPointerGetAttributes(&attr, data) == 0;
|
||||
}
|
||||
|
||||
inline YAML::Node YAMLloadConf(const std::string& conf_file) {
|
||||
std::cerr<<"Loading YAML: "<<conf_file<<"\n";
|
||||
return YAML::LoadFile(conf_file);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
inline T YAMLgetConf(YAML::Node conf, std::string key, T defaultVal) {
|
||||
T val = defaultVal;
|
||||
if(conf && conf[key]) {
|
||||
val = conf[key].as<T>();
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
#endif //UTILS_H
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
import sys
|
||||
import pandas as pd
|
||||
|
||||
if len(sys.argv) < 3:
|
||||
print("Error: two csv files are needed, old first new second")
|
||||
exit(1)
|
||||
|
||||
old_perf_file = str(sys.argv[1])
|
||||
new_perf_file = str(sys.argv[2])
|
||||
|
||||
verbose = False
|
||||
if len(sys.argv) == 4:
|
||||
verbose = bool(sys.argv[3])
|
||||
|
||||
print("Comparing {} vs {}".format(old_perf_file, new_perf_file))
|
||||
|
||||
df_old = pd.read_csv (old_perf_file, sep=';', header=None, index_col=0)
|
||||
df_new = pd.read_csv (new_perf_file, sep=';', header=None, index_col=0)
|
||||
|
||||
for index, row in df_new.iterrows():
|
||||
if index in df_old.index:
|
||||
if verbose:
|
||||
print("New: ",row[1], row[2], row[3])
|
||||
print("Old: ",df_old.loc[index][1], df_old.loc[index][2], df_old.loc[index][3])
|
||||
|
||||
print(index, end=': ')
|
||||
if abs(row[1] - df_old.loc[index][1]) < df_old.loc[index][1]*0.1:
|
||||
print("similar performance")
|
||||
elif (row[1] < df_old.loc[index][1]):
|
||||
print('\x1b[3;30;42m' + 'faster' + '\x1b[0m')
|
||||
elif (row[1] > df_old.loc[index][1]):
|
||||
if row[1] > df_old.loc[index][1] + df_old.loc[index][1] * 0.5 :
|
||||
print('\x1b[3;30;41m' + 'WAY SLOWER' + '\x1b[0m')
|
||||
else:
|
||||
print('\x1b[3;30;41m' + 'slower' + '\x1b[0m')
|
||||
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
import os
|
||||
import urllib.request as dowReq
|
||||
import zipfile
|
||||
|
||||
val = input("Enter BDD or COCO :")
|
||||
if(val == "COCO"):
|
||||
url = "https://cloud.hipert.unimore.it/s/LNxBDk4wzqXPL8c/download"
|
||||
lib = "..\demo\COCO_val2017"
|
||||
lib_zip = "COCO_val2017.zip"
|
||||
elif(val == "BDD"):
|
||||
url = "https://cloud.hipert.unimore.it/s/bikqk3FzCq2tg4D/download"
|
||||
lib = "..\demo\BDD100k_val"
|
||||
lib_zip = "BDD100k_val.zip"
|
||||
|
||||
dowReq.urlretrieve(url,lib_zip)
|
||||
|
||||
with zipfile.ZipFile(lib_zip,'r') as zip_ref:
|
||||
zip_ref.extractall(lib)
|
||||
|
||||
labelFolder = lib + "\labels"
|
||||
imageFolder = lib + "\images"
|
||||
|
||||
file1 = open(".\\..\\demo\\all_labels.txt","a")
|
||||
path1 = os.path.realpath(labelFolder)
|
||||
for file in os.listdir(labelFolder):
|
||||
valTemp = path1 + "\\" + file
|
||||
valTemp = valTemp + '\n'
|
||||
file1.write(valTemp)
|
||||
file1.close()
|
||||
|
||||
file2 = open(".\\..\\demo\\all_images.txt","a")
|
||||
path2 = os.path.realpath(imageFolder)
|
||||
for file in os.listdir(imageFolder):
|
||||
pathtemp = path2 + "\\" + file
|
||||
pathtemp = pathtemp + '\n'
|
||||
file2.write(pathtemp)
|
||||
file2.close()
|
||||
|
||||
print("Completed")
|
||||
@@ -0,0 +1,26 @@
|
||||
#!/bin/bash
|
||||
|
||||
|
||||
function elaborate_testset {
|
||||
wget $1 -O $2.zip
|
||||
unzip -d $2 $2.zip
|
||||
rm $2.zip
|
||||
cd $2/
|
||||
realpath labels/* > all_labels.txt
|
||||
realpath images/* > all_images.txt
|
||||
cd ..
|
||||
}
|
||||
|
||||
cd demo
|
||||
for valset in $@
|
||||
do
|
||||
|
||||
if [ $valset = "COCO" ]; then
|
||||
echo "Downloading $valset validation set in demo"
|
||||
elaborate_testset "https://cloud.hipert.unimore.it/s/LNxBDk4wzqXPL8c/download" "COCO_val2017"
|
||||
elif [ $valset = "BDD" ]; then
|
||||
echo "Downloading $valset validation set in demo"
|
||||
elaborate_testset "https://cloud.hipert.unimore.it/s/bikqk3FzCq2tg4D/download" "BDD100K_val"
|
||||
fi
|
||||
|
||||
done
|
||||
@@ -0,0 +1,72 @@
|
||||
#!/bin/bash
|
||||
|
||||
#based on https://devtalk.nvidia.com/default/topic/1042035/installing-opencv4-on-xavier/ & https://github.com/markste-in/OpenCV4XAVIER/blob/master/buildOpenCV4.sh
|
||||
|
||||
# Compute Capabilities can be found here https://developer.nvidia.com/cuda-gpus#compute
|
||||
ARCH_BIN=7.2 # AGX Xavier
|
||||
#ARCH_BIN=6.2 # Tx2
|
||||
|
||||
cd ~/Downloads
|
||||
sudo apt-get install -y build-essential \
|
||||
unzip \
|
||||
pkg-config \
|
||||
libjpeg-dev \
|
||||
libpng-dev \
|
||||
libtiff-dev \
|
||||
libavcodec-dev \
|
||||
libavformat-dev \
|
||||
libswscale-dev \
|
||||
libv4l-dev \
|
||||
libxvidcore-dev \
|
||||
libx264-dev \
|
||||
libgtk-3-dev \
|
||||
libatlas-base-dev \
|
||||
gfortran \
|
||||
python3-dev \
|
||||
python3-venv \
|
||||
libgstreamer1.0-dev \
|
||||
libgstreamer-plugins-base1.0-dev \
|
||||
libdc1394-22-dev \
|
||||
libavresample-dev \
|
||||
libtbb-dev \
|
||||
|
||||
git clone https://github.com/opencv/opencv.git
|
||||
cd opencv && git checkout 4.5.4 && cd ..
|
||||
git clone https://github.com/opencv/opencv_contrib.git
|
||||
cd opencv_contrib && git checkout 4.5.4 && cd ..
|
||||
|
||||
|
||||
python3 -m venv opencv4
|
||||
source opencv4/bin/activate
|
||||
pip install wheel
|
||||
pip install numpy
|
||||
|
||||
cd opencv && mkdir build && cd build
|
||||
|
||||
cmake -D CMAKE_BUILD_TYPE=RELEASE \
|
||||
-D CMAKE_INSTALL_PREFIX=/usr/local \
|
||||
-D INSTALL_PYTHON_EXAMPLES=ON \
|
||||
-D INSTALL_C_EXAMPLES=OFF \
|
||||
-D OPENCV_EXTRA_MODULES_PATH='~/Downloads/opencv_contrib/modules' \
|
||||
-D PYTHON_EXECUTABLE='~/Downloads/opencv4/bin/python' \
|
||||
-D BUILD_EXAMPLES=ON \
|
||||
-D WITH_CUDA=ON \
|
||||
-D CUDA_ARCH_BIN=${ARCH_BIN} \
|
||||
-D CUDA_ARCH_PTX="" \
|
||||
-D ENABLE_FAST_MATH=ON \
|
||||
-D CUDA_FAST_MATH=ON \
|
||||
-D WITH_CUBLAS=ON \
|
||||
-D WITH_LIBV4L=ON \
|
||||
-D WITH_GSTREAMER=ON \
|
||||
-D WITH_GSTREAMER_0_10=OFF \
|
||||
-D WITH_TBB=ON \
|
||||
-D WITH_OPENGL=ON \
|
||||
-D WITH_VULKAN=ON \
|
||||
../
|
||||
|
||||
make -j4
|
||||
sudo make install
|
||||
sudo ldconfig
|
||||
|
||||
cd ~/Downloads/opencv4/lib/python3.6/site-packages
|
||||
ln -s /usr/local/lib/python3.6/site-packages/cv2.cpython-36m-aarch64-linux-gnu.so cv2.so
|
||||
@@ -0,0 +1,113 @@
|
||||
#!/bin/bash
|
||||
|
||||
#cd build
|
||||
|
||||
RED='\033[1;31m'
|
||||
GREEN='\033[1;32m'
|
||||
ORANGE='\033[1;33m'
|
||||
PINK='\033[1;95m'
|
||||
NC='\033[0m' # No Color
|
||||
|
||||
function print_output {
|
||||
if [ $1 -eq 0 ]; then
|
||||
echo -e "$2 ${GREEN}OK${NC}"
|
||||
elif [ $1 -eq 1 ]; then
|
||||
echo -e "$2 ${RED}FATAL ERROR${NC}"
|
||||
elif [ $1 -eq 2 ] || [ $1 -eq 10 ]; then
|
||||
echo -e "$2 ${PINK}CUDNN ERROR${NC}"
|
||||
elif [ $1 -eq 4 ] || [ $1 -eq 12 ]; then
|
||||
echo -e "$2 ${PINK}TENSORRT ERROR${NC}"
|
||||
elif [ $1 -eq 8 ]; then
|
||||
echo -e "$2 ${PINK}CUDNN vs TENSORRT ERROR${NC}"
|
||||
elif [ $1 -eq 6 ]; then
|
||||
echo -e "$2 ${PINK}CUDNN & TENSORTRT ERROR${NC}"
|
||||
elif [ $1 -eq 14 ]; then
|
||||
echo -e "$2 ${PINK}ERROR FOR EVERY CHECK${NC}"
|
||||
else
|
||||
echo -e "$2 ${RED}NOT OKAY (OPENCV maybe)${NC}"
|
||||
fi
|
||||
|
||||
}
|
||||
|
||||
out_dir=results
|
||||
out_file=results.log
|
||||
rm -rf $out_dir/
|
||||
mkdir -p $out_dir
|
||||
|
||||
function test_net {
|
||||
./test_$1 &> $out_dir/$1_${TKDNN_MODE}_build_$out_file
|
||||
print_output $? $1
|
||||
./test_rtinference $1*.rt 1 &> $out_dir/$1_${TKDNN_MODE}_inference_batch1_$out_file
|
||||
print_output $? "infer $1"
|
||||
./test_rtinference $1*.rt $TKDNN_BATCHSIZE &> $out_dir/$1_${TKDNN_MODE}_inference_batch${TKDNN_BATCHSIZE}_$out_file
|
||||
print_output $? "batched $1"
|
||||
}
|
||||
|
||||
|
||||
# modes=( 1 ) # only FP32
|
||||
modes=( 1 2 ) # FP32 and FP16
|
||||
# modes=( 1 2 3 ) # FP32, FP16 and INT8
|
||||
|
||||
for i in "${modes[@]}"
|
||||
do
|
||||
rm -f *rt
|
||||
if [ $i -eq 1 ]
|
||||
then
|
||||
export TKDNN_MODE=FP32
|
||||
echo -e "${ORANGE}Test FP32${NC}"
|
||||
fi
|
||||
if [ $i -eq 2 ]
|
||||
then
|
||||
export TKDNN_MODE=FP16
|
||||
echo -e "${ORANGE}Test FP16${NC}"
|
||||
fi
|
||||
if [ $i -eq 3 ]
|
||||
then
|
||||
export TKDNN_MODE=INT8
|
||||
export TKDNN_CALIB_LABEL_PATH=../demo/COCO_val2017/all_labels.txt
|
||||
export TKDNN_CALIB_IMG_PATH=../demo/COCO_val2017/all_images.txt
|
||||
echo -e "${ORANGE}Test INT8${NC}"
|
||||
fi
|
||||
|
||||
export TKDNN_BATCHSIZE=2
|
||||
echo -e "${ORANGE}Batch $TKDNN_BATCHSIZE ${NC}"
|
||||
|
||||
test_net mnist
|
||||
# ./test_imuodom &>> $out_file
|
||||
# print_output $? imuodom
|
||||
|
||||
test_net yolo4
|
||||
# test_net yolo4_320
|
||||
# test_net yolo4_320_coco2
|
||||
# test_net yolo4_512
|
||||
# test_net yolo4_608
|
||||
# test_net yolo4-csp
|
||||
# test_net yolo4x
|
||||
# test_net yolo4_berkeley
|
||||
# test_net yolo4_berkeley_f1
|
||||
# test_net yolo4tiny
|
||||
# test_net yolo4tiny_512
|
||||
# test_net yolo3
|
||||
# test_net yolo3_berkeley
|
||||
# test_net yolo3_coco4
|
||||
# test_net yolo3_flir
|
||||
# test_net yolo3_512
|
||||
# test_net yolo3tiny
|
||||
# test_net yolo3tiny_512
|
||||
# test_net yolo2
|
||||
# test_net yolo2_voc
|
||||
# test_net yolo2tiny
|
||||
# test_net csresnext50-panet-spp
|
||||
# test_net csresnext50-panet-spp_berkeley
|
||||
# test_net resnet101_cnet
|
||||
# test_net dla34_cnet
|
||||
# test_net dla34_cnet3d
|
||||
# test_net mobilenetv2ssd
|
||||
# test_net mobilenetv2ssd512
|
||||
# test_net bdd-mobilenetv2ssd
|
||||
# test_net dla34_ctrack
|
||||
# test_net shelfnet
|
||||
# test_net shelfnet_berkeley
|
||||
done
|
||||
|
||||
echo "If errors occured, check logfiles in directory: $out_dir"
|
||||
@@ -0,0 +1,52 @@
|
||||
#!/bin/bash
|
||||
|
||||
function test_inference {
|
||||
./test_$1
|
||||
./test_rtinference $1_$2.rt 1
|
||||
./test_rtinference $1_$2.rt 4
|
||||
}
|
||||
|
||||
sudo jeston_clock
|
||||
|
||||
# modes=( 1 ) # only FP32
|
||||
# modes=( 1 2 ) # FP32 and FP16
|
||||
modes=( 1 2 3 ) # FP32, FP16 and INT8
|
||||
|
||||
rm times_rtinference.csv
|
||||
for i in "${modes[@]}"
|
||||
do
|
||||
rm *rt
|
||||
if [ $i -eq 1 ]
|
||||
then
|
||||
export TKDNN_MODE=FP32
|
||||
mode=fp32
|
||||
echo -e "${ORANGE}Test FP32${NC}"
|
||||
fi
|
||||
if [ $i -eq 2 ]
|
||||
then
|
||||
export TKDNN_MODE=FP16
|
||||
mode=fp16
|
||||
echo -e "${ORANGE}Test FP16${NC}"
|
||||
fi
|
||||
if [ $i -eq 3 ]
|
||||
then
|
||||
export TKDNN_MODE=INT8
|
||||
export TKDNN_CALIB_LABEL_PATH=../demo/COCO_val2017/all_labels.txt
|
||||
export TKDNN_CALIB_IMG_PATH=../demo/COCO_val2017/all_images.txt
|
||||
mode=int8
|
||||
echo -e "${ORANGE}Test INT8${NC}"
|
||||
|
||||
fi
|
||||
|
||||
export TKDNN_BATCHSIZE=4
|
||||
echo -e "${ORANGE}Batch $TKDNN_BATCHSIZE ${NC}"
|
||||
|
||||
test_inference yolo4_320 $mode
|
||||
test_inference yolo4 $mode
|
||||
test_inference yolo4_512 $mode
|
||||
test_inference yolo4_608 $mode
|
||||
test_inference yolo4tiny $mode
|
||||
done
|
||||
|
||||
|
||||
|
||||
+16
-5
@@ -5,10 +5,12 @@
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
Activation::Activation(Network *net, int act_mode) :
|
||||
Activation::Activation(Network *net, int act_mode, const float ceiling, const float slope) :
|
||||
Layer(net) {
|
||||
|
||||
this->act_mode = act_mode;
|
||||
this->act_mode = act_mode;
|
||||
this->ceiling = ceiling;
|
||||
this->slope = slope;
|
||||
checkCuda( cudaMalloc(&dstData, input_dim.tot()*sizeof(dnnType)) );
|
||||
|
||||
if(int(act_mode) < 100) {
|
||||
@@ -31,7 +33,7 @@ Activation::Activation(Network *net, int act_mode) :
|
||||
checkCUDNN( cudnnSetActivationDescriptor(activDesc,
|
||||
(cudnnActivationMode_t) act_mode,
|
||||
CUDNN_PROPAGATE_NAN,
|
||||
0.0) );
|
||||
ceiling) );
|
||||
}
|
||||
}
|
||||
|
||||
@@ -44,9 +46,18 @@ Activation::~Activation() {
|
||||
}
|
||||
|
||||
dnnType* Activation::infer(dataDim_t &dim, dnnType* srcData) {
|
||||
|
||||
if(act_mode == ACTIVATION_LEAKY) {
|
||||
activationLEAKYForward(srcData, dstData, dim.tot());
|
||||
activationLEAKYForward(srcData, dstData, dim.tot(), this->slope);
|
||||
}
|
||||
else if(act_mode == ACTIVATION_MISH) {
|
||||
activationMishForward(srcData, dstData, dim.tot());
|
||||
|
||||
}
|
||||
else if(act_mode == ACTIVATION_LOGISTIC) {
|
||||
activationLOGISTICForward(srcData, dstData, dim.tot());
|
||||
|
||||
} else if(act_mode == ACTIVATION_ELU) {
|
||||
activationELUForward(srcData, dstData, dim.tot());
|
||||
|
||||
} else {
|
||||
dnnType alpha = dnnType(1);
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
#include "BoundingBox.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
float BoundingBox::overlap(const float p1, const float d1, const float p2, const float d2){
|
||||
float l1 = p1 - d1/2;
|
||||
float l2 = p2 - d2/2;
|
||||
float left = l1 > l2 ? l1 : l2;
|
||||
float r1 = p1 + d1/2;
|
||||
float r2 = p2 + d2/2;
|
||||
float right = r1 < r2 ? r1 : r2;
|
||||
return right - left;
|
||||
}
|
||||
|
||||
float BoundingBox::boxesIntersection(const BoundingBox &b){
|
||||
float width = this->overlap(x, w, b.x, b.w);
|
||||
float height = this->overlap(y, h, b.y, b.h);
|
||||
if(width < 0 || height < 0)
|
||||
return 0;
|
||||
float area = width*height;
|
||||
return area;
|
||||
}
|
||||
|
||||
float BoundingBox::boxesUnion(const BoundingBox &b){
|
||||
float i = this->boxesIntersection(b);
|
||||
float u = w*h + b.w*b.h - i;
|
||||
return u;
|
||||
}
|
||||
|
||||
float BoundingBox::IoU(const BoundingBox &b){
|
||||
float I = this->boxesIntersection(b);
|
||||
float U = this->boxesUnion(b);
|
||||
if (I == 0 || U == 0)
|
||||
return 0;
|
||||
return I / U;
|
||||
}
|
||||
|
||||
void BoundingBox::clear(){
|
||||
uniqueTruthIndex = -1;
|
||||
truthFlag = 0;
|
||||
maxIoU = 0;
|
||||
}
|
||||
|
||||
std::ostream& operator<<(std::ostream& os, const BoundingBox& bb){
|
||||
os <<"w: "<< bb.w << ", h: "<< bb.h << ", x: "<< bb.x << ", y: "<< bb.y <<
|
||||
", cat: "<< bb.cl << ", conf: "<< bb.prob<< ", truth: "<<
|
||||
bb.truthFlag<< ", assignedGT: "<< bb.uniqueTruthIndex<<
|
||||
", maxIoU: "<< bb.maxIoU<<"\n";
|
||||
return os;
|
||||
}
|
||||
|
||||
bool boxComparison (const BoundingBox& a,const BoundingBox& b) {
|
||||
return (a.prob>b.prob);
|
||||
}
|
||||
|
||||
}}
|
||||
@@ -0,0 +1,897 @@
|
||||
#include "CenterTrack.h"
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
|
||||
bool CenterTrack::init(const std::string& tensor_path, const int n_classes, const int n_batches,
|
||||
const float conf_thresh, const bool mode_3d, const std::vector<cv::Mat>& k_calibs) {
|
||||
netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() );
|
||||
dim = netRT->input_dim;
|
||||
dim.c = 3;
|
||||
nBatches = n_batches;
|
||||
confThreshold = conf_thresh;
|
||||
mode3D = mode_3d;
|
||||
inputCalibs = k_calibs;
|
||||
init_preprocessing();
|
||||
init_pre_inf();
|
||||
init_postprocessing();
|
||||
init_visualization(n_classes);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CenterTrack::init_preprocessing(){
|
||||
//image transformation
|
||||
src = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
dst = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
dst2 = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
trans = cv::Mat(cv::Size(3,2), CV_32F);
|
||||
trans2 = cv::Mat(cv::Size(3,2), CV_32F);
|
||||
transOut = cv::Mat(cv::Size(3,2), CV_32F);
|
||||
|
||||
dst2.at<float>(0,0) = width * 0.5;
|
||||
dst2.at<float>(0,1) = width * 0.5;
|
||||
dst2.at<float>(1,0) = width * 0.5;
|
||||
dst2.at<float>(1,1) = width * 0.5 + width * -0.5;
|
||||
dst2.at<float>(2,0) = dst2.at<float>(1,0) + (-dst2.at<float>(0,1)+dst2.at<float>(1,1) );
|
||||
dst2.at<float>(2,1) = dst2.at<float>(1,1) + (dst2.at<float>(0,0)-dst2.at<float>(1,0) );
|
||||
|
||||
for(int bi=0; bi<nBatches; bi++) {
|
||||
szOld.push_back(cv::Size(0,0));
|
||||
}
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
std::cout<<"OPENCV CPMTROB\n";
|
||||
checkCuda( cudaMalloc(&mean_d, 3 * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&stddev_d, 3 * sizeof(float)) );
|
||||
float mean[3] = {0.40789655, 0.44719303, 0.47026116};
|
||||
float stddev[3] = {0.2886383, 0.27408165, 0.27809834};
|
||||
|
||||
checkCuda( cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice));
|
||||
checkCuda( cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice));
|
||||
#else
|
||||
std::cout<<"NO OPENCV CPMTROB\n";
|
||||
checkCuda( cudaMallocHost(&input, sizeof(dnnType)*dim.tot() * nBatches));
|
||||
mean << 0.40789655, 0.44719303, 0.47026116;
|
||||
stddev << 0.2886383, 0.27408165, 0.27809834;
|
||||
|
||||
#endif
|
||||
|
||||
checkCuda( cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches));
|
||||
checkCuda( cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot()));
|
||||
checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) );
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CenterTrack::init_pre_inf(){
|
||||
// initial steps: the first part of the network
|
||||
const char *pre_img_conv1_bin = "dla34_ctrack/layers/base-pre_img_layer-0.bin";
|
||||
const char *pre_hm_conv1_bin = "dla34_ctrack/layers/base-pre_hm_layer-0.bin";
|
||||
const char *conv1_bin = "dla34_ctrack/layers/base-base_layer-0.bin";
|
||||
const char *conv2_bin = "dla34_ctrack/layers/base-level0-0.bin";
|
||||
dim_in0 = tk::dnn::dataDim_t(1, 3, 512, 512, 1);
|
||||
dim_in1 = tk::dnn::dataDim_t(1, 1, 512, 512, 1);
|
||||
|
||||
checkCuda( cudaMalloc(&out_d, netRT->input_dim.tot()*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&img_d, dim_in0.tot()*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&hm_d, dim_in1.tot()*sizeof(dnnType)) );
|
||||
// init to zeros hm
|
||||
dnnType *hm_h;
|
||||
checkCuda( cudaMallocHost(&hm_h, 1 * dim.h * dim.w*sizeof(dnnType)) );
|
||||
for(int i=0; i<1 * dim.h * dim.w; i++)
|
||||
hm_h[i] = 0.0f;
|
||||
checkCuda( cudaMemcpy(hm_d, hm_h, 1 * dim.h * dim.w * sizeof(dnnType), cudaMemcpyHostToDevice) );
|
||||
checkCuda( cudaFreeHost(hm_h) );
|
||||
dnnType *i0_h, *i1_h, *i2_h;
|
||||
// dnnType *i0_d, *i1_d, *i2_d;
|
||||
|
||||
// const char *input_bin = "dla34_ctrack/debug/input.bin";
|
||||
// const char *pre_img_bin = "dla34_ctrack/debug/pre_imgages.bin";
|
||||
// const char *pre_hm_bin = "dla34_ctrack/debug/pre_hms.bin";
|
||||
// readBinaryFile(pre_img_bin, dim_in0.tot(), &i0_h, &img_d);
|
||||
// readBinaryFile(pre_hm_bin, dim_in1.tot(), &i1_h, &hm_d);
|
||||
// readBinaryFile(input_bin, dim_in0.tot(), &i2_h, &input_pre_inf_d);
|
||||
|
||||
pre_phase_net = new tk::dnn::Network(dim_in0);
|
||||
//pre-img
|
||||
tk::dnn::Input *in_pre_img = new tk::dnn::Input(pre_phase_net, dim_in0, img_d);
|
||||
tk::dnn::Conv2d *pre_img_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_img_conv1_bin, true);
|
||||
tk::dnn::Activation *pre_img_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU);
|
||||
//pre-hm
|
||||
tk::dnn::Input *in_pre_hm = new tk::dnn::Input(pre_phase_net, dim_in1, hm_d);
|
||||
tk::dnn::Conv2d *pre_hm_conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, pre_hm_conv1_bin, true);
|
||||
tk::dnn::Activation *pre_hm_relu = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU);
|
||||
// image input
|
||||
tk::dnn::Input *input_image = new tk::dnn::Input(pre_phase_net, dim_in0, input_pre_inf_d);
|
||||
tk::dnn::Conv2d *conv1 = new tk::dnn::Conv2d(pre_phase_net, 16, 7, 7, 1, 1, 3, 3, conv1_bin, true);
|
||||
tk::dnn::Activation *relu1 = new tk::dnn::Activation(pre_phase_net, CUDNN_ACTIVATION_RELU);
|
||||
|
||||
tk::dnn::Shortcut *s0_input = new tk::dnn::Shortcut(pre_phase_net, pre_img_relu);
|
||||
tk::dnn::Shortcut *s1_input = new tk::dnn::Shortcut(pre_phase_net, pre_hm_relu);
|
||||
// output data
|
||||
out_d = s1_input->dstData;
|
||||
//print network model
|
||||
pre_phase_net->print();
|
||||
|
||||
iter0=true; // in the first iteration the last input is equal to the current input.
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CenterTrack::init_postprocessing(){
|
||||
srand(0); //seed = 0 for random colors
|
||||
|
||||
dim_hm = tk::dnn::dataDim_t(1, 10, 128, 128, 1);
|
||||
dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1);
|
||||
dim_reg = tk::dnn::dataDim_t(1, 2, 128, 128, 1);
|
||||
dim_track = tk::dnn::dataDim_t(1, 2, 128, 128, 1);
|
||||
dim_dep = tk::dnn::dataDim_t(1, 1, 128, 128, 1);
|
||||
dim_rot = tk::dnn::dataDim_t(1, 8, 128, 128, 1);
|
||||
dim_dim = tk::dnn::dataDim_t(1, 3, 128, 128, 1);
|
||||
dim_amodel_offset = tk::dnn::dataDim_t(1, 2, 128, 128, 1);
|
||||
|
||||
checkCuda( cudaMalloc(&topk_scores, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_inds_, dim_hm.c * K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&topk_ys_, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_xs_, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&ids_d, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) );
|
||||
checkCuda( cudaMallocHost(&ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) );
|
||||
for(int i=0; i<dim_hm.c * dim_hm.h * dim_hm.w; i++){
|
||||
ids_[i] = i;
|
||||
}
|
||||
|
||||
checkCuda( cudaMalloc(&ones, dim_dep.c * dim_dep.h * dim_dep.w * sizeof(float)) );
|
||||
float *ones_h;
|
||||
checkCuda( cudaMallocHost(&ones_h, dim_dep.c * dim_dep.h * dim_dep.w * sizeof(float)) );
|
||||
for(int i=0; i<dim_dep.c * dim_dep.h * dim_dep.w; i++)
|
||||
ones_h[i] = 1.0f;
|
||||
checkCuda( cudaMemcpy(ones, ones_h, dim_dep.c * dim_dep.h * dim_dep.w * sizeof(float), cudaMemcpyHostToDevice) );
|
||||
checkCuda( cudaFreeHost(ones_h) );
|
||||
|
||||
checkCuda( cudaMallocHost(&scores, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&scores_d, K *sizeof(float)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&clses, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&clses_d, K *sizeof(int)) );
|
||||
|
||||
checkCuda( cudaMalloc(&topk_inds_d, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&topk_ys_d, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_xs_d, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&inttopk_ys_d, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&inttopk_xs_d, K *sizeof(int)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&bbx0, K * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&bby0, K * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&bbx1, K * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&bby1, K * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&bbx0_d, K * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&bby0_d, K * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&bbx1_d, K * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&bby1_d, K * sizeof(float)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&intxs, K * sizeof(int)) );
|
||||
checkCuda( cudaMallocHost(&intys, K * sizeof(int)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&track, K * dim_track.c * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&dep, K * dim_dep.c * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&rot, K * dim_rot.c * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&dim_, K * dim_dim.c * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&wh, K * dim_wh.c * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&amodel_offset, K * dim_amodel_offset.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&track_d, K * dim_track.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&dep_d, K * dim_dep.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&rot_d, K * dim_rot.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&dim_d, K * dim_dim.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&wh_d, K * dim_wh.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&amodel_offset_d, K * dim_amodel_offset.c * sizeof(float)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&target_coords, 4 * K *sizeof(float)) );
|
||||
|
||||
for(int bi=0; bi<nBatches; bi++) {
|
||||
cv::Mat calibs_ = cv::Mat::zeros(cv::Size(4,3), CV_32F);
|
||||
if(inputCalibs.size() == 0 || inputCalibs[bi].empty()) {
|
||||
calibs_.at<float>(0,0) = 633.0;
|
||||
calibs_.at<float>(1,1) = 633.0;
|
||||
calibs_.at<float>(2,2) = 1.0;
|
||||
}
|
||||
calibs_.at<float>(2,2) = 1.0;
|
||||
calibs.push_back(calibs_);
|
||||
}
|
||||
|
||||
// Alloc array used in the kernel
|
||||
checkCuda( cudaMalloc(&src_out, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&ids_out, K *sizeof(int)) );
|
||||
|
||||
trRes.resize(nBatches);
|
||||
countTr.resize(nBatches, 0);
|
||||
trackId.resize(nBatches, 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool CenterTrack::init_visualization(const int n_classes){
|
||||
classes = n_classes;
|
||||
// const char *kitti_class_name[] = {
|
||||
// "person", "car", "bicycle"};
|
||||
// classesNames = std::vector<std::string>(kitti_class_name, std::end( kitti_class_name));
|
||||
|
||||
const char *class_name[] = {"car", "truck", "bus", "trailer", "construction_vehicle", "pedestrian",
|
||||
"motorcycle", "bicycle", "traffic_cone", "barrier"};
|
||||
classesNames = std::vector<std::string>(class_name, std::end( class_name));
|
||||
|
||||
// const char *coco_class_name[] = {
|
||||
// "person", "bicycle", "car", "motorcycle", "airplane",
|
||||
// "bus", "train", "truck", "boat", "traffic light", "fire hydrant",
|
||||
// "stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse",
|
||||
// "sheep", "cow", "elephant", "bear", "zebra", "giraffe", "backpack",
|
||||
// "umbrella", "handbag", "tie", "suitcase", "frisbee", "skis",
|
||||
// "snowboard", "sports ball", "kite", "baseball bat", "baseball glove",
|
||||
// "skateboard", "surfboard", "tennis racket", "bottle", "wine glass",
|
||||
// "cup", "fork", "knife", "spoon", "bowl", "banana", "apple", "sandwich",
|
||||
// "orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake",
|
||||
// "chair", "couch", "potted plant", "bed", "dining table", "toilet", "tv",
|
||||
// "laptop", "mouse", "remote", "keyboard", "cell phone", "microwave",
|
||||
// "oven", "toaster", "sink", "refrigerator", "book", "clock", "vase",
|
||||
// "scissors", "teddy bear", "hair drier", "toothbrush"
|
||||
// };
|
||||
// classesNames = std::vector<std::string>(coco_class_name, std::end( coco_class_name));
|
||||
|
||||
for(int c=0; c<classes; c++) {
|
||||
int offset = c*123457 % classes;
|
||||
float r = getColor(2, offset, classes);
|
||||
float g = getColor(1, offset, classes);
|
||||
float b = getColor(0, offset, classes);
|
||||
colors[c] = cv::Scalar(int(255.0*b), int(255.0*g), int(255.0*r));
|
||||
}
|
||||
for(int c=0; c<256; c++) {
|
||||
int offset = c * 123457 % 256;
|
||||
float r = getColor(2, offset, 256);
|
||||
float g = getColor(1, offset, 256);
|
||||
float b = getColor(0, offset, 256);
|
||||
trColors[c] = cv::Scalar(int(255.0*b), int(255.0*g), int(255.0*r));
|
||||
}
|
||||
|
||||
r = cv::Mat(cv::Size(3,3), CV_32F);
|
||||
r.at<float>(0,1) = 0.0;
|
||||
r.at<float>(1,0) = 0.0;
|
||||
r.at<float>(1,1) = 1.0;
|
||||
r.at<float>(1,2) = 0.0;
|
||||
r.at<float>(2,1) = 0.0;
|
||||
|
||||
corners = cv::Mat(cv::Size(8,3), CV_32F);
|
||||
corners.at<float>(1,0) = 0.0;
|
||||
corners.at<float>(1,1) = 0.0;
|
||||
corners.at<float>(1,2) = 0.0;
|
||||
corners.at<float>(1,3) = 0.0;
|
||||
|
||||
pts3DHomo = cv::Mat(cv::Size(8,4), CV_32F);
|
||||
pts3DHomo.at<float>(3,0) = 1.0;
|
||||
pts3DHomo.at<float>(3,1) = 1.0;
|
||||
pts3DHomo.at<float>(3,2) = 1.0;
|
||||
pts3DHomo.at<float>(3,3) = 1.0;
|
||||
pts3DHomo.at<float>(3,4) = 1.0;
|
||||
pts3DHomo.at<float>(3,5) = 1.0;
|
||||
pts3DHomo.at<float>(3,6) = 1.0;
|
||||
pts3DHomo.at<float>(3,7) = 1.0;
|
||||
|
||||
faceId.push_back({0,1,5,4});
|
||||
faceId.push_back({1,2,6, 5});
|
||||
faceId.push_back({3,0,4,7});
|
||||
faceId.push_back({2,3,7,6});
|
||||
// ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
void CenterTrack::_get_additional_inputs(){
|
||||
//None no additional input
|
||||
}
|
||||
|
||||
void CenterTrack::pre_inf(const int bi){
|
||||
TKDNN_TSTART
|
||||
tk::dnn::dataDim_t dim_aus;
|
||||
pre_phase_net->infer(dim_aus, nullptr);
|
||||
TKDNN_TSTOP
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
checkCuda( cudaMemcpy(input_d+ netRT->input_dim.tot()*bi, pre_phase_net->layers[pre_phase_net->num_layers-1]->dstData, netRT->input_dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice) );
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
}
|
||||
|
||||
void CenterTrack::preprocess(cv::Mat &frame, const int bi){
|
||||
cv::Size sz = originalSize[bi];
|
||||
// float scale = 1.0;
|
||||
float new_height = dim.h;//sz.height * scale;
|
||||
float new_width = dim.w;//sz.width * scale;
|
||||
if(sz.height != szOld[bi].height && sz.width != szOld[bi].width){
|
||||
if(inputCalibs.size() == 0 || inputCalibs[bi].empty()) {
|
||||
calibs[bi].at<float>(0,2) = new_width / 2.0f;
|
||||
calibs[bi].at<float>(1,2) = new_height /2.0f;
|
||||
}
|
||||
else {
|
||||
calibs[bi].at<float>(0,0) = inputCalibs[bi].at<float>(0,0) * dim.w / sz.width;
|
||||
calibs[bi].at<float>(0,2) = inputCalibs[bi].at<float>(0,2) * dim.w / sz.width;
|
||||
calibs[bi].at<float>(1,1) = inputCalibs[bi].at<float>(1,1) * dim.h / sz.height;
|
||||
calibs[bi].at<float>(1,2) = inputCalibs[bi].at<float>(1,2) * dim.h / sz.height;
|
||||
}
|
||||
|
||||
float c[] = {new_width / 2.0f, new_height /2.0f};
|
||||
float s[] = {static_cast<float>(dim.w), static_cast<float>(dim.h)};
|
||||
// float s = new_width >= new_height ? new_width : new_height;
|
||||
// ----------- get_affine_transform
|
||||
// rot_rad = pi * 0 / 100 --> 0
|
||||
//dim.print();
|
||||
src.at<float>(0,0) = c[0];
|
||||
src.at<float>(0,1) = c[1];
|
||||
src.at<float>(1,0) = c[0];
|
||||
src.at<float>(1,1) = c[1] + s[0] * -0.5;
|
||||
dst.at<float>(0,0) = dim.w * 0.5;
|
||||
dst.at<float>(0,1) = dim.h * 0.5;
|
||||
dst.at<float>(1,0) = dim.w * 0.5;
|
||||
dst.at<float>(1,1) = dim.h * 0.5 + dim.w * -0.5;
|
||||
|
||||
src.at<float>(2,0) = src.at<float>(1,0) + (-src.at<float>(0,1)+src.at<float>(1,1) );
|
||||
src.at<float>(2,1) = src.at<float>(1,1) + (src.at<float>(0,0)-src.at<float>(1,0) );
|
||||
dst.at<float>(2,0) = dst.at<float>(1,0) + (-dst.at<float>(0,1)+dst.at<float>(1,1) );
|
||||
dst.at<float>(2,1) = dst.at<float>(1,1) + (dst.at<float>(0,0)-dst.at<float>(1,0) );
|
||||
|
||||
|
||||
trans = cv::getAffineTransform( src, dst );
|
||||
trans2 = cv::getAffineTransform( dst2, src );
|
||||
trans2.convertTo(transOut, CV_32F);
|
||||
}
|
||||
szOld[bi] = sz;
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
cv::cuda::GpuMat im_Orig;
|
||||
cv::cuda::GpuMat imageF1_d, imageF2_d;
|
||||
|
||||
im_Orig = cv::cuda::GpuMat(frame);
|
||||
cv::cuda::resize (im_Orig, imageF1_d, cv::Size(dim.w, dim.h));
|
||||
// imageF1_d = im_Orig;
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
sz = imageF1_d.size();
|
||||
|
||||
cv::cuda::warpAffine(imageF1_d, imageF2_d, trans, cv::Size(dim.w, dim.h), cv::INTER_LINEAR );
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
imageF2_d.convertTo(imageF1_d, CV_32FC3, 1/255.0);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
dim2 = dim;
|
||||
cv::cuda::GpuMat bgr[3];
|
||||
cv::cuda::split(imageF1_d,bgr);//split source
|
||||
|
||||
for(int i=0; i<dim.c; i++)
|
||||
checkCuda( cudaMemcpy(d_ptrs + i*dim.h * dim.w, (float*)bgr[i].data, dim.h * dim.w * sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
|
||||
normalize(d_ptrs, dim.c, dim.h, dim.w, mean_d, stddev_d);
|
||||
|
||||
checkCuda( cudaMemcpy(input_pre_inf_d, d_ptrs, dim2.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
#else
|
||||
cv::Mat imageF;
|
||||
resize(frame, imageF, cv::Size(dim.w, dim.h));
|
||||
// imageF = frame;
|
||||
sz = imageF.size();
|
||||
cv::warpAffine(imageF, imageF, trans, cv::Size(dim.w, dim.h), cv::INTER_LINEAR );
|
||||
|
||||
// cv::imshow("warp", imageF);
|
||||
|
||||
sz = imageF.size();
|
||||
imageF.convertTo(imageF, CV_32FC3, 1/255.0);
|
||||
|
||||
dim2 = dim;
|
||||
//split channels
|
||||
cv::Mat bgr[3];
|
||||
cv::split(imageF,bgr);//split source
|
||||
|
||||
for(int i=0; i<3; i++){
|
||||
bgr[i] = bgr[i] - mean[i];
|
||||
bgr[i] = bgr[i] / stddev[i];
|
||||
}
|
||||
for(int i=0; i<dim2.c; i++) {
|
||||
int idx = i * imageF.rows * imageF.cols;
|
||||
int ch = i;
|
||||
memcpy((void*)&input[idx], (void*)bgr[ch].data, imageF.rows*imageF.cols*sizeof(dnnType));
|
||||
}
|
||||
checkCuda( cudaMemcpyAsync(input_pre_inf_d, input, dim2.tot()*sizeof(dnnType), cudaMemcpyHostToDevice));
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
#endif
|
||||
|
||||
if(iter0) {
|
||||
checkCuda( cudaMemcpy(img_d, input_pre_inf_d, dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice) );
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
iter0=false;
|
||||
}
|
||||
pre_inf(bi);
|
||||
|
||||
checkCuda( cudaMemcpy(img_d, input_pre_inf_d, dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice) );
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
}
|
||||
|
||||
cv::Mat CenterTrack::transform_preds_with_trans(float x1, float x2){
|
||||
cv::Mat target_coords(cv::Size(1,3), CV_32F);
|
||||
target_coords.at<float>(0,0) = x1;
|
||||
target_coords.at<float>(0,1) = x2;
|
||||
target_coords.at<float>(0,2) = 1.0;
|
||||
return transOut * target_coords;
|
||||
}
|
||||
|
||||
void CenterTrack::tracking(const int bi) {
|
||||
std::vector<float> item_size(countDet);
|
||||
std::vector<int> item_cl(countDet);
|
||||
std::vector<float> dets(2*countDet);
|
||||
for(int i=0; i<countDet; i++){
|
||||
item_size[i] = (detRes[i].bb1.at<float>(0,0) - detRes[i].bb0.at<float>(0,0)) *
|
||||
(detRes[i].bb1.at<float>(0,1) - detRes[i].bb0.at<float>(0,1));
|
||||
item_cl[i] = detRes[i].cl;
|
||||
dets[i*2] = detRes[i].ct.at<float>(0,0);
|
||||
dets[i*2+1] = detRes[i].ct.at<float>(0,1);
|
||||
}
|
||||
|
||||
std::vector<float> track_size(countTr[bi]);
|
||||
std::vector<int> track_cl(countTr[bi]);
|
||||
std::vector<float> tracks(2*countTr[bi]);
|
||||
for(int i=0; i<countTr[bi]; i++){
|
||||
track_size[i] = (trRes[bi][i].det_res.bb1.at<float>(0,0) - trRes[bi][i].det_res.bb0.at<float>(0,0)) *
|
||||
(trRes[bi][i].det_res.bb1.at<float>(0,1) - trRes[bi][i].det_res.bb0.at<float>(0,1));
|
||||
track_cl[i] = trRes[bi][i].det_res.cl;
|
||||
tracks[i*2] = trRes[bi][i].det_res.ct.at<float>(0,0);
|
||||
tracks[i*2+1] = trRes[bi][i].det_res.ct.at<float>(0,1);
|
||||
}
|
||||
std::vector<float> dist(countTr[bi]*countDet);
|
||||
bool invalid;
|
||||
for(int i=0; i<countTr[bi]; i++){
|
||||
for(int j=0; j<countDet; j++){
|
||||
dist[j*countTr[bi]+i] = pow((tracks[i*2] - dets[j*2]), 2) +
|
||||
pow((tracks[i*2+1] - dets[j*2+1]), 2);
|
||||
invalid = dist[j*countTr[bi]+i] > track_size[i] ||
|
||||
dist[j*countTr[bi]+i] > item_size[j] ||
|
||||
item_cl[j] != track_cl[i];
|
||||
dist[j*countTr[bi]+i] = dist[j*countTr[bi]+i] + invalid * (1 << 18);
|
||||
}
|
||||
}
|
||||
std::vector<int> matched_indices(2*countTr[bi]);
|
||||
float min_tr;
|
||||
int min_idtr = -1;
|
||||
for(int i=0; i<countTr[bi]; i++) {
|
||||
matched_indices[i*2] = -1;
|
||||
matched_indices[i*2+1] = -1;
|
||||
}
|
||||
for(int i=0; i<countDet; i++){
|
||||
min_tr=(1 << 18);
|
||||
for(int j=0; j<countTr[bi]; j++){
|
||||
if(dist[i*countTr[bi]+j]<min_tr) {
|
||||
min_tr = dist[i*countTr[bi]+j];
|
||||
min_idtr = j;
|
||||
}
|
||||
}
|
||||
if(min_tr < (1<<16)) {
|
||||
for(int j=0; j<countDet; j++)
|
||||
dist[j*countTr[bi]+min_idtr] = (1 << 18);
|
||||
matched_indices[2*min_idtr] = min_idtr;
|
||||
matched_indices[2*min_idtr+1] = i;
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<bool> unmatched_dets(countDet);
|
||||
for(int i=0; i<countDet; i++)
|
||||
unmatched_dets[i] = false;
|
||||
std::vector<bool> unmatched_tracks(countTr[bi]);
|
||||
for(int i=0; i<countTr[bi]; i++)
|
||||
unmatched_tracks[i] = false;
|
||||
for(int i=0; i<countTr[bi]; i++) {
|
||||
if(matched_indices[2*i] != -1)
|
||||
unmatched_tracks[matched_indices[2*i]]=true;
|
||||
if(matched_indices[2*i+1] != -1)
|
||||
unmatched_dets[matched_indices[2*i+1]]=true;
|
||||
}
|
||||
|
||||
//match
|
||||
for(int i=0; i<countTr[bi]; i++) {
|
||||
if(matched_indices[2*i+1] != -1 && matched_indices[2*i] != -1) { //second condition is optional
|
||||
int tr_id = matched_indices[2*i];
|
||||
int d_id = matched_indices[2*i+1];
|
||||
|
||||
// trRes[tr_id].det_res = detRes[d_id];
|
||||
trRes[bi][tr_id].det_res.score = detRes[d_id].score;
|
||||
trRes[bi][tr_id].det_res.cl = detRes[d_id].cl;
|
||||
trRes[bi][tr_id].det_res.ct = detRes[d_id].ct;
|
||||
trRes[bi][tr_id].det_res.tr = detRes[d_id].tr;
|
||||
trRes[bi][tr_id].det_res.bb0 = detRes[d_id].bb0;
|
||||
trRes[bi][tr_id].det_res.bb1 = detRes[d_id].bb1;
|
||||
trRes[bi][tr_id].det_res.dep = detRes[d_id].dep;
|
||||
trRes[bi][tr_id].det_res.dim[0] = detRes[d_id].dim[0];
|
||||
trRes[bi][tr_id].det_res.dim[1] = detRes[d_id].dim[1];
|
||||
trRes[bi][tr_id].det_res.dim[2] = detRes[d_id].dim[2];
|
||||
trRes[bi][tr_id].det_res.alpha = detRes[d_id].alpha;
|
||||
trRes[bi][tr_id].det_res.x = detRes[d_id].x;
|
||||
trRes[bi][tr_id].det_res.y = detRes[d_id].y;
|
||||
trRes[bi][tr_id].det_res.z = detRes[d_id].z;
|
||||
trRes[bi][tr_id].det_res.rot_y = detRes[d_id].rot_y;
|
||||
// trRes[bi][matched_indices[2*i]].tracking_id = ; is the same
|
||||
// trRes[bi][matched_indices[2*i]].color = ; is the same
|
||||
trRes[bi][tr_id].age = 1;
|
||||
trRes[bi][tr_id].active = trRes[bi][tr_id].active+1;
|
||||
}
|
||||
}
|
||||
//delete target umatched track
|
||||
int new_count_tr = 0;
|
||||
for(int i=0; i<countTr[bi]; i++) {
|
||||
if(unmatched_tracks[i])
|
||||
new_count_tr++;
|
||||
}
|
||||
if(new_count_tr == 0 && countTr[bi] != 0) { //reset
|
||||
trRes[bi].clear();
|
||||
countTr[bi] = 0;
|
||||
}
|
||||
int old_count_tr = countTr[bi];
|
||||
if(countTr[bi] != 0 && new_count_tr != countTr[bi]) {
|
||||
std::vector<struct trackingRes> new_tr_res;
|
||||
int id_new_tr=0;
|
||||
for(int i=0; i<countTr[bi]; i++) {
|
||||
if(unmatched_tracks[i]) {
|
||||
struct trackingRes new_tr_res_;
|
||||
// new_tr_res_new_det_res.det_res = trRes[i].det_res;
|
||||
new_tr_res_.det_res.score = trRes[bi][i].det_res.score;
|
||||
new_tr_res_.det_res.cl = trRes[bi][i].det_res.cl;
|
||||
new_tr_res_.det_res.ct = trRes[bi][i].det_res.ct;
|
||||
new_tr_res_.det_res.tr = trRes[bi][i].det_res.tr;
|
||||
new_tr_res_.det_res.bb0 = trRes[bi][i].det_res.bb0;
|
||||
new_tr_res_.det_res.bb1 = trRes[bi][i].det_res.bb1;
|
||||
new_tr_res_.det_res.dep = trRes[bi][i].det_res.dep;
|
||||
new_tr_res_.det_res.dim[0] = trRes[bi][i].det_res.dim[0];
|
||||
new_tr_res_.det_res.dim[1] = trRes[bi][i].det_res.dim[1];
|
||||
new_tr_res_.det_res.dim[2] = trRes[bi][i].det_res.dim[2];
|
||||
new_tr_res_.det_res.alpha = trRes[bi][i].det_res.alpha;
|
||||
new_tr_res_.det_res.x = trRes[bi][i].det_res.x;
|
||||
new_tr_res_.det_res.y = trRes[bi][i].det_res.y;
|
||||
new_tr_res_.det_res.z = trRes[bi][i].det_res.z;
|
||||
new_tr_res_.det_res.rot_y = trRes[bi][i].det_res.rot_y;
|
||||
new_tr_res_.tracking_id = trRes[bi][i].tracking_id;
|
||||
new_tr_res_.age = trRes[bi][i].age;
|
||||
new_tr_res_.active = trRes[bi][i].active;
|
||||
new_tr_res_.color = trRes[bi][i].color;
|
||||
id_new_tr ++;
|
||||
new_tr_res.push_back(new_tr_res_);
|
||||
}
|
||||
}
|
||||
|
||||
if(countTr[bi]) {
|
||||
trRes[bi].clear();
|
||||
}
|
||||
countTr[bi] = new_count_tr;
|
||||
trRes[bi] = new_tr_res;
|
||||
}
|
||||
|
||||
int count_tr_ = countTr[bi];
|
||||
for(int i=0; i<countDet; i++) {
|
||||
if((!unmatched_dets[i]) && detRes[i].score > newThresh) {
|
||||
count_tr_ ++;
|
||||
struct trackingRes new_tr_res_;
|
||||
new_tr_res_.det_res.score = detRes[i].score;
|
||||
new_tr_res_.det_res.cl = detRes[i].cl;
|
||||
new_tr_res_.det_res.ct = detRes[i].ct;
|
||||
new_tr_res_.det_res.tr = detRes[i].tr;
|
||||
new_tr_res_.det_res.bb0 = detRes[i].bb0;
|
||||
new_tr_res_.det_res.bb1 = detRes[i].bb1;
|
||||
new_tr_res_.det_res.dep = detRes[i].dep;
|
||||
new_tr_res_.det_res.dim[0] = detRes[i].dim[0];
|
||||
new_tr_res_.det_res.dim[1] = detRes[i].dim[1];
|
||||
new_tr_res_.det_res.dim[2] = detRes[i].dim[2];
|
||||
new_tr_res_.det_res.alpha = detRes[i].alpha;
|
||||
new_tr_res_.det_res.x = detRes[i].x;
|
||||
new_tr_res_.det_res.y = detRes[i].y;
|
||||
new_tr_res_.det_res.z = detRes[i].z;
|
||||
new_tr_res_.det_res.rot_y = detRes[i].rot_y;
|
||||
new_tr_res_.tracking_id = trackId[bi]++;
|
||||
new_tr_res_.age = 1;
|
||||
new_tr_res_.active = 1;
|
||||
new_tr_res_.color = rand() % 256;
|
||||
if(trRes.size() <= bi) {
|
||||
std::vector<struct trackingRes> v_new_tr_res_;
|
||||
v_new_tr_res_.push_back(new_tr_res_);
|
||||
trRes.push_back(v_new_tr_res_);
|
||||
}
|
||||
else
|
||||
trRes[bi].push_back(new_tr_res_);
|
||||
}
|
||||
}
|
||||
|
||||
countTr[bi] = count_tr_;
|
||||
//reset the tracker id
|
||||
if(trackId[bi] == 1000)
|
||||
trackId[bi] = 0;
|
||||
detRes.clear();
|
||||
|
||||
}
|
||||
|
||||
void CenterTrack::postprocess(const int bi, const bool mAP) {
|
||||
dnnType *rt_out[9];
|
||||
rt_out[0] = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi;
|
||||
rt_out[1] = (dnnType *)netRT->buffersRT[2]+ netRT->buffersDIM[2].tot()*bi;
|
||||
rt_out[2] = (dnnType *)netRT->buffersRT[3]+ netRT->buffersDIM[3].tot()*bi;
|
||||
rt_out[3] = (dnnType *)netRT->buffersRT[4]+ netRT->buffersDIM[4].tot()*bi;
|
||||
rt_out[4] = (dnnType *)netRT->buffersRT[5]+ netRT->buffersDIM[5].tot()*bi;
|
||||
rt_out[5] = (dnnType *)netRT->buffersRT[6]+ netRT->buffersDIM[6].tot()*bi;
|
||||
rt_out[6] = (dnnType *)netRT->buffersRT[7]+ netRT->buffersDIM[7].tot()*bi;
|
||||
rt_out[7] = (dnnType *)netRT->buffersRT[8]+ netRT->buffersDIM[8].tot()*bi;
|
||||
rt_out[8] = (dnnType *)netRT->buffersRT[9]+ netRT->buffersDIM[9].tot()*bi;
|
||||
|
||||
// ------------------------------------ process --------------------------------------------
|
||||
|
||||
activationSIGMOIDForward(rt_out[0], rt_out[0], dim_hm.tot());
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
// output['dep'] = 1. / (output['dep'].sigmoid() + 1e-6) - 1.
|
||||
activationSIGMOIDForward(rt_out[5], rt_out[5], dim_dep.tot());
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
transformDep(ones, ones + dim_dep.tot(), rt_out[5], rt_out[5] + dim_dep.tot());
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
// nms
|
||||
subtractWithThreshold(rt_out[0], rt_out[0] + dim_hm.tot(), rt_out[1], rt_out[0], op);
|
||||
|
||||
// ----------- nms end
|
||||
// ----------- topk
|
||||
|
||||
if(K > dim_hm.h * dim_hm.w){
|
||||
printf ("Error topk (K is too large)\n");
|
||||
return;
|
||||
}
|
||||
|
||||
checkCuda( cudaMemcpy(ids_d, ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyHostToDevice) );
|
||||
|
||||
sort(rt_out[0],rt_out[0]+dim_hm.tot(),ids_d);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
topk(rt_out[0], ids_d, K, scores_d, topk_inds_d, topk_ys_d, topk_xs_d);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
checkCuda( cudaMemcpy(scores, scores_d, K *sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
topKxyclasses(topk_inds_d, topk_inds_d+K, K, width, dim_hm.w*dim_hm.h, clses_d, inttopk_xs_d, inttopk_ys_d);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
checkCuda( cudaMemcpy(topk_xs_d, (float *)inttopk_xs_d, K*sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
checkCuda( cudaMemcpy(topk_ys_d, (float *)inttopk_ys_d, K*sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
|
||||
checkCuda( cudaMemcpy(intxs, inttopk_xs_d, K * sizeof(int), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(intys, inttopk_ys_d, K * sizeof(int), cudaMemcpyDeviceToHost) );
|
||||
|
||||
checkCuda( cudaMemcpy(clses, clses_d, K*sizeof(int), cudaMemcpyDeviceToHost) );
|
||||
|
||||
// ----------- topk end
|
||||
|
||||
topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3], src_out, ids_out);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
bboxes(topk_inds_d, K, dim_wh.h*dim_wh.w, topk_xs_d, topk_ys_d, rt_out[2], bbx0_d, bbx1_d, bby0_d, bby1_d, src_out, ids_out);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
checkCuda( cudaMemcpy(bbx0, bbx0_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(bby0, bby0_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(bbx1, bbx1_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(bby1, bby1_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
//regression heads
|
||||
// ['tracking', 'dep', 'rot', 'dim', 'amodel_offset',
|
||||
// 'nuscenes_att', 'velocity']
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_track.c, dim_track.h * dim_track.w, rt_out[4], track_d, ids_out);
|
||||
checkCuda( cudaMemcpy(track, track_d, K * dim_track.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_dep.c, dim_dep.h * dim_dep.w, rt_out[5], dep_d, ids_out);
|
||||
checkCuda( cudaMemcpy(dep, dep_d, K * dim_dep.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_rot.c, dim_rot.h * dim_rot.w, rt_out[6], rot_d, ids_out);
|
||||
checkCuda( cudaMemcpy(rot, rot_d, K * dim_rot.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_dim.c, dim_dim.h * dim_dim.w, rt_out[7], dim_d, ids_out);
|
||||
checkCuda( cudaMemcpy(dim_, dim_d, K * dim_dim.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_amodel_offset.c, dim_amodel_offset.h * dim_amodel_offset.w, rt_out[8], amodel_offset_d, ids_out);
|
||||
checkCuda( cudaMemcpy(amodel_offset, amodel_offset_d, K * dim_amodel_offset.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
// ---------------------------------- post-process -----------------------------------------
|
||||
|
||||
countDet = 0;
|
||||
detRes.clear();
|
||||
for(int i=0; i<K; i++){
|
||||
if(scores[i] < outThresh)
|
||||
break;
|
||||
|
||||
countDet ++;
|
||||
struct detectionRes new_det_res;
|
||||
new_det_res.score = scores[i];
|
||||
new_det_res.cl = clses[i]+1;
|
||||
// ret_s=scores[i];
|
||||
// ret_c=clses[i]+1;
|
||||
new_det_res.ct = transform_preds_with_trans(intxs[i], intys[i]);
|
||||
new_det_res.tr = transform_preds_with_trans(intxs[i] + track[i], intys[i] + track[i+K]);
|
||||
new_det_res.tr = new_det_res.tr -new_det_res.ct;
|
||||
new_det_res.bb0 = transform_preds_with_trans(bbx0[i], bby0[i]);
|
||||
new_det_res.bb1 = transform_preds_with_trans(bbx1[i], bby1[i]);
|
||||
new_det_res.ct = transform_preds_with_trans(((bbx0[i]+bbx1[i])/2 + amodel_offset[i]),
|
||||
((bby0[i]+bby1[i])/2 + amodel_offset[i+K]));
|
||||
new_det_res.dep = dep[i];
|
||||
new_det_res.dim[0] = dim_[i];
|
||||
new_det_res.dim[1] = dim_[i+K];
|
||||
new_det_res.dim[2] = dim_[i+2*K];
|
||||
|
||||
// unproject_2d_to_3d
|
||||
new_det_res.z = dep[i] - calibs[bi].at<float>(2,3);
|
||||
new_det_res.x = ((float)new_det_res.ct.at<float>(0,0) * dep[i] - calibs[bi].at<float>(0,3) -
|
||||
calibs[bi].at<float>(0,2) * new_det_res.z) / calibs[bi].at<float>(0,0);
|
||||
new_det_res.y = ((float)new_det_res.ct.at<float>(0,1) * dep[i] - calibs[bi].at<float>(1,3) -
|
||||
calibs[bi].at<float>(1,2) * new_det_res.z) / calibs[bi].at<float>(1,1) + (dim_[i] / 2);
|
||||
|
||||
// alpha2rot_y
|
||||
// idx = rot[:, 1] > rot[:, 5]
|
||||
// alpha1 = np.arctan2(rot[:, 2], rot[:, 3]) + (-0.5 * np.pi)
|
||||
// alpha2 = np.arctan2(rot[:, 6], rot[:, 7]) + ( 0.5 * np.pi)
|
||||
// return alpha1 * idx + alpha2 * (1 - idx)
|
||||
if(rot[1*K + i] > rot[5*K + i])
|
||||
new_det_res.alpha = std::atan2(rot[2*K + i], rot[3*K + i]) -0.5 * M_PI;
|
||||
else
|
||||
new_det_res.alpha = std::atan2(rot[6*K + i], rot[7*K + i]) +0.5 * M_PI;
|
||||
new_det_res.rot_y = (new_det_res.alpha + std::atan2((float)new_det_res.ct.at<float>(0,0) - calibs[bi].at<float>(0,2), calibs[bi].at<float>(0,0)));
|
||||
new_det_res.ct = new_det_res.ct + new_det_res.tr; //dest
|
||||
detRes.push_back(new_det_res);
|
||||
}
|
||||
// track step
|
||||
tracking(bi);
|
||||
}
|
||||
|
||||
void CenterTrack::draw(std::vector<cv::Mat>& frames) {
|
||||
struct trackingRes t;
|
||||
float sc;
|
||||
int id;
|
||||
std::string txt;
|
||||
int baseline = 0;
|
||||
float font_scale = 0.8;
|
||||
int thickness = 2;
|
||||
|
||||
for(int bi=0; bi<frames.size(); ++bi) {
|
||||
float scale_x = float(originalSize[bi].width)/dim.w;
|
||||
float scale_y = float(originalSize[bi].height)/dim.h;
|
||||
resize(frames[bi], frames[bi], originalSize[bi]);
|
||||
// draw dets
|
||||
for(int i=0; trRes.size() != 0 && i<trRes[bi].size(); i++) {
|
||||
t = trRes[bi][i];
|
||||
id = t.tracking_id;
|
||||
txt = classesNames[t.det_res.cl-1]+'-'+std::to_string(id); //forse ha bisogno di cl-1
|
||||
cv::Size text_size = getTextSize(txt, cv::FONT_HERSHEY_SIMPLEX, font_scale, thickness, &baseline);
|
||||
|
||||
if(t.det_res.score > confThreshold){// && t.active!=0) {
|
||||
if(!mode3D) {
|
||||
cv::rectangle(frames[bi],
|
||||
cv::Point(t.det_res.bb0.at<float>(0,0) * scale_x, t.det_res.bb0.at<float>(0,1) * scale_y),
|
||||
cv::Point(t.det_res.bb1.at<float>(0,0) * scale_x, t.det_res.bb1.at<float>(0,1) * scale_y),
|
||||
trColors[t.color], thickness);
|
||||
cv::rectangle(frames[bi],
|
||||
cv::Point(t.det_res.bb0.at<float>(0,0) * scale_x, t.det_res.bb0.at<float>(0,1) * scale_y - text_size.height - thickness),
|
||||
cv::Point(t.det_res.bb0.at<float>(0,0) * scale_x + text_size.width, t.det_res.bb0.at<float>(0,1) * scale_y),
|
||||
trColors[t.color], -1);
|
||||
|
||||
cv::putText(frames[bi], txt,
|
||||
cv::Point(t.det_res.bb0.at<float>(0,0) * scale_x, t.det_res.bb0.at<float>(0,1) * scale_y - thickness -1),
|
||||
cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1);
|
||||
|
||||
cv::arrowedLine(frames[bi],
|
||||
cv::Point((int)t.det_res.ct.at<float>(0,0) * scale_x, (int)t.det_res.ct.at<float>(0,1) * scale_y),
|
||||
cv::Point((int)(t.det_res.ct.at<float>(0,0) * scale_x + t.det_res.tr.at<float>(0,0) * scale_x),
|
||||
(int)(t.det_res.ct.at<float>(0,1) * scale_y + t.det_res.tr.at<float>(0,1) * scale_y)),
|
||||
cv::Scalar(255, 0, 255), 2);
|
||||
}
|
||||
//3d
|
||||
if(mode3D && t.det_res.z > 1){
|
||||
r.at<float>(0,0) = std::cos(t.det_res.rot_y);
|
||||
r.at<float>(0,2) = std::sin(t.det_res.rot_y);
|
||||
r.at<float>(2,0) = -std::sin(t.det_res.rot_y);
|
||||
r.at<float>(2,2) = std::cos(t.det_res.rot_y);
|
||||
|
||||
corners.at<float>(0,0) = t.det_res.dim[2]/2;
|
||||
corners.at<float>(0,1) = t.det_res.dim[2]/2;
|
||||
corners.at<float>(0,2) = -t.det_res.dim[2]/2;
|
||||
corners.at<float>(0,3) = -t.det_res.dim[2]/2;
|
||||
corners.at<float>(0,4) = t.det_res.dim[2]/2;
|
||||
corners.at<float>(0,5) = t.det_res.dim[2]/2;
|
||||
corners.at<float>(0,6) = -t.det_res.dim[2]/2;
|
||||
corners.at<float>(0,7) = -t.det_res.dim[2]/2;
|
||||
|
||||
corners.at<float>(1,4) = -t.det_res.dim[0];
|
||||
corners.at<float>(1,5) = -t.det_res.dim[0];
|
||||
corners.at<float>(1,6) = -t.det_res.dim[0];
|
||||
corners.at<float>(1,7) = -t.det_res.dim[0];
|
||||
|
||||
corners.at<float>(2,0) = t.det_res.dim[1]/2;
|
||||
corners.at<float>(2,1) = -t.det_res.dim[1]/2;
|
||||
corners.at<float>(2,2) = -t.det_res.dim[1]/2;
|
||||
corners.at<float>(2,3) = t.det_res.dim[1]/2;
|
||||
corners.at<float>(2,4) = t.det_res.dim[1]/2;
|
||||
corners.at<float>(2,5) = -t.det_res.dim[1]/2;
|
||||
corners.at<float>(2,6) = -t.det_res.dim[1]/2;
|
||||
corners.at<float>(2,7) = t.det_res.dim[1]/2;
|
||||
|
||||
cv::Mat aus = r * corners;
|
||||
|
||||
for(int k=0; k<8; k++) {
|
||||
aus.at<float>(0,k) += t.det_res.x;
|
||||
aus.at<float>(1,k) += t.det_res.y;
|
||||
aus.at<float>(2,k) += t.det_res.z;
|
||||
}
|
||||
|
||||
// corners.copyTo(pts3DHomo(cv::Rect(0, 0, 8, 3)));
|
||||
for(int k1=0; k1<3; k1++) {
|
||||
for(int k2=0; k2<8; k2++)
|
||||
pts3DHomo.at<float>(k1,k2) = aus.at<float>(k1,k2);
|
||||
}
|
||||
|
||||
aus.release();
|
||||
aus = calibs[bi] * pts3DHomo;
|
||||
std::vector<float> res_corners;
|
||||
for(int k=0; k<8; k++) {
|
||||
res_corners.push_back(aus.at<float>(0,k) / aus.at<float>(2,k));
|
||||
res_corners.push_back(aus.at<float>(1,k) / aus.at<float>(2,k));
|
||||
}
|
||||
aus.release();
|
||||
for(int ind_f=3; ind_f>=0; ind_f--) {
|
||||
for(int j=0; j<4; j++) {
|
||||
cv::line(frames[bi],
|
||||
cv::Point((int)res_corners.at(faceId.at(ind_f).at(j) * 2) * scale_x,
|
||||
(int)res_corners.at(faceId.at(ind_f).at(j) * 2 + 1) * scale_y),
|
||||
cv::Point((int)res_corners.at(faceId.at(ind_f).at((j+1)%4) * 2) * scale_x,
|
||||
(int)res_corners.at(faceId.at(ind_f).at((j+1)%4) * 2 + 1) * scale_y),
|
||||
trColors[t.color], 2);
|
||||
if(ind_f == 0 && j==3) {
|
||||
cv::line(frames[bi],
|
||||
cv::Point((int)res_corners.at(faceId.at(ind_f).at(0) * 2) * scale_x,
|
||||
(int)res_corners.at(faceId.at(ind_f).at(0) * 2 + 1) * scale_y),
|
||||
cv::Point((int)res_corners.at(faceId.at(ind_f).at(2) * 2) * scale_x,
|
||||
(int)res_corners.at(faceId.at(ind_f).at(2) * 2 + 1) * scale_y), trColors[t.color], 2);
|
||||
cv::line(frames[bi],
|
||||
cv::Point((int)res_corners.at(faceId.at(ind_f).at(1) * 2) * scale_x,
|
||||
(int)res_corners.at(faceId.at(ind_f).at(1) * 2 + 1) * scale_y),
|
||||
cv::Point((int)res_corners.at(faceId.at(ind_f).at(3) * 2) * scale_x,
|
||||
(int)res_corners.at(faceId.at(ind_f).at(3) * 2 + 1) * scale_y), trColors[t.color], 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
float bb0=(1 << 10), bb1=0, bb2=(1 << 10), bb3=0;
|
||||
for(int k=0; k<8; k++) {
|
||||
if(res_corners[2*k] < bb0)
|
||||
bb0 = res_corners[2*k];
|
||||
if(res_corners[2*k] > bb1)
|
||||
bb1 = res_corners[2*k];
|
||||
if(res_corners[2*k+1] < bb2)
|
||||
bb2 = res_corners[2*k+1];
|
||||
if(res_corners[2*k+1] > bb3)
|
||||
bb3 = res_corners[2*k+1];
|
||||
|
||||
}
|
||||
// if(not no_bbox):
|
||||
// cv::rectangle(frame,
|
||||
// cv::Point(bb0, bb2),
|
||||
// cv::Point(bb1, bb3),
|
||||
// trColors[t.color], thickness);
|
||||
cv::rectangle(frames[bi],
|
||||
cv::Point(bb0 * scale_x, bb2 * scale_y - text_size.height - thickness),
|
||||
cv::Point(bb0 * scale_x + text_size.width, bb2 * scale_y),
|
||||
trColors[t.color], -1);
|
||||
|
||||
cv::putText(frames[bi], txt,
|
||||
cv::Point(bb0 * scale_x, bb2 * scale_y - thickness -1),
|
||||
cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), 1);
|
||||
|
||||
cv::arrowedLine(frames[bi],
|
||||
cv::Point((int)((bb0 + bb1)/2) * scale_x, (int)((bb2 + bb3)/2) * scale_y),
|
||||
cv::Point((int)((bb0 + bb1)/2 + t.det_res.tr.at<float>(0,0)) * scale_x,
|
||||
(int)((bb2 + bb3)/2 + t.det_res.tr.at<float>(0,1)) * scale_y),
|
||||
cv::Scalar(255, 0, 255), 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}}
|
||||
|
||||
|
||||
@@ -0,0 +1,405 @@
|
||||
#include "CenternetDetection.h"
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
bool CenternetDetection::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh){
|
||||
std::cout<<(tensor_path).c_str()<<"\n";
|
||||
netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() );
|
||||
classes = n_classes;
|
||||
nBatches = n_batches;
|
||||
confThreshold = conf_thresh;
|
||||
|
||||
dim = netRT->input_dim;
|
||||
|
||||
const char *coco_class_name[] = {
|
||||
"person", "bicycle", "car", "motorcycle", "airplane",
|
||||
"bus", "train", "truck", "boat", "traffic light", "fire hydrant",
|
||||
"stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse",
|
||||
"sheep", "cow", "elephant", "bear", "zebra", "giraffe", "backpack",
|
||||
"umbrella", "handbag", "tie", "suitcase", "frisbee", "skis",
|
||||
"snowboard", "sports ball", "kite", "baseball bat", "baseball glove",
|
||||
"skateboard", "surfboard", "tennis racket", "bottle", "wine glass",
|
||||
"cup", "fork", "knife", "spoon", "bowl", "banana", "apple", "sandwich",
|
||||
"orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake",
|
||||
"chair", "couch", "potted plant", "bed", "dining table", "toilet", "tv",
|
||||
"laptop", "mouse", "remote", "keyboard", "cell phone", "microwave",
|
||||
"oven", "toaster", "sink", "refrigerator", "book", "clock", "vase",
|
||||
"scissors", "teddy bear", "hair drier", "toothbrush"
|
||||
};
|
||||
classesNames = std::vector<std::string>(coco_class_name, std::end( coco_class_name));
|
||||
|
||||
for(int c=0; c<classes; c++) {
|
||||
int offset = c*123457 % classes;
|
||||
float r = getColor(2, offset, classes);
|
||||
float g = getColor(1, offset, classes);
|
||||
float b = getColor(0, offset, classes);
|
||||
colors[c] = cv::Scalar(int(255.0*b), int(255.0*g), int(255.0*r));
|
||||
}
|
||||
|
||||
src = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
dst = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
dst2 = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
trans = cv::Mat(cv::Size(3,2), CV_32F);
|
||||
trans2 = cv::Mat(cv::Size(3,2), CV_32F);
|
||||
|
||||
checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches));
|
||||
|
||||
dim_hm = tk::dnn::dataDim_t(1, 80, 128, 128, 1);
|
||||
dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1);
|
||||
dim_reg = tk::dnn::dataDim_t(1, 2, 128, 128, 1);
|
||||
|
||||
checkCuda( cudaMalloc(&topk_scores, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_inds_, dim_hm.c * K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&topk_ys_, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_xs_, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&ids_d, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&ids_2d, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) );
|
||||
checkCuda( cudaMallocHost(&ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) );
|
||||
checkCuda( cudaMallocHost(&ids_2, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) );
|
||||
for(int i =0; i<dim_hm.c * dim_hm.h * dim_hm.w; i++){
|
||||
ids_[i] = i;
|
||||
}
|
||||
int val = 0;
|
||||
for(int i =0; i <dim_hm.c * dim_hm.h * dim_hm.w; i++){
|
||||
ids_2[i] = val;
|
||||
if(i%dim_hm.c == 0)
|
||||
val = 0;
|
||||
}
|
||||
|
||||
checkCuda( cudaMallocHost(&scores, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&scores_d, K *sizeof(float)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&clses, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&clses_d, K *sizeof(int)) );
|
||||
|
||||
checkCuda( cudaMalloc(&topk_inds_d, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&topk_ys_d, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_xs_d, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&inttopk_ys_d, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&inttopk_xs_d, K *sizeof(int)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&bbx0, K * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&bby0, K * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&bbx1, K * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&bby1, K * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&bbx0_d, K * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&bby0_d, K * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&bbx1_d, K * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&bby1_d, K * sizeof(float)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&target_coords, 4 * K *sizeof(float)) );
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
|
||||
checkCuda( cudaMalloc(&mean_d, 3 * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&stddev_d, 3 * sizeof(float)) );
|
||||
float mean[3] = {0.408, 0.447, 0.47};
|
||||
float stddev[3] = {0.289, 0.274, 0.278};
|
||||
|
||||
checkCuda(cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice));
|
||||
checkCuda(cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice));
|
||||
#else
|
||||
checkCuda(cudaMallocHost(&input, sizeof(dnnType)*netRT->input_dim.tot()* nBatches));
|
||||
mean << 0.408, 0.447, 0.47;
|
||||
stddev << 0.289, 0.274, 0.278;
|
||||
#endif
|
||||
|
||||
checkCuda( cudaMalloc(&d_ptrs, dim.c * dim.h*dim.w * sizeof(float)) );
|
||||
|
||||
// Alloc array used in the kernel
|
||||
checkCuda( cudaMalloc(&src_out, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&ids_out, K *sizeof(int)) );
|
||||
|
||||
dst2.at<float>(0,0)=width * 0.5;
|
||||
dst2.at<float>(0,1)=width * 0.5;
|
||||
dst2.at<float>(1,0)=width * 0.5;
|
||||
dst2.at<float>(1,1)=width * 0.5 + width * -0.5;
|
||||
|
||||
dst2.at<float>(2,0)=dst2.at<float>(1,0) + (-dst2.at<float>(0,1)+dst2.at<float>(1,1) );
|
||||
dst2.at<float>(2,1)=dst2.at<float>(1,1) + (dst2.at<float>(0,0)-dst2.at<float>(1,0) );
|
||||
return true;
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
void CenternetDetection::preprocess(cv::Mat &frame, const int bi){
|
||||
// -----------------------------------pre-process ------------------------------------------
|
||||
|
||||
// auto start_t = std::chrono::steady_clock::now();
|
||||
// auto step_t = std::chrono::steady_clock::now();
|
||||
// auto end_t = std::chrono::steady_clock::now();
|
||||
cv::Size sz = originalSize[bi];
|
||||
// std::cout<<"image: "<<sz.width<<", "<<sz.height<<std::endl;
|
||||
cv::Size sz_old;
|
||||
float scale = 1.0;
|
||||
float new_height = sz.height * scale;
|
||||
float new_width = sz.width * scale;
|
||||
if(sz.height != sz_old.height && sz.width != sz_old.width){
|
||||
float c[] = {new_width / 2.0f, new_height /2.0f};
|
||||
float s[2];
|
||||
|
||||
if(sz.width > sz.height){
|
||||
s[0] = sz.width * 1.0;
|
||||
s[1] = sz.width * 1.0;
|
||||
}
|
||||
else{
|
||||
s[0] = sz.height * 1.0;
|
||||
s[1] = sz.height * 1.0;
|
||||
}
|
||||
|
||||
// ----------- get_affine_transform
|
||||
// rot_rad = pi * 0 / 100 --> 0
|
||||
|
||||
src.at<float>(0,0)=c[0];
|
||||
src.at<float>(0,1)=c[1];
|
||||
src.at<float>(1,0)=c[0];
|
||||
src.at<float>(1,1)=c[1] + s[0] * -0.5;
|
||||
dst.at<float>(0,0)=netRT->input_dim.w * 0.5;
|
||||
dst.at<float>(0,1)=netRT->input_dim.h * 0.5;
|
||||
dst.at<float>(1,0)=netRT->input_dim.w * 0.5;
|
||||
dst.at<float>(1,1)=netRT->input_dim.h * 0.5 + netRT->input_dim.w * -0.5;
|
||||
|
||||
src.at<float>(2,0)=src.at<float>(1,0) + (-src.at<float>(0,1)+src.at<float>(1,1) );
|
||||
src.at<float>(2,1)=src.at<float>(1,1) + (src.at<float>(0,0)-src.at<float>(1,0) );
|
||||
dst.at<float>(2,0)=dst.at<float>(1,0) + (-dst.at<float>(0,1)+dst.at<float>(1,1) );
|
||||
dst.at<float>(2,1)=dst.at<float>(1,1) + (dst.at<float>(0,0)-dst.at<float>(1,0) );
|
||||
|
||||
trans = cv::getAffineTransform( src, dst );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME gett affine trans: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
trans2 = cv::getAffineTransform( dst2, src );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME getAffineTrans 2: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
}
|
||||
sz_old = sz;
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
cv::cuda::GpuMat im_Orig;
|
||||
cv::cuda::GpuMat imageF1_d, imageF2_d;
|
||||
|
||||
im_Orig = cv::cuda::GpuMat(frame);
|
||||
cv::cuda::resize (im_Orig, imageF1_d, cv::Size(new_width, new_height));
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
sz = imageF1_d.size();
|
||||
// std::cout<<"size: "<<sz.height<<" "<<sz.width<<" - "<<std::endl;
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME resize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
cv::cuda::warpAffine(imageF1_d, imageF2_d, trans, cv::Size(netRT->input_dim.w, netRT->input_dim.h), cv::INTER_LINEAR );
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
imageF2_d.convertTo(imageF1_d, CV_32FC3, 1/255.0);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME convert: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
dim2 = dim;
|
||||
cv::cuda::GpuMat bgr[3];
|
||||
cv::cuda::split(imageF1_d,bgr);//split source
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME split: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
for(int i=0; i<dim.c; i++)
|
||||
checkCuda( cudaMemcpy(d_ptrs + i*dim.h * dim.w, (float*)bgr[i].data, dim.h * dim.w * sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
|
||||
normalize(d_ptrs, dim.c, dim.h, dim.w, mean_d, stddev_d);
|
||||
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME normalize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
checkCuda(cudaMemcpy(input_d+ netRT->input_dim.tot()*bi, d_ptrs, dim2.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME Memcpy to input_d: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
#else
|
||||
|
||||
cv::Mat imageF;
|
||||
resize(frame, imageF, cv::Size(new_width, new_height));
|
||||
sz = imageF.size();
|
||||
// std::cout<<"size: "<<sz.height<<" "<<sz.width<<" - "<<std::endl;
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME resize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
cv::Mat trans = cv::getAffineTransform( src, dst );
|
||||
cv::warpAffine(imageF, imageF, trans, cv::Size(netRT->input_dim.w, netRT->input_dim.h), cv::INTER_LINEAR );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME warpAffine: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
sz = imageF.size();
|
||||
// std::cout<<"size: "<<sz.height<<" "<<sz.width<<" - "<<std::endl;
|
||||
imageF.convertTo(imageF, CV_32FC3, 1/255.0);
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME convertto: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
dim2 = dim;
|
||||
//split channels
|
||||
cv::Mat bgr[3];
|
||||
cv::split(imageF,bgr);//split source
|
||||
for(int i=0; i<3; i++){
|
||||
bgr[i] = bgr[i] - mean[i];
|
||||
bgr[i] = bgr[i] / stddev[i];
|
||||
}
|
||||
|
||||
//write channels
|
||||
for(int i=0; i<dim2.c; i++) {
|
||||
int idx = i*imageF.rows*imageF.cols;
|
||||
int ch = dim2.c-3 +i;
|
||||
// std::cout<<"i: "<<i<<", idx: "<<idx<<", ch: "<<ch<<std::endl;
|
||||
memcpy((void*)&input[idx+ netRT->input_dim.tot()*bi], (void*)bgr[ch].data, imageF.rows*imageF.cols*sizeof(dnnType));
|
||||
}
|
||||
checkCuda(cudaMemcpyAsync(input_d+ netRT->input_dim.tot()*bi, input+ netRT->input_dim.tot()*bi, dim2.tot()*sizeof(dnnType), cudaMemcpyHostToDevice));
|
||||
#endif
|
||||
}
|
||||
|
||||
void CenternetDetection::postprocess(const int bi, const bool mAP){
|
||||
dnnType *rt_out[4];
|
||||
rt_out[0] = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi;
|
||||
rt_out[1] = (dnnType *)netRT->buffersRT[2]+ netRT->buffersDIM[2].tot()*bi;
|
||||
rt_out[2] = (dnnType *)netRT->buffersRT[3]+ netRT->buffersDIM[3].tot()*bi;
|
||||
rt_out[3] = (dnnType *)netRT->buffersRT[4]+ netRT->buffersDIM[4].tot()*bi;
|
||||
|
||||
// auto start_t = std::chrono::steady_clock::now();
|
||||
// auto step_t = std::chrono::steady_clock::now();
|
||||
// auto end_t = std::chrono::steady_clock::now();
|
||||
// ------------------------------------ process --------------------------------------------
|
||||
activationSIGMOIDForward(rt_out[0], rt_out[0], dim_hm.tot());
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
subtractWithThreshold(rt_out[0], rt_out[0] + dim_hm.tot(), rt_out[1], rt_out[0], op);
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME threshold: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
// ----------- nms end
|
||||
// ----------- topk
|
||||
|
||||
if(K > dim_hm.h * dim_hm.w){
|
||||
printf ("Error topk (K is too large)\n");
|
||||
return;
|
||||
}
|
||||
|
||||
checkCuda( cudaMemcpy(ids_d, ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyHostToDevice) );
|
||||
|
||||
sort(rt_out[0],rt_out[0]+dim_hm.tot(),ids_d);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME sort: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
topk(rt_out[0], ids_d, K, scores_d, topk_inds_d, topk_ys_d, topk_xs_d);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME topk: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
checkCuda( cudaMemcpy(scores, scores_d, K *sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
topKxyclasses(topk_inds_d, topk_inds_d+K, K, width, dim_hm.w*dim_hm.h, clses_d, inttopk_xs_d, inttopk_ys_d);
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME topk x y clses 2: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
checkCuda( cudaMemcpy(topk_xs_d, (float *)inttopk_xs_d, K*sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
checkCuda( cudaMemcpy(topk_ys_d, (float *)inttopk_ys_d, K*sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
|
||||
checkCuda( cudaMemcpy(clses, clses_d, K*sizeof(int), cudaMemcpyDeviceToHost) );
|
||||
|
||||
// ----------- topk end
|
||||
|
||||
topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3], src_out, ids_out);
|
||||
// checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME add offset: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
bboxes(topk_inds_d, K, dim_wh.h*dim_wh.w, topk_xs_d, topk_ys_d, rt_out[2], bbx0_d, bbx1_d, bby0_d, bby1_d, src_out, ids_out);
|
||||
// checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
checkCuda( cudaMemcpy(bbx0, bbx0_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(bby0, bby0_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(bbx1, bbx1_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(bby1, bby1_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME bboxes: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
// ---------------------------------- post-process -----------------------------------------
|
||||
|
||||
// --------- ctdet_post_process
|
||||
// --------- transform_preds
|
||||
cv::Mat new_pt1(cv::Size(1,2), CV_32F);
|
||||
cv::Mat new_pt2(cv::Size(1,2), CV_32F);
|
||||
|
||||
for(int i = 0; i<K; i++){
|
||||
new_pt1.at<float>(0,0)=static_cast<float>(trans2.at<double>(0,0))*bbx0[i] +
|
||||
static_cast<float>(trans2.at<double>(0,1))*bby0[i] +
|
||||
static_cast<float>(trans2.at<double>(0,2))*1.0;
|
||||
new_pt1.at<float>(1,0)=static_cast<float>(trans2.at<double>(1,0))*bbx0[i] +
|
||||
static_cast<float>(trans2.at<double>(1,1))*bby0[i] +
|
||||
static_cast<float>(trans2.at<double>(1,2))*1.0;
|
||||
|
||||
new_pt2.at<float>(0,0)=static_cast<float>(trans2.at<double>(0,0))*bbx1[i] +
|
||||
static_cast<float>(trans2.at<double>(0,1))*bby1[i] +
|
||||
static_cast<float>(trans2.at<double>(0,2))*1.0;
|
||||
new_pt2.at<float>(1,0)=static_cast<float>(trans2.at<double>(1,0))*bbx1[i] +
|
||||
static_cast<float>(trans2.at<double>(1,1))*bby1[i] +
|
||||
static_cast<float>(trans2.at<double>(1,2))*1.0;
|
||||
|
||||
target_coords[i*4] = new_pt1.at<float>(0,0);
|
||||
target_coords[i*4+1] = new_pt1.at<float>(1,0);
|
||||
target_coords[i*4+2] = new_pt2.at<float>(0,0);
|
||||
target_coords[i*4+3] = new_pt2.at<float>(1,0);
|
||||
}
|
||||
|
||||
detected.clear();
|
||||
for(int i = 0; i<classes; i++){
|
||||
for(int j=0; j<K; j++)
|
||||
if(clses[j] == i){
|
||||
if(scores[j] > confThreshold){
|
||||
// std::cout<<"th: "<<scores[j]<<" - cl: "<<clses[j]<<" i: "<<i<<std::endl;
|
||||
//add coco bbox
|
||||
//det[0:4], i, det[4]
|
||||
float x0 = target_coords[j*4];
|
||||
float y0 = target_coords[j*4+1];
|
||||
float x1 = target_coords[j*4+2];
|
||||
float y1 = target_coords[j*4+3];
|
||||
int obj_class = clses[j];
|
||||
float prob = scores[j];
|
||||
// std::cout<<"("<<x0<<", "<<y0<<"),("<<x1<<", "<<y1<<")"<<std::endl;
|
||||
tk::dnn::box res;
|
||||
res.cl = obj_class;
|
||||
res.prob = prob;
|
||||
res.x = x0;
|
||||
res.y = y0;
|
||||
res.w = x1 - x0;
|
||||
res.h = y1 - y0;
|
||||
detected.push_back(res);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
batchDetected.push_back(detected);
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME detections: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
}
|
||||
|
||||
|
||||
}}
|
||||
|
||||
|
||||
@@ -0,0 +1,540 @@
|
||||
#include "CenternetDetection3D.h"
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
bool CenternetDetection3D::init(const std::string& tensor_path, const int n_classes, const int n_batches,
|
||||
const float conf_thresh, const std::vector<cv::Mat>& k_calibs) {
|
||||
std::cout<<(tensor_path).c_str()<<"\n";
|
||||
netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() );
|
||||
classes = n_classes;
|
||||
nBatches = n_batches;
|
||||
confThreshold = conf_thresh;
|
||||
inputCalibs = k_calibs;
|
||||
dim = netRT->input_dim;
|
||||
|
||||
const char *kitti_class_name[] = {
|
||||
"person", "car", "bicycle"};
|
||||
classesNames = std::vector<std::string>(kitti_class_name, std::end( kitti_class_name));
|
||||
|
||||
for(int c=0; c<classes; c++) {
|
||||
int offset = c*123457 % classes;
|
||||
float r = getColor(2, offset, classes);
|
||||
float g = getColor(1, offset, classes);
|
||||
float b = getColor(0, offset, classes);
|
||||
colors[c] = cv::Scalar(int(255.0*b), int(255.0*g), int(255.0*r));
|
||||
}
|
||||
|
||||
src = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
dst = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
dst2 = cv::Mat(cv::Size(2,3), CV_32F);
|
||||
trans = cv::Mat(cv::Size(3,2), CV_32F);
|
||||
trans2 = cv::Mat(cv::Size(3,2), CV_32F);
|
||||
|
||||
checkCuda(cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches));
|
||||
|
||||
dim_hm = tk::dnn::dataDim_t(1, 3, 128, 128, 1);
|
||||
dim_wh = tk::dnn::dataDim_t(1, 2, 128, 128, 1);
|
||||
dim_reg = tk::dnn::dataDim_t(1, 2, 128, 128, 1);
|
||||
dim_dep = tk::dnn::dataDim_t(1, 1, 128, 128, 1);
|
||||
dim_rot = tk::dnn::dataDim_t(1, 8, 128, 128, 1);
|
||||
dim_dim = tk::dnn::dataDim_t(1, 3, 128, 128, 1);
|
||||
|
||||
checkCuda( cudaMalloc(&topk_scores, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_inds_, dim_hm.c * K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&topk_ys_, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_xs_, dim_hm.c * K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&ids_d, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) );
|
||||
checkCuda( cudaMallocHost(&ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int)) );
|
||||
for(int i =0; i<dim_hm.c * dim_hm.h * dim_hm.w; i++){
|
||||
ids_[i] = i;
|
||||
}
|
||||
|
||||
checkCuda( cudaMalloc(&ones, dim_dep.c * dim_dep.h * dim_dep.w * sizeof(float)) );
|
||||
float *ones_h;
|
||||
checkCuda( cudaMallocHost(&ones_h, dim_dep.c * dim_dep.h * dim_dep.w * sizeof(float)) );
|
||||
for(int i=0; i<dim_dep.c * dim_dep.h * dim_dep.w; i++)
|
||||
ones_h[i]=1.0f;
|
||||
checkCuda( cudaMemcpy(ones, ones_h, dim_dep.c * dim_dep.h * dim_dep.w * sizeof(float), cudaMemcpyHostToDevice) );
|
||||
checkCuda( cudaFreeHost(ones_h) );
|
||||
|
||||
checkCuda( cudaMallocHost(&scores, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&scores_d, K *sizeof(float)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&clses, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&clses_d, K *sizeof(int)) );
|
||||
|
||||
checkCuda( cudaMalloc(&topk_inds_d, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&topk_ys_d, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&topk_xs_d, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&inttopk_ys_d, K *sizeof(int)) );
|
||||
checkCuda( cudaMalloc(&inttopk_xs_d, K *sizeof(int)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&xs, K * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&ys, K * sizeof(float)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&dep, K * dim_dep.c * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&rot, K * dim_rot.c * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&dim_, K * dim_dim.c * sizeof(float)) );
|
||||
checkCuda( cudaMallocHost(&wh, K * dim_wh.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&dep_d, K * dim_dep.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&rot_d, K * dim_rot.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&dim_d, K * dim_dim.c * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&wh_d, K * dim_wh.c * sizeof(float)) );
|
||||
|
||||
checkCuda( cudaMallocHost(&target_coords, 4 * K *sizeof(float)) );
|
||||
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
|
||||
checkCuda( cudaMalloc(&mean_d, 3 * sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&stddev_d, 3 * sizeof(float)) );
|
||||
float mean[3] = {0.485, 0.456, 0.406};
|
||||
float stddev[3] = {0.229, 0.224, 0.225};
|
||||
|
||||
checkCuda(cudaMemcpy(mean_d, mean, 3*sizeof(float), cudaMemcpyHostToDevice));
|
||||
checkCuda(cudaMemcpy(stddev_d, stddev, 3*sizeof(float), cudaMemcpyHostToDevice));
|
||||
#else
|
||||
checkCuda(cudaMallocHost(&input, sizeof(dnnType)*netRT->input_dim.tot() * nBatches));
|
||||
mean << 0.485, 0.456, 0.406;
|
||||
stddev << 0.229, 0.224, 0.225;
|
||||
#endif
|
||||
|
||||
for(int bi=0; bi<nBatches; bi++) {
|
||||
cv::Mat calibs_ = cv::Mat::zeros(cv::Size(4,3), CV_32F);
|
||||
if(inputCalibs.size() == 0 || inputCalibs[bi].empty()) {
|
||||
calibs_.at<float>(0,0) = 707.0493;
|
||||
calibs_.at<float>(0,2) = 604.0814;
|
||||
calibs_.at<float>(1,1) = 707.0493;
|
||||
calibs_.at<float>(1,2) = 180.5066;
|
||||
calibs_.at<float>(0,3) = 45.75831;
|
||||
calibs_.at<float>(1,3) = -0.3454157;
|
||||
calibs_.at<float>(2,2) = 1.0;
|
||||
calibs_.at<float>(2,3) = 0.004981016;
|
||||
}
|
||||
else {
|
||||
calibs_.at<float>(0,0) = inputCalibs[bi].at<float>(0,0);// * (1440.0/dim.w);// / 1440;
|
||||
calibs_.at<float>(0,2) = inputCalibs[bi].at<float>(0,2);// * (1440.0/dim.w);// / 1440;
|
||||
calibs_.at<float>(1,1) = inputCalibs[bi].at<float>(1,1);// * (1080.0/dim.h);//dim.h / 1080;
|
||||
calibs_.at<float>(1,2) = inputCalibs[bi].at<float>(1,2);// * (1080.0/dim.h);//dim.h / 1080;
|
||||
calibs_.at<float>(2,2) = 1.0;
|
||||
}
|
||||
// calibs_.at<float>(0,3) = 45.75831;
|
||||
// calibs_.at<float>(1,3) = -0.3454157;
|
||||
// calibs_.at<float>(2,2) = 1.0;
|
||||
// calibs_.at<float>(2,3) = 0.004981016;
|
||||
calibs.push_back(calibs_);
|
||||
}
|
||||
|
||||
r = cv::Mat(cv::Size(3,3), CV_32F);
|
||||
r.at<float>(0,1) = 0.0;
|
||||
r.at<float>(1,0) = 0.0;
|
||||
r.at<float>(1,1) = 1.0;
|
||||
r.at<float>(1,2) = 0.0;
|
||||
r.at<float>(2,1) = 0.0;
|
||||
|
||||
corners = cv::Mat(cv::Size(8,3), CV_32F);
|
||||
corners.at<float>(1,0) = 0.0;
|
||||
corners.at<float>(1,1) = 0.0;
|
||||
corners.at<float>(1,2) = 0.0;
|
||||
corners.at<float>(1,3) = 0.0;
|
||||
|
||||
pts3DHomo = cv::Mat(cv::Size(8,4), CV_32F);
|
||||
pts3DHomo.at<float>(3,0) = 1.0;
|
||||
pts3DHomo.at<float>(3,1) = 1.0;
|
||||
pts3DHomo.at<float>(3,2) = 1.0;
|
||||
pts3DHomo.at<float>(3,3) = 1.0;
|
||||
pts3DHomo.at<float>(3,4) = 1.0;
|
||||
pts3DHomo.at<float>(3,5) = 1.0;
|
||||
pts3DHomo.at<float>(3,6) = 1.0;
|
||||
pts3DHomo.at<float>(3,7) = 1.0;
|
||||
|
||||
checkCuda( cudaMalloc(&d_ptrs, dim.c * dim.h*dim.w * sizeof(float)) );
|
||||
|
||||
// Alloc array used in the kernel
|
||||
checkCuda( cudaMalloc(&srcOut, K *sizeof(float)) );
|
||||
checkCuda( cudaMalloc(&idsOut, K *sizeof(int)) );
|
||||
|
||||
dst2.at<float>(0,0)=width * 0.5;
|
||||
dst2.at<float>(0,1)=width * 0.5;
|
||||
dst2.at<float>(1,0)=width * 0.5;
|
||||
dst2.at<float>(1,1)=width * 0.5 + width * -0.5;
|
||||
|
||||
dst2.at<float>(2,0)=dst2.at<float>(1,0) + (-dst2.at<float>(0,1)+dst2.at<float>(1,1) );
|
||||
dst2.at<float>(2,1)=dst2.at<float>(1,1) + (dst2.at<float>(0,0)-dst2.at<float>(1,0) );
|
||||
|
||||
faceId.push_back({0,1,5,4});
|
||||
faceId.push_back({1,2,6, 5});
|
||||
faceId.push_back({2,3,7,6});
|
||||
faceId.push_back({3,0,4,7});
|
||||
// ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]);
|
||||
return true;
|
||||
}
|
||||
|
||||
void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi){
|
||||
cv::Size sz = originalSize[bi];
|
||||
float new_height = dim.h;//sz.height * scale;
|
||||
float new_width = dim.w;//sz.width * scale;
|
||||
if(sz.height != sz_old.height && sz.width != sz_old.width){
|
||||
|
||||
if(inputCalibs.size() == 0 || inputCalibs[bi].empty()) {
|
||||
calibs[bi].at<float>(0,2) = new_width / 2.0f;
|
||||
calibs[bi].at<float>(1,2) = new_height /2.0f;
|
||||
}
|
||||
else {
|
||||
calibs[bi].at<float>(0,0) = inputCalibs[bi].at<float>(0,0) * 2.0 * dim.w / sz.width;
|
||||
calibs[bi].at<float>(0,2) = inputCalibs[bi].at<float>(0,2) * dim.w / sz.width ;
|
||||
calibs[bi].at<float>(1,1) = inputCalibs[bi].at<float>(1,1) * 2.0 * dim.h / sz.height;
|
||||
calibs[bi].at<float>(1,2) = inputCalibs[bi].at<float>(1,2) * dim.h / sz.height;
|
||||
}
|
||||
float c[] = {new_width / 2.0f, new_height /2.0f};
|
||||
float s[] = {new_width, new_height};
|
||||
// ----------- get_affine_transform
|
||||
// rot_rad = pi * 0 / 100 --> 0
|
||||
|
||||
src.at<float>(0,0)=c[0];
|
||||
src.at<float>(0,1)=c[1];
|
||||
src.at<float>(1,0)=c[0];
|
||||
src.at<float>(1,1)=c[1] + s[0] * -0.5;
|
||||
dst.at<float>(0,0)=netRT->input_dim.w * 0.5;
|
||||
dst.at<float>(0,1)=netRT->input_dim.h * 0.5;
|
||||
dst.at<float>(1,0)=netRT->input_dim.w * 0.5;
|
||||
dst.at<float>(1,1)=netRT->input_dim.h * 0.5 + netRT->input_dim.w * -0.5;
|
||||
|
||||
src.at<float>(2,0)=src.at<float>(1,0) + (-src.at<float>(0,1)+src.at<float>(1,1) );
|
||||
src.at<float>(2,1)=src.at<float>(1,1) + (src.at<float>(0,0)-src.at<float>(1,0) );
|
||||
dst.at<float>(2,0)=dst.at<float>(1,0) + (-dst.at<float>(0,1)+dst.at<float>(1,1) );
|
||||
dst.at<float>(2,1)=dst.at<float>(1,1) + (dst.at<float>(0,0)-dst.at<float>(1,0) );
|
||||
|
||||
trans = cv::getAffineTransform( src, dst );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME gett affine trans: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
trans2 = cv::getAffineTransform( dst2, src );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME getAffineTrans 2: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
}
|
||||
sz_old = sz;
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
// std::cout<<"OPENCV CPMTROB\n";
|
||||
cv::cuda::GpuMat im_Orig;
|
||||
cv::cuda::GpuMat imageF1_d, imageF2_d;
|
||||
|
||||
im_Orig = cv::cuda::GpuMat(frame);
|
||||
cv::cuda::resize (im_Orig, imageF1_d, cv::Size(dim.w, dim.h));//cv::Size(new_width, new_height));
|
||||
// imageF1_d = im_Orig;
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
sz = imageF1_d.size();
|
||||
// std::cout<<"size: "<<sz.height<<" "<<sz.width<<" - "<<std::endl;
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME resize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
cv::cuda::warpAffine(imageF1_d, imageF2_d, trans, cv::Size(netRT->input_dim.w, netRT->input_dim.h), cv::INTER_LINEAR );
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
imageF2_d.convertTo(imageF1_d, CV_32FC3, 1/255.0);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME convert: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
dim2 = dim;
|
||||
cv::cuda::GpuMat bgr[3];
|
||||
cv::cuda::split(imageF1_d,bgr);//split source
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME split: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
for(int i=0; i<dim.c; i++)
|
||||
checkCuda( cudaMemcpy(d_ptrs + i*dim.h * dim.w, (float*)bgr[i].data, dim.h * dim.w * sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
|
||||
normalize(d_ptrs, dim.c, dim.h, dim.w, mean_d, stddev_d);
|
||||
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME normalize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
checkCuda(cudaMemcpy(input_d+ netRT->input_dim.tot()*bi, d_ptrs, dim2.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME Memcpy to input_d: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
#else
|
||||
// std::cout<<"NO OPENCV CPMTROB\n";
|
||||
cv::Mat imageF;
|
||||
resize(frame, imageF, cv::Size(dim.w, dim.h));//cv::Size(new_width, new_height));
|
||||
// imageF = frame;
|
||||
sz = imageF.size();
|
||||
// std::cout<<"size: "<<sz.height<<" "<<sz.width<<" - "<<std::endl;
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME resize: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
cv::Mat trans = cv::getAffineTransform( src, dst );
|
||||
cv::warpAffine(imageF, imageF, trans, cv::Size(netRT->input_dim.w, netRT->input_dim.h), cv::INTER_LINEAR );
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME warpAffine: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
|
||||
sz = imageF.size();
|
||||
// std::cout<<"size: "<<sz.height<<" "<<sz.width<<" - "<<std::endl;
|
||||
imageF.convertTo(imageF, CV_32FC3, 1/255.0);
|
||||
// end_t = std::chrono::steady_clock::now();
|
||||
// std::cout << " TIME convertto: " << std::chrono::duration_cast<std::chrono:: microseconds>(end_t - step_t).count() << " us" << std::endl;
|
||||
// step_t = end_t;
|
||||
dim2 = dim;
|
||||
//split channels
|
||||
cv::Mat bgr[3];
|
||||
cv::split(imageF,bgr);//split source
|
||||
for(int i=0; i<3; i++){
|
||||
bgr[i] = bgr[i] - mean[i];
|
||||
bgr[i] = bgr[i] / stddev[i];
|
||||
}
|
||||
|
||||
//write channels
|
||||
for(int i=0; i<dim2.c; i++) {
|
||||
int idx = i*imageF.rows*imageF.cols;
|
||||
int ch = dim2.c-3 +i;
|
||||
// std::cout<<"i: "<<i<<", idx: "<<idx<<", ch: "<<ch<<std::endl;
|
||||
memcpy((void*)&input[idx+ netRT->input_dim.tot()*bi], (void*)bgr[ch].data, imageF.rows*imageF.cols*sizeof(dnnType));
|
||||
}
|
||||
checkCuda(cudaMemcpyAsync(input_d+ netRT->input_dim.tot()*bi, input+ netRT->input_dim.tot()*bi, dim2.tot()*sizeof(dnnType), cudaMemcpyHostToDevice));
|
||||
#endif
|
||||
}
|
||||
|
||||
void CenternetDetection3D::postprocess(const int bi, const bool mAP) {
|
||||
dnnType *rt_out[7];
|
||||
rt_out[0] = (dnnType *)netRT->buffersRT[1]+ netRT->buffersDIM[1].tot()*bi;
|
||||
rt_out[1] = (dnnType *)netRT->buffersRT[2]+ netRT->buffersDIM[2].tot()*bi;
|
||||
rt_out[2] = (dnnType *)netRT->buffersRT[3]+ netRT->buffersDIM[3].tot()*bi;
|
||||
rt_out[3] = (dnnType *)netRT->buffersRT[4]+ netRT->buffersDIM[4].tot()*bi;
|
||||
rt_out[4] = (dnnType *)netRT->buffersRT[5]+ netRT->buffersDIM[5].tot()*bi;
|
||||
rt_out[5] = (dnnType *)netRT->buffersRT[6]+ netRT->buffersDIM[6].tot()*bi;
|
||||
rt_out[6] = (dnnType *)netRT->buffersRT[7]+ netRT->buffersDIM[7].tot()*bi;
|
||||
|
||||
// ------------------------------------ process --------------------------------------------
|
||||
activationSIGMOIDForward(rt_out[0], rt_out[0], dim_hm.tot());
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
// output['dep'] = 1. / (output['dep'].sigmoid() + 1e-6) - 1.
|
||||
activationSIGMOIDForward(rt_out[4], rt_out[4], dim_dep.tot());
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
transformDep(ones, ones + dim_dep.tot(), rt_out[4], rt_out[4] + dim_dep.tot());
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
subtractWithThreshold(rt_out[0], rt_out[0] + dim_hm.tot(), rt_out[1], rt_out[0], op);
|
||||
|
||||
// ----------- nms end
|
||||
// ----------- topk
|
||||
|
||||
if(K > dim_hm.h * dim_hm.w){
|
||||
printf ("Error topk (K is too large)\n");
|
||||
return;
|
||||
}
|
||||
|
||||
checkCuda( cudaMemcpy(ids_d, ids_, dim_hm.c * dim_hm.h * dim_hm.w*sizeof(int), cudaMemcpyHostToDevice) );
|
||||
|
||||
sort(rt_out[0],rt_out[0]+dim_hm.tot(),ids_d);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
topk(rt_out[0], ids_d, K, scores_d, topk_inds_d, topk_ys_d, topk_xs_d);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
checkCuda( cudaMemcpy(scores, scores_d, K *sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
topKxyclasses(topk_inds_d, topk_inds_d+K, K, width, dim_hm.w*dim_hm.h, clses_d, inttopk_xs_d, inttopk_ys_d);
|
||||
|
||||
checkCuda( cudaMemcpy(topk_xs_d, (float *)inttopk_xs_d, K*sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
checkCuda( cudaMemcpy(topk_ys_d, (float *)inttopk_ys_d, K*sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
|
||||
checkCuda( cudaMemcpy(clses, clses_d, K*sizeof(int), cudaMemcpyDeviceToHost) );
|
||||
|
||||
// ----------- topk end
|
||||
|
||||
topKxyAddOffset(topk_inds_d, K, dim_reg.h*dim_reg.w, inttopk_xs_d, inttopk_ys_d, topk_xs_d, topk_ys_d, rt_out[3], srcOut, idsOut);
|
||||
// checkCuda( cudaDeviceSynchronize() );
|
||||
|
||||
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_dep.c, dim_dep.h * dim_dep.w, rt_out[4], dep_d, idsOut);
|
||||
checkCuda( cudaMemcpy(dep, dep_d, K * dim_dep.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_rot.c, dim_rot.h * dim_rot.w, rt_out[5], rot_d, idsOut);
|
||||
checkCuda( cudaMemcpy(rot, rot_d, K * dim_rot.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_dim.c, dim_dim.h * dim_dim.w, rt_out[6], dim_d, idsOut);
|
||||
checkCuda( cudaMemcpy(dim_, dim_d, K * dim_dim.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
getRecordsFromTopKId(topk_inds_d, K, dim_wh.c, dim_wh.h * dim_wh.w, rt_out[2], wh_d, idsOut);
|
||||
checkCuda( cudaMemcpy(wh, wh_d, K * dim_wh.c * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
checkCuda( cudaMemcpy(xs, topk_xs_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
checkCuda( cudaMemcpy(ys, topk_ys_d, K * sizeof(float), cudaMemcpyDeviceToHost) );
|
||||
|
||||
// ---------------------------------- post-process -----------------------------------------
|
||||
|
||||
// ddd_post_process_2d
|
||||
cv::Mat new_pt1(cv::Size(1,2), CV_32F);
|
||||
cv::Mat new_pt2(cv::Size(1,2), CV_32F);
|
||||
|
||||
for(int i = 0; i<K; i++){
|
||||
new_pt1.at<float>(0,0)=static_cast<float>(trans2.at<double>(0,0))*xs[i] +
|
||||
static_cast<float>(trans2.at<double>(0,1))*ys[i] +
|
||||
static_cast<float>(trans2.at<double>(0,2))*1.0;
|
||||
new_pt1.at<float>(0,1)=static_cast<float>(trans2.at<double>(1,0))*xs[i] +
|
||||
static_cast<float>(trans2.at<double>(1,1))*ys[i] +
|
||||
static_cast<float>(trans2.at<double>(1,2))*1.0;
|
||||
|
||||
new_pt2.at<float>(0,0)=static_cast<float>(trans2.at<double>(0,0))*wh[i] +
|
||||
static_cast<float>(trans2.at<double>(0,1))*wh[K+i] +
|
||||
static_cast<float>(trans2.at<double>(0,2))*1.0;
|
||||
new_pt2.at<float>(0,1)=static_cast<float>(trans2.at<double>(1,0))*wh[i] +
|
||||
static_cast<float>(trans2.at<double>(1,1))*wh[K+i] +
|
||||
static_cast<float>(trans2.at<double>(1,2))*1.0;
|
||||
|
||||
target_coords[i*4] = new_pt1.at<float>(0,0);
|
||||
target_coords[i*4+1] = new_pt1.at<float>(0,1);
|
||||
target_coords[i*4+2] = new_pt2.at<float>(0,0);
|
||||
target_coords[i*4+3] = new_pt2.at<float>(0,1);
|
||||
}
|
||||
|
||||
float alpha;
|
||||
float x, y, z, rot_y;
|
||||
detected3D.clear();
|
||||
for(int i = 0; i<classes; i++){
|
||||
for(int j=0; j<K; j++){
|
||||
if(clses[j] == i){
|
||||
//get alpha
|
||||
if(rot[1*K + j] > rot[5*K + j])
|
||||
alpha = std::atan2(rot[2*K + j], rot[3*K + j]) -0.5 * M_PI;
|
||||
else
|
||||
alpha = std::atan2(rot[6*K + j], rot[7*K + j]) +0.5 * M_PI;
|
||||
|
||||
// unproject_2d_to_3d
|
||||
z = dep[j] - calibs[bi].at<float>(2,3);// z = depth - P[2, 3]
|
||||
x = (target_coords[j*4] * dep[j] - calibs[bi].at<float>(0,3) - calibs[bi].at<float>(0,2) * z) / calibs[bi].at<float>(0,0);
|
||||
y = (target_coords[j*4+1] * dep[j] - calibs[bi].at<float>(1,3) - calibs[bi].at<float>(1,2) * z) / calibs[bi].at<float>(1,1) + (dim_[j] / 2);
|
||||
// alpha2rot_y
|
||||
rot_y = (alpha + std::atan2(target_coords[j*4] - calibs[bi].at<float>(0,2), calibs[bi].at<float>(0,0)));
|
||||
if(rot_y>M_PI)
|
||||
rot_y -= 2*M_PI;
|
||||
if(rot_y<M_PI)
|
||||
rot_y += 2*M_PI;
|
||||
|
||||
if(scores[j] > confThreshold) {
|
||||
if(z>0) {
|
||||
// compute_box_3d
|
||||
r.at<float>(0,0) = std::cos(rot_y);
|
||||
r.at<float>(0,2) = std::sin(rot_y);
|
||||
r.at<float>(2,0) = -std::sin(rot_y);
|
||||
r.at<float>(2,2) = std::cos(rot_y);
|
||||
|
||||
corners.at<float>(0,0) = dim_[2*K+j]/2;
|
||||
corners.at<float>(0,1) = dim_[2*K+j]/2;
|
||||
corners.at<float>(0,2) = -dim_[2*K+j]/2;
|
||||
corners.at<float>(0,3) = -dim_[2*K+j]/2;
|
||||
corners.at<float>(0,4) = dim_[2*K+j]/2;
|
||||
corners.at<float>(0,5) = dim_[2*K+j]/2;
|
||||
corners.at<float>(0,6) = -dim_[2*K+j]/2;
|
||||
corners.at<float>(0,7) = -dim_[2*K+j]/2;
|
||||
|
||||
corners.at<float>(1,4) = -dim_[j];
|
||||
corners.at<float>(1,5) = -dim_[j];
|
||||
corners.at<float>(1,6) = -dim_[j];
|
||||
corners.at<float>(1,7) = -dim_[j];
|
||||
|
||||
corners.at<float>(2,0) = dim_[K+j]/2;
|
||||
corners.at<float>(2,1) = -dim_[K+j]/2;
|
||||
corners.at<float>(2,2) = -dim_[K+j]/2;
|
||||
corners.at<float>(2,3) = dim_[K+j]/2;
|
||||
corners.at<float>(2,4) = dim_[K+j]/2;
|
||||
corners.at<float>(2,5) = -dim_[K+j]/2;
|
||||
corners.at<float>(2,6) = -dim_[K+j]/2;
|
||||
corners.at<float>(2,7) = dim_[K+j]/2;
|
||||
cv::Mat aus = r * corners;
|
||||
|
||||
for(int k=0; k<8; k++) {
|
||||
aus.at<float>(0,k) += x;
|
||||
aus.at<float>(1,k) += y;
|
||||
aus.at<float>(2,k) += z;
|
||||
}
|
||||
// corners.copyTo(pts3DHomo(cv::Rect(0, 0, 8, 3)));
|
||||
for(int k1=0; k1<3; k1++) {
|
||||
for(int k2=0; k2<8; k2++)
|
||||
pts3DHomo.at<float>(k1,k2) = aus.at<float>(k1,k2);
|
||||
}
|
||||
aus.release();
|
||||
aus = calibs[bi] * pts3DHomo;
|
||||
|
||||
tk::dnn::box3D res;
|
||||
for(int k=0; k<8; k++) {
|
||||
res.corners.push_back(aus.at<float>(0,k) / aus.at<float>(2,k));
|
||||
res.corners.push_back(aus.at<float>(1,k) / aus.at<float>(2,k));
|
||||
}
|
||||
res.cl = i;
|
||||
res.prob = scores[j];
|
||||
//res.print();
|
||||
detected3D.push_back(res);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
batchDetected.push_back(detected3D);
|
||||
}
|
||||
|
||||
void CenternetDetection3D::draw(std::vector<cv::Mat>& frames) {
|
||||
tk::dnn::box3D b;
|
||||
int x0, w, x1, y0, h, y1;
|
||||
int objClass;
|
||||
std::string det_class;
|
||||
|
||||
int baseline = 0;
|
||||
float font_scale = 0.5;
|
||||
int thickness = 2;
|
||||
|
||||
for(int bi=0; bi<frames.size(); ++bi){
|
||||
float scale_x = float(originalSize[bi].width)/dim.w;
|
||||
float scale_y = float(originalSize[bi].height)/dim.h;
|
||||
resize(frames[bi], frames[bi], originalSize[bi]);
|
||||
// draw dets
|
||||
for(int i=0; i<batchDetected[bi].size(); i++) {
|
||||
b = batchDetected[bi][i];
|
||||
|
||||
for(int ind_f = 3; ind_f>=0; ind_f--) {
|
||||
for(int j=0; j<4; j++) {
|
||||
cv::line(frames[bi], cv::Point(b.corners.at(faceId.at(ind_f).at(j) * 2) * scale_x,
|
||||
b.corners.at(faceId.at(ind_f).at(j) * 2 + 1) * scale_y),
|
||||
cv::Point(b.corners.at(faceId.at(ind_f).at((j+1)%4) * 2) * scale_x,
|
||||
b.corners.at(faceId.at(ind_f).at((j+1)%4) * 2 + 1) * scale_y),
|
||||
colors[b.cl], 2);
|
||||
if(ind_f == 0) {
|
||||
cv::line(frames[bi], cv::Point(b.corners.at(faceId.at(ind_f).at(0) * 2) * scale_x,
|
||||
b.corners.at(faceId.at(ind_f).at(0) * 2 + 1)* scale_y),
|
||||
cv::Point(b.corners.at(faceId.at(ind_f).at(2) * 2) * scale_x,
|
||||
b.corners.at(faceId.at(ind_f).at(2) * 2 + 1) * scale_y), colors[b.cl], 2);
|
||||
cv::line(frames[bi], cv::Point(b.corners.at(faceId.at(ind_f).at(1) * 2)* scale_x,
|
||||
b.corners.at(faceId.at(ind_f).at(1) * 2 + 1)* scale_y),
|
||||
cv::Point(b.corners.at(faceId.at(ind_f).at(3) * 2)* scale_x,
|
||||
b.corners.at(faceId.at(ind_f).at(3) * 2 + 1)* scale_y), colors[b.cl], 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
// draw label
|
||||
cv::Size text_size = getTextSize(classesNames[b.cl], cv::FONT_HERSHEY_SIMPLEX, font_scale, thickness, &baseline);
|
||||
cv::rectangle(frames[bi], cv::Point(b.corners.at(faceId.at(0).at(0) * 2)* scale_x,
|
||||
b.corners.at(faceId.at(0).at(0) * 2 + 1)* scale_y),
|
||||
cv::Point((b.corners.at(faceId.at(0).at(0) * 2)* scale_x + text_size.width - 2),
|
||||
(b.corners.at(faceId.at(0).at(0) * 2 + 1)* scale_y - text_size.height - 2)), colors[b.cl], -1);
|
||||
cv::putText(frames[bi], classesNames[b.cl], cv::Point(b.corners.at(faceId.at(0).at(0) * 2)* scale_x,
|
||||
(b.corners.at(faceId.at(0).at(0) * 2 + 1)* scale_y - (baseline / 2))),
|
||||
cv::FONT_HERSHEY_SIMPLEX, font_scale, cv::Scalar(255, 255, 255), thickness);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}}
|
||||
|
||||
|
||||
+170
-79
@@ -4,83 +4,184 @@
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
Conv2d::Conv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
||||
void Conv2d::initCUDNN(bool back) {
|
||||
|
||||
cudnnTensorDescriptor_t srcTensor = srcTensorDesc;
|
||||
cudnnTensorDescriptor_t dstTensor = dstTensorDesc;
|
||||
|
||||
dataDim_t idim, odim;
|
||||
if(!back) {
|
||||
idim = input_dim;
|
||||
odim = output_dim;
|
||||
} else {
|
||||
idim = output_dim;
|
||||
odim = input_dim;
|
||||
}
|
||||
|
||||
checkCUDNN( cudnnCreateFilterDescriptor(&filterDesc) );
|
||||
checkCUDNN( cudnnCreateConvolutionDescriptor(&convDesc) );
|
||||
checkCUDNN( cudnnCreateTensorDescriptor(&biasTensorDesc) );
|
||||
|
||||
// input tensor dim
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(srcTensor,
|
||||
net->tensorFormat, net->dataType, idim.n, idim.c, idim.h, idim.w) );
|
||||
|
||||
checkCUDNN( cudnnSetFilter4dDescriptor(filterDesc,
|
||||
net->dataType, net->tensorFormat, odim.c, idim.c/groups,
|
||||
kernelH, kernelW) );
|
||||
|
||||
checkCUDNN( cudnnSetConvolution2dDescriptor(convDesc,
|
||||
paddingH, paddingW, // padding
|
||||
strideH, strideW, // stride
|
||||
1,1, // upscale
|
||||
CUDNN_CROSS_CORRELATION, CUDNN_DATA_FLOAT) );
|
||||
|
||||
checkCUDNN( cudnnSetConvolutionGroupCount(convDesc,
|
||||
groups) );
|
||||
|
||||
// check dimension of convolution output
|
||||
dataDim_t tmpdim;
|
||||
checkCUDNN( cudnnGetConvolution2dForwardOutputDim(
|
||||
convDesc, srcTensor, filterDesc,
|
||||
&tmpdim.n, &tmpdim.c, &tmpdim.h, &tmpdim.w) );
|
||||
|
||||
if(odim.n != tmpdim.n || odim.c != tmpdim.c || odim.h != tmpdim.h || odim.w != tmpdim.w) {
|
||||
std::cout<<"tkdim input: "; idim.print();
|
||||
std::cout<<"tkdim output: "; odim.print();
|
||||
std::cout<<"cudnndim: "; tmpdim.print();
|
||||
FatalError("Error conv dimension mismatch");
|
||||
}
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(dstTensor,
|
||||
net->tensorFormat, net->dataType, odim.n, odim.c, odim.h, odim.w) );
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(biasTensorDesc,
|
||||
net->tensorFormat, net->dataType,
|
||||
1, output_dim.c, 1, 1) );
|
||||
|
||||
// init workspace
|
||||
workSpace = NULL;
|
||||
ws_sizeInBytes = 0;
|
||||
int algo_count = 0;
|
||||
if(back) {
|
||||
checkCUDNN( cudnnGetConvolutionBackwardDataAlgorithm_v7(net->cudnnHandle,
|
||||
filterDesc, dstTensor, convDesc, srcTensor, 1, &algo_count, &bwAlgo) );
|
||||
checkCUDNN(cudnnGetConvolutionBackwardDataWorkspaceSize(net->cudnnHandle,
|
||||
filterDesc, dstTensor, convDesc, srcTensor,
|
||||
bwAlgo.algo, &ws_sizeInBytes));
|
||||
|
||||
|
||||
// invert tensors
|
||||
srcTensorDesc = dstTensor;
|
||||
dstTensorDesc = srcTensor;
|
||||
} else {
|
||||
|
||||
checkCUDNN( cudnnGetConvolutionForwardAlgorithm_v7(net->cudnnHandle,
|
||||
srcTensor, filterDesc, convDesc, dstTensor,
|
||||
1, &algo_count, &algo) );
|
||||
checkCUDNN(cudnnGetConvolutionForwardWorkspaceSize(net->cudnnHandle,
|
||||
srcTensor, filterDesc, convDesc, dstTensor,
|
||||
algo.algo, &ws_sizeInBytes));
|
||||
}
|
||||
|
||||
if(algo_count < 1)
|
||||
FatalError("Cannot retrieve convolutional algo");
|
||||
}
|
||||
|
||||
void Conv2d::inferCUDNN(dnnType* srcData, bool back) {
|
||||
|
||||
dnnType alpha = dnnType(1);
|
||||
dnnType beta = dnnType(0);
|
||||
if(back) {
|
||||
checkCUDNN(cudnnConvolutionBackwardData(net->cudnnHandle,
|
||||
&alpha, filterDesc, data_d,
|
||||
srcTensorDesc, srcData,
|
||||
convDesc, bwAlgo.algo, workSpace, ws_sizeInBytes,
|
||||
&beta, dstTensorDesc, dstData));
|
||||
} else {
|
||||
checkCUDNN(cudnnConvolutionForward(net->cudnnHandle,
|
||||
&alpha, srcTensorDesc, srcData, filterDesc,
|
||||
data_d, convDesc, algo.algo, workSpace, ws_sizeInBytes,
|
||||
&beta, dstTensorDesc, dstData));
|
||||
}
|
||||
|
||||
if(!batchnorm && !additional_bias) { //CHECK WITH IF CORRECT
|
||||
// bias
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(1);
|
||||
checkCUDNN( cudnnAddTensor(net->cudnnHandle,
|
||||
&alpha, biasTensorDesc, bias_d,
|
||||
&beta, dstTensorDesc, dstData) );
|
||||
} else {
|
||||
if(additional_bias)
|
||||
{
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(1);
|
||||
checkCUDNN( cudnnAddTensor(net->cudnnHandle,
|
||||
&alpha, biasTensorDesc, bias2_d,
|
||||
&beta, dstTensorDesc, dstData) );
|
||||
}
|
||||
if(batchnorm)
|
||||
{
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(0);
|
||||
checkCUDNN( cudnnBatchNormalizationForwardInference(net->cudnnHandle,
|
||||
CUDNN_BATCHNORM_SPATIAL, &alpha, &beta,
|
||||
dstTensorDesc, dstData, dstTensorDesc,
|
||||
dstData, biasTensorDesc, //same tensor descriptor as bias
|
||||
scales_d, bias_d, mean_d, variance_d,
|
||||
TKDNN_BN_MIN_EPSILON) );
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Conv2d::Conv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
||||
int strideH, int strideW, int paddingH, int paddingW,
|
||||
std::string fname_weights, bool batchnorm) :
|
||||
std::string fname_weights, bool batchnorm, bool deConv, int groups, bool additional_bias) :
|
||||
|
||||
LayerWgs(net, net->getOutputDim().c, out_ch, kernelH, kernelW, 1,
|
||||
fname_weights, batchnorm) {
|
||||
|
||||
fname_weights, batchnorm, additional_bias, deConv, groups) {
|
||||
this->kernelH = kernelH;
|
||||
this->kernelW = kernelW;
|
||||
this->strideH = strideH;
|
||||
this->strideW = strideW;
|
||||
this->paddingH = paddingH;
|
||||
this->paddingW = paddingW;
|
||||
this->deConv = deConv;
|
||||
this->groups = groups;
|
||||
this->additional_bias = additional_bias;
|
||||
|
||||
checkCUDNN( cudnnCreateFilterDescriptor(&filterDesc) );
|
||||
checkCUDNN( cudnnCreateConvolutionDescriptor(&convDesc) );
|
||||
checkCUDNN( cudnnCreateTensorDescriptor(&biasTensorDesc) );
|
||||
if(!deConv) {
|
||||
output_dim.n = input_dim.n;
|
||||
output_dim.c = out_ch;
|
||||
output_dim.h = (input_dim.h + 2 * paddingH - kernelH) / strideH + 1;
|
||||
output_dim.w = (input_dim.w + 2 * paddingW - kernelW) / strideW + 1;
|
||||
output_dim.l = 1;
|
||||
} else {
|
||||
output_dim.n = input_dim.n;
|
||||
output_dim.c = out_ch;
|
||||
output_dim.h = ((input_dim.h-1) * strideH) - 2*paddingH + kernelH;
|
||||
output_dim.w = ((input_dim.w-1) * strideW) - 2*paddingW + kernelW;
|
||||
output_dim.l = 1;
|
||||
}
|
||||
initCUDNN(deConv);
|
||||
|
||||
int n = input_dim.n;
|
||||
int c = input_dim.c;
|
||||
int h = input_dim.h;
|
||||
int w = input_dim.w;
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(srcTensorDesc,
|
||||
net->tensorFormat, net->dataType, n, c, h, w) );
|
||||
|
||||
checkCUDNN( cudnnSetFilter4dDescriptor(filterDesc,
|
||||
net->dataType, net->tensorFormat, out_ch, input_dim.c,
|
||||
kernelH, kernelW) );
|
||||
|
||||
checkCUDNN( cudnnSetConvolution2dDescriptor(convDesc,
|
||||
paddingH, paddingW, // padding
|
||||
strideH, strideW, // stride
|
||||
1,1, // upscale
|
||||
CUDNN_CROSS_CORRELATION, CUDNN_DATA_FLOAT) );
|
||||
|
||||
// find dimension of convolution output
|
||||
checkCUDNN( cudnnGetConvolution2dForwardOutputDim(
|
||||
convDesc, srcTensorDesc, filterDesc,
|
||||
&n, &c, &h, &w) );
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(dstTensorDesc,
|
||||
net->tensorFormat, net->dataType, n, c, h, w) );
|
||||
|
||||
checkCUDNN( cudnnGetConvolutionForwardAlgorithm(net->cudnnHandle,
|
||||
srcTensorDesc, filterDesc, convDesc, dstTensorDesc,
|
||||
CUDNN_CONVOLUTION_FWD_PREFER_FASTEST, 0, &algo) );
|
||||
|
||||
workSpace = NULL;
|
||||
ws_sizeInBytes = 0;
|
||||
|
||||
checkCUDNN( cudnnGetConvolutionForwardWorkspaceSize(net->cudnnHandle,
|
||||
srcTensorDesc, filterDesc, convDesc, dstTensorDesc,
|
||||
algo, &ws_sizeInBytes) );
|
||||
if(this->groups != 1)
|
||||
MACC = kernelH*kernelW*output_dim.c*output_dim.w*output_dim.h;
|
||||
else
|
||||
MACC = input_dim.c*kernelH*kernelW*output_dim.c*output_dim.w*output_dim.h;
|
||||
|
||||
// allocate warkspace
|
||||
if (ws_sizeInBytes!=0) {
|
||||
checkCuda( cudaMalloc(&workSpace, ws_sizeInBytes) );
|
||||
}
|
||||
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(biasTensorDesc,
|
||||
net->tensorFormat, net->dataType,
|
||||
1, out_ch, 1, 1) );
|
||||
|
||||
|
||||
output_dim.n = n;
|
||||
output_dim.c = c;
|
||||
output_dim.h = h;
|
||||
output_dim.w = w;
|
||||
output_dim.l = 1;
|
||||
|
||||
//allocate data for infer result
|
||||
checkCuda( cudaMalloc(&dstData, output_dim.tot()*sizeof(dnnType)) );
|
||||
}
|
||||
|
||||
Conv2d::~Conv2d() {
|
||||
|
||||
|
||||
checkCUDNN( cudnnDestroyFilterDescriptor(filterDesc) );
|
||||
checkCUDNN( cudnnDestroyConvolutionDescriptor(convDesc) );
|
||||
checkCUDNN( cudnnDestroyTensorDescriptor(biasTensorDesc) );
|
||||
@@ -93,35 +194,25 @@ Conv2d::~Conv2d() {
|
||||
|
||||
dnnType* Conv2d::infer(dataDim_t &dim, dnnType* srcData) {
|
||||
|
||||
if(deConv) {
|
||||
FatalError("you must use DeConv class for Deconvolutional layers");
|
||||
}
|
||||
|
||||
// convolution
|
||||
dnnType alpha = dnnType(1);
|
||||
dnnType beta = dnnType(0);
|
||||
checkCUDNN( cudnnConvolutionForward(net->cudnnHandle,
|
||||
&alpha, srcTensorDesc, srcData, filterDesc,
|
||||
data_d, convDesc, algo, workSpace, ws_sizeInBytes,
|
||||
&beta, dstTensorDesc, dstData) );
|
||||
inferCUDNN(srcData, false);
|
||||
|
||||
if(!batchnorm) {
|
||||
// bias
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(1);
|
||||
checkCUDNN( cudnnAddTensor(net->cudnnHandle,
|
||||
&alpha, biasTensorDesc, bias_d,
|
||||
&beta, dstTensorDesc, dstData) );
|
||||
} else {
|
||||
float one = 1;
|
||||
float zero = 0;
|
||||
cudnnBatchNormalizationForwardInference(net->cudnnHandle,
|
||||
CUDNN_BATCHNORM_SPATIAL, &one, &zero,
|
||||
dstTensorDesc, dstData, dstTensorDesc,
|
||||
dstData, biasTensorDesc, //same tensor descriptor as bias
|
||||
scales_d, bias_d, mean_d, variance_d,
|
||||
TKDNN_BN_MIN_EPSILON);
|
||||
}
|
||||
//update data dimensions
|
||||
dim = output_dim;
|
||||
return dstData;
|
||||
}
|
||||
|
||||
dnnType* DeConv2d::infer(dataDim_t &dim, dnnType* srcData) {
|
||||
|
||||
// convolution
|
||||
inferCUDNN(srcData, true);
|
||||
|
||||
//update data dimensions
|
||||
dim = output_dim;
|
||||
return dstData;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,410 @@
|
||||
#include "tkDNN/DarknetParser.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
std::string darknetParseType(const std::string& line){
|
||||
size_t start = line.find("[");
|
||||
size_t end = line.find("]");
|
||||
if( start == std::string::npos || end == std::string::npos)
|
||||
return "";
|
||||
start++;
|
||||
std::string type = line.substr(start, end-start);
|
||||
return type;
|
||||
}
|
||||
|
||||
bool divideNameAndValue(const std::string& line, std::string&name, std::string& value){
|
||||
size_t sep = line.find("=");
|
||||
if(sep == std::string::npos)
|
||||
return false;
|
||||
|
||||
name = line.substr(0, sep);
|
||||
value = line.substr(sep+1, line.size() - (sep+1));
|
||||
return true;
|
||||
}
|
||||
|
||||
std::vector<int> fromStringToIntVec(const std::string& line, const char delimiter){
|
||||
std::stringstream linestream(line);
|
||||
std::string value;
|
||||
std::vector<int> values;
|
||||
|
||||
while(getline(linestream,value,delimiter))
|
||||
values.push_back(std::stoi(value));
|
||||
return values;
|
||||
}
|
||||
|
||||
std::vector<float> fromStringToFloatVec(const std::string& line, const char delimiter){
|
||||
std::stringstream linestream(line);
|
||||
std::string value;
|
||||
std::vector<float> values;
|
||||
|
||||
while(getline(linestream,value,delimiter))
|
||||
values.push_back(std::stof(value));
|
||||
return values;
|
||||
}
|
||||
|
||||
bool darknetParseFields(const std::string& line, darknetFields_t& fields){
|
||||
|
||||
std::string name,value;
|
||||
if(!divideNameAndValue(line, name, value))
|
||||
return false;
|
||||
|
||||
if(name.find("new_coords") != std::string::npos)
|
||||
fields.new_coords = std::stoi(value);
|
||||
else if(name.find("width") != std::string::npos)
|
||||
fields.width = std::stoi(value);
|
||||
else if(name.find("height") != std::string::npos)
|
||||
fields.height = std::stoi(value);
|
||||
else if(name.find("channels") != std::string::npos)
|
||||
fields.channels = std::stoi(value);
|
||||
else if(name.find("batch_normalize") != std::string::npos)
|
||||
fields.batch_normalize = std::stoi(value);
|
||||
else if(name.find("filters") != std::string::npos)
|
||||
fields.filters = std::stoi(value);
|
||||
else if(name.find("activation") != std::string::npos)
|
||||
fields.activation = value;
|
||||
else if(name.find("size") != std::string::npos){
|
||||
fields.size_x = std::stoi(value);
|
||||
fields.size_y = std::stoi(value);
|
||||
}
|
||||
else if(name.find("size_x") != std::string::npos)
|
||||
fields.size_x = std::stoi(value);
|
||||
else if(name.find("size_y") != std::string::npos)
|
||||
fields.size_y = std::stoi(value);
|
||||
else if(name.find("stride") != std::string::npos){
|
||||
fields.stride_x = std::stoi(value);
|
||||
fields.stride_y = std::stoi(value);
|
||||
}
|
||||
else if(name.find("stride_x") != std::string::npos)
|
||||
fields.stride_x = std::stoi(value);
|
||||
else if(name.find("stride_y") != std::string::npos)
|
||||
fields.stride_y = std::stoi(value);
|
||||
else if(name.find("pad") != std::string::npos)
|
||||
fields.pad = std::stoi(value);
|
||||
else if(name.find("classes") != std::string::npos)
|
||||
fields.classes = std::stoi(value);
|
||||
else if(name.find("num") != std::string::npos)
|
||||
fields.num = std::stoi(value);
|
||||
else if(name.find("coords") != std::string::npos)
|
||||
fields.coords = std::stoi(value);
|
||||
else if(name.find("groups") != std::string::npos)
|
||||
fields.groups = std::stoi(value);
|
||||
else if(name.find("group_id") != std::string::npos)
|
||||
fields.group_id = std::stoi(value);
|
||||
else if(name.find("scale_x_y") != std::string::npos)
|
||||
fields.scale_xy = std::stof(value);
|
||||
else if(name.find("beta_nms") != std::string::npos)
|
||||
fields.nms_thresh = std::stof(value);
|
||||
else if(name.find("nms_kind") != std::string::npos){
|
||||
if(value == "greedynms") fields.nms_kind = 0;
|
||||
else if(value == "diounms") fields.nms_kind = 1;
|
||||
else std::cout<<"Not supported nms_kind "<<value<<", setting to greedynms"<<std::endl;
|
||||
}
|
||||
else if(name.find("from") != std::string::npos)
|
||||
fields.layers.push_back(std::stof(value));
|
||||
else if(name.find("mask") != std::string::npos){
|
||||
auto vec = fromStringToIntVec(value, ',');
|
||||
fields.n_mask = vec.size();
|
||||
}
|
||||
else if(name.find("layers") != std::string::npos)
|
||||
fields.layers = fromStringToIntVec(value, ',');
|
||||
|
||||
else
|
||||
std::cout<<"Not supported field: "<<line<<std::endl;
|
||||
return true;
|
||||
}
|
||||
|
||||
tk::dnn::Network *darknetAddNet(darknetFields_t &fields) {
|
||||
//std::cout<<"Add Net: "<<fields.type<<"\n";
|
||||
dataDim_t dim(1, fields.channels, fields.height, fields.width);
|
||||
return new tk::dnn::Network(dim);
|
||||
}
|
||||
|
||||
|
||||
void darknetAddLayer(tk::dnn::Network *net, darknetFields_t &f, std::string wgs_path, std::vector<tk::dnn::Layer*> &netLayers, const std::vector<std::string>& names) {
|
||||
if(net == nullptr)
|
||||
FatalError("Cant add a layer without a Net\n");
|
||||
|
||||
// padding compute
|
||||
if(f.pad == 1) {
|
||||
f.padding_x = f.padding_y = f.size_x /2;
|
||||
}
|
||||
//std::cout<<"Add layer: "<<f.type<<"\n";
|
||||
if(f.type == "convolutional") {
|
||||
std::string wgs = wgs_path + "/c" + std::to_string(netLayers.size()) + ".bin";
|
||||
//printf("%d (%d,%d) (%d,%d) (%d,%d) %s %d %d\n", f.filters, f.size_x, f.size_y, f.stride_x, f.stride_y, f.padding_x, f.padding_y, wgs.c_str(), f.batch_normalize, f.groups);
|
||||
tk::dnn::Conv2d *l= new tk::dnn::Conv2d(net, f.filters, f.size_x, f.size_y, f.stride_x,
|
||||
f.stride_y, f.padding_x, f.padding_y, wgs, f.batch_normalize, false, f.groups);
|
||||
netLayers.push_back(l);
|
||||
} else if(f.type == "maxpool") {
|
||||
if(f.stride_x == 1 && f.stride_y == 1)
|
||||
netLayers.push_back(new tk::dnn::Pooling(net, f.size_x, f.size_y, f.stride_x, f.stride_y,
|
||||
f.padding_x, f.padding_y, tk::dnn::POOLING_MAX_FIXEDSIZE));
|
||||
else
|
||||
netLayers.push_back(new tk::dnn::Pooling(net, f.size_x, f.size_y, f.stride_x, f.stride_y,
|
||||
f.padding_x, f.padding_y, tk::dnn::POOLING_MAX));
|
||||
|
||||
} else if(f.type == "avgpool") {
|
||||
netLayers.push_back(new tk::dnn::Pooling(net, f.size_x, f.size_y, f.stride_x, f.stride_y,
|
||||
f.padding_x, f.padding_y, tk::dnn::POOLING_AVERAGE));
|
||||
|
||||
} else if(f.type == "shortcut") {
|
||||
if(f.layers.size() != 1) FatalError("no layers to shortcut\n");
|
||||
int layerIdx = f.layers[0];
|
||||
if(layerIdx < 0)
|
||||
layerIdx = netLayers.size() + layerIdx;
|
||||
if(layerIdx < 0 || layerIdx >= netLayers.size()) FatalError("impossible to shortcut\n");
|
||||
//std::cout<<"shortcut to "<<layerIdx<<" "<<netLayers[layerIdx]->getLayerName()<<"\n";
|
||||
netLayers.push_back(new tk::dnn::Shortcut(net, netLayers[layerIdx]));
|
||||
|
||||
} else if(f.type == "upsample") {
|
||||
netLayers.push_back(new tk::dnn::Upsample(net, f.stride_x));
|
||||
|
||||
} else if(f.type == "route") {
|
||||
if(f.layers.size() == 0) FatalError("no layers to Route\n");
|
||||
std::vector<tk::dnn::Layer*> layers;
|
||||
for(int i=0; i<f.layers.size(); i++) {
|
||||
int layerIdx = f.layers[i];
|
||||
if(layerIdx < 0)
|
||||
layerIdx = netLayers.size() + layerIdx;
|
||||
if(layerIdx < 0 || layerIdx >= netLayers.size()) FatalError("impossible to route\n");
|
||||
//std::cout<<"Route to "<<layerIdx<<" "<<netLayers[layerIdx]->getLayerName()<<"\n";
|
||||
layers.push_back(netLayers[layerIdx]);
|
||||
}
|
||||
netLayers.push_back(new tk::dnn::Route(net, layers.data(), layers.size(), f.groups, f.group_id));
|
||||
|
||||
} else if(f.type == "reorg") {
|
||||
netLayers.push_back(new tk::dnn::Reorg(net, f.stride_x));
|
||||
|
||||
} else if(f.type == "region") {
|
||||
netLayers.push_back(new tk::dnn::Region(net, f.classes, f.coords, f.num));
|
||||
|
||||
} else if(f.type == "yolo") {
|
||||
std::string wgs = wgs_path + "/g" + std::to_string(netLayers.size()) + ".bin";
|
||||
//printf("%d %d %s %d %f\n", f.classes, f.num/f.n_mask, wgs.c_str(), f.n_mask, f.scale_xy);
|
||||
tk::dnn::Yolo *l = new tk::dnn::Yolo(net, f.classes, f.num/f.n_mask, wgs, f.n_mask, f.scale_xy, f.nms_thresh, (tk::dnn::Yolo::nmsKind_t) f.nms_kind, f.new_coords);
|
||||
if(names.size() != f.classes)
|
||||
FatalError("Mismatch between number of classes and names");
|
||||
l->classesNames = names;
|
||||
netLayers.push_back(l);
|
||||
|
||||
} else{
|
||||
FatalError("layer not supported: " + f.type);
|
||||
}
|
||||
|
||||
// add activation
|
||||
if(netLayers.size() > 0 && f.activation != "linear") {
|
||||
tkdnnActivationMode_t act;
|
||||
if(f.activation == "relu") act = tkdnnActivationMode_t(CUDNN_ACTIVATION_RELU);
|
||||
else if(f.activation == "leaky") act = tk::dnn::ACTIVATION_LEAKY;
|
||||
else if(f.activation == "mish") act = tk::dnn::ACTIVATION_MISH;
|
||||
else if(f.activation == "logistic") act = tk::dnn::ACTIVATION_LOGISTIC;
|
||||
else { FatalError("activation not supported: " + f.activation); }
|
||||
netLayers[netLayers.size()-1] = new tk::dnn::Activation(net, act);
|
||||
};
|
||||
}
|
||||
|
||||
std::vector<std::string> darknetReadNames(const std::string& names_file){
|
||||
std::ifstream if_names(names_file);
|
||||
if(!if_names.is_open())
|
||||
FatalError("cloud not open names file: " + names_file);
|
||||
|
||||
std::vector<std::string> names;
|
||||
std::string line;
|
||||
while(std::getline(if_names, line))
|
||||
if(line != "")
|
||||
names.push_back(line);
|
||||
|
||||
if_names.close();
|
||||
return names;
|
||||
}
|
||||
|
||||
tk::dnn::Network* darknetParser(const std::string& cfg_file, const std::string& wgs_path, const std::string& names_file) {
|
||||
|
||||
tk::dnn::Network *net = nullptr;
|
||||
|
||||
// layers without activations to retrieve correct id number
|
||||
std::vector<tk::dnn::Layer*> netLayers;
|
||||
|
||||
std::ifstream if_cfg(cfg_file);
|
||||
if(!if_cfg.is_open())
|
||||
FatalError("cloud not open cfg file: " + cfg_file);
|
||||
|
||||
std::vector<std::string> names = darknetReadNames(names_file);
|
||||
|
||||
darknetFields_t fields; // will be filled with layers fields
|
||||
std::string line;
|
||||
while(std::getline(if_cfg, line)) {
|
||||
// remove comments
|
||||
std::size_t found = line.find("#");
|
||||
if ( found != std::string::npos ) {
|
||||
line = line.substr(0, found);
|
||||
}
|
||||
|
||||
// skip empty lines
|
||||
if(line.size() == 0)
|
||||
continue;
|
||||
|
||||
std::string type = darknetParseType(line);
|
||||
if(type.size() > 0) {
|
||||
// end of filled type
|
||||
if(fields.type != "") {
|
||||
if(fields.type == "net")
|
||||
net = darknetAddNet(fields);
|
||||
else
|
||||
darknetAddLayer(net, fields, wgs_path, netLayers, names);
|
||||
}
|
||||
|
||||
// new type
|
||||
//std::cout<<"type: "<<type<<"\n";
|
||||
fields = darknetFields_t(); // reset to default
|
||||
fields.type = type;
|
||||
continue;
|
||||
}
|
||||
|
||||
if(darknetParseFields(line, fields)) {
|
||||
// already parsed do nothing
|
||||
} else {
|
||||
FatalError("could not parse line: " + line);
|
||||
}
|
||||
}
|
||||
|
||||
// end of filled type
|
||||
if(fields.type != "") {
|
||||
darknetAddLayer(net, fields, wgs_path, netLayers, names);
|
||||
}
|
||||
|
||||
if(net == nullptr) {
|
||||
FatalError("net not found\n");
|
||||
}
|
||||
return net;
|
||||
}
|
||||
std::vector<int> noYolosLine(const std::string &cfg_file){
|
||||
std::ifstream if_cfg(cfg_file);
|
||||
if(!if_cfg.is_open())
|
||||
FatalError("cloud not open cfg file: " + cfg_file);
|
||||
std::string line;
|
||||
std::vector<int> lineNo;
|
||||
int count = 0;
|
||||
while(std::getline(if_cfg,line)){
|
||||
std::size_t found = line.find("#");
|
||||
if ( found != std::string::npos ) {
|
||||
line = line.substr(0, found);
|
||||
}
|
||||
// skip empty lines
|
||||
if(line.empty())
|
||||
continue;
|
||||
if(line == "[yolo]"){
|
||||
lineNo.push_back(count);
|
||||
|
||||
|
||||
}
|
||||
count++;
|
||||
}
|
||||
return lineNo;
|
||||
}
|
||||
void loadYoloInfo(const std::string &cfg_file,int lineNo,std::vector<float> &mask,std::vector<float> &anchors,int &num,int &classes,float &nms_thresh,int &nms_kind,int &coords){
|
||||
std::vector<float> maskTemp,anchorsTemp;
|
||||
int classesTemp,numTemp,nmsKindTemp;
|
||||
int new_coordsTemp=0;
|
||||
float nmsThreshTemp=0.45;
|
||||
|
||||
std::ifstream if_cfg(cfg_file);
|
||||
if(!if_cfg.is_open())
|
||||
FatalError("cloud not open cfg file: " + cfg_file);
|
||||
std::string line;
|
||||
int count = 0;
|
||||
while(std::getline(if_cfg,line)){
|
||||
std::string name,value;
|
||||
std::size_t found = line.find("#");
|
||||
if ( found != std::string::npos ) {
|
||||
line = line.substr(0, found);
|
||||
}
|
||||
// skip empty lines
|
||||
if(line.empty())
|
||||
continue;
|
||||
if(count > lineNo && count <=lineNo+30){
|
||||
divideNameAndValue(line,name,value);
|
||||
if(name == "mask "){
|
||||
maskTemp = fromStringToFloatVec(value,',');
|
||||
}
|
||||
if(name == "anchors "){
|
||||
anchorsTemp = fromStringToFloatVec(value,',');
|
||||
}
|
||||
if(name == "classes"){
|
||||
classesTemp = std::stoi(value);
|
||||
}
|
||||
if(name == "num"){
|
||||
numTemp = std::stoi(value);
|
||||
}
|
||||
if(name == "nms_kind"){
|
||||
if(value == "greedynms"){
|
||||
nmsKindTemp = 0;
|
||||
}else if(value == "diounms"){
|
||||
nmsKindTemp=1;
|
||||
}
|
||||
else{
|
||||
std::cout<<"NMS NOT SUPPORTED DEFAULTING TO GREEDYNMS"<<std::endl;
|
||||
nmsKindTemp=0;
|
||||
}
|
||||
}
|
||||
if(name == "new_coords"){
|
||||
new_coordsTemp = std::stoi(value);
|
||||
}
|
||||
if(name == "beta_nms"){
|
||||
nmsThreshTemp = std::stof(value);
|
||||
}
|
||||
}
|
||||
count++;
|
||||
}
|
||||
mask = maskTemp;
|
||||
anchors = anchorsTemp;
|
||||
num = numTemp;
|
||||
nms_kind = nmsKindTemp;
|
||||
nms_thresh = nmsThreshTemp;
|
||||
coords = new_coordsTemp;
|
||||
classes = classesTemp;
|
||||
|
||||
}
|
||||
void loadYoloInitInfo(int &channels,int &width,int &height,const std::string &cfg_file){
|
||||
std::ifstream if_cfg(cfg_file);
|
||||
if(!if_cfg.is_open())
|
||||
FatalError("cloud not open cfg file: " + cfg_file);
|
||||
std::string line;
|
||||
int count = 0;
|
||||
|
||||
while(std::getline(if_cfg,line)){
|
||||
if(count == 7){
|
||||
std::string name,value;
|
||||
divideNameAndValue(line,name,value);
|
||||
if(name == "width"){
|
||||
width = std::stoi(value);
|
||||
}
|
||||
}
|
||||
|
||||
if(count == 8){
|
||||
std::string name,value;
|
||||
divideNameAndValue(line,name,value);
|
||||
if(name == "height"){
|
||||
height = std::stoi(value);
|
||||
}
|
||||
}
|
||||
|
||||
if(count == 9){
|
||||
std::string name,value;
|
||||
divideNameAndValue(line,name,value);
|
||||
if(name == "channels"){
|
||||
channels = std::stoi(value);
|
||||
break;
|
||||
}
|
||||
else{
|
||||
std::cerr<<"EXITING PROGRAM DUE TO INSUFFICENT DATA FROM CFG"<<std::endl;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
count++;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}}
|
||||
@@ -0,0 +1,148 @@
|
||||
#include <iostream>
|
||||
|
||||
#include "Layer.h"
|
||||
#include "kernels.h"
|
||||
#include <math.h>
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
void DeformConv2d::initCUDNN() {
|
||||
|
||||
stat = cublasCreate(&handle);
|
||||
if (stat != CUBLAS_STATUS_SUCCESS)
|
||||
FatalError("CUBLAS initialization failed\n");
|
||||
|
||||
checkCUDNN( cudnnCreateTensorDescriptor(&biasTensorDesc) );
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(biasTensorDesc,
|
||||
net->tensorFormat, net->dataType,
|
||||
1, output_dim.c, 1, 1) );
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(dstTensorDesc,
|
||||
net->tensorFormat, net->dataType, output_dim.n, output_dim.c, output_dim.h, output_dim.w));
|
||||
|
||||
const int height_ones = (preconv->input_dim.h + 2 * this->paddingH - (1 * (this->kernelH - 1) + 1)) / this->strideH + 1;
|
||||
const int width_ones = (preconv->input_dim.w + 2 * this->paddingW - (1 * (this->kernelW - 1) + 1)) / this->strideW + 1;
|
||||
const int dim_ones = preconv->input_dim.c * this->kernelH * this->kernelW * 1 * height_ones * width_ones;
|
||||
|
||||
int dst_dim = preconv->output_dim.tot();
|
||||
if( dst_dim % 3 != 0 )
|
||||
FatalError("DeformConv2d: the Conv2d output is not divisible by three");
|
||||
chunk_dim = dst_dim/3;
|
||||
checkCuda( cudaMalloc(&offset, 2*chunk_dim*sizeof(dnnType)));
|
||||
checkCuda( cudaMalloc(&mask, chunk_dim*sizeof(dnnType)));
|
||||
|
||||
// kernel ones
|
||||
checkCuda( cudaMalloc(&ones_d1, (height_ones*width_ones)*sizeof(dnnType)) );
|
||||
dnnType *ones_h1;
|
||||
checkCuda( cudaMallocHost(&ones_h1, (height_ones*width_ones)*sizeof(dnnType)) );
|
||||
for(int i=0; i<height_ones*width_ones; i++)
|
||||
ones_h1[i]=1.0f;
|
||||
checkCuda( cudaMemcpy(ones_d1, ones_h1, (height_ones*width_ones)*sizeof(dnnType), cudaMemcpyHostToDevice) );
|
||||
checkCuda( cudaFreeHost(ones_h1) );
|
||||
checkCuda( cudaMalloc(&ones_d2, dim_ones*sizeof(dnnType)) );
|
||||
dnnType *ones_h2;
|
||||
checkCuda( cudaMallocHost(&ones_h2, dim_ones*sizeof(dnnType)) );
|
||||
for(int i=0; i<dim_ones; i++)
|
||||
ones_h2[i]=1.0f;
|
||||
checkCuda( cudaMemcpy(ones_d2, ones_h2, (dim_ones)*sizeof(dnnType), cudaMemcpyHostToDevice) );
|
||||
checkCuda( cudaFreeHost(ones_h2) );
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
}
|
||||
|
||||
DeformConv2d::DeformConv2d( Network *net, int out_ch, int deformable_group, int kernelH, int kernelW,
|
||||
int strideH, int strideW, int paddingH, int paddingW,
|
||||
std::string d_fname_weights, std::string fname_weights, bool batchnorm) :
|
||||
|
||||
LayerWgs(net, net->getOutputDim().c, out_ch, kernelH, kernelW, 1,
|
||||
d_fname_weights, batchnorm, true) {
|
||||
this->out_ch = out_ch;
|
||||
this->deformableGroup = deformable_group;
|
||||
this->kernelH = kernelH;
|
||||
this->kernelW = kernelW;
|
||||
this->strideH = strideH;
|
||||
this->strideW = strideW;
|
||||
this->paddingH = paddingH;
|
||||
this->paddingW = paddingW;
|
||||
|
||||
preconv = new tk::dnn::Conv2d(net, deformable_group * 3 * kernelH * kernelW, kernelH, kernelW,
|
||||
strideH, strideW, paddingH, paddingW, fname_weights, false);
|
||||
net->num_layers--;
|
||||
|
||||
output_dim = preconv->output_dim;
|
||||
|
||||
output_dim.c = out_ch;
|
||||
initCUDNN();
|
||||
|
||||
if(this->deformableGroup != 1)
|
||||
MACC = kernelH*kernelW*output_dim.c*output_dim.w*output_dim.h;
|
||||
else
|
||||
MACC = input_dim.c*kernelH*kernelW*output_dim.c*output_dim.w*output_dim.h;
|
||||
|
||||
//allocate data for infer result
|
||||
checkCuda( cudaMalloc(&dstData, output_dim.tot()*sizeof(dnnType)) );
|
||||
}
|
||||
|
||||
DeformConv2d::~DeformConv2d() {
|
||||
checkCUDNN( cudnnDestroyTensorDescriptor(biasTensorDesc) );
|
||||
checkCuda( cudaFree(dstData) );
|
||||
checkCuda( cudaFree(ones_d1) );
|
||||
checkCuda( cudaFree(ones_d2) );
|
||||
checkCuda( cudaFree(offset) );
|
||||
checkCuda( cudaFree(mask) );
|
||||
checkCuda( cudaFree(output_conv) );
|
||||
cublasDestroy(handle);
|
||||
}
|
||||
|
||||
dnnType* DeformConv2d::infer(dataDim_t &dim, dnnType* srcData) {
|
||||
|
||||
// conv2d
|
||||
output_conv = preconv->infer(dim, srcData);
|
||||
// split conv2d outputs into offset and mask
|
||||
checkCuda(cudaMemcpy(offset, output_conv, 2*chunk_dim*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
checkCuda(cudaMemcpy(mask, output_conv + 2*chunk_dim, chunk_dim*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
// kernel sigmoid
|
||||
activationSIGMOIDForward(mask, mask, chunk_dim);
|
||||
|
||||
// deformable convolution
|
||||
dcnV2CudaForward(stat, handle,
|
||||
srcData, this->data_d,
|
||||
this->bias2_d, ones_d1,
|
||||
offset, mask,
|
||||
dstData, ones_d2,
|
||||
this->kernelH, this->kernelW,
|
||||
this->strideH, this->strideW,
|
||||
this->paddingH, this->paddingW,
|
||||
1, 1,
|
||||
this->deformableGroup, 0, //batch_id for cudnn is set to 0 (no batch)
|
||||
preconv->input_dim.n, preconv->input_dim.c, preconv->input_dim.h, preconv->input_dim.w,
|
||||
this->output_dim.n, this->output_dim.c, this->output_dim.h, this->output_dim.w,
|
||||
chunk_dim);
|
||||
|
||||
dnnType alpha = dnnType(1);
|
||||
dnnType beta = dnnType(0);
|
||||
if(!batchnorm) {
|
||||
// bias
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(1);
|
||||
checkCUDNN( cudnnAddTensor(net->cudnnHandle,
|
||||
&alpha, biasTensorDesc, bias_d,
|
||||
&beta, dstTensorDesc, dstData) );
|
||||
} else {
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(0);
|
||||
checkCUDNN( cudnnBatchNormalizationForwardInference(net->cudnnHandle,
|
||||
CUDNN_BATCHNORM_SPATIAL, &alpha, &beta,
|
||||
dstTensorDesc, dstData, dstTensorDesc,
|
||||
dstData, biasTensorDesc, //same tensor descriptor as bias
|
||||
scales_d, bias_d, mean_d, variance_d,
|
||||
TKDNN_BN_MIN_EPSILON) );
|
||||
}
|
||||
|
||||
//update data dimensions
|
||||
dim = output_dim;
|
||||
return dstData;
|
||||
}
|
||||
|
||||
|
||||
}}
|
||||
+1
-1
@@ -37,7 +37,7 @@ dnnType* Dense::infer(dataDim_t &dim, dnnType* srcData) {
|
||||
// place bias into dstData
|
||||
checkCuda( cudaMemcpy(dstData, bias_d, dim_y*sizeof(dnnType), cudaMemcpyDeviceToDevice) );
|
||||
|
||||
//do matrix moltiplication
|
||||
//do matrix multiplication
|
||||
checkERROR( cublasSgemv(net->cublasHandle, CUBLAS_OP_T,
|
||||
dim_x, dim_y,
|
||||
&alpha,
|
||||
|
||||
@@ -15,6 +15,11 @@ Flatten::Flatten(Network *net) : Layer(net) {
|
||||
output_dim.w = 1;
|
||||
output_dim.l = 1;
|
||||
|
||||
this->h = 1;
|
||||
this->w = 1;
|
||||
this->rows = input_dim.c;
|
||||
this->cols = input_dim.h * input_dim.w;
|
||||
this->c = input_dim.w * input_dim.h * input_dim.c;
|
||||
}
|
||||
|
||||
Flatten::~Flatten() {
|
||||
|
||||
@@ -0,0 +1,165 @@
|
||||
#include "Int8BatchStream.h"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/dnn/dnn.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
|
||||
BatchStream::BatchStream(tk::dnn::dataDim_t dim, int batchSize, int maxBatches, const std::string& fileimglist, const std::string& filelabellist) {
|
||||
mBatchSize = batchSize;
|
||||
mMaxBatches = maxBatches;
|
||||
mDims = nvinfer1::Dims4{ dim.n, dim.c, dim.h, dim.w };
|
||||
mHeight = dim.h;
|
||||
mWidth = dim.w;
|
||||
mImageSize = mDims.d[1]*mDims.d[2]*mDims.d[3];
|
||||
mBatch.resize(mBatchSize*mImageSize, 0);
|
||||
mLabels.resize(mBatchSize, 0);
|
||||
mFileBatch.resize(mDims.d[0]*mImageSize, 0);
|
||||
mFileLabels.resize(mDims.d[0], 0);
|
||||
mFileImgList = fileimglist;
|
||||
readInListFile(fileimglist, mListImg);
|
||||
mFileLabelList = filelabellist;
|
||||
readInListFile(filelabellist, mListLabel);
|
||||
|
||||
reset(0);
|
||||
}
|
||||
|
||||
void BatchStream::reset(int firstBatch) {
|
||||
mBatchCount = 0;
|
||||
mFileCount = 0;
|
||||
mFileBatchPos = mDims.d[0];
|
||||
skip(firstBatch);
|
||||
}
|
||||
|
||||
bool BatchStream::next() {
|
||||
std::cout<<"Next batch: "<<mBatchCount<<" of "<<mMaxBatches<<"\n";
|
||||
if (mBatchCount == mMaxBatches-1)
|
||||
return false;
|
||||
|
||||
for (int csize = 1, batchPos = 0; batchPos < mBatchSize; batchPos += csize, mFileBatchPos += csize) {
|
||||
assert(mFileBatchPos > 0 && mFileBatchPos <= mDims.d[0]);
|
||||
if (mFileBatchPos == mDims.d[0] && !update())
|
||||
return false;
|
||||
|
||||
csize = std::min(mBatchSize - batchPos, mDims.d[0] - mFileBatchPos);
|
||||
std::copy_n(getFileBatch() + mFileBatchPos * mImageSize, csize * mImageSize, getBatch() + batchPos * mImageSize);
|
||||
std::copy_n(getFileLabels() + mFileBatchPos, csize, getLabels() + batchPos);
|
||||
}
|
||||
mBatchCount++;
|
||||
return true;
|
||||
}
|
||||
|
||||
void BatchStream::skip(int skipCount) {
|
||||
if (mBatchSize >= mDims.d[0] && mBatchSize%mDims.d[0] == 0 && mFileBatchPos == mDims.d[0]) {
|
||||
mFileCount += skipCount * mBatchSize / mDims.d[0];
|
||||
return;
|
||||
}
|
||||
|
||||
int x = mBatchCount;
|
||||
for (int i = 0; i < skipCount; i++)
|
||||
next();
|
||||
mBatchCount = x;
|
||||
}
|
||||
|
||||
void BatchStream::readInListFile(const std::string& dataFilePath, std::vector<std::string>& mListIn) {
|
||||
// dataFilePath contains the list of image paths
|
||||
int count = 0;
|
||||
FILE* f = fopen(dataFilePath.c_str(), "r");
|
||||
if (!f)
|
||||
FatalError("failed to open " + dataFilePath);
|
||||
|
||||
char str[512];
|
||||
while (fgets(str, 512, f) != NULL) {
|
||||
for (int i = 0; str[i] != '\0'; ++i) {
|
||||
if (str[i] == '\n'){
|
||||
str[i] = '\0';
|
||||
break;
|
||||
}
|
||||
}
|
||||
count ++;
|
||||
mListIn.push_back(str);
|
||||
if(count == mMaxBatches)
|
||||
break;
|
||||
}
|
||||
fclose(f);
|
||||
}
|
||||
|
||||
void BatchStream::readCVimage(std::string inputFileName, std::vector<float>& res, bool fixshape) {
|
||||
// unaltered original DsImage
|
||||
cv::Mat m_OrigImage;
|
||||
// letterboxed DsImage given to the network as input
|
||||
cv::Mat m_LetterboxImage;
|
||||
m_OrigImage = cv::imread(inputFileName, cv::IMREAD_COLOR);
|
||||
|
||||
if (!m_OrigImage.data || m_OrigImage.cols <= 0 || m_OrigImage.rows <= 0)
|
||||
FatalError("Unable to open " + inputFileName);
|
||||
|
||||
int m_Height = m_OrigImage.rows;
|
||||
int m_Width = m_OrigImage.cols;
|
||||
if(fixshape) {
|
||||
m_Height = mHeight;
|
||||
m_Width = mWidth;
|
||||
}
|
||||
std::cout<<"image is "<<inputFileName<<": "<<m_Height<<" * "<<m_Width<<std::endl;
|
||||
// resize the DsImage with scale
|
||||
float dim = std::max(m_Height, m_Width);
|
||||
int resizeH = ((m_Height / dim) * m_Height);
|
||||
int resizeW = ((m_Width / dim) * m_Width);
|
||||
float m_ScalingFactor = static_cast<float>(resizeH) / static_cast<float>(m_Height);
|
||||
|
||||
// Additional checks for images with non even dims
|
||||
if ((m_Width - resizeW) % 2) resizeW--;
|
||||
if ((m_Height - resizeH) % 2) resizeH--;
|
||||
assert((m_Width - resizeW) % 2 == 0);
|
||||
assert((m_Height - resizeH) % 2 == 0);
|
||||
|
||||
int m_XOffset = (m_Width - resizeW) / 2;
|
||||
int m_YOffset = (m_Height - resizeH) / 2;
|
||||
|
||||
assert(2 * m_XOffset + resizeW == m_Width);
|
||||
assert(2 * m_YOffset + resizeH == m_Height);
|
||||
|
||||
// resizing
|
||||
cv::resize(m_OrigImage, m_LetterboxImage, cv::Size(resizeW, resizeH), 0, 0, cv::INTER_CUBIC);
|
||||
// letterboxing
|
||||
cv::copyMakeBorder(m_LetterboxImage, m_LetterboxImage, m_YOffset, m_YOffset, m_XOffset,
|
||||
m_XOffset, cv::BORDER_CONSTANT, cv::Scalar(128, 128, 128));
|
||||
m_LetterboxImage.convertTo(m_LetterboxImage, CV_32FC3, 1 / 255.0);
|
||||
// converting to RGB and NCHW format
|
||||
m_LetterboxImage = cv::dnn::blobFromImage(m_LetterboxImage);
|
||||
res.assign(m_LetterboxImage.begin<float>(), m_LetterboxImage.end<float>());
|
||||
}
|
||||
|
||||
void BatchStream::readLabels(std::string inputFileName, std::vector<float>& ris) {
|
||||
std::ifstream is(inputFileName.c_str());
|
||||
|
||||
std::string line;
|
||||
while (std::getline(is, line))
|
||||
{
|
||||
std::istringstream iss(line);
|
||||
float val;
|
||||
if(!(iss >> val)) { break; } // error
|
||||
ris.push_back(val);
|
||||
}
|
||||
}
|
||||
|
||||
bool BatchStream::update() {
|
||||
std::string imgFileName = mListImg[mFileCount];
|
||||
std::string labelFileName = mListLabel[mFileCount];
|
||||
mFileCount++;
|
||||
|
||||
//read image
|
||||
mFileBatch.clear();
|
||||
readCVimage(imgFileName, mFileBatch);
|
||||
// std::transform(
|
||||
// singleImg_rawData.begin(), singleImg_rawData.end(), mFileBatch.begin(), [](uint8_t val) { return static_cast<float>(val); });
|
||||
|
||||
//read label
|
||||
mFileLabels.clear();
|
||||
readLabels(labelFileName, mFileLabels);
|
||||
// std::transform(
|
||||
// singleLabels_rawData.begin(), singleLabels_rawData.end(), mFileLabels.begin(), [](uint8_t val) { return static_cast<float>(val); });
|
||||
|
||||
mFileBatchPos = 0;
|
||||
return true;
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
#include "Int8Calibrator.h"
|
||||
|
||||
Int8EntropyCalibrator::Int8EntropyCalibrator(BatchStream& stream, int firstBatch,
|
||||
const std::string& calibTableFilePath,
|
||||
const std::string& inputBlobName,
|
||||
bool readCache):
|
||||
mStream(stream),
|
||||
mCalibTableFilePath(calibTableFilePath),
|
||||
mInputBlobName(inputBlobName.c_str()),
|
||||
mReadCache(readCache) {
|
||||
nvinfer1::Dims4 dims = mStream.getDims();
|
||||
mInputCount = mStream.getBatchSize() + dims.d[1]*dims.d[2]*dims.d[3];
|
||||
checkCuda(cudaMalloc(&mDeviceInput, mInputCount * sizeof(float)));
|
||||
mStream.reset(firstBatch);
|
||||
}
|
||||
|
||||
bool Int8EntropyCalibrator::getBatch(void* bindings[], const char* names[], int nbBindings) NOEXCEPT {
|
||||
if (!mStream.next())
|
||||
return false;
|
||||
|
||||
checkCuda(cudaMemcpy(mDeviceInput, mStream.getBatch(), mInputCount * sizeof(float), cudaMemcpyHostToDevice));
|
||||
assert(!strcmp(names[0], mInputBlobName.c_str()));
|
||||
bindings[0] = mDeviceInput;
|
||||
return true;
|
||||
}
|
||||
|
||||
const void* Int8EntropyCalibrator::readCalibrationCache(size_t& length) NOEXCEPT {
|
||||
mCalibrationCache.clear();
|
||||
assert(!mCalibTableFilePath.empty());
|
||||
std::ifstream input(mCalibTableFilePath, std::ios::binary);
|
||||
input >> std::noskipws;
|
||||
input >> std::noskipws;
|
||||
if (mReadCache && input.good())
|
||||
std::copy(std::istream_iterator<char>(input), std::istream_iterator<char>(),
|
||||
std::back_inserter(mCalibrationCache));
|
||||
|
||||
length = mCalibrationCache.size();
|
||||
return length ? &mCalibrationCache[0] : nullptr;
|
||||
}
|
||||
|
||||
void Int8EntropyCalibrator::writeCalibrationCache(const void* cache, size_t length) NOEXCEPT {
|
||||
assert(!mCalibTableFilePath.empty());
|
||||
std::ofstream output(mCalibTableFilePath, std::ios::binary);
|
||||
output.write(reinterpret_cast<const char*>(cache), length);
|
||||
output.close();
|
||||
}
|
||||
+336
@@ -0,0 +1,336 @@
|
||||
#include <iostream>
|
||||
|
||||
#include "Layer.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
LSTM::LSTM( Network *net, int hiddensize, bool returnSeq, std::string fname_weights) :
|
||||
Layer(net) {
|
||||
|
||||
this->returnSeq = returnSeq;
|
||||
int batchSize = input_dim.n;
|
||||
int inputSize = input_dim.c;
|
||||
seqLen = input_dim.w;
|
||||
stateSize = hiddensize;
|
||||
|
||||
// init Tensor Descriptors
|
||||
std::vector<cudnnTensorDescriptor_t> x_vec(seqLen);
|
||||
std::vector<cudnnTensorDescriptor_t> y_vec(seqLen);
|
||||
|
||||
int dimA[3];
|
||||
int strideA[3];
|
||||
for (int i = 0; i < seqLen; i++) {
|
||||
checkCUDNN(cudnnCreateTensorDescriptor(&x_vec[i]));
|
||||
checkCUDNN(cudnnCreateTensorDescriptor(&y_vec[i]));
|
||||
|
||||
dimA[0] = batchSize;
|
||||
dimA[1] = inputSize;
|
||||
dimA[2] = 1;
|
||||
dimA[0] = batchSize;
|
||||
dimA[1] = inputSize;
|
||||
strideA[0] = dimA[2] * dimA[1];
|
||||
strideA[1] = dimA[2];
|
||||
strideA[2] = 1;
|
||||
checkCUDNN(cudnnSetTensorNdDescriptor(x_vec[i],
|
||||
net->dataType, 3, dimA, strideA));
|
||||
|
||||
dimA[0] = batchSize;
|
||||
dimA[1] = stateSize;
|
||||
dimA[2] = 1;
|
||||
strideA[0] = dimA[2] * dimA[1];
|
||||
strideA[1] = dimA[2];
|
||||
strideA[2] = 1;
|
||||
checkCUDNN(cudnnSetTensorNdDescriptor(y_vec[i],
|
||||
net->dataType, 3, dimA, strideA));
|
||||
}
|
||||
// apply tensordesc
|
||||
x_desc_vec_ = x_vec;
|
||||
y_desc_vec_ = y_vec;
|
||||
|
||||
|
||||
// set the state tensors
|
||||
dimA[0] = numLayers;
|
||||
dimA[1] = batchSize;
|
||||
dimA[2] = stateSize;
|
||||
strideA[0] = dimA[2] * dimA[1];
|
||||
strideA[1] = dimA[2];
|
||||
strideA[2] = 1;
|
||||
checkCUDNN(cudnnCreateTensorDescriptor(&hx_desc_));
|
||||
checkCUDNN(cudnnCreateTensorDescriptor(&cx_desc_));
|
||||
checkCUDNN(cudnnCreateTensorDescriptor(&hy_desc_));
|
||||
checkCUDNN(cudnnCreateTensorDescriptor(&cy_desc_));
|
||||
checkCUDNN(cudnnSetTensorNdDescriptor(hx_desc_, net->dataType, 3, dimA, strideA));
|
||||
checkCUDNN(cudnnSetTensorNdDescriptor(cx_desc_, net->dataType, 3, dimA, strideA));
|
||||
checkCUDNN(cudnnSetTensorNdDescriptor(hy_desc_, net->dataType, 3, dimA, strideA));
|
||||
checkCUDNN(cudnnSetTensorNdDescriptor(cy_desc_, net->dataType, 3, dimA, strideA));
|
||||
// allocate dnnType *hx_ptr, *cx_ptr, *hy_ptr, *cy_ptr;
|
||||
stateDataDim = dimA[0]*dimA[1]*dimA[2];
|
||||
checkCuda( cudaMalloc(&hx_ptr, stateDataDim*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&cx_ptr, stateDataDim*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&hy_ptr, stateDataDim*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&cy_ptr, stateDataDim*sizeof(dnnType)) );
|
||||
|
||||
|
||||
|
||||
// Create Dropout descriptors // TODO: ??? IS IT NECESSARY ???
|
||||
float dropoutprob = 0.1f; // random val ????
|
||||
checkCUDNN(cudnnCreateDropoutDescriptor(&dropoutDesc));
|
||||
checkCUDNN(cudnnDropoutGetStatesSize(net->cudnnHandle, &dropout_byte_));
|
||||
dropout_size_ = dropout_byte_ / sizeof(dnnType);
|
||||
checkCuda( cudaMalloc(&dropout_states_, dropout_byte_) );
|
||||
uint64_t seed_ = 17 + rand() % 4096; // NOLINT(runtime/threadsafe_fn)
|
||||
checkCUDNN(cudnnSetDropoutDescriptor(dropoutDesc,
|
||||
net->cudnnHandle, dropoutprob, dropout_states_, dropout_byte_, seed_));
|
||||
|
||||
|
||||
// RNN descriptors
|
||||
checkCUDNN(cudnnCreateRNNDescriptor(&rnnDesc));
|
||||
|
||||
#if CUDNN_MAJOR > 7
|
||||
checkCUDNN(cudnnSetRNNDescriptor_v6(net->cudnnHandle,rnnDesc, stateSize, numLayers, dropoutDesc,
|
||||
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
|
||||
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
|
||||
cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL,
|
||||
cudnnRNNMode_t::CUDNN_LSTM,
|
||||
cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD,
|
||||
net->dataType));
|
||||
#else
|
||||
checkCUDNN(cudnnSetRNNDescriptor(net->cudnnHandle,rnnDesc, stateSize, numLayers, dropoutDesc,
|
||||
cudnnRNNInputMode_t::CUDNN_LINEAR_INPUT,
|
||||
//(bidirectional ? cudnnDirectionMode_t::CUDNN_BIDIRECTIONAL : cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL),
|
||||
cudnnDirectionMode_t::CUDNN_UNIDIRECTIONAL,
|
||||
cudnnRNNMode_t::CUDNN_LSTM,
|
||||
cudnnRNNAlgo_t::CUDNN_RNN_ALGO_STANDARD,
|
||||
net->dataType));
|
||||
#endif
|
||||
|
||||
|
||||
// Get temp space sizes
|
||||
checkCUDNN(cudnnGetRNNWorkspaceSize(net->cudnnHandle,
|
||||
rnnDesc, seqLen, x_desc_vec_.data(), &workspace_byte_));
|
||||
workspace_size_ = workspace_byte_ / sizeof(dnnType);
|
||||
checkCuda( cudaMalloc(&work_space_, workspace_byte_) );
|
||||
|
||||
|
||||
// Check that number of params are correct
|
||||
size_t cudnn_param_size;
|
||||
checkCUDNN(cudnnGetRNNParamsSize(net->cudnnHandle,
|
||||
rnnDesc,x_desc_vec_[0], &cudnn_param_size, net->dataType));
|
||||
int cudnn_params = cudnn_param_size/sizeof(dnnType);
|
||||
//std::cout<<"LSTM params size: "<<cudnn_params << ", bytes: "<<cudnn_param_size<<"\n";
|
||||
|
||||
// Set param descriptors
|
||||
checkCUDNN(cudnnCreateFilterDescriptor(&w_desc_));
|
||||
int dim_w[3] = {1, 1, 1};
|
||||
dim_w[0] = cudnn_params;
|
||||
checkCUDNN(cudnnSetFilterNdDescriptor(w_desc_,
|
||||
net->dataType, net->tensorFormat, 3, dim_w));
|
||||
|
||||
// load params
|
||||
std::cout<<"Reading weights: PARAMS="<<cudnn_params*2<<"\n";
|
||||
readBinaryFile(fname_weights, cudnn_params*2, &w_h, &w_ptr);
|
||||
// set forward and backward params
|
||||
wf_ptr = w_ptr;
|
||||
wb_ptr = w_ptr + cudnn_params;
|
||||
//std::cout<<"wf: "<<wf_ptr<<" wb "<<wb_ptr<<"\n";
|
||||
|
||||
// set output dim
|
||||
output_dim = input_dim;
|
||||
output_dim.c = stateSize*(bidirectional ? 2 : 1);
|
||||
|
||||
// if retunseq is disabled only the last timestamp is returned
|
||||
if(!returnSeq) {
|
||||
output_dim.h = 1;
|
||||
output_dim.w = 1;
|
||||
}
|
||||
|
||||
//allocate data for infer result
|
||||
checkCuda( cudaMalloc(&dstData, output_dim.tot()*sizeof(dnnType)) );
|
||||
|
||||
// used during inference
|
||||
one_output_dim = input_dim;
|
||||
one_output_dim.c = stateSize;
|
||||
checkCuda( cudaMalloc(&srcF, input_dim.tot()*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&srcB, input_dim.tot()*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&dstF, one_output_dim.tot()*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&dstB_NR, one_output_dim.tot()*sizeof(dnnType)) );
|
||||
checkCuda( cudaMalloc(&dstB, one_output_dim.tot()*sizeof(dnnType)) );
|
||||
|
||||
|
||||
/*
|
||||
// Query weight layout
|
||||
cudnnFilterDescriptor_t m_desc;
|
||||
checkCUDNN(cudnnCreateFilterDescriptor(&m_desc));
|
||||
dnnType *p;
|
||||
int n = 8; // lstm layers
|
||||
|
||||
printCenteredTitle("WEIGHTS", '=', 20);
|
||||
for (int i = 0; i < numLayers; ++i) {
|
||||
for (int j = 0; j < n; ++j) {
|
||||
|
||||
checkCUDNN(cudnnGetRNNLinLayerMatrixParams(net->cudnnHandle, rnnDesc,
|
||||
i, x_desc_vec_[0], w_desc_, 0, j, m_desc, (void**)&p));
|
||||
|
||||
std::cout << "ptr: " << ((int64_t)(p - NULL))/sizeof(dnnType)<<"\n";
|
||||
|
||||
cudnnDataType_t t;
|
||||
cudnnTensorFormat_t f;
|
||||
int ndim = 5;
|
||||
int dims[5] = {0, 0, 0, 0, 0};
|
||||
checkCUDNN(cudnnGetFilterNdDescriptor(m_desc, ndim, &t, &f, &ndim, &dims[0]));
|
||||
std::cout << "(layer, linlayer): " << i << " " << j << "\n";
|
||||
|
||||
int tot = 1;
|
||||
for (int i = 0; i < ndim; ++i) {
|
||||
std::cout << dims[i] << " ";
|
||||
tot *= dims[i];
|
||||
}
|
||||
std::cout<<"\t-> "<<tot<<"\n\n";
|
||||
}
|
||||
}
|
||||
|
||||
printCenteredTitle("BIAS", '=', 20);
|
||||
for (int i = 0; i < numLayers; ++i) {
|
||||
for (int j = 0; j < n; ++j) {
|
||||
checkCUDNN(cudnnGetRNNLinLayerBiasParams(net->cudnnHandle, rnnDesc,
|
||||
i, x_desc_vec_[0], w_desc_, 0, j, m_desc, (void**)&p));
|
||||
|
||||
std::cout << "ptr: " << ((int64_t)(p - NULL))/sizeof(dnnType)<<"\n";
|
||||
|
||||
cudnnDataType_t t;
|
||||
cudnnTensorFormat_t f;
|
||||
int ndim = 5;
|
||||
int dims[5] = {0, 0, 0, 0, 0};
|
||||
checkCUDNN(cudnnGetFilterNdDescriptor(m_desc, ndim, &t, &f, &ndim, &dims[0]));
|
||||
std::cout << "(layer, linlayer): " << i << " " << j << "\n";
|
||||
|
||||
int tot = 1;
|
||||
for (int i = 0; i < ndim; ++i) {
|
||||
std::cout << dims[i] << " ";
|
||||
tot *= dims[i];
|
||||
}
|
||||
std::cout<<"\t-> "<<tot<<"\n\n";
|
||||
}
|
||||
}
|
||||
|
||||
checkCUDNN(cudnnDestroyFilterDescriptor(m_desc));
|
||||
*/
|
||||
}
|
||||
|
||||
LSTM::~LSTM() {
|
||||
checkCuda(cudaFree(hx_ptr));
|
||||
checkCuda(cudaFree(cx_ptr));
|
||||
checkCuda(cudaFree(hy_ptr));
|
||||
checkCuda(cudaFree(cy_ptr));
|
||||
checkCuda(cudaFree(w_ptr ));
|
||||
|
||||
checkCuda(cudaFree(work_space_ ));
|
||||
checkCuda(cudaFree(dropout_states_));
|
||||
|
||||
checkCuda(cudaFree(srcF));
|
||||
checkCuda(cudaFree(srcB));
|
||||
checkCuda(cudaFree(dstF));
|
||||
checkCuda(cudaFree(dstB_NR));
|
||||
checkCuda(cudaFree(dstB));
|
||||
checkCuda(cudaFree(dstData));
|
||||
}
|
||||
|
||||
dnnType* LSTM::infer(dataDim_t &dim, dnnType* srcData) {
|
||||
|
||||
// transpose input
|
||||
matrixTranspose(net->cublasHandle, srcData, srcF, dim.c, dim.h*dim.w*dim.l);
|
||||
|
||||
// build srcB as reversed srcF
|
||||
for(int i=0; i<input_dim.w; i++) {
|
||||
int off_0 = i*(input_dim.c);
|
||||
int off_1 = (i+1)*(input_dim.c);
|
||||
checkCuda( cudaMemcpy(srcB + dim.tot() - off_1, srcF + off_0,
|
||||
input_dim.c*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
}
|
||||
|
||||
// forward
|
||||
{
|
||||
// reset states
|
||||
checkCuda( cudaMemset(hx_ptr, 0, stateDataDim*sizeof(float)) );
|
||||
checkCuda( cudaMemset(cx_ptr, 0, stateDataDim*sizeof(float)) );
|
||||
|
||||
|
||||
checkCUDNN(cudnnRNNForwardInference(net->cudnnHandle,
|
||||
rnnDesc,
|
||||
seqLen, // number of time steps (nT)
|
||||
x_desc_vec_.data(), // input array of desc (nT*nC_in)
|
||||
srcF, // input pointer
|
||||
hx_desc_, // initial hidden state desc
|
||||
hx_ptr, // initial hidden state pointer
|
||||
cx_desc_, // initial cell state desc
|
||||
cx_ptr, // initial cell state pointer
|
||||
w_desc_, // weights desc
|
||||
wf_ptr, // weights pointer
|
||||
y_desc_vec_.data(), // output desc (nT*nC_out)
|
||||
dstF, // output pointer
|
||||
hy_desc_, // final hidden state desc
|
||||
hy_ptr, // final hidden state pointer
|
||||
cy_desc_, // final cell state desc
|
||||
cy_ptr, // final cell state pointer
|
||||
work_space_, // workspace pointer
|
||||
workspace_byte_)); // workspace size
|
||||
}
|
||||
|
||||
// backward
|
||||
{
|
||||
// reset states
|
||||
checkCuda( cudaMemset(hx_ptr, 0, stateDataDim*sizeof(float)) );
|
||||
checkCuda( cudaMemset(cx_ptr, 0, stateDataDim*sizeof(float)) );
|
||||
|
||||
checkCUDNN(cudnnRNNForwardInference(net->cudnnHandle,
|
||||
rnnDesc,
|
||||
seqLen, // number of time steps (nT)
|
||||
x_desc_vec_.data(), // input array of desc (nT*nC_in)
|
||||
srcB, // input pointer
|
||||
hx_desc_, // initial hidden state desc
|
||||
hx_ptr, // initial hidden state pointer
|
||||
cx_desc_, // initial cell state desc
|
||||
cx_ptr, // initial cell state pointer
|
||||
w_desc_, // weights desc
|
||||
wb_ptr, // weights pointer
|
||||
y_desc_vec_.data(), // output desc (nT*nC_out)
|
||||
dstB_NR, // output pointer
|
||||
hy_desc_, // final hidden state desc
|
||||
hy_ptr, // final hidden state pointer
|
||||
cy_desc_, // final cell state desc
|
||||
cy_ptr, // final cell state pointer
|
||||
work_space_, // workspace pointer
|
||||
workspace_byte_)); // workspace size
|
||||
}
|
||||
|
||||
|
||||
// reverse order of dstB
|
||||
for(int i=0; i<one_output_dim.w; i++) {
|
||||
int off_0 = i*(one_output_dim.c);
|
||||
int off_1 = (i+1)*(one_output_dim.c);
|
||||
checkCuda( cudaMemcpy(dstB + one_output_dim.tot() - off_1, dstB_NR + off_0,
|
||||
one_output_dim.c*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
}
|
||||
|
||||
// if retunseq is disabled only the last timestamp is returned
|
||||
if(returnSeq) {
|
||||
// forward transpose
|
||||
matrixTranspose(net->cublasHandle, dstF, dstData,
|
||||
one_output_dim.h* one_output_dim.w*one_output_dim.l, one_output_dim.c);
|
||||
// backward transpose
|
||||
matrixTranspose(net->cublasHandle, dstB, dstData + one_output_dim.tot(),
|
||||
one_output_dim.h* one_output_dim.w*one_output_dim.l, one_output_dim.c);
|
||||
} else {
|
||||
// copy last of forward
|
||||
checkCuda( cudaMemcpy(dstData, dstF + one_output_dim.tot() - one_output_dim.c,
|
||||
one_output_dim.c*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
// copy first of backward
|
||||
checkCuda( cudaMemcpy(dstData + one_output_dim.c, dstB,
|
||||
one_output_dim.c*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
}
|
||||
|
||||
dim = output_dim;
|
||||
return dstData;
|
||||
}
|
||||
|
||||
}}
|
||||
+8
-1
@@ -7,7 +7,7 @@ namespace tk { namespace dnn {
|
||||
Layer::Layer(Network *net) {
|
||||
|
||||
this->net = net;
|
||||
|
||||
this->final = false;
|
||||
if(net != nullptr) {
|
||||
this->input_dim = net->getOutputDim();
|
||||
this->output_dim = input_dim;
|
||||
@@ -18,12 +18,19 @@ Layer::Layer(Network *net) {
|
||||
if(!net->addLayer(this))
|
||||
FatalError("Net reached max number of layers");
|
||||
}
|
||||
|
||||
feature_map_size = input_dim.tot() + output_dim.tot();
|
||||
}
|
||||
|
||||
Layer::~Layer() {
|
||||
|
||||
checkCUDNN( cudnnDestroyTensorDescriptor(srcTensorDesc) );
|
||||
checkCUDNN( cudnnDestroyTensorDescriptor(dstTensorDesc) );
|
||||
|
||||
if(dstData != nullptr) {
|
||||
cudaFree(dstData);
|
||||
dstData = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
}}
|
||||
+29
-21
@@ -8,26 +8,37 @@ namespace tk { namespace dnn {
|
||||
|
||||
LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
||||
int kh, int kw, int kl,
|
||||
std::string fname_weights, bool batchnorm) : Layer(net) {
|
||||
|
||||
std::string fname_weights, bool batchnorm, bool additional_bias, bool deConv, int groups) : Layer(net) {
|
||||
inputs = inputs/groups;
|
||||
|
||||
this->inputs = inputs;
|
||||
this->outputs = outputs;
|
||||
this->weights_path = std::string(fname_weights);
|
||||
|
||||
|
||||
std::cout<<"Reading weights: I="<<inputs<<" O="<<outputs<<" KERNEL="<<kh<<"x"<<kw<<"x"<<kl<<"\n";
|
||||
int seek = 0;
|
||||
readBinaryFile(weights_path.c_str(), inputs*outputs*kh*kw*kl, &data_h, &data_d, seek);
|
||||
seek += inputs*outputs*kh*kw*kl;
|
||||
readBinaryFile(weights_path.c_str(), outputs, &bias_h, &bias_d, seek);
|
||||
n_params = seek;
|
||||
|
||||
this->additional_bias = additional_bias;
|
||||
if(additional_bias) {
|
||||
readBinaryFile(weights_path.c_str(), outputs, &bias2_h, &bias2_d, seek);
|
||||
seek += outputs;
|
||||
}
|
||||
|
||||
readBinaryFile(weights_path.c_str(), outputs, &bias_h, &bias_d, seek);
|
||||
seek += outputs;
|
||||
|
||||
this->batchnorm = batchnorm;
|
||||
if(batchnorm) {
|
||||
seek += outputs;
|
||||
|
||||
readBinaryFile(weights_path.c_str(), outputs, &scales_h, &scales_d, seek);
|
||||
seek += outputs;
|
||||
readBinaryFile(weights_path.c_str(), outputs, &mean_h, &mean_d, seek);
|
||||
seek += outputs;
|
||||
readBinaryFile(weights_path.c_str(), outputs, &variance_h, &variance_d, seek);
|
||||
seek += outputs;
|
||||
|
||||
float eps = TKDNN_BN_MIN_EPSILON;
|
||||
|
||||
@@ -52,6 +63,14 @@ LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
||||
float2half(data_d, data16_d, w_size);
|
||||
cudaMemcpy(data16_h, data16_d, w_size*sizeof(__half), cudaMemcpyDeviceToHost);
|
||||
|
||||
if(additional_bias){
|
||||
int b2_size = outputs;
|
||||
bias216_h = new __half[b2_size];
|
||||
cudaMalloc(&bias216_d, w_size*sizeof(__half));
|
||||
float2half(bias2_d, bias216_d, b2_size);
|
||||
cudaMemcpy(bias216_h, bias216_d, b2_size*sizeof(__half), cudaMemcpyDeviceToHost);
|
||||
}
|
||||
|
||||
int b_size = outputs;
|
||||
bias16_h = new __half[b_size];
|
||||
cudaMalloc(&bias16_d, w_size*sizeof(__half));
|
||||
@@ -80,7 +99,6 @@ LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
||||
cudaMemcpy(power16_h, power16_d, b_size*sizeof(__half), cudaMemcpyDeviceToHost);
|
||||
|
||||
//mean array
|
||||
|
||||
cudaMemcpy(tmp_d, mean_h, b_size*sizeof(float), cudaMemcpyHostToDevice);
|
||||
float2half(tmp_d, mean16_d, b_size);
|
||||
cudaMemcpy(mean16_h, mean16_d, b_size*sizeof(__half), cudaMemcpyDeviceToHost);
|
||||
@@ -91,27 +109,17 @@ LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
||||
float2half(tmp_d, variance16_d, b_size);
|
||||
cudaMemcpy(variance16_h, variance16_d, b_size*sizeof(__half), cudaMemcpyDeviceToHost);
|
||||
|
||||
//conver scales
|
||||
//convert scales
|
||||
float2half(scales_d, scales16_d, b_size);
|
||||
cudaMemcpy(scales16_h, scales16_d, b_size*sizeof(__half), cudaMemcpyDeviceToHost);
|
||||
|
||||
cudaFree(tmp_d);
|
||||
}
|
||||
}
|
||||
|
||||
LayerWgs::~LayerWgs() {
|
||||
|
||||
delete [] data_h;
|
||||
delete [] bias_h;
|
||||
checkCuda( cudaFree(data_d) );
|
||||
checkCuda( cudaFree(bias_d) );
|
||||
|
||||
if(batchnorm) {
|
||||
delete [] scales_h;
|
||||
delete [] mean_h;
|
||||
delete [] variance_h;
|
||||
checkCuda( cudaFree(scales_d) );
|
||||
checkCuda( cudaFree(mean_d) );
|
||||
checkCuda( cudaFree(variance_d) );
|
||||
}
|
||||
releaseHost();
|
||||
releaseDevice();
|
||||
}
|
||||
|
||||
}}
|
||||
|
||||
@@ -0,0 +1,312 @@
|
||||
#include "MobilenetDetection.h"
|
||||
|
||||
bool boxProbCmp(const tk::dnn::box &a, const tk::dnn::box &b){
|
||||
return (a.prob > b.prob);
|
||||
}
|
||||
|
||||
namespace tk{ namespace dnn{
|
||||
|
||||
void MobilenetDetection::generate_ssd_priors(const SSDSpec *specs, const int n_specs, bool clamp){
|
||||
nPriors = 0;
|
||||
for (int i = 0; i < n_specs; i++){
|
||||
nPriors += specs[i].featureSize * specs[i].featureSize * 6;
|
||||
}
|
||||
|
||||
priors = (float *)malloc(N_COORDS * nPriors * sizeof(float));
|
||||
|
||||
int i_prio = 0;
|
||||
float scale, x_center, y_center, h, w, size, ratio;
|
||||
int min, max;
|
||||
for (int i = 0; i < n_specs; i++){
|
||||
scale = (float)imageSize / (float)specs[i].shrinkage;
|
||||
min = specs[i].boxHeight > specs[i].boxWidth ? specs[i].boxWidth : specs[i].boxHeight;
|
||||
max = specs[i].boxHeight < specs[i].boxWidth ? specs[i].boxWidth : specs[i].boxHeight;
|
||||
for (int j = 0; j < specs[i].featureSize; j++){
|
||||
for (int k = 0; k < specs[i].featureSize; k++){
|
||||
//small sized square box
|
||||
size = min;
|
||||
x_center = (k + 0.5f) / scale;
|
||||
y_center = (j + 0.5f) / scale;
|
||||
h = w = (float)size / (float)imageSize;
|
||||
|
||||
priors[i_prio * N_COORDS + 0] = x_center;
|
||||
priors[i_prio * N_COORDS + 1] = y_center;
|
||||
priors[i_prio * N_COORDS + 2] = w;
|
||||
priors[i_prio * N_COORDS + 3] = h;
|
||||
++i_prio;
|
||||
|
||||
//big sized square box
|
||||
size = sqrt(max * min);
|
||||
h = w = (float)size / (float)imageSize;
|
||||
|
||||
priors[i_prio * N_COORDS + 0] = x_center;
|
||||
priors[i_prio * N_COORDS + 1] = y_center;
|
||||
priors[i_prio * N_COORDS + 2] = w;
|
||||
priors[i_prio * N_COORDS + 3] = h;
|
||||
++i_prio;
|
||||
|
||||
//change h/w ratio of the small sized box
|
||||
size = min;
|
||||
h = w = size / (float)imageSize;
|
||||
ratio = sqrt(specs[i].ratio1);
|
||||
priors[i_prio * N_COORDS + 0] = x_center;
|
||||
priors[i_prio * N_COORDS + 1] = y_center;
|
||||
priors[i_prio * N_COORDS + 2] = w * ratio;
|
||||
priors[i_prio * N_COORDS + 3] = h / ratio;
|
||||
++i_prio;
|
||||
|
||||
priors[i_prio * N_COORDS + 0] = x_center;
|
||||
priors[i_prio * N_COORDS + 1] = y_center;
|
||||
priors[i_prio * N_COORDS + 2] = w / ratio;
|
||||
priors[i_prio * N_COORDS + 3] = h * ratio;
|
||||
++i_prio;
|
||||
|
||||
ratio = sqrt(specs[i].ratio2);
|
||||
priors[i_prio * N_COORDS + 0] = x_center;
|
||||
priors[i_prio * N_COORDS + 1] = y_center;
|
||||
priors[i_prio * N_COORDS + 2] = w * ratio;
|
||||
priors[i_prio * N_COORDS + 3] = h / ratio;
|
||||
++i_prio;
|
||||
|
||||
priors[i_prio * N_COORDS + 0] = x_center;
|
||||
priors[i_prio * N_COORDS + 1] = y_center;
|
||||
priors[i_prio * N_COORDS + 2] = w / ratio;
|
||||
priors[i_prio * N_COORDS + 3] = h * ratio;
|
||||
++i_prio;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (clamp){
|
||||
for (int i = 0; i < nPriors * N_COORDS; i++){
|
||||
priors[i] = priors[i] > 1.0f ? 1.0f : priors[i];
|
||||
priors[i] = priors[i] < 0.0f ? 0.0f : priors[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void MobilenetDetection::convert_locatios_to_boxes_and_center(){
|
||||
float cur_x, cur_y;
|
||||
for (int i = 0; i < nPriors; i++){
|
||||
locations_h[i * N_COORDS + 0] = locations_h[i * N_COORDS + 0] * centerVariance * priors[i * N_COORDS + 2] + priors[i * N_COORDS + 0];
|
||||
locations_h[i * N_COORDS + 1] = locations_h[i * N_COORDS + 1] * centerVariance * priors[i * N_COORDS + 3] + priors[i * N_COORDS + 1];
|
||||
locations_h[i * N_COORDS + 2] = exp(locations_h[i * N_COORDS + 2] * sizeVariance) * priors[i * N_COORDS + 2];
|
||||
locations_h[i * N_COORDS + 3] = exp(locations_h[i * N_COORDS + 3] * sizeVariance) * priors[i * N_COORDS + 3];
|
||||
|
||||
cur_x = locations_h[i * N_COORDS + 0];
|
||||
cur_y = locations_h[i * N_COORDS + 1];
|
||||
|
||||
locations_h[i * N_COORDS + 0] = cur_x - locations_h[i * N_COORDS + 2] / 2;
|
||||
locations_h[i * N_COORDS + 1] = cur_y - locations_h[i * N_COORDS + 3] / 2;
|
||||
locations_h[i * N_COORDS + 2] = cur_x + locations_h[i * N_COORDS + 2] / 2;
|
||||
locations_h[i * N_COORDS + 3] = cur_y + locations_h[i * N_COORDS + 3] / 2;
|
||||
}
|
||||
}
|
||||
|
||||
float MobilenetDetection::iou(const tk::dnn::box &a, const tk::dnn::box &b){
|
||||
float max_x = a.x > b.x ? a.x : b.x;
|
||||
float max_y = a.y > b.y ? a.y : b.y;
|
||||
float min_w = a.w < b.w ? a.w : b.w;
|
||||
float min_h = a.h < b.h ? a.h : b.h;
|
||||
|
||||
float ao_w = min_w - max_x > 0 ? min_w - max_x : 0;
|
||||
float ao_h = min_h - max_y > 0 ? min_h - max_y : 0;
|
||||
|
||||
float area_overlap = ao_w * ao_h;
|
||||
float area_0_w = a.w - a.x > 0 ? a.w - a.x : 0;
|
||||
float area_0_h = a.h - a.y > 0 ? a.h - a.y : 0;
|
||||
|
||||
float area_1_w = b.w - b.x > 0 ? b.w - b.x : 0;
|
||||
float area_1_h = b.h - b.y > 0 ? b.h - b.y : 0;
|
||||
|
||||
float area_0 = area_0_h * area_0_w;
|
||||
float area_1 = area_1_h * area_1_w;
|
||||
|
||||
float iou = area_overlap / (area_0 + area_1 - area_overlap + 1e-5);
|
||||
return iou;
|
||||
}
|
||||
|
||||
bool MobilenetDetection::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh){
|
||||
std::cout<<(tensor_path).c_str()<<"\n";
|
||||
netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str());
|
||||
imageSize = netRT->input_dim.h;
|
||||
classes = n_classes;
|
||||
nBatches = n_batches;
|
||||
confThreshold = conf_thresh;
|
||||
|
||||
SSDSpec specs[N_SSDSPEC];
|
||||
|
||||
if(imageSize == 300){
|
||||
specs[0].setAll(19, 16, 60, 105, 2, 3);
|
||||
specs[1].setAll(10, 32, 105, 150, 2, 3);
|
||||
specs[2].setAll(5, 64, 150, 195, 2, 3);
|
||||
specs[3].setAll(3, 100, 195, 240, 2, 3);
|
||||
specs[4].setAll(2, 150, 240, 285, 2, 3);
|
||||
specs[5].setAll(1, 300, 285, 330, 2, 3);
|
||||
}
|
||||
else if(imageSize == 512){
|
||||
specs[0].setAll(32, 16, 60, 105, 2, 3);
|
||||
specs[1].setAll(16, 32, 105, 150, 2, 3);
|
||||
specs[2].setAll(8, 64, 150, 195, 2, 3);
|
||||
specs[3].setAll(4, 100, 195, 240, 2, 3);
|
||||
specs[4].setAll(2, 150, 240, 285, 2, 3);
|
||||
specs[5].setAll(1, 300, 285, 330, 2, 3);
|
||||
}
|
||||
else{
|
||||
FatalError("Input size for mobilenet not supported");
|
||||
}
|
||||
|
||||
generate_ssd_priors(specs, N_SSDSPEC);
|
||||
|
||||
#ifndef OPENCV_CUDACONTRIB
|
||||
checkCuda(cudaMallocHost(&input, sizeof(dnnType) * netRT->input_dim.tot() * nBatches));
|
||||
#endif
|
||||
checkCuda(cudaMalloc(&input_d, sizeof(dnnType) * netRT->input_dim.tot() * nBatches));
|
||||
|
||||
locations_h = (float *)malloc(N_COORDS * nPriors * sizeof(float));
|
||||
confidences_h = (float *)malloc(nPriors * classes * sizeof(float));
|
||||
|
||||
for (int c = 0; c < classes; c++){
|
||||
int offset = c * 123457 % classes;
|
||||
float r = getColor(2, offset, classes);
|
||||
float g = getColor(1, offset, classes);
|
||||
float b = getColor(0, offset, classes);
|
||||
colors[c] = cv::Scalar(int(255.0 * b), int(255.0 * g), int(255.0 * r));
|
||||
}
|
||||
|
||||
if(classes == 11){ //BDD
|
||||
const char *classes_names_[] = {
|
||||
"person","car","truck","bus","motor","bike","rider","traffic light","traffic sign","train"};
|
||||
classesNames = std::vector<std::string>(classes_names_, std::end(classes_names_));
|
||||
}
|
||||
else if(classes == 21){ //VOC
|
||||
const char *classes_names_[] = {
|
||||
"aeroplane", "bicycle", "bird", "boat", "bottle", "bus",
|
||||
"car", "cat", "chair", "cow", "diningtable", "dog", "horse", "motorbike",
|
||||
"person", "pottedplant", "sheep", "sofa", "train", "tvmonitor"};
|
||||
classesNames = std::vector<std::string>(classes_names_, std::end(classes_names_));
|
||||
|
||||
}
|
||||
else if (classes == 81){ //COCO
|
||||
const char *classes_names_[] = {
|
||||
"person" , "bicycle" , "car" , "motorbike" , "aeroplane" , "bus" ,
|
||||
"train" , "truck" , "boat" , "traffic light" , "fire hydrant" , "stop sign" ,
|
||||
"parking meter" , "bench" , "bird" , "cat" , "dog" , "horse" , "sheep" , "cow" ,
|
||||
"elephant" , "bear" , "zebra" , "giraffe" , "backpack" , "umbrella" , "handbag" ,
|
||||
"tie" , "suitcase" , "frisbee" , "skis" , "snowboard" , "sports ball" , "kite" ,
|
||||
"baseball bat" , "baseball glove" , "skateboard" , "surfboard" , "tennis racket" ,
|
||||
"bottle" , "wine glass" , "cup" , "fork" , "knife" , "spoon" , "bowl" , "banana" ,
|
||||
"apple" , "sandwich" , "orange" , "broccoli" , "carrot" , "hot dog" , "pizza" ,
|
||||
"donut" , "cake" , "chair" , "sofa" , "pottedplant" , "bed" , "diningtable" ,
|
||||
"toilet" , "tvmonitor" , "laptop" , "mouse" , "remote" , "keyboard" ,
|
||||
"cell phone" , "microwave" , "oven" , "toaster" , "sink" , "refrigerator" ,
|
||||
"book" , "clock" , "vase" , "scissors" , "teddy bear" , "hair drier" , "toothbrush"};
|
||||
classesNames = std::vector<std::string>(classes_names_, std::end(classes_names_));
|
||||
|
||||
}
|
||||
else{
|
||||
FatalError("Number of classes not supported for mobilenet");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void MobilenetDetection::preprocess(cv::Mat &frame, const int bi){
|
||||
#ifdef OPENCV_CUDACONTRIB
|
||||
//move original image on GPU
|
||||
cv::cuda::GpuMat orig_img, frame_nomean;
|
||||
orig_img = cv::cuda::GpuMat(frame);
|
||||
|
||||
//resize image, remove mean, divide by std
|
||||
cv::cuda::resize (orig_img, orig_img, cv::Size(netRT->input_dim.w, netRT->input_dim.h));
|
||||
orig_img.convertTo(frame_nomean, CV_32FC3, 1, -127);
|
||||
frame_nomean.convertTo(imagePreproc, CV_32FC3, 1 / 128.0, 0);
|
||||
|
||||
//copy image into tensors
|
||||
cv::cuda::split(imagePreproc, bgr);
|
||||
|
||||
for(int i=0; i < netRT->input_dim.c; i++){
|
||||
int idx = i * imagePreproc.rows * imagePreproc.cols;
|
||||
checkCuda( cudaMemcpy((void *)&input_d[idx + netRT->input_dim.tot()*bi], (void *)bgr[i].data, imagePreproc.rows * imagePreproc.cols* sizeof(float), cudaMemcpyDeviceToDevice) );
|
||||
}
|
||||
#else
|
||||
//resize image, remove mean, divide by std
|
||||
cv::Mat frame_nomean;
|
||||
resize(frame, frame, cv::Size(netRT->input_dim.w, netRT->input_dim.h));
|
||||
frame.convertTo(frame_nomean, CV_32FC3, 1, -127);
|
||||
frame_nomean.convertTo(imagePreproc, CV_32FC3, 1 / 128.0, 0);
|
||||
|
||||
//copy image into tensor and copy it into GPU
|
||||
cv::split(imagePreproc, bgr);
|
||||
for (int i = 0; i < netRT->input_dim.c; i++){
|
||||
int idx = i * imagePreproc.rows * imagePreproc.cols;
|
||||
memcpy((void *)&input[idx + netRT->input_dim.tot()*bi], (void *)bgr[i].data, imagePreproc.rows * imagePreproc.cols * sizeof(dnnType));
|
||||
}
|
||||
checkCuda(cudaMemcpyAsync(input_d+ netRT->input_dim.tot()*bi, input + netRT->input_dim.tot()*bi, netRT->input_dim.tot() * sizeof(dnnType), cudaMemcpyHostToDevice, netRT->stream));
|
||||
#endif
|
||||
}
|
||||
|
||||
void MobilenetDetection::postprocess(const int bi, const bool mAP){
|
||||
//get confidences and locations_h
|
||||
dnnType *rt_out[2];
|
||||
rt_out[0] = (dnnType *)netRT->buffersRT[3]+ netRT->buffersDIM[3].tot()*bi;
|
||||
rt_out[1] = (dnnType *)netRT->buffersRT[4]+ netRT->buffersDIM[4].tot()*bi;
|
||||
|
||||
detected.clear();
|
||||
|
||||
checkCuda(cudaMemcpy(confidences_h, rt_out[0], nPriors * classes * sizeof(float), cudaMemcpyDeviceToHost));
|
||||
checkCuda(cudaMemcpy(locations_h, rt_out[1], N_COORDS * nPriors * sizeof(float), cudaMemcpyDeviceToHost));
|
||||
convert_locatios_to_boxes_and_center();
|
||||
|
||||
int width = originalSize[bi].width;
|
||||
int height = originalSize[bi].height;
|
||||
|
||||
float *conf_per_class;
|
||||
for (int i = 1; i < classes; i++){
|
||||
conf_per_class = &confidences_h[i * nPriors];
|
||||
std::vector<tk::dnn::box> boxes;
|
||||
for (int j = 0; j < nPriors; j++){
|
||||
|
||||
if (conf_per_class[j] > confThreshold){
|
||||
tk::dnn::box b;
|
||||
b.cl = i;
|
||||
b.prob = conf_per_class[j];
|
||||
b.x = locations_h[j * N_COORDS + 0];
|
||||
b.y = locations_h[j * N_COORDS + 1];
|
||||
b.w = locations_h[j * N_COORDS + 2];
|
||||
b.h = locations_h[j * N_COORDS + 3];
|
||||
|
||||
if(mAP)
|
||||
for(int c=1; c<classes; c++)
|
||||
b.probs.push_back(confidences_h[c * nPriors + j]);
|
||||
|
||||
boxes.push_back(b);
|
||||
}
|
||||
}
|
||||
std::sort(boxes.begin(), boxes.end(), boxProbCmp);
|
||||
|
||||
std::vector<tk::dnn::box> remaining;
|
||||
while (boxes.size() > 0){
|
||||
remaining.clear();
|
||||
|
||||
tk::dnn::box b;
|
||||
b.cl = boxes[0].cl -1 ; //remove background class
|
||||
b.prob = boxes[0].prob;
|
||||
b.x = boxes[0].x * width;
|
||||
b.y = boxes[0].y * height;
|
||||
b.w = boxes[0].w * width - b.x; //convert from x1 to width
|
||||
b.h = boxes[0].h * height - b.y; //convert from y1 to height
|
||||
detected.push_back(b);
|
||||
for (size_t j = 1; j < boxes.size(); j++){
|
||||
if (iou(boxes[0], boxes[j]) <= IoUThreshold){
|
||||
remaining.push_back(boxes[j]);
|
||||
}
|
||||
}
|
||||
boxes = remaining;
|
||||
}
|
||||
}
|
||||
batchDetected.push_back(detected);
|
||||
}
|
||||
|
||||
|
||||
} // namespace dnn
|
||||
} // namespace tk
|
||||
+1
-1
@@ -12,7 +12,7 @@ MulAdd::MulAdd(Network *net, dnnType mul, dnnType add) : Layer(net) {
|
||||
|
||||
int size = input_dim.tot();
|
||||
|
||||
// create a vector with all value setted to add
|
||||
// create a vector with all value set to add
|
||||
dnnType *add_vector_h = new dnnType[size];
|
||||
for(int i=0; i<size; i++)
|
||||
add_vector_h[i] = add;
|
||||
|
||||
+96
-7
@@ -17,23 +17,40 @@ Network::Network(dataDim_t input_dim) {
|
||||
<<", CUDNN v"<<cu_ver<<")\n";
|
||||
dataType = CUDNN_DATA_FLOAT;
|
||||
tensorFormat = CUDNN_TENSOR_NCHW;
|
||||
dontLoadWeights = false;
|
||||
num_layers = 0;
|
||||
|
||||
fp16 = false;
|
||||
dla = false;
|
||||
int8 = false;
|
||||
if(const char* env_p = std::getenv("TKDNN_MODE")) {
|
||||
if(strcmp(env_p, "FP16") == 0)
|
||||
fp16 = true;
|
||||
else if(strcmp(env_p, "DLA") == 0) {
|
||||
dla = true;
|
||||
fp16 = true;
|
||||
}
|
||||
else if(strcmp(env_p, "DLA") == 0) {
|
||||
dla = true;
|
||||
fp16 = true;
|
||||
}
|
||||
else if(strcmp(env_p, "INT8") == 0) {
|
||||
int8 = true;
|
||||
}
|
||||
}
|
||||
maxBatchSize = 1;
|
||||
if(const char* env_p = std::getenv("TKDNN_BATCHSIZE")) {
|
||||
maxBatchSize = atoi(env_p);
|
||||
}
|
||||
if(const char* env_p = std::getenv("TKDNN_CALIB_IMG_PATH"))
|
||||
fileImgList = env_p;
|
||||
|
||||
if(const char* env_p = std::getenv("TKDNN_CALIB_LABEL_PATH"))
|
||||
fileLabelList = env_p;
|
||||
|
||||
|
||||
if(fp16)
|
||||
std::cout<<COL_REDB<<"!! FP16 INERENCE ENABLED !!"<<COL_END<<"\n";
|
||||
std::cout<<COL_REDB<<"!! FP16 INFERENCE ENABLED !!"<<COL_END<<"\n";
|
||||
if(dla)
|
||||
std::cout<<COL_GREENB<<"!! DLA INERENCE ENABLED !!"<<COL_END<<"\n";
|
||||
std::cout<<COL_GREENB<<"!! DLA INFERENCE ENABLED !!"<<COL_END<<"\n";
|
||||
if(int8)
|
||||
std::cout<<COL_ORANGEB<<"!! INT8 INFERENCE ENABLED !!"<<COL_END<<"\n";
|
||||
|
||||
|
||||
checkCUDNN( cudnnCreate(&cudnnHandle) );
|
||||
@@ -42,11 +59,16 @@ Network::Network(dataDim_t input_dim) {
|
||||
}
|
||||
|
||||
Network::~Network() {
|
||||
|
||||
checkCUDNN( cudnnDestroy(cudnnHandle) );
|
||||
checkERROR( cublasDestroy(cublasHandle) );
|
||||
}
|
||||
|
||||
void Network::releaseLayers() {
|
||||
for(int i=0; i<num_layers; i++)
|
||||
delete layers[i];
|
||||
num_layers = 0;
|
||||
}
|
||||
|
||||
dnnType* Network::infer(dataDim_t &dim, dnnType* data) {
|
||||
|
||||
//do infer for every layer
|
||||
@@ -61,6 +83,7 @@ bool Network::addLayer(Layer *l) {
|
||||
if(num_layers == MAX_LAYERS)
|
||||
return false;
|
||||
|
||||
l->id = num_layers;
|
||||
layers[num_layers++] = l;
|
||||
return true;
|
||||
}
|
||||
@@ -73,6 +96,28 @@ dataDim_t Network::getOutputDim() {
|
||||
return layers[num_layers-1]->output_dim;
|
||||
}
|
||||
|
||||
void Network::adjustFeatureMapSizeWithShortcuts(){
|
||||
layerType_t layer_type;
|
||||
int shortcutted_idx;
|
||||
|
||||
for(int i=0; i<num_layers; i++) {
|
||||
layer_type = layers[i]->getLayerType();
|
||||
if(layer_type == LAYER_SHORTCUT){
|
||||
shortcutted_idx = -1;
|
||||
for(int j=0; j<num_layers; j++) {
|
||||
if(static_cast<tk::dnn::Shortcut*>(layers[i])->backLayer == layers[j]){
|
||||
shortcutted_idx = j;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if(shortcutted_idx == -1)
|
||||
FatalError("Problem when computing featuer_map_size with shortcuts");
|
||||
for(int j=shortcutted_idx+1; j<i; ++j)
|
||||
layers[j]->feature_map_size += layers[shortcutted_idx]->output_dim.tot();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Network::print() {
|
||||
|
||||
printCenteredTitle(" NETWORK MODEL ", '=', 60);
|
||||
@@ -83,10 +128,21 @@ void Network::print() {
|
||||
std::cout.width(16); std::cout<<std::left<<"output (H*W,CH)";
|
||||
std::cout<<"\n";
|
||||
|
||||
adjustFeatureMapSizeWithShortcuts();
|
||||
|
||||
long long unsigned int tot_params = 0;
|
||||
long long unsigned int max_feature_map_size = 0;
|
||||
long long unsigned int tot_MACC = 0;
|
||||
|
||||
for(int i=0; i<num_layers; i++) {
|
||||
dataDim_t in = layers[i]->input_dim;
|
||||
dataDim_t out = layers[i]->output_dim;
|
||||
|
||||
tot_params += layers[i]->n_params;
|
||||
tot_MACC += layers[i]->MACC;
|
||||
if(layers[i]->feature_map_size> max_feature_map_size)
|
||||
max_feature_map_size = layers[i]->feature_map_size;
|
||||
|
||||
std::cout.width(3); std::cout<<std::right<<i;
|
||||
std::cout<<" ";
|
||||
std::cout.width(16); std::cout<<std::left<<layers[i]->getLayerName();
|
||||
@@ -105,6 +161,39 @@ void Network::print() {
|
||||
}
|
||||
printCenteredTitle("", '=', 60);
|
||||
std::cout<<"\n";
|
||||
std::cout<<"N params: "<<tot_params<<std::endl;
|
||||
std::cout<<"Max feature map size: "<<max_feature_map_size<<std::endl;
|
||||
std::cout<<"N MACC: "<<tot_MACC<<std::endl<<std::endl;
|
||||
printCudaMemUsage();
|
||||
}
|
||||
const char *Network::getNetworkRTName(const char *network_name){
|
||||
networkName = network_name;
|
||||
int network_name_len = strlen(network_name);
|
||||
char *RTName = (char *)malloc((network_name_len + 9)*sizeof(char));
|
||||
if (fp16){
|
||||
strcpy(RTName, network_name);
|
||||
strcat(RTName, "_fp16.rt");
|
||||
RTName[network_name_len + 8] = '\0';
|
||||
}
|
||||
else if (dla){
|
||||
strcpy(RTName, network_name);
|
||||
strcat(RTName, "_dla.rt");
|
||||
RTName[network_name_len + 7] = '\0';
|
||||
}
|
||||
|
||||
else if (int8){
|
||||
strcpy(RTName, network_name);
|
||||
strcat(RTName, "_int8.rt");
|
||||
RTName[network_name_len + 8] = '\0';
|
||||
}
|
||||
|
||||
else{
|
||||
strcpy(RTName, network_name);
|
||||
strcat(RTName, "_fp32.rt");
|
||||
RTName[network_name_len + 8] = '\0';
|
||||
}
|
||||
networkNameRT = RTName;
|
||||
return RTName;
|
||||
}
|
||||
|
||||
|
||||
|
||||
+713
-185
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,434 @@
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/highgui/highgui.hpp>
|
||||
#include <opencv2/videoio.hpp>
|
||||
#include <opencv2/imgproc/imgproc.hpp>
|
||||
#include "tkDNN/NetworkViz.h"
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
cv::Mat mapillary_15_map(cv::Mat adjMap){
|
||||
|
||||
// cv::imshow("test", adjMap);
|
||||
// cv::waitKey(0);
|
||||
cv::Mat M1(1, 256, CV_8UC1), M2(1, 256, CV_8UC1), M3(1, 256, CV_8UC1);
|
||||
|
||||
//animal
|
||||
M3.at<uchar>(0)=165;
|
||||
M2.at<uchar>(0)=42;
|
||||
M1.at<uchar>(0)=45;
|
||||
|
||||
//curb
|
||||
M3.at<uchar>(1)=196;
|
||||
M2.at<uchar>(1)=196;
|
||||
M1.at<uchar>(1)=196;
|
||||
|
||||
//barrier
|
||||
M3.at<uchar>(2)=90;
|
||||
M2.at<uchar>(2)=120;
|
||||
M1.at<uchar>(2)=150;
|
||||
|
||||
//road
|
||||
M3.at<uchar>(3)=128;
|
||||
M2.at<uchar>(3)=64;
|
||||
M1.at<uchar>(3)=128;
|
||||
|
||||
//building
|
||||
M3.at<uchar>(4)=70;
|
||||
M2.at<uchar>(4)=70;
|
||||
M1.at<uchar>(4)=70;
|
||||
|
||||
//person
|
||||
M3.at<uchar>(5)=220;
|
||||
M2.at<uchar>(5)=20;
|
||||
M1.at<uchar>(5)=60;
|
||||
|
||||
//roadmark
|
||||
M3.at<uchar>(6)=255;
|
||||
M2.at<uchar>(6)=255;
|
||||
M1.at<uchar>(6)=255;
|
||||
|
||||
//nature
|
||||
M3.at<uchar>(7)=107;
|
||||
M2.at<uchar>(7)=142;
|
||||
M1.at<uchar>(7)=35;
|
||||
|
||||
//sky
|
||||
M3.at<uchar>(8)=70;
|
||||
M2.at<uchar>(8)=130;
|
||||
M1.at<uchar>(8)=180;
|
||||
|
||||
//billboard
|
||||
M3.at<uchar>(9)=220;
|
||||
M2.at<uchar>(9)=220;
|
||||
M1.at<uchar>(9)=220;
|
||||
|
||||
//pole
|
||||
M3.at<uchar>(10)=153;
|
||||
M2.at<uchar>(10)=153;
|
||||
M1.at<uchar>(10)=153;
|
||||
|
||||
//traffic sign
|
||||
M3.at<uchar>(11)=128;
|
||||
M2.at<uchar>(11)=128;
|
||||
M1.at<uchar>(11)=128;
|
||||
|
||||
//bike
|
||||
M3.at<uchar>(12)=119;
|
||||
M2.at<uchar>(12)=11;
|
||||
M1.at<uchar>(12)=32;
|
||||
|
||||
//vehicle
|
||||
M3.at<uchar>(13)=0;
|
||||
M2.at<uchar>(13)=0;
|
||||
M1.at<uchar>(13)=142;
|
||||
|
||||
//void
|
||||
for(int i=14;i<256;i++)
|
||||
{
|
||||
M1.at<uchar>(i)=0;
|
||||
M2.at<uchar>(i)=0;
|
||||
M3.at<uchar>(i)=0;
|
||||
}
|
||||
|
||||
cv::Mat r1,r2,r3;
|
||||
|
||||
cv::LUT(adjMap,M1,r1);
|
||||
cv::LUT(adjMap,M2,r2);
|
||||
cv::LUT(adjMap,M3,r3);
|
||||
|
||||
std::vector<cv::Mat> planes;
|
||||
planes.push_back(r1);
|
||||
planes.push_back(r2);
|
||||
planes.push_back(r3);
|
||||
|
||||
cv::Mat dst;
|
||||
cv::merge(planes,dst);
|
||||
return dst;
|
||||
|
||||
|
||||
}
|
||||
|
||||
cv::Mat berkeley_20_map(cv::Mat adjMap){
|
||||
|
||||
cv::Mat M1(1, 256, CV_8UC1), M2(1, 256, CV_8UC1), M3(1, 256, CV_8UC1);
|
||||
|
||||
//road
|
||||
M3.at<uchar>(0)=128;
|
||||
M2.at<uchar>(0)=64;
|
||||
M1.at<uchar>(0)=128;
|
||||
|
||||
//sidewalk
|
||||
M3.at<uchar>(1)=244;
|
||||
M2.at<uchar>(1)=35;
|
||||
M1.at<uchar>(1)=232;
|
||||
|
||||
//building
|
||||
M3.at<uchar>(2)=70;
|
||||
M2.at<uchar>(2)=70;
|
||||
M1.at<uchar>(2)=70;
|
||||
|
||||
//wall
|
||||
M3.at<uchar>(3)=102;
|
||||
M2.at<uchar>(3)=102;
|
||||
M1.at<uchar>(3)=156;
|
||||
|
||||
//fence
|
||||
M3.at<uchar>(4)=90;
|
||||
M2.at<uchar>(4)=120;
|
||||
M1.at<uchar>(4)=150;
|
||||
|
||||
//pole
|
||||
M3.at<uchar>(5)=153;
|
||||
M2.at<uchar>(5)=153;
|
||||
M1.at<uchar>(5)=153;
|
||||
|
||||
//traffic light
|
||||
M3.at<uchar>(6)=250;
|
||||
M2.at<uchar>(6)=170;
|
||||
M1.at<uchar>(6)=30;
|
||||
|
||||
//traffic sign
|
||||
M3.at<uchar>(7)=128;
|
||||
M2.at<uchar>(7)=128;
|
||||
M1.at<uchar>(7)=128;
|
||||
|
||||
//nature
|
||||
M3.at<uchar>(8)=107;
|
||||
M2.at<uchar>(8)=142;
|
||||
M1.at<uchar>(8)=35;
|
||||
|
||||
//ground
|
||||
M3.at<uchar>(9)=0;
|
||||
M2.at<uchar>(9)=192;
|
||||
M1.at<uchar>(9)=0;
|
||||
|
||||
//sky
|
||||
M3.at<uchar>(10)=70;
|
||||
M2.at<uchar>(10)=130;
|
||||
M1.at<uchar>(10)=180;
|
||||
|
||||
//person
|
||||
M3.at<uchar>(11)=220;
|
||||
M2.at<uchar>(11)=20;
|
||||
M1.at<uchar>(11)=60;
|
||||
|
||||
//rider
|
||||
M3.at<uchar>(12)=255;
|
||||
M2.at<uchar>(12)=0;
|
||||
M1.at<uchar>(12)=100;
|
||||
|
||||
//car
|
||||
M3.at<uchar>(13)=0;
|
||||
M2.at<uchar>(13)=0;
|
||||
M1.at<uchar>(13)=142;
|
||||
|
||||
//truck
|
||||
M3.at<uchar>(14)=0;
|
||||
M2.at<uchar>(14)=0;
|
||||
M1.at<uchar>(14)=70;
|
||||
|
||||
//bus
|
||||
M3.at<uchar>(15)=0;
|
||||
M2.at<uchar>(15)=60;
|
||||
M1.at<uchar>(15)=100;
|
||||
|
||||
//train
|
||||
M3.at<uchar>(16)=0;
|
||||
M2.at<uchar>(16)=0;
|
||||
M1.at<uchar>(16)=192;
|
||||
|
||||
//motorbike
|
||||
M3.at<uchar>(17)=0;
|
||||
M2.at<uchar>(17)=0;
|
||||
M1.at<uchar>(17)=230;
|
||||
|
||||
//bike
|
||||
M3.at<uchar>(18)=119;
|
||||
M2.at<uchar>(18)=11;
|
||||
M1.at<uchar>(18)=32;
|
||||
|
||||
//void
|
||||
for(int i=19;i<256;i++)
|
||||
{
|
||||
M1.at<uchar>(i)=0;
|
||||
M2.at<uchar>(i)=0;
|
||||
M3.at<uchar>(i)=0;
|
||||
}
|
||||
|
||||
cv::Mat r1,r2,r3;
|
||||
|
||||
cv::LUT(adjMap,M1,r1);
|
||||
cv::LUT(adjMap,M2,r2);
|
||||
cv::LUT(adjMap,M3,r3);
|
||||
|
||||
std::vector<cv::Mat> planes;
|
||||
planes.push_back(r1);
|
||||
planes.push_back(r2);
|
||||
planes.push_back(r3);
|
||||
|
||||
cv::Mat dst;
|
||||
cv::merge(planes,dst);
|
||||
return dst;
|
||||
|
||||
}
|
||||
|
||||
cv::Mat cityscapes_19_map(cv::Mat adjMap){
|
||||
|
||||
cv::Mat M1(1, 256, CV_8UC1), M2(1, 256, CV_8UC1), M3(1, 256, CV_8UC1);
|
||||
|
||||
//road
|
||||
M3.at<uchar>(0)=128;
|
||||
M2.at<uchar>(0)=64;
|
||||
M1.at<uchar>(0)=128;
|
||||
|
||||
//sidewalk
|
||||
M3.at<uchar>(1)=244;
|
||||
M2.at<uchar>(1)=35;
|
||||
M1.at<uchar>(1)=232;
|
||||
|
||||
//building
|
||||
M3.at<uchar>(2)=70;
|
||||
M2.at<uchar>(2)=70;
|
||||
M1.at<uchar>(2)=70;
|
||||
|
||||
//wall
|
||||
M3.at<uchar>(3)=102;
|
||||
M2.at<uchar>(3)=102;
|
||||
M1.at<uchar>(3)=156;
|
||||
|
||||
//fence
|
||||
M3.at<uchar>(4)=190;
|
||||
M2.at<uchar>(4)=153;
|
||||
M1.at<uchar>(4)=153;
|
||||
|
||||
//pole
|
||||
M3.at<uchar>(5)=153;
|
||||
M2.at<uchar>(5)=153;
|
||||
M1.at<uchar>(5)=153;
|
||||
|
||||
//traffic light
|
||||
M3.at<uchar>(6)=250;
|
||||
M2.at<uchar>(6)=170;
|
||||
M1.at<uchar>(6)=30;
|
||||
|
||||
//traffic sign
|
||||
M3.at<uchar>(7)=220;
|
||||
M2.at<uchar>(7)=220;
|
||||
M1.at<uchar>(7)=0;
|
||||
|
||||
//vegetation
|
||||
M3.at<uchar>(8)=107;
|
||||
M2.at<uchar>(8)=142;
|
||||
M1.at<uchar>(8)=35;
|
||||
|
||||
//terrain
|
||||
M3.at<uchar>(9)=152;
|
||||
M2.at<uchar>(9)=251;
|
||||
M1.at<uchar>(9)=152;
|
||||
|
||||
//sky
|
||||
M3.at<uchar>(10)=70;
|
||||
M2.at<uchar>(10)=130;
|
||||
M1.at<uchar>(10)=180;
|
||||
|
||||
//person
|
||||
M3.at<uchar>(11)=220;
|
||||
M2.at<uchar>(11)=20;
|
||||
M1.at<uchar>(11)=60;
|
||||
|
||||
//rider
|
||||
M3.at<uchar>(12)=255;
|
||||
M2.at<uchar>(12)=0;
|
||||
M1.at<uchar>(12)=0;
|
||||
|
||||
//car
|
||||
M3.at<uchar>(13)=0;
|
||||
M2.at<uchar>(13)=0;
|
||||
M1.at<uchar>(13)=142;
|
||||
|
||||
//truck
|
||||
M3.at<uchar>(14)=0;
|
||||
M2.at<uchar>(14)=0;
|
||||
M1.at<uchar>(14)=70;
|
||||
|
||||
//bus
|
||||
M3.at<uchar>(15)=0;
|
||||
M2.at<uchar>(15)=60;
|
||||
M1.at<uchar>(15)=100;
|
||||
|
||||
//train
|
||||
M3.at<uchar>(16)=0;
|
||||
M2.at<uchar>(16)=80;
|
||||
M1.at<uchar>(16)=100;
|
||||
|
||||
//motorcycle
|
||||
M3.at<uchar>(17)=0;
|
||||
M2.at<uchar>(17)=0;
|
||||
M1.at<uchar>(17)=230;
|
||||
|
||||
//bicycle
|
||||
M3.at<uchar>(18)=119;
|
||||
M2.at<uchar>(18)=11;
|
||||
M1.at<uchar>(18)=32;
|
||||
|
||||
//void
|
||||
for(int i=19;i<256;i++)
|
||||
{
|
||||
M1.at<uchar>(i)=0;
|
||||
M2.at<uchar>(i)=0;
|
||||
M3.at<uchar>(i)=0;
|
||||
}
|
||||
|
||||
cv::Mat r1,r2,r3;
|
||||
|
||||
cv::LUT(adjMap,M1,r1);
|
||||
cv::LUT(adjMap,M2,r2);
|
||||
cv::LUT(adjMap,M3,r3);
|
||||
|
||||
std::vector<cv::Mat> planes;
|
||||
planes.push_back(r1);
|
||||
planes.push_back(r2);
|
||||
planes.push_back(r3);
|
||||
|
||||
cv::Mat dst;
|
||||
cv::merge(planes,dst);
|
||||
return dst;
|
||||
|
||||
}
|
||||
|
||||
|
||||
cv::Mat vizFloat2colorMap(cv::Mat map,double min, double max, int classes) {
|
||||
|
||||
if(min == 0 && max == 0)
|
||||
cv::minMaxIdx(map, &min, &max);
|
||||
|
||||
cv::Mat adjMap;
|
||||
cv::Mat falseColorsMap;
|
||||
|
||||
switch (classes)
|
||||
{
|
||||
case 15:
|
||||
map.convertTo(adjMap,CV_8UC1);
|
||||
falseColorsMap = mapillary_15_map(adjMap);
|
||||
break;
|
||||
case 20:
|
||||
map.convertTo(adjMap,CV_8UC1);
|
||||
falseColorsMap = berkeley_20_map(adjMap);
|
||||
break;
|
||||
case 19:
|
||||
map.convertTo(adjMap,CV_8UC1);
|
||||
falseColorsMap = cityscapes_19_map(adjMap);
|
||||
break;
|
||||
|
||||
default:
|
||||
// expand your range to 0..255. Similar to histEq();
|
||||
map.convertTo(adjMap,CV_8UC1, 255 / (max-min), -min);
|
||||
applyColorMap(adjMap, falseColorsMap, cv::COLORMAP_PARULA);
|
||||
}
|
||||
return falseColorsMap;
|
||||
}
|
||||
|
||||
cv::Mat vizData2Mat(dnnType *dataInput, tk::dnn::dataDim_t dim, int img_h, int img_w, double min, double max, int classes) {
|
||||
dnnType *data = nullptr;
|
||||
|
||||
// copy to CPU
|
||||
if(isCudaPointer(dataInput)) {
|
||||
data = new dnnType[dim.tot()];
|
||||
checkCuda( cudaMemcpy(data, dataInput, dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToHost) );
|
||||
} else {
|
||||
data = dataInput;
|
||||
}
|
||||
|
||||
int gridDim = ceil(sqrt(dim.c));
|
||||
cv::Size gridSize(dim.w*gridDim, dim.h*gridDim);
|
||||
cv::Mat grid = cv::Mat(gridSize, CV_8UC3, cv::Scalar(0));
|
||||
|
||||
for(int i=0; i<dim.c;i++) {
|
||||
cv::Mat raw = vizFloat2colorMap(cv::Mat(cv::Size(dim.w, dim.h),CV_32FC1, data + dim.w*dim.h*i), min, max, classes);
|
||||
int r = i / gridDim;
|
||||
int c = i - r * gridDim;
|
||||
raw.copyTo(grid.rowRange(r*dim.h, r*dim.h + dim.h).colRange(c*dim.w, c*dim.w + dim.w));
|
||||
}
|
||||
|
||||
cv::Size vdim(img_w, img_h);
|
||||
cv::Mat viz;
|
||||
cv::resize(grid, viz, vdim, 0, 0, 0);
|
||||
|
||||
// free memory
|
||||
if(isCudaPointer(dataInput)) {
|
||||
delete [] data;
|
||||
}
|
||||
return viz;
|
||||
}
|
||||
|
||||
cv::Mat vizLayer2Mat(tk::dnn::Network *net, int layer, int imgdim) {
|
||||
if(layer >= net->num_layers)
|
||||
FatalError("Could not viz layer\n");
|
||||
return vizData2Mat(net->layers[layer]->dstData, net->layers[layer]->output_dim, imgdim, imgdim);
|
||||
|
||||
//cv::imwrite("viz/layer" + std::to_string(layer) + ".png", viz);
|
||||
//cv::imshow("layer", viz);
|
||||
//cv::waitKey(0);
|
||||
}
|
||||
|
||||
}}
|
||||
@@ -0,0 +1,45 @@
|
||||
//
|
||||
// Created by perseusdg on 03/01/22.
|
||||
//
|
||||
|
||||
#include <iostream>
|
||||
#include "Layer.h"
|
||||
#include "kernels.h"
|
||||
|
||||
namespace tk{ namespace dnn {
|
||||
Padding::Padding(Network *net, int32_t pad_h, int32_t pad_w, tkdnnPaddingMode_t padding_mode,float constant) : Layer(net) {
|
||||
this->paddingH = pad_h;
|
||||
this->paddingW = pad_w;
|
||||
this->padding_mode = padding_mode;
|
||||
output_dim.c = input_dim.c;
|
||||
output_dim.n = input_dim.n;
|
||||
output_dim.h = input_dim.h + 2 * (this->paddingH);
|
||||
output_dim.w = input_dim.w + 2 * (this->paddingW);
|
||||
if(padding_mode == tkdnnPaddingMode_t::PADDING_MODE_CONSTANT){
|
||||
this->constant = constant;
|
||||
}else{
|
||||
this->constant = 0;
|
||||
}
|
||||
checkCuda(cudaMalloc(&dstData,output_dim.tot()*sizeof(dnnType)));
|
||||
}
|
||||
|
||||
Padding::~Padding() {
|
||||
checkCuda(cudaFree(dstData));
|
||||
}
|
||||
dnnType* Padding::infer(dataDim_t &dim, float *srcData) {
|
||||
fill(dstData,output_dim.tot(),0.0);
|
||||
if(padding_mode == tkdnnPaddingMode_t::PADDING_MODE_REFLECTION)
|
||||
{
|
||||
reflection_pad2d_out_forward(paddingH, paddingW, srcData, dstData, input_dim.h, input_dim.w, input_dim.c,
|
||||
input_dim.n);
|
||||
}
|
||||
else if(padding_mode == tkdnnPaddingMode_t::PADDING_MODE_CONSTANT){
|
||||
constant_pad2d_forward(srcData,dstData,input_dim.h,input_dim.w,output_dim.h,output_dim.w,input_dim.c,
|
||||
input_dim.n,paddingH,paddingW,constant);
|
||||
}
|
||||
|
||||
dim = output_dim;
|
||||
return dstData;
|
||||
}
|
||||
|
||||
}}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user