diff --git a/CMakeLists.txt b/CMakeLists.txt
index 0a390c6..b959c19 100644
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -3,10 +3,10 @@ cmake_minimum_required(VERSION 3.15)
project (tkDNN)
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${CMAKE_CURRENT_SOURCE_DIR}/cmake)
if(UNIX)
-set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++11 -fPIC -Wno-deprecated-declarations -Wno-unused-variable ")
+set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=c++14 -fPIC -Wno-deprecated-declarations")
endif()
if(WIN32)
-set(CMAKE_CXX_STANDARD 11)
+set(CMAKE_CXX_STANDARD 14)
set(CMAKE_CXX_FLAGS "/O2 /FS /EHsc")
set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON)
endif(WIN32)
@@ -140,6 +140,9 @@ target_link_libraries(test_shelfnet_berkeley tkDNN)
add_executable(test_shelfnet_mapillary tests/shelfnet/shelfnet_mapillary.cpp)
target_link_libraries(test_shelfnet_mapillary tkDNN)
+add_executable(test_shelfnet_coco tests/shelfnet/shelfnet_coco.cpp)
+target_link_libraries(test_shelfnet_coco tkDNN)
+
# DEMOS
add_executable(test_rtinference tests/test_rtinference/rtinference.cpp)
target_link_libraries(test_rtinference tkDNN)
diff --git a/README.md b/README.md
index b630fec..ddfd4ef 100644
--- a/README.md
+++ b/README.md
@@ -17,10 +17,15 @@ If you use tkDNN in your research, please cite the [following paper](https://iee
}
```
-### What's new (20 July 2021)
+### What's new
+#### 20 July 2021
- [x] Support to sematic segmentation [README](docs/README_seg.md)
- [x] Support 2D/3D Object Detection and Tracking [README](docs/README_2d3dtracking.md)
-- [ ] Support to TensorRT8 (WIP)
+#### 24 November 2021
+- [x] Support to sematic segmentation on cuda 11
+- [x] Support to TensorRT8 (thanks to [Harshvardhan Chandirasekar](https://github.com/perseusdg)).
+
+TensorRT8 (and therefore Jetpack 4.6) is currently supported only on the branch tensorrt8 due to [performance issue with TensorRT8](https://docs.nvidia.com/deeplearning/tensorrt/release-notes/tensorrt-8.html)). We will merge it to the master as soon as those issues are fixed (probably in future minor releases).
## FPS Results
Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimension as the input size, on
@@ -168,11 +173,10 @@ For specific details on how to run tkDNN on Windows 10 see [HERE](./docs/windows
| yolo4_320 | Yolov4 8 | [COCO 2017](http://cocodataset.org/) | 80 | 320x320 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
| yolo4_512 | Yolov4 8 | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
| yolo4_608 | Yolov4 8 | [COCO 2017](http://cocodataset.org/) | 80 | 608x608 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) |
-| yolo4_berkeley | Yolov4 8 | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 540x320 | [weights](https://cloud.hipert.unimore.it/s/nkWFa5fgb4NTdnB/download) |
+| yolo4_berkeley | Yolov4 8 | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 544x320 | [weights](https://cloud.hipert.unimore.it/s/nkWFa5fgb4NTdnB/download) |
| yolo4tiny | Yolov4 tiny 9 | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) |
-| yolo4x | Yolov4x-mish 9 | [COCO 2017](http://cocodataset.org/) |
+| yolo4x | Yolov4x-mish 9 | [COCO 2017](http://cocodataset.org/) | 80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) |
| yolo4tiny_512 | Yolov4 tiny 9 | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) |
-80 | 640x640 | [weights](https://cloud.hipert.unimore.it/s/5MFjtNtgbDGdJEo/download) |
| yolo4x-cps | Scaled Yolov4 10 | [COCO 2017](http://cocodataset.org/) | 80 | 512x512 | [weights](https://cloud.hipert.unimore.it/s/AfzHE4BfTeEm2gH/download) |
| shelfnet | ShelfNet18_realtime11 | [Cityscapes](https://www.cityscapes-dataset.com/) | 19 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/mEDZMRJaGCFWSJF/download) |
| shelfnet_berkeley | ShelfNet18_realtime11 | [DeepDrive](https://bdd-data.berkeley.edu/) | 20 | 1024x1024 | [weights](https://cloud.hipert.unimore.it/s/m92e7QdD9gYMF7f/download) |
diff --git a/demo/demo/demo.cpp b/demo/demo/demo.cpp
index 317a574..5445d86 100644
--- a/demo/demo/demo.cpp
+++ b/demo/demo/demo.cpp
@@ -9,7 +9,6 @@
#include "Yolo3Detection.h"
bool gRun;
-bool SAVE_RESULT = false;
void sig_handler(int signo) {
std::cout<<"request gateway stop\n";
@@ -18,43 +17,53 @@ void sig_handler(int signo) {
int main(int argc, char *argv[]) {
- std::cout<<"detection\n";
signal(SIGINT, sig_handler);
-
- std::string net = "yolo4tiny_fp32.rt";
- if(argc > 1)
- net = argv[1];
+ // get config file path and read it
#ifdef __linux__
- std::string input = "../demo/yolo_test.mp4";
+ std::string config_file = "../demo/demoConfig.yaml";
#elif _WIN32
- std::string input = "..\\..\\..\\demo\\yolo_test.mp4";
+ std::string config_file = "..\\..\\..\\demo\\demoConfig.yaml";
#endif
+ if(argc > 1)
+ config_file = argv[1];
+
+ YAML::Node conf = YAMLloadConf(config_file);
+ if(!conf)
+ FatalError("Problem with config file");
- if(argc > 2)
- input = argv[2];
- char ntype = 'y';
- if(argc > 3)
- ntype = argv[3][0];
- int n_classes = 80;
- if(argc > 4)
- n_classes = atoi(argv[4]);
- int n_batch = 1;
- if(argc > 5)
- n_batch = atoi(argv[5]);
- bool show = true;
- if(argc > 6)
- show = atoi(argv[6]);
- float conf_thresh=0.3;
- if(argc > 7)
- conf_thresh = atof(argv[7]);
+ // read settings from config file
+ std::string net = YAMLgetConf(conf, "net", "yolo4tiny_fp32.rt");
+ if(!fileExist(net.c_str()))
+ FatalError("The given network does not exist. Create the rt first.");
+ #ifdef __linux__
+ std::string input = YAMLgetConf(conf, "input", "../demo/yolo_test.mp4");
+ #elif _WIN32
+ std::string input = YAMLgetConf(conf, "win_input", "..\\..\\..\\demo\\yolo_test.mp4");
+ #endif
+ if(!fileExist(input.c_str()))
+ FatalError("The given input video does not exist.");
+
+ char ntype = YAMLgetConf(conf, "ntype", 'y');
+ int n_classes = YAMLgetConf(conf, "n_classes", 80);
+ int n_batch = YAMLgetConf(conf, "n_batch", 1);
if(n_batch < 1 || n_batch > 64)
FatalError("Batch dim not supported");
+ float conf_thresh = YAMLgetConf(conf, "conf_thresh", 0.3);
+ bool show = YAMLgetConf(conf, "show", true);
+ bool save = YAMLgetConf(conf, "save", false);
- if(!show)
- SAVE_RESULT = true;
-
+ std::cout <<"Net settings - net: "<< net
+ <<", ntype: "<< ntype
+ <<", n_classes: "<< n_classes
+ <<", n_batch: "<< n_batch
+ <<", conf_thresh: "<< conf_thresh<<"\n";
+ std::cout <<"Demo settings - input: "<< input
+ <<", show: "<< show
+ <<", save: "<< save<<"\n\n";
+
+ // create detection network
tk::dnn::Yolo3Detection yolo;
tk::dnn::CenternetDetection cnet;
tk::dnn::MobilenetDetection mbnet;
@@ -79,8 +88,7 @@ int main(int argc, char *argv[]) {
detNN->init(net, n_classes, n_batch, conf_thresh);
- gRun = true;
-
+ // open video stream
cv::VideoCapture cap(input);
if(!cap.isOpened())
gRun = false;
@@ -88,19 +96,21 @@ int main(int argc, char *argv[]) {
std::cout<<"camera started\n";
cv::VideoWriter resultVideo;
- if(SAVE_RESULT) {
+ if(save) {
int w = cap.get(cv::CAP_PROP_FRAME_WIDTH);
int h = cap.get(cv::CAP_PROP_FRAME_HEIGHT);
resultVideo.open("result.mp4", cv::VideoWriter::fourcc('M','P','4','V'), 30, cv::Size(w, h));
}
- cv::Mat frame;
if(show)
cv::namedWindow("detection", cv::WINDOW_NORMAL);
+ cv::Mat frame;
std::vector batch_frame;
std::vector batch_dnn_input;
+ // start detection loop
+ gRun = true;
while(gRun) {
batch_dnn_input.clear();
batch_frame.clear();
@@ -128,19 +138,18 @@ int main(int argc, char *argv[]) {
cv::waitKey(1);
}
}
- if(n_batch == 1 && SAVE_RESULT)
+ if(n_batch == 1 && save)
resultVideo << frame;
}
std::cout<<"detection end\n";
- double mean = 0;
+ double mean = 0;
std::cout<stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
+ std::cout<<"Min: "<<*std::min_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
std::cout<<"Max: "<<*std::max_element(detNN->stats.begin(), detNN->stats.end())/n_batch<<" ms\n";
for(int i=0; istats.size(); i++) mean += detNN->stats[i]; mean /= detNN->stats.size();
std::cout<<"Avg: "<
```
-In general the demo program takes 7 parameters:
-```
-./demo
-```
-where
+In general the demo program takes 1 parameter, the `````` that is the path to che configuration file. The parameter is optional and its default value is ```"../demo/demoConfig.yaml"```.
-* `````` is the rt file generated by a test
-* ```<``` is the path to a video file or a camera input
-* `````` is the type of network. Thee types are currently supported: ```y``` (YOLO family), ```c``` (CenterNet family) and ```m``` (MobileNet-SSD family)
-* ``````is the number of classes the network is trained on
-* `````` number of batches to use in inference (N.B. you should first export TKDNN_BATCHSIZE to the required n_batches and create again the rt file for the network).
-* `````` if set to 0 the demo will not show the visualization but save the video into result.mp4 (if n-batches ==1)
-* `````` confidence threshold for the detector. Only bounding boxes with threshold greater than conf-thresh will be displayed.
+The config file is a yaml file with the following attributes:
+* ```net``` is the rt file generated by a test
+* ```input``` is the path to a video file or a camera input (on Linux)
+* ```win_input``` is the path to a video file or a camera input (on Windows)
+* ```ntype``` is the type of network. Thee types are currently supported: ```y``` (YOLO family), ```c``` (CenterNet family) and ```m``` (MobileNet-SSD family)
+* ```n_classes``` is the number of classes the network is trained on
+* ```n_batch``` number of batches to use in inference (N.B. you should first export TKDNN_BATCHSIZE to the required n_batches and create again the rt file for the network).
+* ```conf_thresh``` confidence threshold for the detector. Only bounding boxes with threshold greater than conf-thresh will be displayed.
+* ```show``` if set to 0 the demo will not show the visualization (if n-batches ==1)
+* ```save``` if set to 1 the demo will save the video of the demo into result.mp4 (if n-batches ==1)
N.B. By default it is used FP32 inference
@@ -61,7 +60,8 @@ To run the demo with FP16 inference follow these steps (example with yolov3):
export TKDNN_MODE=FP16 # set the half floating point optimization
rm yolo3_fp16.rt # be sure to delete(or move) old tensorRT files
./test_yolo3 # run the yolo test (is slow)
-./demo yolo3_fp16.rt ../demo/yolo_test.mp4 y
+# set net: yolo3_fp16.rt in the config-file
+./demo
```
N.B. Using FP16 inference will lead to some errors in the results (first or second decimal).
@@ -86,7 +86,8 @@ export TKDNN_CALIB_LABEL_PATH=../demo/COCO_val2017/all_labels.txt
export TKDNN_CALIB_IMG_PATH=../demo/COCO_val2017/all_images.txt
rm yolo3_int8.rt # be sure to delete(or move) old tensorRT files
./test_yolo3 # run the yolo test (is slow)
-./demo yolo3_int8.rt ../demo/yolo_test.mp4 y
+# set net: yolo3_int8.rt in the config-file
+./demo
```
N.B.
diff --git a/include/tkDNN/SegmentationNN.h b/include/tkDNN/SegmentationNN.h
index b691cbc..b73ffa2 100644
--- a/include/tkDNN/SegmentationNN.h
+++ b/include/tkDNN/SegmentationNN.h
@@ -182,6 +182,7 @@ class SegmentationNN {
checkCuda(cudaMemcpyAsync(mean_d, mean.data(), mean.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream));
checkCuda(cudaMemcpyAsync(stddev_d, stddev.data(), stddev.size() * sizeof(float), cudaMemcpyHostToDevice, netRT->stream));
+ return true;
}
/**
diff --git a/include/tkDNN/utils.h b/include/tkDNN/utils.h
index 65375ba..4965a7a 100644
--- a/include/tkDNN/utils.h
+++ b/include/tkDNN/utils.h
@@ -21,6 +21,7 @@
#include
#include
+#include
#define dnnType float
@@ -137,4 +138,19 @@ static inline bool isCudaPointer(void *data) {
cudaPointerAttributes attr;
return cudaPointerGetAttributes(&attr, data) == 0;
}
+
+inline YAML::Node YAMLloadConf(const std::string& conf_file) {
+ std::cerr<<"Loading YAML: "<
+inline T YAMLgetConf(YAML::Node conf, std::string key, T defaultVal) {
+ T val = defaultVal;
+ if(conf && conf[key]) {
+ val = conf[key].as();
+ }
+ return val;
+}
+
#endif //UTILS_H
diff --git a/scripts/checkExecTimes.py b/scripts/checkExecTimes.py
new file mode 100644
index 0000000..c3e1b0b
--- /dev/null
+++ b/scripts/checkExecTimes.py
@@ -0,0 +1,37 @@
+import sys
+import pandas as pd
+
+if len(sys.argv) < 3:
+ print("Error: two csv files are needed, old first new second")
+ exit(1)
+
+old_perf_file = str(sys.argv[1])
+new_perf_file = str(sys.argv[2])
+
+verbose = False
+if len(sys.argv) == 4:
+ verbose = bool(sys.argv[3])
+
+print("Comparing {} vs {}".format(old_perf_file, new_perf_file))
+
+df_old = pd.read_csv (old_perf_file, sep=';', header=None, index_col=0)
+df_new = pd.read_csv (new_perf_file, sep=';', header=None, index_col=0)
+
+for index, row in df_new.iterrows():
+ if index in df_old.index:
+ if verbose:
+ print("New: ",row[1], row[2], row[3])
+ print("Old: ",df_old.loc[index][1], df_old.loc[index][2], df_old.loc[index][3])
+
+ print(index, end=': ')
+ if abs(row[1] - df_old.loc[index][1]) < df_old.loc[index][1]*0.1:
+ print("similar performance")
+ elif (row[1] < df_old.loc[index][1]):
+ print('\x1b[3;30;42m' + 'faster' + '\x1b[0m')
+ elif (row[1] > df_old.loc[index][1]):
+ if row[1] > df_old.loc[index][1] + df_old.loc[index][1] * 0.5 :
+ print('\x1b[3;30;41m' + 'WAY SLOWER' + '\x1b[0m')
+ else:
+ print('\x1b[3;30;41m' + 'slower' + '\x1b[0m')
+
+
diff --git a/src/CenterTrack.cpp b/src/CenterTrack.cpp
index dc15823..991ff69 100644
--- a/src/CenterTrack.cpp
+++ b/src/CenterTrack.cpp
@@ -17,6 +17,8 @@ bool CenterTrack::init(const std::string& tensor_path, const int n_classes, cons
init_pre_inf();
init_postprocessing();
init_visualization(n_classes);
+
+ return true;
}
bool CenterTrack::init_preprocessing(){
@@ -59,6 +61,8 @@ bool CenterTrack::init_preprocessing(){
checkCuda( cudaMalloc(&input_d, sizeof(dnnType)*netRT->input_dim.tot() * nBatches));
checkCuda( cudaMalloc(&input_pre_inf_d, sizeof(dnnType)*dim.tot()));
checkCuda( cudaMalloc(&d_ptrs, dim.tot() * sizeof(float)) );
+
+ return true;
}
bool CenterTrack::init_pre_inf(){
@@ -202,6 +206,8 @@ bool CenterTrack::init_postprocessing(){
trRes.resize(nBatches);
countTr.resize(nBatches, 0);
trackId.resize(nBatches, 0);
+
+ return true;
}
bool CenterTrack::init_visualization(const int n_classes){
@@ -274,6 +280,8 @@ bool CenterTrack::init_visualization(const int n_classes){
faceId.push_back({3,0,4,7});
faceId.push_back({2,3,7,6});
// ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]);
+
+ return true;
}
void CenterTrack::_get_additional_inputs(){
@@ -308,7 +316,7 @@ void CenterTrack::preprocess(cv::Mat &frame, const int bi){
}
float c[] = {new_width / 2.0f, new_height /2.0f};
- float s[] = {dim.w, dim.h};
+ float s[] = {float(dim.w), float(dim.h)};
// float s = new_width >= new_height ? new_width : new_height;
// ----------- get_affine_transform
// rot_rad = pi * 0 / 100 --> 0
diff --git a/src/CenternetDetection.cpp b/src/CenternetDetection.cpp
index 46757f4..dce1455 100644
--- a/src/CenternetDetection.cpp
+++ b/src/CenternetDetection.cpp
@@ -119,6 +119,7 @@ bool CenternetDetection::init(const std::string& tensor_path, const int n_classe
dst2.at(2,0)=dst2.at(1,0) + (-dst2.at(0,1)+dst2.at(1,1) );
dst2.at(2,1)=dst2.at(1,1) + (dst2.at(0,0)-dst2.at(1,0) );
+ return true;
}
diff --git a/src/CenternetDetection3D.cpp b/src/CenternetDetection3D.cpp
index 653624b..14f674d 100644
--- a/src/CenternetDetection3D.cpp
+++ b/src/CenternetDetection3D.cpp
@@ -167,6 +167,8 @@ bool CenternetDetection3D::init(const std::string& tensor_path, const int n_clas
faceId.push_back({2,3,7,6});
faceId.push_back({3,0,4,7});
// ([[0,1,5,4], [1,2,6, 5], [2,3,7,6], [3,0,4,7]]);
+
+ return true;
}
void CenternetDetection3D::preprocess(cv::Mat &frame, const int bi){
diff --git a/src/MobilenetDetection.cpp b/src/MobilenetDetection.cpp
index 3c54e28..13b836f 100644
--- a/src/MobilenetDetection.cpp
+++ b/src/MobilenetDetection.cpp
@@ -198,7 +198,7 @@ bool MobilenetDetection::init(const std::string& tensor_path, const int n_classe
"bottle" , "wine glass" , "cup" , "fork" , "knife" , "spoon" , "bowl" , "banana" ,
"apple" , "sandwich" , "orange" , "broccoli" , "carrot" , "hot dog" , "pizza" ,
"donut" , "cake" , "chair" , "sofa" , "pottedplant" , "bed" , "diningtable" ,
- "toilet" , "tvmonitor" , "laptop" , "mouse" , "remote" , "keyboard" ,
+ "toilet" , "tvmonitor" , "laptop" , "mouse" , "remote" , "keyboard" ,
"cell phone" , "microwave" , "oven" , "toaster" , "sink" , "refrigerator" ,
"book" , "clock" , "vase" , "scissors" , "teddy bear" , "hair drier" , "toothbrush"};
classesNames = std::vector(classes_names_, std::end(classes_names_));
@@ -207,7 +207,7 @@ bool MobilenetDetection::init(const std::string& tensor_path, const int n_classe
else{
FatalError("Number of classes not supported for mobilenet");
}
- return 1;
+ return true;
}
void MobilenetDetection::preprocess(cv::Mat &frame, const int bi){
diff --git a/tests/darknet/cfg/yolo4-csp_crowd.cfg b/tests/darknet/cfg/yolo4-csp_crowd.cfg
new file mode 100644
index 0000000..4c50f4d
--- /dev/null
+++ b/tests/darknet/cfg/yolo4-csp_crowd.cfg
@@ -0,0 +1,1279 @@
+[net]
+# Testing
+#batch=1
+#subdivisions=1
+# Training
+batch=64
+subdivisions=16
+width=512
+height=512
+channels=3
+momentum=0.949
+decay=0.0005
+angle=0
+saturation = 1.5
+exposure = 1.5
+hue=.1
+
+learning_rate=0.001
+burn_in=1000
+max_batches = 8000
+policy=steps
+steps=6400,7200
+scales=.1,.1
+
+mosaic=1
+
+letter_box=1
+
+ema_alpha=0.9998
+
+#optimized_memory=1
+
+#23:104x104 54:52x52 85:26x26 104:13x13 for 416
+
+
+
+[convolutional]
+batch_normalize=1
+filters=32
+size=3
+stride=1
+pad=1
+activation=mish
+
+# Downsample
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=3
+stride=2
+pad=1
+activation=mish
+
+#[convolutional]
+#batch_normalize=1
+#filters=64
+#size=1
+#stride=1
+#pad=1
+#activation=mish
+
+#[route]
+#layers = -2
+
+#[convolutional]
+#batch_normalize=1
+#filters=64
+#size=1
+#stride=1
+#pad=1
+#activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=32
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+#[convolutional]
+#batch_normalize=1
+#filters=64
+#size=1
+#stride=1
+#pad=1
+#activation=mish
+
+#[route]
+#layers = -1,-7
+
+#[convolutional]
+#batch_normalize=1
+#filters=64
+#size=1
+#stride=1
+#pad=1
+#activation=mish
+
+# Downsample
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=2
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=64
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -1,-10
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+# Downsample
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=2
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -1,-28
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+# Downsample
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=3
+stride=2
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -1,-28
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+# Downsample
+
+[convolutional]
+batch_normalize=1
+filters=1024
+size=3
+stride=2
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=3
+stride=1
+pad=1
+activation=mish
+
+[shortcut]
+from=-3
+activation=linear
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -1,-16
+
+[convolutional]
+batch_normalize=1
+filters=1024
+size=1
+stride=1
+pad=1
+activation=mish
+
+##########################
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=512
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+### SPP ###
+[maxpool]
+stride=1
+size=5
+
+[route]
+layers=-2
+
+[maxpool]
+stride=1
+size=9
+
+[route]
+layers=-4
+
+[maxpool]
+stride=1
+size=13
+
+[route]
+layers=-1,-3,-5,-6
+### End SPP ###
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=512
+activation=mish
+
+[route]
+layers = -1, -13
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[upsample]
+stride=2
+
+[route]
+layers = 79
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -1, -3
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=256
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=256
+activation=mish
+
+[route]
+layers = -1, -6
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[upsample]
+stride=2
+
+[route]
+layers = 48
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -1, -3
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=128
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=128
+activation=mish
+
+[route]
+layers = -1, -6
+
+[convolutional]
+batch_normalize=1
+filters=128
+size=1
+stride=1
+pad=1
+activation=mish
+
+##########################
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=256
+activation=mish
+
+[convolutional]
+size=1
+stride=1
+pad=1
+filters=21
+activation=logistic
+
+
+[yolo]
+mask = 0,1,2
+anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401
+classes=2
+num=9
+jitter=.1
+scale_x_y = 2.0
+objectness_smooth=0
+ignore_thresh = .7
+truth_thresh = 1
+#random=1
+resize=1.5
+iou_thresh=0.2
+iou_normalizer=0.05
+cls_normalizer=0.5
+obj_normalizer=4.0
+iou_loss=ciou
+nms_kind=diounms
+beta_nms=0.6
+new_coords=1
+max_delta=5
+
+[route]
+layers = -4
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=2
+pad=1
+filters=256
+activation=mish
+
+[route]
+layers = -1, -20
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=256
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=256
+activation=mish
+
+[route]
+layers = -1,-6
+
+[convolutional]
+batch_normalize=1
+filters=256
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=512
+activation=mish
+
+[convolutional]
+size=1
+stride=1
+pad=1
+filters=21
+activation=logistic
+
+
+[yolo]
+mask = 3,4,5
+anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401
+classes=2
+num=9
+jitter=.1
+scale_x_y = 2.0
+objectness_smooth=1
+ignore_thresh = .7
+truth_thresh = 1
+#random=1
+resize=1.5
+iou_thresh=0.2
+iou_normalizer=0.05
+cls_normalizer=0.5
+obj_normalizer=1.0
+iou_loss=ciou
+nms_kind=diounms
+beta_nms=0.6
+new_coords=1
+max_delta=5
+
+[route]
+layers = -4
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=2
+pad=1
+filters=512
+activation=mish
+
+[route]
+layers = -1, -49
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[route]
+layers = -2
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=512
+activation=mish
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=512
+activation=mish
+
+[route]
+layers = -1,-6
+
+[convolutional]
+batch_normalize=1
+filters=512
+size=1
+stride=1
+pad=1
+activation=mish
+
+[convolutional]
+batch_normalize=1
+size=3
+stride=1
+pad=1
+filters=1024
+activation=mish
+
+[convolutional]
+size=1
+stride=1
+pad=1
+filters=21
+activation=logistic
+
+
+[yolo]
+mask = 6,7,8
+anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401
+classes=2
+num=9
+jitter=.1
+scale_x_y = 2.0
+objectness_smooth=1
+ignore_thresh = .7
+truth_thresh = 1
+#random=1
+resize=1.5
+iou_thresh=0.2
+iou_normalizer=0.05
+cls_normalizer=0.5
+obj_normalizer=0.4
+iou_loss=ciou
+nms_kind=diounms
+beta_nms=0.6
+new_coords=1
+max_delta=2
diff --git a/tests/darknet/names/crowdhuman.names b/tests/darknet/names/crowdhuman.names
new file mode 100644
index 0000000..5ba1275
--- /dev/null
+++ b/tests/darknet/names/crowdhuman.names
@@ -0,0 +1,2 @@
+person
+head
\ No newline at end of file
diff --git a/tests/darknet/yolo4-csp_crowd.cpp b/tests/darknet/yolo4-csp_crowd.cpp
new file mode 100644
index 0000000..b8f0e6b
--- /dev/null
+++ b/tests/darknet/yolo4-csp_crowd.cpp
@@ -0,0 +1,34 @@
+#include
+#include
+#include "tkdnn.h"
+#include "test.h"
+#include "DarknetParser.h"
+
+int main() {
+ std::string bin_path = "yolo4-csp_crowd";
+ std::vector input_bins = {
+ bin_path + "/layers/input.bin"
+ };
+ std::vector output_bins = {
+ bin_path + "/debug/layer144_out.bin",
+ bin_path + "/debug/layer159_out.bin",
+ bin_path + "/debug/layer174_out.bin"
+ };
+ std::string wgs_path = bin_path + "/layers";
+ std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4-csp_crowd.cfg";
+ std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/crowdhuman.names";
+ downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/RKWfWNmWXfJigsK/download");
+
+ // parse darknet network
+ tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path);
+ net->print();
+
+ //convert network to tensorRT
+ tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str()));
+
+ int ret = testInference(input_bins, output_bins, net, netRT);
+ net->releaseLayers();
+ delete net;
+ delete netRT;
+ return ret;
+}
\ No newline at end of file
diff --git a/tests/shelfnet/shelfnet_coco.cpp b/tests/shelfnet/shelfnet_coco.cpp
new file mode 100644
index 0000000..276a09d
--- /dev/null
+++ b/tests/shelfnet/shelfnet_coco.cpp
@@ -0,0 +1,295 @@
+#include
+#include
+#include
+
+#include "tkdnn.h"
+#include "NetworkViz.h"
+
+
+const char *input_bin = "shelfnet_coco/debug/input.bin";
+
+const char *backbone[] = {
+ "shelfnet_coco/layers/backbone-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer1-0-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer1-0-conv2.bin",
+ "shelfnet_coco/layers/backbone-layer1-1-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer1-1-conv2.bin",
+ "shelfnet_coco/layers/backbone-layer2-0-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer2-0-conv2.bin",
+ "shelfnet_coco/layers/backbone-layer2-0-downsample-0.bin",
+ "shelfnet_coco/layers/backbone-layer2-1-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer2-1-conv2.bin",
+ "shelfnet_coco/layers/backbone-layer3-0-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer3-0-conv2.bin",
+ "shelfnet_coco/layers/backbone-layer3-0-downsample-0.bin",
+ "shelfnet_coco/layers/backbone-layer3-1-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer3-1-conv2.bin",
+ "shelfnet_coco/layers/backbone-layer4-0-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer4-0-conv2.bin",
+ "shelfnet_coco/layers/backbone-layer4-0-downsample-0.bin",
+ "shelfnet_coco/layers/backbone-layer4-1-conv1.bin",
+ "shelfnet_coco/layers/backbone-layer4-1-conv2.bin"};
+
+const char *conv_out[] = {
+ "shelfnet_coco/layers/conv_out-conv-conv.bin",
+ "shelfnet_coco/layers/conv_out-conv_out.bin",
+ "shelfnet_coco/layers/conv_out16-conv-conv.bin",
+ "shelfnet_coco/layers/conv_out16-conv_out.bin",
+ "shelfnet_coco/layers/conv_out32-conv-conv.bin",
+ "shelfnet_coco/layers/conv_out32-conv_out.bin"
+ };
+
+const char *decoder[] = {
+ "shelfnet_coco/layers/decoder-bottom-conv1.bin",
+ "shelfnet_coco/layers/decoder-bottom-conv12.bin",
+ "shelfnet_coco/layers/decoder-up_conv_list-0-conv-conv.bin",
+ "shelfnet_coco/layers/decoder-up_conv_list-0-conv_atten.bin",
+ "shelfnet_coco/layers/decoder-up_dense_list-0-conv.bin",
+ "shelfnet_coco/layers/decoder-up_conv_list-1-conv-conv.bin",
+ "shelfnet_coco/layers/decoder-up_conv_list-1-conv_atten.bin",
+ "shelfnet_coco/layers/decoder-up_dense_list-1-conv.bin"
+ };
+
+
+const char *ladder[] = {
+ "shelfnet_coco/layers/ladder-inconv-conv1.bin",
+ "shelfnet_coco/layers/ladder-inconv-conv12.bin",
+ "shelfnet_coco/layers/ladder-down_module_list-0-conv1.bin",
+ "shelfnet_coco/layers/ladder-down_module_list-0-conv12.bin",
+ "shelfnet_coco/layers/ladder-down_conv_list-0.bin",
+
+ "shelfnet_coco/layers/ladder-down_module_list-1-conv1.bin",
+ "shelfnet_coco/layers/ladder-down_module_list-1-conv12.bin",
+ "shelfnet_coco/layers/ladder-down_conv_list-1.bin",
+
+ "shelfnet_coco/layers/ladder-bottom-conv1.bin",
+ "shelfnet_coco/layers/ladder-bottom-conv12.bin",
+
+
+
+ "shelfnet_coco/layers/ladder-up_conv_list-0-conv-conv.bin",
+ "shelfnet_coco/layers/ladder-up_conv_list-0-conv_atten.bin",
+ "shelfnet_coco/layers/ladder-up_dense_list-0-conv.bin",
+
+
+ "shelfnet_coco/layers/ladder-up_conv_list-1-conv-conv.bin",
+ "shelfnet_coco/layers/ladder-up_conv_list-1-conv_atten.bin",
+ "shelfnet_coco/layers/ladder-up_dense_list-1-conv.bin"};
+
+const char *trans[] = {
+ "shelfnet_coco/layers/trans1-conv.bin",
+ "shelfnet_coco/layers/trans2-conv.bin",
+ "shelfnet_coco/layers/trans3-conv.bin"};
+int main()
+{
+
+ downloadWeightsifDoNotExist(input_bin, "shelfnet_coco", "https://cloud.hipert.unimore.it/s/KfQ9fGJQsgzNbiW/download");
+
+ int classes = 183;
+
+ // Network layout
+ tk::dnn::dataDim_t dim(1, 3, 1024, 1024, 1);
+ tk::dnn::Network net(dim);
+
+ int bi = 0, di = 0, li = 0, ci = 0;
+ new tk::dnn::Conv2d(&net, 64, 7, 7, 2, 2, 3, 3, backbone[bi++], true);
+ new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ tk::dnn::Layer* last = new tk::dnn::Pooling (&net, 3, 3, 2, 2, 1, 1, tk::dnn::POOLING_MAX);
+
+
+
+ for(int i=0; i<2; ++i){
+ new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true);
+ new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, backbone[bi++], true);
+ new tk::dnn::Shortcut(&net, last);
+ last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
+ }
+
+ std::vector features;
+ for(int i=0;i<3;++i){
+ int out_channel = pow(2,7+i);
+ std::cout< up_out;
+ //bottom
+ new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, decoder[di++], true, false, 1, true);
+ new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, decoder[di++], true, false, 1, true);
+ new tk::dnn::Shortcut(&net, last);
+ last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
+ up_out.push_back(last);
+
+ for(int i=0; i<2; ++i){
+ int out_channel = pow(2,7-i);
+ //up-conv
+ std::cout<output_dim.w, last->output_dim.h, last->output_dim.w, last->output_dim.h, 0, 0, tk::dnn::POOLING_AVERAGE);
+ new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, decoder[di++], true);
+
+ tk::dnn::Layer* act = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_SIGMOID);
+ new tk::dnn::Route(&net, &last, 1);
+ new tk::dnn::Shortcut(&net, act, true);
+
+ //interpolate
+ new tk::dnn::Resize(&net, 1,2,2);
+ new tk::dnn::Shortcut(&net, features[1-i]);
+
+ //up-dense
+ new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, decoder[di++], true);
+ last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ up_out.push_back(last);
+ }
+
+ //LADDER
+
+ std::vector down_out;
+ new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
+ new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
+ new tk::dnn::Shortcut(&net, last);
+ new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
+
+ for(int i=0; i<2;++i){
+ int out_channel = pow(2,6+i);
+ tk::dnn::Layer* l_last = new tk::dnn::Shortcut(&net, up_out[2-i]);
+
+ new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
+ new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
+ new tk::dnn::Shortcut(&net, l_last);
+ l_last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
+ down_out.push_back(l_last);
+
+ new tk::dnn::Conv2d (&net, out_channel*2, 3, 3, 2, 2, 1, 1, ladder[li++], false);
+ last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.0f); //should be ReLU
+ }
+
+ new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
+ new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ new tk::dnn::Conv2d (&net, 256, 3, 3, 1, 1, 1, 1, ladder[li++], true, false, 1, true);
+ new tk::dnn::Shortcut(&net, last);
+ last = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_RELU);
+ up_out.clear();
+ up_out.push_back(last);
+
+ for(int i=0; i<2; ++i){
+ int out_channel = pow(2,7-i);
+ //up-conv
+ new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true);
+ last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+
+ new tk::dnn::Pooling(&net, last->output_dim.w, last->output_dim.h, last->output_dim.w, last->output_dim.h, 0, 0, tk::dnn::POOLING_AVERAGE);
+ new tk::dnn::Conv2d (&net, out_channel, 1, 1, 1, 1, 0, 0, ladder[li++], true);
+
+ tk::dnn::Layer* act = new tk::dnn::Activation (&net, CUDNN_ACTIVATION_SIGMOID);
+ new tk::dnn::Route(&net, &last, 1);
+ new tk::dnn::Shortcut(&net, act, true);
+
+ //interpolate
+ new tk::dnn::Resize(&net, 1,2,2);
+ new tk::dnn::Shortcut(&net, down_out[1-i]);
+
+ // //up-dense
+ new tk::dnn::Conv2d (&net, out_channel, 3, 3, 1, 1, 1, 1, ladder[li++], true);
+ last = new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ up_out.push_back(last);
+ }
+
+
+ // for(int i=2;i>=0;--i){
+ // new tk::dnn::Route(&net, &up_out[i], 1);
+ new tk::dnn::Conv2d (&net, 64, 3, 3, 1, 1, 1, 1, conv_out[ci++], true);
+ new tk::dnn::Activation (&net, tk::dnn::ACTIVATION_LEAKY, 0.0f, 0.01);
+ new tk::dnn::Conv2d (&net, classes, 3, 3, 1, 1, 1, 1, conv_out[ci++], false);
+ /*up_out[i] =*/ new tk::dnn::Resize(&net, classes, net.input_dim.h, net.input_dim.w, true, tk::dnn::ResizeMode_t::LINEAR);
+ // // }
+
+ new tk::dnn::Softmax(&net);
+
+ const char *output_bin = "shelfnet_coco/debug/softmax.bin";
+
+ // Load input
+ dnnType *data;
+ dnnType *input_h;
+ readBinaryFile(input_bin, dim.tot(), &input_h, &data);
+ std::cout<<"Input:"<