From 3a0802d70c7a6286ac1ca871efc2498550295aa5 Mon Sep 17 00:00:00 2001 From: Francesco Gatti Date: Mon, 27 Jul 2020 13:45:42 +0200 Subject: [PATCH 01/12] Resolve detection objects pick by prob threshold. Before this it will only pick the last object with prob > thresh wich is absolutely wrong Now it picks all the objects with prob > thesh. fixes #94 --- src/Yolo3Detection.cpp | 47 +++++++++++++++++++++--------------------- 1 file changed, 24 insertions(+), 23 deletions(-) diff --git a/src/Yolo3Detection.cpp b/src/Yolo3Detection.cpp index c76af20..27f393f 100644 --- a/src/Yolo3Detection.cpp +++ b/src/Yolo3Detection.cpp @@ -113,34 +113,35 @@ void Yolo3Detection::postprocess(const int bi, const bool mAP){ int x1 = (b.x+b.w/2.); int y0 = (b.y-b.h/2.); int y1 = (b.y+b.h/2.); - int obj_class = -1; - float prob = 0; + for(int c=0; c= confThreshold) { - obj_class = c; - prob = dets[j].prob[c]; + int obj_class = c; + float prob = dets[j].prob[c]; + + // convert to image coords + x0 = x_ratio*x0; + x1 = x_ratio*x1; + y0 = y_ratio*y0; + y1 = y_ratio*y1; + + tk::dnn::box res; + res.cl = obj_class; + res.prob = prob; + res.x = x0; + res.y = y0; + res.w = x1 - x0; + res.h = y1 - y0; + + // FIXME: this shuld be useless + // if(mAP) + // for(int c=0; c= 0) { - // convert to image coords - x0 = x_ratio*x0; - x1 = x_ratio*x1; - y0 = y_ratio*y0; - y1 = y_ratio*y1; - - tk::dnn::box res; - res.cl = obj_class; - res.prob = prob; - res.x = x0; - res.y = y0; - res.w = x1 - x0; - res.h = y1 - y0; - if(mAP) - for(int c=0; c Date: Mon, 27 Jul 2020 13:52:39 +0200 Subject: [PATCH 02/12] fix coords convert --- src/Yolo3Detection.cpp | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/src/Yolo3Detection.cpp b/src/Yolo3Detection.cpp index 27f393f..606c6d1 100644 --- a/src/Yolo3Detection.cpp +++ b/src/Yolo3Detection.cpp @@ -114,17 +114,17 @@ void Yolo3Detection::postprocess(const int bi, const bool mAP){ int y0 = (b.y-b.h/2.); int y1 = (b.y+b.h/2.); + // convert to image coords + x0 = x_ratio*x0; + x1 = x_ratio*x1; + y0 = y_ratio*y0; + y1 = y_ratio*y1; + for(int c=0; c= confThreshold) { int obj_class = c; float prob = dets[j].prob[c]; - // convert to image coords - x0 = x_ratio*x0; - x1 = x_ratio*x1; - y0 = y_ratio*y0; - y1 = y_ratio*y1; - tk::dnn::box res; res.cl = obj_class; res.prob = prob; From f778e1aa998f894654b24c0ab9ad759c0eb14019 Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Wed, 5 Aug 2020 19:55:10 +0200 Subject: [PATCH 03/12] Fixed boxes to float, add conf thresh as param Signed-off-by: Micaela Verucchi --- README.md | 3 ++- demo/config.yaml | 2 +- demo/demo/demo.cpp | 5 ++++- demo/demo/map.cpp | 2 +- include/tkDNN/CenternetDetection.h | 2 +- include/tkDNN/DetectionNN.h | 2 +- include/tkDNN/MobilenetDetection.h | 2 +- include/tkDNN/Yolo3Detection.h | 2 +- src/CenternetDetection.cpp | 3 ++- src/MobilenetDetection.cpp | 3 ++- src/Yolo3Detection.cpp | 11 ++++++----- 11 files changed, 22 insertions(+), 15 deletions(-) diff --git a/README.md b/README.md index a1b5b16..d9b5755 100644 --- a/README.md +++ b/README.md @@ -193,7 +193,7 @@ Once you have succesfully created your rt file, run the demo: ``` ./demo yolo4_fp32.rt ../demo/yolo_test.mp4 y ``` -In general the demo program takes 6 parameters: +In general the demo program takes 7 parameters: ``` ./demo ``` @@ -204,6 +204,7 @@ where * ``````is the number of classes the network is trained on * `````` number of batches to use in inference (N.B. you should first export TKDNN_BATCHSIZE to the required n_batches and create again the rt file for the network). * `````` if set to 0 the demo will not show the visualization but save the video into result.mp4 (if n-batches ==1) +* `````` confidence threshold for the detector. Only bounding boxes with threshold greater than conf-thresh will be displayed. N.b. By default it is used FP32 inference diff --git a/demo/config.yaml b/demo/config.yaml index e6f91a7..31ac599 100644 --- a/demo/config.yaml +++ b/demo/config.yaml @@ -3,5 +3,5 @@ map_points : 101 #number of recall points (0 for all, 101 for COCO, 11 Pascal map_levels : 10 #number of IoU step for the AP map_step : 0.05 #step of IoU IoU_thresh : 0.5 #starting IoU threshold -conf_thresh : 0.0 #threshold on the condifence of the bbox +conf_thresh : 0.001 #threshold on the condifence of the bbox verbose : false #print on screen information diff --git a/demo/demo/demo.cpp b/demo/demo/demo.cpp index 76b451d..9f50d0b 100644 --- a/demo/demo/demo.cpp +++ b/demo/demo/demo.cpp @@ -40,6 +40,9 @@ int main(int argc, char *argv[]) { bool show = true; if(argc > 6) show = atoi(argv[6]); + float conf_thresh=0.3; + if(argc > 7) + conf_thresh = atof(argv[7]); if(n_batch < 1 || n_batch > 64) FatalError("Batch dim not supported"); @@ -69,7 +72,7 @@ int main(int argc, char *argv[]) { FatalError("Network type not allowed (3rd parameter)\n"); } - detNN->init(net, n_classes, n_batch); + detNN->init(net, n_classes, n_batch, conf_thresh); gRun = true; diff --git a/demo/demo/map.cpp b/demo/demo/map.cpp index d724db0..356e35a 100644 --- a/demo/demo/map.cpp +++ b/demo/demo/map.cpp @@ -105,7 +105,7 @@ int main(int argc, char *argv[]) default: FatalError("Network type not allowed (3rd parameter)\n"); } - detNN->init(net, n_classes); + detNN->init(net, n_classes, 1, conf_thresh); //read images std::ifstream all_labels(labels_path); diff --git a/include/tkDNN/CenternetDetection.h b/include/tkDNN/CenternetDetection.h index 227cb78..3c8cfbb 100644 --- a/include/tkDNN/CenternetDetection.h +++ b/include/tkDNN/CenternetDetection.h @@ -73,7 +73,7 @@ public: CenternetDetection() {}; ~CenternetDetection() {}; - bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1); + bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1, const float conf_thresh=0.3); void preprocess(cv::Mat &frame, const int bi=0); void postprocess(const int bi=0,const bool mAP=false); }; diff --git a/include/tkDNN/DetectionNN.h b/include/tkDNN/DetectionNN.h index 030cf8f..ba42834 100644 --- a/include/tkDNN/DetectionNN.h +++ b/include/tkDNN/DetectionNN.h @@ -84,7 +84,7 @@ class DetectionNN { * @param n_batches maximum number of batches to use in inference * @return true if everything is correct, false otherwise. */ - virtual bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1) = 0; + virtual bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1, const float conf_thresh=0.3) = 0; /** * This method performs the whole detection of the NN. diff --git a/include/tkDNN/MobilenetDetection.h b/include/tkDNN/MobilenetDetection.h index cabd7eb..9a5fedc 100644 --- a/include/tkDNN/MobilenetDetection.h +++ b/include/tkDNN/MobilenetDetection.h @@ -65,7 +65,7 @@ public: MobilenetDetection() {}; ~MobilenetDetection() {}; - bool init(const std::string& tensor_path, const int n_classes, const int n_batches=1); + bool init(const std::string& tensor_path, const int n_classes, const int n_batches=1, const float conf_thresh=0.3); void preprocess(cv::Mat &frame, const int bi=0); void postprocess(const int bi=0,const bool mAP=false); }; diff --git a/include/tkDNN/Yolo3Detection.h b/include/tkDNN/Yolo3Detection.h index 6d38514..100a720 100644 --- a/include/tkDNN/Yolo3Detection.h +++ b/include/tkDNN/Yolo3Detection.h @@ -24,7 +24,7 @@ public: Yolo3Detection() {}; ~Yolo3Detection() {}; - bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1); + bool init(const std::string& tensor_path, const int n_classes=80, const int n_batches=1, const float conf_thresh=0.3); void preprocess(cv::Mat &frame, const int bi=0); void postprocess(const int bi=0,const bool mAP=false); }; diff --git a/src/CenternetDetection.cpp b/src/CenternetDetection.cpp index 9d8df38..394e24a 100644 --- a/src/CenternetDetection.cpp +++ b/src/CenternetDetection.cpp @@ -3,11 +3,12 @@ namespace tk { namespace dnn { -bool CenternetDetection::init(const std::string& tensor_path, const int n_classes, const int n_batches){ +bool CenternetDetection::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh){ std::cout<<(tensor_path).c_str()<<"\n"; netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); classes = n_classes; nBatches = n_batches; + confThreshold = conf_thresh; dim = netRT->input_dim; diff --git a/src/MobilenetDetection.cpp b/src/MobilenetDetection.cpp index c905fea..3c54e28 100644 --- a/src/MobilenetDetection.cpp +++ b/src/MobilenetDetection.cpp @@ -126,12 +126,13 @@ float MobilenetDetection::iou(const tk::dnn::box &a, const tk::dnn::box &b){ return iou; } -bool MobilenetDetection::init(const std::string& tensor_path, const int n_classes, const int n_batches){ +bool MobilenetDetection::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh){ std::cout<<(tensor_path).c_str()<<"\n"; netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str()); imageSize = netRT->input_dim.h; classes = n_classes; nBatches = n_batches; + confThreshold = conf_thresh; SSDSpec specs[N_SSDSPEC]; diff --git a/src/Yolo3Detection.cpp b/src/Yolo3Detection.cpp index 606c6d1..e9b0064 100644 --- a/src/Yolo3Detection.cpp +++ b/src/Yolo3Detection.cpp @@ -3,13 +3,14 @@ namespace tk { namespace dnn { -bool Yolo3Detection::init(const std::string& tensor_path, const int n_classes, const int n_batches) { +bool Yolo3Detection::init(const std::string& tensor_path, const int n_classes, const int n_batches, const float conf_thresh) { //convert network to tensorRT std::cout<<(tensor_path).c_str()<<"\n"; netRT = new tk::dnn::NetworkRT(NULL, (tensor_path).c_str() ); nBatches = n_batches; + confThreshold = conf_thresh; tk::dnn::dataDim_t idim = netRT->input_dim; idim.n = nBatches; @@ -109,10 +110,10 @@ void Yolo3Detection::postprocess(const int bi, const bool mAP){ detected.clear(); for(int j=0; j Date: Thu, 6 Aug 2020 11:05:26 +0200 Subject: [PATCH 04/12] Update README.md --- README.md | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index d9b5755..ea55785 100644 --- a/README.md +++ b/README.md @@ -14,7 +14,7 @@ M. Verucchi, G. Brilli, D. Sapienza, M. Verasani, M. Arena, F. Gatti, A. Capoton "A Systematic Assessment of Embedded Neural Networks for Object Detection", in IEEE International Conference on Emerging Technologies and Factory Automation (2020) ``` -## Results +## FPS Results Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimesion as the input size, on * RTX 2080Ti (CUDA 10.2, TensorRT 7.0.0, Cudnn 7.6.5); * Xavier AGX, Jetpack 4.3 (CUDA 10.0, CUDNN 7.6.3, tensorrt 6.0.1 ); @@ -40,6 +40,20 @@ Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimesio | Nano | yolo4 512 | 2,32 | 2,34 | 3,02 | 3,04 | - | - | | Nano | yolo4 608 | 1,40 | 1,41 | 1,92 | 1,93 | - | - | +## MAP Results +Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001 + +| | CodaLab | CodaLab | CodaLab | CodaLab | tkDNN map | tkDNN map | +| -------------------- | :-----------: | :-------: | :-----------: | :---------: | :-----------: | :-------: | +| | **tkDNN** | **tkDNN** | **darknet** | **darknet** | **tkDNN** | **tkDNN** | +| | MAP(0.5:0.95) | AP50 | MAP(0.5:0.95) | AP50 | MAP(0.5:0.95) | AP50 | +| Yolov3 (416x416) | 0.381 | 0.675 | 0.380 | 0.675 | 0.372 | 0.663 | +| yolov4 (416x416) | 0.468 | 0.705 | 0.471 | 0.710 | 0.459 | 0.695 | +| yolov3tiny (416x416) | 0.096 | 0.202 | 0.096 | 0.201 | 0.093 | 0.198 | +| yolov4tiny (416x416) | 0.202 | 0.400 | 0.201 | 0.400 | 0.197 | 0.395 | +| Cnet-dla34 (512x512) | 0.366 | 0.543 | \- | \- | 0.361 | 0.535 | +| mv2SSD (512x512) | 0.226 | 0.381 | \- | \- | 0.223 | 0.378 | + ## Index - [tkDNN](#tkdnn) - [Index](#index) From df5443e017f9390b3f282a5c18ecf335ffafc5f8 Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Thu, 6 Aug 2020 15:53:57 +0200 Subject: [PATCH 05/12] Fix boxes also for Centernet Signed-off-by: Micaela Verucchi --- src/CenternetDetection.cpp | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/src/CenternetDetection.cpp b/src/CenternetDetection.cpp index 394e24a..46757f4 100644 --- a/src/CenternetDetection.cpp +++ b/src/CenternetDetection.cpp @@ -372,10 +372,10 @@ void CenternetDetection::postprocess(const int bi, const bool mAP){ // std::cout<<"th: "< Date: Fri, 11 Sep 2020 09:13:59 +0200 Subject: [PATCH 06/12] Fix typos (#107) Signed-off-by: micaela --- README.md | 14 +++++++------- include/tkDNN/DetectionNN.h | 8 ++++---- include/tkDNN/ImuOdom.h | 4 ++-- include/tkDNN/Layer.h | 20 ++++++++++---------- include/tkDNN/Network.h | 8 ++++---- include/tkDNN/NetworkRT.h | 2 +- include/tkDNN/evaluation.h | 8 ++++---- include/tkDNN/pluginsRT/DeformableConvRT.h | 2 +- include/tkDNN/test.h | 2 +- src/DarknetParser.cpp | 2 +- src/DeformConv2d.cpp | 2 +- src/Dense.cpp | 2 +- src/LSTM.cpp | 10 +++++----- src/LayerWgs.cpp | 2 +- src/MulAdd.cpp | 2 +- src/NetworkRT.cpp | 2 +- src/Region.cpp | 2 +- src/Shortcut.cpp | 2 +- src/evaluation.cpp | 6 +++--- 19 files changed, 50 insertions(+), 50 deletions(-) diff --git a/README.md b/README.md index ea55785..f17f75e 100644 --- a/README.md +++ b/README.md @@ -15,7 +15,7 @@ M. Verucchi, G. Brilli, D. Sapienza, M. Verasani, M. Arena, F. Gatti, A. Capoton ``` ## FPS Results -Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimesion as the input size, on +Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimension as the input size, on * RTX 2080Ti (CUDA 10.2, TensorRT 7.0.0, Cudnn 7.6.5); * Xavier AGX, Jetpack 4.3 (CUDA 10.0, CUDNN 7.6.3, tensorrt 6.0.1 ); * Tx2, Jetpack 4.2 (CUDA 10.0, CUDNN 7.3.1, tensorrt 5.0.6 ); @@ -169,7 +169,7 @@ tkDNN implement and easy parser for darknet cfg files, a network can be converte tk::dnn::Network *net = tk::dnn::darknetParser("yolov4.cfg", "yolov4/layers", "coco.names"); net->print(); ``` -All models from darknet are now parsed directly from cfg, you still need to export the weights with the descripted tools in the previus section. +All models from darknet are now parsed directly from cfg, you still need to export the weights with the described tools in the previous section.
Supported layers convolutional @@ -203,7 +203,7 @@ cmake .. -DDEBUG=True make ``` -Once you have succesfully created your rt file, run the demo: +Once you have successfully created your rt file, run the demo: ``` ./demo yolo4_fp32.rt ../demo/yolo_test.mp4 y ``` @@ -247,7 +247,7 @@ You should provide image_list.txt and label_list.txt, using training images. How ``` bash scripts/download_validation.sh COCO ``` -to automatically download COCO2017 validation (inside demo folder) and create those needed file. Use BDD insted of COCO to download BDD validation. +to automatically download COCO2017 validation (inside demo folder) and create those needed file. Use BDD instead of COCO to download BDD validation. Then a complete example using yolo3 and COCO dataset would be: ``` @@ -269,8 +269,8 @@ N.B. export TKDNN_BATCHSIZE=2 # build tensorRT files ``` -This will create a TensorRT file with the desidered **max** batch size. -The test will still run with a batch of 1, but the created tensorRT can manage the desidered batch size. +This will create a TensorRT file with the desired **max** batch size. +The test will still run with a batch of 1, but the created tensorRT can manage the desired batch size. ### Test batch Inference This will test the network with random input and check if the output of each batch is the same. @@ -316,7 +316,7 @@ cd build ./map_demo dla34_cnet_FP32.rt c ../demo/COCO_val2017/all_labels.txt ../demo/config.yaml ``` -This demo also creates a json file named ```net_name_COCO_res.json``` containing all the detections computed. The detections are in COCO format, the correct format to subit the results to [CodaLab COCO detection challenge](https://competitions.codalab.org/competitions/20794#participate). +This demo also creates a json file named ```net_name_COCO_res.json``` containing all the detections computed. The detections are in COCO format, the correct format to submit the results to [CodaLab COCO detection challenge](https://competitions.codalab.org/competitions/20794#participate). ## Existing tests and supported networks diff --git a/include/tkDNN/DetectionNN.h b/include/tkDNN/DetectionNN.h index ba42834..0498d41 100644 --- a/include/tkDNN/DetectionNN.h +++ b/include/tkDNN/DetectionNN.h @@ -76,10 +76,10 @@ class DetectionNN { ~DetectionNN(){}; /** - * Method used to inialize the class, allocate memory and compute + * Method used to initialize the class, allocate memory and compute * needed data. * - * @param tensor_path path to the rt file og the NN. + * @param tensor_path path to the rt file of the NN. * @param n_classes number of classes for the given dataset. * @param n_batches maximum number of batches to use in inference * @return true if everything is correct, false otherwise. @@ -141,9 +141,9 @@ class DetectionNN { } /** - * Method to draw boundixg boxes and labels on a frame. + * Method to draw bounding boxes and labels on a frame. * - * @param frames orginal frame to draw bounding box on. + * @param frames original frame to draw bounding box on. */ void draw(std::vector& frames) { tk::dnn::box b; diff --git a/include/tkDNN/ImuOdom.h b/include/tkDNN/ImuOdom.h index 58def96..6d8d4cb 100644 --- a/include/tkDNN/ImuOdom.h +++ b/include/tkDNN/ImuOdom.h @@ -44,7 +44,7 @@ class ImuOdom { virtual ~ImuOdom() {} /** - * Method used for inizialize the class + * Method used for initialize the class * * @return Success of the initialization */ @@ -141,7 +141,7 @@ class ImuOdom { //odomPOS = odomPOS + deltaP.cast(); // V2 odomROT = odomROT * q.normalized().toRotationMatrix(); - // compute euler + // compute Euler auto newEULER = odomROT.eulerAngles(0, 1, 2); for(int i=0; i<3; i++) { while( fabs(newEULER(i) - odomEULER(i)) > M_PI_2 ) { diff --git a/include/tkDNN/Layer.h b/include/tkDNN/Layer.h index f2ec56d..790a431 100644 --- a/include/tkDNN/Layer.h +++ b/include/tkDNN/Layer.h @@ -171,7 +171,7 @@ public: /** - Input layer (it doesnt need weigths) + Input layer (it doesn't need weights) */ class Input : public Layer { @@ -207,7 +207,7 @@ public: /** - Avaible activation functions + Available activation functions */ typedef enum { ACTIVATION_ELU = 100, @@ -216,7 +216,7 @@ typedef enum { } tkdnnActivationMode_t; /** - Activation layer (it doesnt need weigths) + Activation layer (it doesn't need weights) */ class Activation : public Layer { @@ -318,9 +318,9 @@ public: virtual dnnType* infer(dataDim_t &dim, dnnType* srcData); const bool bidirectional = true; /**> is the net bidir */ - bool returnSeq = false; /**> if false return only the result of last timestep */ + bool returnSeq = false; /**> if false return only the result of last timestamp */ int stateSize = 0; /**> number of hidden states */ - int seqLen = 0; /**> number of timesteps */ + int seqLen = 0; /**> number of timestamp */ int numLayers = 1; /**> number of internal layers */ protected: @@ -367,7 +367,7 @@ public: /** - Deformable Convolutionl 2d layer + Deformable Convolutional 2d layer */ class DeformConv2d : public LayerWgs { @@ -449,7 +449,7 @@ protected: /** - Avaible pooling functions (padding on tkDNN is not supported) + Available pooling functions (padding on tkDNN is not supported) */ typedef enum { POOLING_MAX = 0, @@ -460,7 +460,7 @@ typedef enum { /** Pooling layer - currenty supported only 2d pooing (also on 3d input) + currently supported only 2d pooing (also on 3d input) */ class Pooling : public Layer { @@ -526,7 +526,7 @@ public: /** Reorg layer - Mantain same dimension but change C*H*W distribution + Maintains same dimension but change C*H*W distribution */ class Reorg : public Layer { @@ -559,7 +559,7 @@ public: /** Upsample layer - Mantain same dimension but change C*H*W distribution + Maintains same dimension but change C*H*W distribution */ class Upsample : public Layer { diff --git a/include/tkDNN/Network.h b/include/tkDNN/Network.h index 2d95215..b78acff 100644 --- a/include/tkDNN/Network.h +++ b/include/tkDNN/Network.h @@ -7,12 +7,12 @@ namespace tk { namespace dnn { /** - Data rapresentation beetween layers + Data representation between layers n = batch size c = channels - h = heigth (lines) + h = height (lines) w = width (rows) - l = lenght (3rd dimension) + l = length (3rd dimension) */ struct dataDim_t { @@ -43,7 +43,7 @@ public: void releaseLayers(); /** - Do inferece for every added layer + Do inference for every added layer */ dnnType* infer(dataDim_t &dim, dnnType* data); diff --git a/include/tkDNN/NetworkRT.h b/include/tkDNN/NetworkRT.h index 66b4f3d..4c6c816 100644 --- a/include/tkDNN/NetworkRT.h +++ b/include/tkDNN/NetworkRT.h @@ -91,7 +91,7 @@ public: } /** - Do inferece + Do inference */ dnnType* infer(dataDim_t &dim, dnnType* data); void enqueue(int batchSize = 1); diff --git a/include/tkDNN/evaluation.h b/include/tkDNN/evaluation.h index 8907d9d..128eba0 100644 --- a/include/tkDNN/evaluation.h +++ b/include/tkDNN/evaluation.h @@ -73,12 +73,12 @@ double computeMap( std::vector &images,const int classes, * all the recall levels are evaluated, otherwise only * map_point recall levels are used. For COCO evaluation * 101 points are used. - * @param map_step step used to increment IoU theshold + * @param map_step step used to increment IoU threshold * @param map_levels number of IoU step to perform * @param verbose is set to true, prints on screen additional info * @param write_on_file if set to true, the results produced by this function * are written on file - * @param net name of the considerd neural network + * @param net name of the considered neural network * * @return mAP IoU_tresh:IoU_tresh+map_step*map_levels (e.g. mAP 0.5:0.95 when * map_step=0.05 and map_levels=10) @@ -89,7 +89,7 @@ double computeMapNIoULevels(std::vector &images,const int classes, const int map_levels=10, const bool verbose=false, const bool write_on_file = false, std::string net = ""); /** - * This method computes the numper of True Positive (TP), False Positive (FP), + * This method computes the number of True Positive (TP), False Positive (FP), * False Negative (FN), precision, recall and f1-score. * Those values are computer over all the detections, over all the classes. * @@ -101,7 +101,7 @@ double computeMapNIoULevels(std::vector &images,const int classes, * @param verbose is set to true, prints on screen additional info * @param write_on_file if set to true, the results produced by this function * are written on file - * @param net name of the considerd neural network + * @param net name of the considered neural network */ void computeTPFPFN( std::vector &images,const int classes, const float IoU_thresh=0.5, const float conf_thresh=0.3, diff --git a/include/tkDNN/pluginsRT/DeformableConvRT.h b/include/tkDNN/pluginsRT/DeformableConvRT.h index bff6370..225a24e 100644 --- a/include/tkDNN/pluginsRT/DeformableConvRT.h +++ b/include/tkDNN/pluginsRT/DeformableConvRT.h @@ -89,7 +89,7 @@ public: for(int b=0; b input_bins, std::vector } if(output_bins.size() != outputs.size()) { std::cout< netLayers; std::ifstream if_cfg(cfg_file); diff --git a/src/DeformConv2d.cpp b/src/DeformConv2d.cpp index b161a22..dbb71e1 100644 --- a/src/DeformConv2d.cpp +++ b/src/DeformConv2d.cpp @@ -95,7 +95,7 @@ dnnType* DeformConv2d::infer(dataDim_t &dim, dnnType* srcData) { // split conv2d outputs into offset and mask checkCuda(cudaMemcpy(offset, output_conv, 2*chunk_dim*sizeof(dnnType), cudaMemcpyDeviceToDevice)); checkCuda(cudaMemcpy(mask, output_conv + 2*chunk_dim, chunk_dim*sizeof(dnnType), cudaMemcpyDeviceToDevice)); - // kernel sigmoide + // kernel sigmoid activationSIGMOIDForward(mask, mask, chunk_dim); // deformable convolution diff --git a/src/Dense.cpp b/src/Dense.cpp index b6a9af2..4371d06 100644 --- a/src/Dense.cpp +++ b/src/Dense.cpp @@ -37,7 +37,7 @@ dnnType* Dense::infer(dataDim_t &dim, dnnType* srcData) { // place bias into dstData checkCuda( cudaMemcpy(dstData, bias_d, dim_y*sizeof(dnnType), cudaMemcpyDeviceToDevice) ); - //do matrix moltiplication + //do matrix multiplication checkERROR( cublasSgemv(net->cublasHandle, CUBLAS_OP_T, dim_x, dim_y, &alpha, diff --git a/src/LSTM.cpp b/src/LSTM.cpp index 511fbee..7b87711 100644 --- a/src/LSTM.cpp +++ b/src/LSTM.cpp @@ -133,7 +133,7 @@ LSTM::LSTM( Network *net, int hiddensize, bool returnSeq, std::string fname_weig output_dim = input_dim; output_dim.c = stateSize*(bidirectional ? 2 : 1); - // if retunseq is disabled only the last timestep is returned + // if retunseq is disabled only the last timestamp is returned if(!returnSeq) { output_dim.h = 1; output_dim.w = 1; @@ -254,7 +254,7 @@ dnnType* LSTM::infer(dataDim_t &dim, dnnType* srcData) { rnnDesc, seqLen, // number of time steps (nT) x_desc_vec_.data(), // input array of desc (nT*nC_in) - srcF, // input pointer + srcF, // input pointer hx_desc_, // initial hidden state desc hx_ptr, // initial hidden state pointer cx_desc_, // initial cell state desc @@ -281,7 +281,7 @@ dnnType* LSTM::infer(dataDim_t &dim, dnnType* srcData) { rnnDesc, seqLen, // number of time steps (nT) x_desc_vec_.data(), // input array of desc (nT*nC_in) - srcB, // input pointer + srcB, // input pointer hx_desc_, // initial hidden state desc hx_ptr, // initial hidden state pointer cx_desc_, // initial cell state desc @@ -289,7 +289,7 @@ dnnType* LSTM::infer(dataDim_t &dim, dnnType* srcData) { w_desc_, // weights desc wb_ptr, // weights pointer y_desc_vec_.data(), // output desc (nT*nC_out) - dstB_NR, // output pointer + dstB_NR, // output pointer hy_desc_, // final hidden state desc hy_ptr, // final hidden state pointer cy_desc_, // final cell state desc @@ -307,7 +307,7 @@ dnnType* LSTM::infer(dataDim_t &dim, dnnType* srcData) { one_output_dim.c*sizeof(dnnType), cudaMemcpyDeviceToDevice)); } - // if retunseq is disabled only the last timestep is returned + // if retunseq is disabled only the last timestamp is returned if(returnSeq) { // forward transpose matrixTranspose(net->cublasHandle, dstF, dstData, diff --git a/src/LayerWgs.cpp b/src/LayerWgs.cpp index 4afb7cc..a761327 100644 --- a/src/LayerWgs.cpp +++ b/src/LayerWgs.cpp @@ -105,7 +105,7 @@ LayerWgs::LayerWgs(Network *net, int inputs, int outputs, float2half(tmp_d, variance16_d, b_size); cudaMemcpy(variance16_h, variance16_d, b_size*sizeof(__half), cudaMemcpyDeviceToHost); - //conver scales + //convert scales float2half(scales_d, scales16_d, b_size); cudaMemcpy(scales16_h, scales16_d, b_size*sizeof(__half), cudaMemcpyDeviceToHost); diff --git a/src/MulAdd.cpp b/src/MulAdd.cpp index 0c2a962..25cec8d 100644 --- a/src/MulAdd.cpp +++ b/src/MulAdd.cpp @@ -12,7 +12,7 @@ MulAdd::MulAdd(Network *net, dnnType mul, dnnType add) : Layer(net) { int size = input_dim.tot(); - // create a vector with all value setted to add + // create a vector with all value set to add dnnType *add_vector_h = new dnnType[size]; for(int i=0; igetBindingIndex("data"); buf_output_idx = engineRT->getBindingIndex("out"); - std::cout<<"input idex = "< output index = "< output index = "<getBindingDimensions(buf_input_idx); diff --git a/src/Region.cpp b/src/Region.cpp index 65bb786..7c26208 100644 --- a/src/Region.cpp +++ b/src/Region.cpp @@ -63,7 +63,7 @@ dnnType* Region::infer(dataDim_t &dim, dnnType* srcData) { } -/* Intepret class */ +/* Interpret class */ RegionInterpret::RegionInterpret(dataDim_t input_dim, dataDim_t output_dim, int classes, int coords, int num, float thresh, std::string fname_weights) { diff --git a/src/Shortcut.cpp b/src/Shortcut.cpp index 78a2f23..2c7a4f4 100644 --- a/src/Shortcut.cpp +++ b/src/Shortcut.cpp @@ -13,7 +13,7 @@ Shortcut::Shortcut(Network *net, Layer *backLayer) : Layer(net) { if( /*backLayer->output_dim.c != input_dim.c ||*/ backLayer->output_dim.w != input_dim.w || backLayer->output_dim.h != input_dim.h ) - FatalError("Shortcut dim missmatch"); + FatalError("Shortcut dim mismatch"); } Shortcut::~Shortcut() { diff --git a/src/evaluation.cpp b/src/evaluation.cpp index 58c951d..f23c380 100644 --- a/src/evaluation.cpp +++ b/src/evaluation.cpp @@ -63,7 +63,7 @@ double computeMap( std::vector &images,const int classes, int gt_checked = 0; - // for each detection comput IoU with groundtruth and match detetcion and + // for each detection compute IoU with groundtruth and match detetcion and // groundtruth with IoU greater than IoU_thresh for(auto &img:images){ for(size_t i=0; i &images,const int classes, } } - //compute average precision for each class. Two methods are avaible, + //compute average precision for each class. Two methods are available, //based on map_points required double mean_average_precision = 0; double last_recall, last_precision, delta_recall; @@ -287,7 +287,7 @@ void computeTPFPFN( std::vector &images,const int classes, } } - //count all TP, FP, FN and compute precsion, recall and f1-score + //count all TP, FP, FN and compute precision, recall and f1-score double avg_precision = 0, avg_recall = 0, f1_score = 0; int TP = 0, FP = 0, FN = 0; for(size_t i=0; i Date: Tue, 15 Sep 2020 10:50:24 +0200 Subject: [PATCH 07/12] Update README with Xavier NX FPS results Signed-off-by: Micaela Verucchi --- README.md | 37 +++++++++++++++++++++---------------- 1 file changed, 21 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index f17f75e..54b8140 100644 --- a/README.md +++ b/README.md @@ -18,27 +18,32 @@ M. Verucchi, G. Brilli, D. Sapienza, M. Verasani, M. Arena, F. Gatti, A. Capoton Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimension as the input size, on * RTX 2080Ti (CUDA 10.2, TensorRT 7.0.0, Cudnn 7.6.5); * Xavier AGX, Jetpack 4.3 (CUDA 10.0, CUDNN 7.6.3, tensorrt 6.0.1 ); + * Xavier NX, Jetpack 4.4 (CUDA 10.2, CUDNN 8.0.0, tensorrt 7.1.0 ). * Tx2, Jetpack 4.2 (CUDA 10.0, CUDNN 7.3.1, tensorrt 5.0.6 ); * Jetson Nano, Jetpack 4.4 (CUDA 10.2, CUDNN 8.0.0, tensorrt 7.1.0 ). | Platform | Network | FP32, B=1 | FP32, B=4 | FP16, B=1 | FP16, B=4 | INT8, B=1 | INT8, B=4 | | :------: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | :-----: | -| RTX 2080Ti | yolo4 320 | 118,59 |237,31 | 207,81 | 443,32 | 262,37 | 530,93 | -| RTX 2080Ti | yolo4 416 | 104,81 |162,86 | 169,06 | 293,78 | 206,93 | 353,26 | -| RTX 2080Ti | yolo4 512 | 92,98 |132,43 | 140,36 | 215,17 | 165,35 | 254,96 | -| RTX 2080Ti | yolo4 608 | 63,77 |81,53 | 111,39 | 152,89 | 127,79 | 184,72 | -| AGX Xavier | yolo4 320 | 26,78 |32,05 | 57,14 | 79,05 | 73,15 | 97,56 | -| AGX Xavier | yolo4 416 | 19,96 |21,52 | 41,01 | 49,00 | 50,81 | 60,61 | -| AGX Xavier | yolo4 512 | 16,58 |16,98 | 31,12 | 33,84 | 37,82 | 41,28 | -| AGX Xavier | yolo4 608 | 9,45 |10,13 | 21,92 | 23,36 | 27,05 | 28,93 | -| Tx2 | yolo4 320 | 11,18 | 12,07 | 15,32 | 16,31 | - | - | -| Tx2 | yolo4 416 | 7,30 | 7,58 | 9,45 | 9,90 | - | - | -| Tx2 | yolo4 512 | 5,96 | 5,95 | 7,22 | 7,23 | - | - | -| Tx2 | yolo4 608 | 3,63 | 3,65 | 4,67 | 4,70 | - | - | -| Nano | yolo4 320 | 4,23 | 4,55 | 6,14 | 6,53 | - | - | -| Nano | yolo4 416 | 2,88 | 3,00 | 3,90 | 4,04 | - | - | -| Nano | yolo4 512 | 2,32 | 2,34 | 3,02 | 3,04 | - | - | -| Nano | yolo4 608 | 1,40 | 1,41 | 1,92 | 1,93 | - | - | +| RTX 2080Ti | yolo4 320 | 118.59 | 237.31 | 207.81 | 443.32 | 262.37 | 530.93 | +| RTX 2080Ti | yolo4 416 | 104.81 | 162.86 | 169.06 | 293.78 | 206.93 | 353.26 | +| RTX 2080Ti | yolo4 512 | 92.98 | 132.43 | 140.36 | 215.17 | 165.35 | 254.96 | +| RTX 2080Ti | yolo4 608 | 63.77 | 81.53 | 111.39 | 152.89 | 127.79 | 184.72 | +| AGX Xavier | yolo4 320 | 26.78 | 32.05 | 57.14 | 79.05 | 73.15 | 97.56 | +| AGX Xavier | yolo4 416 | 19.96 | 21.52 | 41.01 | 49.00 | 50.81 | 60.61 | +| AGX Xavier | yolo4 512 | 16.58 | 16.98 | 31.12 | 33.84 | 37.82 | 41.28 | +| AGX Xavier | yolo4 608 | 9.45 | 10.13 | 21.92 | 23.36 | 27.05 | 28.93 | +| Xavier NX | yolo4 320 | 11.49 | 13.79 | 25.26 | 35.51 | 33.77 | 45.66 | +| Xavier NX | yolo4 416 | 8.38 | 9.65 | 18.72 | 22.97 | 23.71 | 29.28 | +| Xavier NX | yolo4 512 | 7.03 | 7.57 | 13.50 | 15.43 | 17.95 | 19.70 | +| Xavier NX | yolo4 608 | 4.49 | 4.56 | 10.10 | 11.02 | 13.09 | 14.04 | +| Tx2 | yolo4 320 | 11.18 | 12.07 | 15.32 | 16.31 | - | - | +| Tx2 | yolo4 416 | 7.30 | 7.58 | 9.45 | 9.90 | - | - | +| Tx2 | yolo4 512 | 5.96 | 5.95 | 7.22 | 7.23 | - | - | +| Tx2 | yolo4 608 | 3.63 | 3.65 | 4.67 | 4.70 | - | - | +| Nano | yolo4 320 | 4.23 | 4.55 | 6.14 | 6.53 | - | - | +| Nano | yolo4 416 | 2.88 | 3.00 | 3.90 | 4.04 | - | - | +| Nano | yolo4 512 | 2.32 | 2.34 | 3.02 | 3.04 | - | - | +| Nano | yolo4 608 | 1.40 | 1.41 | 1.92 | 1.93 | - | - | ## MAP Results Results for COCO val 2017 (5k images), on RTX 2080Ti, with conf threshold=0.001 From d3372aad31d27d68593209f13e7c189752ec6e42 Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Tue, 15 Sep 2020 14:09:02 +0200 Subject: [PATCH 08/12] Update README with Xavier NX FPS results 15W4Core Signed-off-by: Micaela Verucchi --- README.md | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 54b8140..70ddd26 100644 --- a/README.md +++ b/README.md @@ -32,10 +32,10 @@ Inference FPS of yolov4 with tkDNN, average of 1200 images with the same dimensi | AGX Xavier | yolo4 416 | 19.96 | 21.52 | 41.01 | 49.00 | 50.81 | 60.61 | | AGX Xavier | yolo4 512 | 16.58 | 16.98 | 31.12 | 33.84 | 37.82 | 41.28 | | AGX Xavier | yolo4 608 | 9.45 | 10.13 | 21.92 | 23.36 | 27.05 | 28.93 | -| Xavier NX | yolo4 320 | 11.49 | 13.79 | 25.26 | 35.51 | 33.77 | 45.66 | -| Xavier NX | yolo4 416 | 8.38 | 9.65 | 18.72 | 22.97 | 23.71 | 29.28 | -| Xavier NX | yolo4 512 | 7.03 | 7.57 | 13.50 | 15.43 | 17.95 | 19.70 | -| Xavier NX | yolo4 608 | 4.49 | 4.56 | 10.10 | 11.02 | 13.09 | 14.04 | +| Xavier NX | yolo4 320 | 14.56 | 16.25 | 30.14 | 41.15 | 42.13 | 53.42 | +| Xavier NX | yolo4 416 | 10.02 | 10.60 | 22.43 | 25.59 | 29.08 | 32.94 | +| Xavier NX | yolo4 512 | 8.10 | 8.32 | 15.78 | 17.13 | 20.51 | 22.46 | +| Xavier NX | yolo4 608 | 5.26 | 5.18 | 11.54 | 12.06 | 15.09 | 15.82 | | Tx2 | yolo4 320 | 11.18 | 12.07 | 15.32 | 16.31 | - | - | | Tx2 | yolo4 416 | 7.30 | 7.58 | 9.45 | 9.90 | - | - | | Tx2 | yolo4 512 | 5.96 | 5.95 | 7.22 | 7.23 | - | - | From a0e7f05a50e5bc639a3c843139c39884d2c5a7fc Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Sat, 10 Oct 2020 13:03:01 +0200 Subject: [PATCH 09/12] Update README.md --- README.md | 17 ++++++++++------- 1 file changed, 10 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 70ddd26..cdfa25b 100644 --- a/README.md +++ b/README.md @@ -3,15 +3,18 @@ tkDNN is a Deep Neural Network library built with cuDNN and tensorRT primitives, The main goal of this project is to exploit NVIDIA boards as much as possible to obtain the best inference performance. It does not allow training. -If you use tkDNN in your research, please cite one of the following papers. For use in commercial solutions, write at gattifrancesco@hotmail.it and micaela.verucchi@unimore.it or refer to https://hipert.unimore.it/ . +If you use tkDNN in your research, please cite the [following paper](https://ieeexplore.ieee.org/stamp/stamp.jsp?arnumber=9212130&casa_token=sQTJXi7tJNoAAAAA:BguH9xCIY48MxbtDS3LXzIXzO-9sWArm7Hd7y7BwaLmqRuM_Gx8bOYizFPNMNtpo5K0kB-P-). For use in commercial solutions, write at gattifrancesco@hotmail.it and micaela.verucchi@unimore.it or refer to https://hipert.unimore.it/ . ``` -Accepted paper @ IRC 2020, will soon be published. -M. Verucchi, L. Bartoli, F. Bagni, F. Gatti, P. Burgio and M. Bertogna, "Real-Time clustering and LiDAR-camera fusion on embedded platforms for self-driving cars", in proceedings in IEEE Robotic Computing (2020) - -Accepted paper @ ETFA 2020, will soon be published. -M. Verucchi, G. Brilli, D. Sapienza, M. Verasani, M. Arena, F. Gatti, A. Capotondi, R. Cavicchioli, M. Bertogna, M. Solieri -"A Systematic Assessment of Embedded Neural Networks for Object Detection", in IEEE International Conference on Emerging Technologies and Factory Automation (2020) +@inproceedings{verucchi2020systematic, + title={A Systematic Assessment of Embedded Neural Networks for Object Detection}, + author={Verucchi, Micaela and Brilli, Gianluca and Sapienza, Davide and Verasani, Mattia and Arena, Marco and Gatti, Francesco and Capotondi, Alessandro and Cavicchioli, Roberto and Bertogna, Marko and Solieri, Marco}, + booktitle={2020 25th IEEE International Conference on Emerging Technologies and Factory Automation (ETFA)}, + volume={1}, + pages={937--944}, + year={2020}, + organization={IEEE} +} ``` ## FPS Results From 86478f9384eef13d68a9406ee12fbcb4df6ab892 Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Fri, 23 Oct 2020 11:40:55 +0200 Subject: [PATCH 10/12] Add yolo4_mmr test Signed-off-by: Micaela Verucchi --- tests/darknet/cfg/yolo4_mmr.cfg | 1158 +++++++++++++++++++++++++++++++ tests/darknet/names/mmr.names | 4 + tests/darknet/yolo4_mmr.cpp | 34 + 3 files changed, 1196 insertions(+) create mode 100644 tests/darknet/cfg/yolo4_mmr.cfg create mode 100644 tests/darknet/names/mmr.names create mode 100644 tests/darknet/yolo4_mmr.cpp diff --git a/tests/darknet/cfg/yolo4_mmr.cfg b/tests/darknet/cfg/yolo4_mmr.cfg new file mode 100644 index 0000000..90a7204 --- /dev/null +++ b/tests/darknet/cfg/yolo4_mmr.cfg @@ -0,0 +1,1158 @@ +[net] +batch=1 +subdivisions=1 +# Training +width=512 +height=512 +# width=608 +# height=608 +channels=3 +momentum=0.949 +decay=0.0005 +angle=0 +saturation = 1.5 +exposure = 1.5 +hue=.1 + +learning_rate=0.0013 +burn_in=1000 +max_batches = 16000 +policy=steps +steps=12800,14400 +scales=.1,.1 + +#cutmix=1 +mosaic=1 + +#:104x104 54:52x52 85:26x26 104:13x13 for 416 + +[convolutional] +batch_normalize=1 +filters=32 +size=3 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=32 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-7 + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=64 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=64 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-10 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=128 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-28 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=256 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-28 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=1024 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=512 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-16 + +[convolutional] +batch_normalize=1 +filters=1024 +size=1 +stride=1 +pad=1 +activation=mish + +########################## + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +### SPP ### +[maxpool] +stride=1 +size=5 + +[route] +layers=-2 + +[maxpool] +stride=1 +size=9 + +[route] +layers=-4 + +[maxpool] +stride=1 +size=13 + +[route] +layers=-1,-3,-5,-6 +### End SPP ### + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[upsample] +stride=2 + +[route] +layers = 85 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +[upsample] +stride=2 + +[route] +layers = 54 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=128 +size=1 +stride=1 +pad=1 +activation=leaky + +########################## + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=256 +activation=leaky + +[convolutional] +size=1 +stride=1 +pad=1 +filters=27 +activation=linear + + +[yolo] +mask = 0,1,2 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=4 +num=9 +jitter=.3 +ignore_thresh = .7 +truth_thresh = 1 +scale_x_y = 1.2 +iou_thresh=0.213 +cls_normalizer=1.0 +iou_normalizer=0.07 +iou_loss=ciou +nms_kind=greedynms +beta_nms=0.6 +max_delta=5 + + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=256 +activation=leaky + +[route] +layers = -1, -16 + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=256 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=512 +activation=leaky + +[convolutional] +size=1 +stride=1 +pad=1 +filters=27 +activation=linear + + +[yolo] +mask = 3,4,5 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=4 +num=9 +jitter=.3 +ignore_thresh = .7 +truth_thresh = 1 +scale_x_y = 1.1 +iou_thresh=0.213 +cls_normalizer=1.0 +iou_normalizer=0.07 +iou_loss=ciou +nms_kind=greedynms +beta_nms=0.6 +max_delta=5 + + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=512 +activation=leaky + +[route] +layers = -1, -37 + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +batch_normalize=1 +filters=512 +size=1 +stride=1 +pad=1 +activation=leaky + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1024 +activation=leaky + +[convolutional] +size=1 +stride=1 +pad=1 +filters=27 +activation=linear + + +[yolo] +mask = 6,7,8 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=4 +num=9 +jitter=.3 +ignore_thresh = .7 +truth_thresh = 1 +random=1 +scale_x_y = 1.05 +iou_thresh=0.213 +cls_normalizer=1.0 +iou_normalizer=0.07 +iou_loss=ciou +nms_kind=greedynms +beta_nms=0.6 +max_delta=5 + diff --git a/tests/darknet/names/mmr.names b/tests/darknet/names/mmr.names new file mode 100644 index 0000000..701a1fc --- /dev/null +++ b/tests/darknet/names/mmr.names @@ -0,0 +1,4 @@ +blue-cone +yellow-cone +orange-cone +big-orange-cone \ No newline at end of file diff --git a/tests/darknet/yolo4_mmr.cpp b/tests/darknet/yolo4_mmr.cpp new file mode 100644 index 0000000..85649b2 --- /dev/null +++ b/tests/darknet/yolo4_mmr.cpp @@ -0,0 +1,34 @@ +#include +#include +#include "tkdnn.h" +#include "test.h" +#include "DarknetParser.h" + +int main() { + std::string bin_path = "yolo4_mmr"; + std::vector input_bins = { + bin_path + "/layers/input.bin" + }; + std::vector output_bins = { + bin_path + "/debug/layer139_out.bin", + bin_path + "/debug/layer150_out.bin", + bin_path + "/debug/layer161_out.bin" + }; + std::string wgs_path = bin_path + "/layers"; + std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4_mmr.cfg"; + std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/mmr.names"; + // downloadWeightsifDoNotExist(input_bins[0], bin_path, ""); + + // parse darknet network + tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); + net->print(); + + //convert network to tensorRT + tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str())); + + int ret = testInference(input_bins, output_bins, net, netRT); + net->releaseLayers(); + delete net; + delete netRT; + return ret; +} From 702791e41ac302ed0396034cf80d6d76303ac20a Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Mon, 23 Nov 2020 11:25:52 +0100 Subject: [PATCH 11/12] Add support for yolov4x-mish. Changes: - add parameters nms_kind, nms_thresh, new_coords to yolo layer and darknet parser - added diou nms, new method to compute the BBs - created test for yolov4x-mish called yolo4x Tested, all tests work. Problem to solve: little loss in mAP of yolo4x Signed-off-by: Micaela Verucchi --- include/tkDNN/DarknetParser.h | 3 + include/tkDNN/Layer.h | 14 +- include/tkDNN/pluginsRT/YoloRT.h | 20 +- scripts/test_all_tests.sh | 1 + src/DarknetParser.cpp | 14 +- src/NetworkRT.cpp | 12 +- src/Yolo.cpp | 68 +- src/Yolo3Detection.cpp | 7 +- tests/darknet/cfg/yolo4x.cfg | 1427 ++++++++++++++++++++++++++++++ tests/darknet/yolo4x.cpp | 36 + 10 files changed, 1571 insertions(+), 31 deletions(-) create mode 100644 tests/darknet/cfg/yolo4x.cfg create mode 100644 tests/darknet/yolo4x.cpp diff --git a/include/tkDNN/DarknetParser.h b/include/tkDNN/DarknetParser.h index 29d1e8e..089c4d6 100644 --- a/include/tkDNN/DarknetParser.h +++ b/include/tkDNN/DarknetParser.h @@ -24,7 +24,10 @@ namespace tk { namespace dnn { int num = 1; int pad = 0; int coords = 4; + int nms_kind = 0; + int new_coords= 0; float scale_xy = 1; + float nms_thresh = 0.45; std::vector layers; std::string activation = "linear"; diff --git a/include/tkDNN/Layer.h b/include/tkDNN/Layer.h index 790a431..25c4565 100644 --- a/include/tkDNN/Layer.h +++ b/include/tkDNN/Layer.h @@ -610,24 +610,28 @@ public: int sort_class; }; - Yolo(Network *net, int classes, int num, std::string fname_weights,int n_masks=3, float scale_xy=1); + enum nmsKind_t {GREEDY_NMS=0, DIOU_NMS=1}; + + Yolo(Network *net, int classes, int num, std::string fname_weights,int n_masks=3, float scale_xy=1, double nms_thresh=0.45, nmsKind_t nsm_kind=GREEDY_NMS, int new_coords=0); virtual ~Yolo(); virtual layerType_t getLayerType() { return LAYER_YOLO; }; - int classes, num, n_masks; + int classes, num, n_masks, new_coords; dnnType *mask_h, *mask_d; //anchors dnnType *bias_h, *bias_d; //anchors float scaleXY; + double nms_thresh; + nmsKind_t nsm_kind; std::vector classesNames; virtual dnnType* infer(dataDim_t &dim, dnnType* srcData); - int computeDetections(Yolo::detection *dets, int &ndets, int netw, int neth, float thresh); + int computeDetections(Yolo::detection *dets, int &ndets, int netw, int neth, float thresh, int new_coords=0); dnnType *predictions; - static const int MAX_DETECTIONS = 8192; + static const int MAX_DETECTIONS = 8192*2; static Yolo::detection *allocateDetections(int nboxes, int classes); - static void mergeDetections(Yolo::detection *dets, int ndets, int classes); + static void mergeDetections(Yolo::detection *dets, int ndets, int classes, double nms_thresh=0.45, nmsKind_t nsm_kind=GREEDY_NMS); }; /** diff --git a/include/tkDNN/pluginsRT/YoloRT.h b/include/tkDNN/pluginsRT/YoloRT.h index f8e596c..9af8587 100644 --- a/include/tkDNN/pluginsRT/YoloRT.h +++ b/include/tkDNN/pluginsRT/YoloRT.h @@ -8,12 +8,15 @@ class YoloRT : public IPlugin { public: - YoloRT(int classes, int num, tk::dnn::Yolo *yolo = nullptr, int n_masks=3, float scale_xy=1) { + YoloRT(int classes, int num, tk::dnn::Yolo *yolo = nullptr, int n_masks=3, float scale_xy=1, float nms_thresh=0.45, int nms_kind=0, int new_coords=0) { this->classes = classes; this->num = num; this->n_masks = n_masks; this->scaleXY = scale_xy; + this->nms_thresh = nms_thresh; + this->nms_kind = nms_kind; + this->new_coords = new_coords; mask = new dnnType[n_masks]; bias = new dnnType[num*n_masks*2]; @@ -64,7 +67,10 @@ public: for (int b = 0; b < batchSize; ++b){ for(int n = 0; n < n_masks; ++n){ int index = entry_index(b, n*w*h, 0); - activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); + if (new_coords == 1) + activationLOGISTICForward(srcData + index, dstData + index, 4*w*h, stream); //x,y,w,h + else + activationLOGISTICForward(srcData + index, dstData + index, 2*w*h, stream); //x,y if (this->scaleXY != 1) scalAdd(dstData + index, 2 * w*h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); @@ -79,7 +85,7 @@ public: virtual size_t getSerializationSize() override { - return 6*sizeof(int) + sizeof(float)+ n_masks*sizeof(dnnType) + num*n_masks*2*sizeof(dnnType) + YOLORT_CLASSNAME_W*classes*sizeof(char); + return 8*sizeof(int) + 2*sizeof(float)+ n_masks*sizeof(dnnType) + num*n_masks*2*sizeof(dnnType) + YOLORT_CLASSNAME_W*classes*sizeof(char); } virtual void serialize(void* buffer) override { @@ -87,10 +93,13 @@ public: tk::dnn::writeBUF(buf, classes); tk::dnn::writeBUF(buf, num); tk::dnn::writeBUF(buf, n_masks); + tk::dnn::writeBUF(buf, scaleXY); + tk::dnn::writeBUF(buf, nms_thresh); + tk::dnn::writeBUF(buf, nms_kind); + tk::dnn::writeBUF(buf, new_coords); tk::dnn::writeBUF(buf, c); tk::dnn::writeBUF(buf, h); tk::dnn::writeBUF(buf, w); - tk::dnn::writeBUF(buf, scaleXY); for(int i=0; i classesNames; dnnType *mask; diff --git a/scripts/test_all_tests.sh b/scripts/test_all_tests.sh index 770aa22..af04aff 100644 --- a/scripts/test_all_tests.sh +++ b/scripts/test_all_tests.sh @@ -73,6 +73,7 @@ do print_output $? imuodom test_net yolo4 + test_net yolo4x test_net yolo4_berkeley test_net yolo4tiny test_net yolo3 diff --git a/src/DarknetParser.cpp b/src/DarknetParser.cpp index 7d7d989..7b5410c 100644 --- a/src/DarknetParser.cpp +++ b/src/DarknetParser.cpp @@ -37,7 +37,10 @@ namespace tk { namespace dnn { std::string name,value; if(!divideNameAndValue(line, name, value)) return false; - if(name.find("width") != std::string::npos) + + if(name.find("new_coords") != std::string::npos) + fields.new_coords = std::stoi(value); + else if(name.find("width") != std::string::npos) fields.width = std::stoi(value); else if(name.find("height") != std::string::npos) fields.height = std::stoi(value); @@ -79,6 +82,13 @@ namespace tk { namespace dnn { fields.group_id = std::stoi(value); else if(name.find("scale_x_y") != std::string::npos) fields.scale_xy = std::stof(value); + else if(name.find("beta_nms") != std::string::npos) + fields.nms_thresh = std::stof(value); + else if(name.find("nms_kind") != std::string::npos){ + if(value == "greedynms") fields.nms_kind = 0; + else if(value == "diounms") fields.nms_kind = 1; + else std::cout<<"Not supported nms_kind "<classesNames = names; diff --git a/src/NetworkRT.cpp b/src/NetworkRT.cpp index 9d38440..501ade4 100644 --- a/src/NetworkRT.cpp +++ b/src/NetworkRT.cpp @@ -529,7 +529,7 @@ ILayer* NetworkRT::convert_layer(ITensor *input, Yolo *l) { //std::cout<<"convert Yolo\n"; //std::cout<<"New plugin YOLO\n"; - IPlugin *plugin = new YoloRT(l->classes, l->num, l, l->n_masks, l->scaleXY); + IPlugin *plugin = new YoloRT(l->classes, l->num, l, l->n_masks, l->scaleXY, l->nms_thresh, l->nsm_kind, l->new_coords); IPluginLayer *lRT = networkRT->addPlugin(&input, 1, *plugin); checkNULL(lRT); return lRT; @@ -739,12 +739,16 @@ IPlugin* PluginFactory::createPlugin(const char* layerName, const void* serialDa if(name.find("Yolo") == 0) { YoloRT *r = new YoloRT(readBUF(buf), //classes readBUF(buf), //num - nullptr, - readBUF(buf)); //n_masks + nullptr, //yolo + readBUF(buf), //n_masks + readBUF(buf), //scale_xy + readBUF(buf), //nms_thresh + readBUF(buf), //nms_kind + readBUF(buf) //new_coords + ); r->c = readBUF(buf); r->h = readBUF(buf); r->w = readBUF(buf); - r->scaleXY = readBUF(buf); for(int i=0; in_masks; i++) r->mask[i] = readBUF(buf); for(int i=0; in_masks*2*r->num; i++) diff --git a/src/Yolo.cpp b/src/Yolo.cpp index a4416be..9737e74 100644 --- a/src/Yolo.cpp +++ b/src/Yolo.cpp @@ -11,7 +11,7 @@ namespace tk { namespace dnn { -Yolo::Yolo(Network *net, int classes, int num, std::string fname_weights, int n_masks, float scale_xy) : +Yolo::Yolo(Network *net, int classes, int num, std::string fname_weights, int n_masks, float scale_xy, double nms_thresh, nmsKind_t nsm_kind, int new_coords) : Layer(net) { this->final = true; @@ -19,6 +19,9 @@ Yolo::Yolo(Network *net, int classes, int num, std::string fname_weights, int n_ this->num = num; this->n_masks = n_masks; this->scaleXY = scale_xy; + this->nms_thresh = nms_thresh; + this->nsm_kind = nsm_kind; + this->new_coords = new_coords; // load anchors if(fname_weights != "") { @@ -59,12 +62,21 @@ int entry_index(int batch, int location, int entry, entry*input_dim.w*input_dim.h + loc; } -Yolo::box get_yolo_box(float *x, float *biases, int n, int index, int i, int j, int lw, int lh, int w, int h, int stride) { +Yolo::box get_yolo_box(float *x, float *biases, int n, int index, int i, int j, int lw, int lh, int w, int h, int stride, int new_coords) { Yolo::box b; - b.x = (i + x[index + 0*stride]) / lw; - b.y = (j + x[index + 1*stride]) / lh; - b.w = exp(x[index + 2*stride]) * biases[2*n] / w; - b.h = exp(x[index + 3*stride]) * biases[2*n+1] / h; + + if(new_coords == 0){ + b.x = (i + x[index + 0*stride]) / lw; + b.y = (j + x[index + 1*stride]) / lh; + b.w = exp(x[index + 2*stride]) * biases[2*n] / w; + b.h = exp(x[index + 3*stride]) * biases[2*n+1] / h; + } + else{ + b.x = (i + x[index + 0 * stride] * 2 - 0.5) / lw; + b.y = (j + x[index + 1 * stride] * 2 - 0.5) / lh; + b.w = x[index + 2 * stride] * x[index + 2 * stride] * 4 * biases[2 * n] / w; + b.h = x[index + 3 * stride] * x[index + 3 * stride] * 4 * biases[2 * n + 1] / h; + } return b; } @@ -75,7 +87,10 @@ dnnType* Yolo::infer(dataDim_t &dim, dnnType* srcData) { for (int b = 0; b < dim.n; ++b){ for(int n = 0; n < n_masks; ++n){ int index = entry_index(b, n*dim.w*dim.h, 0, classes, input_dim, output_dim); - activationLOGISTICForward(srcData + index, dstData + index, 2*dim.w*dim.h); + if (new_coords == 1) + activationLOGISTICForward(srcData + index, dstData + index, 4*dim.w*dim.h); + else + activationLOGISTICForward(srcData + index, dstData + index, 2*dim.w*dim.h); if (this->scaleXY != 1) scalAdd(dstData + index, 2 * dim.w*dim.h, this->scaleXY, -0.5*(this->scaleXY - 1), 1); @@ -116,7 +131,7 @@ void correct_yolo_boxes(Yolo::detection *dets, int n, int w, int h, int netw, in } } -int Yolo::computeDetections(Yolo::detection *dets, int &ndets, int netw, int neth, float thresh) { +int Yolo::computeDetections(Yolo::detection *dets, int &ndets, int netw, int neth, float thresh, int new_coords) { if(predictions == nullptr) predictions = new dnnType[output_dim.tot()]; @@ -140,7 +155,7 @@ int Yolo::computeDetections(Yolo::detection *dets, int &ndets, int netw, int net if(objectness <= thresh) continue; int box_index = entry_index(0, n*lw*lh + i, 0, classes, input_dim, output_dim); - dets[count].bbox = get_yolo_box(predictions, bias_h, mask_h[n], box_index, col, row, lw, lh, netw, neth, lw*lh); + dets[count].bbox = get_yolo_box(predictions, bias_h, mask_h[n], box_index, col, row, lw, lh, netw, neth, lw*lh, new_coords); dets[count].objectness = objectness; dets[count].classes = classes; for(j = 0; j < classes; ++j){ @@ -193,6 +208,32 @@ float yolo_box_iou(Yolo::box a, Yolo::box b) return yolo_box_intersection(a, b)/yolo_box_union(a, b); } +void box_c(const Yolo::box a, const Yolo::box b, float& top, float& bot, float& left, float& right) { + top = std::min(a.y - a.h / 2, b.y - b.h / 2); + bot = std::max(a.y + a.h / 2, b.y + b.h / 2); + left = std::min(a.x - a.w / 2, b.x - b.w / 2); + right = std::max(a.x + a.w / 2, b.x + b.w / 2); +} + +// https://github.com/Zzh-tju/DIoU-darknet +// https://arxiv.org/abs/1911.08287 +float yolo_box_diou(const Yolo::box a, const Yolo::box b, const float nms_thresh=0.6) +{ + float top, bot, left, right; + box_c(a, b, top, bot, left, right); + float w = right - left; + float h = bot - top; + float c = w * w + h * h; + float iou = yolo_box_iou(a, b); + if (c == 0) + return iou; + + float d = (a.x - b.x) * (a.x - b.x) + (a.y - b.y) * (a.y - b.y); + float u = pow(d / c, nms_thresh); + float diou_term = u; + return iou - diou_term; +} + int yolo_nms_comparator(const void *pa, const void *pb) { Yolo::detection a = *(Yolo::detection *)pa; @@ -219,8 +260,7 @@ Yolo::detection *Yolo::allocateDetections(int nboxes, int classes) { return dets; } -void Yolo::mergeDetections(Yolo::detection *dets, int ndets, int classes) { - double nms_thresh = 0.45; +void Yolo::mergeDetections(Yolo::detection *dets, int ndets, int classes, double nms_thresh, nmsKind_t nsm_kind) { int total = ndets; int i, j, k; @@ -246,13 +286,13 @@ void Yolo::mergeDetections(Yolo::detection *dets, int ndets, int classes) { box a = dets[i].bbox; for(j = i+1; j < total; ++j){ box b = dets[j].bbox; - if (yolo_box_iou(a, b) > nms_thresh){ + if (nsm_kind == GREEDY_NMS && yolo_box_iou(a, b) > nms_thresh) + dets[j].prob[k] = 0; + else if (nsm_kind == DIOU_NMS && yolo_box_diou(a, b, nms_thresh) > nms_thresh) dets[j].prob[k] = 0; - } } } } - } }} diff --git a/src/Yolo3Detection.cpp b/src/Yolo3Detection.cpp index e9b0064..b94eea9 100644 --- a/src/Yolo3Detection.cpp +++ b/src/Yolo3Detection.cpp @@ -32,6 +32,9 @@ bool Yolo3Detection::init(const std::string& tensor_path, const int n_classes, c memcpy(yolo[i]->bias_h, yRT->bias, sizeof(dnnType)*num*nMasks*2); yolo[i]->input_dim = yolo[i]->output_dim = tk::dnn::dataDim_t(1, yRT->c, yRT->h, yRT->w); yolo[i]->classesNames = yRT->classesNames; + yolo[i]->nms_thresh = yRT->nms_thresh; + yolo[i]->nsm_kind = (tk::dnn::Yolo::nmsKind_t) yRT->nms_kind; + yolo[i]->new_coords = yRT->new_coords; } dets = tk::dnn::Yolo::allocateDetections(tk::dnn::Yolo::MAX_DETECTIONS, classes); @@ -102,9 +105,9 @@ void Yolo3Detection::postprocess(const int bi, const bool mAP){ nDets = 0; for(int i=0; ipluginFactory->n_yolos; i++) { yolo[i]->dstData = rt_out[i]; - yolo[i]->computeDetections(dets, nDets, netRT->input_dim.w, netRT->input_dim.h, confThreshold); + yolo[i]->computeDetections(dets, nDets, netRT->input_dim.w, netRT->input_dim.h, confThreshold, yolo[i]->new_coords); } - tk::dnn::Yolo::mergeDetections(dets, nDets, classes); + tk::dnn::Yolo::mergeDetections(dets, nDets, classes, yolo[0]->nms_thresh, yolo[0]->nsm_kind); // fill detected detected.clear(); diff --git a/tests/darknet/cfg/yolo4x.cfg b/tests/darknet/cfg/yolo4x.cfg new file mode 100644 index 0000000..89f2564 --- /dev/null +++ b/tests/darknet/cfg/yolo4x.cfg @@ -0,0 +1,1427 @@ +[net] +# Testing +#batch=1 +#subdivisions=1 +# Training +batch=64 +subdivisions=8 +width=672 +height=672 +channels=3 +momentum=0.949 +decay=0.0005 +angle=0 +saturation = 1.5 +exposure = 1.5 +hue=.1 + +learning_rate=0.00261 +burn_in=1000 +max_batches = 500500 +policy=steps +steps=400000,450000 +scales=.1,.1 + +mosaic=1 + +letter_box=1 + +[convolutional] +batch_normalize=1 +filters=32 +size=3 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=80 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=40 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=80 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +# Downsample + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=80 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=80 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=80 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=80 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=80 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=80 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=80 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=80 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=80 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-13 + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-34 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=640 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-34 + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +# Downsample + +[convolutional] +batch_normalize=1 +filters=1280 +size=3 +stride=2 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=3 +stride=1 +pad=1 +activation=mish + +[shortcut] +from=-3 +activation=linear + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1,-19 + +[convolutional] +batch_normalize=1 +filters=1280 +size=1 +stride=1 +pad=1 +activation=mish + +########################## 6 0 6 6 3 + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=640 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +### SPP ### +[maxpool] +stride=1 +size=5 + +[route] +layers=-2 + +[maxpool] +stride=1 +size=9 + +[route] +layers=-4 + +[maxpool] +stride=1 +size=13 + +[route] +layers=-1,-3,-5,-6 +### End SPP ### + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=640 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=640 +activation=mish + +[route] +layers = -1, -15 + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[upsample] +stride=2 + +[route] +layers = 94 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=320 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=320 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=320 +activation=mish + +[route] +layers = -1, -8 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[upsample] +stride=2 + +[route] +layers = 57 + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -1, -3 + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=160 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=160 +activation=mish + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=160 +activation=mish + +[route] +layers = -1, -8 + +[convolutional] +batch_normalize=1 +filters=160 +size=1 +stride=1 +pad=1 +activation=mish + +########################## + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=320 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=linear + + +[yolo] +mask = 0,1,2 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +objectness_smooth=0 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=4.0 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=320 +activation=mish + +[route] +layers = -1, -22 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=320 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=320 +activation=mish + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=320 +activation=mish + +[route] +layers = -1,-8 + +[convolutional] +batch_normalize=1 +filters=320 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=640 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=linear + + +[yolo] +mask = 3,4,5 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +objectness_smooth=1 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=1.0 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 + +[route] +layers = -4 + +[convolutional] +batch_normalize=1 +size=3 +stride=2 +pad=1 +filters=640 +activation=mish + +[route] +layers = -1, -55 + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[route] +layers = -2 + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=640 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=640 +activation=mish + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=640 +activation=mish + +[route] +layers = -1,-8 + +[convolutional] +batch_normalize=1 +filters=640 +size=1 +stride=1 +pad=1 +activation=mish + +[convolutional] +batch_normalize=1 +size=3 +stride=1 +pad=1 +filters=1280 +activation=mish + +[convolutional] +size=1 +stride=1 +pad=1 +filters=255 +activation=linear + + +[yolo] +mask = 6,7,8 +anchors = 12, 16, 19, 36, 40, 28, 36, 75, 76, 55, 72, 146, 142, 110, 192, 243, 459, 401 +classes=80 +num=9 +jitter=.1 +objectness_smooth=1 +ignore_thresh = .7 +truth_thresh = 1 +#random=1 +resize=1.5 +iou_thresh=0.2 +iou_normalizer=0.05 +cls_normalizer=0.5 +obj_normalizer=0.4 +iou_loss=ciou +nms_kind=diounms +beta_nms=0.6 +new_coords=1 diff --git a/tests/darknet/yolo4x.cpp b/tests/darknet/yolo4x.cpp new file mode 100644 index 0000000..b9ad003 --- /dev/null +++ b/tests/darknet/yolo4x.cpp @@ -0,0 +1,36 @@ +#include +#include +#include "tkdnn.h" +#include "test.h" +#include "DarknetParser.h" + +int main() { + std::string bin_path = "yolo4x"; + std::vector input_bins = { + bin_path + "/layers/input.bin" + }; + std::vector output_bins = { + bin_path + "/debug/layer168_out.bin", + bin_path + "/debug/layer185_out.bin", + bin_path + "/debug/layer202_out.bin" + }; + std::string wgs_path = bin_path + "/layers"; + std::string cfg_path = std::string(TKDNN_PATH) + "/tests/darknet/cfg/yolo4x.cfg"; + std::string name_path = std::string(TKDNN_PATH) + "/tests/darknet/names/coco.names"; + downloadWeightsifDoNotExist(input_bins[0], bin_path, "https://cloud.hipert.unimore.it/s/BLPpiAigZJLorQD/download"); + + + + // parse darknet network + tk::dnn::Network *net = tk::dnn::darknetParser(cfg_path, wgs_path, name_path); + net->print(); + + //convert network to tensorRT + tk::dnn::NetworkRT *netRT = new tk::dnn::NetworkRT(net, net->getNetworkRTName(bin_path.c_str())); + + int ret = testInference(input_bins, output_bins, net, netRT); + net->releaseLayers(); + delete net; + delete netRT; + return ret; +} From b8855b9599e52a51b371e99255063cd6f00fecd7 Mon Sep 17 00:00:00 2001 From: Micaela Verucchi Date: Mon, 23 Nov 2020 11:34:06 +0100 Subject: [PATCH 12/12] Update README Signed-off-by: Micaela Verucchi --- README.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index cdfa25b..84e0037 100644 --- a/README.md +++ b/README.md @@ -352,7 +352,8 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing | csresnext50-panet-spp | Cross Stage Partial Network 7 | [COCO 2014](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/Kcs4xBozwY4wFx8/download) | | yolo4 | Yolov4 8 | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/d97CFzYqCPCp5Hg/download) | | yolo4_berkeley | Yolov4 8 | [BDD100K ](https://bair.berkeley.edu/blog/2018/05/30/bdd/) | 10 | 540x320 | [weights](https://cloud.hipert.unimore.it/s/nkWFa5fgb4NTdnB/download) | -| yolo4tiny | Yolov4 tiny | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) | +| yolo4tiny | Yolov4 tiny 9 | [COCO 2017](http://cocodataset.org/) | 80 | 416x416 | [weights](https://cloud.hipert.unimore.it/s/iRnc4pSqmx78gJs/download) | +| yolo4x | Yolov4x-mish 9 | [COCO 2017](http://cocodataset.org/) | 80 | 672x672 | [weights](https://cloud.hipert.unimore.it/s/BLPpiAigZJLorQD/download) | ## References @@ -365,3 +366,4 @@ This demo also creates a json file named ```net_name_COCO_res.json``` containing 6. He, Kaiming, et al. "Deep residual learning for image recognition." Proceedings of the IEEE conference on computer vision and pattern recognition. 2016. 7. Wang, Chien-Yao, et al. "CSPNet: A New Backbone that can Enhance Learning Capability of CNN." arXiv preprint arXiv:1911.11929 (2019). 8. Bochkovskiy, Alexey, Chien-Yao Wang, and Hong-Yuan Mark Liao. "YOLOv4: Optimal Speed and Accuracy of Object Detection." arXiv preprint arXiv:2004.10934 (2020). +9. Bochkovskiy, Alexey, "Yolo v4, v3 and v2 for Windows and Linux" (https://github.com/AlexeyAB/darknet)