Batchnorm eps fix, works on jetpack 4.3
This commit is contained in:
@@ -3,9 +3,10 @@ tkDNN is a Deep Neural Network library built with cuDNN primitives specifically
|
|||||||
The main scope is to do high performance inference on already trained models.
|
The main scope is to do high performance inference on already trained models.
|
||||||
|
|
||||||
this branch actually work on every NVIDIA GPU that support the dependencies:
|
this branch actually work on every NVIDIA GPU that support the dependencies:
|
||||||
* CUDA 9
|
* CUDA 10.0
|
||||||
* CUDNN 7.105
|
* CUDNN 7.603
|
||||||
* TENSORRT 4.02
|
* TENSORRT 6.01
|
||||||
|
* OPENCV 4.1
|
||||||
|
|
||||||
## Dependencies
|
## Dependencies
|
||||||
|
|
||||||
|
|||||||
@@ -27,6 +27,8 @@ enum layerType_t
|
|||||||
LAYER_YOLO
|
LAYER_YOLO
|
||||||
};
|
};
|
||||||
|
|
||||||
|
#define TKDNN_BN_MIN_EPSILON 1e-5
|
||||||
|
|
||||||
/**
|
/**
|
||||||
Simple layer Father class
|
Simple layer Father class
|
||||||
*/
|
*/
|
||||||
|
|||||||
+5
-5
@@ -120,11 +120,11 @@ dnnType *Conv2d::infer(dataDim_t &dim, dnnType *srcData)
|
|||||||
float one = 1;
|
float one = 1;
|
||||||
float zero = 0;
|
float zero = 0;
|
||||||
cudnnBatchNormalizationForwardInference(net->cudnnHandle,
|
cudnnBatchNormalizationForwardInference(net->cudnnHandle,
|
||||||
CUDNN_BATCHNORM_SPATIAL, &one, &zero,
|
CUDNN_BATCHNORM_SPATIAL, &one, &zero,
|
||||||
dstTensorDesc, dstData, dstTensorDesc,
|
dstTensorDesc, dstData, dstTensorDesc,
|
||||||
dstData, biasTensorDesc, //same tensor descriptor as bias
|
dstData, biasTensorDesc, //same tensor descriptor as bias
|
||||||
scales_d, bias_d, mean_d, variance_d,
|
scales_d, bias_d, mean_d, variance_d,
|
||||||
CUDNN_BN_MIN_EPSILON);
|
TKDNN_BN_MIN_EPSILON);
|
||||||
}
|
}
|
||||||
//update data dimensions
|
//update data dimensions
|
||||||
dim = output_dim;
|
dim = output_dim;
|
||||||
|
|||||||
+1
-1
@@ -34,7 +34,7 @@ LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
|||||||
seek += outputs;
|
seek += outputs;
|
||||||
readBinaryFile(weights_path.c_str(), outputs, &variance_h, &variance_d, seek);
|
readBinaryFile(weights_path.c_str(), outputs, &variance_h, &variance_d, seek);
|
||||||
|
|
||||||
float eps = CUDNN_BN_MIN_EPSILON;
|
float eps = TKDNN_BN_MIN_EPSILON;
|
||||||
|
|
||||||
power_h = new dnnType[outputs];
|
power_h = new dnnType[outputs];
|
||||||
for (int i = 0; i < outputs; i++)
|
for (int i = 0; i < outputs; i++)
|
||||||
|
|||||||
+9
-8
@@ -21,10 +21,6 @@ Network::Network(dataDim_t input_dim)
|
|||||||
<< ", CUDNN v" << cu_ver << ")\n";
|
<< ", CUDNN v" << cu_ver << ")\n";
|
||||||
dataType = CUDNN_DATA_FLOAT;
|
dataType = CUDNN_DATA_FLOAT;
|
||||||
tensorFormat = CUDNN_TENSOR_NCHW;
|
tensorFormat = CUDNN_TENSOR_NCHW;
|
||||||
|
|
||||||
checkCUDNN(cudnnCreate(&cudnnHandle));
|
|
||||||
checkERROR(cublasCreate(&cublasHandle));
|
|
||||||
|
|
||||||
num_layers = 0;
|
num_layers = 0;
|
||||||
|
|
||||||
fp16 = false;
|
fp16 = false;
|
||||||
@@ -39,11 +35,16 @@ Network::Network(dataDim_t input_dim)
|
|||||||
fp16 = true;
|
fp16 = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if(fp16)
|
||||||
|
std::cout<<COL_REDB<<"!! FP16 INERENCE ENABLED !!"<<COL_END<<"\n";
|
||||||
|
if(dla)
|
||||||
|
std::cout<<COL_GREENB<<"!! DLA INERENCE ENABLED !!"<<COL_END<<"\n";
|
||||||
|
|
||||||
|
|
||||||
|
checkCUDNN( cudnnCreate(&cudnnHandle) );
|
||||||
|
checkERROR( cublasCreate(&cublasHandle) );
|
||||||
|
|
||||||
if (fp16)
|
|
||||||
std::cout << COL_REDB << "!! FP16 INERENCE ENABLED !!" << COL_END << "\n";
|
|
||||||
if (dla)
|
|
||||||
std::cout << COL_GREENB << "!! DLA INERENCE ENABLED !!" << COL_END << "\n";
|
|
||||||
}
|
}
|
||||||
|
|
||||||
Network::~Network()
|
Network::~Network()
|
||||||
|
|||||||
Reference in New Issue
Block a user