Add CenterNet based on Resnet101, TensorRT not implemented.
Signed-off-by: Davide Sapienza <sapienza.dav@gmail.com>
This commit is contained in:
@@ -44,7 +44,6 @@ Activation::~Activation() {
|
||||
}
|
||||
|
||||
dnnType* Activation::infer(dataDim_t &dim, dnnType* srcData) {
|
||||
|
||||
if(act_mode == ACTIVATION_LEAKY) {
|
||||
activationLEAKYForward(srcData, dstData, dim.tot());
|
||||
|
||||
|
||||
+6
-5
@@ -44,7 +44,8 @@ void Conv2d::initCUDNN(bool back) {
|
||||
convDesc, srcTensor, filterDesc,
|
||||
&tmpdim.n, &tmpdim.c, &tmpdim.h, &tmpdim.w) );
|
||||
if(odim.n != tmpdim.n || odim.c != tmpdim.c || odim.h != tmpdim.h || odim.w != tmpdim.w) {
|
||||
std::cout<<"tkdim: "; odim.print();
|
||||
std::cout<<"tkdim input: "; idim.print();
|
||||
std::cout<<"tkdim output: "; odim.print();
|
||||
std::cout<<"cudnndim: "; tmpdim.print();
|
||||
FatalError("Eror conv dimension mismatch");
|
||||
}
|
||||
@@ -107,12 +108,12 @@ void Conv2d::inferCUDNN(dnnType* srcData, bool back) {
|
||||
} else {
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(0);
|
||||
cudnnBatchNormalizationForwardInference(net->cudnnHandle,
|
||||
checkCUDNN( cudnnBatchNormalizationForwardInference(net->cudnnHandle,
|
||||
CUDNN_BATCHNORM_SPATIAL, &alpha, &beta,
|
||||
dstTensorDesc, dstData, dstTensorDesc,
|
||||
dstData, biasTensorDesc, //same tensor descriptor as bias
|
||||
scales_d, bias_d, mean_d, variance_d,
|
||||
CUDNN_BN_MIN_EPSILON);
|
||||
CUDNN_BN_MIN_EPSILON) );
|
||||
}
|
||||
}
|
||||
|
||||
@@ -140,8 +141,8 @@ Conv2d::Conv2d( Network *net, int out_ch, int kernelH, int kernelW,
|
||||
} else {
|
||||
output_dim.n = input_dim.n;
|
||||
output_dim.c = out_ch;
|
||||
output_dim.h = (input_dim.h * strideH) - 2*paddingH + kernelH -1;
|
||||
output_dim.w = (input_dim.w * strideW) - 2*paddingW + kernelW -1;
|
||||
output_dim.h = ((input_dim.h-1) * strideH) - 2*paddingH + kernelH;
|
||||
output_dim.w = ((input_dim.w-1) * strideW) - 2*paddingW + kernelW;
|
||||
output_dim.l = 1;
|
||||
}
|
||||
initCUDNN(deConv);
|
||||
|
||||
@@ -0,0 +1,206 @@
|
||||
#include <iostream>
|
||||
|
||||
#include "Layer.h"
|
||||
#include "kernels.h"
|
||||
#include <math.h>
|
||||
|
||||
|
||||
namespace tk { namespace dnn {
|
||||
|
||||
void DeformConv2d::initCUDNN() {
|
||||
|
||||
checkCUDNN( cudnnCreateTensorDescriptor(&biasTensorDesc) );
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(biasTensorDesc,
|
||||
net->tensorFormat, net->dataType,
|
||||
1, output_dim.c, 1, 1) );
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(dstTensorDesc,
|
||||
net->tensorFormat, net->dataType, output_dim.n, output_dim.c, output_dim.h, output_dim.w));
|
||||
|
||||
}
|
||||
|
||||
DeformConv2d::DeformConv2d( Network *net, int out_ch, int deformable_group, int kernelH, int kernelW,
|
||||
int strideH, int strideW, int paddingH, int paddingW,
|
||||
std::string d_fname_weights, std::string fname_weights, bool batchnorm) :
|
||||
|
||||
LayerWgs(net, net->getOutputDim().c, out_ch, kernelH, kernelW, 1,
|
||||
d_fname_weights, batchnorm, true){
|
||||
|
||||
this->out_ch = out_ch;
|
||||
this->deformableGroup = deformable_group;
|
||||
this->kernelH = kernelH;
|
||||
this->kernelW = kernelW;
|
||||
this->strideH = strideH;
|
||||
this->strideW = strideW;
|
||||
this->paddingH = paddingH;
|
||||
this->paddingW = paddingW;
|
||||
|
||||
preconv = new tk::dnn::Conv2d(net, deformable_group * 3 * kernelH * kernelW, kernelH, kernelW,
|
||||
strideH, strideW, paddingH, paddingW, fname_weights, false);
|
||||
net->num_layers--;
|
||||
|
||||
output_dim = preconv->output_dim;
|
||||
|
||||
output_dim.c = out_ch;
|
||||
initCUDNN();
|
||||
//allocate data for infer result
|
||||
checkCuda( cudaMalloc(&dstData, output_dim.tot()*sizeof(dnnType)) );
|
||||
}
|
||||
|
||||
DeformConv2d::~DeformConv2d() {
|
||||
|
||||
checkCUDNN( cudnnDestroyTensorDescriptor(biasTensorDesc) );
|
||||
checkCuda( cudaFree(dstData) );
|
||||
}
|
||||
|
||||
void Conv2dToChunk(int dim, dnnType* srcData, dnnType* offset, dnnType* mask)
|
||||
{
|
||||
// std::cout<<"9\n";
|
||||
// cudaMemcpyFromArray(offset, (const struct cudaArray *)srcData, 0, 2*dim.tot()/3, dim.tot()/3, cudaMemcpyDeviceToHost);
|
||||
// checkCuda(cudaMemcpyFromArray(offset, (const struct cudaArray *)srcData, 0, 0, 2*dim, cudaMemcpyDeviceToDevice));
|
||||
checkCuda(cudaMemcpy(offset, srcData, 2*dim*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
|
||||
// std::cout<<"9a\n";
|
||||
cudaDeviceSynchronize();
|
||||
// cudaMemcpyFromArray(mask, (const struct cudaArray *)srcData, 2*dim.tot()/3, dim.tot(), dim.tot()/3, cudaMemcpyDeviceToHost);
|
||||
// checkCuda(cudaMemcpyFromArray(mask, (const struct cudaArray *)srcData, 0, 2*dim, dim, cudaMemcpyDeviceToDevice));
|
||||
checkCuda(cudaMemcpy(mask, srcData + 2*dim, dim*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
// std::cout<<"9b\n";
|
||||
}
|
||||
|
||||
dnnType* DeformConv2d::infer(dataDim_t &dim, dnnType* srcData) {
|
||||
dnnType *input;
|
||||
checkCuda(cudaMalloc(&input, dim.tot()*sizeof(dnnType)));
|
||||
checkCuda(cudaMemcpy(input, srcData, dim.tot()*sizeof(dnnType), cudaMemcpyDeviceToDevice));
|
||||
cudaDeviceSynchronize();
|
||||
srcData = preconv->infer(dim, srcData);
|
||||
dim = preconv->output_dim;
|
||||
|
||||
//split to chank
|
||||
dnnType *offset, *mask;
|
||||
int dst_dim = dim.tot();
|
||||
if (dst_dim % 3 != 0 )
|
||||
std::cout<<"take attention\n\n";
|
||||
int chunk_dim = dst_dim/3;
|
||||
checkCuda(cudaMalloc(&offset, 2*chunk_dim*sizeof(dnnType)));
|
||||
checkCuda(cudaMalloc(&mask, chunk_dim*sizeof(dnnType)));
|
||||
cudaDeviceSynchronize();
|
||||
|
||||
Conv2dToChunk(chunk_dim, srcData, offset, mask);
|
||||
|
||||
// kernel sigmoide
|
||||
dnnType *vec;
|
||||
vec = new dnnType[chunk_dim];
|
||||
cudaDeviceSynchronize();
|
||||
cudaMemcpy(vec, mask, chunk_dim*sizeof(dnnType), cudaMemcpyDeviceToHost);
|
||||
cudaDeviceSynchronize();
|
||||
for(int i=0; i<chunk_dim; i++){
|
||||
// std::cout<<i<<" -- "<<vec[i]<<", ";
|
||||
vec[i] = 1.0f / (1.0f + exp(-vec[i]));
|
||||
// std::cout<<vec[i]<<std::endl;
|
||||
}
|
||||
|
||||
cudaDeviceSynchronize();
|
||||
cudaMemcpy(mask, vec, chunk_dim*sizeof(dnnType), cudaMemcpyHostToDevice);
|
||||
cudaDeviceSynchronize();
|
||||
free(vec);
|
||||
|
||||
// dnnType *tmp;
|
||||
// cudaMallocHost(&tmp, chunk_dim*sizeof(dnnType));
|
||||
// cudaMemcpy(tmp, mask, chunk_dim*sizeof(dnnType), cudaMemcpyDeviceToHost);
|
||||
// std::cout<<"conv2d output before chunking"<<std::endl;
|
||||
// for (size_t i = 0; i < chunk_dim; i++)
|
||||
// {
|
||||
// std::cout<<i<<" -- "<<tmp[i]<<", ";
|
||||
// }
|
||||
// std::cout<<std::endl;
|
||||
// cudaFreeHost(tmp);
|
||||
|
||||
|
||||
const int height_ones = (preconv->input_dim.h + 2 * this->paddingH - (1 * (this->kernelH - 1) + 1)) / this->strideH + 1;
|
||||
const int width_ones = (preconv->input_dim.w + 2 * this->paddingW - (1 * (this->kernelW - 1) + 1)) / this->strideW + 1;
|
||||
const int dim_ones = preconv->input_dim.c * this->kernelH * this->kernelW * 1 * height_ones * width_ones;
|
||||
|
||||
// kernel ones
|
||||
dnnType *ones_d1;
|
||||
cudaMallocHost(&ones_d1, (height_ones*width_ones)*sizeof(dnnType));
|
||||
float aus1[height_ones*width_ones];
|
||||
for(int i=0; i<height_ones*width_ones; i++)
|
||||
aus1[i]=1.0f;
|
||||
cudaMemcpy(ones_d1, aus1, (height_ones*width_ones)*sizeof(dnnType), cudaMemcpyHostToDevice);
|
||||
cudaDeviceSynchronize();
|
||||
dnnType *ones_d2;
|
||||
cudaMallocHost(&ones_d2, dim_ones*sizeof(dnnType));
|
||||
float aus2[dim_ones];
|
||||
for(int i=0; i<dim_ones; i++)
|
||||
aus2[i]=1.0f;
|
||||
cudaMemcpy(ones_d2, aus2, (dim_ones)*sizeof(dnnType), cudaMemcpyHostToDevice);
|
||||
cudaDeviceSynchronize();
|
||||
|
||||
dcn_v2_cuda_forward(input, this->data_d,
|
||||
this->bias2_d, ones_d1,
|
||||
offset, mask,
|
||||
dstData, ones_d2,
|
||||
this->kernelH, this->kernelW,
|
||||
this->strideH, this->strideW,
|
||||
this->paddingH, this->paddingW,
|
||||
1, 1,
|
||||
this->deformableGroup,
|
||||
preconv->input_dim.n, preconv->input_dim.c, preconv->input_dim.h, preconv->input_dim.w,
|
||||
this->output_dim.n, this->output_dim.c, this->output_dim.h, this->output_dim.w,
|
||||
dst_dim);
|
||||
|
||||
cudaFree(offset);
|
||||
cudaFree(mask);
|
||||
cudaFree(input);
|
||||
cudaFreeHost(ones_d1);
|
||||
cudaFreeHost(ones_d2);
|
||||
|
||||
|
||||
// dnnType *aus3;
|
||||
// cudaMallocHost(&aus3, 256*7*7*sizeof(dnnType));
|
||||
// cudaMemcpy(aus3, dstData, (256*7*7)*sizeof(dnnType), cudaMemcpyDeviceToHost);
|
||||
// checkCuda(cudaDeviceSynchronize());
|
||||
// std::cout<<"OutDim:\n";
|
||||
// this->output_dim.print();
|
||||
// std::cout<<"\n\n\nprint dstData: \n";
|
||||
// for (int i = 0 ; i < 256*7*7; i++){
|
||||
// if(i==294)
|
||||
// std::cout<<"\n\n\n";
|
||||
// std::cout<<aus3[i]<<" ";
|
||||
// }
|
||||
// std::cout<<"\n";
|
||||
// cudaFreeHost(aus3);
|
||||
|
||||
std::cout<<"srcData BN:\n";
|
||||
printDeviceVector(64, dstData);
|
||||
|
||||
dnnType alpha = dnnType(1);
|
||||
dnnType beta = dnnType(0);
|
||||
if(!batchnorm) {
|
||||
// // // bias
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(1);
|
||||
checkCUDNN( cudnnAddTensor(net->cudnnHandle,
|
||||
&alpha, biasTensorDesc, bias_d,
|
||||
&beta, dstTensorDesc, dstData) );
|
||||
} else {
|
||||
std::cout<<"LOL\n";
|
||||
alpha = dnnType(1);
|
||||
beta = dnnType(0);
|
||||
checkCUDNN( cudnnBatchNormalizationForwardInference(net->cudnnHandle,
|
||||
CUDNN_BATCHNORM_SPATIAL, &alpha, &beta,
|
||||
dstTensorDesc, dstData, dstTensorDesc,
|
||||
dstData, biasTensorDesc, //same tensor descriptor as bias
|
||||
scales_d, bias_d, mean_d, variance_d,
|
||||
CUDNN_BN_MIN_EPSILON) );
|
||||
}
|
||||
//update data dimensions
|
||||
std::cout<<"dstData BN:\n";
|
||||
printDeviceVector(64, dstData);
|
||||
dim = output_dim;
|
||||
return dstData;
|
||||
}
|
||||
|
||||
|
||||
}}
|
||||
+7
-1
@@ -8,7 +8,7 @@ namespace tk { namespace dnn {
|
||||
|
||||
LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
||||
int kh, int kw, int kl,
|
||||
std::string fname_weights, bool batchnorm) : Layer(net) {
|
||||
std::string fname_weights, bool batchnorm, bool additional_bias) : Layer(net) {
|
||||
|
||||
this->inputs = inputs;
|
||||
this->outputs = outputs;
|
||||
@@ -18,6 +18,12 @@ LayerWgs::LayerWgs(Network *net, int inputs, int outputs,
|
||||
int seek = 0;
|
||||
readBinaryFile(weights_path.c_str(), inputs*outputs*kh*kw*kl, &data_h, &data_d, seek, net->dontLoadWeights);
|
||||
seek += inputs*outputs*kh*kw*kl;
|
||||
this->additional_bias = additional_bias;
|
||||
if(additional_bias) {
|
||||
readBinaryFile(weights_path.c_str(), outputs, &bias2_h, &bias2_d, seek, net->dontLoadWeights);
|
||||
seek += outputs;
|
||||
}
|
||||
|
||||
readBinaryFile(weights_path.c_str(), outputs, &bias_h, &bias_d, seek, net->dontLoadWeights);
|
||||
|
||||
this->batchnorm = batchnorm;
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
#include <cstdio>
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include "kernels.h"
|
||||
#include <errno.h>
|
||||
|
||||
#define CUDA_KERNEL_LOOP(i, n) \
|
||||
for (int i = blockIdx.x * blockDim.x + threadIdx.x; \
|
||||
@@ -134,3 +136,109 @@ void modulated_deformable_im2col_cuda(cudaStream_t stream,
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
void dcn_v2_cuda_forward(float *input, float *weight,
|
||||
float *bias, float *ones,
|
||||
float *offset, float *mask,
|
||||
float *output, float *columns,
|
||||
int kernel_h, int kernel_w,
|
||||
const int stride_h, const int stride_w,
|
||||
const int pad_h, const int pad_w,
|
||||
const int dilation_h, const int dilation_w,
|
||||
const int deformable_group,
|
||||
const int in_n, const int in_c, const int in_h, const int in_w,
|
||||
const int out_n, const int out_c, const int out_h, const int out_w,
|
||||
const int dst_dim, cudaStream_t stream)
|
||||
{
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
cudaError_t cudaStat;
|
||||
cublasStatus_t stat;
|
||||
cublasHandle_t handle;
|
||||
stat = cublasCreate(&handle);
|
||||
if (stat != CUBLAS_STATUS_SUCCESS) {
|
||||
printf ("CUBLAS initialization failed\n");
|
||||
return;
|
||||
}
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
const int batch = in_n;
|
||||
const int channels = in_c;
|
||||
const int height = in_h;
|
||||
const int width = in_w;
|
||||
|
||||
const int channels_out = out_c;
|
||||
const int channels_kernel = in_c;
|
||||
const int kernel_h_ = kernel_h;
|
||||
const int kernel_w_ = kernel_w;
|
||||
|
||||
const int height_out = (height + 2 * pad_h - (dilation_h * (kernel_h - 1) + 1)) / stride_h + 1;
|
||||
const int width_out = (width + 2 * pad_w - (dilation_w * (kernel_w - 1) + 1)) / stride_w + 1;
|
||||
|
||||
float *input_n;
|
||||
cudaMalloc(&input_n, (in_n*in_c*in_h*in_w)*sizeof(float));
|
||||
cudaMemcpy(input_n, input, (in_n*in_c*in_h*in_w)*sizeof(float), cudaMemcpyDeviceToDevice);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
float *offset_n;
|
||||
cudaMalloc(&offset_n, ((dst_dim/3)*2)*sizeof(float));
|
||||
cudaMemcpy(offset_n, offset, ((dst_dim/3)*2)*sizeof(float), cudaMemcpyDeviceToDevice);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
float *mask_n;
|
||||
cudaMalloc(&mask_n, (dst_dim/3)*sizeof(float));
|
||||
cudaMemcpy(mask_n, mask, (dst_dim/3)*sizeof(float), cudaMemcpyDeviceToDevice);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
float *output_n;
|
||||
checkCuda(cudaMalloc(&output_n, (channels_out*height_out*width_out)*sizeof(float)));
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
long m_ = channels_out;
|
||||
long n_ = height_out * width_out;
|
||||
long k_ = 1;
|
||||
float alpha = 1.0;
|
||||
float beta = 0.0;
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
stat = cublasSgemm(handle, CUBLAS_OP_T, CUBLAS_OP_N,
|
||||
n_, m_, k_, &alpha,
|
||||
ones, k_, bias, k_,
|
||||
&beta, output_n, n_);
|
||||
if (stat != CUBLAS_STATUS_SUCCESS) {
|
||||
printf ("CUBLAS initialization failed\n");
|
||||
return ;
|
||||
}
|
||||
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
modulated_deformable_im2col_cuda(stream,
|
||||
input_n, offset_n,
|
||||
mask_n,
|
||||
1, channels, height, width,
|
||||
height_out, width_out, kernel_h, kernel_w,
|
||||
pad_h, pad_w, stride_h, stride_w, dilation_h, dilation_w,
|
||||
deformable_group, columns);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
//(k * m) x (m * n)
|
||||
// Y = WC
|
||||
long m = channels_out;
|
||||
long n = height_out * width_out;
|
||||
long k = channels * kernel_h * kernel_w;
|
||||
|
||||
alpha = 1.0;
|
||||
beta = 1.0;
|
||||
|
||||
stat = cublasSgemm(handle, CUBLAS_OP_N, CUBLAS_OP_N,
|
||||
n, m, k, &alpha,
|
||||
columns, n, weight, k,
|
||||
&beta, output_n, n);
|
||||
|
||||
cudaMemcpy(output, output_n, (n*m)*sizeof(float), cudaMemcpyDeviceToDevice);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
if (stat != CUBLAS_STATUS_SUCCESS) {
|
||||
printf ("CUBLAS initialization failed\n");
|
||||
return ;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -89,7 +89,6 @@ int checkResult(int size, dnnType *data_d, dnnType *correct_d, bool device) {
|
||||
data_h = data_d;
|
||||
correct_h = correct_d;
|
||||
}
|
||||
|
||||
int diffs = 0;
|
||||
for(int i=0; i<size; i++) {
|
||||
if(data_h[i] != data_h[i] || correct_h[i] != correct_h[i] || //nan control
|
||||
|
||||
Reference in New Issue
Block a user