merge
This commit is contained in:
+4
-6
@@ -8,12 +8,10 @@ cuda_include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS
|
||||
cuda_add_library(kernels SHARED src/kernels/activation_elu.cu)
|
||||
|
||||
include_directories(${CMAKE_CURRENT_SOURCE_DIR}/include ${CUDA_INCLUDE_DIRS})
|
||||
add_library(tkDNN SHARED src/Layer.cpp src/LayerWgs.cpp
|
||||
src/Dense.cpp src/Activation.cpp src/Conv2d.cpp
|
||||
add_library(tkDNN SHARED src/Layer.cpp src/LayerWgs.cpp
|
||||
src/Dense.cpp src/Activation.cpp src/Conv2d.cpp src/Conv3d.cpp src/Flatten.cpp src/MulAdd.cpp src/Pooling.cpp
|
||||
src/Network.cpp src/utils.cpp)
|
||||
target_link_libraries(tkDNN kernels)
|
||||
target_link_libraries(tkDNN kernels ${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDA_TOOLKIT_ROOT_DIR}/lib/libcudnn.so)
|
||||
|
||||
add_executable(tkDNNtest tests/test.cpp)
|
||||
message(${CUDA_LIBRARIES})
|
||||
target_link_libraries(tkDNNtest tkDNN
|
||||
${CUDA_LIBRARIES} ${CUDA_CUBLAS_LIBRARIES} ${CUDA_TOOLKIT_ROOT_DIR}/lib/libcudnn.so)
|
||||
target_link_libraries(tkDNNtest tkDNN)
|
||||
@@ -0,0 +1,85 @@
|
||||
# tkDNN
|
||||
tkDNN is a Deep Neural Network library built with cuDNN primitives specifically thought to work on NVIDIA TK1 board.<br>
|
||||
The main scope is to do high performance inference on already trained models.
|
||||
Currently supports the following layers:
|
||||
|
||||
* Dense, fully interconnected
|
||||
* Activation (RELU, ELU, SIGMOID, TANH)
|
||||
* Convolutional 2D
|
||||
* Convolutional 3D
|
||||
* Max and Average Pooling
|
||||
* Flatten
|
||||
* Data preprocessing
|
||||
|
||||
## Workflow
|
||||
The recommended workflow follow these step:
|
||||
* Build and train a model in Keras (on any PC)
|
||||
* Export weights and bias
|
||||
* Define the model on tkDNN
|
||||
* Do inference (on TK1)
|
||||
|
||||
## Compile the library
|
||||
Build with cmake
|
||||
```
|
||||
mkdir build
|
||||
cd build
|
||||
cmake ..
|
||||
make
|
||||
```
|
||||
|
||||
## Test
|
||||
There is a ready to use example on *test* directory, to try it you must generate the weights with Keras
|
||||
```
|
||||
cd tests
|
||||
python test_model.py
|
||||
```
|
||||
And then execute the inference on build directory
|
||||
```
|
||||
cd build
|
||||
./tkDNNtest
|
||||
```
|
||||
this should output the same prediction as Keras.
|
||||
|
||||
## Simple example
|
||||
Here is a example of the entire workflow on a simple model.
|
||||
Using the following Keras model save it to a file
|
||||
```python
|
||||
model = Sequential()
|
||||
model.add(Reshape((20, 1), input_shape=(20)))
|
||||
model.add(Dense(256))
|
||||
model.compile()
|
||||
|
||||
# save model
|
||||
model.save("path/to/model.h5")
|
||||
```
|
||||
|
||||
After the model is created the weights can be exported for tkDNN inference
|
||||
```
|
||||
python weights_exporter model.h5 dense --output=weights/path
|
||||
```
|
||||
the exporter take as arguments, in order:
|
||||
* input model
|
||||
* layer type ["dense", "conv2d", conv3d"]
|
||||
* { layer type ["dense", "conv2d", conv3d"] for each layer to export }
|
||||
* optional argument --output define path where export weights
|
||||
|
||||
Then we can create a c++ program to do inference on tk1
|
||||
```c++
|
||||
#include<tkdnn.h> //library include
|
||||
|
||||
//Network object
|
||||
tkDNN::Network net;
|
||||
//input dimension
|
||||
tkDNN::dataDim_t dim(1, 20, 1, 1, 1);
|
||||
//Dense layer
|
||||
tkDNN::Dense d0(&net, dim, 256, "weights/path", "bias/path");
|
||||
|
||||
//here load the input data to CUDA
|
||||
//value_type is an alias of "float"
|
||||
value_type *data_d = [...]
|
||||
|
||||
//do inference
|
||||
value_type *output_d = d0.infer(dim, data_d);
|
||||
//dim will be updated with the output dimension
|
||||
```
|
||||
The result is finally stored on output_d in device memory.
|
||||
+100
-1
@@ -136,14 +136,113 @@ protected:
|
||||
value_type *dstData; //where results will be putted
|
||||
int kernelH, kernelW, strideH, strideW;
|
||||
|
||||
cudnnTensorDescriptor_t biasTensorDesc;
|
||||
cudnnFilterDescriptor_t filterDesc;
|
||||
cudnnConvolutionDescriptor_t convDesc;
|
||||
cudnnConvolutionFwdAlgo_t algo;
|
||||
cudnnTensorDescriptor_t biasTensorDesc;
|
||||
|
||||
void* workSpace;
|
||||
size_t ws_sizeInBytes;
|
||||
};
|
||||
|
||||
/**
|
||||
Convolutional 3D layer
|
||||
*/
|
||||
class Conv3d : public LayerWgs {
|
||||
|
||||
public:
|
||||
Conv3d(Network *net, dataDim_t in_dim, int out_ch,
|
||||
int kernelH, int kernelW, int kernelL,
|
||||
int strideH, int strideW, int strideL,
|
||||
const char* fname_weights, const char* fname_bias);
|
||||
virtual ~Conv3d();
|
||||
|
||||
value_type* infer(dataDim_t &dim, value_type* srcData);
|
||||
|
||||
protected:
|
||||
value_type *dstData; //where results will be putted
|
||||
int kernelH, kernelW, kernelL;
|
||||
int strideH, strideW, strideL;
|
||||
|
||||
cudnnFilterDescriptor_t filterDesc;
|
||||
cudnnConvolutionDescriptor_t convDesc;
|
||||
cudnnConvolutionFwdAlgo_t algo;
|
||||
cudnnTensorDescriptor_t biasTensorDesc;
|
||||
cudnnTensorDescriptor_t biasDstTensorDesc;
|
||||
|
||||
void* workSpace;
|
||||
size_t ws_sizeInBytes;
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
Flatten layer
|
||||
is actually a matrix transposition
|
||||
*/
|
||||
class Flatten : public Layer {
|
||||
|
||||
public:
|
||||
Flatten(Network *net, dataDim_t input_dim);
|
||||
virtual ~Flatten();
|
||||
|
||||
value_type* infer(dataDim_t &dim, value_type* srcData);
|
||||
|
||||
protected:
|
||||
value_type *dstData; //where results will be putted
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
MulAdd layer
|
||||
apply a multiplication and then an addition for each data
|
||||
*/
|
||||
class MulAdd : public Layer {
|
||||
|
||||
public:
|
||||
MulAdd(Network *net, dataDim_t input_dim, value_type mul, value_type add);
|
||||
virtual ~MulAdd();
|
||||
|
||||
value_type* infer(dataDim_t &dim, value_type* srcData);
|
||||
|
||||
protected:
|
||||
value_type mul, add;
|
||||
value_type *dstData, *add_vector; //where results will be putted
|
||||
};
|
||||
|
||||
|
||||
|
||||
/**
|
||||
Avaible pooling functions (padding on tkDNN is not supported)
|
||||
*/
|
||||
typedef enum {
|
||||
POOLING_MAX = 0,
|
||||
POOLING_AVERAGE = 1, // count for average includes padded values
|
||||
POOLING_AVERAGE_EXCLUDE_PADDING = 2 // count for average does not include padded values
|
||||
} tkdnnPoolingMode_t;
|
||||
|
||||
/**
|
||||
Pooling layer
|
||||
currenty supported only 2d pooing (also on 3d input)
|
||||
*/
|
||||
class Pooling : public Layer {
|
||||
|
||||
public:
|
||||
Pooling(Network *net, dataDim_t input_dim, int winH, int winW,
|
||||
int strideH, int strideW, tkdnnPoolingMode_t pool_mode);
|
||||
virtual ~Pooling();
|
||||
|
||||
value_type* infer(dataDim_t &dim, value_type* srcData);
|
||||
|
||||
protected:
|
||||
|
||||
cudnnPoolingDescriptor_t poolingDesc;
|
||||
|
||||
int winH, winW;
|
||||
int strideH, strideW;
|
||||
tkdnnPoolingMode_t pool_mode;
|
||||
value_type *dstData, *tmpInputData, *tmpOutputData; //where results will be putted
|
||||
bool poolOn3d;
|
||||
};
|
||||
|
||||
}
|
||||
#endif //LAYER_H
|
||||
@@ -0,0 +1,16 @@
|
||||
/**
|
||||
This is the core header of the library, it should be used only this
|
||||
*/
|
||||
#include "Network.h"
|
||||
#include "Layer.h"
|
||||
|
||||
namespace tkDNN {
|
||||
|
||||
/**
|
||||
Return the tkDNN version
|
||||
*/
|
||||
int getVersion() {
|
||||
|
||||
return 100;
|
||||
}
|
||||
}
|
||||
@@ -66,4 +66,8 @@ void readBinaryFile(const char* fname, int size, value_type** data_h, value_type
|
||||
void printDeviceVector(int size, value_type* vec_d);
|
||||
void resize(int size, value_type **data);
|
||||
|
||||
void matrixTranspose(cublasHandle_t handle, value_type* srcData, value_type* dstData, int rows, int cols);
|
||||
|
||||
void matrixMulAdd( cublasHandle_t handle, value_type* srcData, value_type* dstData,
|
||||
value_type* add_vector, int dim, value_type mul);
|
||||
#endif //UTILS_H
|
||||
+4
-2
@@ -14,12 +14,14 @@ Activation::Activation(Network *net, dataDim_t input_dim, tkdnnActivationMode_t
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(srcTensorDesc,
|
||||
net->tensorFormat,
|
||||
net->dataType,
|
||||
input_dim.n, input_dim.c,
|
||||
input_dim.n*input_dim.l,
|
||||
input_dim.c,
|
||||
input_dim.h, input_dim.w) );
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(dstTensorDesc,
|
||||
net->tensorFormat,
|
||||
net->dataType,
|
||||
input_dim.n, input_dim.c,
|
||||
input_dim.n*input_dim.l,
|
||||
input_dim.c,
|
||||
input_dim.h, input_dim.w) );
|
||||
}
|
||||
|
||||
|
||||
+157
@@ -0,0 +1,157 @@
|
||||
#include <iostream>
|
||||
|
||||
#include "Layer.h"
|
||||
|
||||
namespace tkDNN {
|
||||
|
||||
Conv3d::Conv3d( Network *net, dataDim_t in_dim, int out_ch,
|
||||
int kernelH, int kernelW, int kernelL,
|
||||
int strideH, int strideW, int strideL,
|
||||
const char* fname_weights, const char* fname_bias) :
|
||||
|
||||
LayerWgs(net, in_dim, in_dim.c, out_ch, kernelH, kernelW, kernelL,
|
||||
fname_weights, fname_bias) {
|
||||
|
||||
this->kernelH = kernelH;
|
||||
this->kernelW = kernelW;
|
||||
this->kernelL = kernelL;
|
||||
this->strideH = strideH;
|
||||
this->strideW = strideW;
|
||||
this->strideL = strideL;
|
||||
|
||||
checkCUDNN( cudnnCreateTensorDescriptor(&biasDstTensorDesc) );
|
||||
checkCUDNN( cudnnCreateFilterDescriptor(&filterDesc) );
|
||||
checkCUDNN( cudnnCreateConvolutionDescriptor(&convDesc) );
|
||||
checkCUDNN( cudnnCreateTensorDescriptor(&biasTensorDesc) );
|
||||
|
||||
int n = input_dim.n;
|
||||
int c = input_dim.c;
|
||||
int h = input_dim.h;
|
||||
int w = input_dim.w;
|
||||
int l = input_dim.l;
|
||||
|
||||
//create a tensor Nd descriptor with N = 4
|
||||
int dimA[5];
|
||||
dimA[0] = n; dimA[1] = c; dimA[2] = h; dimA[3] = w; dimA[4] = l;
|
||||
int strideA[5];
|
||||
strideA[0] = c*h*w*l;
|
||||
strideA[1] = h*w*l;
|
||||
strideA[2] = w*l;
|
||||
strideA[3] = l;
|
||||
strideA[4] = 1;
|
||||
checkCUDNN( cudnnSetTensorNdDescriptor(srcTensorDesc,
|
||||
net->dataType, 5, dimA, strideA));
|
||||
|
||||
//filter descriptor
|
||||
int filterDim[5];
|
||||
filterDim[0] = out_ch;
|
||||
filterDim[1] = in_dim.c;
|
||||
filterDim[2] = kernelH;
|
||||
filterDim[3] = kernelW;
|
||||
filterDim[4] = kernelL;
|
||||
checkCUDNN( cudnnSetFilterNdDescriptor(filterDesc,
|
||||
net->dataType, 5, filterDim));
|
||||
|
||||
//convolutional descriptor
|
||||
int padA[3] = {0, 0, 0};
|
||||
int filterStride[3] = {strideH, strideW, strideL};
|
||||
int upscale[3] = {1, 1, 1};
|
||||
checkCUDNN( cudnnSetConvolutionNdDescriptor(convDesc, 3,
|
||||
padA, filterStride, upscale, CUDNN_CROSS_CORRELATION));
|
||||
|
||||
|
||||
//get output dimension
|
||||
int outputDim[5];
|
||||
checkCUDNN(cudnnGetConvolutionNdForwardOutputDim(convDesc, srcTensorDesc, filterDesc, 5, outputDim));
|
||||
n = outputDim[0];
|
||||
c = outputDim[1];
|
||||
h = outputDim[2];
|
||||
w = outputDim[3];
|
||||
l = outputDim[4];
|
||||
|
||||
//destination sensor
|
||||
int outputStride[5];
|
||||
outputStride[0] = c*h*w*l;
|
||||
outputStride[1] = h*w*l;
|
||||
outputStride[2] = w*l;
|
||||
outputStride[3] = l;
|
||||
outputStride[4] = 1;
|
||||
checkCUDNN( cudnnSetTensorNdDescriptor(dstTensorDesc, net->dataType,
|
||||
5, outputDim, outputStride));
|
||||
|
||||
|
||||
//conv algo
|
||||
checkCUDNN( cudnnGetConvolutionForwardAlgorithm(net->cudnnHandle,
|
||||
srcTensorDesc, filterDesc, convDesc, dstTensorDesc,
|
||||
CUDNN_CONVOLUTION_FWD_PREFER_FASTEST, 0, &algo) );
|
||||
|
||||
checkCUDNN( cudnnGetConvolutionForwardWorkspaceSize(net->cudnnHandle,
|
||||
srcTensorDesc,
|
||||
filterDesc,
|
||||
convDesc,
|
||||
dstTensorDesc,
|
||||
algo,
|
||||
&ws_sizeInBytes) );
|
||||
if (ws_sizeInBytes!=0)
|
||||
checkCuda( cudaMalloc(&workSpace, ws_sizeInBytes) );
|
||||
|
||||
|
||||
// bias on N dimensional is not SUPPORTED so i have to use 2d method
|
||||
//the trick is to upscale the 2d matrix width by the factor of 3d thickness
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(biasDstTensorDesc,
|
||||
net->tensorFormat, net->dataType, n, c, h*l, w) );
|
||||
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(biasTensorDesc,
|
||||
net->tensorFormat, net->dataType,
|
||||
1, c, 1, 1) );
|
||||
|
||||
output_dim.n = n;
|
||||
output_dim.c = c;
|
||||
output_dim.h = h;
|
||||
output_dim.w = w;
|
||||
output_dim.l = l;
|
||||
|
||||
//allocate data for infer result
|
||||
checkCuda( cudaMalloc(&dstData, output_dim.tot()*sizeof(value_type)) );
|
||||
}
|
||||
|
||||
Conv3d::~Conv3d() {
|
||||
|
||||
checkCUDNN( cudnnDestroyFilterDescriptor(filterDesc) );
|
||||
checkCUDNN( cudnnDestroyConvolutionDescriptor(convDesc) );
|
||||
checkCUDNN( cudnnDestroyTensorDescriptor(biasTensorDesc) );
|
||||
checkCUDNN( cudnnDestroyTensorDescriptor(biasDstTensorDesc) );
|
||||
|
||||
if (ws_sizeInBytes!=0)
|
||||
checkCuda( cudaFree(workSpace) );
|
||||
|
||||
checkCuda( cudaFree(dstData) );
|
||||
}
|
||||
|
||||
value_type* Conv3d::infer(dataDim_t &dim, value_type* srcData) {
|
||||
|
||||
|
||||
// convolution
|
||||
value_type alpha = value_type(1);
|
||||
value_type beta = value_type(0);
|
||||
checkCUDNN( cudnnConvolutionForward(net->cudnnHandle,
|
||||
&alpha, srcTensorDesc, srcData, filterDesc,
|
||||
data_d, convDesc, algo, workSpace, ws_sizeInBytes,
|
||||
&beta, dstTensorDesc, dstData) );
|
||||
|
||||
|
||||
// bias
|
||||
alpha = value_type(1);
|
||||
beta = value_type(1);
|
||||
checkCUDNN( cudnnAddTensor(net->cudnnHandle, CUDNN_ADD_SAME_C,
|
||||
&alpha, biasTensorDesc, bias_d,
|
||||
&beta, biasDstTensorDesc, dstData) );
|
||||
|
||||
//update data dimensions
|
||||
dim = output_dim;
|
||||
|
||||
return dstData;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
#include <iostream>
|
||||
|
||||
#include "Layer.h"
|
||||
#include "kernels.h"
|
||||
|
||||
namespace tkDNN {
|
||||
|
||||
Flatten::Flatten(Network *net, dataDim_t input_dim) :
|
||||
Layer(net, input_dim) {
|
||||
|
||||
checkCuda( cudaMalloc(&dstData, input_dim.tot()*sizeof(value_type)) );
|
||||
|
||||
output_dim.n = 1;
|
||||
output_dim.c = input_dim.tot();
|
||||
output_dim.h = 1;
|
||||
output_dim.w = 1;
|
||||
output_dim.l = 1;
|
||||
|
||||
}
|
||||
|
||||
Flatten::~Flatten() {
|
||||
|
||||
checkCuda( cudaFree(dstData) );
|
||||
}
|
||||
|
||||
value_type* Flatten::infer(dataDim_t &dim, value_type* srcData) {
|
||||
|
||||
//transpose per channel
|
||||
matrixTranspose(net->cublasHandle, srcData, dstData, dim.c, dim.h*dim.w*dim.l);
|
||||
|
||||
//update data dimensions
|
||||
dim = output_dim;
|
||||
|
||||
return dstData;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
#include <iostream>
|
||||
|
||||
#include "Layer.h"
|
||||
#include "kernels.h"
|
||||
|
||||
namespace tkDNN {
|
||||
|
||||
MulAdd::MulAdd(Network *net, dataDim_t input_dim, value_type mul, value_type add) :
|
||||
Layer(net, input_dim) {
|
||||
|
||||
this->mul = mul;
|
||||
this->add = add;
|
||||
|
||||
int size = input_dim.tot();
|
||||
|
||||
// create a vector with all value setted to add
|
||||
value_type *add_vector_h = new value_type[size];
|
||||
for(int i=0; i<size; i++)
|
||||
add_vector_h[i] = add;
|
||||
|
||||
checkCuda( cudaMalloc(&add_vector, size*sizeof(value_type)));
|
||||
checkCuda( cudaMemcpy(add_vector, add_vector_h, size*sizeof(value_type), cudaMemcpyHostToDevice));
|
||||
delete [] add_vector_h;
|
||||
|
||||
|
||||
checkCuda( cudaMalloc(&dstData, input_dim.tot()*sizeof(value_type)) );
|
||||
}
|
||||
|
||||
MulAdd::~MulAdd() {
|
||||
|
||||
checkCuda( cudaFree(add_vector) );
|
||||
checkCuda( cudaFree(dstData) );
|
||||
}
|
||||
|
||||
value_type* MulAdd::infer(dataDim_t &dim, value_type* srcData) {
|
||||
|
||||
matrixMulAdd(net->cublasHandle, srcData, dstData, add_vector, input_dim.tot(), mul);
|
||||
|
||||
//update data dimensions
|
||||
dim = output_dim;
|
||||
|
||||
return dstData;
|
||||
}
|
||||
|
||||
}
|
||||
+5
-1
@@ -1,5 +1,6 @@
|
||||
#include <iostream>
|
||||
|
||||
#include "tkdnn.h"
|
||||
#include "Network.h"
|
||||
#include "Layer.h"
|
||||
|
||||
@@ -7,7 +8,10 @@ namespace tkDNN {
|
||||
|
||||
Network::Network() {
|
||||
|
||||
std::cout<<"New NETWORK with CUDNN v"<<float(cudnnGetVersion())/1000<<"\n";
|
||||
float tk_ver = float(tkDNN::getVersion())/1000;
|
||||
float cu_ver = float(cudnnGetVersion())/1000;
|
||||
|
||||
std::cout<<"New NETWORK (tkDNN v"<<tk_ver<<", CUDNN v"<<cu_ver<<")\n";
|
||||
dataType = CUDNN_DATA_FLOAT;
|
||||
tensorFormat = CUDNN_TENSOR_NCHW;
|
||||
|
||||
|
||||
+111
@@ -0,0 +1,111 @@
|
||||
#include <iostream>
|
||||
|
||||
#include "Layer.h"
|
||||
#include "kernels.h"
|
||||
|
||||
namespace tkDNN {
|
||||
|
||||
Pooling::Pooling( Network *net, dataDim_t input_dim,
|
||||
int winH, int winW, int strideH, int strideW, tkdnnPoolingMode_t pool_mode) :
|
||||
Layer(net, input_dim) {
|
||||
|
||||
|
||||
if(winH != strideH || winW != strideW)
|
||||
FatalError("stride pooling not yet implemented");
|
||||
|
||||
this->winH = winH;
|
||||
this->winW = winW;
|
||||
this->strideH = strideH;
|
||||
this->strideW = strideW;
|
||||
this->pool_mode = pool_mode;
|
||||
|
||||
checkCUDNN( cudnnCreatePoolingDescriptor(&poolingDesc) );
|
||||
|
||||
int n = input_dim.n;
|
||||
int c = input_dim.c;
|
||||
int h = input_dim.h;
|
||||
int w = input_dim.w;
|
||||
int l = input_dim.l;
|
||||
|
||||
poolOn3d = false;
|
||||
|
||||
if(l > 1) {
|
||||
poolOn3d = true;
|
||||
|
||||
if(n != 1)
|
||||
FatalError("N value on 3d pool must be 1");
|
||||
|
||||
//use batch as l
|
||||
n = l;
|
||||
}
|
||||
|
||||
checkCUDNN( cudnnSetPooling2dDescriptor(poolingDesc, cudnnPoolingMode_t(pool_mode),
|
||||
winH, winW, 0, 0, strideH, strideW) );
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(srcTensorDesc,
|
||||
net->tensorFormat, net->dataType, n, c, h, w) );
|
||||
|
||||
//get out dim
|
||||
h = h / winH; w = w / winW;
|
||||
|
||||
checkCUDNN( cudnnSetTensor4dDescriptor(dstTensorDesc,
|
||||
net->tensorFormat, net->dataType, n, c, h, w) );
|
||||
|
||||
|
||||
output_dim.n = n;
|
||||
output_dim.c = c;
|
||||
output_dim.h = h;
|
||||
output_dim.w = w;
|
||||
output_dim.l = l;
|
||||
|
||||
checkCuda( cudaMalloc(&dstData, output_dim.tot()*sizeof(value_type)) );
|
||||
|
||||
//pool on 3d data need transposition at the enter and on the exit
|
||||
//allocate for initial and final transposition
|
||||
if(poolOn3d) {
|
||||
output_dim.n = 1;
|
||||
|
||||
checkCuda( cudaMalloc(&tmpInputData, input_dim.tot()*sizeof(value_type)) );
|
||||
checkCuda( cudaMalloc(&tmpOutputData, output_dim.tot()*sizeof(value_type)) );
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
Pooling::~Pooling() {
|
||||
|
||||
if(poolOn3d) {
|
||||
checkCuda( cudaFree(tmpInputData) );
|
||||
checkCuda( cudaFree(tmpOutputData) );
|
||||
}
|
||||
|
||||
checkCUDNN( cudnnDestroyPoolingDescriptor(poolingDesc) );
|
||||
checkCuda( cudaFree(dstData) );
|
||||
}
|
||||
|
||||
value_type* Pooling::infer(dataDim_t &dim, value_type* srcData) {
|
||||
|
||||
value_type *poolSrc = srcData;
|
||||
value_type *poolDst = dstData;
|
||||
|
||||
if(poolOn3d) {
|
||||
matrixTranspose(net->cublasHandle, srcData, tmpInputData, dim.h*dim.w*dim.c, dim.l);
|
||||
poolSrc = tmpInputData;
|
||||
poolDst = tmpOutputData;
|
||||
}
|
||||
|
||||
value_type alpha = value_type(1);
|
||||
value_type beta = value_type(0);
|
||||
checkCUDNN( cudnnPoolingForward(net->cudnnHandle, poolingDesc,
|
||||
&alpha, srcTensorDesc, poolSrc,
|
||||
&beta, dstTensorDesc, poolDst) );
|
||||
|
||||
//update dim
|
||||
dim = output_dim;
|
||||
|
||||
if(poolOn3d)
|
||||
matrixTranspose(net->cublasHandle, tmpOutputData, dstData, dim.l, dim.h*dim.w*dim.c);
|
||||
|
||||
return dstData;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -9,7 +9,7 @@
|
||||
__global__
|
||||
void activation_elu(value_type *input, value_type *output, int size) {
|
||||
|
||||
int i = threadIdx.x*(blockIdx.x +1);
|
||||
int i = blockDim.x*blockIdx.x + threadIdx.x;
|
||||
|
||||
if(i<size) {
|
||||
value_type k0, k1;
|
||||
@@ -30,6 +30,9 @@ void activation_elu(value_type *input, value_type *output, int size) {
|
||||
*/
|
||||
void activationELUForward(value_type* srcData, value_type* dstData, int size)
|
||||
{
|
||||
activation_elu<<<(size+255)/256, 256>>>(srcData, dstData, size);
|
||||
int blocks = (size+255)/256;
|
||||
int threads = 256;
|
||||
|
||||
activation_elu<<<blocks, threads>>>(srcData, dstData, size);
|
||||
checkCuda( cudaDeviceSynchronize() );
|
||||
}
|
||||
@@ -42,4 +42,25 @@ void resize(int size, value_type **data)
|
||||
if (*data != NULL)
|
||||
checkCuda( cudaFree(*data) );
|
||||
checkCuda( cudaMalloc(data, size*sizeof(value_type)) );
|
||||
}
|
||||
|
||||
void matrixTranspose(cublasHandle_t handle, value_type* srcData, value_type* dstData, int rows, int cols) {
|
||||
|
||||
value_type *A = srcData, *clone = dstData;
|
||||
int m = rows, n= cols;
|
||||
checkCuda( cudaMemcpy(clone, A, m*n*sizeof(value_type), cudaMemcpyDeviceToDevice));
|
||||
|
||||
float const alpha(1.0);
|
||||
float const beta(0.0);
|
||||
checkERROR( cublasSgeam( handle, CUBLAS_OP_T, CUBLAS_OP_N, m, n, &alpha, A, n, &beta, A, m, clone, m ));
|
||||
}
|
||||
|
||||
void matrixMulAdd( cublasHandle_t handle, value_type* srcData, value_type* dstData,
|
||||
value_type* add_vector, int dim, value_type mul) {
|
||||
|
||||
checkCuda( cudaMemcpy(dstData, add_vector, dim*sizeof(value_type), cudaMemcpyDeviceToDevice));
|
||||
|
||||
value_type alpha = mul;
|
||||
checkERROR( cublasSaxpy(handle, dim, &alpha, srcData, 1, dstData, 1));
|
||||
|
||||
}
|
||||
@@ -1,45 +0,0 @@
|
||||
import keras
|
||||
import numpy as np
|
||||
import pickle
|
||||
from keras.models import Sequential
|
||||
from keras.layers import Input, Dense, Activation, Flatten, Dropout, ELU, Reshape
|
||||
from keras.layers.convolutional import Convolution2D, Convolution3D
|
||||
from keras.layers.pooling import MaxPooling2D, MaxPooling3D
|
||||
from keras.models import Sequential, Model
|
||||
from keras.layers import Cropping2D
|
||||
import keras.backend.tensorflow_backend as KTF
|
||||
from weights_exporter import *
|
||||
|
||||
def dense_model():
|
||||
model = Sequential()
|
||||
|
||||
model.add(Reshape((10, 10, 1), input_shape=(10, 10)))
|
||||
model.add(Convolution2D(2, (4, 4), subsample=(2, 2),
|
||||
bias_initializer='random_uniform'))
|
||||
model.add(ELU())
|
||||
model.add(Convolution2D(4, (2, 2), subsample=(1, 1),
|
||||
bias_initializer='random_uniform', activation="relu"))
|
||||
sgd = keras.optimizers.Adam(lr=1e-4, decay=1e-8)
|
||||
model.compile(optimizer=sgd, loss="mse")
|
||||
return model
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
print "DATA FORMAT: ", keras.backend.image_data_format()
|
||||
|
||||
model = dense_model()
|
||||
wg = model.get_weights()
|
||||
export_conv2d("conv0", wg[0], wg[1])
|
||||
export_conv2d("conv1", wg[2], wg[3])
|
||||
|
||||
grid = np.random.rand(10,10)
|
||||
X = grid[None,:,:]
|
||||
i = np.array(grid.flatten(), dtype=np.float32)
|
||||
print i
|
||||
i.tofile("input.bin", format="f")
|
||||
print "Input: ", X
|
||||
|
||||
r = model.predict( X, batch_size=1)
|
||||
print np.shape(r)
|
||||
print "Result: ", r
|
||||
print "Result shape: ", np.shape(r)
|
||||
+37
-8
@@ -1,23 +1,40 @@
|
||||
#include<iostream>
|
||||
#include "Layer.h"
|
||||
#include "tkdnn.h"
|
||||
|
||||
const char *input_bin = "../tests/input.bin";
|
||||
const char *c0_bin = "../tests/conv0.bin";
|
||||
const char *c0_bias_bin = "../tests/conv0.bias.bin";
|
||||
const char *c1_bin = "../tests/conv1.bin";
|
||||
const char *c1_bias_bin = "../tests/conv1.bias.bin";
|
||||
const char *d2_bin = "../tests/dense2.bin";
|
||||
const char *d2_bias_bin = "../tests/dense2.bias.bin";
|
||||
const char *c2_bin = "../tests/conv2.bin";
|
||||
const char *c2_bias_bin = "../tests/conv2.bias.bin";
|
||||
const char *d3_bin = "../tests/dense3.bin";
|
||||
const char *d3_bias_bin = "../tests/dense3.bias.bin";
|
||||
const char *d4_bin = "../tests/dense4.bin";
|
||||
const char *d4_bias_bin = "../tests/dense4.bias.bin";
|
||||
const char *d5_bin = "../tests/dense5.bin";
|
||||
const char *d5_bias_bin = "../tests/dense5.bias.bin";
|
||||
|
||||
int main() {
|
||||
|
||||
// Network layout
|
||||
tkDNN::Network net;
|
||||
tkDNN::dataDim_t dim(1, 1, 10, 10);
|
||||
tkDNN::Conv2d c0 (&net, dim, 2, 4, 4, 2, 2, c0_bin, c0_bias_bin);
|
||||
tkDNN::dataDim_t dim(1, 1, 100, 100, 4);
|
||||
tkDNN::MulAdd m0 (&net, dim, 2, -1);
|
||||
tkDNN::Conv3d c0 (&net, m0.output_dim, 16, 8, 8, 2, 4, 4, 1, c0_bin, c0_bias_bin);
|
||||
tkDNN::Activation a0 (&net, c0.output_dim, tkDNN::ACTIVATION_ELU);
|
||||
tkDNN::Conv2d c1 (&net, a0.output_dim, 4, 2, 2, 1, 1, c1_bin, c1_bias_bin);
|
||||
tkDNN::Activation a1 (&net, c1.output_dim, tkDNN::ACTIVATION_RELU);
|
||||
tkDNN::Pooling p0 (&net, a0.output_dim, 2, 2, 2, 2, tkDNN::POOLING_AVERAGE);
|
||||
tkDNN::Conv3d c1 (&net, p0.output_dim, 16, 4, 4, 2, 2, 2, 1, c1_bin, c1_bias_bin);
|
||||
tkDNN::Activation a1 (&net, c1.output_dim, tkDNN::ACTIVATION_ELU);
|
||||
tkDNN::Conv3d c2 (&net, a1.output_dim, 24, 3, 3, 2, 1, 1, 1, c2_bin, c2_bias_bin);
|
||||
tkDNN::Activation a2 (&net, c2.output_dim, tkDNN::ACTIVATION_ELU);
|
||||
tkDNN::Flatten f2 (&net, a2.output_dim);
|
||||
tkDNN::Dense d3 (&net, f2.output_dim, 256, d3_bin, d3_bias_bin);
|
||||
tkDNN::Activation a3 (&net, d3.output_dim, tkDNN::ACTIVATION_ELU);
|
||||
tkDNN::Dense d4 (&net, a3.output_dim, 32, d4_bin, d4_bias_bin);
|
||||
tkDNN::Activation a4 (&net, d4.output_dim, tkDNN::ACTIVATION_RELU);
|
||||
tkDNN::Dense d5 (&net, a4.output_dim, 2, d5_bin, d5_bias_bin);
|
||||
|
||||
|
||||
// Load input
|
||||
value_type *data;
|
||||
@@ -32,13 +49,25 @@ int main() {
|
||||
data = net.infer(dim, data); dim.print();
|
||||
|
||||
/*
|
||||
//old Inference method
|
||||
//old inference
|
||||
data = m0.infer(dim, data); dim.print();
|
||||
data = c0.infer(dim, data); dim.print();
|
||||
data = a0.infer(dim, data); dim.print();
|
||||
data = p0.infer(dim, data); dim.print();
|
||||
data = c1.infer(dim, data); dim.print();
|
||||
data = a1.infer(dim, data); dim.print();
|
||||
data = c2.infer(dim, data); dim.print();
|
||||
data = a2.infer(dim, data); dim.print();
|
||||
data = f2.infer(dim, data); dim.print();
|
||||
data = d3.infer(dim, data); dim.print();
|
||||
data = a3.infer(dim, data); dim.print();
|
||||
data = d4.infer(dim, data); dim.print();
|
||||
data = a4.infer(dim, data); dim.print();
|
||||
data = d5.infer(dim, data); dim.print();
|
||||
*/
|
||||
|
||||
TIMER_STOP
|
||||
|
||||
// Print result
|
||||
printDeviceVector(dim.tot(), data);
|
||||
return 0;
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
import keras
|
||||
import numpy as np
|
||||
from keras.models import Sequential
|
||||
from keras.layers import Input, Dense, Activation, Flatten, Dropout, ELU, Reshape, Lambda
|
||||
from keras.layers.convolutional import Convolution2D, Convolution3D
|
||||
from keras.layers.pooling import MaxPooling2D, MaxPooling3D, AveragePooling3D
|
||||
from keras.models import Sequential, Model
|
||||
from keras.layers import Cropping2D
|
||||
import keras.backend.tensorflow_backend as KTF
|
||||
from weights_exporter import *
|
||||
|
||||
def dense_model():
|
||||
model = Sequential()
|
||||
model.add(Reshape((100, 100, 4, 1), input_shape=(100, 100, 4)))
|
||||
model.add(Lambda(lambda x: 2*x - 1.,
|
||||
batch_input_shape=(1, 100, 100, 4), # 100by100by2
|
||||
output_shape=(100, 100, 4, 1))) # 100by100by2
|
||||
model.add(Convolution3D(16, kernel_size=(8, 8, 2), subsample=(4, 4, 1), border_mode="valid",
|
||||
bias_initializer="random_uniform"))
|
||||
model.add(ELU())
|
||||
model.add(AveragePooling3D(pool_size=(2, 2, 1)))
|
||||
model.add(Convolution3D(16, kernel_size=(4, 4, 2), subsample=(2, 2, 1), border_mode="valid",
|
||||
bias_initializer="random_uniform"))
|
||||
model.add(ELU())
|
||||
model.add(Convolution3D(24, kernel_size=(3, 3, 2), subsample=(1, 1, 1), border_mode="valid",
|
||||
bias_initializer="random_uniform"))
|
||||
model.add(ELU())
|
||||
model.add(Flatten())
|
||||
model.add(Dense(256, bias_initializer="random_uniform"))
|
||||
model.add(ELU())
|
||||
model.add(Dense(32, activation="relu", bias_initializer="random_uniform"))
|
||||
model.add(Dense(2, bias_initializer="random_uniform"))
|
||||
|
||||
sgd = keras.optimizers.Adam(lr=1e-4, decay=1e-8)
|
||||
model.compile(optimizer=sgd, loss="mse")
|
||||
return model
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
print "DATA FORMAT: ", keras.backend.image_data_format()
|
||||
|
||||
model = dense_model()
|
||||
wg = model.get_weights()
|
||||
export_conv3d("conv0", wg[0], wg[1])
|
||||
export_conv3d("conv1", wg[2], wg[3])
|
||||
export_conv3d("conv2", wg[4], wg[5])
|
||||
export_dense ("dense3", wg[6], wg[7])
|
||||
export_dense ("dense4", wg[8], wg[9])
|
||||
export_dense ("dense5", wg[10], wg[11])
|
||||
|
||||
grid = np.random.rand(100, 100,4)
|
||||
X = grid[None,:,:]
|
||||
i = np.array(grid.flatten(), dtype=np.float32)
|
||||
print i
|
||||
i.tofile("input.bin", format="f")
|
||||
print "Input: ", X
|
||||
|
||||
r = model.predict( X, batch_size=1)
|
||||
print np.shape(r)
|
||||
print "Result: ", r
|
||||
print "Result shape: ", np.shape(r)
|
||||
Reference in New Issue
Block a user