mirror of
https://github.com/opencv/opencv.git
synced 2026-07-29 15:23:05 +04:00
Add new layer forward interface
Add layer forward interface with InputArrayOfArrays and OutputArrayOfArrays parameters, it allows UMat buffer to be processed and transferred in the layers. Signed-off-by: Li Peng <peng.li@intel.com>
This commit is contained in:
@@ -671,14 +671,20 @@ public:
|
||||
};
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
bool forward_ocl(std::vector<Mat*> &inputs, std::vector<Mat> &outputs, std::vector<Mat> &internals)
|
||||
bool forward_ocl(InputArrayOfArrays inps, OutputArrayOfArrays outs, OutputArrayOfArrays internals)
|
||||
{
|
||||
int group = inputs[0]->size[1] / umat_blobs[0].size[1];
|
||||
std::vector<UMat> inputs;
|
||||
std::vector<UMat> outputs;
|
||||
|
||||
inps.getUMatVector(inputs);
|
||||
outs.getUMatVector(outputs);
|
||||
|
||||
int group = inputs[0].size[1] / umat_blobs[0].size[1];
|
||||
|
||||
if (convolutionOp.empty())
|
||||
{
|
||||
OCL4DNNConvConfig config;
|
||||
config.in_shape = shape(*inputs[0]);
|
||||
config.in_shape = shape(inputs[0]);
|
||||
config.out_shape = shape(outputs[0]);
|
||||
config.kernel = kernel;
|
||||
config.pad = pad;
|
||||
@@ -690,6 +696,112 @@ public:
|
||||
convolutionOp = Ptr<OCL4DNNConvSpatial<float> >(new OCL4DNNConvSpatial<float>(config));
|
||||
}
|
||||
|
||||
int k, outCn = umat_blobs[0].size[0];
|
||||
if( weightsMat.empty() )
|
||||
{
|
||||
// prepare weightsMat where each row is aligned and has enough zero padding on the right to
|
||||
// use vectorized (i.e. with intrinsics) loops without tail processing
|
||||
Mat wm = blobs[0].reshape(1, outCn).clone();
|
||||
if( wm.step1() % VEC_ALIGN != 0 )
|
||||
{
|
||||
int newcols = (int)alignSize(wm.step1(), VEC_ALIGN);
|
||||
Mat wm_buffer = Mat(outCn, newcols, wm.type());
|
||||
Mat wm_padding = wm_buffer.colRange(wm.cols, newcols);
|
||||
wm_padding.setTo(Scalar::all(0.));
|
||||
Mat wm_aligned = wm_buffer.colRange(0, wm.cols);
|
||||
wm.copyTo(wm_aligned);
|
||||
wm = wm_aligned;
|
||||
}
|
||||
weightsMat = wm;
|
||||
|
||||
Mat biasMat = hasBias() ? blobs[1].reshape(1, outCn) : Mat();
|
||||
biasvec.resize(outCn+2);
|
||||
if( biasMat.empty() )
|
||||
{
|
||||
for( k = 0; k < outCn; k++ )
|
||||
biasvec[k] = 0.f;
|
||||
}
|
||||
else
|
||||
{
|
||||
for( k = 0; k < outCn; k++ )
|
||||
biasvec[k] = biasMat.at<float>(k);
|
||||
}
|
||||
|
||||
if( !bnorm.empty() || !scaleLayer.empty() )
|
||||
{
|
||||
Mat scale, shift, scale2, shift2;
|
||||
const float *scaleptr = 0, *shiftptr = 0;
|
||||
const float *scaleptr2 = 0, *shiftptr2 = 0;
|
||||
|
||||
if( !bnorm.empty() )
|
||||
{
|
||||
bnorm->getScaleShift(scale, shift);
|
||||
CV_Assert( scale.isContinuous() && shift.isContinuous() &&
|
||||
scale.type() == CV_32F && shift.type() == CV_32F &&
|
||||
scale.total() == (size_t)outCn &&
|
||||
shift.total() == (size_t)outCn );
|
||||
scaleptr = scale.ptr<float>();
|
||||
shiftptr = shift.ptr<float>();
|
||||
}
|
||||
if( !scaleLayer.empty() )
|
||||
{
|
||||
scale2 = scaleLayer->blobs[0];
|
||||
CV_Assert( scale2.isContinuous() && scale2.type() == CV_32F &&
|
||||
scale2.total() == (size_t)outCn );
|
||||
scaleptr2 = scale2.ptr<float>();
|
||||
if( scaleLayer->hasBias )
|
||||
{
|
||||
shift2 = scaleLayer->blobs[1];
|
||||
CV_Assert( shift2.isContinuous() && shift2.type() == CV_32F &&
|
||||
shift2.total() == (size_t)outCn );
|
||||
shiftptr2 = shift2.ptr<float>();
|
||||
}
|
||||
}
|
||||
|
||||
if (shiftptr || shiftptr2)
|
||||
fusedBias = true;
|
||||
|
||||
for( int i = 0; i < outCn; i++ )
|
||||
{
|
||||
float s1 = scaleptr ? scaleptr[i] : 1.f;
|
||||
float delta1 = shiftptr ? shiftptr[i] : 0.f;
|
||||
float s2 = scaleptr2 ? scaleptr2[i] : 1.f;
|
||||
float delta2 = shiftptr2 ? shiftptr2[i] : 0.f;
|
||||
float* w_i = weightsMat.ptr<float>(i);
|
||||
int j, wcols = weightsMat.cols;
|
||||
|
||||
for( j = 0; j < wcols; j++ )
|
||||
w_i[j] *= (s1*s2);
|
||||
|
||||
biasvec[i] = biasvec[i]*(s1*s2) + (delta1*s2 + delta2);
|
||||
}
|
||||
}
|
||||
biasvec[outCn] = biasvec[outCn+1] = biasvec[outCn-1];
|
||||
}
|
||||
|
||||
reluslope.clear();
|
||||
if( activ )
|
||||
{
|
||||
Ptr<ReLULayer> activ_relu = activ.dynamicCast<ReLULayer>();
|
||||
if( !activ_relu.empty() )
|
||||
{
|
||||
reluslope.assign(outCn+2, activ_relu->negativeSlope);
|
||||
activType = OCL4DNN_CONV_FUSED_ACTIV_RELU;
|
||||
}
|
||||
|
||||
Ptr<ChannelsPReLULayer> activ_chprelu = activ.dynamicCast<ChannelsPReLULayer>();
|
||||
if( !activ_chprelu.empty() )
|
||||
{
|
||||
const Mat& m = activ_chprelu->blobs[0];
|
||||
CV_Assert(m.isContinuous() && m.type() == CV_32F && (int)m.total() == outCn);
|
||||
const float* mdata = m.ptr<float>();
|
||||
reluslope.resize(outCn+2);
|
||||
std::copy(mdata, mdata + outCn, reluslope.begin());
|
||||
reluslope[outCn] = reluslope[outCn+1] = reluslope[outCn-1];
|
||||
activType = OCL4DNN_CONV_FUSED_ACTIV_PRELU;
|
||||
}
|
||||
}
|
||||
|
||||
if ( newWeightAndBias )
|
||||
{
|
||||
weightsMat.copyTo(umat_blobs[0]);
|
||||
@@ -723,9 +835,8 @@ public:
|
||||
newActiv = false;
|
||||
}
|
||||
|
||||
UMat inpMat, outMat;
|
||||
inpMat = inputs[0]->getUMat(ACCESS_READ);
|
||||
outMat = outputs[0].getUMat(ACCESS_WRITE);
|
||||
UMat& inpMat = inputs[0];
|
||||
UMat& outMat = outputs[0];
|
||||
int batch_size = inpMat.size[0];
|
||||
|
||||
return convolutionOp->Forward(inpMat,
|
||||
@@ -736,6 +847,18 @@ public:
|
||||
}
|
||||
#endif
|
||||
|
||||
void forward(InputArrayOfArrays inputs_arr, OutputArrayOfArrays outputs_arr, OutputArrayOfArrays internals_arr)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
CV_TRACE_ARG_VALUE(name, "name", name.c_str());
|
||||
|
||||
CV_OCL_RUN((preferableTarget == DNN_TARGET_OPENCL) &&
|
||||
OCL_PERFORMANCE_CHECK(ocl::Device::getDefault().isIntel()),
|
||||
forward_ocl(inputs_arr, outputs_arr, internals_arr))
|
||||
|
||||
Layer::forward_fallback(inputs_arr, outputs_arr, internals_arr);
|
||||
}
|
||||
|
||||
void forward(std::vector<Mat*> &inputs, std::vector<Mat> &outputs, std::vector<Mat> &internals)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
@@ -811,11 +934,6 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
if (shiftptr || shiftptr2)
|
||||
fusedBias = true;
|
||||
#endif
|
||||
|
||||
for( int i = 0; i < outCn; i++ )
|
||||
{
|
||||
float s1 = scaleptr ? scaleptr[i] : 1.f;
|
||||
@@ -841,9 +959,6 @@ public:
|
||||
if( !activ_relu.empty() )
|
||||
{
|
||||
reluslope.assign(outCn+2, activ_relu->negativeSlope);
|
||||
#ifdef HAVE_OPENCL
|
||||
activType = OCL4DNN_CONV_FUSED_ACTIV_RELU;
|
||||
#endif
|
||||
}
|
||||
|
||||
Ptr<ChannelsPReLULayer> activ_chprelu = activ.dynamicCast<ChannelsPReLULayer>();
|
||||
@@ -855,16 +970,9 @@ public:
|
||||
reluslope.resize(outCn+2);
|
||||
std::copy(mdata, mdata + outCn, reluslope.begin());
|
||||
reluslope[outCn] = reluslope[outCn+1] = reluslope[outCn-1];
|
||||
#ifdef HAVE_OPENCL
|
||||
activType = OCL4DNN_CONV_FUSED_ACTIV_PRELU;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
CV_OCL_RUN((preferableTarget == DNN_TARGET_OPENCL) &&
|
||||
OCL_PERFORMANCE_CHECK(ocl::Device::getDefault().isIntel()),
|
||||
forward_ocl(inputs, outputs, internals))
|
||||
|
||||
int nstripes = std::max(getNumThreads(), 1);
|
||||
|
||||
ParallelConv::run(*inputs[0], outputs[0], weightsMat, biasvec, reluslope,
|
||||
@@ -1173,6 +1281,14 @@ public:
|
||||
}
|
||||
};
|
||||
|
||||
void forward(InputArrayOfArrays inputs_arr, OutputArrayOfArrays outputs_arr, OutputArrayOfArrays internals_arr)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
CV_TRACE_ARG_VALUE(name, "name", name.c_str());
|
||||
|
||||
Layer::forward_fallback(inputs_arr, outputs_arr, internals_arr);
|
||||
}
|
||||
|
||||
void forward(std::vector<Mat *> &inputs, std::vector<Mat> &outputs, std::vector<Mat> &internals)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
|
||||
Reference in New Issue
Block a user