1
0
mirror of https://github.com/opencv/opencv.git synced 2026-07-28 23:03:03 +04:00

Merge pull request #14827 from YashasSamaga:cuda4dnn-csl-low

CUDA backend for the DNN module

* stub cuda4dnn design

* minor fixes for tests and doxygen

* add csl public api directory to module headers

* add low-level CSL components

* add high-level CSL components

* integrate csl::Tensor into backbone code

* switch to CPU iff unsupported; otherwise, fail on error

* add fully connected layer

* add softmax layer

* add activation layers

* support arbitary rank TensorDescriptor

* pass input wrappers to `initCUDA()`

* add 1d/2d/3d-convolution

* add pooling layer

* reorganize and refactor code

* fixes for gcc, clang and doxygen; remove cxx14/17 code

* add blank_layer

* add LRN layer

* add rounding modes for pooling layer

* split tensor.hpp into tensor.hpp and tensor_ops.hpp

* add concat layer

* add scale layer

* add batch normalization layer

* split math.cu into activations.cu and math.hpp

* add eltwise layer

* add flatten layer

* add tensor transform api

* add asymmetric padding support for convolution layer

* add reshape layer

* fix rebase issues

* add permute layer

* add padding support for concat layer

* refactor and reorganize code

* add normalize layer

* optimize bias addition in scale layer

* add prior box layer

* fix and optimize normalize layer

* add asymmetric padding support for pooling layer

* add event API

* improve pooling performance for some padding scenarios

* avoid over-allocation of compute resources to kernels

* improve prior box performance

* enable layer fusion

* add const layer

* add resize layer

* add slice layer

* add padding layer

* add deconvolution layer

* fix channelwise  ReLU initialization

* add vector traits

* add vectorized versions of relu, clipped_relu, power

* add vectorized concat kernels

* improve concat_with_offsets performance

* vectorize scale and bias kernels

* add support for multi-billion element tensors

* vectorize prior box kernels

* fix address alignment check

* improve bias addition performance of conv/deconv/fc layers

* restructure code for supporting multiple targets

* add DNN_TARGET_CUDA_FP64

* add DNN_TARGET_FP16

* improve vectorization

* add region layer

* improve tensor API, add dynamic ranks

1. use ManagedPtr instead of a Tensor in backend wrapper
2. add new methods to tensor classes
  - size_range: computes the combined size of for a given axis range
  - tensor span/view can be constructed from a raw pointer and shape
3. the tensor classes can change their rank at runtime (previously rank was fixed at compile-time)
4. remove device code from tensor classes (as they are unused)
5. enforce strict conditions on tensor class APIs to improve debugging ability

* fix parametric relu activation

* add squeeze/unsqueeze tensor API

* add reorg layer

* optimize permute and enable 2d permute

* enable 1d and 2d slice

* add split layer

* add shuffle channel layer

* allow tensors of different ranks in reshape primitive

* patch SliceOp to allow Crop Layer

* allow extra shape inputs in reshape layer

* use `std::move_backward` instead of `std::move` for insert in resizable_static_array

* improve workspace management

* add spatial LRN

* add nms (cpu) to region layer

* add max pooling with argmax ( and a fix to limits.hpp)

* add max unpooling layer

* rename DNN_TARGET_CUDA_FP32 to DNN_TARGET_CUDA

* update supportBackend to be more rigorous

* remove stray include from preventing non-cuda build

* include op_cuda.hpp outside condition #if

* refactoring, fixes and many optimizations

* drop DNN_TARGET_CUDA_FP64

* fix gcc errors

* increase max. tensor rank limit to six

* add Interp layer

* drop custom layers; use BackendNode

* vectorize activation kernels

* fixes for gcc

* remove wrong assertion

* fix broken assertion in unpooling primitive

* fix build errors in non-CUDA build

* completely remove workspace from public API

* fix permute layer

* enable accuracy and perf. tests for DNN_TARGET_CUDA

* add asynchronous forward

* vectorize eltwise ops

* vectorize fill kernel

* fixes for gcc

* remove CSL headers from public API

* remove csl header source group from cmake

* update min. cudnn version in cmake

* add numerically stable FP32 log1pexp

* refactor code

* add FP16 specialization to cudnn based tensor addition

* vectorize scale1 and bias1 + minor refactoring

* fix doxygen build

* fix invalid alignment assertion

* clear backend wrappers before allocateLayers

* ignore memory lock failures

* do not allocate internal blobs

* integrate NVTX

* add numerically stable half precision log1pexp

* fix indentation, following coding style,  improve docs

* remove accidental modification of IE code

* Revert "add asynchronous forward"

This reverts commit 1154b9da9da07e9b52f8a81bdcea48cf31c56f70.

* [cmake] throw error for unsupported CC versions

* fix rebase issues

* add more docs, refactor code, fix bugs

* minor refactoring and fixes

* resolve warnings/errors from clang

* remove haveCUDA() checks from supportBackend()

* remove NVTX integration

* changes based on review comments

* avoid exception when no CUDA device is present

* add color code for CUDA in Net::dump
This commit is contained in:
Yashas Samaga B L
2019-10-21 16:58:00 +05:30
committed by Alexander Alekhin
parent 8ec6544624
commit 613c12e590
122 changed files with 13024 additions and 99 deletions
+148 -3
View File
@@ -42,6 +42,7 @@
#include "../precomp.hpp"
#include "layers_common.hpp"
#include "../op_cuda.hpp"
#include "../op_halide.hpp"
#include "../op_inf_engine.hpp"
#include "../op_vkcom.hpp"
@@ -55,6 +56,12 @@
using namespace cv::dnn::ocl4dnn;
#endif
#ifdef HAVE_CUDA
#include "../cuda4dnn/primitives/convolution.hpp"
#include "../cuda4dnn/primitives/transpose_convolution.hpp"
using namespace cv::dnn::cuda4dnn;
#endif
namespace cv
{
namespace dnn
@@ -253,6 +260,15 @@ public:
virtual bool supportBackend(int backendId) CV_OVERRIDE
{
if (backendId == DNN_BACKEND_CUDA)
{
/* only convolution 2d and 3d supported */
if(kernel_size.size() == 2 || kernel_size.size() == 3)
return true;
return false;
}
#ifdef HAVE_INF_ENGINE
if (backendId == DNN_BACKEND_INFERENCE_ENGINE)
{
@@ -491,8 +507,6 @@ public:
return Ptr<BackendNode>();
}
virtual Ptr<BackendNode> initHalide(const std::vector<Ptr<BackendWrapper> > &inputs) CV_OVERRIDE
{
#ifdef HAVE_HALIDE
@@ -1281,6 +1295,66 @@ public:
kernel_size, strides, pads_begin, pads_end, dilations, activ.get(), ngroups, nstripes);
}
#ifdef HAVE_CUDA
Ptr<BackendNode> initCUDA(
void *context_,
const std::vector<Ptr<BackendWrapper>>& inputs,
const std::vector<Ptr<BackendWrapper>>& outputs
) override
{
auto context = reinterpret_cast<csl::CSLContext*>(context_);
CV_Assert(inputs.size() == 1);
auto input_wrapper = inputs[0].dynamicCast<CUDABackendWrapper>();
auto input_shape = input_wrapper->getShape();
CV_Assert(outputs.size() == 1);
auto output_wrapper = outputs[0].dynamicCast<CUDABackendWrapper>();
auto output_shape = output_wrapper->getShape();
const auto output_feature_maps = blobs[0].size[0];
const auto input_feature_maps = input_shape[1];
const auto input_feature_maps_per_group = blobs[0].size[1];
const auto groups = input_feature_maps / input_feature_maps_per_group;
ConvolutionConfiguration config;
config.kernel_size.assign(std::begin(kernel_size), std::end(kernel_size));
config.dilations.assign(std::begin(dilations), std::end(dilations));
config.strides.assign(std::begin(strides), std::end(strides));
if (padMode.empty())
{
config.padMode = ConvolutionConfiguration::PaddingMode::MANUAL;
config.pads_begin.assign(std::begin(pads_begin), std::end(pads_begin));
config.pads_end.assign(std::begin(pads_end), std::end(pads_end));
}
else if (padMode == "VALID")
{
config.padMode = ConvolutionConfiguration::PaddingMode::VALID;
}
else if (padMode == "SAME")
{
config.padMode = ConvolutionConfiguration::PaddingMode::SAME;
}
else
{
CV_Error(Error::StsNotImplemented, padMode + " padding mode not supported by ConvolutionLayer");
}
config.input_shape.assign(std::begin(input_shape), std::end(input_shape));
config.output_shape.assign(std::begin(output_shape), std::end(output_shape));
config.groups = groups;
Mat filtersMat = fusedWeights ? weightsMat : blobs[0];
Mat biasMat = (hasBias() || fusedBias) ? Mat(output_feature_maps, 1, CV_32F, biasvec.data()) : Mat();
if (countNonZero(biasMat) == 0)
biasMat = Mat();
return make_cuda_node<cuda4dnn::ConvolutionOp>(
preferableTarget, std::move(context->stream), std::move(context->cudnn_handle), config, filtersMat, biasMat);
}
#endif
virtual int64 getFLOPS(const std::vector<MatShape> &inputs,
const std::vector<MatShape> &outputs) const CV_OVERRIDE
{
@@ -1323,6 +1397,15 @@ public:
virtual bool supportBackend(int backendId) CV_OVERRIDE
{
if (backendId == DNN_BACKEND_CUDA)
{
/* only deconvolution 2d and 3d supported */
if (kernel_size.size() == 2 || kernel_size.size() == 3)
return true;
return false;
}
#ifdef HAVE_INF_ENGINE
const int outGroupCn = blobs[0].size[1]; // Weights are in IOHW or IODHW layout
const int group = numOutput / outGroupCn;
@@ -1372,7 +1455,8 @@ public:
}
else
#endif // HAVE_INF_ENGINE
return kernel_size.size() == 2 && (backendId == DNN_BACKEND_OPENCV || backendId == DNN_BACKEND_HALIDE);
return backendId == DNN_BACKEND_CUDA ||
(kernel_size.size() == 2 && (backendId == DNN_BACKEND_OPENCV || backendId == DNN_BACKEND_HALIDE));
}
bool getMemoryShapes(const std::vector<MatShape> &inputs,
@@ -1898,6 +1982,67 @@ public:
}
}
#ifdef HAVE_CUDA
Ptr<BackendNode> initCUDA(
void *context_,
const std::vector<Ptr<BackendWrapper>>& inputs,
const std::vector<Ptr<BackendWrapper>>& outputs
) override
{
auto context = reinterpret_cast<csl::CSLContext*>(context_);
CV_Assert(inputs.size() == 1);
auto input_wrapper = inputs[0].dynamicCast<CUDABackendWrapper>();
auto input_shape = input_wrapper->getShape();
CV_Assert(outputs.size() == 1);
auto output_wrapper = outputs[0].dynamicCast<CUDABackendWrapper>();
auto output_shape = output_wrapper->getShape();
const auto output_feature_maps = numOutput;
const auto output_feature_maps_per_group = blobs[0].size[1];
const auto groups = output_feature_maps / output_feature_maps_per_group;
TransposeConvolutionConfiguration config;
config.kernel_size.assign(std::begin(kernel_size), std::end(kernel_size));
config.dilations.assign(std::begin(dilations), std::end(dilations));
config.strides.assign(std::begin(strides), std::end(strides));
if (padMode.empty())
{
config.padMode = TransposeConvolutionConfiguration::PaddingMode::MANUAL;
config.pads_begin.assign(std::begin(pads_begin), std::end(pads_begin));
config.pads_end.assign(std::begin(pads_end), std::end(pads_end));
}
else if (padMode == "VALID")
{
config.padMode = TransposeConvolutionConfiguration::PaddingMode::VALID;
}
else if (padMode == "SAME")
{
config.padMode = TransposeConvolutionConfiguration::PaddingMode::SAME;
}
else
{
CV_Error(Error::StsNotImplemented, padMode + " padding mode not supported by DeconvolutionLayer");
}
config.input_shape.assign(std::begin(input_shape), std::end(input_shape));
config.output_shape.assign(std::begin(output_shape), std::end(output_shape));
config.groups = groups;
CV_Assert(blobs.size() >= 1);
Mat filtersMat = fusedWeights ? weightsMat.t() : blobs[0];
Mat biasMat = (hasBias() || fusedBias) ? biasesMat : Mat();
if (countNonZero(biasMat) == 0)
biasMat = Mat();
return make_cuda_node<cuda4dnn::TransposeConvolutionOp>(
preferableTarget, std::move(context->stream), std::move(context->cudnn_handle), config, filtersMat, biasMat);
}
#endif
virtual Ptr<BackendNode> initHalide(const std::vector<Ptr<BackendWrapper> > &inputs) CV_OVERRIDE
{
#ifdef HAVE_HALIDE