1
0
mirror of https://github.com/opencv/opencv.git synced 2026-07-29 15:23:05 +04:00

Merge pull request #27560 from abhishek-gola:convTranspose_layer_add

Added fully functional convTranspose layer to new DNN engine #27560

Closes https://github.com/opencv/opencv/issues/26307

### Pull Request Readiness Checklist

See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request

- [x] I agree to contribute to the project under Apache 2 License.
- [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV
- [x] The PR is proposed to the proper branch
- [x] There is a reference to the original bug report and related work
- [x] There is accuracy test, performance test and test data in opencv_extra repository, if applicable
      Patch to opencv_extra has the same branch name.
- [x] The feature is well documented and sample code can be built with the project CMake
This commit is contained in:
Abhishek Gola
2026-05-07 23:50:39 +05:30
committed by GitHub
parent e519173241
commit 642a7307c4
16 changed files with 1870 additions and 930 deletions
@@ -415,6 +415,19 @@ CV__DNN_INLINE_NS_BEGIN
bool ceil_mode;
};
class CV_EXPORTS ConvTranspose2Layer : public Layer
{
public:
static Ptr<ConvTranspose2Layer> create(const LayerParams& params);
virtual void setWeights(InputArray weights, InputArray bias,
int C0, int accuracy) = 0;
virtual bool fuseAddBias(InputArray bias) = 0;
std::vector<int> strides, dilations, pads, adjust_pads;
int ngroups;
AutoPadding auto_pad;
};
class CV_EXPORTS LRNLayer : public Layer
{
public:
+8
View File
@@ -66,6 +66,7 @@ struct ConstArgs
}
Conv2Layer* conv = dynamic_cast<Conv2Layer*>(layer_ptr);
ConvTranspose2Layer* deconv = dynamic_cast<ConvTranspose2Layer*>(layer_ptr);
BatchNorm2Layer* bn = dynamic_cast<BatchNorm2Layer*>(layer_ptr);
//ActivationLayer* activ = dynamic_cast<ActivationLayer*>(layer_ptr);
@@ -77,6 +78,13 @@ struct ConstArgs
netimpl->defaultC0, netimpl->accuracy);
conv->inputs.resize(1);
unuse_tail = true;
} else if (deconv) {
// deconvolution with constant weights and bias
deconv->setWeights(netimpl->__tensors__[inputs[1]],
ninputs > 2 ? netimpl->__tensors__[inputs[2]] : Mat(),
netimpl->defaultC0, netimpl->accuracy);
deconv->inputs.resize(1);
unuse_tail = true;
} else if (bn && bn->freezeScaleBias()) {
// batch norm with constant parameters
unuse_tail = true;
+1
View File
@@ -136,6 +136,7 @@ void initializeLayerFactory()
CV_DNN_REGISTER_LAYER_CLASS(Convolution, ConvolutionLayer);
CV_DNN_REGISTER_LAYER_CLASS(Conv2, Conv2Layer);
CV_DNN_REGISTER_LAYER_CLASS(Deconvolution, DeconvolutionLayer);
CV_DNN_REGISTER_LAYER_CLASS(ConvTranspose2, ConvTranspose2Layer);
CV_DNN_REGISTER_LAYER_CLASS(Pooling, PoolingLayer);
CV_DNN_REGISTER_LAYER_CLASS(MaxPool, MaxPoolLayer);
CV_DNN_REGISTER_LAYER_CLASS(AveragePool, AveragePoolLayer);
+173
View File
@@ -364,6 +364,179 @@ void ConvState::initPooling(const MatShape& inpshape_,
initOfs();
}
static MatShape getDeconvWpackShape(const MatShape& wshape, int ngroups, int C0)
{
CV_Assert(wshape.dims >= 3);
int C_in = wshape[0], Kg = wshape[1];
CV_Assert(C_in % ngroups == 0);
int Cg = C_in / ngroups;
int ksize = int(wshape.total()) / (C_in * Kg);
int K0 = C0;
int Kblk = (Kg + K0 - 1) / K0;
int C1Max = 0;
for (int g = 0; g < ngroups; ++g) {
int c_start = g * Cg;
int c00 = c_start & (C0 - 1);
int cblocks = (c00 + Cg + C0 - 1) / C0;
C1Max = std::max(C1Max, cblocks);
}
return MatShape({ngroups, Kblk, ksize, C1Max, C0 * K0}, DATA_LAYOUT_UNKNOWN);
}
MatShape deconvInferShape(const MatShape& inpShape, const MatShape& wshape,
const std::vector<int>& kernelShape, int ngroups,
const std::vector<int>& strides,
const std::vector<int>& dilations,
const std::vector<int>& pads,
const std::vector<int>& adjustPads,
AutoPadding autoPad)
{
bool blockLayout = true;
int ndims = inpShape.dims;
int nspatialdims = ndims - 2 - int(blockLayout);
CV_Assert(nspatialdims >= 1);
MatShape outshape = inpShape;
int kshape_[MatShape::MAX_DIMS];
if (!kernelShape.empty()) {
int kshape_size = (int)kernelShape.size();
for (int i = 0; i < nspatialdims; i++)
kshape_[i] = kernelShape[kshape_size - nspatialdims + i];
} else {
CV_Assert(!wshape.empty() && wshape.dims == nspatialdims + 2);
for (int i = 0; i < nspatialdims; i++)
kshape_[i] = wshape[i + 2];
}
int C0 = inpShape[ndims - 1];
int K_out = ngroups * wshape[1];
outshape[1] = (K_out + C0 - 1) / C0;
CV_Assert(strides.empty() || (int)strides.size() == nspatialdims);
CV_Assert(dilations.empty() || (int)dilations.size() == nspatialdims);
CV_Assert(pads.empty() || (int)pads.size() == nspatialdims * 2);
CV_Assert(adjustPads.empty() || (int)adjustPads.size() == nspatialdims);
for (int i = 0; i < nspatialdims; i++) {
int inpsz = inpShape[i + 2];
int k_i = kshape_[i];
int stride = strides.empty() ? 1 : strides[i];
int dilation = dilations.empty() ? 1 : dilations[i];
int adj = adjustPads.empty() ? 0 : adjustPads[i];
int outsz;
if (autoPad == AUTO_PAD_NONE || autoPad == AUTO_PAD_VALID) {
int pad_total = 0;
if (!pads.empty())
pad_total = pads[i] + pads[i + nspatialdims];
outsz = (inpsz - 1) * stride - pad_total + dilation * (k_i - 1) + 1 + adj;
} else {
outsz = (inpsz - 1) * stride + 1 + adj;
}
outshape[i + 2] = outsz;
}
outshape.C = K_out;
return outshape;
}
void repackDeconvWeights(const Mat& weights, Mat& Wpack, int outtype, int ngroups, int C0_)
{
CV_Assert(weights.isContinuous());
CV_Assert_N(weights.type() == CV_32F, outtype == CV_32F);
CV_Assert(ngroups > 0);
CV_Assert((C0_ & (C0_ - 1)) == 0 && C0_ >= 4);
MatShape wshape = weights.shape();
CV_Assert(wshape.dims >= 3);
int C_in = wshape[0], Kg = wshape[1];
CV_Assert(C_in % ngroups == 0);
int K_out = ngroups * Kg;
if (!Wpack.isContinuous())
Wpack.release();
MatShape wpackShape = getDeconvWpackShape(wshape, ngroups, C0_);
Wpack.create(wpackShape, CV_32F);
Wpack.setZero();
parallel_for_(Range(0, K_out), [&](const Range& range) {
int Cg = C_in / ngroups;
int ksize = wpackShape[2], Kblk = wpackShape[1], C1Max = wpackShape[3];
int C0 = C0_, K0 = C0;
const float* wdata = weights.ptr<float>();
float* Wpackdata = Wpack.ptr<float>();
for (int k = range.start; k < range.end; ++k) {
int g = k / Kg;
int kin = k - g * Kg; // output channel within group
int kblk = kin / K0;
int k0 = kin & (K0 - 1);
int c_start = g * Cg;
int c00 = c_start & (C0 - 1);
for (int c = 0; c < Cg; ++c) {
int ch = c00 + c;
int c1 = ch / C0;
int c0 = ch & (C0 - 1);
int c_global = g * Cg + c;
const float* wptr = wdata + (c_global * Kg + kin) * ksize;
float* wpackptr = Wpackdata + (((g * Kblk + kblk) * ksize * C1Max + c1) * C0 + c0) * K0 + k0;
for (int i = 0; i < ksize; ++i) {
wpackptr[i * (C1Max * C0 * K0)] = wptr[i];
}
}
}
});
}
void ConvState::initDeconv(const MatShape& inpshape_,
const MatShape& wshape_,
const MatShape& outshape_,
int ngroups_,
const std::vector<int>& strides_,
const std::vector<int>& dilations_,
const std::vector<int>& pads_)
{
nspatialdims = wshape_.dims - 2;
CV_Assert(0 < nspatialdims && nspatialdims <= ConvState::MAX_CONV_DIMS);
CV_Assert(strides_.empty() || (int)strides_.size() == nspatialdims);
CV_Assert(dilations_.empty() || (int)dilations_.size() == nspatialdims);
CV_Assert(pads_.empty() || (int)pads_.size() == nspatialdims * 2);
CV_Assert(inpshape_.dims == outshape_.dims);
CV_Assert(inpshape_.dims == nspatialdims + 2 + int(inpshape_.layout == DATA_LAYOUT_BLOCK));
inpshape = inpshape_;
outshape = outshape_;
ngroups = ngroups_;
depthwise = false;
fastActivation = FAST_ACTIV_NONE;
activation = nullptr;
activParams.clear();
for (int i = 0; i < MAX_CONV_DIMS; i++) {
kshape[i] = strides[i] = dilations[i] = 1;
pads[i] = pads[i + MAX_CONV_DIMS] = 0;
inner[i] = inner[i + MAX_CONV_DIMS] = 0;
}
for (int i = 0; i < nspatialdims; i++) {
int j = i + (MAX_CONV_DIMS - nspatialdims);
kshape[j] = wshape_[i + 2];
strides[j] = strides_.empty() ? 1 : strides_[i];
dilations[j] = dilations_.empty() ? 1 : dilations_[i];
pads[j] = pads_.empty() ? 0 : pads_[i];
pads[j + MAX_CONV_DIMS] = pads_.empty() ? 0 : pads_[i + nspatialdims];
}
if (inpshape.layout == DATA_LAYOUT_BLOCK) {
int C0 = inpshape.back();
wshape = getDeconvWpackShape(wshape_, ngroups, C0);
CV_Assert(wshape.dims == 5);
}
}
void ConvState::initOfs()
{
CV_Assert(MAX_CONV_DIMS == 3);
+23 -2
View File
@@ -76,9 +76,15 @@ struct ConvState
const std::vector<int>& pads,
AutoPadding auto_pad, bool ceil_mode);
// internal-use method to initialize coordtab and ofstab.
// it's called from initConv and initPooling
void initOfs();
void initDeconv(const MatShape& inpShape,
const MatShape& wshape,
const MatShape& outShape,
int ngroups,
const std::vector<int>& strides,
const std::vector<int>& dilations,
const std::vector<int>& pads);
};
AutoPadding getAutoPadding(const LayerParams& params);
@@ -93,6 +99,21 @@ ConvFunc getDepthwiseConvFunc(int depth);
void repackDepthwiseConvWeights(const Mat& weights, Mat& Wpack, int outtype, int C0);
void repackConvWeights(const Mat& weights, Mat& Wpack, int outtype, int ngroups, int C0);
MatShape deconvInferShape(const MatShape& inpShape, const MatShape& wshape,
const std::vector<int>& kernelShape, int ngroups,
const std::vector<int>& strides,
const std::vector<int>& dilations,
const std::vector<int>& pads,
const std::vector<int>& adjustPads,
AutoPadding autoPad);
typedef void (*DeconvFunc)(const void* inp, const void* residual, void* out,
const ConvState& cs, const void* weights,
const float* scale, const float* bias);
DeconvFunc getDeconvFunc(int depth);
void repackDeconvWeights(const Mat& weights, Mat& Wpack, int outtype, int ngroups, int C0);
CV__DNN_INLINE_NS_END
}
}
@@ -1319,876 +1319,11 @@ public:
}
};
class DeConvolutionLayerImpl CV_FINAL : public BaseConvolutionLayerImpl
{
public:
Mat weightsMat, biasesMat;
UMat umat_weights;
UMat umat_biases;
DeConvolutionLayerImpl(const LayerParams& params) : BaseConvolutionLayerImpl(params) {}
MatShape computeColRowShape(const MatShape &inpShape, const MatShape &outShape) const CV_OVERRIDE
{
int dims = inpShape.size();
int inpD = dims == 5 ? inpShape[2] : 1;
int inpH = inpShape[dims - 2];
int inpW = inpShape.back();
int outCn = outShape[1];
int outGroupCn = outCn / groups;
int ksize = outGroupCn * std::accumulate(kernel_size.begin(), kernel_size.end(),
1, std::multiplies<size_t>());
return shape(ksize, inpD * inpH * inpW);
}
virtual bool supportBackend(int backendId) CV_OVERRIDE
{
if (backendId == DNN_BACKEND_CUDA)
{
/* only deconvolution 2d and 3d supported */
if (kernel_size.size() == 2 || kernel_size.size() == 3)
return true;
return false;
}
#ifdef HAVE_INF_ENGINE
const int outGroupCn = blobs[0].size[1]; // Weights are in IOHW or IODHW layout
const int group = numOutput / outGroupCn;
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH) {
return group == 1;
}
#endif // HAVE_INF_ENGINE
{
return backendId == DNN_BACKEND_CUDA ||
(kernel_size.size() == 2 && backendId == DNN_BACKEND_OPENCV) ||
(kernel_size.size() == 2 && backendId == DNN_BACKEND_CANN);
}
}
bool getMemoryShapes(const std::vector<MatShape> &inputs,
const int requiredOutputs,
std::vector<MatShape> &outputs,
std::vector<MatShape> &internals) const CV_OVERRIDE
{
CV_Assert(inputs.size() != 0);
int outCn = numOutput;
if (outCn < 0) {
CV_Assert(inputs.size() > 1 || !blobs.empty());
MatShape weightShape = blobs.empty() ? inputs[1] : blobs[0].shape();
outCn = weightShape[1]*groups;
}
std::vector<int> outShape;
outShape.push_back(inputs[0][0]); // batch
outShape.push_back(outCn);
if (padMode.empty())
{
for (int i = 0; i < kernel_size.size(); i++)
outShape.push_back(strides[i] * (inputs[0][2 + i] - 1) + kernel_size[i] - pads_begin[i] - pads_end[i] + adjust_pads[i]);
}
else if (padMode == "VALID")
{
for (int i = 0; i < kernel_size.size(); i++)
outShape.push_back(strides[i] * (inputs[0][2 + i] - 1) + kernel_size[i] + adjust_pads[i]);
}
else if (padMode == "SAME")
{
for (int i = 0; i < kernel_size.size(); i++)
outShape.push_back(strides[i] * (inputs[0][2 + i] - 1) + 1 + adjust_pads[i]);
}
else
CV_Error(Error::StsError, "Unsupported padding mode " + padMode);
CV_Assert(outCn % blobs[0].size[1] == 0);
int inpCn = inputs[0][1];
CV_Assert(inpCn % groups == 0 && outCn % groups == 0);
CV_Assert(blobs[0].size[0] == inpCn);
outputs.resize(1, MatShape(outShape));
if (!is1x1())
internals.push_back(computeColRowShape(inputs[0], outputs[0]));
return false;
}
void getTypes(const std::vector<MatType> &inputs,
const int requiredOutputs,
const int requiredInternals,
std::vector<MatType> &outputs,
std::vector<MatType> &internals) const CV_OVERRIDE
{
CV_Assert(inputs.size() > 0);
outputs.assign(requiredOutputs, inputs[0]);
internals.assign(requiredInternals, CV_32F);
}
void finalize(InputArrayOfArrays inputs_arr, OutputArrayOfArrays outputs_arr) CV_OVERRIDE
{
BaseConvolutionLayerImpl::finalize(inputs_arr, outputs_arr);
std::vector<Mat> inputs, outputs;
inputs_arr.getMatVector(inputs);
outputs_arr.getMatVector(outputs);
CV_Assert(inputs.size() > 1 || !blobs.empty());
MatShape weightShape = blobs.empty() ? inputs[1].shape() : blobs[0].shape();
numOutput = weightShape[1]*groups;
std::vector<int> inpShape;
std::vector<int> outShape;
for (int i = 2; i < inputs[0].dims; i++) {
inpShape.push_back(inputs[0].size[i]);
outShape.push_back(outputs[0].size[i]);
}
getConvPoolPaddings(outShape, kernel_size, strides, padMode, pads_begin, pads_end);
if (pads_begin.size() == 2) {
for (int i = 0; i < pads_begin.size(); i++) {
if (pads_begin[i] != pads_end[i])
CV_Error(Error::StsNotImplemented, "Unsupported asymmetric padding in deconvolution layer");
}
pad = Size(pads_begin[1], pads_begin[0]);
}
weightsMultipliers.assign(numOutput, 1.0);
if (weightsMat.empty() && !blobs.empty()) {
transpose(blobs[0].reshape(1, blobs[0].size[0]), weightsMat);
}
if (biasesMat.empty() && blobs.size() >= 2) {
biasesMat = blobs[1].reshape(1, numOutput);
}
}
void fuseWeights(const Mat& w_, const Mat& b_) CV_OVERRIDE
{
Mat w = w_.total() == 1 ? Mat(1, numOutput, CV_32F, Scalar(w_.at<float>(0))) : w_;
Mat b = b_.total() == 1 ? Mat(1, numOutput, CV_32F, Scalar(b_.at<float>(0))) : b_;
CV_Assert_N(!weightsMat.empty(),
w.empty() || numOutput == w.total(),
b.empty() || numOutput == b.total());
if (!w.empty())
{
transpose(blobs[0].reshape(1, blobs[0].size[0]), weightsMat);
weightsMat = weightsMat.reshape(1, numOutput);
for (int i = 0; i < numOutput; ++i)
{
double wi = w.at<float>(i);
weightsMultipliers[i] *= wi;
cv::multiply(weightsMat.row(i), weightsMultipliers[i], weightsMat.row(i));
biasesMat.at<float>(i) *= wi;
}
weightsMat = weightsMat.reshape(1, weightsMat.total() / blobs[0].size[0]);
}
if (!b.empty())
{
cv::add(biasesMat, b.reshape(1, numOutput), biasesMat);
}
}
class MatMulInvoker : public ParallelLoopBody
{
public:
MatMulInvoker(const Mat& a, const Mat& b, Mat& c, int nstripes)
{
a_ = &a;
b_ = &b;
c_ = &c;
nstripes_ = nstripes;
useAVX = checkHardwareSupport(CPU_AVX);
useAVX2 = checkHardwareSupport(CPU_AVX2);
useAVX512 = CV_CPU_HAS_SUPPORT_AVX512_SKX;
useRVV = checkHardwareSupport(CPU_RVV);
useLASX = checkHardwareSupport(CPU_LASX);
}
void operator()(const Range& range_) const CV_OVERRIDE
{
int stripeSize = (int)alignSize((b_->cols + nstripes_ - 1)/nstripes_, 16);
Range range(range_.start*stripeSize, std::min(range_.end*stripeSize, b_->cols));
int mmax = a_->rows;
int nmax = range.end - range.start;
int kmax = a_->cols;
int m, n, k;
const float* aptr = a_->ptr<float>();
const float* bptr = b_->ptr<float>() + range.start;
float* cptr = c_->ptr<float>() + range.start;
size_t astep = a_->step1();
size_t bstep = b_->step1();
size_t cstep = c_->step1();
#if CV_TRY_AVX512_SKX
if( useAVX512 )
opt_AVX512_SKX::fastGEMM( aptr, astep, bptr, bstep, cptr, cstep, mmax, kmax, nmax );
else
#endif
#if CV_TRY_AVX2
if( useAVX2 )
opt_AVX2::fastGEMM( aptr, astep, bptr, bstep, cptr, cstep, mmax, kmax, nmax );
else
#endif
#if CV_TRY_AVX
if( useAVX )
opt_AVX::fastGEMM( aptr, astep, bptr, bstep, cptr, cstep, mmax, kmax, nmax );
else
#endif
#if CV_TRY_RVV && CV_RVV
if( useRVV ) {
opt_RVV::fastGEMM( aptr, astep, bptr, bstep, cptr, cstep, mmax, kmax, nmax );
}
else
#endif
#if CV_TRY_LASX
if( useLASX )
opt_LASX::fastGEMM( aptr, astep, bptr, bstep, cptr, cstep, mmax, kmax, nmax );
else
#endif
for( m = 0; m < mmax; m += 2 )
{
float* dst0 = cptr + cstep*m;
float* dst1 = cptr + cstep*std::min(m+1, mmax-1);
const float* aptr0 = aptr + astep*m;
const float* aptr1 = aptr + astep*std::min(m+1, mmax-1);
for( n = 0; n < nmax; n++ )
{
dst0[n] = 0.f;
dst1[n] = 0.f;
}
for( k = 0; k < kmax; k += 4 )
{
float alpha00 = aptr0[k];
float alpha01 = aptr1[k];
float alpha10 = 0.f, alpha11 = 0.f;
float alpha20 = 0.f, alpha21 = 0.f;
float alpha30 = 0.f, alpha31 = 0.f;
const float* bptr0 = bptr + k*bstep;
const float* bptr1 = bptr0;
const float* bptr2 = bptr0;
const float* bptr3 = bptr0;
if( k+1 < kmax )
{
alpha10 = aptr0[k+1];
alpha11 = aptr1[k+1];
bptr1 = bptr0 + bstep;
if( k+2 < kmax )
{
alpha20 = aptr0[k+2];
alpha21 = aptr1[k+2];
bptr2 = bptr1 + bstep;
if( k+3 < kmax )
{
alpha30 = aptr0[k+3];
alpha31 = aptr1[k+3];
bptr3 = bptr2 + bstep;
}
}
}
n = 0;
#if CV_SIMD128
v_float32x4 a00 = v_setall_f32(alpha00);
v_float32x4 a01 = v_setall_f32(alpha01);
v_float32x4 a10 = v_setall_f32(alpha10);
v_float32x4 a11 = v_setall_f32(alpha11);
v_float32x4 a20 = v_setall_f32(alpha20);
v_float32x4 a21 = v_setall_f32(alpha21);
v_float32x4 a30 = v_setall_f32(alpha30);
v_float32x4 a31 = v_setall_f32(alpha31);
for( ; n <= nmax - 4; n += 4 )
{
v_float32x4 d0 = v_load(dst0 + n);
v_float32x4 d1 = v_load(dst1 + n);
v_float32x4 b0 = v_load(bptr0 + n);
v_float32x4 b1 = v_load(bptr1 + n);
v_float32x4 b2 = v_load(bptr2 + n);
v_float32x4 b3 = v_load(bptr3 + n);
// TODO try to improve pipeline width
d0 = v_fma(b0, a00, d0);
d1 = v_fma(b0, a01, d1);
d0 = v_fma(b1, a10, d0);
d1 = v_fma(b1, a11, d1);
d0 = v_fma(b2, a20, d0);
d1 = v_fma(b2, a21, d1);
d0 = v_fma(b3, a30, d0);
d1 = v_fma(b3, a31, d1);
v_store(dst0 + n, d0);
v_store(dst1 + n, d1);
}
#endif
for( ; n < nmax; n++ )
{
float b0 = bptr0[n];
float b1 = bptr1[n];
float b2 = bptr2[n];
float b3 = bptr3[n];
float d0 = dst0[n] + alpha00*b0 + alpha10*b1 + alpha20*b2 + alpha30*b3;
float d1 = dst1[n] + alpha01*b0 + alpha11*b1 + alpha21*b2 + alpha31*b3;
dst0[n] = d0;
dst1[n] = d1;
}
}
}
}
const Mat *a_, *b_;
Mat* c_;
int nstripes_;
bool useAVX;
bool useAVX2;
bool useAVX512;
bool useRVV;
bool useLASX;
};
class Col2ImInvoker : public cv::ParallelLoopBody
{
public:
const float* data_col;
const float* biasvec;
int channels, height, width;
int kernel_h, kernel_w;
int pad_h, pad_w;
int stride_h, stride_w;
float* data_im;
int height_col, width_col;
int nstripes;
bool is1x1;
Col2ImInvoker()
: data_col(0), biasvec(0), channels(0), height(0), width(0),
kernel_h(0), kernel_w(0), pad_h(0), pad_w(0), stride_h(0), stride_w(0), data_im(0),
height_col(0), width_col(0), nstripes(0), is1x1(0)
{}
static void run(const float* data_col,
int channels, int height, int width,
int kernel_h, int kernel_w,
int pad_h, int pad_w,
int stride_h, int stride_w,
int height_col, int width_col,
float* data_im,
const float* biasvec,
bool is1x1)
{
const int nstripes = getNumThreads();
Col2ImInvoker t;
t.data_col = data_col;
t.data_im = data_im;
t.channels = channels; t.height = height; t.width = width;
t.kernel_h = kernel_h; t.kernel_w = kernel_w;
t.pad_h = pad_h; t.pad_w = pad_w;
t.stride_h = stride_h; t.stride_w = stride_w;
t.height_col = height_col;
t.width_col = width_col;
t.nstripes = nstripes;
t.is1x1 = is1x1;
t.biasvec = biasvec;
parallel_for_(Range(0, nstripes), t, nstripes);
}
virtual void operator ()(const Range &r) const CV_OVERRIDE
{
const float* data_col_ = data_col;
float* data_im_ = data_im;
int coeff_h = (1 - stride_h * kernel_w * height_col) * width_col;
int coeff_w = (1 - stride_w * height_col * width_col);
size_t total = (size_t)channels * height * width;
size_t stripeSize = (total + nstripes - 1)/nstripes;
size_t startIndex = r.start*stripeSize;
size_t endIndex = std::min(r.end*stripeSize, total);
int w = (int)(startIndex % width + pad_w);
int h = (int)((startIndex / width) % height + pad_h);
int c = (int)(startIndex / (width * height));
int h_col_start = (h < kernel_h) ? 0 : (h - kernel_h) / stride_h + 1;
int h_col_end = std::min(h / stride_h + 1, height_col);
int plane_size_col = height_col * width_col;
int offset = (c * kernel_h * kernel_w + h * kernel_w + w) * plane_size_col;
bool is1x1_ = is1x1;
const float* biasvec_ = biasvec;
for (size_t index = startIndex; index < endIndex; index++)
{
// compute the start and end of the output
int w_col_start = (w < kernel_w) ? 0 : (w - kernel_w) / stride_w + 1;
int w_col_end = std::min(w / stride_w + 1, width_col);
float val;
if( is1x1_ )
val = data_im_[index];
else
{
val = 0.f;
for (int h_col = h_col_start; h_col < h_col_end; ++h_col) {
for (int w_col = w_col_start; w_col < w_col_end; ++w_col) {
val += data_col_[offset + h_col * coeff_h + w_col * coeff_w];
}
}
}
data_im_[index] = val + biasvec_[c];
offset += plane_size_col;
if( ++w >= width + pad_w )
{
w = (int)((index + 1)% width + pad_w);
h = (int)(((index + 1) / width) % height + pad_h);
c = (int)((index + 1) / (width * height));
h_col_start = (h < kernel_h) ? 0 : (h - kernel_h) / stride_h + 1;
h_col_end = std::min(h / stride_h + 1, height_col);
offset = (c * kernel_h * kernel_w + h * kernel_w + w) * plane_size_col;
}
}
}
};
#ifdef HAVE_OPENCL
bool forward_ocl(InputArrayOfArrays inputs_, OutputArrayOfArrays outputs_, OutputArrayOfArrays internals_)
{
std::vector<UMat> inputs;
std::vector<UMat> outputs;
std::vector<UMat> internals;
if (inputs_.depth() == CV_16F)
return false;
inputs_.getUMatVector(inputs);
outputs_.getUMatVector(outputs);
internals_.getUMatVector(internals);
int outCn = numOutput;
int inpCn = inputs[0].size[1];
if (is1x1())
return false;
if (umat_weights.empty() || inputs.size() >= 2) {
Mat temp;
if (fusedWeights)
weightsMat.copyTo(umat_weights);
else if (!blobs.empty()) {
transpose(blobs[0].reshape(1, inpCn), temp);
temp.copyTo(umat_weights);
}
else {
transpose(inputs[1].reshape(1, inpCn), temp);
temp.copyTo(umat_weights);
}
}
if (umat_biases.empty() || inputs.size() >= 3) {
if (fusedBias)
biasesMat.copyTo(umat_biases);
else if (blobs.size() > 1)
blobs[1].reshape(1, outCn).copyTo(umat_biases);
else if (inputs.size() >= 3)
inputs[2].reshape(1, outCn).copyTo(umat_biases);
else
umat_biases = UMat::zeros(outCn, 1, CV_32F);
}
String buildopt = format("-DT=%s ", ocl::typeToStr(inputs[0].type()));
buildopt += format("-DPAD_H=%d -DPAD_W=%d -DKERNEL_H=%d -DKERNEL_W=%d -DSTRIDE_H=%d -DSTRIDE_W=%d ",
pad.height, pad.width, kernel.height, kernel.width, stride.height, stride.width);
//for (size_t ii = 0; ii < outputs.size(); ii++)
{
int ii = 0;
int inpGroupCn = inpCn / groups;
int outGroupCn = outCn / groups;
const UMat& inp = inputs[ii];
UMat& out = outputs[ii];
int numImg = inp.size[0];
int inpH = inp.size[2], inpW = inp.size[3];
int outH = out.size[2], outW = out.size[3];
MatShape inpshape = shape(numImg*inpCn, inpH*inpW);
MatShape outshape = shape(numImg*outCn, outH*outW);
UMat convBlob = inputs[ii].reshape(1, inpshape);
UMat decnBlob = out.reshape(1, outshape);
int rows = internals[0].rows / groups;
for (int n = 0; n < numImg; n++)
{
for (int g = 0; g < groups; g++)
{
UMat colMat = internals[0].rowRange(_Range(g * rows, rows));
UMat convMat = convBlob.rowRange(_Range((g + n * groups) * inpGroupCn, inpGroupCn));
UMat wghtMat = umat_weights.colRange(_Range(g * inpGroupCn, inpGroupCn));
gemm(wghtMat, convMat, 1, noArray(), 0, colMat, 0);
}
for (int g = 0; g < groups; g++)
{
int total = outGroupCn * decnBlob.cols;
int index = 0;
int height_col = inpH;
int width_col = inpW;
int coeff_h = (1 - stride.height * kernel.width * height_col) * width_col;
int coeff_w = (1 - stride.width * height_col * width_col);
ocl::Kernel k("col2im", ocl::dnn::col2im_oclsrc, buildopt);
k.set(index++, total);
k.set(index++, ocl::KernelArg::PtrReadOnly(internals[0]));
k.set(index++, (int)(g * rows * internals[0].cols));
k.set(index++, outGroupCn);
k.set(index++, outH);
k.set(index++, outW);
k.set(index++, height_col);
k.set(index++, width_col);
k.set(index++, coeff_h);
k.set(index++, coeff_w);
k.set(index++, ocl::KernelArg::PtrReadOnly(umat_biases));
k.set(index++, (int)(g * outGroupCn * umat_biases.cols));
k.set(index++, ocl::KernelArg::PtrWriteOnly(decnBlob));
k.set(index++, (int)((g + n * groups) * outGroupCn * decnBlob.cols));
size_t global[] = { (size_t)total };
bool ret = k.run(1, global, NULL, false);
if (!ret)
return false;
}
}
}
return true;
}
#endif
void forward(InputArrayOfArrays inputs_arr, OutputArrayOfArrays outputs_arr, OutputArrayOfArrays internals_arr) CV_OVERRIDE
{
CV_TRACE_FUNCTION();
CV_TRACE_ARG_VALUE(name, "name", name.c_str());
// For some reason, tests for deconvolution fail;
// Also, the current implementation is super-inefficient,
// Just disabled it. Need to rewrite it and then uncomment back these lines
//CV_OCL_RUN(IS_DNN_OPENCL_TARGET(preferableTarget),
// forward_ocl(inputs_arr, outputs_arr, internals_arr));
if (inputs_arr.depth(0) == CV_16F)
{
forward_fallback(inputs_arr, outputs_arr, internals_arr);
return;
}
auto kind = outputs_arr.kind();
std::vector<Mat> inputs, internals;
inputs_arr.getMatVector(inputs);
internals_arr.getMatVector(internals);
int outCn = numOutput;
int inpCn = inputs[0].size[1];
bool is1x1flag = is1x1();
int nstripes = getNumThreads();
/*CV_Assert(outputs.size() == 1);
CV_Assert(inputs[0].size[0] == outputs[0].size[0]);
CV_Assert(outCn == outputs[0].size[1]);*/
if (weightsMat.empty() || inputs.size() >= 2) {
Mat inpWeights = !blobs.empty() ? blobs[0] : inputs[1];
transpose(inpWeights.reshape(1, inpCn), weightsMat);
}
if (biasesMat.empty() || inputs.size() >= 3) {
Mat inpBias = blobs.size() >= 2 ? blobs[1] : inputs.size() >= 3 ? inputs[2] : Mat();
Mat biasesMat_ = !inpBias.empty() ? inpBias.reshape(1, outCn) : Mat::zeros(outCn, 1, CV_32F);
biasesMat_.copyTo(biasesMat);
}
/*printf("DeConvolution Input: ");
pprint(std::cout, inputs[0], 0, 3, 100, '[');
printf("\nDeConvolution Weights: ");
pprint(std::cout, weightsMat, 0, 3, 100, '[');
printf("\nDeConvolution Bias: ");
pprint(std::cout, biasesMat, 0, 3, 100, '[');
printf("\n");*/
//for (size_t ii = 0; ii < outputs.size(); ii++)
{
int ii = 0;
int inpGroupCn = inpCn / groups;
int outGroupCn = outCn / groups;
const Mat& inp = inputs[ii];
MatShape outshape = outputs_arr.shape(0);
CV_Assert(outshape.dims == inp.dims);
CV_Assert(outshape[0] == inp.size[0]);
CV_Assert(outshape[1] == outCn);
Mat out;
if (kind == _InputArray::STD_VECTOR_MAT) {
out = outputs_arr.getMat(0);
}
else {
out.create(outshape, inp.type());
}
int numImg = inp.size[0];
int inpH = inp.size[2], inpW = inp.size[3];
int outH = out.size[2], outW = out.size[3];
Mat convBlob = inputs[ii].reshape(1, numImg*inpCn);
Mat decnBlob = out.reshape(1, numImg*outCn);
for (int n = 0; n < numImg; n++)
{
for (int g = 0; g < groups; g++)
{
Mat dstMat = decnBlob.rowRange(_Range((g + n * groups) * outGroupCn, outGroupCn));
Mat &colMat = is1x1flag ? dstMat : internals[0];
Mat convMat = convBlob.rowRange(_Range((g + n * groups) * inpGroupCn, inpGroupCn));
Mat wghtMat = weightsMat.colRange(_Range(g * inpGroupCn, inpGroupCn));
Mat curBiasMat = biasesMat.rowRange(_Range(g * outGroupCn, outGroupCn));
//gemm(wghtMat, convMat, 1, colMat, 0, colMat, 0);
MatMulInvoker mminvoker(wghtMat, convMat, colMat, nstripes);
parallel_for_(Range(0, nstripes), mminvoker, nstripes);
Col2ImInvoker::run(colMat.ptr<float>(), outGroupCn, outH, outW,
kernel.height, kernel.width, pad.height, pad.width,
stride.height, stride.width, inpH, inpW, dstMat.ptr<float>(),
curBiasMat.ptr<float>(), is1x1flag);
}
}
if (kind == _InputArray::STD_VECTOR_UMAT) {
std::vector<UMat>& u_outputs = outputs_arr.getUMatVecRef();
out.copyTo(u_outputs[0]);
}
}
}
#ifdef HAVE_CUDA
Ptr<BackendNode> initCUDA(
void *context_,
const std::vector<Ptr<BackendWrapper>>& inputs,
const std::vector<Ptr<BackendWrapper>>& outputs
) override
{
CV_Assert(!blobs.empty());
auto context = reinterpret_cast<csl::CSLContext*>(context_);
CV_Assert(inputs.size() == 1);
auto input_wrapper = inputs[0].dynamicCast<CUDABackendWrapper>();
auto input_shape = input_wrapper->getShape();
CV_Assert(outputs.size() == 1);
auto output_wrapper = outputs[0].dynamicCast<CUDABackendWrapper>();
auto output_shape = output_wrapper->getShape();
const auto output_feature_maps = numOutput;
const auto output_feature_maps_per_group = blobs[0].size[1];
const auto groups = output_feature_maps / output_feature_maps_per_group;
TransposeConvolutionConfiguration config;
config.kernel_size.assign(std::begin(kernel_size), std::end(kernel_size));
config.dilations.assign(std::begin(dilations), std::end(dilations));
config.strides.assign(std::begin(strides), std::end(strides));
if (padMode.empty())
{
config.padMode = TransposeConvolutionConfiguration::PaddingMode::MANUAL;
config.pads_begin.assign(std::begin(pads_begin), std::end(pads_begin));
config.pads_end.assign(std::begin(pads_end), std::end(pads_end));
}
else if (padMode == "VALID")
{
config.padMode = TransposeConvolutionConfiguration::PaddingMode::VALID;
}
else if (padMode == "SAME")
{
config.padMode = TransposeConvolutionConfiguration::PaddingMode::SAME;
}
else
{
CV_Error(Error::StsNotImplemented, padMode + " padding mode not supported by DeconvolutionLayer");
}
config.input_shape.assign(std::begin(input_shape), std::end(input_shape));
config.output_shape.assign(std::begin(output_shape), std::end(output_shape));
config.groups = groups;
CV_Assert(blobs.size() >= 1);
Mat filtersMat = fusedWeights ? weightsMat.t() : blobs[0];
Mat biasMat = (hasBias() || fusedBias) ? biasesMat : Mat();
if (countNonZero(biasMat) == 0)
biasMat = Mat();
return make_cuda_node<cuda4dnn::TransposeConvolutionOp>(
preferableTarget, std::move(context->stream), std::move(context->cudnn_handle), config, filtersMat, biasMat);
}
#endif
#ifdef HAVE_CANN
virtual Ptr<BackendNode> initCann(const std::vector<Ptr<BackendWrapper> > &inputs,
const std::vector<Ptr<BackendWrapper> > &outputs,
const std::vector<Ptr<BackendNode> >& nodes) CV_OVERRIDE
{
CV_Assert(!blobs.empty());
CV_Assert(inputs.size() == 1);
CV_Assert(nodes.size() == 1);
bool has_bias = hasBias() || fusedBias;
auto x = inputs[0].dynamicCast<CannBackendWrapper>();
auto y = outputs[0].dynamicCast<CannBackendWrapper>();
const auto shape_x = x->host->size; // [N, C, H, W]
const auto shape_y = y->host->size; // [N, C, H, W]
const int filter_out_channel = blobs[0].size[0];
const int groups = shape_x[1] / filter_out_channel;
// create operator
auto op = std::make_shared<ge::op::Conv2DTransposeD>(name);
// set attributes
op->set_attr_input_size(
ge::Operator::OpListInt({(int64_t)shape_y[0],
(int64_t)shape_y[1],
(int64_t)shape_y[2],
(int64_t)shape_y[3],})
);
op->set_attr_strides(
ge::Operator::OpListInt({1, 1, (int64_t)strides[0], (int64_t)strides[1]})
);
op->set_attr_pads(ge::Operator::OpListInt(
{(int64_t)pads_begin[1], (int64_t)pads_end[1], (int64_t)pads_begin[0], (int64_t)pads_end[0]}
));
op->set_attr_dilations(ge::Operator::OpListInt(
{1, 1, (int64_t)dilations[0], (int64_t)dilations[1]}
));
op->set_attr_groups(groups);
op->set_attr_data_format("NCHW");
op->set_attr_output_padding(
ge::Operator::OpListInt({0, 0, (int64_t)adjust_pads[0], (int64_t)adjust_pads[1]}) // adjust_pads: [height, width]
);
// set inputs
// set inputs : x
auto op_x = nodes[0].dynamicCast<CannBackendNode>()->getOp();
op->set_input_x_by_name(*op_x, x->name.c_str());
auto desc_x = x->getTensorDesc();
op->update_input_desc_x(*desc_x);
// set inputs : weight
const Mat& mat_w = blobs[0];
auto op_const_w = std::make_shared<CannConstOp>(mat_w.data, mat_w.type(), shape(mat_w), cv::format("%s_w", name.c_str()));
op->set_input_filter(*(op_const_w->getOp()));
op->update_input_desc_filter(*(op_const_w->getTensorDesc()));
// set inputs : bias
if (has_bias)
{
int out_channel = blobs[0].size[0];
const Mat& mat_b = blobs[1];
std::vector<int> shape_b{out_channel};
auto op_const_b = std::make_shared<CannConstOp>(mat_b.data, mat_b.type(), shape_b, cv::format("%s_b", name.c_str()));
op->set_input_bias(*(op_const_b->getOp()));
op->update_input_desc_bias(*(op_const_b->getTensorDesc()));
}
// set outputs
auto desc_output = std::make_shared<ge::TensorDesc>(ge::Shape(), ge::FORMAT_NCHW, ge::DT_FLOAT);
op->update_output_desc_y(*desc_output);
return Ptr<BackendNode>(new CannBackendNode(op));
}
#endif // HAVE_CANN
#ifdef HAVE_DNN_NGRAPH
virtual Ptr<BackendNode> initNgraph(const std::vector<Ptr<BackendWrapper> > &inputs,
const std::vector<Ptr<BackendNode> >& nodes) CV_OVERRIDE
{
CV_Assert(!blobs.empty());
const int outGroupCn = blobs[0].size[1];
const int group = numOutput / outGroupCn;
CV_Assert(group == 1);
auto& ieInpNode = nodes[0].dynamicCast<InfEngineNgraphNode>()->node;
std::vector<size_t> kernel_shape = getShape<size_t>(blobs[0]);
auto ieWeights = std::make_shared<ov::op::v0::Constant>(ov::element::f32, kernel_shape, blobs[0].data);
if (fusedWeights)
{
Mat newWeights;
transpose(weightsMat, newWeights);
ieWeights = std::make_shared<ov::op::v0::Constant>(ov::element::f32, kernel_shape, newWeights.data);
}
std::vector<size_t> paddings_end;
if (padMode == "SAME")
{
for (int i = 0; i < pads_begin.size(); i++) {
paddings_end.push_back(kernel_size[i] - pads_begin[i] - 1 - adjust_pads[i]);
}
adjust_pads = std::vector<size_t>(pads_begin.size(), 0);
} else {
paddings_end = pads_end;
}
ov::op::PadType pad_type = padMode == "VALID" ? ov::op::PadType::VALID : ov::op::PadType::EXPLICIT;
auto deconv = std::make_shared<ov::op::v1::ConvolutionBackpropData>(
ieInpNode,
ieWeights,
ov::Strides(strides),
ov::CoordinateDiff(std::vector<std::ptrdiff_t>(pads_begin.begin(), pads_begin.end())),
ov::CoordinateDiff(std::vector<std::ptrdiff_t>(paddings_end.begin(), paddings_end.end())),
ov::Strides(dilations),
pad_type,
ov::CoordinateDiff(std::vector<std::ptrdiff_t>(adjust_pads.begin(), adjust_pads.end())));
if (hasBias() || fusedBias)
{
std::vector<size_t> shape(deconv->get_shape().size(), 1);
shape[1] = numOutput;
auto bias = std::make_shared<ov::op::v0::Constant>(ov::element::f32, ov::Shape(shape), blobs[1].data);
auto deconv_bias = std::make_shared<ov::op::v1::Add>(deconv, bias, ov::op::AutoBroadcastType::NUMPY);
return Ptr<BackendNode>(new InfEngineNgraphNode(deconv_bias));
}
return Ptr<BackendNode>(new InfEngineNgraphNode(deconv));
}
#endif // HAVE_DNN_NGRAPH
virtual int64 getFLOPS(const std::vector<MatShape> &inputs,
const std::vector<MatShape> &outputs) const CV_OVERRIDE
{
CV_Assert(inputs.size() == outputs.size());
float flops = 0;
int outChannels = blobs[0].size[0];
size_t karea = std::accumulate(kernel_size.begin(), kernel_size.end(),
1, std::multiplies<size_t>());
for (int i = 0; i < inputs.size(); i++)
{
flops += CV_BIG_INT(2)*outChannels*karea*total(inputs[i]);
}
return flops;
}
};
Ptr<BaseConvolutionLayer> ConvolutionLayer::create(const LayerParams &params)
{
Ptr<ConvolutionLayerImpl> l(new ConvolutionLayerImpl(params));
return l;
}
Ptr<BaseConvolutionLayer> DeconvolutionLayer::create(const LayerParams &params)
{
return Ptr<BaseConvolutionLayer>(new DeConvolutionLayerImpl(params));
}
}
}
@@ -0,0 +1,286 @@
// This file is part of OpenCV project.
// It is subject to the license terms in the LICENSE file found in the top-level directory
// of this distribution and at http://opencv.org/license.html.
// Copyright (C) 2026, BigVision LLC, all rights reserved.
// Third party copyrights are property of their respective owners.
#include "../precomp.hpp"
#include "../net_impl.hpp"
#include "layers_common.hpp"
#include "conv2_common.hpp"
namespace cv
{
namespace dnn
{
/*
ONNX Det operator
Spec: https://onnx.ai/onnx/operators/onnx__ConvTranspose.html
Supported opsets: 1-22
*/
class ConvTranspose2LayerImpl : public ConvTranspose2Layer
{
public:
ConvTranspose2LayerImpl(const LayerParams& params)
{
setParamsFrom(params);
auto_pad = getAutoPadding(params);
strides = params.getVector<int>("stride");
dilations = params.getVector<int>("dilation");
pads = params.getVector<int>("pad");
adjust_pads = params.getVector<int>("adj");
ngroups = params.get<int>("group", 1);
}
virtual std::ostream& dumpAttrs(std::ostream& strm, int indent) const CV_OVERRIDE
{
prindent(strm, indent);
strm << "group: " << ngroups << ",\n";
if (!wshape0.empty()) {
prindent(strm, indent);
strm << "ksize: [";
for (int k = 0; k < wshape0.dims; k++)
strm << (k > 0 ? ", " : "") << wshape0[k];
strm << "],\n";
}
prindent(strm, indent);
strm << "stride: [";
for (size_t k = 0; k < strides.size(); k++)
strm << (k > 0 ? ", " : "") << strides[k];
strm << "],\n";
prindent(strm, indent);
strm << "dilation: [";
for (size_t k = 0; k < dilations.size(); k++)
strm << (k > 0 ? ", " : "") << dilations[k];
strm << "],\n";
prindent(strm, indent);
strm << "pad: [";
for (size_t k = 0; k < pads.size(); k++)
strm << (k > 0 ? ", " : "") << pads[k];
strm << "],\n";
if (!adjust_pads.empty()) {
prindent(strm, indent);
strm << "adj: [";
for (size_t k = 0; k < adjust_pads.size(); k++)
strm << (k > 0 ? ", " : "") << adjust_pads[k];
strm << "],\n";
}
return strm;
}
int inferType(int inptype0) const
{
return inptype0;
}
virtual void setWeights(InputArray weights_arr, InputArray bias_arr,
int C0, int /*accuracy*/) CV_OVERRIDE
{
Mat rawWeights = weights_arr.getMat();
Mat rawBias = bias_arr.getMat();
CV_Assert(!rawWeights.empty());
wshape0 = rawWeights.shape();
Mat wfloat;
if (rawWeights.type() != CV_32F)
rawWeights.convertTo(wfloat, CV_32F);
else
wfloat = rawWeights;
repackDeconvWeights(wfloat, weights, CV_32F, ngroups, C0);
if (!rawBias.empty())
rawBias.convertTo(bias, CV_32F);
}
virtual bool fuseAddBias(InputArray arr) CV_OVERRIDE
{
if (inputs.size() > 1)
return false;
Mat new_bias = arr.getMat();
CV_Assert(new_bias.isContinuous() && new_bias.dims == 1);
if (new_bias.type() != CV_32F) {
Mat temp;
new_bias.convertTo(temp, CV_32F);
new_bias = temp;
}
if (!bias.empty()) {
CV_Assert(bias.shape() == new_bias.shape());
add(bias, new_bias, bias);
} else {
new_bias.copyTo(bias);
}
return true;
}
virtual int64_t getFLOPS(const std::vector<MatShape>& inputs,
const std::vector<MatShape>& outputs) const CV_OVERRIDE
{
CV_Assert(inputs.size() >= 1);
CV_Assert(outputs.size() == 1);
MatShape wshape = inputs.size() > 1 ? inputs[1] : wshape0;
int K = wshape[1] * ngroups;
size_t ksize = wshape.total() / (wshape[0] * wshape[1]);
int C = inputs[0][1] * inputs[0].back();
return (int64_t)((inputs[0].total() / C) * ksize * K);
}
virtual void getTypes(const std::vector<MatType>& inptypes,
const int, const int,
std::vector<MatType>& outtypes,
std::vector<MatType>& temptypes) const CV_OVERRIDE
{
int ninputs = (int)inptypes.size();
CV_Assert(ninputs >= 1);
outtypes.assign(1, inferType(inptypes[0]));
temptypes.clear();
}
virtual bool getMemoryShapes(const std::vector<MatShape>& inpshapes,
const int,
std::vector<MatShape>& outshapes,
std::vector<MatShape>& tempshapes) const CV_OVERRIDE
{
size_t ninputs = inpshapes.size();
CV_Assert(ninputs >= 1);
MatShape wshape = ninputs > 1 ? inpshapes[1] : wshape0;
outshapes.assign(1, deconvInferShape(inpshapes[0], wshape, emptyKernelShape,
ngroups, strides, dilations,
pads, adjust_pads, auto_pad));
tempshapes.clear();
return true;
}
int getLayouts(const std::vector<DataLayout>& actualInputs,
std::vector<DataLayout>& desiredInputs,
const int requiredOutputs,
std::vector<DataLayout>& outputs) const CV_OVERRIDE
{
size_t ninputs = actualInputs.size();
CV_Assert(ninputs >= 1u && requiredOutputs == 1u);
desiredInputs = actualInputs;
desiredInputs[0] = DATA_LAYOUT_BLOCK;
for (size_t i = 1; i < ninputs; i++)
desiredInputs[i] = DATA_LAYOUT_UNKNOWN;
outputs.assign(requiredOutputs, DATA_LAYOUT_BLOCK);
return getNetImpl(this)->defaultC0;
}
void finalize(InputArrayOfArrays, OutputArrayOfArrays) CV_OVERRIDE
{
}
void forward(InputArrayOfArrays input_arrs,
OutputArrayOfArrays output_arrs,
OutputArrayOfArrays) CV_OVERRIDE
{
auto* netimpl_ = getNetImpl(this);
int ninputs = (int)input_arrs.total();
CV_Assert(ninputs >= 1);
const Mat& inp = input_arrs.getMat(0);
int inptype = inp.type();
MatShape inpshape = inp.shape();
CV_Assert(inpshape.layout == DATA_LAYOUT_BLOCK);
CV_Assert(inp.isContinuous());
bool dynamicWeights = false;
for (int i = 1; i < ninputs; i++) {
if (!netimpl_->isConstArg(inputs[i]))
dynamicWeights = true;
}
if (dynamicWeights || weights.empty()) {
setWeights(input_arrs.getMat(1), ninputs > 2 ? input_arrs.getMat(2) : Mat(),
inpshape.back(), netimpl_->accuracy);
}
MatShape outshape = deconvInferShape(inpshape, wshape0, emptyKernelShape,
ngroups, strides, dilations,
pads, adjust_pads, auto_pad);
// compute actual pads for SAME/VALID auto-padding
int nsd = inpshape.dims - 3;
std::vector<int> pads_resolved = pads;
if (auto_pad != AUTO_PAD_NONE) {
pads_resolved.resize(nsd * 2, 0);
for (int i = 0; i < nsd; i++) {
int inpsz = inpshape[2 + i];
int outsz = outshape[2 + i];
int adj_i = adjust_pads.empty() ? 0 : adjust_pads[i];
int stride = strides.empty() ? 1 : strides[i];
int dil = dilations.empty() ? 1 : dilations[i];
int ki = wshape0[2 + i];
int total = (inpsz - 1) * stride + dil * (ki - 1) + 1 + adj_i - outsz;
int pb;
if (auto_pad == AUTO_PAD_SAME_UPPER && stride <= ki * dil) {
pb = std::max((total - (outsz - 1 + stride) % stride) / 2, 0);
} else {
pb = total / 2;
}
pads_resolved[i] = pb;
pads_resolved[nsd + i] = total - pb;
}
}
int outtype = inferType(inptype);
if (inpshape != prevInpshape) {
cs.initDeconv(inpshape, wshape0, outshape, ngroups,
strides, dilations, pads_resolved);
prevInpshape = inpshape;
}
int outkind = output_arrs.kind();
CV_Assert(outkind == _InputArray::STD_VECTOR_MAT ||
outkind == _InputArray::STD_VECTOR_UMAT);
std::vector<Mat>* outs = nullptr;
std::vector<UMat>* uouts = nullptr;
Mat out;
if (outkind == _InputArray::STD_VECTOR_MAT) {
outs = &output_arrs.getMatVecRef();
outs->resize(1);
outs->at(0).fit(outshape, outtype);
out = outs->at(0);
} else {
uouts = &output_arrs.getUMatVecRef();
uouts->resize(1);
uouts->at(0).fit(outshape, outtype);
out.fit(outshape, outtype);
}
DeconvFunc func = getDeconvFunc(inptype);
CV_Assert(func != nullptr);
const float* bias_data = bias.empty() ? nullptr : bias.ptr<float>();
func(inp.data, nullptr, out.data, cs, weights.data, nullptr, bias_data);
if (uouts) {
out.copyTo(uouts->at(0));
}
if (dynamicWeights) {
weights.release();
}
}
std::vector<int> emptyKernelShape;
Mat weights, bias;
MatShape wshape0, prevInpshape;
ConvState cs;
};
Ptr<ConvTranspose2Layer> ConvTranspose2Layer::create(const LayerParams& params)
{
return Ptr<ConvTranspose2Layer>(new ConvTranspose2LayerImpl(params));
}
}}
@@ -0,0 +1,160 @@
// This file is part of OpenCV project.
// It is subject to the license terms in the LICENSE file found in the top-level directory
// of this distribution and at http://opencv.org/license.html.
// Copyright (C) 2026, BigVision LLC, all rights reserved.
// Third party copyrights are property of their respective owners.
#include "../../precomp.hpp"
#include "../conv2_common.hpp"
#include "opencv2/core/hal/intrin.hpp"
namespace cv { namespace dnn {
CV__DNN_INLINE_NS_BEGIN
static void deconvBlock32f(const void* inp__, const void* /*residual*/,
void* out__, const ConvState& cs,
const void* weights__, const float* /*scale*/,
const float* bias__)
{
CV_Assert(cs.inpshape.layout == DATA_LAYOUT_BLOCK);
CV_Assert(cs.outshape.layout == DATA_LAYOUT_BLOCK);
CV_Assert(cs.wshape.dims == 5);
enum { MAX_DIMS = ConvState::MAX_CONV_DIMS };
const int sdims = cs.nspatialdims;
const int C0 = cs.inpshape.back();
const int K0 = C0;
const int N = cs.inpshape[0];
const int C1 = cs.inpshape[1];
const int K1 = cs.outshape[1];
const int C = cs.inpshape.C;
const int K = cs.outshape.C;
const int ngroups = cs.ngroups;
const int Kg = K / ngroups;
const int Cg = C / ngroups;
const int Kblk = cs.wshape[1];
const int ksize = cs.wshape[2];
const int C1Max = cs.wshape[3];
int ispatial = 1, ospatial = 1;
for (int i = 0; i < sdims; i++) {
ispatial *= cs.inpshape[2 + i];
ospatial *= cs.outshape[2 + i];
}
int oDims[MAX_DIMS], iDims[MAX_DIMS];
for (int i = 0; i < MAX_DIMS; i++) oDims[i] = iDims[i] = 1;
for (int i = 0; i < sdims; i++) {
oDims[MAX_DIMS - sdims + i] = cs.outshape[2 + i];
iDims[MAX_DIMS - sdims + i] = cs.inpshape[2 + i];
}
std::vector<std::array<int, MAX_DIMS>> kcoords_tab(ksize);
for (int ks = 0; ks < ksize; ks++) {
int ktmp = ks;
for (int i = sdims - 1; i >= 0; i--) {
int di = MAX_DIMS - sdims + i;
kcoords_tab[ks][di] = ktmp % cs.kshape[di];
ktmp /= cs.kshape[di];
}
}
const float* wdata = (const float*)weights__;
const float* bias = bias__;
const int NK1 = N * K1;
parallel_for_(Range(0, NK1), [&](const Range& range) {
for (int nk1 = range.start; nk1 < range.end; nk1++) {
const int n = nk1 / K1;
const int k1 = nk1 % K1;
const int k_base = k1 * C0;
const int currK0 = std::min(C0, K - k_base);
if (currK0 <= 0) continue;
float* out_k1 = (float*)out__ + ((int64_t)n * K1 + k1) * ospatial * C0;
for (int opos = 0; opos < ospatial; opos++) {
float* p = out_k1 + opos * C0;
if (bias) {
for (int k0 = 0; k0 < currK0; k0++) p[k0] = bias[k_base + k0];
} else {
for (int k0 = 0; k0 < currK0; k0++) p[k0] = 0.f;
}
for (int k0 = currK0; k0 < C0; k0++) p[k0] = 0.f;
}
const float* inp_n = (const float*)inp__ + (int64_t)n * C1 * ispatial * C0;
for (int opos_flat = 0; opos_flat < ospatial; opos_flat++) {
float* out_ptr = out_k1 + opos_flat * C0;
int ocoords[MAX_DIMS];
{
int tmp = opos_flat;
for (int i = sdims - 1; i >= 0; i--) {
int di = MAX_DIMS - sdims + i;
ocoords[di] = tmp % oDims[di];
tmp /= oDims[di];
}
}
for (int ks = 0; ks < ksize; ks++) {
bool valid = true;
int ipos_flat = 0;
for (int i = 0; i < sdims; i++) {
int di = MAX_DIMS - sdims + i;
int raw = ocoords[di] + cs.pads[di]
- kcoords_tab[ks][di] * cs.dilations[di];
if (raw < 0 || raw % cs.strides[di] != 0) {
valid = false; break;
}
int ic = raw / cs.strides[di];
if (ic >= iDims[di]) { valid = false; break; }
ipos_flat = ipos_flat * iDims[di] + ic;
}
if (!valid) continue;
for (int k0 = 0; k0 < currK0; k0++) {
const int k = k_base + k0;
const int g = k / Kg;
const int kin = k - g * Kg;
const int kblk = kin / K0;
const int k0l = kin & (K0 - 1);
const float* w_base = wdata +
((int64_t)(g * Kblk + kblk) * ksize + ks) * C1Max * C0 * K0;
const int c_start = g * Cg;
const int c1_abs_base = c_start / C0;
float sum = 0.f;
for (int c1p = 0; c1p < C1Max; c1p++) {
const int c1_abs = c1_abs_base + c1p;
if (c1_abs >= C1) break;
const float* inp_ptr = inp_n +
(int64_t)(c1_abs * ispatial + ipos_flat) * C0;
const float* w_c1p = w_base + (int64_t)c1p * C0 * K0;
for (int c0 = 0; c0 < C0; c0++)
sum += w_c1p[c0 * K0 + k0l] * inp_ptr[c0];
}
out_ptr[k0] += sum;
}
}
}
}
});
}
DeconvFunc getDeconvFunc(int depth)
{
if (depth == CV_32F)
return deconvBlock32f;
return nullptr;
}
CV__DNN_INLINE_NS_END
}}
File diff suppressed because it is too large Load Diff
+7
View File
@@ -1674,6 +1674,13 @@ void Net::Impl::setParam(const std::string& outputTensorName, int numParam, cons
return;
}
ConvTranspose2Layer* deconv = dynamic_cast<ConvTranspose2Layer*>(layer.get());
if (deconv && numParam == 0) {
deconv->setWeights(blob, Mat(), defaultC0, accuracy);
finalizeLayers = true;
return;
}
CV_Error_(Error::StsOutOfRange,
("DNN: op producing '%s' has fewer than %d params",
outputTensorName.c_str(), numParam + 1));
+25 -17
View File
@@ -1389,28 +1389,28 @@ void ONNXImporter2::parseConvTranspose(LayerParams& layerParams, const opencv_on
{
int n_inputs = node_proto.input_size();
CV_Assert(2 <= n_inputs && n_inputs <= 3);
layerParams.type = "Deconvolution";
layerParams.set("bias_term", node_proto.input_size() == 3);
if (net.isConstArg(node_inputs[1]) && (n_inputs == 2 || net.isConstArg(node_inputs[2]))) {
Mat weights = net.argTensor(node_inputs[1]);
layerParams.blobs.push_back(weights);
if (n_inputs > 2) {
Mat bias = net.argTensor(node_inputs[2]);
layerParams.blobs.push_back(bias);
}
n_inputs = 1;
}
if (!layerParams.has("kernel_size"))
CV_Error(Error::StsNotImplemented,
"Required attribute 'kernel_size' is not present.");
layerParams.type = "ConvTranspose2";
if (layerParams.has("output_shape"))
{
const DictValue& outShape = layerParams.get("output_shape");
DictValue strides = layerParams.get("stride");
// Infer kernel_size from weight shape if not provided
if (!layerParams.has("kernel_size"))
{
const Arg& warg = node_inputs[1];
const ArgData& wdata = netimpl->args.at(warg.idx);
if (wdata.shape.size() >= 3)
{
int kdims = (int)wdata.shape.size() - 2;
std::vector<int> kshape(kdims);
for (int i = 0; i < kdims; ++i)
kshape[i] = wdata.shape[2 + i];
layerParams.set("kernel_size", DictValue::arrayInt(kshape.data(), kdims));
}
}
DictValue kernel = layerParams.get("kernel_size");
String padMode;
@@ -1430,6 +1430,14 @@ void ONNXImporter2::parseConvTranspose(LayerParams& layerParams, const opencv_on
}
layerParams.set("adj", DictValue::arrayInt(&adjust_pads[0], (int)adjust_pads.size()));
}
else
{
for (int i = 0; i < strides.size(); i++)
{
adjust_pads.push_back(1);
}
layerParams.set("adj", DictValue::arrayInt(&adjust_pads[0], (int)adjust_pads.size()));
}
}
else if (layerParams.has("output_padding"))
{
+4 -1
View File
@@ -788,7 +788,10 @@ TEST_P(Test_Model, TextDetectionByDB)
{
SCOPED_TRACE("Original DB");
testTextDetectionModelByDB(weightPathDB, "", imgPath, gt, binThresh, polyThresh, maxCandidates, unclipRatio, size, meanDB, scaleDB, 0.05f);
float boxes_iou_diff = 0.05f;
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
boxes_iou_diff = 0.11f;
testTextDetectionModelByDB(weightPathDB, "", imgPath, gt, binThresh, polyThresh, maxCandidates, unclipRatio, size, meanDB, scaleDB, boxes_iou_diff);
}
{
@@ -537,25 +537,29 @@ CASE(test_convinteger_with_padding)
CASE(test_convinteger_without_padding)
// no filter
CASE(test_convtranspose)
// no filter
SKIP;
CASE(test_convtranspose_1d)
// no filter
SKIP;
CASE(test_convtranspose_3d)
// no filter
SKIP;
CASE(test_convtranspose_autopad_same)
// no filter
SKIP;
CASE(test_convtranspose_dilations)
// no filter
SKIP;
CASE(test_convtranspose_kernel_shape)
// no filter
SKIP;
CASE(test_convtranspose_group_2)
SKIP;
CASE(test_convtranspose_group_2_image_3)
SKIP;
CASE(test_convtranspose_output_shape)
// no filter
SKIP;
CASE(test_convtranspose_pad)
// no filter
SKIP;
CASE(test_convtranspose_pads)
// no filter
SKIP;
CASE(test_convtranspose_with_kernel)
// no filter
SKIP;
CASE(test_cos)
// no filter
CASE(test_cos_example)
@@ -742,3 +742,14 @@
"test_qlinearconv",
"test_qlinearmatmul_2D",
"test_qlinearmatmul_3D",
"test_convtranspose",
"test_convtranspose_1d",
"test_convtranspose_3d",
"test_convtranspose_dilations",
"test_convtranspose_group_2",
"test_convtranspose_group_2_image_3",
"test_convtranspose_kernel_shape",
"test_convtranspose_output_shape",
"test_convtranspose_pad",
"test_convtranspose_pads",
"test_convtranspose_with_kernel",
@@ -269,18 +269,7 @@
"test_constantofshape_int_shape_zero", // Issue::Parser::Weights are required as inputs
"test_convinteger_with_padding", // Issues::Layer::Can't create layer "onnx_node_output_0!y" of type "ConvInteger" in function 'getLayerInstance'
"test_convinteger_without_padding", //Issues::Layer::Can't create layer "onnx_node_output_0!y" of type "ConvInteger" in function 'getLayerInstance'
"test_convtranspose", // Issue::Parser::Weights are required as inputs
"test_convtranspose_1d", // Issue::Parser::Weights are required as inputs
"test_convtranspose_3d", // Issue::Parser::Weights are required as inputs
"test_convtranspose_autopad_same", // Issue::Parser::Weights are required as inputs
"test_convtranspose_dilations", // Issue::Parser::Weights are required as inputs
"test_convtranspose_group_2",
"test_convtranspose_group_2_image_3",
"test_convtranspose_kernel_shape", // Issue::Parser::Weights are required as inputs
"test_convtranspose_output_shape", // Issue::Parser::Weights are required as inputs
"test_convtranspose_pad", // Issue::Parser::Weights are required as inputs
"test_convtranspose_pads", // Issue::Parser::Weights are required as inputs
"test_convtranspose_with_kernel", // Issue::Parser::Weights are required as inputs
"test_convtranspose_autopad_same",
"test_deform_conv_with_mask_bias",
"test_deform_conv_with_multiple_offset_groups",
"test_dequantizelinear_e4m3fn",
+29 -23
View File
@@ -335,8 +335,7 @@ TEST_P(Test_ONNX_layers, Deconvolution)
testONNXModels("deconv_adjpad_2d", npy, 0, 0, false, false);
}
// BUG: https://github.com/opencv/opencv/issues/26307
TEST_P(Test_ONNX_layers, DISABLED_Deconvolution3D)
TEST_P(Test_ONNX_layers, Deconvolution3D)
{
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2022010000)
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
@@ -356,17 +355,13 @@ TEST_P(Test_ONNX_layers, DISABLED_Deconvolution3D)
}
#endif
if (backend == DNN_BACKEND_OPENCV)
throw SkipTestException("OpenCV backend is not supported"); // FIXIT use tags
if (backend == DNN_BACKEND_VKCOM)
applyTestTag(CV_TEST_TAG_DNN_SKIP_VULKAN);
testONNXModels("deconv3d");
}
// BUG: https://github.com/opencv/opencv/issues/26307
TEST_P(Test_ONNX_layers, DISABLED_Deconvolution3D_bias)
TEST_P(Test_ONNX_layers, Deconvolution3D_bias)
{
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2022010000)
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
@@ -386,17 +381,13 @@ TEST_P(Test_ONNX_layers, DISABLED_Deconvolution3D_bias)
}
#endif
if (backend == DNN_BACKEND_OPENCV)
throw SkipTestException("OpenCV backend is not supported"); // FIXIT use tags
if (backend == DNN_BACKEND_VKCOM)
applyTestTag(CV_TEST_TAG_DNN_SKIP_VULKAN);
testONNXModels("deconv3d_bias");
}
// BUG: https://github.com/opencv/opencv/issues/26307
TEST_P(Test_ONNX_layers, DISABLED_Deconvolution3D_pad)
TEST_P(Test_ONNX_layers, Deconvolution3D_pad)
{
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2022010000)
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
@@ -416,17 +407,13 @@ TEST_P(Test_ONNX_layers, DISABLED_Deconvolution3D_pad)
}
#endif
//if (backend == DNN_BACKEND_OPENCV)
throw SkipTestException("OpenCV backend is not supported"); // FIXIT use tags
//if (backend == DNN_BACKEND_VKCOM)
// applyTestTag(CV_TEST_TAG_DNN_SKIP_VULKAN);
//testONNXModels("deconv3d_pad");
testONNXModels("deconv3d_pad");
}
// BUG: https://github.com/opencv/opencv/issues/26307
TEST_P(Test_ONNX_layers, DISABLED_Deconvolution3D_adjpad)
TEST_P(Test_ONNX_layers, Deconvolution3D_adjpad)
{
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2022010000)
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
@@ -446,9 +433,6 @@ TEST_P(Test_ONNX_layers, DISABLED_Deconvolution3D_adjpad)
}
#endif
if (backend == DNN_BACKEND_OPENCV)
throw SkipTestException("OpenCV backend is not supported"); // FIXIT use tags
if (backend == DNN_BACKEND_VKCOM)
applyTestTag(CV_TEST_TAG_DNN_SKIP_VULKAN);
@@ -1548,7 +1532,18 @@ TEST_P(Test_ONNX_layers, LSTM_cell_forward)
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && target == DNN_TARGET_OPENCL_FP16)
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
#endif
testONNXModels("lstm_cell_forward", npy, 0, 0, false, false);
double l1 = 0, lInf = 0;
if (backend == DNN_BACKEND_CUDA && target == DNN_TARGET_CUDA)
{
l1 = 5e-5;
lInf = 3e-4;
}
else if (backend == DNN_BACKEND_CUDA && target == DNN_TARGET_CUDA_FP16)
{
l1 = 3e-4;
lInf = 1e-3;
}
testONNXModels("lstm_cell_forward", npy, l1, lInf, false, false);
}
TEST_P(Test_ONNX_layers, LSTM_cell_bidirectional)
{
@@ -1559,7 +1554,18 @@ TEST_P(Test_ONNX_layers, LSTM_cell_bidirectional)
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && target == DNN_TARGET_OPENCL_FP16)
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
#endif
testONNXModels("lstm_cell_bidirectional", npy, 0, 0, false, false);
double l1 = 0, lInf = 0;
if (backend == DNN_BACKEND_CUDA && target == DNN_TARGET_CUDA)
{
l1 = 5e-5;
lInf = 3e-4;
}
else if (backend == DNN_BACKEND_CUDA && target == DNN_TARGET_CUDA_FP16)
{
l1 = 3e-4;
lInf = 3e-3;
}
testONNXModels("lstm_cell_bidirectional", npy, l1, lInf, false, false);
}
TEST_P(Test_ONNX_layers, LSTM_cell_with_peepholes)
{