1
0
mirror of https://github.com/opencv/opencv.git synced 2026-07-21 19:33:03 +04:00

Merge pull request #27017 from nklskyoy:gemm-dynamic-inputs

determine gemm op mode at runtime #27017

See https://github.com/opencv/opencv/issues/26209

TODOs:

- [x] Determine OP mode on runtime

### Pull Request Readiness Checklist

See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request

- [x] I agree to contribute to the project under Apache 2 License.
- [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV
- [x] The PR is proposed to the proper branch
- [x] There is a reference to the original bug report and related work
- [ ] There is accuracy test, performance test and test data in opencv_extra repository, if applicable
      Patch to opencv_extra has the same branch name.
- [x] The feature is well documented and sample code can be built with the project CMake
This commit is contained in:
nklskyoy
2025-03-19 07:19:46 +01:00
committed by GitHub
parent 7635d9a012
commit db2ba6a145
+86 -33
View File
@@ -20,6 +20,35 @@ using namespace cv::dnn::cuda4dnn;
namespace cv { namespace dnn {
enum class LayerGemmOpMode {
blobB,
blobBC,
blobC,
noblob
};
bool constB(LayerGemmOpMode mode){
switch (mode) {
case LayerGemmOpMode::blobB:
case LayerGemmOpMode::blobBC:
return true;
default:
return false;
}
}
bool constC(LayerGemmOpMode mode){
switch (mode) {
case LayerGemmOpMode::blobC:
case LayerGemmOpMode::blobBC:
return true;
default:
return false;
}
}
// Y = alpha * A * B + beta * C
class GemmLayerImpl CV_FINAL : public GemmLayer {
public:
GemmLayerImpl(const LayerParams& params) {
@@ -30,28 +59,10 @@ public:
alpha = params.get<float>("alpha", 1.0f);
beta = params.get<float>("beta", 1.0f);
if (params.has("constB") || params.has("constB") || params.has("have_bias"))
{
// The params are not part of ONNX, but set by old ONNX parser
const_B = params.get<bool>("constB", false); // true means blobs[0] is B
const_C = params.get<bool>("constC", false); // true means blobs.back() is C
have_bias = params.get<bool>("have_bias", false); // NOTE: have_bias being true does not mean bias is constant
}
else
{
// TODO: With the new parser the function should be smart enough to figure out
// the operation mode from the number of 'inputs' and number of 'blobs'.
// note, however, that 'inputs' may not be set yet in the constructor
// Ticket: https://github.com/opencv/opencv/issues/26209
if (!blobs.empty()) {
const_B = const_C = true;
} else {
const_B = const_C = false;
}
have_bias = blobs.size() > 1 || params.get<bool>("have_bias", false); // NOTE: have_bias being true does not mean bias is constant
}
// The params are not part of ONNX, but set by old ONNX parser
const_B = params.get<bool>("constB", false);
const_C = params.get<bool>("constC", false);
have_bias = params.get<bool>("have_bias", false);
real_ndims_C = params.get<int>("real_ndims_C", -1);
}
@@ -64,17 +75,52 @@ public:
(backendId == DNN_BACKEND_VKCOM && haveVulkan() && !have_bias && !trans_a);
}
LayerGemmOpMode getOpMode(size_t n_inputs, size_t n_blobs) const {
if (n_blobs == 0) return LayerGemmOpMode::noblob;
if (n_inputs == 3) return LayerGemmOpMode::noblob ; // if all inputs are given, then no blobs are used
if (n_inputs == 2) {
if (have_bias) {
// check where the input comes from
if(const_B)
return LayerGemmOpMode::blobB;
if(const_C)
return LayerGemmOpMode::blobC;
return LayerGemmOpMode::blobC;
} else {
if (n_blobs == 1)
// 2 inputs, no bias => input[1] is B
return LayerGemmOpMode::noblob;
if (n_blobs == 2)
return LayerGemmOpMode::blobC;
}
}
if (n_inputs == 1) {
// only A is given per input
if (n_blobs == 2)
return LayerGemmOpMode::blobBC;
else
return LayerGemmOpMode::blobB;
}
CV_Error(Error::StsError, "DNN/Gemm: could not derive OP mode");
}
virtual bool getMemoryShapes(const std::vector<MatShape> &inputs,
const int requiredOutputs,
std::vector<MatShape> &outputs,
std::vector<MatShape> &internals) const CV_OVERRIDE {
int num_inputs = static_cast<int>(inputs.size() + blobs.size());
CV_CheckGE(num_inputs, 2, "DNN/Gemm: Gemm takes at least two inputs");
CV_CheckLE(num_inputs, 3, "DNN/Gemm: Gemm takes at most three inputs");
LayerGemmOpMode mode = getOpMode(inputs.size(), blobs.size());
// Check whether A and B are two dimensional
const auto shape_A = inputs[0];
const auto shape_B = const_B ? shape(blobs[0]) : inputs[1];
const auto shape_B = constB(mode) ? shape(blobs[0]) : inputs[1];
CV_CheckGE(shape_A.size(), static_cast<size_t>(2), "DNN/Gemm: Tensor A must be n-dimensional (n >= 2)");
CV_CheckEQ(shape_B.size(), static_cast<size_t>(2), "DNN/Gemm: Tensor B must be two dimensional");
@@ -90,11 +136,9 @@ public:
CV_CheckEQ(K_a, K_b, "DNN/Gemm: Invalid dimension of dim K");
bool have_bias_ = have_bias || inputs.size() == 3;
// Check whether C can be unidirectional broadcast to (M, N). Handle carefully with 1D Mat.
if (have_bias_) {
const auto shape_C = const_C ? shape(blobs.back()) : inputs.back();
if (have_bias) {
const auto shape_C = constC(mode) ? shape(blobs.back()) : inputs.back();
auto ndims_C = shape_C.size();
CV_CheckLE(ndims_C, static_cast<size_t>(2), "DNN/Gemm: C can only be 0d (scalar) / 1d / 2d tensor");
@@ -169,16 +213,21 @@ public:
}
}
virtual void finalize(InputArrayOfArrays inputs_arr, OutputArrayOfArrays outputs_arr) CV_OVERRIDE {
opt.init();
std::vector<Mat> inputs;
inputs_arr.getMatVector(inputs);
LayerGemmOpMode mode = getOpMode(inputs.size(), blobs.size());
// pack B if it is const
if (const_B) {
if (constB(mode)) {
fastGemmPackB(blobs[0], packed_B, trans_b, opt);
}
// also pre-broadcast bias
if (const_C) {
if (constC(mode)) {
const auto &C = blobs.back();
std::vector<Mat> outputs;
@@ -208,6 +257,8 @@ public:
inputs_arr.getMatVector(inputs);
outputs_arr.getMatVector(outputs);
LayerGemmOpMode mode = getOpMode(inputs.size(), blobs.size());
const auto &A = inputs[0];
auto &Y = outputs[0];
@@ -217,11 +268,10 @@ public:
size_t dims_Y = shape_Y.size();
int M = shape_Y[dims_Y - 2], N = shape_Y[dims_Y - 1];
int K = trans_a ? ma : na;
bool have_bias_ = have_bias || inputs.size() == 3;
// broadcast C and copy C to output
if (have_bias_) {
if (!const_C || broadcast_C.empty()) {
if (constC(mode) || inputs.size() >= 3) {
if (!constC(mode) || broadcast_C.empty()) {
broadcastCWtihBeta(M, N, (inputs.size() >= 3 ? inputs.back() : blobs.back()));
}
int step = M * N;
@@ -234,7 +284,7 @@ public:
std::memset(ptr_y, 0, total * sizeof(float));
}
if (const_B) {
if (constB(mode)) {
CV_CheckGT(packed_B.size(), static_cast<size_t>(0), "DNN/Gemm: constant B is not pre-packed");
fastGemm(trans_a, M, N, K, alpha, A.ptr<const float>(), na, packed_B.data(), 1.f, Y.ptr<float>(), N, opt);
} else {
@@ -295,6 +345,7 @@ public:
op->update_input_desc_x2(*desc_x2);
}
// set inputs : bias
// TODO: clearify if the bias needs to be constant here
auto mat_C = have_bias && const_C ? blobs.back() : Mat::zeros(1, 1, CV_32F);
auto shape_C = shape(mat_C);
if (real_ndims_C == 1) {
@@ -391,6 +442,8 @@ public:
}
#endif
private:
bool const_B;
bool const_C;