From d0820dac382943cb35b7ca4a0d51458164010803 Mon Sep 17 00:00:00 2001 From: Abduragim Shtanchaev <44877829+Abdurrahheem@users.noreply.github.com> Date: Wed, 20 Nov 2024 15:12:58 +0400 Subject: [PATCH] Merge pull request #26391 from Abdurrahheem:ash/lstm-new-graph-engine-latest LSTM layer for new graph engine. #26391 Merge with extra: https://github.com/opencv/opencv_extra/pull/1218 This PR updates/creates LSTM layer compatible with new graph engine. It is based on previous LSTM implementation with some modification on how initializers blobs are processed. Note: Following tests are currently are disabled Two following two tests are disbled since ONNNRuntime does not support `layout=1` attiribute inference. See a detailed issue #26456 on this. - `LSTM_layout_seq` - `LSTM_layout_batch` Following test fails with the new engine as it is not able to deal with shapes of the form [?, C, H, W] - `LSTM_Activations` Works: - [x] One directional case any batch type - [x] Fix directional case when batch size large than 1 - [x] Add peepholes attribute TODO with the next PRs: - [ ] Activation support Note: > Currently `LSTM_layout_seq`, `LSTM_layout_batch` are disabled as the tests are incorrect. They do not comply with the ONNX standard. Particularly test outputs are of incorrect dimensionality. They produce 3-dimentinal output instead of 4-dimentional. ------ ### Pull Request Readiness Checklist See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request - [x] I agree to contribute to the project under Apache 2 License. - [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV - [x] The PR is proposed to the proper branch - [x] There is a reference to the original bug report and related work - [x] There is accuracy test, performance test and test data in opencv_extra repository, if applicable Patch to opencv_extra has the same branch name. - [x] The feature is well documented and sample code can be built with the project CMake --- .../dnn/include/opencv2/dnn/all_layers.hpp | 7 + modules/dnn/src/init.cpp | 1 + modules/dnn/src/layers/recurrent2_layers.cpp | 605 ++++++++++++++++++ modules/dnn/src/onnx/onnx_importer2.cpp | 27 +- modules/dnn/test/test_layers.cpp | 50 ++ modules/dnn/test/test_onnx_importer.cpp | 12 +- 6 files changed, 698 insertions(+), 4 deletions(-) create mode 100644 modules/dnn/src/layers/recurrent2_layers.cpp diff --git a/modules/dnn/include/opencv2/dnn/all_layers.hpp b/modules/dnn/include/opencv2/dnn/all_layers.hpp index 7d8ff6b058..eb0d32079b 100644 --- a/modules/dnn/include/opencv2/dnn/all_layers.hpp +++ b/modules/dnn/include/opencv2/dnn/all_layers.hpp @@ -174,6 +174,13 @@ CV__DNN_INLINE_NS_BEGIN int outputNameToIndex(const String& outputName) CV_OVERRIDE; }; + class CV_EXPORTS LSTM2Layer : public Layer + { + public: + /** Creates instance of LSTM layer */ + static Ptr create(const LayerParams& params); + }; + /** @brief GRU recurrent one-layer * * Accepts input sequence and computes the final hidden state for each element in the batch. diff --git a/modules/dnn/src/init.cpp b/modules/dnn/src/init.cpp index 8e8932aab6..8bad602431 100644 --- a/modules/dnn/src/init.cpp +++ b/modules/dnn/src/init.cpp @@ -209,6 +209,7 @@ void initializeLayerFactory() CV_DNN_REGISTER_LAYER_CLASS(FlowWarp, FlowWarpLayer); CV_DNN_REGISTER_LAYER_CLASS(LSTM, LSTMLayer); + CV_DNN_REGISTER_LAYER_CLASS(LSTM2, LSTM2Layer); CV_DNN_REGISTER_LAYER_CLASS(GRU, GRULayer); CV_DNN_REGISTER_LAYER_CLASS(CumSum, CumSumLayer); CV_DNN_REGISTER_LAYER_CLASS(Einsum, EinsumLayer); diff --git a/modules/dnn/src/layers/recurrent2_layers.cpp b/modules/dnn/src/layers/recurrent2_layers.cpp new file mode 100644 index 0000000000..bdfbbef1a8 --- /dev/null +++ b/modules/dnn/src/layers/recurrent2_layers.cpp @@ -0,0 +1,605 @@ +// This file is part of OpenCV project. +// It is subject to the license terms in the LICENSE file found in the top-level directory +// of this distribution and at http://opencv.org/license.html. + +#include "../precomp.hpp" +#include +#include +#include +#include "layers_common.hpp" +#include "../net_impl.hpp" + +namespace cv +{ +namespace dnn +{ + +template +static void tanh(const Mat &src, Mat &dst) +{ + MatConstIterator_ itSrc = src.begin(); + MatIterator_ itDst = dst.begin(); + + for (; itSrc != src.end(); itSrc++, itDst++) + *itDst = std::tanh(*itSrc); +} + +static void tanh(const Mat &src, Mat &dst) +{ + dst.create(src.dims, (const int*)src.size, src.type()); + if (src.type() == CV_32F) + tanh(src, dst); + else if (src.type() == CV_64F) + tanh(src, dst); + else + CV_Error(Error::StsUnsupportedFormat, "Function supports only floating point types"); +} + +static void sigmoid(const Mat &src, Mat &dst) +{ + cv::exp(-src, dst); + cv::pow(1 + dst, -1, dst); +} + +typedef void (*ActivationFunction)(const Mat &src, Mat &dst); +static ActivationFunction get_activation_function(const String& activation) { + if (activation == "Tanh"){ + return tanh; + } + else if (activation == "Sigmoid"){ + return sigmoid; + } + else + { + CV_Error(Error::StsNotImplemented, + cv::format("Activation function [%s] for layer LSTM is not supported", activation.c_str())); + } +} + +class LSTM2LayerImpl CV_FINAL : public LSTM2Layer +{ + int seqLenth, batchSize, numHidden; + MatShape outTailShape; //shape of single output sample + enum layout_t : int { + SEQ_BATCH_HID = 0, + BATCH_SEQ_HID = 1 + }; + + bool useTimestampDim; + bool produceCellOutput, produceOutputYh; + bool useCellClip, usePeephole; + bool reverse; // If true, go in negative direction along the time axis + bool bidirectional; // If true, produces both forward and reversed directions along time axis + float forgetBias, cellClip; + layout_t layout; // If layout == BATCH_SEQ_HID, uses batch_size x seq_length x num_hidden for input and output + // else uses seq_length x batch_size x num_hidden + + ActivationFunction f_activation; + ActivationFunction g_activation; + ActivationFunction h_activation; + + public: + LSTM2LayerImpl(const LayerParams& params) + { + setParamsFrom(params); + numHidden = params.get("hidden_size", 1); + layout = (layout_t) params.get("layout", SEQ_BATCH_HID); + + produceCellOutput = params.get("produce_cell_output", false); + produceOutputYh = params.get("produce_output_yh", false); + bidirectional = params.get("bidirectional", false); + reverse = params.get("reverse", false); + useTimestampDim = params.get("use_timestamp_dim", true); + usePeephole = params.get("use_peephole", false); + useCellClip = params.get("use_cell_clip", false); + + forgetBias = params.get("forget_bias", 0.0f); + cellClip = params.get("cell_clip", 0.0f); + + CV_Assert(!reverse || !bidirectional); + + // read activations + DictValue activations = params.get("activations", DictValue(String())); + if (activations.size() == 1) // if activations wasn't specified use default + { + f_activation = sigmoid; + g_activation = tanh; + h_activation = tanh; + } else { + CV_Assert(activations.size() == 3); + f_activation = get_activation_function(activations.getStringValue(0)); + g_activation = get_activation_function(activations.getStringValue(1)); + h_activation = get_activation_function(activations.getStringValue(2)); + } + + outTailShape.clear(); + } + + bool getMemoryShapes(const std::vector &inputs, + const int requiredOutputs, + std::vector &outputs, + std::vector &internals) const CV_OVERRIDE + { + const MatShape& inp0 = inputs[0]; + const MatShape& Wx = inputs[1]; + const MatShape& Wh = inputs[2]; + + int _hidSize = Wh[2]; + int _inpSize = Wx[2]; + MatShape outTailShape_(outTailShape), outResShape; + + if (!outTailShape_.empty()) + CV_Assert(total(outTailShape_) == _hidSize); + else + outTailShape_.assign(1, _hidSize); + + // compute output shape of y + // figure out batch size + int _batchSize; + int _seqLen; + if (useTimestampDim) + { + CV_Assert(inp0.size() >= 2 && total(inp0, 2) == _inpSize); + if (layout == SEQ_BATCH_HID) { + _batchSize = inp0[1]; + _seqLen = inp0[0]; + } else { + _batchSize = inp0[0]; + _seqLen = inp0[1]; + } + outResShape.push_back(_seqLen); + } + else + { + CV_Assert(inp0.size() >= 2 && total(inp0, 1) == _inpSize); + _batchSize = inp0[0]; + } + + outResShape.push_back(1 + static_cast(bidirectional)); + outResShape.push_back(_batchSize); + outResShape.push_back(_hidSize); + outputs.assign(1, outResShape); + + int shp[] = {1 + static_cast(bidirectional), _batchSize, numHidden}; + MatShape newShape(shp, shp + sizeof(shp)/sizeof(shp[0])); + + // compute output shape of yc + if (produceCellOutput) + { + outputs.push_back(newShape); + } + // compute output shape of yh + if (produceOutputYh) + { + outputs.push_back(newShape); + } + + // internal shapes need during forward pass + internals.assign(1, shape(_batchSize, _hidSize)); // hInternal + internals.push_back(shape(_batchSize, _hidSize)); // cInternal + internals.push_back(shape(_batchSize, 1)); // dummyOnes + internals.push_back(shape(_batchSize, 4*_hidSize)); // gates + + return false; + } + + void getTypes(const std::vector& inputs, + const int requiredOutputs, + const int requiredInternals, + std::vector& outputs, + std::vector& internals) const CV_OVERRIDE + { + CV_Assert(inputs[0] == CV_32F || inputs[0] == CV_64F); // Only floating-point types are supported currently + outputs.assign(requiredOutputs, inputs[0]); + internals.assign(4, inputs[0]); + } + + + void forward(InputArrayOfArrays inputs_arr, + OutputArrayOfArrays outputs_arr, + OutputArrayOfArrays internals_arr) CV_OVERRIDE + { + + std::vector input, output, internals; + inputs_arr.getMatVector(input); + outputs_arr.getMatVector(output); + internals_arr.getMatVector(internals); + + int numInputs = input.size(); + int inpSize = input[0].size[2]; + int hidSize = numHidden; + + // determine seqLen and batchSize + if (useTimestampDim) + { + CV_Assert(input[0].dims >= 2 && (int)input[0].total(2) == inpSize); + if (layout == SEQ_BATCH_HID){ + seqLenth = input[0].size[0]; + batchSize = input[0].size[1]; + }else{ + seqLenth = input[0].size[1]; + batchSize = input[0].size[0]; + } + } else { + CV_Assert(input[0].dims >= 2 && (int)input[0].total(1) == inpSize); + seqLenth = 1; + batchSize = input[0].size[0]; + } + + std::vector blobs_; + int hidShape [] = {1 + static_cast(bidirectional), batchSize, numHidden}; + int biasShape [] = {1 + static_cast(bidirectional), 8 * numHidden}; + + blobs_.push_back(input[1].clone()); + blobs_.push_back(input[2].clone()); + switch (numInputs) { + case 3: + // X, W, R are given + // create bias + blobs_.push_back(Mat::zeros(2, biasShape, input[0].type())); + // create h0, c0 + blobs_.push_back(Mat::zeros(3, hidShape, input[0].type())); + blobs_.push_back(Mat::zeros(3, hidShape, input[0].type())); + break; + case 4: + // X, W, R, B are given + blobs_.push_back(input[3]); + // create h0, c0 + blobs_.push_back(Mat::zeros(3, hidShape, input[0].type())); + blobs_.push_back(Mat::zeros(3, hidShape, input[0].type())); + break; + case 7: + // X, W, R, B, h0, c0 are given + blobs_.push_back(input[3]); + blobs_.push_back(input[5]); + blobs_.push_back(input[6]); + break; + case 8: + // X, W, R, B, seqlen, h0, c0, P are given + blobs_.push_back(input[3]); + blobs_.push_back(input[5]); + blobs_.push_back(input[6]); + blobs_.push_back(input[7]); + break; + default: + CV_Error(Error::StsNotImplemented, "Insufficient inputs for LSTM layer. " + "Required inputs: X, W, R, B, seqLen, h0, c0 [, P for peephole]"); + } + + // set outputs to 0 + for (auto& out : output) + out.setTo(0); + + + // convert weights to 2d matrices ease of use later in the forward pass + transformBlobs(blobs_); + + const int numDirs = 1 + static_cast(bidirectional); + const int batchSizeTotal = seqLenth * batchSize; + + Mat hInternal = internals[0], + cInternal = internals[1], + dummyOnes = internals[2], + gates = internals[3]; + + + Mat cOutTs; + Mat cOut = produceCellOutput ? output[0].clone() : Mat(); + Mat hOutTs = Mat::zeros(seqLenth * batchSize, hidSize, output[0].type()); + Mat xTs = input[0].reshape(1, batchSizeTotal); + + // Prepare output[0] buffer to store the results + int shp0[] = {seqLenth * batchSize, numDirs * numHidden}; + output[0] = output[0].reshape(1, sizeof(shp0)/sizeof(shp0[0]), shp0); + + // Initialize Wx, Wh, bias, h_0, c_0, pI, pF, pO + Mat Wx, Wh, bias, h_0, c_0, pI, pF, pO; + + for (int i = 0; i < numDirs; i++) + { + // slice required weights for each direction + Wx = blobs_[0].rowRange(i * blobs_[0].rows / numDirs, (i + 1) * blobs_[0].rows / numDirs); + Wh = blobs_[1].rowRange(i * blobs_[1].rows / numDirs, (i + 1) * blobs_[1].rows / numDirs); + bias = blobs_[2].colRange(i * blobs_[2].cols / numDirs, (i + 1) * blobs_[2].cols / numDirs); + h_0 = blobs_[3].rowRange(i * blobs_[3].rows / numDirs, (i + 1) * blobs_[3].rows / numDirs); + c_0 = blobs_[4].rowRange(i * blobs_[4].rows / numDirs, (i + 1) * blobs_[4].rows / numDirs); + + if (usePeephole) + { + // slice required weights for each direction + pI = blobs_[5].rowRange(i * blobs_[5].rows / numDirs, (i + 1) * blobs_[5].rows / numDirs); + pF = blobs_[6].rowRange(i * blobs_[6].rows / numDirs, (i + 1) * blobs_[6].rows / numDirs); + pO = blobs_[7].rowRange(i * blobs_[7].rows / numDirs, (i + 1) * blobs_[7].rows / numDirs); + } + + h_0.copyTo(hInternal); + c_0.copyTo(cInternal); + dummyOnes.setTo(1.); + gates.setTo(0.); + + if (produceCellOutput) + { + cOutTs = cOut.reshape(1, batchSizeTotal); + cOutTs = cOutTs.colRange(i * cOutTs.cols / numDirs, (i + 1) * cOutTs.cols / numDirs); + } + + int tsStart, tsEnd, tsInc; + if (reverse || i == 1) { + tsStart = seqLenth - 1; + tsEnd = -1; + tsInc = -1; + } + else { + tsStart = 0; + tsEnd = seqLenth; + tsInc = 1; + } + + // main loop of LSTM forward pass + for (int ts = tsStart; ts != tsEnd; ts += tsInc) + { + Range curRowRange(ts*batchSize, (ts + 1)*batchSize); + Mat xCurr = xTs.rowRange(curRowRange); + + gemm(xCurr, Wx, 1, gates, 0, gates, GEMM_2_T); // Wx * x_t + gemm(dummyOnes, bias, 1, gates, 1, gates); //+b + gemm(hInternal, Wh, 1, gates, 1, gates, GEMM_2_T); //+Wh * h_{t-1} + + Mat gateI = gates.colRange(0*hidSize, 1*hidSize); + Mat gateF = gates.colRange(1*hidSize, 2*hidSize); + Mat gateO = gates.colRange(2*hidSize, 3*hidSize); + Mat gateG = gates.colRange(3*hidSize, 4*hidSize); + + if (forgetBias){ + add(gateF, forgetBias, gateF); + } + + if (usePeephole) + { + Mat gatesIF = gates.colRange(0, 2*hidSize); + gemm(cInternal, pI, 1, gateI, 1, gateI); + gemm(cInternal, pF, 1, gateF, 1, gateF); + f_activation(gatesIF, gatesIF); + } + else + { + Mat gatesIFO = gates.colRange(0, 3*hidSize); + f_activation(gatesIFO, gatesIFO); + } + + g_activation(gateG, gateG); + + //compute c_t + multiply(gateF, cInternal, gateF); // f_t (*) c_{t-1} + multiply(gateI, gateG, gateI); // i_t (*) g_t + add(gateF, gateI, cInternal); // c_t = f_t (*) c_{t-1} + i_t (*) g_t + + if (useCellClip) + { + min(cInternal, cellClip, cInternal); + max(cInternal, -cellClip, cInternal); + } + + if (usePeephole) + { + gemm(cInternal, pO, 1, gateO, 1, gateO); + f_activation(gateO, gateO); + } + + //compute h_t + h_activation(cInternal, hInternal); + multiply(gateO, hInternal, hInternal); + + //save results in output blobs + hInternal.copyTo(hOutTs.rowRange(curRowRange)); + + if (produceCellOutput) + cInternal.copyTo(cOutTs.rowRange(curRowRange)); + + } + + // slice in the result from each direction to the output[0] buffer + hOutTs.copyTo(output[0].colRange(i * hOutTs.cols, (i + 1) * hOutTs.cols)); + } + + // this one is needed to make make the output[0] compatible with ONNX LSTM layer standard + int shp1[] = {seqLenth, batchSize, numDirs, numHidden}; + output[0] = output[0].reshape(1, sizeof(shp1)/sizeof(shp1[0]), shp1); + + // this transpose is needed to make the output[0] compatible with ONNX LSTM layer standard + Mat tmp = output[0].clone(); + cv::transposeND(tmp, {0, 2, 1, 3}, output[0]); + + if (produceOutputYh){ + getCellStateYh(output[0], output[1], numDirs); + } + + if (produceCellOutput){ + getCellStateYc(cOut, output[2], numDirs); + } + + if (layout == BATCH_SEQ_HID) { + cv::transposeND(output[0], {2, 0, 1, 3}, output[0]); + } + + // Make sure changes are written back to outputs_arr + outputs_arr.assign(output); + } + + void getCellStateYh(Mat& scr, Mat& dst, int numDirs) + { + // TODO: implement + if (numDirs == 1){ + // take a slice of output[0] + Mat hOut = scr.rowRange(scr.size[0] - 1, scr.size[0]); + + // reshape 1x1xBxH -> 1xBxH + int shp[] = {1, batchSize, numHidden}; + hOut = hOut.reshape(1, sizeof(shp)/sizeof(shp[0]), shp); + + if (layout == BATCH_SEQ_HID){ + cv::transposeND(hOut, {1, 0, 2}, dst); + } + else{ + hOut.copyTo(dst); + } + + } else { + // there is issue here. + // Slice: SxDxBxH -> last sequence, first direction + Range ranges1[] = {cv::Range(scr.size[0] - 1, scr.size[0]), cv::Range(0, 1), cv::Range::all(), cv::Range::all()}; + Mat part1 = scr(ranges1); + + + // Slice: SxDxBxH -> first sequence, last direction + Range ranges2[] = {cv::Range(0, 1), cv::Range(scr.size[1] - 1, scr.size[1]), cv::Range::all(), cv::Range::all()}; + Mat part2 = scr(ranges2); + + + int shp[] = {1, part1.size[2] * part1.size[3]}; + part1 = part1.reshape(1, sizeof(shp)/sizeof(shp[0]), shp); + part2 = part2.reshape(1, sizeof(shp)/sizeof(shp[0]), shp); + + vconcat(part1, part2, dst); + + int finalShape[] = {2, batchSize, numHidden}; + dst = dst.reshape(1, sizeof(finalShape)/sizeof(finalShape[0]), finalShape); + + if (layout == BATCH_SEQ_HID){ + cv::transposeND(dst, {1, 0, 2}, dst); + } + } + } + + + void getCellStateYc(Mat& cOut, Mat& dst, int numDirs) + { + // seq, batch, dirs, hidden + int shp[] = {0, batchSize, numDirs, numHidden}; + cOut = cOut.reshape(1, sizeof(shp)/sizeof(shp[0]), shp); + + // permute to {0, 2, 1, 3}; + cv::Mat newCellState; + // transpose to match batch first output + if (layout == BATCH_SEQ_HID){ + cv::transposeND(cOut, {2, 0, 1, 3}, newCellState); + } + else{ + cv::transposeND(cOut, {0, 2, 1, 3}, newCellState); + } + cOut = newCellState; + + if (numDirs == 1) + { + // Slice: Yh = Y[-1, :, :, :] + Range ranges[] = {cv::Range(cOut.size[0] - 1, cOut.size[0]), cv::Range::all(), cv::Range::all(), cv::Range::all()}; + cOut = cOut(ranges); + // Reshape: 1x1xBxH -> 1xBxH + int shp[] = {1, batchSize, numHidden}; + cOut = cOut.reshape(1, sizeof(shp)/sizeof(shp[0]), shp); + cOut.copyTo(dst); + } + else + { + // Slice: SxDxBxH -> last sequence, first direction + Range ranges1[] = {cv::Range(cOut.size[0] - 1, cOut.size[0]), cv::Range(0, 1), cv::Range::all(), cv::Range::all()}; + Mat part1 = cOut(ranges1); + + // Slice: SxDxBxH -> first sequence, last direction + Range ranges2[] = {cv::Range(0, 1), cv::Range(cOut.size[1] - 1, cOut.size[1]), cv::Range::all(), cv::Range::all()}; + Mat part2 = cOut(ranges2); + + int shp[] = {1, part1.size[2] * part1.size[3]}; + part1 = part1.reshape(1, sizeof(shp)/sizeof(shp[0]), shp); + part2 = part2.reshape(1, sizeof(shp)/sizeof(shp[0]), shp); + + vconcat(part1, part2, cOut); + + // Reshape: 1x2xBxH -> 2xBxH + int finalShape[] = {2, batchSize, numHidden}; + cOut = cOut.reshape(1, sizeof(finalShape)/sizeof(finalShape[0]), finalShape); + cOut.copyTo(dst); + } + } + + void transformBlobs(std::vector& blobs) + { + Mat &Wx = blobs[0]; + Mat &Wh = blobs[1]; + Mat &b = blobs[2]; + + const int numHidden = Wh.size[2]; + + Mat h0, c0; + // check weather input is dynamic or not: hx, cx are given by user. + // Resahpe if only they are given + if (!blobs[3].empty()){ + h0 = blobs[3]; + h0 = h0.reshape(1, h0.size[0] * h0.size[1]); + } + if (!blobs[4].empty()){ + c0 = blobs[4]; + c0 = c0.reshape(1, c0.size[0] * c0.size[1]); + } + + b = b.reshape(1, b.size[0]); + Mat bx = b.colRange(0, b.cols / 2); + Mat bh = b.colRange(b.cols / 2, b.cols); + b = bx + bh; + + auto toIFOC = [] (Mat& in) { + int first = in.size[0]; + int rest = in.total() / first / 4; + // every weight blob contains weights for Input, Output, Forget and Cell gates + Mat m = in.reshape(1, {first, 4, rest}); + Mat outputGate = m.col(1); + Mat forgetGate = m.col(2); + std::swap_ranges(outputGate.begin(), outputGate.end(), forgetGate.begin()); + }; + + toIFOC(Wx); + toIFOC(Wh); + toIFOC(b); + + + Wx = Wx.reshape(1, Wx.size[0] * Wx.size[1]); + Wh = Wh.reshape(1, Wh.size[0] * Wh.size[1]); + + + blobs[0] = Wx; + blobs[1] = Wh; + blobs[2] = b.reshape(1, 1); + + if (!blobs[3].empty()){ + blobs[3] = h0; + } + if (!blobs[4].empty()){ + blobs[4] = c0; + } + + if (blobs.size() == 5) { + return; + } + + Mat P = blobs[5]; + blobs[5] = P.colRange(0, numHidden); + blobs[5] = blobs[5].clone().reshape(1, blobs[5].total()); // Single column. + blobs[5] = Mat::diag(blobs[5]); + + blobs.push_back(P.colRange(numHidden, 2 * numHidden)); + blobs[6] = blobs[6].clone().reshape(1, blobs[6].total()); // Single column. + blobs[6] = Mat::diag(blobs[6]); + + blobs.push_back(P.colRange(2 * numHidden, 3 * numHidden)); + blobs[7] = blobs[7].clone().reshape(1, blobs[7].total()); // Single column. + blobs[7] = Mat::diag(blobs[7]); + return; + } +}; + +Ptr LSTM2Layer::create(const LayerParams& params) +{ + return Ptr(new LSTM2LayerImpl(params)); +}; + +}} diff --git a/modules/dnn/src/onnx/onnx_importer2.cpp b/modules/dnn/src/onnx/onnx_importer2.cpp index 92054fa18b..cb0279873a 100644 --- a/modules/dnn/src/onnx/onnx_importer2.cpp +++ b/modules/dnn/src/onnx/onnx_importer2.cpp @@ -1113,7 +1113,32 @@ void ONNXImporter2::parseConstant(LayerParams& layerParams, const opencv_onnx::N // BUG: https://github.com/opencv/opencv/issues/26308 void ONNXImporter2::parseLSTM(LayerParams& layerParams, const opencv_onnx::NodeProto& node_proto_) { - rememberMissingOp(node_proto_.op_type()); + opencv_onnx::NodeProto lstm_proto = node_proto_; + layerParams.type = "LSTM2"; + + layerParams.set("is_onnx", true); + layerParams.set("reverse", layerParams.get("direction", "") == "reverse"); + layerParams.set("bidirectional", layerParams.get("direction", "") == "bidirectional"); + + + + bool need_yc = lstm_proto.output_size() > 2 && !lstm_proto.output(2).empty(); + bool need_yh = lstm_proto.output_size() > 1 && !lstm_proto.output(1).empty(); + bool need_y = lstm_proto.output_size() > 0 && !lstm_proto.output(0).empty(); + + const std::string y_name = need_y ? lstm_proto.output(0) : ""; + const std::string yh_name = need_yh ? lstm_proto.output(1) : ""; + const std::string yc_name = need_yc ? lstm_proto.output(2) : ""; + + layerParams.set("produce_cell_output", need_yc); + layerParams.set("produce_output_yh", need_yh); + + + if (lstm_proto.input_size() == 8) + layerParams.set("use_peephole", true); + + + addLayer(layerParams, lstm_proto); } // BUG: https://github.com/opencv/opencv/issues/26309 diff --git a/modules/dnn/test/test_layers.cpp b/modules/dnn/test/test_layers.cpp index 55c2efe33c..115c85c3ec 100644 --- a/modules/dnn/test/test_layers.cpp +++ b/modules/dnn/test/test_layers.cpp @@ -2740,4 +2740,54 @@ INSTANTIATE_TEST_CASE_P(TestLayerFusion, ConvolutionActivationEltwiseFusion, Com TestLayerFusion::dnnBackendsAndTargetsForFusionTests() )); +TEST(Layer_LSTM, repeatedInference) +{ + std::string onnx_file_path = findDataFile("dnn/onnx/models/onnxscript_lstm.onnx", false); + + // Test parameters + const int batch_size = 1; + const int seq_length = 5; + const int input_size = 6; + const int hidden_size = 4; + const int num_directions = 1; + + // Create random input tensors + int x_shape [] = {seq_length, batch_size, input_size}; + int h_0_shape [] = {num_directions, batch_size, hidden_size}; + int c_0_shape [] = {num_directions, batch_size, hidden_size}; + + Mat X(3, x_shape, CV_32F); + Mat h_0(3, h_0_shape, CV_32F); + Mat c_0(3, c_0_shape, CV_32F); + + randu(X, 0, 1); + randu(h_0, 0, 1); + randu(c_0, 0, 1); + + // Load and run ONNX model + dnn::Net net = dnn::readNet(onnx_file_path); + std::vector outputNames = {"Y", "Y_h", "Y_c"}; + + std::vector> all_outputs; + // Run the model twice + for (int i = 0; i < 2; i++) { + + net.setInput(X, "X"); + net.setInput(h_0, "initial_h"); + net.setInput(c_0, "initial_c"); + + std::vector outputs; + net.forward(outputs, outputNames); + // std::cout << "........pass " << (i + 1) << " done........" << std::endl; + all_outputs.push_back(outputs); + } + // convert to assertions + double diff0 = cv::norm(all_outputs[0][0], all_outputs[1][0], NORM_L1); + double diff1 = cv::norm(all_outputs[0][1], all_outputs[1][1], NORM_L1); + double diff2 = cv::norm(all_outputs[0][2], all_outputs[1][2], NORM_L1); + EXPECT_EQ(diff0, 0.); + EXPECT_EQ(diff1, 0.); + EXPECT_EQ(diff2, 0.); +} + }} // namespace diff --git a/modules/dnn/test/test_onnx_importer.cpp b/modules/dnn/test/test_onnx_importer.cpp index 4c8bd9aea6..7963741b70 100644 --- a/modules/dnn/test/test_onnx_importer.cpp +++ b/modules/dnn/test/test_onnx_importer.cpp @@ -1327,7 +1327,7 @@ TEST_P(Test_ONNX_layers, Split_EltwiseMax) #endif testONNXModels("split_max"); } - +// Fails with the new engine. Output shape [?, N, OutputSize] not supported by new graph engine TEST_P(Test_ONNX_layers, LSTM_Activations) { if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH) @@ -1497,15 +1497,21 @@ TEST_P(Test_ONNX_layers, LSTM_init_h0_c0) applyTestTag(CV_TEST_TAG_DNN_SKIP_CUDA); testONNXModels("lstm_init_h0_c0", npy, 0, 0, false, false, 3); } + // epsilon is larger because onnx does not match with torch/opencv exactly -TEST_P(Test_ONNX_layers, LSTM_layout_seq) +// Test uses incorrect ONNX and test data with 3 dims instead of 4. +// ONNNRuntime does not support layout=1 attiribute inference. See a detailed issue #26456 +TEST_P(Test_ONNX_layers, DISABLED_LSTM_layout_seq) { if(backend == DNN_BACKEND_CUDA) applyTestTag(CV_TEST_TAG_DNN_SKIP_CUDA); testONNXModels("lstm_layout_0", npy, 0.005, 0.005, false, false, 3); } + // epsilon is larger because onnx does not match with torch/opencv exactly -TEST_P(Test_ONNX_layers, LSTM_layout_batch) +// Test uses incorrect ONNX and test data with 3 dims instead of 4. +// ONNNRuntime does not support layout=1 attiribute inference. See a detailed issue #26456 +TEST_P(Test_ONNX_layers, DISABLED_LSTM_layout_batch) { if(backend == DNN_BACKEND_CUDA) applyTestTag(CV_TEST_TAG_DNN_SKIP_CUDA);