1
0
mirror of https://github.com/opencv/opencv.git synced 2026-07-30 07:43:03 +04:00

Merge pull request #29221 from abhishek-gola:oom_issue_fixed

Fixed Out-of-Memory issue and added VLM sample #29221

### Pull Request Readiness Checklist

See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request

- [x] I agree to contribute to the project under Apache 2 License.
- [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV
- [x] The PR is proposed to the proper branch
- [x] There is a reference to the original bug report and related work
- [x] There is accuracy test, performance test and test data in opencv_extra repository, if applicable
      Patch to opencv_extra has the same branch name.
- [x] The feature is well documented and sample code can be built with the project CMake
This commit is contained in:
Abhishek Gola
2026-06-03 19:59:30 +05:30
committed by GitHub
parent b0b77e7b32
commit e3fc091de4
4 changed files with 189 additions and 46 deletions
+1 -1
View File
@@ -145,7 +145,7 @@ public:
const bool big_enough = (int64_t)Cout * (int64_t)Cin >= (int64_t)(256 * 256);
if (ksize_all_one && strides_all_one && big_enough) {
size_t pack_bytes = mlasSgemmPackBSize(false, true, Cout, Cin);
if (pack_bytes > 0) {
if (pack_bytes > 0 && pack_bytes <= (size_t)INT_MAX) {
mlas_packed_B_.create(1, (int)pack_bytes, CV_8U);
// Weight is (Cout, Cin, 1, ..., 1) contiguous; reshape to
// (Cout, Cin) and PackB with trans_b=true.
+53 -42
View File
@@ -256,56 +256,67 @@ public:
// pack B if it is const
if (constB(mode) && blobs[0].data != last_packed_blob_data) {
fastGemmPackB(blobs[0], packed_B, trans_b, opt);
// Pre-pack B in the "thin" layout when the gemm shape has a
// small leading dim (M <= FAST_GEMM_THIN_MAX_M).
packed_B.clear();
packed_B.shrink_to_fit();
thin_packed_B.clear();
if (!trans_a && blobs[0].type() == CV_32F) {
thin_packed_B.shrink_to_fit();
bool mlas_packed = false;
#ifdef HAVE_MLAS
packed_B_mlas.release();
packed_B_mlas_M = packed_B_mlas_N = packed_B_mlas_K = 0;
if (mlasAvailable()) {
std::vector<Mat> outputs;
outputs_arr.getMatVector(outputs);
if (!outputs.empty()) {
const auto &Y = outputs[0];
const auto shape_Y = shape(Y);
const int N = shape_Y.back();
const int K = trans_b ? blobs[0].size[1] : blobs[0].size[0];
const int rows_thin = flatten_a ? shape_Y[shape_Y.size() - 2]
: (int)(Y.total() / (size_t)N);
if (fastGemmThinEligible(rows_thin, N, K)) {
thin_packed_B.resize(fastGemmThinPackBSize(N, K));
const size_t ldb_K = trans_b ? 1 : N;
const size_t ldb_N = trans_b ? K : 1;
fastGemmThinPackB(N, K, blobs[0].ptr<const float>(),
ldb_K, ldb_N, thin_packed_B.data());
const auto shape_A = shape(inputs[0]);
const auto shape_Y = shape(outputs[0]);
const int na = shape_A[shape_A.size() - 1];
const int ma = shape_A[shape_A.size() - 2];
const int N = shape_Y[shape_Y.size() - 1];
const int M = shape_Y[shape_Y.size() - 2];
const int K = trans_a ? ma : na;
const Mat& Bmat = blobs[0];
const int ldb = Bmat.size[Bmat.dims - 1];
const size_t packed_bytes = mlasSgemmPackBSize(trans_a, trans_b, N, K);
if (packed_bytes > 0 && packed_bytes <= static_cast<size_t>(INT_MAX)) {
packed_B_mlas.create(1, static_cast<int>(packed_bytes), CV_8U);
if (mlasSgemmPackB(trans_a, trans_b, N, K,
Bmat.ptr<const float>(), ldb,
packed_B_mlas.data)) {
packed_B_mlas_M = M;
packed_B_mlas_N = N;
packed_B_mlas_K = K;
mlas_packed = true;
} else {
packed_B_mlas.release();
}
}
}
#ifdef HAVE_MLAS
std::vector<Mat> outputs;
outputs_arr.getMatVector(outputs);
const auto shape_A = shape(inputs[0]);
const auto shape_Y = shape(outputs[0]);
const int na = shape_A[shape_A.size() - 1];
const int ma = shape_A[shape_A.size() - 2];
const int N = shape_Y[shape_Y.size() - 1];
const int M = shape_Y[shape_Y.size() - 2];
const int K = trans_a ? ma : na;
const Mat& Bmat = blobs[0];
const int ldb = Bmat.size[Bmat.dims - 1];
const size_t packed_bytes = mlasSgemmPackBSize(trans_a, trans_b, N, K);
if (packed_bytes > 0) {
packed_B_mlas.create(1, static_cast<int>(packed_bytes), CV_8U);
if (mlasSgemmPackB(trans_a, trans_b, N, K,
Bmat.ptr<const float>(), ldb,
packed_B_mlas.data)) {
packed_B_mlas_M = M;
packed_B_mlas_N = N;
packed_B_mlas_K = K;
} else {
packed_B_mlas.release();
#endif
if (!mlas_packed) {
fastGemmPackB(blobs[0], packed_B, trans_b, opt);
if (!trans_a && blobs[0].type() == CV_32F) {
std::vector<Mat> outputs;
outputs_arr.getMatVector(outputs);
if (!outputs.empty()) {
const auto &Y = outputs[0];
const auto shape_Y = shape(Y);
const int N = shape_Y.back();
const int K = trans_b ? blobs[0].size[1] : blobs[0].size[0];
const int rows_thin = flatten_a ? shape_Y[shape_Y.size() - 2]
: (int)(Y.total() / (size_t)N);
if (fastGemmThinEligible(rows_thin, N, K)) {
thin_packed_B.resize(fastGemmThinPackBSize(N, K));
const size_t ldb_K = trans_b ? 1 : N;
const size_t ldb_N = trans_b ? K : 1;
fastGemmThinPackB(N, K, blobs[0].ptr<const float>(),
ldb_K, ldb_N, thin_packed_B.data());
}
}
}
}
#endif
last_packed_blob_data = blobs[0].data;
}
+5 -3
View File
@@ -158,8 +158,9 @@ class MatMulLayerImpl CV_FINAL : public MatMulLayer {
const Mat* B_mat = !blobs.empty() ? &blobs[0] :
(inputs.size() >= 2 && inputs[1].dims == 2 ? &inputs[1] : nullptr);
if (B_mat && B_mat->data != last_packed_input_B_data) {
fastGemmPackB(*B_mat, packed_input_B, trans_b, opt);
helper.updatePackedBOffsets(packed_input_B.size());
packed_input_B.clear();
packed_input_B.shrink_to_fit();
thin_packed_B.clear();
if (helper.batch == 1 && B_mat->type() == CV_32F &&
fastGemmThinEligible(helper.M, helper.N, helper.K)) {
@@ -169,7 +170,8 @@ class MatMulLayerImpl CV_FINAL : public MatMulLayer {
(size_t)helper.ldb0, (size_t)helper.ldb1,
thin_packed_B.data());
} else {
thin_packed_B.clear();
fastGemmPackB(*B_mat, packed_input_B, trans_b, opt);
helper.updatePackedBOffsets(packed_input_B.size());
}
last_packed_input_B_data = B_mat->data;
}