mirror of
https://github.com/opencv/opencv.git
synced 2026-07-30 15:53:03 +04:00
Merge pull request #28836 from varun-jaiswal17:fix-dnn-nan-bugs
Fix dnn NaN bugs in GELU SIMD and softmax for large inputs #28836 ### Bug Description - **GELU SIMD bug:** `exp(2*inner)` overflows to `inf` for large inputs (e.g. `x=10.6` → `inner≈51`), causing `inf/inf = NaN`; fixed by clamping `inner` to [-9, 9] as `tanh` already saturates to ±1.0 beyond this range. - **Softmax bug:** All `-inf` inputs (masked attention rows) produce `sum=0`, then `1/0 = inf` and `0*inf = NaN`; fixed by outputting zeros when `sum == 0`. - Add regression tests for both fixes Depends on : https://github.com/opencv/opencv/pull/28837 ### Pull Request Readiness Checklist See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request - [x] I agree to contribute to the project under Apache 2 License. - [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV - [x] The PR is proposed to the proper branch - [x] There is a reference to the original bug report and related work - [x] There is accuracy test, performance test and test data in opencv_extra repository, if applicable Patch to opencv_extra has the same branch name. - [x] The feature is well documented and sample code can be built with the project CMake
This commit is contained in:
@@ -41,6 +41,7 @@
|
||||
|
||||
#include "test_precomp.hpp"
|
||||
#include <opencv2/core/ocl.hpp>
|
||||
#include <opencv2/core/fast_math.hpp>
|
||||
#include "npy_blob.hpp"
|
||||
#include <opencv2/dnn/shape_utils.hpp>
|
||||
#include <opencv2/dnn/all_layers.hpp>
|
||||
@@ -2959,4 +2960,59 @@ TEST(ConvolutionWinograd, Accuracy)
|
||||
normAssert(outLarge, refLarge, "Large input after small", 0.0, 0.0);
|
||||
}
|
||||
|
||||
TEST(Layer_Test_GeluApprox, NoNaN_LargeInput)
|
||||
{
|
||||
LayerParams lp;
|
||||
lp.type = "GeluApproximation";
|
||||
lp.name = "test_gelu_approx";
|
||||
Ptr<Layer> layer = LayerFactory::createLayerInstance("GeluApproximation", lp);
|
||||
ASSERT_TRUE(layer != nullptr);
|
||||
|
||||
float data[] = {-15.f, -10.f, -7.4f, -1.f, 0.f, 1.f, 5.f, 10.6f, 15.f, 20.f};
|
||||
int dims[] = {1, 1, 10};
|
||||
Mat inp(3, dims, CV_32F, data);
|
||||
std::vector<Mat> inpVec = {inp};
|
||||
std::vector<Mat> outVec;
|
||||
|
||||
runLayer(layer, inpVec, outVec);
|
||||
ASSERT_EQ(outVec.size(), (size_t)1);
|
||||
|
||||
Mat& out = outVec[0];
|
||||
for (int i = 0; i < 10; i++) {
|
||||
float val = out.ptr<float>()[i];
|
||||
EXPECT_FALSE(cvIsNaN(val)) << "NaN at index " << i << " (input=" << data[i] << ")";
|
||||
EXPECT_FALSE(cvIsInf(val)) << "Inf at index " << i << " (input=" << data[i] << ")";
|
||||
}
|
||||
|
||||
EXPECT_NEAR(out.ptr<float>()[9], 20.f, 0.01f);
|
||||
EXPECT_NEAR(out.ptr<float>()[0], 0.f, 1e-6f);
|
||||
EXPECT_NEAR(out.ptr<float>()[4], 0.f, 1e-6f);
|
||||
}
|
||||
|
||||
TEST(Layer_Test_Softmax, NoNaN_AllNegInf)
|
||||
{
|
||||
LayerParams lp;
|
||||
lp.type = "Softmax";
|
||||
lp.name = "test_softmax";
|
||||
lp.set("axis", 1);
|
||||
Ptr<Layer> layer = LayerFactory::createLayerInstance("Softmax", lp);
|
||||
ASSERT_TRUE(layer != nullptr);
|
||||
|
||||
int dims[] = {1, 8};
|
||||
Mat inp(2, dims, CV_32F, Scalar(-std::numeric_limits<float>::infinity()));
|
||||
std::vector<Mat> inpVec = {inp};
|
||||
std::vector<Mat> outVec;
|
||||
|
||||
runLayer(layer, inpVec, outVec);
|
||||
ASSERT_EQ(outVec.size(), (size_t)1);
|
||||
|
||||
Mat& out = outVec[0];
|
||||
for (int i = 0; i < 8; i++) {
|
||||
float val = out.ptr<float>()[i];
|
||||
EXPECT_FALSE(cvIsNaN(val)) << "NaN at index " << i;
|
||||
EXPECT_FALSE(cvIsInf(val)) << "Inf at index " << i;
|
||||
EXPECT_EQ(val, 0.f) << "Expected 0 at index " << i;
|
||||
}
|
||||
}
|
||||
|
||||
}} // namespace
|
||||
|
||||
Reference in New Issue
Block a user