1
0
mirror of https://github.com/opencv/opencv.git synced 2026-07-21 19:33:03 +04:00

Merge pull request #28836 from varun-jaiswal17:fix-dnn-nan-bugs

Fix dnn NaN bugs in GELU SIMD and softmax for large inputs #28836

### Bug Description

- **GELU SIMD bug:** `exp(2*inner)` overflows to `inf` for large inputs (e.g. `x=10.6` → `inner≈51`), causing `inf/inf = NaN`; fixed by clamping `inner` to [-9, 9] as `tanh` already saturates to ±1.0 beyond this range.
- **Softmax bug:** All `-inf` inputs (masked attention rows) produce `sum=0`, then `1/0 = inf` and `0*inf = NaN`; 
fixed by outputting zeros when `sum == 0`.
- Add regression tests for both fixes

Depends on : https://github.com/opencv/opencv/pull/28837

### Pull Request Readiness Checklist

See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request

- [x] I agree to contribute to the project under Apache 2 License.
- [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV
- [x] The PR is proposed to the proper branch
- [x] There is a reference to the original bug report and related work
- [x] There is accuracy test, performance test and test data in opencv_extra repository, if applicable
      Patch to opencv_extra has the same branch name.
- [x] The feature is well documented and sample code can be built with the project CMake
This commit is contained in:
Varun Jaiswal
2026-04-22 12:30:51 +05:30
committed by GitHub
parent 5e315bcc97
commit 562628eb71
3 changed files with 66 additions and 2 deletions
@@ -232,10 +232,13 @@ static void activationGELUApprox(const void* input, void* output,
v_float32 half = vx_setall_f32(0.5f), one = vx_setall_f32(1.f);
v_float32 v_s2pi = vx_setall_f32(sqrt2_pi), v_coeff = vx_setall_f32(coeff);
v_float32 two = vx_setall_f32(2.f);
// Clamp to [-9, 9] to prevent overflow in exp(2*inner); tanh saturates here anyway
v_float32 clamp_hi = vx_setall_f32(9.f), clamp_lo = vx_setall_f32(-9.f);
for (; i + vlanes <= len; i += vlanes) {
v_float32 x = vx_load(inp + i);
// inner = sqrt(2/pi) * x + coeff * x^3 = x * (sqrt(2/pi) + coeff * x^2)
v_float32 inner = v_mul(x, v_add(v_s2pi, v_mul(v_coeff, v_mul(x, x))));
inner = v_min(v_max(inner, clamp_lo), clamp_hi);
// tanh via exp: (exp(2*inner)-1)/(exp(2*inner)+1)
v_float32 e2 = v_exp(v_mul(two, inner));
v_float32 t = v_div(v_sub(e2, one), v_add(e2, one));
@@ -11,6 +11,7 @@
#include "../../precomp.hpp"
#include "softmax.hpp"
#include "opencv2/core/fast_math.hpp"
namespace cv { namespace dnn {
@@ -101,10 +102,13 @@ void softmax(Mat &dst, const Mat &src, int axis, int axisBias, int axisStep){
s += axisBuf[cnDim];
}
s = 1.f / s;
// copy back the result to src
_cnDim = 0;
if (s == 0.f || cvIsInf(1.f / s)) {
for (; _cnDim < axisStep; _cnDim++)
dstPtr[srcOffset + (_cnDim + axisBias) * cnStep] = 0.f;
} else {
s = 1.f / s;
#if CV_ENABLE_UNROLLED && defined(_M_ARM64)
for (; _cnDim + 3 < axisStep; _cnDim += 4) {
dstPtr[srcOffset + (_cnDim + 0 + axisBias) * cnStep] = axisBuf[_cnDim + 0] * s;
@@ -115,6 +119,7 @@ void softmax(Mat &dst, const Mat &src, int axis, int axisBias, int axisStep){
#endif
for (; _cnDim < axisStep; _cnDim++)
dstPtr[srcOffset + (_cnDim + axisBias) * cnStep] = axisBuf[_cnDim] * s;
}
}
}, nstripes);
}
+56
View File
@@ -41,6 +41,7 @@
#include "test_precomp.hpp"
#include <opencv2/core/ocl.hpp>
#include <opencv2/core/fast_math.hpp>
#include "npy_blob.hpp"
#include <opencv2/dnn/shape_utils.hpp>
#include <opencv2/dnn/all_layers.hpp>
@@ -2959,4 +2960,59 @@ TEST(ConvolutionWinograd, Accuracy)
normAssert(outLarge, refLarge, "Large input after small", 0.0, 0.0);
}
TEST(Layer_Test_GeluApprox, NoNaN_LargeInput)
{
LayerParams lp;
lp.type = "GeluApproximation";
lp.name = "test_gelu_approx";
Ptr<Layer> layer = LayerFactory::createLayerInstance("GeluApproximation", lp);
ASSERT_TRUE(layer != nullptr);
float data[] = {-15.f, -10.f, -7.4f, -1.f, 0.f, 1.f, 5.f, 10.6f, 15.f, 20.f};
int dims[] = {1, 1, 10};
Mat inp(3, dims, CV_32F, data);
std::vector<Mat> inpVec = {inp};
std::vector<Mat> outVec;
runLayer(layer, inpVec, outVec);
ASSERT_EQ(outVec.size(), (size_t)1);
Mat& out = outVec[0];
for (int i = 0; i < 10; i++) {
float val = out.ptr<float>()[i];
EXPECT_FALSE(cvIsNaN(val)) << "NaN at index " << i << " (input=" << data[i] << ")";
EXPECT_FALSE(cvIsInf(val)) << "Inf at index " << i << " (input=" << data[i] << ")";
}
EXPECT_NEAR(out.ptr<float>()[9], 20.f, 0.01f);
EXPECT_NEAR(out.ptr<float>()[0], 0.f, 1e-6f);
EXPECT_NEAR(out.ptr<float>()[4], 0.f, 1e-6f);
}
TEST(Layer_Test_Softmax, NoNaN_AllNegInf)
{
LayerParams lp;
lp.type = "Softmax";
lp.name = "test_softmax";
lp.set("axis", 1);
Ptr<Layer> layer = LayerFactory::createLayerInstance("Softmax", lp);
ASSERT_TRUE(layer != nullptr);
int dims[] = {1, 8};
Mat inp(2, dims, CV_32F, Scalar(-std::numeric_limits<float>::infinity()));
std::vector<Mat> inpVec = {inp};
std::vector<Mat> outVec;
runLayer(layer, inpVec, outVec);
ASSERT_EQ(outVec.size(), (size_t)1);
Mat& out = outVec[0];
for (int i = 0; i < 8; i++) {
float val = out.ptr<float>()[i];
EXPECT_FALSE(cvIsNaN(val)) << "NaN at index " << i;
EXPECT_FALSE(cvIsInf(val)) << "Inf at index " << i;
EXPECT_EQ(val, 0.f) << "Expected 0 at index " << i;
}
}
}} // namespace