1
0
mirror of https://github.com/opencv/opencv.git synced 2026-07-30 15:53:03 +04:00

Merge pull request #27000 from GenshinImpactStarts:cart_to_polar

[HAL RVV] reuse atan | impl cart_to_polar | add perf test #27000

Implement through the existing `cv_hal_cartToPolar32f` and `cv_hal_cartToPolar64f` interfaces.

Add `cartToPolar` performance tests.

cv_hal_rvv::fast_atan is modified to make it more reusable because it's needed in cartToPolar.

**UPDATE**: UI enabled. Since the vec type of RVV can't be stored in struct. UI implementation of `v_atan_f32` is modified. Both `fastAtan` and `cartToPolar` are affected so the test result for `atan` is also appended. I have tested the modified UI on RVV and AVX2 and no regressions appears.

Perf test done on MUSE-PI. AVX2 test done on Intel(R) Xeon(R) Gold 6140 CPU @ 2.30GHz.

```sh
$ opencv_test_core --gtest_filter="*CartToPolar*:*Core_CartPolar_reverse*:*Phase*" 
$ opencv_perf_core --gtest_filter="*CartToPolar*:*phase*" --perf_min_samples=300 --perf_force_samples=300
```

Test result between enabled UI and HAL:
```
                   Name of Test                       ui    rvv      rvv    
                                                                      vs    
                                                                      ui    
                                                                  (x-factor)
CartToPolar::CartToPolarFixture::(127x61, 32FC1)    0.106  0.059     1.80   
CartToPolar::CartToPolarFixture::(127x61, 64FC1)    0.155  0.070     2.20   
CartToPolar::CartToPolarFixture::(640x480, 32FC1)   4.188  2.317     1.81   
CartToPolar::CartToPolarFixture::(640x480, 64FC1)   6.593  2.889     2.28   
CartToPolar::CartToPolarFixture::(1280x720, 32FC1)  12.600 7.057     1.79   
CartToPolar::CartToPolarFixture::(1280x720, 64FC1)  19.860 8.797     2.26   
CartToPolar::CartToPolarFixture::(1920x1080, 32FC1) 28.295 15.809    1.79   
CartToPolar::CartToPolarFixture::(1920x1080, 64FC1) 44.573 19.398    2.30   
phase32f::VectorLength::128                         0.002  0.002     1.20   
phase32f::VectorLength::1000                        0.008  0.006     1.32   
phase32f::VectorLength::131072                      1.061  0.731     1.45   
phase32f::VectorLength::524288                      3.997  2.976     1.34   
phase32f::VectorLength::1048576                     8.001  5.959     1.34   
phase64f::VectorLength::128                         0.002  0.002     1.33   
phase64f::VectorLength::1000                        0.012  0.008     1.58   
phase64f::VectorLength::131072                      1.648  0.931     1.77   
phase64f::VectorLength::524288                      6.836  3.837     1.78   
phase64f::VectorLength::1048576                     14.060 7.540     1.86   
```

Test result before and after enabling UI on RVV:
```
                   Name of Test                      perf   perf     perf   
                                                      ui     ui       ui    
                                                     orig    pr       pr    
                                                                      vs    
                                                                     perf   
                                                                      ui    
                                                                     orig   
                                                                  (x-factor)
CartToPolar::CartToPolarFixture::(127x61, 32FC1)    0.141  0.106     1.33   
CartToPolar::CartToPolarFixture::(127x61, 64FC1)    0.187  0.155     1.20   
CartToPolar::CartToPolarFixture::(640x480, 32FC1)   5.990  4.188     1.43   
CartToPolar::CartToPolarFixture::(640x480, 64FC1)   8.370  6.593     1.27   
CartToPolar::CartToPolarFixture::(1280x720, 32FC1)  18.214 12.600    1.45   
CartToPolar::CartToPolarFixture::(1280x720, 64FC1)  25.365 19.860    1.28   
CartToPolar::CartToPolarFixture::(1920x1080, 32FC1) 40.437 28.295    1.43   
CartToPolar::CartToPolarFixture::(1920x1080, 64FC1) 56.699 44.573    1.27   
phase32f::VectorLength::128                         0.003  0.002     1.54   
phase32f::VectorLength::1000                        0.016  0.008     1.90   
phase32f::VectorLength::131072                      2.048  1.061     1.93   
phase32f::VectorLength::524288                      8.219  3.997     2.06   
phase32f::VectorLength::1048576                     16.426 8.001     2.05   
phase64f::VectorLength::128                         0.003  0.002     1.44   
phase64f::VectorLength::1000                        0.020  0.012     1.60   
phase64f::VectorLength::131072                      2.621  1.648     1.59   
phase64f::VectorLength::524288                      10.780 6.836     1.58   
phase64f::VectorLength::1048576                     22.723 14.060    1.62   
```

Test result before and after modifying UI on AVX2:
```
                   Name of Test                     perf  perf     perf   
                                                    avx2  avx2     avx2   
                                                    orig   pr       pr    
                                                                    vs    
                                                                   perf   
                                                                   avx2   
                                                                   orig   
                                                                (x-factor)
CartToPolar::CartToPolarFixture::(127x61, 32FC1)    0.006 0.005    1.14   
CartToPolar::CartToPolarFixture::(127x61, 64FC1)    0.010 0.009    1.08   
CartToPolar::CartToPolarFixture::(640x480, 32FC1)   0.273 0.264    1.03   
CartToPolar::CartToPolarFixture::(640x480, 64FC1)   0.511 0.487    1.05   
CartToPolar::CartToPolarFixture::(1280x720, 32FC1)  0.760 0.723    1.05   
CartToPolar::CartToPolarFixture::(1280x720, 64FC1)  2.009 1.937    1.04   
CartToPolar::CartToPolarFixture::(1920x1080, 32FC1) 1.996 1.923    1.04   
CartToPolar::CartToPolarFixture::(1920x1080, 64FC1) 5.721 5.509    1.04   
phase32f::VectorLength::128                         0.000 0.000    0.98   
phase32f::VectorLength::1000                        0.001 0.001    0.97   
phase32f::VectorLength::131072                      0.105 0.111    0.95   
phase32f::VectorLength::524288                      0.402 0.402    1.00   
phase32f::VectorLength::1048576                     0.775 0.767    1.01   
phase64f::VectorLength::128                         0.000 0.000    1.00   
phase64f::VectorLength::1000                        0.001 0.001    1.01   
phase64f::VectorLength::131072                      0.163 0.162    1.01   
phase64f::VectorLength::524288                      0.669 0.653    1.02   
phase64f::VectorLength::1048576                     1.660 1.634    1.02   
```

### Pull Request Readiness Checklist

See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request

- [x] I agree to contribute to the project under Apache 2 License.
- [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV
- [ ] The PR is proposed to the proper branch
- [ ] There is a reference to the original bug report and related work
- [ ] There is accuracy test, performance test and test data in opencv_extra repository, if applicable
      Patch to opencv_extra has the same branch name.
- [ ] The feature is well documented and sample code can be built with the project CMake
This commit is contained in:
GenshinImpactStarts
2025-03-13 20:56:56 +08:00
committed by GitHub
parent b129abfdaa
commit 2a8d4b8e43
5 changed files with 158 additions and 111 deletions
+29 -47
View File
@@ -73,48 +73,30 @@ static inline float atan_f32(float y, float x)
}
#endif
#if CV_SIMD
#if (CV_SIMD || CV_SIMD_SCALABLE)
struct v_atan_f32
v_float32 v_atan_f32(const v_float32& y, const v_float32& x)
{
explicit v_atan_f32(const float& scale)
{
eps = vx_setall_f32((float)DBL_EPSILON);
z = vx_setzero_f32();
p7 = vx_setall_f32(atan2_p7);
p5 = vx_setall_f32(atan2_p5);
p3 = vx_setall_f32(atan2_p3);
p1 = vx_setall_f32(atan2_p1);
val90 = vx_setall_f32(90.f);
val180 = vx_setall_f32(180.f);
val360 = vx_setall_f32(360.f);
s = vx_setall_f32(scale);
}
v_float32 eps = vx_setall_f32((float)DBL_EPSILON);
v_float32 z = vx_setzero_f32();
v_float32 p7 = vx_setall_f32(atan2_p7);
v_float32 p5 = vx_setall_f32(atan2_p5);
v_float32 p3 = vx_setall_f32(atan2_p3);
v_float32 p1 = vx_setall_f32(atan2_p1);
v_float32 val90 = vx_setall_f32(90.f);
v_float32 val180 = vx_setall_f32(180.f);
v_float32 val360 = vx_setall_f32(360.f);
v_float32 compute(const v_float32& y, const v_float32& x)
{
v_float32 ax = v_abs(x);
v_float32 ay = v_abs(y);
v_float32 c = v_div(v_min(ax, ay), v_add(v_max(ax, ay), this->eps));
v_float32 cc = v_mul(c, c);
v_float32 a = v_mul(v_fma(v_fma(v_fma(cc, this->p7, this->p5), cc, this->p3), cc, this->p1), c);
a = v_select(v_ge(ax, ay), a, v_sub(this->val90, a));
a = v_select(v_lt(x, this->z), v_sub(this->val180, a), a);
a = v_select(v_lt(y, this->z), v_sub(this->val360, a), a);
return v_mul(a, this->s);
}
v_float32 eps;
v_float32 z;
v_float32 p7;
v_float32 p5;
v_float32 p3;
v_float32 p1;
v_float32 val90;
v_float32 val180;
v_float32 val360;
v_float32 s;
};
v_float32 ax = v_abs(x);
v_float32 ay = v_abs(y);
v_float32 c = v_div(v_min(ax, ay), v_add(v_max(ax, ay), eps));
v_float32 cc = v_mul(c, c);
v_float32 a = v_mul(v_fma(v_fma(v_fma(cc, p7, p5), cc, p3), cc, p1), c);
a = v_select(v_ge(ax, ay), a, v_sub(val90, a));
a = v_select(v_lt(x, z), v_sub(val180, a), a);
a = v_select(v_lt(y, z), v_sub(val360, a), a);
return a;
}
#endif
@@ -124,9 +106,9 @@ static void cartToPolar32f_(const float *X, const float *Y, float *mag, float *a
{
float scale = angleInDegrees ? 1.f : (float)(CV_PI/180);
int i = 0;
#if CV_SIMD
#if (CV_SIMD || CV_SIMD_SCALABLE)
const int VECSZ = VTraits<v_float32>::vlanes();
v_atan_f32 v(scale);
v_float32 s = vx_setall_f32(scale);
for( ; i < len; i += VECSZ*2 )
{
@@ -148,8 +130,8 @@ static void cartToPolar32f_(const float *X, const float *Y, float *mag, float *a
v_float32 m0 = v_sqrt(v_muladd(x0, x0, v_mul(y0, y0)));
v_float32 m1 = v_sqrt(v_muladd(x1, x1, v_mul(y1, y1)));
v_float32 r0 = v.compute(y0, x0);
v_float32 r1 = v.compute(y1, x1);
v_float32 r0 = v_mul(v_atan_f32(y0, x0), s);
v_float32 r1 = v_mul(v_atan_f32(y1, x1), s);
v_store(mag + i, m0);
v_store(mag + i + VECSZ, m1);
@@ -200,9 +182,9 @@ static void fastAtan32f_(const float *Y, const float *X, float *angle, int len,
{
float scale = angleInDegrees ? 1.f : (float)(CV_PI/180);
int i = 0;
#if CV_SIMD
#if (CV_SIMD || CV_SIMD_SCALABLE)
const int VECSZ = VTraits<v_float32>::vlanes();
v_atan_f32 v(scale);
v_float32 s = vx_setall_f32(scale);
for( ; i < len; i += VECSZ*2 )
{
@@ -221,8 +203,8 @@ static void fastAtan32f_(const float *Y, const float *X, float *angle, int len,
v_float32 y1 = vx_load(Y + i + VECSZ);
v_float32 x1 = vx_load(X + i + VECSZ);
v_float32 r0 = v.compute(y0, x0);
v_float32 r1 = v.compute(y1, x1);
v_float32 r0 = v_mul(v_atan_f32(y0, x0), s);
v_float32 r1 = v_mul(v_atan_f32(y1, x1), s);
v_store(angle + i, r0);
v_store(angle + i + VECSZ, r1);