mirror of
https://github.com/opencv/opencv.git
synced 2026-07-30 15:53:03 +04:00
Merge pull request #27000 from GenshinImpactStarts:cart_to_polar
[HAL RVV] reuse atan | impl cart_to_polar | add perf test #27000 Implement through the existing `cv_hal_cartToPolar32f` and `cv_hal_cartToPolar64f` interfaces. Add `cartToPolar` performance tests. cv_hal_rvv::fast_atan is modified to make it more reusable because it's needed in cartToPolar. **UPDATE**: UI enabled. Since the vec type of RVV can't be stored in struct. UI implementation of `v_atan_f32` is modified. Both `fastAtan` and `cartToPolar` are affected so the test result for `atan` is also appended. I have tested the modified UI on RVV and AVX2 and no regressions appears. Perf test done on MUSE-PI. AVX2 test done on Intel(R) Xeon(R) Gold 6140 CPU @ 2.30GHz. ```sh $ opencv_test_core --gtest_filter="*CartToPolar*:*Core_CartPolar_reverse*:*Phase*" $ opencv_perf_core --gtest_filter="*CartToPolar*:*phase*" --perf_min_samples=300 --perf_force_samples=300 ``` Test result between enabled UI and HAL: ``` Name of Test ui rvv rvv vs ui (x-factor) CartToPolar::CartToPolarFixture::(127x61, 32FC1) 0.106 0.059 1.80 CartToPolar::CartToPolarFixture::(127x61, 64FC1) 0.155 0.070 2.20 CartToPolar::CartToPolarFixture::(640x480, 32FC1) 4.188 2.317 1.81 CartToPolar::CartToPolarFixture::(640x480, 64FC1) 6.593 2.889 2.28 CartToPolar::CartToPolarFixture::(1280x720, 32FC1) 12.600 7.057 1.79 CartToPolar::CartToPolarFixture::(1280x720, 64FC1) 19.860 8.797 2.26 CartToPolar::CartToPolarFixture::(1920x1080, 32FC1) 28.295 15.809 1.79 CartToPolar::CartToPolarFixture::(1920x1080, 64FC1) 44.573 19.398 2.30 phase32f::VectorLength::128 0.002 0.002 1.20 phase32f::VectorLength::1000 0.008 0.006 1.32 phase32f::VectorLength::131072 1.061 0.731 1.45 phase32f::VectorLength::524288 3.997 2.976 1.34 phase32f::VectorLength::1048576 8.001 5.959 1.34 phase64f::VectorLength::128 0.002 0.002 1.33 phase64f::VectorLength::1000 0.012 0.008 1.58 phase64f::VectorLength::131072 1.648 0.931 1.77 phase64f::VectorLength::524288 6.836 3.837 1.78 phase64f::VectorLength::1048576 14.060 7.540 1.86 ``` Test result before and after enabling UI on RVV: ``` Name of Test perf perf perf ui ui ui orig pr pr vs perf ui orig (x-factor) CartToPolar::CartToPolarFixture::(127x61, 32FC1) 0.141 0.106 1.33 CartToPolar::CartToPolarFixture::(127x61, 64FC1) 0.187 0.155 1.20 CartToPolar::CartToPolarFixture::(640x480, 32FC1) 5.990 4.188 1.43 CartToPolar::CartToPolarFixture::(640x480, 64FC1) 8.370 6.593 1.27 CartToPolar::CartToPolarFixture::(1280x720, 32FC1) 18.214 12.600 1.45 CartToPolar::CartToPolarFixture::(1280x720, 64FC1) 25.365 19.860 1.28 CartToPolar::CartToPolarFixture::(1920x1080, 32FC1) 40.437 28.295 1.43 CartToPolar::CartToPolarFixture::(1920x1080, 64FC1) 56.699 44.573 1.27 phase32f::VectorLength::128 0.003 0.002 1.54 phase32f::VectorLength::1000 0.016 0.008 1.90 phase32f::VectorLength::131072 2.048 1.061 1.93 phase32f::VectorLength::524288 8.219 3.997 2.06 phase32f::VectorLength::1048576 16.426 8.001 2.05 phase64f::VectorLength::128 0.003 0.002 1.44 phase64f::VectorLength::1000 0.020 0.012 1.60 phase64f::VectorLength::131072 2.621 1.648 1.59 phase64f::VectorLength::524288 10.780 6.836 1.58 phase64f::VectorLength::1048576 22.723 14.060 1.62 ``` Test result before and after modifying UI on AVX2: ``` Name of Test perf perf perf avx2 avx2 avx2 orig pr pr vs perf avx2 orig (x-factor) CartToPolar::CartToPolarFixture::(127x61, 32FC1) 0.006 0.005 1.14 CartToPolar::CartToPolarFixture::(127x61, 64FC1) 0.010 0.009 1.08 CartToPolar::CartToPolarFixture::(640x480, 32FC1) 0.273 0.264 1.03 CartToPolar::CartToPolarFixture::(640x480, 64FC1) 0.511 0.487 1.05 CartToPolar::CartToPolarFixture::(1280x720, 32FC1) 0.760 0.723 1.05 CartToPolar::CartToPolarFixture::(1280x720, 64FC1) 2.009 1.937 1.04 CartToPolar::CartToPolarFixture::(1920x1080, 32FC1) 1.996 1.923 1.04 CartToPolar::CartToPolarFixture::(1920x1080, 64FC1) 5.721 5.509 1.04 phase32f::VectorLength::128 0.000 0.000 0.98 phase32f::VectorLength::1000 0.001 0.001 0.97 phase32f::VectorLength::131072 0.105 0.111 0.95 phase32f::VectorLength::524288 0.402 0.402 1.00 phase32f::VectorLength::1048576 0.775 0.767 1.01 phase64f::VectorLength::128 0.000 0.000 1.00 phase64f::VectorLength::1000 0.001 0.001 1.01 phase64f::VectorLength::131072 0.163 0.162 1.01 phase64f::VectorLength::524288 0.669 0.653 1.02 phase64f::VectorLength::1048576 1.660 1.634 1.02 ``` ### Pull Request Readiness Checklist See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request - [x] I agree to contribute to the project under Apache 2 License. - [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV - [ ] The PR is proposed to the proper branch - [ ] There is a reference to the original bug report and related work - [ ] There is accuracy test, performance test and test data in opencv_extra repository, if applicable Patch to opencv_extra has the same branch name. - [ ] The feature is well documented and sample code can be built with the project CMake
This commit is contained in:
committed by
GitHub
parent
b129abfdaa
commit
2a8d4b8e43
@@ -73,48 +73,30 @@ static inline float atan_f32(float y, float x)
|
||||
}
|
||||
#endif
|
||||
|
||||
#if CV_SIMD
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
struct v_atan_f32
|
||||
v_float32 v_atan_f32(const v_float32& y, const v_float32& x)
|
||||
{
|
||||
explicit v_atan_f32(const float& scale)
|
||||
{
|
||||
eps = vx_setall_f32((float)DBL_EPSILON);
|
||||
z = vx_setzero_f32();
|
||||
p7 = vx_setall_f32(atan2_p7);
|
||||
p5 = vx_setall_f32(atan2_p5);
|
||||
p3 = vx_setall_f32(atan2_p3);
|
||||
p1 = vx_setall_f32(atan2_p1);
|
||||
val90 = vx_setall_f32(90.f);
|
||||
val180 = vx_setall_f32(180.f);
|
||||
val360 = vx_setall_f32(360.f);
|
||||
s = vx_setall_f32(scale);
|
||||
}
|
||||
v_float32 eps = vx_setall_f32((float)DBL_EPSILON);
|
||||
v_float32 z = vx_setzero_f32();
|
||||
v_float32 p7 = vx_setall_f32(atan2_p7);
|
||||
v_float32 p5 = vx_setall_f32(atan2_p5);
|
||||
v_float32 p3 = vx_setall_f32(atan2_p3);
|
||||
v_float32 p1 = vx_setall_f32(atan2_p1);
|
||||
v_float32 val90 = vx_setall_f32(90.f);
|
||||
v_float32 val180 = vx_setall_f32(180.f);
|
||||
v_float32 val360 = vx_setall_f32(360.f);
|
||||
|
||||
v_float32 compute(const v_float32& y, const v_float32& x)
|
||||
{
|
||||
v_float32 ax = v_abs(x);
|
||||
v_float32 ay = v_abs(y);
|
||||
v_float32 c = v_div(v_min(ax, ay), v_add(v_max(ax, ay), this->eps));
|
||||
v_float32 cc = v_mul(c, c);
|
||||
v_float32 a = v_mul(v_fma(v_fma(v_fma(cc, this->p7, this->p5), cc, this->p3), cc, this->p1), c);
|
||||
a = v_select(v_ge(ax, ay), a, v_sub(this->val90, a));
|
||||
a = v_select(v_lt(x, this->z), v_sub(this->val180, a), a);
|
||||
a = v_select(v_lt(y, this->z), v_sub(this->val360, a), a);
|
||||
return v_mul(a, this->s);
|
||||
}
|
||||
|
||||
v_float32 eps;
|
||||
v_float32 z;
|
||||
v_float32 p7;
|
||||
v_float32 p5;
|
||||
v_float32 p3;
|
||||
v_float32 p1;
|
||||
v_float32 val90;
|
||||
v_float32 val180;
|
||||
v_float32 val360;
|
||||
v_float32 s;
|
||||
};
|
||||
v_float32 ax = v_abs(x);
|
||||
v_float32 ay = v_abs(y);
|
||||
v_float32 c = v_div(v_min(ax, ay), v_add(v_max(ax, ay), eps));
|
||||
v_float32 cc = v_mul(c, c);
|
||||
v_float32 a = v_mul(v_fma(v_fma(v_fma(cc, p7, p5), cc, p3), cc, p1), c);
|
||||
a = v_select(v_ge(ax, ay), a, v_sub(val90, a));
|
||||
a = v_select(v_lt(x, z), v_sub(val180, a), a);
|
||||
a = v_select(v_lt(y, z), v_sub(val360, a), a);
|
||||
return a;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -124,9 +106,9 @@ static void cartToPolar32f_(const float *X, const float *Y, float *mag, float *a
|
||||
{
|
||||
float scale = angleInDegrees ? 1.f : (float)(CV_PI/180);
|
||||
int i = 0;
|
||||
#if CV_SIMD
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
const int VECSZ = VTraits<v_float32>::vlanes();
|
||||
v_atan_f32 v(scale);
|
||||
v_float32 s = vx_setall_f32(scale);
|
||||
|
||||
for( ; i < len; i += VECSZ*2 )
|
||||
{
|
||||
@@ -148,8 +130,8 @@ static void cartToPolar32f_(const float *X, const float *Y, float *mag, float *a
|
||||
v_float32 m0 = v_sqrt(v_muladd(x0, x0, v_mul(y0, y0)));
|
||||
v_float32 m1 = v_sqrt(v_muladd(x1, x1, v_mul(y1, y1)));
|
||||
|
||||
v_float32 r0 = v.compute(y0, x0);
|
||||
v_float32 r1 = v.compute(y1, x1);
|
||||
v_float32 r0 = v_mul(v_atan_f32(y0, x0), s);
|
||||
v_float32 r1 = v_mul(v_atan_f32(y1, x1), s);
|
||||
|
||||
v_store(mag + i, m0);
|
||||
v_store(mag + i + VECSZ, m1);
|
||||
@@ -200,9 +182,9 @@ static void fastAtan32f_(const float *Y, const float *X, float *angle, int len,
|
||||
{
|
||||
float scale = angleInDegrees ? 1.f : (float)(CV_PI/180);
|
||||
int i = 0;
|
||||
#if CV_SIMD
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
const int VECSZ = VTraits<v_float32>::vlanes();
|
||||
v_atan_f32 v(scale);
|
||||
v_float32 s = vx_setall_f32(scale);
|
||||
|
||||
for( ; i < len; i += VECSZ*2 )
|
||||
{
|
||||
@@ -221,8 +203,8 @@ static void fastAtan32f_(const float *Y, const float *X, float *angle, int len,
|
||||
v_float32 y1 = vx_load(Y + i + VECSZ);
|
||||
v_float32 x1 = vx_load(X + i + VECSZ);
|
||||
|
||||
v_float32 r0 = v.compute(y0, x0);
|
||||
v_float32 r1 = v.compute(y1, x1);
|
||||
v_float32 r0 = v_mul(v_atan_f32(y0, x0), s);
|
||||
v_float32 r1 = v_mul(v_atan_f32(y1, x1), s);
|
||||
|
||||
v_store(angle + i, r0);
|
||||
v_store(angle + i + VECSZ, r1);
|
||||
|
||||
Reference in New Issue
Block a user