mirror of
https://github.com/opencv/opencv.git
synced 2026-07-31 08:13:04 +04:00
Merge pull request #15494 from everton1984:hal_vector_get_n
Improving VSX performance of integral function * Adding support for vector get function on VSX datatypes so the integral function gains a bit of performance. * Removing get as a datatype member function and implementing a new HAL instruction v_extract_n to get the n-th element of a vector register. * Adding SSE/NEON/AVX intrinsics. * Implement new HAL instruction v_broadcast_element on VSX/AVX/NEON/SSE. * core(simd): add tests for v_extract_n/v_broadcast_element - updated docs - commented out code to repair compilation - added WASM and MSA default implementations * core(simd): fix compilation - x86: avoid _mm256_extract_epi64/32/16/8 with MSVS 2015 - x86: _mm_extract_epi64 is 64-bit only * cleanup
This commit is contained in:
committed by
Alexander Alekhin
parent
9d14c0b37a
commit
75315fb297
@@ -147,7 +147,8 @@ struct Integral_SIMD<uchar, int, double>
|
||||
v_expand(el8, el4l, el4h);
|
||||
el4l += prev;
|
||||
el4h += el4l;
|
||||
prev = vx_setall_s32(v_rotate_right<v_int32::nlanes - 1>(el4h).get0());
|
||||
|
||||
prev = v_broadcast_element<v_int32::nlanes - 1>(el4h);
|
||||
#endif
|
||||
v_store(sum_row + j , el4l + vx_load(prev_sum_row + j ));
|
||||
v_store(sum_row + j + v_int32::nlanes, el4h + vx_load(prev_sum_row + j + v_int32::nlanes));
|
||||
@@ -215,7 +216,8 @@ struct Integral_SIMD<uchar, float, double>
|
||||
v_expand(el8, el4li, el4hi);
|
||||
el4l = v_cvt_f32(el4li) + prev;
|
||||
el4h = v_cvt_f32(el4hi) + el4l;
|
||||
prev = vx_setall_f32(v_rotate_right<v_float32::nlanes - 1>(el4h).get0());
|
||||
|
||||
prev = v_broadcast_element<v_float32::nlanes - 1>(el4h);
|
||||
#endif
|
||||
v_store(sum_row + j , el4l + vx_load(prev_sum_row + j ));
|
||||
v_store(sum_row + j + v_float32::nlanes, el4h + vx_load(prev_sum_row + j + v_float32::nlanes));
|
||||
|
||||
Reference in New Issue
Block a user