mirror of
https://github.com/opencv/opencv.git
synced 2026-07-30 07:43:03 +04:00
Merge remote-tracking branch 'upstream/3.4' into merge-3.4
This commit is contained in:
@@ -7,6 +7,9 @@ ocv_add_dispatched_file(convert SSE2 AVX2)
|
||||
ocv_add_dispatched_file(convert_scale SSE2 AVX2)
|
||||
ocv_add_dispatched_file(count_non_zero SSE2 AVX2)
|
||||
ocv_add_dispatched_file(matmul SSE2 AVX2)
|
||||
ocv_add_dispatched_file(mean SSE2 AVX2)
|
||||
ocv_add_dispatched_file(merge SSE2 AVX2)
|
||||
ocv_add_dispatched_file(split SSE2 AVX2)
|
||||
ocv_add_dispatched_file(sum SSE2 AVX2)
|
||||
|
||||
# dispatching for accuracy tests
|
||||
|
||||
@@ -234,7 +234,12 @@ CV_CPU_OPTIMIZATION_HAL_NAMESPACE_BEGIN
|
||||
inline vtyp vx_##loadsfx##_low(const typ* ptr) { return prefix##_##loadsfx##_low(ptr); } \
|
||||
inline vtyp vx_##loadsfx##_halves(const typ* ptr0, const typ* ptr1) { return prefix##_##loadsfx##_halves(ptr0, ptr1); } \
|
||||
inline void vx_store(typ* ptr, const vtyp& v) { return v_store(ptr, v); } \
|
||||
inline void vx_store_aligned(typ* ptr, const vtyp& v) { return v_store_aligned(ptr, v); }
|
||||
inline void vx_store_aligned(typ* ptr, const vtyp& v) { return v_store_aligned(ptr, v); } \
|
||||
inline vtyp vx_lut(const typ* ptr, const int* idx) { return prefix##_lut(ptr, idx); } \
|
||||
inline vtyp vx_lut_pairs(const typ* ptr, const int* idx) { return prefix##_lut_pairs(ptr, idx); }
|
||||
|
||||
#define CV_INTRIN_DEFINE_WIDE_LUT_QUAD(typ, vtyp, prefix) \
|
||||
inline vtyp vx_lut_quads(const typ* ptr, const int* idx) { return prefix##_lut_quads(ptr, idx); }
|
||||
|
||||
#define CV_INTRIN_DEFINE_WIDE_LOAD_EXPAND(typ, wtyp, prefix) \
|
||||
inline wtyp vx_load_expand(const typ* ptr) { return prefix##_load_expand(ptr); }
|
||||
@@ -244,6 +249,7 @@ CV_CPU_OPTIMIZATION_HAL_NAMESPACE_BEGIN
|
||||
|
||||
#define CV_INTRIN_DEFINE_WIDE_INTRIN_WITH_EXPAND(typ, vtyp, short_typ, wtyp, qtyp, prefix, loadsfx) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(typ, vtyp, short_typ, prefix, loadsfx) \
|
||||
CV_INTRIN_DEFINE_WIDE_LUT_QUAD(typ, vtyp, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_LOAD_EXPAND(typ, wtyp, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_LOAD_EXPAND_Q(typ, qtyp, prefix)
|
||||
|
||||
@@ -251,14 +257,19 @@ CV_CPU_OPTIMIZATION_HAL_NAMESPACE_BEGIN
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN_WITH_EXPAND(uchar, v_uint8, u8, v_uint16, v_uint32, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN_WITH_EXPAND(schar, v_int8, s8, v_int16, v_int32, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(ushort, v_uint16, u16, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_LUT_QUAD(ushort, v_uint16, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_LOAD_EXPAND(ushort, v_uint32, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(short, v_int16, s16, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_LUT_QUAD(short, v_int16, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_LOAD_EXPAND(short, v_int32, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(int, v_int32, s32, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_LUT_QUAD(int, v_int32, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_LOAD_EXPAND(int, v_int64, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(unsigned, v_uint32, u32, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_LUT_QUAD(unsigned, v_uint32, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_LOAD_EXPAND(unsigned, v_uint64, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(float, v_float32, f32, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_LUT_QUAD(float, v_float32, prefix) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(int64, v_int64, s64, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(uint64, v_uint64, u64, prefix, load) \
|
||||
CV_INTRIN_DEFINE_WIDE_LOAD_EXPAND(float16_t, v_float32, prefix)
|
||||
@@ -335,11 +346,11 @@ namespace CV__SIMD_NAMESPACE {
|
||||
typedef v_uint64x4 v_uint64;
|
||||
typedef v_int64x4 v_int64;
|
||||
typedef v_float32x8 v_float32;
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN_ALL_TYPES(v256)
|
||||
#if CV_SIMD256_64F
|
||||
typedef v_float64x4 v_float64;
|
||||
#endif
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN_ALL_TYPES(v256)
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(double, v_float64, f64, v256, load)
|
||||
#endif
|
||||
inline void vx_cleanup() { v256_cleanup(); }
|
||||
} // namespace
|
||||
using namespace CV__SIMD_NAMESPACE;
|
||||
@@ -358,11 +369,9 @@ namespace CV__SIMD_NAMESPACE {
|
||||
typedef v_uint64x2 v_uint64;
|
||||
typedef v_int64x2 v_int64;
|
||||
typedef v_float32x4 v_float32;
|
||||
#if CV_SIMD128_64F
|
||||
typedef v_float64x2 v_float64;
|
||||
#endif
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN_ALL_TYPES(v)
|
||||
#if CV_SIMD128_64F
|
||||
typedef v_float64x2 v_float64;
|
||||
CV_INTRIN_DEFINE_WIDE_INTRIN(double, v_float64, f64, v, load)
|
||||
#endif
|
||||
inline void vx_cleanup() { v_cleanup(); }
|
||||
|
||||
@@ -1417,6 +1417,97 @@ inline v_float64x4 v_cvt_f64_high(const v_float32x8& a)
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
inline v_int8x32 v256_lut(const schar* tab, const int* idx)
|
||||
{
|
||||
return v_int8x32(_mm256_setr_epi8(tab[idx[ 0]], tab[idx[ 1]], tab[idx[ 2]], tab[idx[ 3]], tab[idx[ 4]], tab[idx[ 5]], tab[idx[ 6]], tab[idx[ 7]],
|
||||
tab[idx[ 8]], tab[idx[ 9]], tab[idx[10]], tab[idx[11]], tab[idx[12]], tab[idx[13]], tab[idx[14]], tab[idx[15]],
|
||||
tab[idx[16]], tab[idx[17]], tab[idx[18]], tab[idx[19]], tab[idx[20]], tab[idx[21]], tab[idx[22]], tab[idx[23]],
|
||||
tab[idx[24]], tab[idx[25]], tab[idx[26]], tab[idx[27]], tab[idx[28]], tab[idx[29]], tab[idx[30]], tab[idx[31]]));
|
||||
}
|
||||
inline v_int8x32 v256_lut_pairs(const schar* tab, const int* idx)
|
||||
{
|
||||
return v_int8x32(_mm256_setr_epi16(*(const short*)(tab + idx[ 0]), *(const short*)(tab + idx[ 1]), *(const short*)(tab + idx[ 2]), *(const short*)(tab + idx[ 3]),
|
||||
*(const short*)(tab + idx[ 4]), *(const short*)(tab + idx[ 5]), *(const short*)(tab + idx[ 6]), *(const short*)(tab + idx[ 7]),
|
||||
*(const short*)(tab + idx[ 8]), *(const short*)(tab + idx[ 9]), *(const short*)(tab + idx[10]), *(const short*)(tab + idx[11]),
|
||||
*(const short*)(tab + idx[12]), *(const short*)(tab + idx[13]), *(const short*)(tab + idx[14]), *(const short*)(tab + idx[15])));
|
||||
}
|
||||
inline v_int8x32 v256_lut_quads(const schar* tab, const int* idx)
|
||||
{
|
||||
return v_int8x32(_mm256_i32gather_epi32((const int*)tab, _mm256_loadu_si256((const __m256i*)idx), 1));
|
||||
}
|
||||
inline v_uint8x32 v256_lut(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v256_lut((const schar *)tab, idx)); }
|
||||
inline v_uint8x32 v256_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v256_lut_pairs((const schar *)tab, idx)); }
|
||||
inline v_uint8x32 v256_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v256_lut_quads((const schar *)tab, idx)); }
|
||||
|
||||
inline v_int16x16 v256_lut(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x16(_mm256_setr_epi16(tab[idx[0]], tab[idx[1]], tab[idx[ 2]], tab[idx[ 3]], tab[idx[ 4]], tab[idx[ 5]], tab[idx[ 6]], tab[idx[ 7]],
|
||||
tab[idx[8]], tab[idx[9]], tab[idx[10]], tab[idx[11]], tab[idx[12]], tab[idx[13]], tab[idx[14]], tab[idx[15]]));
|
||||
}
|
||||
inline v_int16x16 v256_lut_pairs(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x16(_mm256_i32gather_epi32((const int*)tab, _mm256_loadu_si256((const __m256i*)idx), 2));
|
||||
}
|
||||
inline v_int16x16 v256_lut_quads(const short* tab, const int* idx)
|
||||
{
|
||||
#if defined(__GNUC__)
|
||||
return v_int16x16(_mm256_i32gather_epi64((const long long int*)tab, _mm_loadu_si128((const __m128i*)idx), 2));//Looks like intrinsic has wrong definition
|
||||
#else
|
||||
return v_int16x16(_mm256_i32gather_epi64((const int64*)tab, _mm_loadu_si128((const __m128i*)idx), 2));
|
||||
#endif
|
||||
}
|
||||
inline v_uint16x16 v256_lut(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v256_lut((const short *)tab, idx)); }
|
||||
inline v_uint16x16 v256_lut_pairs(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v256_lut_pairs((const short *)tab, idx)); }
|
||||
inline v_uint16x16 v256_lut_quads(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v256_lut_quads((const short *)tab, idx)); }
|
||||
|
||||
inline v_int32x8 v256_lut(const int* tab, const int* idx)
|
||||
{
|
||||
return v_int32x8(_mm256_i32gather_epi32(tab, _mm256_loadu_si256((const __m256i*)idx), 4));
|
||||
}
|
||||
inline v_int32x8 v256_lut_pairs(const int* tab, const int* idx)
|
||||
{
|
||||
#if defined(__GNUC__)
|
||||
return v_int32x8(_mm256_i32gather_epi64((const long long int*)tab, _mm_loadu_si128((const __m128i*)idx), 4));
|
||||
#else
|
||||
return v_int32x8(_mm256_i32gather_epi64((const int64*)tab, _mm_loadu_si128((const __m128i*)idx), 4));
|
||||
#endif
|
||||
}
|
||||
inline v_int32x8 v256_lut_quads(const int* tab, const int* idx)
|
||||
{
|
||||
return v_int32x8(_mm256_insertf128_si256(_mm256_castsi128_si256(_mm_loadu_si128((const __m128i*)(tab + idx[0]))), _mm_loadu_si128((const __m128i*)(tab + idx[1])), 0x1));
|
||||
}
|
||||
inline v_uint32x8 v256_lut(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v256_lut((const int *)tab, idx)); }
|
||||
inline v_uint32x8 v256_lut_pairs(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v256_lut_pairs((const int *)tab, idx)); }
|
||||
inline v_uint32x8 v256_lut_quads(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v256_lut_quads((const int *)tab, idx)); }
|
||||
|
||||
inline v_int64x4 v256_lut(const int64* tab, const int* idx)
|
||||
{
|
||||
#if defined(__GNUC__)
|
||||
return v_int64x4(_mm256_i32gather_epi64((const long long int*)tab, _mm_loadu_si128((const __m128i*)idx), 8));
|
||||
#else
|
||||
return v_int64x4(_mm256_i32gather_epi64(tab, _mm_loadu_si128((const __m128i*)idx), 8));
|
||||
#endif
|
||||
}
|
||||
inline v_int64x4 v256_lut_pairs(const int64* tab, const int* idx)
|
||||
{
|
||||
return v_int64x4(_mm256_insertf128_si256(_mm256_castsi128_si256(_mm_loadu_si128((const __m128i*)(tab + idx[0]))), _mm_loadu_si128((const __m128i*)(tab + idx[1])), 0x1));
|
||||
}
|
||||
inline v_uint64x4 v256_lut(const uint64* tab, const int* idx) { return v_reinterpret_as_u64(v256_lut((const int64 *)tab, idx)); }
|
||||
inline v_uint64x4 v256_lut_pairs(const uint64* tab, const int* idx) { return v_reinterpret_as_u64(v256_lut_pairs((const int64 *)tab, idx)); }
|
||||
|
||||
inline v_float32x8 v256_lut(const float* tab, const int* idx)
|
||||
{
|
||||
return v_float32x8(_mm256_i32gather_ps(tab, _mm256_loadu_si256((const __m256i*)idx), 4));
|
||||
}
|
||||
inline v_float32x8 v256_lut_pairs(const float* tab, const int* idx) { return v_reinterpret_as_f32(v256_lut_pairs((const int *)tab, idx)); }
|
||||
inline v_float32x8 v256_lut_quads(const float* tab, const int* idx) { return v_reinterpret_as_f32(v256_lut_quads((const int *)tab, idx)); }
|
||||
|
||||
inline v_float64x4 v256_lut(const double* tab, const int* idx)
|
||||
{
|
||||
return v_float64x4(_mm256_i32gather_pd(tab, _mm_loadu_si128((const __m128i*)idx), 8));
|
||||
}
|
||||
inline v_float64x4 v256_lut_pairs(const double* tab, const int* idx) { return v_float64x4(_mm256_insertf128_pd(_mm256_castpd128_pd256(_mm_loadu_pd(tab + idx[0])), _mm_loadu_pd(tab + idx[1]), 0x1)); }
|
||||
|
||||
inline v_int32x8 v_lut(const int* tab, const v_int32x8& idxvec)
|
||||
{
|
||||
return v_int32x8(_mm256_i32gather_epi32(tab, idxvec.val, 4));
|
||||
@@ -1476,6 +1567,49 @@ inline void v_lut_deinterleave(const double* tab, const v_int32x8& idxvec, v_flo
|
||||
y = v_float64x4(_mm256_unpackhi_pd(xy02, xy13));
|
||||
}
|
||||
|
||||
inline v_int8x32 v_interleave_pairs(const v_int8x32& vec)
|
||||
{
|
||||
return v_int8x32(_mm256_shuffle_epi8(vec.val, _mm256_set_epi64x(0x0f0d0e0c0b090a08, 0x0705060403010200, 0x0f0d0e0c0b090a08, 0x0705060403010200)));
|
||||
}
|
||||
inline v_uint8x32 v_interleave_pairs(const v_uint8x32& vec) { return v_reinterpret_as_u8(v_interleave_pairs(v_reinterpret_as_s8(vec))); }
|
||||
inline v_int8x32 v_interleave_quads(const v_int8x32& vec)
|
||||
{
|
||||
return v_int8x32(_mm256_shuffle_epi8(vec.val, _mm256_set_epi64x(0x0f0b0e0a0d090c08, 0x0703060205010400, 0x0f0b0e0a0d090c08, 0x0703060205010400)));
|
||||
}
|
||||
inline v_uint8x32 v_interleave_quads(const v_uint8x32& vec) { return v_reinterpret_as_u8(v_interleave_quads(v_reinterpret_as_s8(vec))); }
|
||||
|
||||
inline v_int16x16 v_interleave_pairs(const v_int16x16& vec)
|
||||
{
|
||||
return v_int16x16(_mm256_shuffle_epi8(vec.val, _mm256_set_epi64x(0x0f0e0b0a0d0c0908, 0x0706030205040100, 0x0f0e0b0a0d0c0908, 0x0706030205040100)));
|
||||
}
|
||||
inline v_uint16x16 v_interleave_pairs(const v_uint16x16& vec) { return v_reinterpret_as_u16(v_interleave_pairs(v_reinterpret_as_s16(vec))); }
|
||||
inline v_int16x16 v_interleave_quads(const v_int16x16& vec)
|
||||
{
|
||||
return v_int16x16(_mm256_shuffle_epi8(vec.val, _mm256_set_epi64x(0x0f0e07060d0c0504, 0x0b0a030209080100, 0x0f0e07060d0c0504, 0x0b0a030209080100)));
|
||||
}
|
||||
inline v_uint16x16 v_interleave_quads(const v_uint16x16& vec) { return v_reinterpret_as_u16(v_interleave_quads(v_reinterpret_as_s16(vec))); }
|
||||
|
||||
inline v_int32x8 v_interleave_pairs(const v_int32x8& vec)
|
||||
{
|
||||
return v_int32x8(_mm256_shuffle_epi32(vec.val, _MM_SHUFFLE(3, 1, 2, 0)));
|
||||
}
|
||||
inline v_uint32x8 v_interleave_pairs(const v_uint32x8& vec) { return v_reinterpret_as_u32(v_interleave_pairs(v_reinterpret_as_s32(vec))); }
|
||||
inline v_float32x8 v_interleave_pairs(const v_float32x8& vec) { return v_reinterpret_as_f32(v_interleave_pairs(v_reinterpret_as_s32(vec))); }
|
||||
|
||||
inline v_int8x32 v_pack_triplets(const v_int8x32& vec)
|
||||
{
|
||||
return v_int8x32(_mm256_permutevar8x32_epi32(_mm256_shuffle_epi8(vec.val, _mm256_broadcastsi128_si256(_mm_set_epi64x(0xffffff0f0e0d0c0a, 0x0908060504020100))),
|
||||
_mm256_set_epi64x(0x0000000700000007, 0x0000000600000005, 0x0000000400000002, 0x0000000100000000)));
|
||||
}
|
||||
inline v_uint8x32 v_pack_triplets(const v_uint8x32& vec) { return v_reinterpret_as_u8(v_pack_triplets(v_reinterpret_as_s8(vec))); }
|
||||
|
||||
inline v_int16x16 v_pack_triplets(const v_int16x16& vec)
|
||||
{
|
||||
return v_int16x16(_mm256_permutevar8x32_epi32(_mm256_shuffle_epi8(vec.val, _mm256_broadcastsi128_si256(_mm_set_epi64x(0xffff0f0e0d0c0b0a, 0x0908050403020100))),
|
||||
_mm256_set_epi64x(0x0000000700000007, 0x0000000600000005, 0x0000000400000002, 0x0000000100000000)));
|
||||
}
|
||||
inline v_uint16x16 v_pack_triplets(const v_uint16x16& vec) { return v_reinterpret_as_u16(v_pack_triplets(v_reinterpret_as_s16(vec))); }
|
||||
|
||||
////////// Matrix operations /////////
|
||||
|
||||
inline v_int32x8 v_dotprod(const v_int16x16& a, const v_int16x16& b)
|
||||
|
||||
@@ -1799,6 +1799,28 @@ template<int n> inline v_reg<double, n> v_cvt_f64(const v_reg<float, n*2>& a)
|
||||
return c;
|
||||
}
|
||||
|
||||
template<typename _Tp> inline v_reg<_Tp, V_TypeTraits<_Tp>::nlanes128> v_lut(const _Tp* tab, const int* idx)
|
||||
{
|
||||
v_reg<_Tp, V_TypeTraits<_Tp>::nlanes128> c;
|
||||
for (int i = 0; i < V_TypeTraits<_Tp>::nlanes128; i++)
|
||||
c.s[i] = tab[idx[i]];
|
||||
return c;
|
||||
}
|
||||
template<typename _Tp> inline v_reg<_Tp, V_TypeTraits<_Tp>::nlanes128> v_lut_pairs(const _Tp* tab, const int* idx)
|
||||
{
|
||||
v_reg<_Tp, V_TypeTraits<_Tp>::nlanes128> c;
|
||||
for (int i = 0; i < V_TypeTraits<_Tp>::nlanes128; i++)
|
||||
c.s[i] = tab[idx[i / 2] + i % 2];
|
||||
return c;
|
||||
}
|
||||
template<typename _Tp> inline v_reg<_Tp, V_TypeTraits<_Tp>::nlanes128> v_lut_quads(const _Tp* tab, const int* idx)
|
||||
{
|
||||
v_reg<_Tp, V_TypeTraits<_Tp>::nlanes128> c;
|
||||
for (int i = 0; i < V_TypeTraits<_Tp>::nlanes128; i++)
|
||||
c.s[i] = tab[idx[i / 4] + i % 4];
|
||||
return c;
|
||||
}
|
||||
|
||||
template<int n> inline v_reg<int, n> v_lut(const int* tab, const v_reg<int, n>& idx)
|
||||
{
|
||||
v_reg<int, n> c;
|
||||
@@ -1807,6 +1829,14 @@ template<int n> inline v_reg<int, n> v_lut(const int* tab, const v_reg<int, n>&
|
||||
return c;
|
||||
}
|
||||
|
||||
template<int n> inline v_reg<unsigned, n> v_lut(const unsigned* tab, const v_reg<int, n>& idx)
|
||||
{
|
||||
v_reg<int, n> c;
|
||||
for (int i = 0; i < n; i++)
|
||||
c.s[i] = tab[idx.s[i]];
|
||||
return c;
|
||||
}
|
||||
|
||||
template<int n> inline v_reg<float, n> v_lut(const float* tab, const v_reg<int, n>& idx)
|
||||
{
|
||||
v_reg<float, n> c;
|
||||
@@ -1845,6 +1875,49 @@ template<int n> inline void v_lut_deinterleave(const double* tab, const v_reg<in
|
||||
}
|
||||
}
|
||||
|
||||
template<typename _Tp, int n> inline v_reg<_Tp, n> v_interleave_pairs(const v_reg<_Tp, n>& vec)
|
||||
{
|
||||
v_reg<float, n> c;
|
||||
for (int i = 0; i < n/4; i++)
|
||||
{
|
||||
c.s[4*i ] = vec.s[4*i ];
|
||||
c.s[4*i+1] = vec.s[4*i+2];
|
||||
c.s[4*i+2] = vec.s[4*i+1];
|
||||
c.s[4*i+3] = vec.s[4*i+3];
|
||||
}
|
||||
return c;
|
||||
}
|
||||
|
||||
template<typename _Tp, int n> inline v_reg<_Tp, n> v_interleave_quads(const v_reg<_Tp, n>& vec)
|
||||
{
|
||||
v_reg<float, n> c;
|
||||
for (int i = 0; i < n/8; i++)
|
||||
{
|
||||
c.s[8*i ] = vec.s[8*i ];
|
||||
c.s[8*i+1] = vec.s[8*i+4];
|
||||
c.s[8*i+2] = vec.s[8*i+1];
|
||||
c.s[8*i+3] = vec.s[8*i+5];
|
||||
c.s[8*i+4] = vec.s[8*i+2];
|
||||
c.s[8*i+5] = vec.s[8*i+6];
|
||||
c.s[8*i+6] = vec.s[8*i+3];
|
||||
c.s[8*i+7] = vec.s[8*i+7];
|
||||
}
|
||||
return c;
|
||||
}
|
||||
|
||||
template<typename _Tp, int n> inline v_reg<_Tp, n> v_pack_triplets(const v_reg<_Tp, n>& vec)
|
||||
{
|
||||
v_reg<float, n> c;
|
||||
int j = 0;
|
||||
for (int i = 0; i < n/4; i++)
|
||||
{
|
||||
c.s[3*i ] = vec.s[4*i ];
|
||||
c.s[3*i+1] = vec.s[4*i+1];
|
||||
c.s[3*i+2] = vec.s[4*i+2];
|
||||
}
|
||||
return c;
|
||||
}
|
||||
|
||||
/** @brief Transpose 4x4 matrix
|
||||
|
||||
Scheme:
|
||||
|
||||
@@ -1572,6 +1572,162 @@ inline v_float64x2 v_cvt_f64_high(const v_float32x4& a)
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
inline v_int8x16 v_lut(const schar* tab, const int* idx)
|
||||
{
|
||||
schar CV_DECL_ALIGNED(32) elems[16] =
|
||||
{
|
||||
tab[idx[ 0]],
|
||||
tab[idx[ 1]],
|
||||
tab[idx[ 2]],
|
||||
tab[idx[ 3]],
|
||||
tab[idx[ 4]],
|
||||
tab[idx[ 5]],
|
||||
tab[idx[ 6]],
|
||||
tab[idx[ 7]],
|
||||
tab[idx[ 8]],
|
||||
tab[idx[ 9]],
|
||||
tab[idx[10]],
|
||||
tab[idx[11]],
|
||||
tab[idx[12]],
|
||||
tab[idx[13]],
|
||||
tab[idx[14]],
|
||||
tab[idx[15]]
|
||||
};
|
||||
return v_int8x16(vld1q_s8(elems));
|
||||
}
|
||||
inline v_int8x16 v_lut_pairs(const schar* tab, const int* idx)
|
||||
{
|
||||
short CV_DECL_ALIGNED(32) elems[8] =
|
||||
{
|
||||
*(short*)(tab+idx[0]),
|
||||
*(short*)(tab+idx[1]),
|
||||
*(short*)(tab+idx[2]),
|
||||
*(short*)(tab+idx[3]),
|
||||
*(short*)(tab+idx[4]),
|
||||
*(short*)(tab+idx[5]),
|
||||
*(short*)(tab+idx[6]),
|
||||
*(short*)(tab+idx[7])
|
||||
};
|
||||
return v_int8x16(vreinterpretq_s8_s16(vld1q_s16(elems)));
|
||||
}
|
||||
inline v_int8x16 v_lut_quads(const schar* tab, const int* idx)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) elems[4] =
|
||||
{
|
||||
*(int*)(tab + idx[0]),
|
||||
*(int*)(tab + idx[1]),
|
||||
*(int*)(tab + idx[2]),
|
||||
*(int*)(tab + idx[3])
|
||||
};
|
||||
return v_int8x16(vreinterpretq_s8_s32(vld1q_s32(elems)));
|
||||
}
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut((schar*)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((schar*)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((schar*)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
short CV_DECL_ALIGNED(32) elems[8] =
|
||||
{
|
||||
tab[idx[0]],
|
||||
tab[idx[1]],
|
||||
tab[idx[2]],
|
||||
tab[idx[3]],
|
||||
tab[idx[4]],
|
||||
tab[idx[5]],
|
||||
tab[idx[6]],
|
||||
tab[idx[7]]
|
||||
};
|
||||
return v_int16x8(vld1q_s16(elems));
|
||||
}
|
||||
inline v_int16x8 v_lut_pairs(const short* tab, const int* idx)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) elems[4] =
|
||||
{
|
||||
*(int*)(tab + idx[0]),
|
||||
*(int*)(tab + idx[1]),
|
||||
*(int*)(tab + idx[2]),
|
||||
*(int*)(tab + idx[3])
|
||||
};
|
||||
return v_int16x8(vreinterpretq_s16_s32(vld1q_s32(elems)));
|
||||
}
|
||||
inline v_int16x8 v_lut_quads(const short* tab, const int* idx)
|
||||
{
|
||||
int64 CV_DECL_ALIGNED(32) elems[2] =
|
||||
{
|
||||
*(int64*)(tab + idx[0]),
|
||||
*(int64*)(tab + idx[1])
|
||||
};
|
||||
return v_int16x8(vreinterpretq_s16_s64(vld1q_s64(elems)));
|
||||
}
|
||||
inline v_uint16x8 v_lut(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut((short*)tab, idx)); }
|
||||
inline v_uint16x8 v_lut_pairs(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_pairs((short*)tab, idx)); }
|
||||
inline v_uint16x8 v_lut_quads(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_quads((short*)tab, idx)); }
|
||||
|
||||
inline v_int32x4 v_lut(const int* tab, const int* idx)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) elems[4] =
|
||||
{
|
||||
tab[idx[0]],
|
||||
tab[idx[1]],
|
||||
tab[idx[2]],
|
||||
tab[idx[3]]
|
||||
};
|
||||
return v_int32x4(vld1q_s32(elems));
|
||||
}
|
||||
inline v_int32x4 v_lut_pairs(const int* tab, const int* idx)
|
||||
{
|
||||
int64 CV_DECL_ALIGNED(32) elems[2] =
|
||||
{
|
||||
*(int64*)(tab + idx[0]),
|
||||
*(int64*)(tab + idx[1])
|
||||
};
|
||||
return v_int32x4(vreinterpretq_s32_s64(vld1q_s64(elems)));
|
||||
}
|
||||
inline v_int32x4 v_lut_quads(const int* tab, const int* idx)
|
||||
{
|
||||
return v_int32x4(vld1q_s32(tab + idx[0]));
|
||||
}
|
||||
inline v_uint32x4 v_lut(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut((int*)tab, idx)); }
|
||||
inline v_uint32x4 v_lut_pairs(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut_pairs((int*)tab, idx)); }
|
||||
inline v_uint32x4 v_lut_quads(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut_quads((int*)tab, idx)); }
|
||||
|
||||
inline v_int64x2 v_lut(const int64_t* tab, const int* idx)
|
||||
{
|
||||
return v_int64x2(vcombine_s64(vcreate_s64(tab[idx[0]]), vcreate_s64(tab[idx[1]])));
|
||||
}
|
||||
inline v_int64x2 v_lut_pairs(const int64_t* tab, const int* idx)
|
||||
{
|
||||
return v_int64x2(vld1q_s64(tab + idx[0]));
|
||||
}
|
||||
inline v_uint64x2 v_lut(const uint64_t* tab, const int* idx) { return v_reinterpret_as_u64(v_lut((const int64_t *)tab, idx)); }
|
||||
inline v_uint64x2 v_lut_pairs(const uint64_t* tab, const int* idx) { return v_reinterpret_as_u64(v_lut_pairs((const int64_t *)tab, idx)); }
|
||||
|
||||
inline v_float32x4 v_lut(const float* tab, const int* idx)
|
||||
{
|
||||
float CV_DECL_ALIGNED(32) elems[4] =
|
||||
{
|
||||
tab[idx[0]],
|
||||
tab[idx[1]],
|
||||
tab[idx[2]],
|
||||
tab[idx[3]]
|
||||
};
|
||||
return v_float32x4(vld1q_f32(elems));
|
||||
}
|
||||
inline v_float32x4 v_lut_pairs(const float* tab, const int* idx)
|
||||
{
|
||||
uint64 CV_DECL_ALIGNED(32) elems[2] =
|
||||
{
|
||||
*(uint64*)(tab + idx[0]),
|
||||
*(uint64*)(tab + idx[1])
|
||||
};
|
||||
return v_float32x4(vreinterpretq_f32_u64(vld1q_u64(elems)));
|
||||
}
|
||||
inline v_float32x4 v_lut_quads(const float* tab, const int* idx)
|
||||
{
|
||||
return v_float32x4(vld1q_f32(tab + idx[0]));
|
||||
}
|
||||
|
||||
inline v_int32x4 v_lut(const int* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) elems[4] =
|
||||
@@ -1584,6 +1740,18 @@ inline v_int32x4 v_lut(const int* tab, const v_int32x4& idxvec)
|
||||
return v_int32x4(vld1q_s32(elems));
|
||||
}
|
||||
|
||||
inline v_uint32x4 v_lut(const unsigned* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
unsigned CV_DECL_ALIGNED(32) elems[4] =
|
||||
{
|
||||
tab[vgetq_lane_s32(idxvec.val, 0)],
|
||||
tab[vgetq_lane_s32(idxvec.val, 1)],
|
||||
tab[vgetq_lane_s32(idxvec.val, 2)],
|
||||
tab[vgetq_lane_s32(idxvec.val, 3)]
|
||||
};
|
||||
return v_uint32x4(vld1q_u32(elems));
|
||||
}
|
||||
|
||||
inline v_float32x4 v_lut(const float* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
float CV_DECL_ALIGNED(32) elems[4] =
|
||||
@@ -1614,7 +1782,64 @@ inline void v_lut_deinterleave(const float* tab, const v_int32x4& idxvec, v_floa
|
||||
y = v_float32x4(tab[idx[0]+1], tab[idx[1]+1], tab[idx[2]+1], tab[idx[3]+1]);
|
||||
}
|
||||
|
||||
inline v_int8x16 v_interleave_pairs(const v_int8x16& vec)
|
||||
{
|
||||
return v_int8x16(vcombine_s8(vtbl1_s8(vget_low_s8(vec.val), vcreate_s8(0x0705060403010200)), vtbl1_s8(vget_high_s8(vec.val), vcreate_s8(0x0705060403010200))));
|
||||
}
|
||||
inline v_uint8x16 v_interleave_pairs(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_interleave_pairs(v_reinterpret_as_s8(vec))); }
|
||||
inline v_int8x16 v_interleave_quads(const v_int8x16& vec)
|
||||
{
|
||||
return v_int8x16(vcombine_s8(vtbl1_s8(vget_low_s8(vec.val), vcreate_s8(0x0703060205010400)), vtbl1_s8(vget_high_s8(vec.val), vcreate_s8(0x0703060205010400))));
|
||||
}
|
||||
inline v_uint8x16 v_interleave_quads(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_interleave_quads(v_reinterpret_as_s8(vec))); }
|
||||
|
||||
inline v_int16x8 v_interleave_pairs(const v_int16x8& vec)
|
||||
{
|
||||
return v_int16x8(vreinterpretq_s16_s8(vcombine_s8(vtbl1_s8(vget_low_s8(vreinterpretq_s8_s16(vec.val)), vcreate_s8(0x0706030205040100)), vtbl1_s8(vget_high_s8(vreinterpretq_s8_s16(vec.val)), vcreate_s8(0x0706030205040100)))));
|
||||
}
|
||||
inline v_uint16x8 v_interleave_pairs(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_interleave_pairs(v_reinterpret_as_s16(vec))); }
|
||||
inline v_int16x8 v_interleave_quads(const v_int16x8& vec)
|
||||
{
|
||||
return v_int16x8(vreinterpretq_s16_s8(vcombine_s8(vtbl1_s8(vget_low_s8(vreinterpretq_s8_s16(vec.val)), vcreate_s8(0x0b0a030209080100)), vtbl1_s8(vget_high_s8(vreinterpretq_s8_s16(vec.val)), vcreate_s8(0x0b0a030209080100)))));
|
||||
}
|
||||
inline v_uint16x8 v_interleave_quads(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_interleave_quads(v_reinterpret_as_s16(vec))); }
|
||||
|
||||
inline v_int32x4 v_interleave_pairs(const v_int32x4& vec)
|
||||
{
|
||||
int32x2x2_t res = vzip_s32(vget_low_s32(vec.val), vget_high_s32(vec.val));
|
||||
return v_int32x4(vcombine_s32(res.val[0], res.val[1]));
|
||||
}
|
||||
inline v_uint32x4 v_interleave_pairs(const v_uint32x4& vec) { return v_reinterpret_as_u32(v_interleave_pairs(v_reinterpret_as_s32(vec))); }
|
||||
inline v_float32x4 v_interleave_pairs(const v_float32x4& vec) { return v_reinterpret_as_f32(v_interleave_pairs(v_reinterpret_as_s32(vec))); }
|
||||
|
||||
inline v_int8x16 v_pack_triplets(const v_int8x16& vec)
|
||||
{
|
||||
return v_int8x16(vextq_s8(vcombine_s8(vtbl1_s8(vget_low_s8(vec.val), vcreate_s8(0x0605040201000000)), vtbl1_s8(vget_high_s8(vec.val), vcreate_s8(0x0807060504020100))), vdupq_n_s8(0), 2));
|
||||
}
|
||||
inline v_uint8x16 v_pack_triplets(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_pack_triplets(v_reinterpret_as_s8(vec))); }
|
||||
|
||||
inline v_int16x8 v_pack_triplets(const v_int16x8& vec)
|
||||
{
|
||||
return v_int16x8(vreinterpretq_s16_s8(vextq_s8(vcombine_s8(vtbl1_s8(vget_low_s8(vreinterpretq_s8_s16(vec.val)), vcreate_s8(0x0504030201000000)), vget_high_s8(vreinterpretq_s8_s16(vec.val))), vdupq_n_s8(0), 2)));
|
||||
}
|
||||
inline v_uint16x8 v_pack_triplets(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_pack_triplets(v_reinterpret_as_s16(vec))); }
|
||||
|
||||
#if CV_SIMD128_64F
|
||||
inline v_float64x2 v_lut(const double* tab, const int* idx)
|
||||
{
|
||||
double CV_DECL_ALIGNED(32) elems[2] =
|
||||
{
|
||||
tab[idx[0]],
|
||||
tab[idx[1]]
|
||||
};
|
||||
return v_float64x2(vld1q_f64(elems));
|
||||
}
|
||||
|
||||
inline v_float64x2 v_lut_pairs(const double* tab, const int* idx)
|
||||
{
|
||||
return v_float64x2(vld1q_f64(tab + idx[0]));
|
||||
}
|
||||
|
||||
inline v_float64x2 v_lut(const double* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
double CV_DECL_ALIGNED(32) elems[2] =
|
||||
|
||||
@@ -2699,6 +2699,126 @@ inline void v_store_fp16(short* ptr, const v_float32x4& a)
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
inline v_int8x16 v_lut(const schar* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int8x16(_mm_setr_epi8(tab[idx[0]], tab[idx[1]], tab[idx[ 2]], tab[idx[ 3]], tab[idx[ 4]], tab[idx[ 5]], tab[idx[ 6]], tab[idx[ 7]],
|
||||
tab[idx[8]], tab[idx[9]], tab[idx[10]], tab[idx[11]], tab[idx[12]], tab[idx[13]], tab[idx[14]], tab[idx[15]]));
|
||||
#else
|
||||
return v_int8x16(_mm_setr_epi64(
|
||||
_mm_setr_pi8(tab[idx[0]], tab[idx[1]], tab[idx[ 2]], tab[idx[ 3]], tab[idx[ 4]], tab[idx[ 5]], tab[idx[ 6]], tab[idx[ 7]]),
|
||||
_mm_setr_pi8(tab[idx[8]], tab[idx[9]], tab[idx[10]], tab[idx[11]], tab[idx[12]], tab[idx[13]], tab[idx[14]], tab[idx[15]])
|
||||
));
|
||||
#endif
|
||||
}
|
||||
inline v_int8x16 v_lut_pairs(const schar* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int8x16(_mm_setr_epi16(*(const short*)(tab + idx[0]), *(const short*)(tab + idx[1]), *(const short*)(tab + idx[2]), *(const short*)(tab + idx[3]),
|
||||
*(const short*)(tab + idx[4]), *(const short*)(tab + idx[5]), *(const short*)(tab + idx[6]), *(const short*)(tab + idx[7])));
|
||||
#else
|
||||
return v_int8x16(_mm_setr_epi64(
|
||||
_mm_setr_pi16(*(const short*)(tab + idx[0]), *(const short*)(tab + idx[1]), *(const short*)(tab + idx[2]), *(const short*)(tab + idx[3])),
|
||||
_mm_setr_pi16(*(const short*)(tab + idx[4]), *(const short*)(tab + idx[5]), *(const short*)(tab + idx[6]), *(const short*)(tab + idx[7]))
|
||||
));
|
||||
#endif
|
||||
}
|
||||
inline v_int8x16 v_lut_quads(const schar* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int8x16(_mm_setr_epi32(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1]),
|
||||
*(const int*)(tab + idx[2]), *(const int*)(tab + idx[3])));
|
||||
#else
|
||||
return v_int8x16(_mm_setr_epi64(
|
||||
_mm_setr_pi32(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1])),
|
||||
_mm_setr_pi32(*(const int*)(tab + idx[2]), *(const int*)(tab + idx[3]))
|
||||
));
|
||||
#endif
|
||||
}
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut((const schar *)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((const schar *)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((const schar *)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int16x8(_mm_setr_epi16(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]],
|
||||
tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]]));
|
||||
#else
|
||||
return v_int16x8(_mm_setr_epi64(
|
||||
_mm_setr_pi16(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]),
|
||||
_mm_setr_pi16(tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]])
|
||||
));
|
||||
#endif
|
||||
}
|
||||
inline v_int16x8 v_lut_pairs(const short* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int16x8(_mm_setr_epi32(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1]),
|
||||
*(const int*)(tab + idx[2]), *(const int*)(tab + idx[3])));
|
||||
#else
|
||||
return v_int16x8(_mm_setr_epi64(
|
||||
_mm_setr_pi32(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1])),
|
||||
_mm_setr_pi32(*(const int*)(tab + idx[2]), *(const int*)(tab + idx[3]))
|
||||
));
|
||||
#endif
|
||||
}
|
||||
inline v_int16x8 v_lut_quads(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x8(_mm_set_epi64x(*(const int64_t*)(tab + idx[1]), *(const int64_t*)(tab + idx[0])));
|
||||
}
|
||||
inline v_uint16x8 v_lut(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut((const short *)tab, idx)); }
|
||||
inline v_uint16x8 v_lut_pairs(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_pairs((const short *)tab, idx)); }
|
||||
inline v_uint16x8 v_lut_quads(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_quads((const short *)tab, idx)); }
|
||||
|
||||
inline v_int32x4 v_lut(const int* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int32x4(_mm_setr_epi32(tab[idx[0]], tab[idx[1]],
|
||||
tab[idx[2]], tab[idx[3]]));
|
||||
#else
|
||||
return v_int32x4(_mm_setr_epi64(
|
||||
_mm_setr_pi32(tab[idx[0]], tab[idx[1]]),
|
||||
_mm_setr_pi32(tab[idx[2]], tab[idx[3]])
|
||||
));
|
||||
#endif
|
||||
}
|
||||
inline v_int32x4 v_lut_pairs(const int* tab, const int* idx)
|
||||
{
|
||||
return v_int32x4(_mm_set_epi64x(*(const int64_t*)(tab + idx[1]), *(const int64_t*)(tab + idx[0])));
|
||||
}
|
||||
inline v_int32x4 v_lut_quads(const int* tab, const int* idx)
|
||||
{
|
||||
return v_int32x4(_mm_load_si128((const __m128i*)(tab + idx[0])));
|
||||
}
|
||||
inline v_uint32x4 v_lut(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut((const int *)tab, idx)); }
|
||||
inline v_uint32x4 v_lut_pairs(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut_pairs((const int *)tab, idx)); }
|
||||
inline v_uint32x4 v_lut_quads(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut_quads((const int *)tab, idx)); }
|
||||
|
||||
inline v_int64x2 v_lut(const int64_t* tab, const int* idx)
|
||||
{
|
||||
return v_int64x2(_mm_set_epi64x(tab[idx[1]], tab[idx[0]]));
|
||||
}
|
||||
inline v_int64x2 v_lut_pairs(const int64_t* tab, const int* idx)
|
||||
{
|
||||
return v_int64x2(_mm_load_si128((const __m128i*)(tab + idx[0])));
|
||||
}
|
||||
inline v_uint64x2 v_lut(const uint64_t* tab, const int* idx) { return v_reinterpret_as_u64(v_lut((const int64_t *)tab, idx)); }
|
||||
inline v_uint64x2 v_lut_pairs(const uint64_t* tab, const int* idx) { return v_reinterpret_as_u64(v_lut_pairs((const int64_t *)tab, idx)); }
|
||||
|
||||
inline v_float32x4 v_lut(const float* tab, const int* idx)
|
||||
{
|
||||
return v_float32x4(_mm_setr_ps(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]));
|
||||
}
|
||||
inline v_float32x4 v_lut_pairs(const float* tab, const int* idx) { return v_reinterpret_as_f32(v_lut_pairs((const int *)tab, idx)); }
|
||||
inline v_float32x4 v_lut_quads(const float* tab, const int* idx) { return v_reinterpret_as_f32(v_lut_quads((const int *)tab, idx)); }
|
||||
|
||||
inline v_float64x2 v_lut(const double* tab, const int* idx)
|
||||
{
|
||||
return v_float64x2(_mm_setr_pd(tab[idx[0]], tab[idx[1]]));
|
||||
}
|
||||
inline v_float64x2 v_lut_pairs(const double* tab, const int* idx) { return v_float64x2(_mm_castsi128_pd(_mm_load_si128((const __m128i*)(tab + idx[0])))); }
|
||||
|
||||
inline v_int32x4 v_lut(const int* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) idx[4];
|
||||
@@ -2706,6 +2826,11 @@ inline v_int32x4 v_lut(const int* tab, const v_int32x4& idxvec)
|
||||
return v_int32x4(_mm_setr_epi32(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]));
|
||||
}
|
||||
|
||||
inline v_uint32x4 v_lut(const unsigned* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
return v_reinterpret_as_u32(v_lut((const int *)tab, idxvec));
|
||||
}
|
||||
|
||||
inline v_float32x4 v_lut(const float* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) idx[4];
|
||||
@@ -2751,6 +2876,77 @@ inline void v_lut_deinterleave(const double* tab, const v_int32x4& idxvec, v_flo
|
||||
y = v_float64x2(_mm_unpackhi_pd(xy0, xy1));
|
||||
}
|
||||
|
||||
inline v_int8x16 v_interleave_pairs(const v_int8x16& vec)
|
||||
{
|
||||
#if CV_SSSE3
|
||||
return v_int8x16(_mm_shuffle_epi8(vec.val, _mm_set_epi64x(0x0f0d0e0c0b090a08, 0x0705060403010200)));
|
||||
#else
|
||||
__m128i a = _mm_shufflelo_epi16(vec.val, _MM_SHUFFLE(3, 1, 2, 0));
|
||||
a = _mm_shufflehi_epi16(a, _MM_SHUFFLE(3, 1, 2, 0));
|
||||
a = _mm_shuffle_epi32(a, _MM_SHUFFLE(3, 1, 2, 0));
|
||||
return v_int8x16(_mm_unpacklo_epi8(a, _mm_unpackhi_epi64(a, a)));
|
||||
#endif
|
||||
}
|
||||
inline v_uint8x16 v_interleave_pairs(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_interleave_pairs(v_reinterpret_as_s8(vec))); }
|
||||
inline v_int8x16 v_interleave_quads(const v_int8x16& vec)
|
||||
{
|
||||
#if CV_SSSE3
|
||||
return v_int8x16(_mm_shuffle_epi8(vec.val, _mm_set_epi64x(0x0f0b0e0a0d090c08, 0x0703060205010400)));
|
||||
#else
|
||||
__m128i a = _mm_shuffle_epi32(vec.val, _MM_SHUFFLE(3, 1, 2, 0));
|
||||
return v_int8x16(_mm_unpacklo_epi8(a, _mm_unpackhi_epi64(a, a)));
|
||||
#endif
|
||||
}
|
||||
inline v_uint8x16 v_interleave_quads(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_interleave_quads(v_reinterpret_as_s8(vec))); }
|
||||
|
||||
inline v_int16x8 v_interleave_pairs(const v_int16x8& vec)
|
||||
{
|
||||
#if CV_SSSE3
|
||||
return v_int16x8(_mm_shuffle_epi8(vec.val, _mm_set_epi64x(0x0f0e0b0a0d0c0908, 0x0706030205040100)));
|
||||
#else
|
||||
__m128i a = _mm_shufflelo_epi16(vec.val, _MM_SHUFFLE(3, 1, 2, 0));
|
||||
return v_int16x8(_mm_shufflehi_epi16(a, _MM_SHUFFLE(3, 1, 2, 0)));
|
||||
#endif
|
||||
}
|
||||
inline v_uint16x8 v_interleave_pairs(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_interleave_pairs(v_reinterpret_as_s16(vec))); }
|
||||
inline v_int16x8 v_interleave_quads(const v_int16x8& vec)
|
||||
{
|
||||
#if CV_SSSE3
|
||||
return v_int16x8(_mm_shuffle_epi8(vec.val, _mm_set_epi64x(0x0f0e07060d0c0504, 0x0b0a030209080100)));
|
||||
#else
|
||||
return v_int16x8(_mm_unpacklo_epi16(vec.val, _mm_unpackhi_epi64(vec.val, vec.val)));
|
||||
#endif
|
||||
}
|
||||
inline v_uint16x8 v_interleave_quads(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_interleave_quads(v_reinterpret_as_s16(vec))); }
|
||||
|
||||
inline v_int32x4 v_interleave_pairs(const v_int32x4& vec)
|
||||
{
|
||||
return v_int32x4(_mm_shuffle_epi32(vec.val, _MM_SHUFFLE(3, 1, 2, 0)));
|
||||
}
|
||||
inline v_uint32x4 v_interleave_pairs(const v_uint32x4& vec) { return v_reinterpret_as_u32(v_interleave_pairs(v_reinterpret_as_s32(vec))); }
|
||||
inline v_float32x4 v_interleave_pairs(const v_float32x4& vec) { return v_reinterpret_as_f32(v_interleave_pairs(v_reinterpret_as_s32(vec))); }
|
||||
|
||||
inline v_int8x16 v_pack_triplets(const v_int8x16& vec)
|
||||
{
|
||||
#if CV_SSSE3
|
||||
return v_int8x16(_mm_shuffle_epi8(vec.val, _mm_set_epi64x(0xffffff0f0e0d0c0a, 0x0908060504020100)));
|
||||
#else
|
||||
__m128i mask = _mm_set1_epi64x(0x00000000FFFFFFFF);
|
||||
__m128i a = _mm_or_si128(_mm_andnot_si128(mask, vec.val), _mm_and_si128(mask, _mm_sll_epi32(vec.val, _mm_set_epi64x(0, 8))));
|
||||
return v_int8x16(_mm_srli_si128(_mm_shufflelo_epi16(a, _MM_SHUFFLE(2, 1, 0, 3)), 2));
|
||||
#endif
|
||||
}
|
||||
inline v_uint8x16 v_pack_triplets(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_pack_triplets(v_reinterpret_as_s8(vec))); }
|
||||
|
||||
inline v_int16x8 v_pack_triplets(const v_int16x8& vec)
|
||||
{
|
||||
#if CV_SSSE3
|
||||
return v_int16x8(_mm_shuffle_epi8(vec.val, _mm_set_epi64x(0xffff0f0e0d0c0b0a, 0x0908050403020100)));
|
||||
#else
|
||||
return v_int16x8(_mm_srli_si128(_mm_shufflelo_epi16(vec.val, _MM_SHUFFLE(2, 1, 0, 3)), 2));
|
||||
#endif
|
||||
}
|
||||
inline v_uint16x8 v_pack_triplets(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_pack_triplets(v_reinterpret_as_s16(vec))); }
|
||||
|
||||
////////////// FP16 support ///////////////////////////
|
||||
|
||||
|
||||
@@ -993,6 +993,80 @@ inline v_float64x2 v_cvt_f64_high(const v_float32x4& a)
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
inline v_int8x16 v_lut(const schar* tab, const int* idx)
|
||||
{
|
||||
return v_int8x16(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]], tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]],
|
||||
tab[idx[8]], tab[idx[9]], tab[idx[10]], tab[idx[11]], tab[idx[12]], tab[idx[13]], tab[idx[14]], tab[idx[15]]);
|
||||
}
|
||||
inline v_int8x16 v_lut_pairs(const schar* tab, const int* idx)
|
||||
{
|
||||
return v_reinterpret_as_s8(v_int16x8(*(const short*)(tab+idx[0]), *(const short*)(tab+idx[1]), *(const short*)(tab+idx[2]), *(const short*)(tab+idx[3]),
|
||||
*(const short*)(tab+idx[4]), *(const short*)(tab+idx[5]), *(const short*)(tab+idx[6]), *(const short*)(tab+idx[7])));
|
||||
}
|
||||
inline v_int8x16 v_lut_quads(const schar* tab, const int* idx)
|
||||
{
|
||||
return v_reinterpret_as_s8(v_int32x4(*(const int*)(tab+idx[0]), *(const int*)(tab+idx[1]), *(const int*)(tab+idx[2]), *(const int*)(tab+idx[3])));
|
||||
}
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut((const schar*)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((const schar*)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((const schar*)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x8(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]], tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]]);
|
||||
}
|
||||
inline v_int16x8 v_lut_pairs(const short* tab, const int* idx)
|
||||
{
|
||||
return v_reinterpret_as_s16(v_int32x4(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1]), *(const int*)(tab + idx[2]), *(const int*)(tab + idx[3])));
|
||||
}
|
||||
inline v_int16x8 v_lut_quads(const short* tab, const int* idx)
|
||||
{
|
||||
return v_reinterpret_as_s16(v_int64x2(*(const int64*)(tab + idx[0]), *(const int64*)(tab + idx[1])));
|
||||
}
|
||||
inline v_uint16x8 v_lut(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut((const short*)tab, idx)); }
|
||||
inline v_uint16x8 v_lut_pairs(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_pairs((const short*)tab, idx)); }
|
||||
inline v_uint16x8 v_lut_quads(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_quads((const short*)tab, idx)); }
|
||||
|
||||
inline v_int32x4 v_lut(const int* tab, const int* idx)
|
||||
{
|
||||
return v_int32x4(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]);
|
||||
}
|
||||
inline v_int32x4 v_lut_pairs(const int* tab, const int* idx)
|
||||
{
|
||||
return v_reinterpret_as_s32(v_int64x2(*(const int64*)(tab + idx[0]), *(const int64*)(tab + idx[1])));
|
||||
}
|
||||
inline v_int32x4 v_lut_quads(const int* tab, const int* idx)
|
||||
{
|
||||
return v_int32x4(vsx_ld(0, tab + idx[0]));
|
||||
}
|
||||
inline v_uint32x4 v_lut(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut((const int*)tab, idx)); }
|
||||
inline v_uint32x4 v_lut_pairs(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut_pairs((const int*)tab, idx)); }
|
||||
inline v_uint32x4 v_lut_quads(const unsigned* tab, const int* idx) { return v_reinterpret_as_u32(v_lut_quads((const int*)tab, idx)); }
|
||||
|
||||
inline v_int64x2 v_lut(const int64_t* tab, const int* idx)
|
||||
{
|
||||
return v_int64x2(tab[idx[0]], tab[idx[1]]);
|
||||
}
|
||||
inline v_int64x2 v_lut_pairs(const int64_t* tab, const int* idx)
|
||||
{
|
||||
return v_int64x2(vsx_ld2(0, tab + idx[0]));
|
||||
}
|
||||
inline v_uint64x2 v_lut(const uint64_t* tab, const int* idx) { return v_reinterpret_as_u64(v_lut((const int64_t *)tab, idx)); }
|
||||
inline v_uint64x2 v_lut_pairs(const uint64_t* tab, const int* idx) { return v_reinterpret_as_u64(v_lut_pairs((const int64_t *)tab, idx)); }
|
||||
|
||||
inline v_float32x4 v_lut(const float* tab, const int* idx)
|
||||
{
|
||||
return v_float32x4(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]);
|
||||
}
|
||||
inline v_float32x4 v_lut_pairs(const float* tab, const int* idx) { return v_reinterpret_as_f32(v_lut_pairs((const int*)tab, idx)); }
|
||||
inline v_float32x4 v_lut_quads(const float* tab, const int* idx) { return v_load(tab + *idx); }
|
||||
|
||||
inline v_float64x2 v_lut(const double* tab, const int* idx)
|
||||
{
|
||||
return v_float64x2(tab[idx[0]], tab[idx[1]]);
|
||||
}
|
||||
inline v_float64x2 v_lut_pairs(const double* tab, const int* idx) { return v_load(tab + *idx); }
|
||||
|
||||
inline v_int32x4 v_lut(const int* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) idx[4];
|
||||
@@ -1000,6 +1074,13 @@ inline v_int32x4 v_lut(const int* tab, const v_int32x4& idxvec)
|
||||
return v_int32x4(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]);
|
||||
}
|
||||
|
||||
inline v_uint32x4 v_lut(const unsigned* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) idx[4];
|
||||
v_store_aligned(idx, idxvec);
|
||||
return v_uint32x4(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]);
|
||||
}
|
||||
|
||||
inline v_float32x4 v_lut(const float* tab, const v_int32x4& idxvec)
|
||||
{
|
||||
int CV_DECL_ALIGNED(32) idx[4];
|
||||
@@ -1030,6 +1111,55 @@ inline void v_lut_deinterleave(const double* tab, const v_int32x4& idxvec, v_flo
|
||||
y = v_float64x2(tab[idx[0]+1], tab[idx[1]+1]);
|
||||
}
|
||||
|
||||
inline v_int8x16 v_interleave_pairs(const v_int8x16& vec)
|
||||
{
|
||||
vec_short8 vec0 = vec_mergeh((vec_short8)vec.val, (vec_short8)vec_mergesql(vec.val, vec.val));
|
||||
vec0 = vec_mergeh(vec0, vec_mergesql(vec0, vec0));
|
||||
return v_int8x16(vec_mergeh((vec_char16)vec0, (vec_char16)vec_mergesql(vec0, vec0)));
|
||||
}
|
||||
inline v_uint8x16 v_interleave_pairs(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_interleave_pairs(v_reinterpret_as_s8(vec))); }
|
||||
inline v_int8x16 v_interleave_quads(const v_int8x16& vec)
|
||||
{
|
||||
vec_char16 vec0 = (vec_char16)vec_mergeh((vec_int4)vec.val, (vec_int4)vec_mergesql(vec.val, vec.val));
|
||||
return v_int8x16(vec_mergeh(vec0, vec_mergesql(vec0, vec0)));
|
||||
}
|
||||
inline v_uint8x16 v_interleave_quads(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_interleave_quads(v_reinterpret_as_s8(vec))); }
|
||||
|
||||
inline v_int16x8 v_interleave_pairs(const v_int16x8& vec)
|
||||
{
|
||||
vec_short8 vec0 = (vec_short8)vec_mergeh((vec_int4)vec.val, (vec_int4)vec_mergesql(vec.val, vec.val));
|
||||
return v_int16x8(vec_mergeh(vec0, vec_mergesql(vec0, vec0)));
|
||||
}
|
||||
inline v_uint16x8 v_interleave_pairs(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_interleave_pairs(v_reinterpret_as_s16(vec))); }
|
||||
inline v_int16x8 v_interleave_quads(const v_int16x8& vec)
|
||||
{
|
||||
return v_int16x8(vec_mergeh(vec.val, vec_mergesql(vec.val, vec.val)));
|
||||
}
|
||||
inline v_uint16x8 v_interleave_quads(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_interleave_quads(v_reinterpret_as_s16(vec))); }
|
||||
|
||||
inline v_int32x4 v_interleave_pairs(const v_int32x4& vec)
|
||||
{
|
||||
return v_int32x4(vec_mergeh(vec.val, vec_mergesql(vec.val, vec.val)));
|
||||
}
|
||||
inline v_uint32x4 v_interleave_pairs(const v_uint32x4& vec) { return v_reinterpret_as_u32(v_interleave_pairs(v_reinterpret_as_s32(vec))); }
|
||||
inline v_float32x4 v_interleave_pairs(const v_float32x4& vec) { return v_reinterpret_as_f32(v_interleave_pairs(v_reinterpret_as_s32(vec))); }
|
||||
|
||||
inline v_int8x16 v_pack_triplets(const v_int8x16& vec)
|
||||
{
|
||||
schar CV_DECL_ALIGNED(32) val[16];
|
||||
v_store_aligned(val, vec);
|
||||
return v_int8x16(val[0], val[1], val[2], val[4], val[5], val[6], val[8], val[9], val[10], val[12], val[13], val[14], val[15], val[15], val[15], val[15]);
|
||||
}
|
||||
inline v_uint8x16 v_pack_triplets(const v_uint8x16& vec) { return v_reinterpret_as_u8(v_pack_triplets(v_reinterpret_as_s8(vec))); }
|
||||
|
||||
inline v_int16x8 v_pack_triplets(const v_int16x8& vec)
|
||||
{
|
||||
short CV_DECL_ALIGNED(32) val[8];
|
||||
v_store_aligned(val, vec);
|
||||
return v_int16x8(val[0], val[1], val[2], val[4], val[5], val[6], val[7], val[7]);
|
||||
}
|
||||
inline v_uint16x8 v_pack_triplets(const v_uint16x8& vec) { return v_reinterpret_as_u16(v_pack_triplets(v_reinterpret_as_s16(vec))); }
|
||||
|
||||
/////// FP16 support ////////
|
||||
|
||||
// [TODO] implement these 2 using VSX or universal intrinsics (copy from intrin_sse.cpp and adopt)
|
||||
|
||||
@@ -206,7 +206,7 @@ void cv::cuda::HostMem::create(int rows_, int cols_, int type_)
|
||||
cols = cols_;
|
||||
step = elemSize() * cols;
|
||||
int sz[] = { rows, cols };
|
||||
size_t steps[] = { step, CV_ELEM_SIZE(type_) };
|
||||
size_t steps[] = { step, (size_t)CV_ELEM_SIZE(type_) };
|
||||
flags = updateContinuityFlag(flags, 2, sz, steps);
|
||||
|
||||
if (alloc_type == SHARED)
|
||||
|
||||
@@ -14,9 +14,12 @@
|
||||
#undef CV_IPP_RUN
|
||||
#define CV_IPP_RUN(c, f, ...)
|
||||
|
||||
#include "mean.simd.hpp"
|
||||
#include "mean.simd_declarations.hpp" // defines CV_CPU_DISPATCH_MODES_ALL=AVX2,...,BASELINE based on CMakeLists.txt content
|
||||
|
||||
namespace cv {
|
||||
|
||||
#if defined HAVE_IPP
|
||||
namespace cv
|
||||
{
|
||||
static bool ipp_mean( Mat &src, Mat &mask, Scalar &ret )
|
||||
{
|
||||
CV_INSTRUMENT_REGION_IPP();
|
||||
@@ -107,10 +110,9 @@ static bool ipp_mean( Mat &src, Mat &mask, Scalar &ret )
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
cv::Scalar cv::mean( InputArray _src, InputArray _mask )
|
||||
Scalar mean(InputArray _src, InputArray _mask)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
@@ -173,314 +175,11 @@ cv::Scalar cv::mean( InputArray _src, InputArray _mask )
|
||||
return s*(nz0 ? 1./nz0 : 0);
|
||||
}
|
||||
|
||||
//==================================================================================================
|
||||
|
||||
namespace cv {
|
||||
|
||||
template <typename T, typename ST, typename SQT>
|
||||
struct SumSqr_SIMD
|
||||
static SumSqrFunc getSumSqrFunc(int depth)
|
||||
{
|
||||
int operator () (const T *, const uchar *, ST *, SQT *, int, int) const
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
|
||||
#if CV_SIMD
|
||||
|
||||
template <>
|
||||
struct SumSqr_SIMD<uchar, int, int>
|
||||
{
|
||||
int operator () (const uchar * src0, const uchar * mask, int * sum, int * sqsum, int len, int cn) const
|
||||
{
|
||||
if (mask || (cn != 1 && cn != 2 && cn != 4))
|
||||
return 0;
|
||||
len *= cn;
|
||||
|
||||
int x = 0;
|
||||
v_int32 v_sum = vx_setzero_s32();
|
||||
v_int32 v_sqsum = vx_setzero_s32();
|
||||
|
||||
const int len0 = len & -v_uint8::nlanes;
|
||||
while(x < len0)
|
||||
{
|
||||
const int len_tmp = min(x + 256*v_uint16::nlanes, len0);
|
||||
v_uint16 v_sum16 = vx_setzero_u16();
|
||||
for ( ; x < len_tmp; x += v_uint8::nlanes)
|
||||
{
|
||||
v_uint16 v_src0 = vx_load_expand(src0 + x);
|
||||
v_uint16 v_src1 = vx_load_expand(src0 + x + v_uint16::nlanes);
|
||||
v_sum16 += v_src0 + v_src1;
|
||||
v_int16 v_tmp0, v_tmp1;
|
||||
v_zip(v_reinterpret_as_s16(v_src0), v_reinterpret_as_s16(v_src1), v_tmp0, v_tmp1);
|
||||
v_sqsum += v_dotprod(v_tmp0, v_tmp0) + v_dotprod(v_tmp1, v_tmp1);
|
||||
}
|
||||
v_uint32 v_half0, v_half1;
|
||||
v_expand(v_sum16, v_half0, v_half1);
|
||||
v_sum += v_reinterpret_as_s32(v_half0 + v_half1);
|
||||
}
|
||||
if (x <= len - v_uint16::nlanes)
|
||||
{
|
||||
v_uint16 v_src = vx_load_expand(src0 + x);
|
||||
v_uint16 v_half = v_combine_high(v_src, v_src);
|
||||
|
||||
v_uint32 v_tmp0, v_tmp1;
|
||||
v_expand(v_src + v_half, v_tmp0, v_tmp1);
|
||||
v_sum += v_reinterpret_as_s32(v_tmp0);
|
||||
|
||||
v_int16 v_tmp2, v_tmp3;
|
||||
v_zip(v_reinterpret_as_s16(v_src), v_reinterpret_as_s16(v_half), v_tmp2, v_tmp3);
|
||||
v_sqsum += v_dotprod(v_tmp2, v_tmp2);
|
||||
x += v_uint16::nlanes;
|
||||
}
|
||||
|
||||
if (cn == 1)
|
||||
{
|
||||
*sum += v_reduce_sum(v_sum);
|
||||
*sqsum += v_reduce_sum(v_sqsum);
|
||||
}
|
||||
else
|
||||
{
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) ar[2 * v_int32::nlanes];
|
||||
v_store(ar, v_sum);
|
||||
v_store(ar + v_int32::nlanes, v_sqsum);
|
||||
for (int i = 0; i < v_int32::nlanes; ++i)
|
||||
{
|
||||
sum[i % cn] += ar[i];
|
||||
sqsum[i % cn] += ar[v_int32::nlanes + i];
|
||||
}
|
||||
}
|
||||
v_cleanup();
|
||||
return x / cn;
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct SumSqr_SIMD<schar, int, int>
|
||||
{
|
||||
int operator () (const schar * src0, const uchar * mask, int * sum, int * sqsum, int len, int cn) const
|
||||
{
|
||||
if (mask || (cn != 1 && cn != 2 && cn != 4))
|
||||
return 0;
|
||||
len *= cn;
|
||||
|
||||
int x = 0;
|
||||
v_int32 v_sum = vx_setzero_s32();
|
||||
v_int32 v_sqsum = vx_setzero_s32();
|
||||
|
||||
const int len0 = len & -v_int8::nlanes;
|
||||
while (x < len0)
|
||||
{
|
||||
const int len_tmp = min(x + 256 * v_int16::nlanes, len0);
|
||||
v_int16 v_sum16 = vx_setzero_s16();
|
||||
for (; x < len_tmp; x += v_int8::nlanes)
|
||||
{
|
||||
v_int16 v_src0 = vx_load_expand(src0 + x);
|
||||
v_int16 v_src1 = vx_load_expand(src0 + x + v_int16::nlanes);
|
||||
v_sum16 += v_src0 + v_src1;
|
||||
v_int16 v_tmp0, v_tmp1;
|
||||
v_zip(v_src0, v_src1, v_tmp0, v_tmp1);
|
||||
v_sqsum += v_dotprod(v_tmp0, v_tmp0) + v_dotprod(v_tmp1, v_tmp1);
|
||||
}
|
||||
v_int32 v_half0, v_half1;
|
||||
v_expand(v_sum16, v_half0, v_half1);
|
||||
v_sum += v_half0 + v_half1;
|
||||
}
|
||||
if (x <= len - v_int16::nlanes)
|
||||
{
|
||||
v_int16 v_src = vx_load_expand(src0 + x);
|
||||
v_int16 v_half = v_combine_high(v_src, v_src);
|
||||
|
||||
v_int32 v_tmp0, v_tmp1;
|
||||
v_expand(v_src + v_half, v_tmp0, v_tmp1);
|
||||
v_sum += v_tmp0;
|
||||
|
||||
v_int16 v_tmp2, v_tmp3;
|
||||
v_zip(v_src, v_half, v_tmp2, v_tmp3);
|
||||
v_sqsum += v_dotprod(v_tmp2, v_tmp2);
|
||||
x += v_int16::nlanes;
|
||||
}
|
||||
|
||||
if (cn == 1)
|
||||
{
|
||||
*sum += v_reduce_sum(v_sum);
|
||||
*sqsum += v_reduce_sum(v_sqsum);
|
||||
}
|
||||
else
|
||||
{
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) ar[2 * v_int32::nlanes];
|
||||
v_store(ar, v_sum);
|
||||
v_store(ar + v_int32::nlanes, v_sqsum);
|
||||
for (int i = 0; i < v_int32::nlanes; ++i)
|
||||
{
|
||||
sum[i % cn] += ar[i];
|
||||
sqsum[i % cn] += ar[v_int32::nlanes + i];
|
||||
}
|
||||
}
|
||||
v_cleanup();
|
||||
return x / cn;
|
||||
}
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
template<typename T, typename ST, typename SQT>
|
||||
static int sumsqr_(const T* src0, const uchar* mask, ST* sum, SQT* sqsum, int len, int cn )
|
||||
{
|
||||
const T* src = src0;
|
||||
|
||||
if( !mask )
|
||||
{
|
||||
SumSqr_SIMD<T, ST, SQT> vop;
|
||||
int x = vop(src0, mask, sum, sqsum, len, cn), k = cn % 4;
|
||||
src = src0 + x * cn;
|
||||
|
||||
if( k == 1 )
|
||||
{
|
||||
ST s0 = sum[0];
|
||||
SQT sq0 = sqsum[0];
|
||||
for(int i = x; i < len; i++, src += cn )
|
||||
{
|
||||
T v = src[0];
|
||||
s0 += v; sq0 += (SQT)v*v;
|
||||
}
|
||||
sum[0] = s0;
|
||||
sqsum[0] = sq0;
|
||||
}
|
||||
else if( k == 2 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1];
|
||||
SQT sq0 = sqsum[0], sq1 = sqsum[1];
|
||||
for(int i = x; i < len; i++, src += cn )
|
||||
{
|
||||
T v0 = src[0], v1 = src[1];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
}
|
||||
sum[0] = s0; sum[1] = s1;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1;
|
||||
}
|
||||
else if( k == 3 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1], s2 = sum[2];
|
||||
SQT sq0 = sqsum[0], sq1 = sqsum[1], sq2 = sqsum[2];
|
||||
for(int i = x; i < len; i++, src += cn )
|
||||
{
|
||||
T v0 = src[0], v1 = src[1], v2 = src[2];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
s2 += v2; sq2 += (SQT)v2*v2;
|
||||
}
|
||||
sum[0] = s0; sum[1] = s1; sum[2] = s2;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1; sqsum[2] = sq2;
|
||||
}
|
||||
|
||||
for( ; k < cn; k += 4 )
|
||||
{
|
||||
src = src0 + x * cn + k;
|
||||
ST s0 = sum[k], s1 = sum[k+1], s2 = sum[k+2], s3 = sum[k+3];
|
||||
SQT sq0 = sqsum[k], sq1 = sqsum[k+1], sq2 = sqsum[k+2], sq3 = sqsum[k+3];
|
||||
for(int i = x; i < len; i++, src += cn )
|
||||
{
|
||||
T v0, v1;
|
||||
v0 = src[0], v1 = src[1];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
v0 = src[2], v1 = src[3];
|
||||
s2 += v0; sq2 += (SQT)v0*v0;
|
||||
s3 += v1; sq3 += (SQT)v1*v1;
|
||||
}
|
||||
sum[k] = s0; sum[k+1] = s1;
|
||||
sum[k+2] = s2; sum[k+3] = s3;
|
||||
sqsum[k] = sq0; sqsum[k+1] = sq1;
|
||||
sqsum[k+2] = sq2; sqsum[k+3] = sq3;
|
||||
}
|
||||
return len;
|
||||
}
|
||||
|
||||
int i, nzm = 0;
|
||||
|
||||
if( cn == 1 )
|
||||
{
|
||||
ST s0 = sum[0];
|
||||
SQT sq0 = sqsum[0];
|
||||
for( i = 0; i < len; i++ )
|
||||
if( mask[i] )
|
||||
{
|
||||
T v = src[i];
|
||||
s0 += v; sq0 += (SQT)v*v;
|
||||
nzm++;
|
||||
}
|
||||
sum[0] = s0;
|
||||
sqsum[0] = sq0;
|
||||
}
|
||||
else if( cn == 3 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1], s2 = sum[2];
|
||||
SQT sq0 = sqsum[0], sq1 = sqsum[1], sq2 = sqsum[2];
|
||||
for( i = 0; i < len; i++, src += 3 )
|
||||
if( mask[i] )
|
||||
{
|
||||
T v0 = src[0], v1 = src[1], v2 = src[2];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
s2 += v2; sq2 += (SQT)v2*v2;
|
||||
nzm++;
|
||||
}
|
||||
sum[0] = s0; sum[1] = s1; sum[2] = s2;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1; sqsum[2] = sq2;
|
||||
}
|
||||
else
|
||||
{
|
||||
for( i = 0; i < len; i++, src += cn )
|
||||
if( mask[i] )
|
||||
{
|
||||
for( int k = 0; k < cn; k++ )
|
||||
{
|
||||
T v = src[k];
|
||||
ST s = sum[k] + v;
|
||||
SQT sq = sqsum[k] + (SQT)v*v;
|
||||
sum[k] = s; sqsum[k] = sq;
|
||||
}
|
||||
nzm++;
|
||||
}
|
||||
}
|
||||
return nzm;
|
||||
}
|
||||
|
||||
|
||||
static int sqsum8u( const uchar* src, const uchar* mask, int* sum, int* sqsum, int len, int cn )
|
||||
{ return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum8s( const schar* src, const uchar* mask, int* sum, int* sqsum, int len, int cn )
|
||||
{ return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum16u( const ushort* src, const uchar* mask, int* sum, double* sqsum, int len, int cn )
|
||||
{ return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum16s( const short* src, const uchar* mask, int* sum, double* sqsum, int len, int cn )
|
||||
{ return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum32s( const int* src, const uchar* mask, double* sum, double* sqsum, int len, int cn )
|
||||
{ return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum32f( const float* src, const uchar* mask, double* sum, double* sqsum, int len, int cn )
|
||||
{ return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum64f( const double* src, const uchar* mask, double* sum, double* sqsum, int len, int cn )
|
||||
{ return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
typedef int (*SumSqrFunc)(const uchar*, const uchar* mask, uchar*, uchar*, int, int);
|
||||
|
||||
static SumSqrFunc getSumSqrTab(int depth)
|
||||
{
|
||||
static SumSqrFunc sumSqrTab[] =
|
||||
{
|
||||
(SumSqrFunc)GET_OPTIMIZED(sqsum8u), (SumSqrFunc)sqsum8s, (SumSqrFunc)sqsum16u, (SumSqrFunc)sqsum16s,
|
||||
(SumSqrFunc)sqsum32s, (SumSqrFunc)GET_OPTIMIZED(sqsum32f), (SumSqrFunc)sqsum64f, 0
|
||||
};
|
||||
|
||||
return sumSqrTab[depth];
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getSumSqrFunc, (depth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
@@ -804,9 +503,7 @@ static bool ipp_meanStdDev(Mat& src, OutputArray _mean, OutputArray _sdv, Mat& m
|
||||
}
|
||||
#endif
|
||||
|
||||
} // cv::
|
||||
|
||||
void cv::meanStdDev( InputArray _src, OutputArray _mean, OutputArray _sdv, InputArray _mask )
|
||||
void meanStdDev(InputArray _src, OutputArray _mean, OutputArray _sdv, InputArray _mask)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
@@ -825,7 +522,7 @@ void cv::meanStdDev( InputArray _src, OutputArray _mean, OutputArray _sdv, Input
|
||||
|
||||
int k, cn = src.channels(), depth = src.depth();
|
||||
|
||||
SumSqrFunc func = getSumSqrTab(depth);
|
||||
SumSqrFunc func = getSumSqrFunc(depth);
|
||||
|
||||
CV_Assert( func != 0 );
|
||||
|
||||
@@ -913,3 +610,5 @@ void cv::meanStdDev( InputArray _src, OutputArray _mean, OutputArray _sdv, Input
|
||||
dptr[k] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -0,0 +1,325 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
|
||||
#include "precomp.hpp"
|
||||
#include "stat.hpp"
|
||||
|
||||
namespace cv {
|
||||
typedef int (*SumSqrFunc)(const uchar*, const uchar* mask, uchar*, uchar*, int, int);
|
||||
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
SumSqrFunc getSumSqrFunc(int depth);
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
template <typename T, typename ST, typename SQT>
|
||||
struct SumSqr_SIMD
|
||||
{
|
||||
inline int operator () (const T *, const uchar *, ST *, SQT *, int, int) const
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
|
||||
#if CV_SIMD
|
||||
|
||||
template <>
|
||||
struct SumSqr_SIMD<uchar, int, int>
|
||||
{
|
||||
int operator () (const uchar * src0, const uchar * mask, int * sum, int * sqsum, int len, int cn) const
|
||||
{
|
||||
if (mask || (cn != 1 && cn != 2 && cn != 4))
|
||||
return 0;
|
||||
len *= cn;
|
||||
|
||||
int x = 0;
|
||||
v_int32 v_sum = vx_setzero_s32();
|
||||
v_int32 v_sqsum = vx_setzero_s32();
|
||||
|
||||
const int len0 = len & -v_uint8::nlanes;
|
||||
while(x < len0)
|
||||
{
|
||||
const int len_tmp = min(x + 256*v_uint16::nlanes, len0);
|
||||
v_uint16 v_sum16 = vx_setzero_u16();
|
||||
for ( ; x < len_tmp; x += v_uint8::nlanes)
|
||||
{
|
||||
v_uint16 v_src0 = vx_load_expand(src0 + x);
|
||||
v_uint16 v_src1 = vx_load_expand(src0 + x + v_uint16::nlanes);
|
||||
v_sum16 += v_src0 + v_src1;
|
||||
v_int16 v_tmp0, v_tmp1;
|
||||
v_zip(v_reinterpret_as_s16(v_src0), v_reinterpret_as_s16(v_src1), v_tmp0, v_tmp1);
|
||||
v_sqsum += v_dotprod(v_tmp0, v_tmp0) + v_dotprod(v_tmp1, v_tmp1);
|
||||
}
|
||||
v_uint32 v_half0, v_half1;
|
||||
v_expand(v_sum16, v_half0, v_half1);
|
||||
v_sum += v_reinterpret_as_s32(v_half0 + v_half1);
|
||||
}
|
||||
if (x <= len - v_uint16::nlanes)
|
||||
{
|
||||
v_uint16 v_src = vx_load_expand(src0 + x);
|
||||
v_uint16 v_half = v_combine_high(v_src, v_src);
|
||||
|
||||
v_uint32 v_tmp0, v_tmp1;
|
||||
v_expand(v_src + v_half, v_tmp0, v_tmp1);
|
||||
v_sum += v_reinterpret_as_s32(v_tmp0);
|
||||
|
||||
v_int16 v_tmp2, v_tmp3;
|
||||
v_zip(v_reinterpret_as_s16(v_src), v_reinterpret_as_s16(v_half), v_tmp2, v_tmp3);
|
||||
v_sqsum += v_dotprod(v_tmp2, v_tmp2);
|
||||
x += v_uint16::nlanes;
|
||||
}
|
||||
|
||||
if (cn == 1)
|
||||
{
|
||||
*sum += v_reduce_sum(v_sum);
|
||||
*sqsum += v_reduce_sum(v_sqsum);
|
||||
}
|
||||
else
|
||||
{
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) ar[2 * v_int32::nlanes];
|
||||
v_store(ar, v_sum);
|
||||
v_store(ar + v_int32::nlanes, v_sqsum);
|
||||
for (int i = 0; i < v_int32::nlanes; ++i)
|
||||
{
|
||||
sum[i % cn] += ar[i];
|
||||
sqsum[i % cn] += ar[v_int32::nlanes + i];
|
||||
}
|
||||
}
|
||||
v_cleanup();
|
||||
return x / cn;
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct SumSqr_SIMD<schar, int, int>
|
||||
{
|
||||
int operator () (const schar * src0, const uchar * mask, int * sum, int * sqsum, int len, int cn) const
|
||||
{
|
||||
if (mask || (cn != 1 && cn != 2 && cn != 4))
|
||||
return 0;
|
||||
len *= cn;
|
||||
|
||||
int x = 0;
|
||||
v_int32 v_sum = vx_setzero_s32();
|
||||
v_int32 v_sqsum = vx_setzero_s32();
|
||||
|
||||
const int len0 = len & -v_int8::nlanes;
|
||||
while (x < len0)
|
||||
{
|
||||
const int len_tmp = min(x + 256 * v_int16::nlanes, len0);
|
||||
v_int16 v_sum16 = vx_setzero_s16();
|
||||
for (; x < len_tmp; x += v_int8::nlanes)
|
||||
{
|
||||
v_int16 v_src0 = vx_load_expand(src0 + x);
|
||||
v_int16 v_src1 = vx_load_expand(src0 + x + v_int16::nlanes);
|
||||
v_sum16 += v_src0 + v_src1;
|
||||
v_int16 v_tmp0, v_tmp1;
|
||||
v_zip(v_src0, v_src1, v_tmp0, v_tmp1);
|
||||
v_sqsum += v_dotprod(v_tmp0, v_tmp0) + v_dotprod(v_tmp1, v_tmp1);
|
||||
}
|
||||
v_int32 v_half0, v_half1;
|
||||
v_expand(v_sum16, v_half0, v_half1);
|
||||
v_sum += v_half0 + v_half1;
|
||||
}
|
||||
if (x <= len - v_int16::nlanes)
|
||||
{
|
||||
v_int16 v_src = vx_load_expand(src0 + x);
|
||||
v_int16 v_half = v_combine_high(v_src, v_src);
|
||||
|
||||
v_int32 v_tmp0, v_tmp1;
|
||||
v_expand(v_src + v_half, v_tmp0, v_tmp1);
|
||||
v_sum += v_tmp0;
|
||||
|
||||
v_int16 v_tmp2, v_tmp3;
|
||||
v_zip(v_src, v_half, v_tmp2, v_tmp3);
|
||||
v_sqsum += v_dotprod(v_tmp2, v_tmp2);
|
||||
x += v_int16::nlanes;
|
||||
}
|
||||
|
||||
if (cn == 1)
|
||||
{
|
||||
*sum += v_reduce_sum(v_sum);
|
||||
*sqsum += v_reduce_sum(v_sqsum);
|
||||
}
|
||||
else
|
||||
{
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) ar[2 * v_int32::nlanes];
|
||||
v_store(ar, v_sum);
|
||||
v_store(ar + v_int32::nlanes, v_sqsum);
|
||||
for (int i = 0; i < v_int32::nlanes; ++i)
|
||||
{
|
||||
sum[i % cn] += ar[i];
|
||||
sqsum[i % cn] += ar[v_int32::nlanes + i];
|
||||
}
|
||||
}
|
||||
v_cleanup();
|
||||
return x / cn;
|
||||
}
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
template<typename T, typename ST, typename SQT>
|
||||
static int sumsqr_(const T* src0, const uchar* mask, ST* sum, SQT* sqsum, int len, int cn )
|
||||
{
|
||||
const T* src = src0;
|
||||
|
||||
if( !mask )
|
||||
{
|
||||
SumSqr_SIMD<T, ST, SQT> vop;
|
||||
int x = vop(src0, mask, sum, sqsum, len, cn), k = cn % 4;
|
||||
src = src0 + x * cn;
|
||||
|
||||
if( k == 1 )
|
||||
{
|
||||
ST s0 = sum[0];
|
||||
SQT sq0 = sqsum[0];
|
||||
for(int i = x; i < len; i++, src += cn )
|
||||
{
|
||||
T v = src[0];
|
||||
s0 += v; sq0 += (SQT)v*v;
|
||||
}
|
||||
sum[0] = s0;
|
||||
sqsum[0] = sq0;
|
||||
}
|
||||
else if( k == 2 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1];
|
||||
SQT sq0 = sqsum[0], sq1 = sqsum[1];
|
||||
for(int i = x; i < len; i++, src += cn )
|
||||
{
|
||||
T v0 = src[0], v1 = src[1];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
}
|
||||
sum[0] = s0; sum[1] = s1;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1;
|
||||
}
|
||||
else if( k == 3 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1], s2 = sum[2];
|
||||
SQT sq0 = sqsum[0], sq1 = sqsum[1], sq2 = sqsum[2];
|
||||
for(int i = x; i < len; i++, src += cn )
|
||||
{
|
||||
T v0 = src[0], v1 = src[1], v2 = src[2];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
s2 += v2; sq2 += (SQT)v2*v2;
|
||||
}
|
||||
sum[0] = s0; sum[1] = s1; sum[2] = s2;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1; sqsum[2] = sq2;
|
||||
}
|
||||
|
||||
for( ; k < cn; k += 4 )
|
||||
{
|
||||
src = src0 + x * cn + k;
|
||||
ST s0 = sum[k], s1 = sum[k+1], s2 = sum[k+2], s3 = sum[k+3];
|
||||
SQT sq0 = sqsum[k], sq1 = sqsum[k+1], sq2 = sqsum[k+2], sq3 = sqsum[k+3];
|
||||
for(int i = x; i < len; i++, src += cn )
|
||||
{
|
||||
T v0, v1;
|
||||
v0 = src[0], v1 = src[1];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
v0 = src[2], v1 = src[3];
|
||||
s2 += v0; sq2 += (SQT)v0*v0;
|
||||
s3 += v1; sq3 += (SQT)v1*v1;
|
||||
}
|
||||
sum[k] = s0; sum[k+1] = s1;
|
||||
sum[k+2] = s2; sum[k+3] = s3;
|
||||
sqsum[k] = sq0; sqsum[k+1] = sq1;
|
||||
sqsum[k+2] = sq2; sqsum[k+3] = sq3;
|
||||
}
|
||||
return len;
|
||||
}
|
||||
|
||||
int i, nzm = 0;
|
||||
|
||||
if( cn == 1 )
|
||||
{
|
||||
ST s0 = sum[0];
|
||||
SQT sq0 = sqsum[0];
|
||||
for( i = 0; i < len; i++ )
|
||||
if( mask[i] )
|
||||
{
|
||||
T v = src[i];
|
||||
s0 += v; sq0 += (SQT)v*v;
|
||||
nzm++;
|
||||
}
|
||||
sum[0] = s0;
|
||||
sqsum[0] = sq0;
|
||||
}
|
||||
else if( cn == 3 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1], s2 = sum[2];
|
||||
SQT sq0 = sqsum[0], sq1 = sqsum[1], sq2 = sqsum[2];
|
||||
for( i = 0; i < len; i++, src += 3 )
|
||||
if( mask[i] )
|
||||
{
|
||||
T v0 = src[0], v1 = src[1], v2 = src[2];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
s2 += v2; sq2 += (SQT)v2*v2;
|
||||
nzm++;
|
||||
}
|
||||
sum[0] = s0; sum[1] = s1; sum[2] = s2;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1; sqsum[2] = sq2;
|
||||
}
|
||||
else
|
||||
{
|
||||
for( i = 0; i < len; i++, src += cn )
|
||||
if( mask[i] )
|
||||
{
|
||||
for( int k = 0; k < cn; k++ )
|
||||
{
|
||||
T v = src[k];
|
||||
ST s = sum[k] + v;
|
||||
SQT sq = sqsum[k] + (SQT)v*v;
|
||||
sum[k] = s; sqsum[k] = sq;
|
||||
}
|
||||
nzm++;
|
||||
}
|
||||
}
|
||||
return nzm;
|
||||
}
|
||||
|
||||
|
||||
static int sqsum8u( const uchar* src, const uchar* mask, int* sum, int* sqsum, int len, int cn )
|
||||
{ CV_INSTRUMENT_REGION(); return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum8s( const schar* src, const uchar* mask, int* sum, int* sqsum, int len, int cn )
|
||||
{ CV_INSTRUMENT_REGION(); return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum16u( const ushort* src, const uchar* mask, int* sum, double* sqsum, int len, int cn )
|
||||
{ CV_INSTRUMENT_REGION(); return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum16s( const short* src, const uchar* mask, int* sum, double* sqsum, int len, int cn )
|
||||
{ CV_INSTRUMENT_REGION(); return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum32s( const int* src, const uchar* mask, double* sum, double* sqsum, int len, int cn )
|
||||
{ CV_INSTRUMENT_REGION(); return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum32f( const float* src, const uchar* mask, double* sum, double* sqsum, int len, int cn )
|
||||
{ CV_INSTRUMENT_REGION(); return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
static int sqsum64f( const double* src, const uchar* mask, double* sum, double* sqsum, int len, int cn )
|
||||
{ CV_INSTRUMENT_REGION(); return sumsqr_(src, mask, sum, sqsum, len, cn); }
|
||||
|
||||
SumSqrFunc getSumSqrFunc(int depth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
static SumSqrFunc sumSqrTab[] =
|
||||
{
|
||||
(SumSqrFunc)GET_OPTIMIZED(sqsum8u), (SumSqrFunc)sqsum8s, (SumSqrFunc)sqsum16u, (SumSqrFunc)sqsum16s,
|
||||
(SumSqrFunc)sqsum32s, (SumSqrFunc)GET_OPTIMIZED(sqsum32f), (SumSqrFunc)sqsum64f, 0
|
||||
};
|
||||
|
||||
return sumSqrTab[depth];
|
||||
}
|
||||
|
||||
#endif
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_END
|
||||
} // namespace
|
||||
@@ -6,208 +6,44 @@
|
||||
#include "precomp.hpp"
|
||||
#include "opencl_kernels_core.hpp"
|
||||
|
||||
#include "merge.simd.hpp"
|
||||
#include "merge.simd_declarations.hpp" // defines CV_CPU_DISPATCH_MODES_ALL=AVX2,...,BASELINE based on CMakeLists.txt content
|
||||
|
||||
namespace cv { namespace hal {
|
||||
|
||||
#if CV_SIMD
|
||||
/*
|
||||
The trick with STORE_UNALIGNED/STORE_ALIGNED_NOCACHE is the following:
|
||||
on IA there are instructions movntps and such to which
|
||||
v_store_interleave(...., STORE_ALIGNED_NOCACHE) is mapped.
|
||||
Those instructions write directly into memory w/o touching cache
|
||||
that results in dramatic speed improvements, especially on
|
||||
large arrays (FullHD, 4K etc.).
|
||||
|
||||
Those intrinsics require the destination address to be aligned
|
||||
by 16/32 bits (with SSE2 and AVX2, respectively).
|
||||
So we potentially split the processing into 3 stages:
|
||||
1) the optional prefix part [0:i0), where we use simple unaligned stores.
|
||||
2) the optional main part [i0:len - VECSZ], where we use "nocache" mode.
|
||||
But in some cases we have to use unaligned stores in this part.
|
||||
3) the optional suffix part (the tail) (len - VECSZ:len) where we switch back to "unaligned" mode
|
||||
to process the remaining len - VECSZ elements.
|
||||
In principle there can be very poorly aligned data where there is no main part.
|
||||
For that we set i0=0 and use unaligned stores for the whole array.
|
||||
*/
|
||||
template<typename T, typename VecT> static void
|
||||
vecmerge_( const T** src, T* dst, int len, int cn )
|
||||
{
|
||||
const int VECSZ = VecT::nlanes;
|
||||
int i, i0 = 0;
|
||||
const T* src0 = src[0];
|
||||
const T* src1 = src[1];
|
||||
|
||||
const int dstElemSize = cn * sizeof(T);
|
||||
int r = (int)((size_t)(void*)dst % (VECSZ*sizeof(T)));
|
||||
hal::StoreMode mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
if( r != 0 )
|
||||
{
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
if (r % dstElemSize == 0 && len > VECSZ*2)
|
||||
i0 = VECSZ - (r / dstElemSize);
|
||||
}
|
||||
|
||||
if( cn == 2 )
|
||||
{
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a = vx_load(src0 + i), b = vx_load(src1 + i);
|
||||
v_store_interleave(dst + i*cn, a, b, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if( cn == 3 )
|
||||
{
|
||||
const T* src2 = src[2];
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a = vx_load(src0 + i), b = vx_load(src1 + i), c = vx_load(src2 + i);
|
||||
v_store_interleave(dst + i*cn, a, b, c, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
CV_Assert( cn == 4 );
|
||||
const T* src2 = src[2];
|
||||
const T* src3 = src[3];
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a = vx_load(src0 + i), b = vx_load(src1 + i);
|
||||
VecT c = vx_load(src2 + i), d = vx_load(src3 + i);
|
||||
v_store_interleave(dst + i*cn, a, b, c, d, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
vx_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename T> static void
|
||||
merge_( const T** src, T* dst, int len, int cn )
|
||||
{
|
||||
int k = cn % 4 ? cn % 4 : 4;
|
||||
int i, j;
|
||||
if( k == 1 )
|
||||
{
|
||||
const T* src0 = src[0];
|
||||
for( i = j = 0; i < len; i++, j += cn )
|
||||
dst[j] = src0[i];
|
||||
}
|
||||
else if( k == 2 )
|
||||
{
|
||||
const T *src0 = src[0], *src1 = src[1];
|
||||
i = j = 0;
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst[j] = src0[i];
|
||||
dst[j+1] = src1[i];
|
||||
}
|
||||
}
|
||||
else if( k == 3 )
|
||||
{
|
||||
const T *src0 = src[0], *src1 = src[1], *src2 = src[2];
|
||||
i = j = 0;
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst[j] = src0[i];
|
||||
dst[j+1] = src1[i];
|
||||
dst[j+2] = src2[i];
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const T *src0 = src[0], *src1 = src[1], *src2 = src[2], *src3 = src[3];
|
||||
i = j = 0;
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst[j] = src0[i]; dst[j+1] = src1[i];
|
||||
dst[j+2] = src2[i]; dst[j+3] = src3[i];
|
||||
}
|
||||
}
|
||||
|
||||
for( ; k < cn; k += 4 )
|
||||
{
|
||||
const T *src0 = src[k], *src1 = src[k+1], *src2 = src[k+2], *src3 = src[k+3];
|
||||
for( i = 0, j = k; i < len; i++, j += cn )
|
||||
{
|
||||
dst[j] = src0[i]; dst[j+1] = src1[i];
|
||||
dst[j+2] = src2[i]; dst[j+3] = src3[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void merge8u(const uchar** src, uchar* dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CALL_HAL(merge8u, cv_hal_merge8u, src, dst, len, cn)
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint8::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecmerge_<uchar, v_uint8>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
merge_(src, dst, len, cn);
|
||||
CV_CPU_DISPATCH(merge8u, (src, dst, len, cn),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
void merge16u(const ushort** src, ushort* dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CALL_HAL(merge16u, cv_hal_merge16u, src, dst, len, cn)
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint16::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecmerge_<ushort, v_uint16>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
merge_(src, dst, len, cn);
|
||||
CV_CPU_DISPATCH(merge16u, (src, dst, len, cn),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
void merge32s(const int** src, int* dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CALL_HAL(merge32s, cv_hal_merge32s, src, dst, len, cn)
|
||||
#if CV_SIMD
|
||||
if( len >= v_int32::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecmerge_<int, v_int32>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
merge_(src, dst, len, cn);
|
||||
CV_CPU_DISPATCH(merge32s, (src, dst, len, cn),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
void merge64s(const int64** src, int64* dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CALL_HAL(merge64s, cv_hal_merge64s, src, dst, len, cn)
|
||||
#if CV_SIMD
|
||||
if( len >= v_int64::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecmerge_<int64, v_int64>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
merge_(src, dst, len, cn);
|
||||
CV_CPU_DISPATCH(merge64s, (src, dst, len, cn),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
}} // cv::hal::
|
||||
} // namespace cv::hal::
|
||||
|
||||
|
||||
typedef void (*MergeFunc)(const uchar** src, uchar* dst, int len, int cn);
|
||||
@@ -227,7 +63,6 @@ static MergeFunc getMergeFunc(int depth)
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
|
||||
namespace cv {
|
||||
static bool ipp_merge(const Mat* mv, Mat& dst, int channels)
|
||||
{
|
||||
#ifdef HAVE_IPP_IW_LL
|
||||
@@ -276,10 +111,9 @@ static bool ipp_merge(const Mat* mv, Mat& dst, int channels)
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
void cv::merge(const Mat* mv, size_t n, OutputArray _dst)
|
||||
void merge(const Mat* mv, size_t n, OutputArray _dst)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
@@ -363,8 +197,6 @@ void cv::merge(const Mat* mv, size_t n, OutputArray _dst)
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
|
||||
namespace cv {
|
||||
|
||||
static bool ocl_merge( InputArrayOfArrays _mv, OutputArray _dst )
|
||||
{
|
||||
std::vector<UMat> src, ksrc;
|
||||
@@ -423,11 +255,9 @@ static bool ocl_merge( InputArrayOfArrays _mv, OutputArray _dst )
|
||||
return k.run(2, globalsize, NULL, false);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
void cv::merge(InputArrayOfArrays _mv, OutputArray _dst)
|
||||
void merge(InputArrayOfArrays _mv, OutputArray _dst)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
@@ -438,3 +268,5 @@ void cv::merge(InputArrayOfArrays _mv, OutputArray _dst)
|
||||
_mv.getMatVector(mv);
|
||||
merge(!mv.empty() ? &mv[0] : 0, mv.size(), _dst);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -0,0 +1,219 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
|
||||
#include "precomp.hpp"
|
||||
|
||||
namespace cv { namespace hal {
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
void merge8u(const uchar** src, uchar* dst, int len, int cn);
|
||||
void merge16u(const ushort** src, ushort* dst, int len, int cn);
|
||||
void merge32s(const int** src, int* dst, int len, int cn);
|
||||
void merge64s(const int64** src, int64* dst, int len, int cn);
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
#if CV_SIMD
|
||||
/*
|
||||
The trick with STORE_UNALIGNED/STORE_ALIGNED_NOCACHE is the following:
|
||||
on IA there are instructions movntps and such to which
|
||||
v_store_interleave(...., STORE_ALIGNED_NOCACHE) is mapped.
|
||||
Those instructions write directly into memory w/o touching cache
|
||||
that results in dramatic speed improvements, especially on
|
||||
large arrays (FullHD, 4K etc.).
|
||||
|
||||
Those intrinsics require the destination address to be aligned
|
||||
by 16/32 bits (with SSE2 and AVX2, respectively).
|
||||
So we potentially split the processing into 3 stages:
|
||||
1) the optional prefix part [0:i0), where we use simple unaligned stores.
|
||||
2) the optional main part [i0:len - VECSZ], where we use "nocache" mode.
|
||||
But in some cases we have to use unaligned stores in this part.
|
||||
3) the optional suffix part (the tail) (len - VECSZ:len) where we switch back to "unaligned" mode
|
||||
to process the remaining len - VECSZ elements.
|
||||
In principle there can be very poorly aligned data where there is no main part.
|
||||
For that we set i0=0 and use unaligned stores for the whole array.
|
||||
*/
|
||||
template<typename T, typename VecT> static void
|
||||
vecmerge_( const T** src, T* dst, int len, int cn )
|
||||
{
|
||||
const int VECSZ = VecT::nlanes;
|
||||
int i, i0 = 0;
|
||||
const T* src0 = src[0];
|
||||
const T* src1 = src[1];
|
||||
|
||||
const int dstElemSize = cn * sizeof(T);
|
||||
int r = (int)((size_t)(void*)dst % (VECSZ*sizeof(T)));
|
||||
hal::StoreMode mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
if( r != 0 )
|
||||
{
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
if (r % dstElemSize == 0 && len > VECSZ*2)
|
||||
i0 = VECSZ - (r / dstElemSize);
|
||||
}
|
||||
|
||||
if( cn == 2 )
|
||||
{
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a = vx_load(src0 + i), b = vx_load(src1 + i);
|
||||
v_store_interleave(dst + i*cn, a, b, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if( cn == 3 )
|
||||
{
|
||||
const T* src2 = src[2];
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a = vx_load(src0 + i), b = vx_load(src1 + i), c = vx_load(src2 + i);
|
||||
v_store_interleave(dst + i*cn, a, b, c, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
CV_Assert( cn == 4 );
|
||||
const T* src2 = src[2];
|
||||
const T* src3 = src[3];
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a = vx_load(src0 + i), b = vx_load(src1 + i);
|
||||
VecT c = vx_load(src2 + i), d = vx_load(src3 + i);
|
||||
v_store_interleave(dst + i*cn, a, b, c, d, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
vx_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename T> static void
|
||||
merge_( const T** src, T* dst, int len, int cn )
|
||||
{
|
||||
int k = cn % 4 ? cn % 4 : 4;
|
||||
int i, j;
|
||||
if( k == 1 )
|
||||
{
|
||||
const T* src0 = src[0];
|
||||
for( i = j = 0; i < len; i++, j += cn )
|
||||
dst[j] = src0[i];
|
||||
}
|
||||
else if( k == 2 )
|
||||
{
|
||||
const T *src0 = src[0], *src1 = src[1];
|
||||
i = j = 0;
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst[j] = src0[i];
|
||||
dst[j+1] = src1[i];
|
||||
}
|
||||
}
|
||||
else if( k == 3 )
|
||||
{
|
||||
const T *src0 = src[0], *src1 = src[1], *src2 = src[2];
|
||||
i = j = 0;
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst[j] = src0[i];
|
||||
dst[j+1] = src1[i];
|
||||
dst[j+2] = src2[i];
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const T *src0 = src[0], *src1 = src[1], *src2 = src[2], *src3 = src[3];
|
||||
i = j = 0;
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst[j] = src0[i]; dst[j+1] = src1[i];
|
||||
dst[j+2] = src2[i]; dst[j+3] = src3[i];
|
||||
}
|
||||
}
|
||||
|
||||
for( ; k < cn; k += 4 )
|
||||
{
|
||||
const T *src0 = src[k], *src1 = src[k+1], *src2 = src[k+2], *src3 = src[k+3];
|
||||
for( i = 0, j = k; i < len; i++, j += cn )
|
||||
{
|
||||
dst[j] = src0[i]; dst[j+1] = src1[i];
|
||||
dst[j+2] = src2[i]; dst[j+3] = src3[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void merge8u(const uchar** src, uchar* dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint8::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecmerge_<uchar, v_uint8>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
merge_(src, dst, len, cn);
|
||||
}
|
||||
|
||||
void merge16u(const ushort** src, ushort* dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint16::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecmerge_<ushort, v_uint16>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
merge_(src, dst, len, cn);
|
||||
}
|
||||
|
||||
void merge32s(const int** src, int* dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
#if CV_SIMD
|
||||
if( len >= v_int32::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecmerge_<int, v_int32>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
merge_(src, dst, len, cn);
|
||||
}
|
||||
|
||||
void merge64s(const int64** src, int64* dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
#if CV_SIMD
|
||||
if( len >= v_int64::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecmerge_<int64, v_int64>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
merge_(src, dst, len, cn);
|
||||
}
|
||||
|
||||
#endif
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_END
|
||||
}} // namespace
|
||||
@@ -6,213 +6,44 @@
|
||||
#include "precomp.hpp"
|
||||
#include "opencl_kernels_core.hpp"
|
||||
|
||||
#include "split.simd.hpp"
|
||||
#include "split.simd_declarations.hpp" // defines CV_CPU_DISPATCH_MODES_ALL=AVX2,...,BASELINE based on CMakeLists.txt content
|
||||
|
||||
namespace cv { namespace hal {
|
||||
|
||||
#if CV_SIMD
|
||||
// see the comments for vecmerge_ in merge.cpp
|
||||
template<typename T, typename VecT> static void
|
||||
vecsplit_( const T* src, T** dst, int len, int cn )
|
||||
{
|
||||
const int VECSZ = VecT::nlanes;
|
||||
int i, i0 = 0;
|
||||
T* dst0 = dst[0];
|
||||
T* dst1 = dst[1];
|
||||
|
||||
int r0 = (int)((size_t)(void*)dst0 % (VECSZ*sizeof(T)));
|
||||
int r1 = (int)((size_t)(void*)dst1 % (VECSZ*sizeof(T)));
|
||||
int r2 = cn > 2 ? (int)((size_t)(void*)dst[2] % (VECSZ*sizeof(T))) : r0;
|
||||
int r3 = cn > 3 ? (int)((size_t)(void*)dst[3] % (VECSZ*sizeof(T))) : r0;
|
||||
|
||||
hal::StoreMode mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
if( (r0|r1|r2|r3) != 0 )
|
||||
{
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
if( r0 == r1 && r0 == r2 && r0 == r3 && r0 % sizeof(T) == 0 && len > VECSZ*2 )
|
||||
i0 = VECSZ - (r0 / sizeof(T));
|
||||
}
|
||||
|
||||
if( cn == 2 )
|
||||
{
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a, b;
|
||||
v_load_deinterleave(src + i*cn, a, b);
|
||||
v_store(dst0 + i, a, mode);
|
||||
v_store(dst1 + i, b, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if( cn == 3 )
|
||||
{
|
||||
T* dst2 = dst[2];
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a, b, c;
|
||||
v_load_deinterleave(src + i*cn, a, b, c);
|
||||
v_store(dst0 + i, a, mode);
|
||||
v_store(dst1 + i, b, mode);
|
||||
v_store(dst2 + i, c, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
CV_Assert( cn == 4 );
|
||||
T* dst2 = dst[2];
|
||||
T* dst3 = dst[3];
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a, b, c, d;
|
||||
v_load_deinterleave(src + i*cn, a, b, c, d);
|
||||
v_store(dst0 + i, a, mode);
|
||||
v_store(dst1 + i, b, mode);
|
||||
v_store(dst2 + i, c, mode);
|
||||
v_store(dst3 + i, d, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
vx_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename T> static void
|
||||
split_( const T* src, T** dst, int len, int cn )
|
||||
{
|
||||
int k = cn % 4 ? cn % 4 : 4;
|
||||
int i, j;
|
||||
if( k == 1 )
|
||||
{
|
||||
T* dst0 = dst[0];
|
||||
|
||||
if(cn == 1)
|
||||
{
|
||||
memcpy(dst0, src, len * sizeof(T));
|
||||
}
|
||||
else
|
||||
{
|
||||
for( i = 0, j = 0 ; i < len; i++, j += cn )
|
||||
dst0[i] = src[j];
|
||||
}
|
||||
}
|
||||
else if( k == 2 )
|
||||
{
|
||||
T *dst0 = dst[0], *dst1 = dst[1];
|
||||
i = j = 0;
|
||||
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst0[i] = src[j];
|
||||
dst1[i] = src[j+1];
|
||||
}
|
||||
}
|
||||
else if( k == 3 )
|
||||
{
|
||||
T *dst0 = dst[0], *dst1 = dst[1], *dst2 = dst[2];
|
||||
i = j = 0;
|
||||
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst0[i] = src[j];
|
||||
dst1[i] = src[j+1];
|
||||
dst2[i] = src[j+2];
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
T *dst0 = dst[0], *dst1 = dst[1], *dst2 = dst[2], *dst3 = dst[3];
|
||||
i = j = 0;
|
||||
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst0[i] = src[j]; dst1[i] = src[j+1];
|
||||
dst2[i] = src[j+2]; dst3[i] = src[j+3];
|
||||
}
|
||||
}
|
||||
|
||||
for( ; k < cn; k += 4 )
|
||||
{
|
||||
T *dst0 = dst[k], *dst1 = dst[k+1], *dst2 = dst[k+2], *dst3 = dst[k+3];
|
||||
for( i = 0, j = k; i < len; i++, j += cn )
|
||||
{
|
||||
dst0[i] = src[j]; dst1[i] = src[j+1];
|
||||
dst2[i] = src[j+2]; dst3[i] = src[j+3];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void split8u(const uchar* src, uchar** dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CALL_HAL(split8u, cv_hal_split8u, src,dst, len, cn)
|
||||
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint8::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecsplit_<uchar, v_uint8>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
split_(src, dst, len, cn);
|
||||
CV_CPU_DISPATCH(split8u, (src, dst, len, cn),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
void split16u(const ushort* src, ushort** dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CALL_HAL(split16u, cv_hal_split16u, src,dst, len, cn)
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint16::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecsplit_<ushort, v_uint16>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
split_(src, dst, len, cn);
|
||||
CV_CPU_DISPATCH(split16u, (src, dst, len, cn),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
void split32s(const int* src, int** dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CALL_HAL(split32s, cv_hal_split32s, src,dst, len, cn)
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint32::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecsplit_<int, v_int32>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
split_(src, dst, len, cn);
|
||||
CV_CPU_DISPATCH(split32s, (src, dst, len, cn),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
void split64s(const int64* src, int64** dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CALL_HAL(split64s, cv_hal_split64s, src,dst, len, cn)
|
||||
#if CV_SIMD
|
||||
if( len >= v_int64::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecsplit_<int64, v_int64>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
split_(src, dst, len, cn);
|
||||
CV_CPU_DISPATCH(split64s, (src, dst, len, cn),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
}} // cv::hal::
|
||||
} // namespace cv::hal::
|
||||
|
||||
/****************************************************************************************\
|
||||
* split & merge *
|
||||
@@ -235,7 +66,6 @@ static SplitFunc getSplitFunc(int depth)
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
|
||||
namespace cv {
|
||||
static bool ipp_split(const Mat& src, Mat* mv, int channels)
|
||||
{
|
||||
#ifdef HAVE_IPP_IW_LL
|
||||
@@ -284,10 +114,9 @@ static bool ipp_split(const Mat& src, Mat* mv, int channels)
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
void cv::split(const Mat& src, Mat* mv)
|
||||
void split(const Mat& src, Mat* mv)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
@@ -343,8 +172,6 @@ void cv::split(const Mat& src, Mat* mv)
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
|
||||
namespace cv {
|
||||
|
||||
static bool ocl_split( InputArray _m, OutputArrayOfArrays _mv )
|
||||
{
|
||||
int type = _m.type(), depth = CV_MAT_DEPTH(type), cn = CV_MAT_CN(type),
|
||||
@@ -383,11 +210,9 @@ static bool ocl_split( InputArray _m, OutputArrayOfArrays _mv )
|
||||
return k.run(2, globalsize, NULL, false);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
void cv::split(InputArray _m, OutputArrayOfArrays _mv)
|
||||
void split(InputArray _m, OutputArrayOfArrays _mv)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
@@ -413,3 +238,5 @@ void cv::split(InputArray _m, OutputArrayOfArrays _mv)
|
||||
|
||||
split(m, &dst[0]);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -0,0 +1,223 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
|
||||
#include "precomp.hpp"
|
||||
|
||||
namespace cv { namespace hal {
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
void split8u(const uchar* src, uchar** dst, int len, int cn);
|
||||
void split16u(const ushort* src, ushort** dst, int len, int cn);
|
||||
void split32s(const int* src, int** dst, int len, int cn);
|
||||
void split64s(const int64* src, int64** dst, int len, int cn);
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
#if CV_SIMD
|
||||
// see the comments for vecmerge_ in merge.cpp
|
||||
template<typename T, typename VecT> static void
|
||||
vecsplit_( const T* src, T** dst, int len, int cn )
|
||||
{
|
||||
const int VECSZ = VecT::nlanes;
|
||||
int i, i0 = 0;
|
||||
T* dst0 = dst[0];
|
||||
T* dst1 = dst[1];
|
||||
|
||||
int r0 = (int)((size_t)(void*)dst0 % (VECSZ*sizeof(T)));
|
||||
int r1 = (int)((size_t)(void*)dst1 % (VECSZ*sizeof(T)));
|
||||
int r2 = cn > 2 ? (int)((size_t)(void*)dst[2] % (VECSZ*sizeof(T))) : r0;
|
||||
int r3 = cn > 3 ? (int)((size_t)(void*)dst[3] % (VECSZ*sizeof(T))) : r0;
|
||||
|
||||
hal::StoreMode mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
if( (r0|r1|r2|r3) != 0 )
|
||||
{
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
if( r0 == r1 && r0 == r2 && r0 == r3 && r0 % sizeof(T) == 0 && len > VECSZ*2 )
|
||||
i0 = VECSZ - (r0 / sizeof(T));
|
||||
}
|
||||
|
||||
if( cn == 2 )
|
||||
{
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a, b;
|
||||
v_load_deinterleave(src + i*cn, a, b);
|
||||
v_store(dst0 + i, a, mode);
|
||||
v_store(dst1 + i, b, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
else if( cn == 3 )
|
||||
{
|
||||
T* dst2 = dst[2];
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a, b, c;
|
||||
v_load_deinterleave(src + i*cn, a, b, c);
|
||||
v_store(dst0 + i, a, mode);
|
||||
v_store(dst1 + i, b, mode);
|
||||
v_store(dst2 + i, c, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
CV_Assert( cn == 4 );
|
||||
T* dst2 = dst[2];
|
||||
T* dst3 = dst[3];
|
||||
for( i = 0; i < len; i += VECSZ )
|
||||
{
|
||||
if( i > len - VECSZ )
|
||||
{
|
||||
i = len - VECSZ;
|
||||
mode = hal::STORE_UNALIGNED;
|
||||
}
|
||||
VecT a, b, c, d;
|
||||
v_load_deinterleave(src + i*cn, a, b, c, d);
|
||||
v_store(dst0 + i, a, mode);
|
||||
v_store(dst1 + i, b, mode);
|
||||
v_store(dst2 + i, c, mode);
|
||||
v_store(dst3 + i, d, mode);
|
||||
if( i < i0 )
|
||||
{
|
||||
i = i0 - VECSZ;
|
||||
mode = hal::STORE_ALIGNED_NOCACHE;
|
||||
}
|
||||
}
|
||||
}
|
||||
vx_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename T> static void
|
||||
split_( const T* src, T** dst, int len, int cn )
|
||||
{
|
||||
int k = cn % 4 ? cn % 4 : 4;
|
||||
int i, j;
|
||||
if( k == 1 )
|
||||
{
|
||||
T* dst0 = dst[0];
|
||||
|
||||
if(cn == 1)
|
||||
{
|
||||
memcpy(dst0, src, len * sizeof(T));
|
||||
}
|
||||
else
|
||||
{
|
||||
for( i = 0, j = 0 ; i < len; i++, j += cn )
|
||||
dst0[i] = src[j];
|
||||
}
|
||||
}
|
||||
else if( k == 2 )
|
||||
{
|
||||
T *dst0 = dst[0], *dst1 = dst[1];
|
||||
i = j = 0;
|
||||
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst0[i] = src[j];
|
||||
dst1[i] = src[j+1];
|
||||
}
|
||||
}
|
||||
else if( k == 3 )
|
||||
{
|
||||
T *dst0 = dst[0], *dst1 = dst[1], *dst2 = dst[2];
|
||||
i = j = 0;
|
||||
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst0[i] = src[j];
|
||||
dst1[i] = src[j+1];
|
||||
dst2[i] = src[j+2];
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
T *dst0 = dst[0], *dst1 = dst[1], *dst2 = dst[2], *dst3 = dst[3];
|
||||
i = j = 0;
|
||||
|
||||
for( ; i < len; i++, j += cn )
|
||||
{
|
||||
dst0[i] = src[j]; dst1[i] = src[j+1];
|
||||
dst2[i] = src[j+2]; dst3[i] = src[j+3];
|
||||
}
|
||||
}
|
||||
|
||||
for( ; k < cn; k += 4 )
|
||||
{
|
||||
T *dst0 = dst[k], *dst1 = dst[k+1], *dst2 = dst[k+2], *dst3 = dst[k+3];
|
||||
for( i = 0, j = k; i < len; i++, j += cn )
|
||||
{
|
||||
dst0[i] = src[j]; dst1[i] = src[j+1];
|
||||
dst2[i] = src[j+2]; dst3[i] = src[j+3];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void split8u(const uchar* src, uchar** dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint8::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecsplit_<uchar, v_uint8>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
split_(src, dst, len, cn);
|
||||
}
|
||||
|
||||
void split16u(const ushort* src, ushort** dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint16::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecsplit_<ushort, v_uint16>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
split_(src, dst, len, cn);
|
||||
}
|
||||
|
||||
void split32s(const int* src, int** dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
#if CV_SIMD
|
||||
if( len >= v_uint32::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecsplit_<int, v_int32>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
split_(src, dst, len, cn);
|
||||
}
|
||||
|
||||
void split64s(const int64* src, int64** dst, int len, int cn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
#if CV_SIMD
|
||||
if( len >= v_int64::nlanes && 2 <= cn && cn <= 4 )
|
||||
vecsplit_<int64, v_int64>(src, dst, len, cn);
|
||||
else
|
||||
#endif
|
||||
split_(src, dst, len, cn);
|
||||
}
|
||||
|
||||
#endif
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_END
|
||||
}} // namespace
|
||||
@@ -162,24 +162,6 @@ PERF_TEST_P_(DNNTestNetwork, DenseNet_121)
|
||||
Mat(cv::Size(224, 224), CV_32FC3));
|
||||
}
|
||||
|
||||
PERF_TEST_P_(DNNTestNetwork, OpenPose_pose_coco)
|
||||
{
|
||||
if (backend == DNN_BACKEND_HALIDE ||
|
||||
(backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD))
|
||||
throw SkipTestException("");
|
||||
processNet("dnn/openpose_pose_coco.caffemodel", "dnn/openpose_pose_coco.prototxt", "",
|
||||
Mat(cv::Size(368, 368), CV_32FC3));
|
||||
}
|
||||
|
||||
PERF_TEST_P_(DNNTestNetwork, OpenPose_pose_mpi)
|
||||
{
|
||||
if (backend == DNN_BACKEND_HALIDE ||
|
||||
(backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD))
|
||||
throw SkipTestException("");
|
||||
processNet("dnn/openpose_pose_mpi.caffemodel", "dnn/openpose_pose_mpi.prototxt", "",
|
||||
Mat(cv::Size(368, 368), CV_32FC3));
|
||||
}
|
||||
|
||||
PERF_TEST_P_(DNNTestNetwork, OpenPose_pose_mpi_faster_4_stages)
|
||||
{
|
||||
if (backend == DNN_BACKEND_HALIDE ||
|
||||
@@ -219,11 +201,7 @@ PERF_TEST_P_(DNNTestNetwork, YOLOv3)
|
||||
|
||||
PERF_TEST_P_(DNNTestNetwork, EAST_text_detection)
|
||||
{
|
||||
if (backend == DNN_BACKEND_HALIDE
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
|| (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
#endif
|
||||
)
|
||||
if (backend == DNN_BACKEND_HALIDE)
|
||||
throw SkipTestException("");
|
||||
processNet("dnn/frozen_east_text_detection.pb", "", "", Mat(cv::Size(320, 320), CV_32FC3));
|
||||
}
|
||||
|
||||
@@ -1231,10 +1231,7 @@ public:
|
||||
const int group = numOutput / outGroupCn;
|
||||
if (group != 1)
|
||||
{
|
||||
#if INF_ENGINE_VER_MAJOR_GE(INF_ENGINE_RELEASE_2018R3)
|
||||
return preferableTarget == DNN_TARGET_CPU;
|
||||
#endif
|
||||
return false;
|
||||
}
|
||||
if (preferableTarget == DNN_TARGET_OPENCL || preferableTarget == DNN_TARGET_OPENCL_FP16)
|
||||
return dilation.width == 1 && dilation.height == 1;
|
||||
@@ -1287,12 +1284,8 @@ public:
|
||||
int dims[] = {inputs[0][0], outCn, outH, outW};
|
||||
outputs.resize(inputs.size(), shape(dims, 4));
|
||||
|
||||
internals.push_back(MatShape());
|
||||
if (!is1x1())
|
||||
internals[0] = computeColRowShape(inputs[0], outputs[0]);
|
||||
|
||||
if (hasBias())
|
||||
internals.push_back(shape(1, outH*outW));
|
||||
internals.push_back(computeColRowShape(inputs[0], outputs[0]));
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -179,7 +179,7 @@ public:
|
||||
_backgroundLabelId = getParameter<int>(params, "background_label_id");
|
||||
_varianceEncodedInTarget = getParameter<bool>(params, "variance_encoded_in_target", 0, false, false);
|
||||
_keepTopK = getParameter<int>(params, "keep_top_k");
|
||||
_confidenceThreshold = getParameter<float>(params, "confidence_threshold", 0, false, -FLT_MAX);
|
||||
_confidenceThreshold = getParameter<float>(params, "confidence_threshold", 0, false, 0);
|
||||
_topK = getParameter<int>(params, "top_k", 0, false, -1);
|
||||
_locPredTransposed = getParameter<bool>(params, "loc_pred_transposed", 0, false, false);
|
||||
_bboxesNormalized = getParameter<bool>(params, "normalized_bbox", 0, false, true);
|
||||
|
||||
@@ -91,9 +91,10 @@ public:
|
||||
|
||||
virtual bool supportBackend(int backendId) CV_OVERRIDE
|
||||
{
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE)
|
||||
return (bias == 1) && (preferableTarget != DNN_TARGET_MYRIAD || type == SPATIAL_NRM);
|
||||
return backendId == DNN_BACKEND_OPENCV ||
|
||||
backendId == DNN_BACKEND_HALIDE ||
|
||||
backendId == DNN_BACKEND_INFERENCE_ENGINE ||
|
||||
(backendId == DNN_BACKEND_VKCOM && haveVulkan() && (size % 2 == 1) && (type == CHANNEL_NRM));
|
||||
}
|
||||
|
||||
@@ -393,10 +394,13 @@ public:
|
||||
virtual Ptr<BackendNode> initInfEngine(const std::vector<Ptr<BackendWrapper> >&) CV_OVERRIDE
|
||||
{
|
||||
#ifdef HAVE_INF_ENGINE
|
||||
float alphaSize = alpha;
|
||||
if (!normBySize)
|
||||
alphaSize *= (type == SPATIAL_NRM ? size*size : size);
|
||||
#if INF_ENGINE_VER_MAJOR_GE(INF_ENGINE_RELEASE_2018R5)
|
||||
InferenceEngine::Builder::NormLayer ieLayer(name);
|
||||
ieLayer.setSize(size);
|
||||
ieLayer.setAlpha(alpha);
|
||||
ieLayer.setAlpha(alphaSize);
|
||||
ieLayer.setBeta(beta);
|
||||
ieLayer.setAcrossMaps(type == CHANNEL_NRM);
|
||||
|
||||
@@ -413,7 +417,7 @@ public:
|
||||
ieLayer->_size = size;
|
||||
ieLayer->_k = (int)bias;
|
||||
ieLayer->_beta = beta;
|
||||
ieLayer->_alpha = alpha;
|
||||
ieLayer->_alpha = alphaSize;
|
||||
ieLayer->_isAcrossMaps = (type == CHANNEL_NRM);
|
||||
return Ptr<BackendNode>(new InfEngineBackendNode(ieLayer));
|
||||
#endif
|
||||
|
||||
@@ -392,10 +392,10 @@ void ONNXImporter::populateNet(Net dstNet)
|
||||
layerParams.set("ceil_mode", isCeilMode(layerParams));
|
||||
layerParams.set("ave_pool_padded_area", framework_name == "pytorch");
|
||||
}
|
||||
else if (layer_type == "GlobalAveragePool")
|
||||
else if (layer_type == "GlobalAveragePool" || layer_type == "GlobalMaxPool")
|
||||
{
|
||||
layerParams.type = "Pooling";
|
||||
layerParams.set("pool", "AVE");
|
||||
layerParams.set("pool", layer_type == "GlobalAveragePool" ? "AVE" : "MAX");
|
||||
layerParams.set("global_pooling", true);
|
||||
}
|
||||
else if (layer_type == "Add" || layer_type == "Sum")
|
||||
@@ -448,6 +448,11 @@ void ONNXImporter::populateNet(Net dstNet)
|
||||
layerParams.set("bias_term", false);
|
||||
}
|
||||
}
|
||||
else if (layer_type == "Neg")
|
||||
{
|
||||
layerParams.type = "Power";
|
||||
layerParams.set("scale", -1);
|
||||
}
|
||||
else if (layer_type == "Constant")
|
||||
{
|
||||
CV_Assert(node_proto.input_size() == 0);
|
||||
@@ -584,7 +589,7 @@ void ONNXImporter::populateNet(Net dstNet)
|
||||
for (int j = 1; j < node_proto.input_size(); j++) {
|
||||
layerParams.blobs.push_back(getBlob(node_proto, constBlobs, j));
|
||||
}
|
||||
layerParams.set("num_output", layerParams.blobs[0].size[1]);
|
||||
layerParams.set("num_output", layerParams.blobs[0].size[1] * layerParams.get<int>("group", 1));
|
||||
layerParams.set("bias_term", node_proto.input_size() == 3);
|
||||
}
|
||||
else if (layer_type == "Transpose")
|
||||
@@ -595,21 +600,35 @@ void ONNXImporter::populateNet(Net dstNet)
|
||||
else if (layer_type == "Unsqueeze")
|
||||
{
|
||||
CV_Assert(node_proto.input_size() == 1);
|
||||
Mat input = getBlob(node_proto, constBlobs, 0);
|
||||
|
||||
DictValue axes = layerParams.get("axes");
|
||||
std::vector<int> dims;
|
||||
for (int j = 0; j < input.dims; j++) {
|
||||
dims.push_back(input.size[j]);
|
||||
}
|
||||
CV_Assert(axes.getIntValue(axes.size()-1) <= dims.size());
|
||||
for (int j = 0; j < axes.size(); j++) {
|
||||
dims.insert(dims.begin() + axes.getIntValue(j), 1);
|
||||
if (constBlobs.find(node_proto.input(0)) != constBlobs.end())
|
||||
{
|
||||
// Constant input.
|
||||
Mat input = getBlob(node_proto, constBlobs, 0);
|
||||
|
||||
std::vector<int> dims;
|
||||
for (int j = 0; j < input.dims; j++) {
|
||||
dims.push_back(input.size[j]);
|
||||
}
|
||||
CV_Assert(axes.getIntValue(axes.size()-1) <= dims.size());
|
||||
for (int j = 0; j < axes.size(); j++) {
|
||||
dims.insert(dims.begin() + axes.getIntValue(j), 1);
|
||||
}
|
||||
|
||||
Mat out = input.reshape(0, dims);
|
||||
constBlobs.insert(std::make_pair(layerParams.name, out));
|
||||
continue;
|
||||
}
|
||||
|
||||
Mat out = input.reshape(0, dims);
|
||||
constBlobs.insert(std::make_pair(layerParams.name, out));
|
||||
continue;
|
||||
// Variable input.
|
||||
if (axes.size() != 1)
|
||||
CV_Error(Error::StsNotImplemented, "Multidimensional unsqueeze");
|
||||
|
||||
int dims[] = {1, -1};
|
||||
layerParams.type = "Reshape";
|
||||
layerParams.set("axis", axes.getIntValue(0));
|
||||
layerParams.set("num_axes", 1);
|
||||
layerParams.set("dim", DictValue::arrayInt(&dims[0], 2));
|
||||
}
|
||||
else if (layer_type == "Reshape")
|
||||
{
|
||||
@@ -707,6 +726,25 @@ void ONNXImporter::populateNet(Net dstNet)
|
||||
continue;
|
||||
}
|
||||
}
|
||||
else if (layer_type == "Upsample")
|
||||
{
|
||||
layerParams.type = "Resize";
|
||||
if (layerParams.has("scales"))
|
||||
{
|
||||
// Pytorch layer
|
||||
DictValue scales = layerParams.get("scales");
|
||||
CV_Assert(scales.size() == 4);
|
||||
layerParams.set("zoom_factor_y", scales.getIntValue(2));
|
||||
layerParams.set("zoom_factor_x", scales.getIntValue(3));
|
||||
}
|
||||
else
|
||||
{
|
||||
// Caffe2 layer
|
||||
replaceLayerParam(layerParams, "height_scale", "zoom_factor_y");
|
||||
replaceLayerParam(layerParams, "width_scale", "zoom_factor_x");
|
||||
}
|
||||
replaceLayerParam(layerParams, "mode", "interpolation");
|
||||
}
|
||||
else
|
||||
{
|
||||
for (int j = 0; j < node_proto.input_size(); j++) {
|
||||
|
||||
@@ -227,7 +227,7 @@ void InfEngineBackendNet::addLayer(InferenceEngine::Builder::Layer& layer)
|
||||
// By default, all the weights are connected to last ports ids.
|
||||
for (int i = 0; i < blobsIds.size(); ++i)
|
||||
{
|
||||
netBuilder.connect((size_t)blobsIds[i], {(size_t)id, portIds[i]});
|
||||
netBuilder.connect((size_t)blobsIds[i], {(size_t)id, (size_t)portIds[i]});
|
||||
}
|
||||
#endif
|
||||
}
|
||||
@@ -541,7 +541,6 @@ size_t InfEngineBackendNet::getBatchSize() const CV_NOEXCEPT
|
||||
return batchSize;
|
||||
}
|
||||
|
||||
#if INF_ENGINE_VER_MAJOR_GT(INF_ENGINE_RELEASE_2018R2)
|
||||
InferenceEngine::StatusCode InfEngineBackendNet::AddExtension(const InferenceEngine::IShapeInferExtensionPtr &extension, InferenceEngine::ResponseDesc *resp) CV_NOEXCEPT
|
||||
{
|
||||
CV_Error(Error::StsNotImplemented, "");
|
||||
@@ -553,7 +552,6 @@ InferenceEngine::StatusCode InfEngineBackendNet::reshape(const InferenceEngine::
|
||||
CV_Error(Error::StsNotImplemented, "");
|
||||
return InferenceEngine::StatusCode::OK;
|
||||
}
|
||||
#endif
|
||||
|
||||
void InfEngineBackendNet::init(int targetId)
|
||||
{
|
||||
|
||||
@@ -22,8 +22,6 @@
|
||||
//#pragma GCC diagnostic pop
|
||||
#endif
|
||||
|
||||
#define INF_ENGINE_RELEASE_2018R1 2018010000
|
||||
#define INF_ENGINE_RELEASE_2018R2 2018020000
|
||||
#define INF_ENGINE_RELEASE_2018R3 2018030000
|
||||
#define INF_ENGINE_RELEASE_2018R4 2018040000
|
||||
#define INF_ENGINE_RELEASE_2018R5 2018050000
|
||||
|
||||
@@ -250,7 +250,7 @@ TEST_P(DNNTestNetwork, OpenPose_pose_mpi_faster_4_stages)
|
||||
TEST_P(DNNTestNetwork, OpenFace)
|
||||
{
|
||||
#if defined(INF_ENGINE_RELEASE)
|
||||
#if (INF_ENGINE_RELEASE < 2018030000 || INF_ENGINE_RELEASE == 2018050000)
|
||||
#if INF_ENGINE_RELEASE == 2018050000
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("");
|
||||
#elif INF_ENGINE_RELEASE < 2018040000
|
||||
|
||||
@@ -294,29 +294,9 @@ public:
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
{
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
if (inp && ref && inp->size[0] != 1)
|
||||
{
|
||||
// Myriad plugin supports only batch size 1. Slice a single sample.
|
||||
if (inp->size[0] == ref->size[0])
|
||||
{
|
||||
std::vector<cv::Range> range(inp->dims, Range::all());
|
||||
range[0] = Range(0, 1);
|
||||
*inp = inp->operator()(range);
|
||||
|
||||
range = std::vector<cv::Range>(ref->dims, Range::all());
|
||||
range[0] = Range(0, 1);
|
||||
*ref = ref->operator()(range);
|
||||
}
|
||||
else
|
||||
throw SkipTestException("Myriad plugin supports only batch size 1");
|
||||
}
|
||||
#else
|
||||
if (inp && ref && inp->dims == 4 && ref->dims == 4 &&
|
||||
inp->size[0] != 1 && inp->size[0] != ref->size[0])
|
||||
throw SkipTestException("Inconsistent batch size of input and output blobs for Myriad plugin");
|
||||
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -93,11 +93,6 @@ TEST_P(Convolution, Accuracy)
|
||||
Backend backendId = get<0>(get<7>(GetParam()));
|
||||
Target targetId = get<1>(get<7>(GetParam()));
|
||||
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE && targetId == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("Test is enabled starts from OpenVINO 2018R3");
|
||||
#endif
|
||||
|
||||
bool skipCheck = false;
|
||||
|
||||
int sz[] = {outChannels, inChannels / group, kernel.height, kernel.width};
|
||||
@@ -232,8 +227,6 @@ TEST_P(LRN, Accuracy)
|
||||
std::string nrmType = get<4>(GetParam());
|
||||
Backend backendId = get<0>(get<5>(GetParam()));
|
||||
Target targetId = get<1>(get<5>(GetParam()));
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE)
|
||||
throw SkipTestException("");
|
||||
|
||||
LayerParams lp;
|
||||
lp.set("norm_region", nrmType);
|
||||
@@ -254,8 +247,8 @@ INSTANTIATE_TEST_CASE_P(Layer_Test_Halide, LRN, Combine(
|
||||
/*input ch,w,h*/ Values(Vec3i(6, 5, 8), Vec3i(7, 11, 6)),
|
||||
/*local size*/ Values(3, 5),
|
||||
Values(Vec3f(0.9f, 1.0f, 1.1f), Vec3f(0.9f, 1.1f, 1.0f),
|
||||
/*alpha, beta,*/ Vec3f(1.0f, 0.9f, 1.1f), Vec3f(1.0f, 1.1f, 0.9f),
|
||||
/*bias */ Vec3f(1.1f, 0.9f, 1.0f), Vec3f(1.1f, 1.0f, 0.9f)),
|
||||
/*alpha, beta, bias*/ Vec3f(1.0f, 0.9f, 1.1f), Vec3f(1.0f, 1.1f, 0.9f),
|
||||
Vec3f(1.1f, 0.9f, 1.0f), Vec3f(1.1f, 1.0f, 0.9f)),
|
||||
/*norm_by_size*/ Bool(),
|
||||
/*norm_type*/ Values("ACROSS_CHANNELS", "WITHIN_CHANNEL"),
|
||||
dnnBackendsAndTargetsWithHalide()
|
||||
|
||||
@@ -220,10 +220,6 @@ TEST(Layer_Test_Reshape, Accuracy)
|
||||
|
||||
TEST_P(Test_Caffe_layers, BatchNorm)
|
||||
{
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE)
|
||||
throw SkipTestException("Test is enabled starts from OpenVINO 2018R3");
|
||||
#endif
|
||||
testLayerUsingCaffeModels("layer_batch_norm", true);
|
||||
testLayerUsingCaffeModels("layer_batch_norm_local_stats", true, false);
|
||||
}
|
||||
@@ -741,10 +737,6 @@ INSTANTIATE_TEST_CASE_P(Layer_Test, Crop, Combine(
|
||||
// into the normalization area.
|
||||
TEST_P(Test_Caffe_layers, Average_pooling_kernel_area)
|
||||
{
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("Test is enabled starts from OpenVINO 2018R3");
|
||||
#endif
|
||||
LayerParams lp;
|
||||
lp.name = "testAvePool";
|
||||
lp.type = "Pooling";
|
||||
|
||||
@@ -72,6 +72,7 @@ TEST_P(Test_ONNX_layers, Deconvolution)
|
||||
{
|
||||
testONNXModels("deconvolution");
|
||||
testONNXModels("two_deconvolution");
|
||||
testONNXModels("deconvolution_group");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Dropout)
|
||||
@@ -140,6 +141,11 @@ TEST_P(Test_ONNX_layers, Padding)
|
||||
testONNXModels("padding");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Resize)
|
||||
{
|
||||
testONNXModels("resize_nearest");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, MultyInputs)
|
||||
{
|
||||
const String model = _tf("models/multy_inputs.onnx");
|
||||
@@ -169,6 +175,11 @@ TEST_P(Test_ONNX_layers, DynamicReshape)
|
||||
testONNXModels("dynamic_reshape");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Reshape)
|
||||
{
|
||||
testONNXModels("unsqueeze");
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/*nothing*/, Test_ONNX_layers, dnnBackendsAndTargets());
|
||||
|
||||
class Test_ONNX_nets : public Test_ONNX_layers {};
|
||||
|
||||
@@ -147,10 +147,6 @@ TEST_P(Test_TensorFlow_layers, eltwise)
|
||||
|
||||
TEST_P(Test_TensorFlow_layers, pad_and_concat)
|
||||
{
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("Test is enabled starts from OpenVINO 2018R3");
|
||||
#endif
|
||||
runTensorFlowNet("pad_and_concat");
|
||||
}
|
||||
|
||||
@@ -185,10 +181,6 @@ TEST_P(Test_TensorFlow_layers, pooling)
|
||||
// TODO: fix tests and replace to pooling
|
||||
TEST_P(Test_TensorFlow_layers, ave_pool_same)
|
||||
{
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("Test is enabled starts from OpenVINO 2018R3");
|
||||
#endif
|
||||
runTensorFlowNet("ave_pool_same");
|
||||
}
|
||||
|
||||
@@ -453,11 +445,6 @@ TEST_P(Test_TensorFlow_nets, opencv_face_detector_uint8)
|
||||
TEST_P(Test_TensorFlow_nets, EAST_text_detection)
|
||||
{
|
||||
checkBackend();
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("Test is enabled starts from OpenVINO 2018R3");
|
||||
#endif
|
||||
|
||||
std::string netPath = findDataFile("dnn/frozen_east_text_detection.pb", false);
|
||||
std::string imgPath = findDataFile("cv/ximgproc/sources/08.png", false);
|
||||
std::string refScoresPath = findDataFile("dnn/east_text_detection.scores.npy", false);
|
||||
@@ -516,17 +503,6 @@ TEST_P(Test_TensorFlow_layers, fp16_weights)
|
||||
runTensorFlowNet("fp16_max_pool_odd_valid", false, l1, lInf);
|
||||
runTensorFlowNet("fp16_max_pool_even", false, l1, lInf);
|
||||
runTensorFlowNet("fp16_padding_same", false, l1, lInf);
|
||||
}
|
||||
|
||||
// TODO: fix pad_and_concat and add this test case to fp16_weights
|
||||
TEST_P(Test_TensorFlow_layers, fp16_pad_and_concat)
|
||||
{
|
||||
const float l1 = 0.00071;
|
||||
const float lInf = 0.012;
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE < 2018030000
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("Test is enabled starts from OpenVINO 2018R3");
|
||||
#endif
|
||||
runTensorFlowNet("fp16_pad_and_concat", false, l1, lInf);
|
||||
}
|
||||
|
||||
|
||||
@@ -272,7 +272,7 @@ class Test_Torch_nets : public DNNTestLayer {};
|
||||
|
||||
TEST_P(Test_Torch_nets, OpenFace_accuracy)
|
||||
{
|
||||
#if defined(INF_ENGINE_RELEASE) && (INF_ENGINE_RELEASE < 2018030000 || INF_ENGINE_RELEASE == 2018050000)
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_RELEASE == 2018050000
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("");
|
||||
#endif
|
||||
|
||||
@@ -267,7 +267,7 @@ CV_IMPL void cvResizeWindow( const char* name, int width, int height)
|
||||
CVWindow *window = cvGetWindow(name);
|
||||
if(window && ![window autosize]) {
|
||||
height += [window contentView].sliderHeight;
|
||||
NSSize size = { width, height };
|
||||
NSSize size = { (CGFloat)width, (CGFloat)height };
|
||||
[window setContentSize:size];
|
||||
}
|
||||
[localpool drain];
|
||||
|
||||
@@ -11,7 +11,7 @@ typedef perf::TestBaseWithParam<Size_MatType_OutMatDepth_t> Size_MatType_OutMatD
|
||||
PERF_TEST_P(Size_MatType_OutMatDepth, integral,
|
||||
testing::Combine(
|
||||
testing::Values(TYPICAL_MAT_SIZES),
|
||||
testing::Values(CV_8UC1, CV_8UC3, CV_8UC4),
|
||||
testing::Values(CV_8UC1, CV_8UC2, CV_8UC3, CV_8UC4),
|
||||
testing::Values(CV_32S, CV_32F, CV_64F)
|
||||
)
|
||||
)
|
||||
@@ -32,9 +32,9 @@ PERF_TEST_P(Size_MatType_OutMatDepth, integral,
|
||||
|
||||
PERF_TEST_P(Size_MatType_OutMatDepth, integral_sqsum,
|
||||
testing::Combine(
|
||||
testing::Values(::perf::szVGA, ::perf::sz1080p),
|
||||
testing::Values(CV_8UC1, CV_8UC4),
|
||||
testing::Values(CV_32S, CV_32F)
|
||||
testing::Values(TYPICAL_MAT_SIZES),
|
||||
testing::Values(CV_8UC1, CV_8UC2, CV_8UC3, CV_8UC4),
|
||||
testing::Values(CV_32S, CV_32F, CV_64F)
|
||||
)
|
||||
)
|
||||
{
|
||||
@@ -57,9 +57,9 @@ PERF_TEST_P(Size_MatType_OutMatDepth, integral_sqsum,
|
||||
|
||||
PERF_TEST_P( Size_MatType_OutMatDepth, integral_sqsum_tilted,
|
||||
testing::Combine(
|
||||
testing::Values( ::perf::szVGA, ::perf::szODD , ::perf::sz1080p ),
|
||||
testing::Values( CV_8UC1, CV_8UC4 ),
|
||||
testing::Values( CV_32S, CV_32F )
|
||||
testing::Values(TYPICAL_MAT_SIZES),
|
||||
testing::Values( CV_8UC1, CV_8UC2, CV_8UC3, CV_8UC4 ),
|
||||
testing::Values( CV_32S, CV_32F, CV_64F )
|
||||
)
|
||||
)
|
||||
{
|
||||
|
||||
+101
-167
@@ -340,155 +340,6 @@ static void hlineResizeCn(ET* src, int cn, int *ofst, FT* m, FT* dst, int dst_mi
|
||||
hline<ET, FT, n, mulall, cncnt>::ResizeCn(src, cn, ofst, m, dst, dst_min, dst_max, dst_width);
|
||||
};
|
||||
|
||||
#if CV_SIMD512
|
||||
inline void v_load_indexed1(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
v_expand(v_reinterpret_as_u8(v_uint16(
|
||||
*((uint16_t*)(src + ofst[ 0])), *((uint16_t*)(src + ofst[ 1])), *((uint16_t*)(src + ofst[ 2])), *((uint16_t*)(src + ofst[ 3])),
|
||||
*((uint16_t*)(src + ofst[ 4])), *((uint16_t*)(src + ofst[ 5])), *((uint16_t*)(src + ofst[ 6])), *((uint16_t*)(src + ofst[ 7])),
|
||||
*((uint16_t*)(src + ofst[ 8])), *((uint16_t*)(src + ofst[ 9])), *((uint16_t*)(src + ofst[10])), *((uint16_t*)(src + ofst[11])),
|
||||
*((uint16_t*)(src + ofst[12])), *((uint16_t*)(src + ofst[13])), *((uint16_t*)(src + ofst[14])), *((uint16_t*)(src + ofst[15])),
|
||||
*((uint16_t*)(src + ofst[16])), *((uint16_t*)(src + ofst[17])), *((uint16_t*)(src + ofst[14])), *((uint16_t*)(src + ofst[15])),
|
||||
*((uint16_t*)(src + ofst[20])), *((uint16_t*)(src + ofst[21])), *((uint16_t*)(src + ofst[14])), *((uint16_t*)(src + ofst[15])),
|
||||
*((uint16_t*)(src + ofst[24])), *((uint16_t*)(src + ofst[25])), *((uint16_t*)(src + ofst[14])), *((uint16_t*)(src + ofst[15])),
|
||||
*((uint16_t*)(src + ofst[28])), *((uint16_t*)(src + ofst[29])), *((uint16_t*)(src + ofst[14])), *((uint16_t*)(src + ofst[15])))),
|
||||
v_src0, v_src1);
|
||||
}
|
||||
inline void v_load_indexed2(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
v_expand(v_reinterpret_as_u8(v_uint32(
|
||||
*((uint32_t*)(src + 2 * ofst[ 0])), *((uint32_t*)(src + 2 * ofst[ 1])), *((uint32_t*)(src + 2 * ofst[ 2])), *((uint32_t*)(src + 2 * ofst[ 3])),
|
||||
*((uint32_t*)(src + 2 * ofst[ 4])), *((uint32_t*)(src + 2 * ofst[ 5])), *((uint32_t*)(src + 2 * ofst[ 6])), *((uint32_t*)(src + 2 * ofst[ 7])),
|
||||
*((uint32_t*)(src + 2 * ofst[ 8])), *((uint32_t*)(src + 2 * ofst[ 9])), *((uint32_t*)(src + 2 * ofst[10])), *((uint32_t*)(src + 2 * ofst[11])),
|
||||
*((uint32_t*)(src + 2 * ofst[12])), *((uint32_t*)(src + 2 * ofst[13])), *((uint32_t*)(src + 2 * ofst[14])), *((uint32_t*)(src + 2 * ofst[15])))),
|
||||
v_src0, v_src1);
|
||||
v_uint32 v_tmp0, v_tmp1, v_tmp2, v_tmp3;
|
||||
v_zip(v_reinterpret_as_u32(v_src0), v_reinterpret_as_u32(v_src1), v_tmp2, v_tmp3);
|
||||
v_zip(v_tmp2, v_tmp3, v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_tmp2, v_tmp3);
|
||||
v_zip(v_tmp2, v_tmp3, v_tmp0, v_tmp1);
|
||||
v_zip(v_reinterpret_as_u16(v_tmp0), v_reinterpret_as_u16(v_tmp1), v_src0, v_src1);
|
||||
}
|
||||
inline void v_load_indexed4(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
v_expand(v_reinterpret_as_u8(v_uint64(
|
||||
*((uint64_t*)(src + 4 * ofst[0])), *((uint64_t*)(src + 4 * ofst[1])), *((uint64_t*)(src + 4 * ofst[2])), *((uint64_t*)(src + 4 * ofst[3])),
|
||||
*((uint64_t*)(src + 4 * ofst[4])), *((uint64_t*)(src + 4 * ofst[5])), *((uint64_t*)(src + 4 * ofst[6])), *((uint64_t*)(src + 4 * ofst[7])))),
|
||||
v_src0, v_src1);
|
||||
v_uint64 v_tmp0, v_tmp1, v_tmp2, v_tmp3;
|
||||
v_zip(v_reinterpret_as_u64(v_src0), v_reinterpret_as_u64(v_src1), v_tmp2, v_tmp3);
|
||||
v_zip(v_tmp2, v_tmp3, v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_tmp2, v_tmp3);
|
||||
v_zip(v_reinterpret_as_u16(v_tmp2), v_reinterpret_as_u16(v_tmp3), v_src0, v_src1);
|
||||
}
|
||||
inline void v_load_indexed_deinterleave(uint16_t* src, int *ofst, v_uint32 &v_src0, v_uint32 &v_src1)
|
||||
{
|
||||
v_expand(v_reinterpret_as_u16(v_uint32(
|
||||
*((uint32_t*)(src + ofst[ 0])), *((uint32_t*)(src + ofst[ 1])), *((uint32_t*)(src + ofst[ 2])), *((uint32_t*)(src + ofst[ 3])),
|
||||
*((uint32_t*)(src + ofst[ 4])), *((uint32_t*)(src + ofst[ 5])), *((uint32_t*)(src + ofst[ 6])), *((uint32_t*)(src + ofst[ 7])),
|
||||
*((uint32_t*)(src + ofst[ 8])), *((uint32_t*)(src + ofst[ 9])), *((uint32_t*)(src + ofst[10])), *((uint32_t*)(src + ofst[11])),
|
||||
*((uint32_t*)(src + ofst[12])), *((uint32_t*)(src + ofst[13])), *((uint32_t*)(src + ofst[14])), *((uint32_t*)(src + ofst[15])))),
|
||||
v_src0, v_src1);
|
||||
v_uint32 v_tmp0, v_tmp1;
|
||||
v_zip(v_src0, v_src1, v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_src0, v_src1);
|
||||
v_zip(v_src0, v_src1, v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_src0, v_src1);
|
||||
}
|
||||
#elif CV_SIMD256
|
||||
inline void v_load_indexed1(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
v_expand(v_reinterpret_as_u8(v_uint16(
|
||||
*((uint16_t*)(src + ofst[ 0])), *((uint16_t*)(src + ofst[ 1])), *((uint16_t*)(src + ofst[ 2])), *((uint16_t*)(src + ofst[ 3])),
|
||||
*((uint16_t*)(src + ofst[ 4])), *((uint16_t*)(src + ofst[ 5])), *((uint16_t*)(src + ofst[ 6])), *((uint16_t*)(src + ofst[ 7])),
|
||||
*((uint16_t*)(src + ofst[ 8])), *((uint16_t*)(src + ofst[ 9])), *((uint16_t*)(src + ofst[10])), *((uint16_t*)(src + ofst[11])),
|
||||
*((uint16_t*)(src + ofst[12])), *((uint16_t*)(src + ofst[13])), *((uint16_t*)(src + ofst[14])), *((uint16_t*)(src + ofst[15])))),
|
||||
v_src0, v_src1);
|
||||
}
|
||||
inline void v_load_indexed2(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
v_expand(v_reinterpret_as_u8(v_uint32(
|
||||
*((uint32_t*)(src + 2 * ofst[0])), *((uint32_t*)(src + 2 * ofst[1])), *((uint32_t*)(src + 2 * ofst[2])), *((uint32_t*)(src + 2 * ofst[3])),
|
||||
*((uint32_t*)(src + 2 * ofst[4])), *((uint32_t*)(src + 2 * ofst[5])), *((uint32_t*)(src + 2 * ofst[6])), *((uint32_t*)(src + 2 * ofst[7])))),
|
||||
v_src0, v_src1);
|
||||
v_uint32 v_tmp0, v_tmp1, v_tmp2, v_tmp3;
|
||||
v_zip(v_reinterpret_as_u32(v_src0), v_reinterpret_as_u32(v_src1), v_tmp2, v_tmp3);
|
||||
v_zip(v_tmp2, v_tmp3, v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_tmp2, v_tmp3);
|
||||
v_zip(v_reinterpret_as_u16(v_tmp2), v_reinterpret_as_u16(v_tmp3), v_src0, v_src1);
|
||||
}
|
||||
inline void v_load_indexed4(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
v_expand(v_reinterpret_as_u8(v_uint64(
|
||||
*((uint64_t*)(src + 4 * ofst[0])), *((uint64_t*)(src + 4 * ofst[1])), *((uint64_t*)(src + 4 * ofst[2])), *((uint64_t*)(src + 4 * ofst[3])))),
|
||||
v_src0, v_src1);
|
||||
v_uint64 v_tmp0, v_tmp1, v_tmp2, v_tmp3;
|
||||
v_zip(v_reinterpret_as_u64(v_src0), v_reinterpret_as_u64(v_src1), v_tmp2, v_tmp3);
|
||||
v_zip(v_tmp2, v_tmp3, v_tmp0, v_tmp1);
|
||||
v_zip(v_reinterpret_as_u16(v_tmp0), v_reinterpret_as_u16(v_tmp1), v_src0, v_src1);
|
||||
}
|
||||
inline void v_load_indexed_deinterleave(uint16_t* src, int *ofst, v_uint32 &v_src0, v_uint32 &v_src1)
|
||||
{
|
||||
v_uint32 v_tmp0, v_tmp1;
|
||||
v_expand(v_reinterpret_as_u16(v_uint32(
|
||||
*((uint32_t*)(src + ofst[0])), *((uint32_t*)(src + ofst[1])), *((uint32_t*)(src + ofst[2])), *((uint32_t*)(src + ofst[3])),
|
||||
*((uint32_t*)(src + ofst[4])), *((uint32_t*)(src + ofst[5])), *((uint32_t*)(src + ofst[6])), *((uint32_t*)(src + ofst[7])))),
|
||||
v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_src0, v_src1);
|
||||
v_zip(v_src0, v_src1, v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_src0, v_src1);
|
||||
}
|
||||
#elif CV_SIMD128
|
||||
inline void v_load_indexed1(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
uint16_t buf[8];
|
||||
buf[0] = *((uint16_t*)(src + ofst[0]));
|
||||
buf[1] = *((uint16_t*)(src + ofst[1]));
|
||||
buf[2] = *((uint16_t*)(src + ofst[2]));
|
||||
buf[3] = *((uint16_t*)(src + ofst[3]));
|
||||
buf[4] = *((uint16_t*)(src + ofst[4]));
|
||||
buf[5] = *((uint16_t*)(src + ofst[5]));
|
||||
buf[6] = *((uint16_t*)(src + ofst[6]));
|
||||
buf[7] = *((uint16_t*)(src + ofst[7]));
|
||||
v_src0 = vx_load_expand((uint8_t*)buf);
|
||||
v_src1 = vx_load_expand((uint8_t*)buf + 8);
|
||||
}
|
||||
inline void v_load_indexed2(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
uint32_t buf[4];
|
||||
buf[0] = *((uint32_t*)(src + 2 * ofst[0]));
|
||||
buf[1] = *((uint32_t*)(src + 2 * ofst[1]));
|
||||
buf[2] = *((uint32_t*)(src + 2 * ofst[2]));
|
||||
buf[3] = *((uint32_t*)(src + 2 * ofst[3]));
|
||||
v_uint32 v_tmp0, v_tmp1, v_tmp2, v_tmp3;
|
||||
v_tmp0 = v_reinterpret_as_u32(vx_load_expand((uint8_t*)buf));
|
||||
v_tmp1 = v_reinterpret_as_u32(vx_load_expand((uint8_t*)buf + 8));
|
||||
v_zip(v_tmp0, v_tmp1, v_tmp2, v_tmp3);
|
||||
v_zip(v_tmp2, v_tmp3, v_tmp0, v_tmp1);
|
||||
v_zip(v_reinterpret_as_u16(v_tmp0), v_reinterpret_as_u16(v_tmp1), v_src0, v_src1);
|
||||
}
|
||||
inline void v_load_indexed4(uint8_t* src, int *ofst, v_uint16 &v_src0, v_uint16 &v_src1)
|
||||
{
|
||||
v_uint16 v_tmp0, v_tmp1;
|
||||
v_src0 = vx_load_expand(src + 4 * ofst[0]);
|
||||
v_src1 = vx_load_expand(src + 4 * ofst[1]);
|
||||
v_recombine(v_src0, v_src1, v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_src0, v_src1);
|
||||
}
|
||||
inline void v_load_indexed_deinterleave(uint16_t* src, int *ofst, v_uint32 &v_src0, v_uint32 &v_src1)
|
||||
{
|
||||
uint32_t buf[4];
|
||||
buf[0] = *((uint32_t*)(src + ofst[0]));
|
||||
buf[1] = *((uint32_t*)(src + ofst[1]));
|
||||
buf[2] = *((uint32_t*)(src + ofst[2]));
|
||||
buf[3] = *((uint32_t*)(src + ofst[3]));
|
||||
v_src0 = vx_load_expand((uint16_t*)buf);
|
||||
v_src1 = vx_load_expand((uint16_t*)buf + 4);
|
||||
v_uint32 v_tmp0, v_tmp1;
|
||||
v_zip(v_src0, v_src1, v_tmp0, v_tmp1);
|
||||
v_zip(v_tmp0, v_tmp1, v_src0, v_src1);
|
||||
}
|
||||
#endif
|
||||
template <>
|
||||
void hlineResizeCn<uint8_t, ufixedpoint16, 2, true, 1>(uint8_t* src, int, int *ofst, ufixedpoint16* m, ufixedpoint16* dst, int dst_min, int dst_max, int dst_width)
|
||||
{
|
||||
@@ -507,16 +358,23 @@ void hlineResizeCn<uint8_t, ufixedpoint16, 2, true, 1>(uint8_t* src, int, int *o
|
||||
*(dst++) = src_0;
|
||||
}
|
||||
#if CV_SIMD
|
||||
for (; i <= dst_max - VECSZ; i += VECSZ, m += 2*VECSZ, dst += VECSZ)
|
||||
for (; i <= dst_max - 2*VECSZ; i += 2*VECSZ, m += 4*VECSZ, dst += 2*VECSZ)
|
||||
{
|
||||
v_uint16 v_src0, v_src1;
|
||||
v_load_indexed1(src, ofst + i, v_src0, v_src1);
|
||||
|
||||
v_int16 v_mul0 = vx_load((int16_t*)m);
|
||||
v_int16 v_mul1 = vx_load((int16_t*)m + VECSZ);
|
||||
v_uint32 v_res0 = v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src0), v_mul0));
|
||||
v_uint32 v_res1 = v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src1), v_mul1));
|
||||
v_store((uint16_t*)dst, v_pack(v_res0, v_res1));
|
||||
v_expand(vx_lut_pairs(src, ofst + i), v_src0, v_src1);
|
||||
v_store((uint16_t*)dst , v_pack(v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src0), vx_load((int16_t*)m))),
|
||||
v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src1), vx_load((int16_t*)m + VECSZ)))));
|
||||
v_expand(vx_lut_pairs(src, ofst + i + VECSZ), v_src0, v_src1);
|
||||
v_store((uint16_t*)dst+VECSZ, v_pack(v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src0), vx_load((int16_t*)m + 2*VECSZ))),
|
||||
v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src1), vx_load((int16_t*)m + 3*VECSZ)))));
|
||||
}
|
||||
if (i <= dst_max - VECSZ)
|
||||
{
|
||||
v_uint16 v_src0, v_src1;
|
||||
v_expand(vx_lut_pairs(src, ofst + i), v_src0, v_src1);
|
||||
v_store((uint16_t*)dst, v_pack(v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src0), vx_load((int16_t*)m))),
|
||||
v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src1), vx_load((int16_t*)m + VECSZ)))));
|
||||
i += VECSZ; m += 2*VECSZ; dst += VECSZ;
|
||||
}
|
||||
#endif
|
||||
for (; i < dst_max; i += 1, m += 2)
|
||||
@@ -564,7 +422,7 @@ void hlineResizeCn<uint8_t, ufixedpoint16, 2, true, 2>(uint8_t* src, int, int *o
|
||||
for (; i <= dst_max - VECSZ/2; i += VECSZ/2, m += VECSZ, dst += VECSZ)
|
||||
{
|
||||
v_uint16 v_src0, v_src1;
|
||||
v_load_indexed2(src, ofst + i, v_src0, v_src1);
|
||||
v_expand(v_interleave_pairs(v_reinterpret_as_u8(vx_lut_pairs((uint16_t*)src, ofst + i))), v_src0, v_src1);
|
||||
|
||||
v_uint32 v_mul = vx_load((uint32_t*)m);//AaBbCcDd
|
||||
v_uint32 v_zip0, v_zip1;
|
||||
@@ -595,6 +453,81 @@ void hlineResizeCn<uint8_t, ufixedpoint16, 2, true, 2>(uint8_t* src, int, int *o
|
||||
}
|
||||
}
|
||||
template <>
|
||||
void hlineResizeCn<uint8_t, ufixedpoint16, 2, true, 3>(uint8_t* src, int, int *ofst, ufixedpoint16* m, ufixedpoint16* dst, int dst_min, int dst_max, int dst_width)
|
||||
{
|
||||
int i = 0;
|
||||
union {
|
||||
uint64_t q;
|
||||
uint16_t w[4];
|
||||
} srccn;
|
||||
((ufixedpoint16*)(srccn.w))[0] = src[0];
|
||||
((ufixedpoint16*)(srccn.w))[1] = src[1];
|
||||
((ufixedpoint16*)(srccn.w))[2] = src[2];
|
||||
((ufixedpoint16*)(srccn.w))[3] = 0;
|
||||
#if CV_SIMD
|
||||
const int VECSZ = v_uint16::nlanes;
|
||||
v_uint16 v_srccn = v_pack_triplets(v_reinterpret_as_u16(vx_setall_u64(srccn.q)));
|
||||
for (; i <= dst_min - (VECSZ+2)/3; i += VECSZ/4, m += VECSZ/2, dst += 3*VECSZ/4) // Points that fall left from src image so became equal to leftmost src point
|
||||
{
|
||||
v_store((uint16_t*)dst, v_srccn);
|
||||
}
|
||||
#endif
|
||||
for (; i < dst_min; i++, m += 2)
|
||||
{
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[0];
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[1];
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[2];
|
||||
}
|
||||
#if CV_SIMD
|
||||
CV_DECL_ALIGNED(CV_SIMD_WIDTH) int ofst3[VECSZ/2];
|
||||
for (; i <= dst_max - (3*VECSZ/4 + (VECSZ+2)/3); i += VECSZ/2, m += VECSZ, dst += 3*VECSZ/2)
|
||||
{
|
||||
v_store(ofst3, vx_load(ofst + i) * vx_setall_s32(3));
|
||||
v_uint8 v_src01, v_src23;
|
||||
v_uint16 v_src0, v_src1, v_src2, v_src3;
|
||||
v_zip(vx_lut_quads(src, ofst3), vx_lut_quads(src+3, ofst3), v_src01, v_src23);
|
||||
v_expand(v_src01, v_src0, v_src1);
|
||||
v_expand(v_src23, v_src2, v_src3);
|
||||
|
||||
v_uint32 v_mul0, v_mul1, v_mul2, v_mul3, v_tmp;
|
||||
v_mul0 = vx_load((uint32_t*)m);//AaBbCcDd
|
||||
v_zip(v_mul0, v_mul0, v_mul3, v_tmp );//AaAaBbBb CcCcDdDd
|
||||
v_zip(v_mul3, v_mul3, v_mul0, v_mul1);//AaAaAaAa BbBbBbBb
|
||||
v_zip(v_tmp , v_tmp , v_mul2, v_mul3);//CcCcCcCc DdDdDdDd
|
||||
|
||||
v_uint32 v_res0 = v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src0), v_reinterpret_as_s16(v_mul0)));
|
||||
v_uint32 v_res1 = v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src1), v_reinterpret_as_s16(v_mul1)));
|
||||
v_uint32 v_res2 = v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src2), v_reinterpret_as_s16(v_mul2)));
|
||||
v_uint32 v_res3 = v_reinterpret_as_u32(v_dotprod(v_reinterpret_as_s16(v_src3), v_reinterpret_as_s16(v_mul3)));
|
||||
v_store((uint16_t*)dst , v_pack_triplets(v_pack(v_res0, v_res1)));
|
||||
v_store((uint16_t*)dst + 3*VECSZ/4, v_pack_triplets(v_pack(v_res2, v_res3)));
|
||||
}
|
||||
#endif
|
||||
for (; i < dst_max; i += 1, m += 2)
|
||||
{
|
||||
uint8_t* px = src + 3 * ofst[i];
|
||||
*(dst++) = m[0] * px[0] + m[1] * px[3];
|
||||
*(dst++) = m[0] * px[1] + m[1] * px[4];
|
||||
*(dst++) = m[0] * px[2] + m[1] * px[5];
|
||||
}
|
||||
((ufixedpoint16*)(srccn.w))[0] = (src + 3*ofst[dst_width - 1])[0];
|
||||
((ufixedpoint16*)(srccn.w))[1] = (src + 3*ofst[dst_width - 1])[1];
|
||||
((ufixedpoint16*)(srccn.w))[2] = (src + 3*ofst[dst_width - 1])[2];
|
||||
#if CV_SIMD
|
||||
v_srccn = v_pack_triplets(v_reinterpret_as_u16(vx_setall_u64(srccn.q)));
|
||||
for (; i <= dst_width - (VECSZ+2)/3; i += VECSZ/4, dst += 3*VECSZ/4) // Points that fall right from src image so became equal to rightmost src point
|
||||
{
|
||||
v_store((uint16_t*)dst, v_srccn);
|
||||
}
|
||||
#endif
|
||||
for (; i < dst_width; i++)
|
||||
{
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[0];
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[1];
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[2];
|
||||
}
|
||||
}
|
||||
template <>
|
||||
void hlineResizeCn<uint8_t, ufixedpoint16, 2, true, 4>(uint8_t* src, int, int *ofst, ufixedpoint16* m, ufixedpoint16* dst, int dst_min, int dst_max, int dst_width)
|
||||
{
|
||||
int i = 0;
|
||||
@@ -614,20 +547,19 @@ void hlineResizeCn<uint8_t, ufixedpoint16, 2, true, 4>(uint8_t* src, int, int *o
|
||||
v_store((uint16_t*)dst, v_srccn);
|
||||
}
|
||||
#endif
|
||||
if (i < dst_min) // Points that fall left from src image so became equal to leftmost src point
|
||||
for (; i < dst_min; i++, m += 2)
|
||||
{
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[0];
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[1];
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[2];
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[3];
|
||||
i++; m += 2;
|
||||
}
|
||||
#if CV_SIMD
|
||||
for (; i <= dst_max - VECSZ/2; i += VECSZ/2, m += VECSZ, dst += 2*VECSZ)
|
||||
{
|
||||
v_uint16 v_src0, v_src1, v_src2, v_src3;
|
||||
v_load_indexed4(src, ofst + i, v_src0, v_src1);
|
||||
v_load_indexed4(src, ofst + i + VECSZ/4, v_src2, v_src3);
|
||||
v_expand(v_interleave_quads(v_reinterpret_as_u8(vx_lut_pairs((uint32_t*)src, ofst + i))), v_src0, v_src1);
|
||||
v_expand(v_interleave_quads(v_reinterpret_as_u8(vx_lut_pairs((uint32_t*)src, ofst + i + VECSZ/4))), v_src2, v_src3);
|
||||
|
||||
v_uint32 v_mul0, v_mul1, v_mul2, v_mul3, v_tmp;
|
||||
v_mul0 = vx_load((uint32_t*)m);//AaBbCcDd
|
||||
@@ -660,7 +592,7 @@ void hlineResizeCn<uint8_t, ufixedpoint16, 2, true, 4>(uint8_t* src, int, int *o
|
||||
v_store((uint16_t*)dst, v_srccn);
|
||||
}
|
||||
#endif
|
||||
if (i < dst_width)
|
||||
for (; i < dst_width; i++)
|
||||
{
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[0];
|
||||
*(dst++) = ((ufixedpoint16*)(srccn.w))[1];
|
||||
@@ -689,10 +621,12 @@ void hlineResizeCn<uint16_t, ufixedpoint32, 2, true, 1>(uint16_t* src, int, int
|
||||
for (; i <= dst_max - VECSZ; i += VECSZ, m += 2*VECSZ, dst += VECSZ)
|
||||
{
|
||||
v_uint32 v_src0, v_src1;
|
||||
v_load_indexed_deinterleave(src, ofst + i, v_src0, v_src1);
|
||||
v_uint32 v_mul0, v_mul1;
|
||||
v_load_deinterleave((uint32_t*)m, v_mul0, v_mul1);
|
||||
v_store((uint32_t*)dst, v_src0 * v_mul0 + v_src1 * v_mul1);//abcd
|
||||
v_expand(vx_lut_pairs(src, ofst + i), v_src0, v_src1);
|
||||
|
||||
v_uint64 v_res0 = v_reinterpret_as_u64(v_src0 * vx_load((uint32_t*)m));
|
||||
v_uint64 v_res1 = v_reinterpret_as_u64(v_src1 * vx_load((uint32_t*)m + VECSZ));
|
||||
v_store((uint32_t*)dst, v_pack((v_res0 & vx_setall_u64(0xFFFFFFFF)) + (v_res0 >> 32),
|
||||
(v_res1 & vx_setall_u64(0xFFFFFFFF)) + (v_res1 >> 32)));
|
||||
}
|
||||
#endif
|
||||
for (; i < dst_max; i += 1, m += 2)
|
||||
|
||||
@@ -13,33 +13,34 @@ namespace { // Anonymous namespace to avoid exposing the implementation classes
|
||||
// NOTE: Look at the bottom of the file for the entry-point function for external callers
|
||||
//
|
||||
|
||||
// At the moment only 3 channel support untilted is supported
|
||||
// More channel support coming soon.
|
||||
// TODO: Add support for sqsum and 1,2, and 4 channels
|
||||
class IntegralCalculator_3Channel {
|
||||
// TODO: Add support for 1 channel input (WIP: currently hitting hardware glassjaw)
|
||||
template<size_t num_channels> class IntegralCalculator;
|
||||
|
||||
template<size_t num_channels>
|
||||
class IntegralCalculator {
|
||||
public:
|
||||
IntegralCalculator_3Channel() {};
|
||||
IntegralCalculator() {};
|
||||
|
||||
|
||||
void calculate_integral_avx512(const uchar *src, size_t _srcstep,
|
||||
double *sum, size_t _sumstep,
|
||||
double *sqsum, size_t _sqsumstep,
|
||||
int width, int height, int cn)
|
||||
int width, int height)
|
||||
{
|
||||
const int srcstep = (int)(_srcstep/sizeof(uchar));
|
||||
const int sumstep = (int)(_sumstep/sizeof(double));
|
||||
const int sqsumstep = (int)(_sqsumstep/sizeof(double));
|
||||
const int ops_per_line = width * cn;
|
||||
const int ops_per_line = width * num_channels;
|
||||
|
||||
// Clear the first line of the sum as per spec (see integral documentation)
|
||||
// Also adjust the index of sum and sqsum to be at the real 0th element
|
||||
// and not point to the border pixel so it stays in sync with the src pointer
|
||||
memset( sum, 0, (ops_per_line+cn)*sizeof(double));
|
||||
sum += cn;
|
||||
memset( sum, 0, (ops_per_line+num_channels)*sizeof(double));
|
||||
sum += num_channels;
|
||||
|
||||
if (sqsum) {
|
||||
memset( sqsum, 0, (ops_per_line+cn)*sizeof(double));
|
||||
sqsum += cn;
|
||||
memset( sqsum, 0, (ops_per_line+num_channels)*sizeof(double));
|
||||
sqsum += num_channels;
|
||||
}
|
||||
|
||||
// Now calculate the integral over the whole image one line at a time
|
||||
@@ -50,22 +51,22 @@ public:
|
||||
double * sqsum_above = (sqsum) ? &sqsum[y*sqsumstep] : NULL;
|
||||
double * sqsum_line = (sqsum) ? &sqsum_above[sqsumstep] : NULL;
|
||||
|
||||
integral_line_3channel_avx512(src_line, sum_line, sum_above, sqsum_line, sqsum_above, ops_per_line);
|
||||
calculate_integral_for_line(src_line, sum_line, sum_above, sqsum_line, sqsum_above, ops_per_line);
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
static inline
|
||||
void integral_line_3channel_avx512(const uchar *srcs,
|
||||
double *sums, double *sums_above,
|
||||
double *sqsums, double *sqsums_above,
|
||||
int num_ops_in_line)
|
||||
static CV_ALWAYS_INLINE
|
||||
void calculate_integral_for_line(const uchar *srcs,
|
||||
double *sums, double *sums_above,
|
||||
double *sqsums, double *sqsums_above,
|
||||
int num_ops_in_line)
|
||||
{
|
||||
__m512i sum_accumulator = _mm512_setzero_si512(); // holds rolling sums for the line
|
||||
__m512i sqsum_accumulator = _mm512_setzero_si512(); // holds rolling sqsums for the line
|
||||
|
||||
// The first element on each line must be zeroes as per spec (see integral documentation)
|
||||
set_border_pixel_value(sums, sqsums);
|
||||
zero_out_border_pixel(sums, sqsums);
|
||||
|
||||
// Do all 64 byte chunk operations then do the last bits that don't fit in a 64 byte chunk
|
||||
aligned_integral( srcs, sums, sums_above, sqsums, sqsums_above, sum_accumulator, sqsum_accumulator, num_ops_in_line);
|
||||
@@ -74,20 +75,18 @@ public:
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
void set_border_pixel_value(double *sums, double *sqsums)
|
||||
static CV_ALWAYS_INLINE
|
||||
void zero_out_border_pixel(double *sums, double *sqsums)
|
||||
{
|
||||
// Sets the border pixel value to 0s.
|
||||
// Note the hard coded -3 and the 0x7 mask is because we only support 3 channel right now
|
||||
__m512i zeroes = _mm512_setzero_si512();
|
||||
|
||||
_mm512_mask_storeu_epi64(&sums[-3], 0x7, zeroes);
|
||||
// Note the negative index is because the sums/sqsums pointers point to the first real pixel
|
||||
// after the border pixel so we have to look backwards
|
||||
_mm512_mask_storeu_epi64(&sums[-num_channels], (1<<num_channels)-1, _mm512_setzero_si512());
|
||||
if (sqsums)
|
||||
_mm512_mask_storeu_epi64(&sqsums[-3], 0x7, zeroes);
|
||||
_mm512_mask_storeu_epi64(&sqsums[-num_channels], (1<<num_channels)-1, _mm512_setzero_si512());
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
static CV_ALWAYS_INLINE
|
||||
void aligned_integral(const uchar *&srcs,
|
||||
double *&sums, double *&sums_above,
|
||||
double *&sqsum, double *&sqsum_above,
|
||||
@@ -110,7 +109,7 @@ public:
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
static CV_ALWAYS_INLINE
|
||||
void post_aligned_integral(const uchar *srcs,
|
||||
const double *sums, const double *sums_above,
|
||||
const double *sqsum, const double *sqsum_above,
|
||||
@@ -132,50 +131,54 @@ public:
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
static CV_ALWAYS_INLINE
|
||||
void integral_64_operations_avx512(const __m512i *srcs,
|
||||
__m512i *sums, const __m512i *sums_above,
|
||||
__m512i *sqsums, const __m512i *sqsums_above,
|
||||
__mmask64 data_mask,
|
||||
__m512i &sum_accumulator, __m512i &sqsum_accumulator)
|
||||
{
|
||||
__m512i src_64byte_chunk = read_64_bytes(srcs, data_mask);
|
||||
__m512i src_64byte_chunk = read_64_bytes(srcs, data_mask);
|
||||
|
||||
for(int num_16byte_chunks=0; num_16byte_chunks<4; num_16byte_chunks++) {
|
||||
__m128i src_16bytes = _mm512_extracti64x2_epi64(src_64byte_chunk, 0x0); // Get lower 16 bytes of data
|
||||
while (data_mask) {
|
||||
__m128i src_16bytes = extract_lower_16bytes(src_64byte_chunk);
|
||||
|
||||
for (int num_8byte_chunks = 0; num_8byte_chunks < 2; num_8byte_chunks++) {
|
||||
__m512i src_longs_lo = convert_lower_8bytes_to_longs(src_16bytes);
|
||||
__m512i src_longs_hi = convert_lower_8bytes_to_longs(shift_right_8_bytes(src_16bytes));
|
||||
|
||||
__m512i src_longs = convert_lower_8bytes_to_longs(src_16bytes);
|
||||
// Calculate integral for the sum on the 8 lanes at a time
|
||||
integral_8_operations(src_longs_lo, sums_above, data_mask, sums, sum_accumulator);
|
||||
integral_8_operations(src_longs_hi, sums_above+1, data_mask>>8, sums+1, sum_accumulator);
|
||||
|
||||
// Calculate integral for the sum on the 8 entries
|
||||
integral_8_operations(src_longs, sums_above, data_mask, sums, sum_accumulator);
|
||||
sums++; sums_above++;
|
||||
|
||||
if (sqsums){ // Calculate integral for the sum on the 8 entries
|
||||
__m512i squared_source = _mm512_mullo_epi64(src_longs, src_longs);
|
||||
|
||||
integral_8_operations(squared_source, sqsums_above, data_mask, sqsums, sqsum_accumulator);
|
||||
sqsums++; sqsums_above++;
|
||||
}
|
||||
|
||||
// Prepare for next iteration of inner loop
|
||||
// shift source to align next 8 bytes to lane 0 and shift the mask
|
||||
src_16bytes = shift_right_8_bytes(src_16bytes);
|
||||
data_mask = data_mask >> 8;
|
||||
if (sqsums) {
|
||||
__m512i squared_source_lo = square_m512(src_longs_lo);
|
||||
__m512i squared_source_hi = square_m512(src_longs_hi);
|
||||
|
||||
integral_8_operations(squared_source_lo, sqsums_above, data_mask, sqsums, sqsum_accumulator);
|
||||
integral_8_operations(squared_source_hi, sqsums_above+1, data_mask>>8, sqsums+1, sqsum_accumulator);
|
||||
sqsums += 2;
|
||||
sqsums_above+=2;
|
||||
}
|
||||
|
||||
// Prepare for next iteration of outer loop
|
||||
// Prepare for next iteration of loop
|
||||
// shift source to align next 16 bytes to lane 0, shift the mask, and advance the pointers
|
||||
sums += 2;
|
||||
sums_above += 2;
|
||||
data_mask = data_mask >> 16;
|
||||
src_64byte_chunk = shift_right_16_bytes(src_64byte_chunk);
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
static CV_ALWAYS_INLINE
|
||||
void integral_8_operations(const __m512i src_longs, const __m512i *above_values_ptr, __mmask64 data_mask,
|
||||
__m512i *results_ptr, __m512i &accumulator)
|
||||
{
|
||||
// NOTE that the calculate_integral function referenced here must be implemented in the templated
|
||||
// derivatives because the algorithm depends heavily on the number of channels in the image
|
||||
//
|
||||
_mm512_mask_storeu_pd(
|
||||
results_ptr, // Store the result here
|
||||
data_mask, // Using the data mask to avoid overrunning the line
|
||||
@@ -188,59 +191,178 @@ public:
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
__m512d calculate_integral(__m512i src_longs, const __m512d above_values, __m512i &accumulator)
|
||||
{
|
||||
__m512i carryover_idxs = _mm512_set_epi64(6, 5, 7, 6, 5, 7, 6, 5);
|
||||
|
||||
// Align data to prepare for the adds:
|
||||
// shifts data left by 3 and 6 qwords(lanes) and gets rolling sum in all lanes
|
||||
// Vertical LANES: 76543210
|
||||
// src_longs : HGFEDCBA
|
||||
// shited3lanes : + EDCBA
|
||||
// shifted6lanes : + BA
|
||||
// carry_over_idxs : + 65765765 (index position of result from previous iteration)
|
||||
// = integral
|
||||
__m512i shifted3lanes = _mm512_maskz_expand_epi64(0xF8, src_longs);
|
||||
__m512i shifted6lanes = _mm512_maskz_expand_epi64(0xC0, src_longs);
|
||||
__m512i carry_over = _mm512_permutex2var_epi64(accumulator, carryover_idxs, accumulator);
|
||||
|
||||
// Do the adds in tree form (shift3 + shift 6) + (current_source_values + accumulator)
|
||||
__m512i sum_shift3and6 = _mm512_add_epi64(shifted3lanes, shifted6lanes);
|
||||
__m512i sum_src_carry = _mm512_add_epi64(src_longs, carry_over);
|
||||
accumulator = _mm512_add_epi64(sum_shift3and6, sum_src_carry);
|
||||
|
||||
// Convert to packed double and add to the line above to get the true integral value
|
||||
__m512d accumulator_pd = _mm512_cvtepu64_pd(accumulator);
|
||||
__m512d integral_pd = _mm512_add_pd(accumulator_pd, above_values);
|
||||
return integral_pd;
|
||||
}
|
||||
// The calculate_integral function referenced here must be implemented in the templated derivatives
|
||||
// because the algorithm depends heavily on the number of channels in the image
|
||||
// This is the incomplete definition (just the prototype) here.
|
||||
//
|
||||
static CV_ALWAYS_INLINE
|
||||
__m512d calculate_integral(__m512i src_longs, const __m512d above_values, __m512i &accumulator);
|
||||
|
||||
|
||||
static inline
|
||||
static CV_ALWAYS_INLINE
|
||||
__m512i read_64_bytes(const __m512i *srcs, __mmask64 data_mask) {
|
||||
return _mm512_maskz_loadu_epi8(data_mask, srcs);
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
static CV_ALWAYS_INLINE
|
||||
__m128i extract_lower_16bytes(__m512i src_64byte_chunk) {
|
||||
return _mm512_extracti64x2_epi64(src_64byte_chunk, 0x0);
|
||||
}
|
||||
|
||||
|
||||
static CV_ALWAYS_INLINE
|
||||
__m512i convert_lower_8bytes_to_longs(__m128i src_16bytes) {
|
||||
return _mm512_cvtepu8_epi64(src_16bytes);
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
static CV_ALWAYS_INLINE
|
||||
__m512i square_m512(__m512i src_longs) {
|
||||
return _mm512_mullo_epi64(src_longs, src_longs);
|
||||
}
|
||||
|
||||
|
||||
static CV_ALWAYS_INLINE
|
||||
__m128i shift_right_8_bytes(__m128i src_16bytes) {
|
||||
return _mm_maskz_compress_epi64(2, src_16bytes);
|
||||
}
|
||||
|
||||
|
||||
static inline
|
||||
|
||||
static CV_ALWAYS_INLINE
|
||||
__m512i shift_right_16_bytes(__m512i src_64byte_chunk) {
|
||||
return _mm512_maskz_compress_epi64(0xFC, src_64byte_chunk);
|
||||
}
|
||||
|
||||
|
||||
};
|
||||
|
||||
|
||||
//============================================================================================================
|
||||
// This the only section that needs to change with respect to algorithm based on the number of channels
|
||||
// It is responsible for returning the calculation of 8 lanes worth of the integral and returning in the
|
||||
// accumulated sums in the accumulator parameter (NOTE: accumulator is an input and output parameter)
|
||||
//
|
||||
// The function prototype that needs to be implemented is:
|
||||
//
|
||||
// __m512d calculate_integral(__m512i src_longs, const __m512d above_values, __m512i &accumulator){ ... }
|
||||
//
|
||||
// Description of parameters:
|
||||
// INPUTS:
|
||||
// src_longs : 8 lanes worth of the source bytes converted to 64 bit integers
|
||||
// above_values: 8 lanes worth of the result values from the line above (See the integral spec)
|
||||
// accumulator : 8 lanes worth of sums from the previous iteration
|
||||
// IMPORTANT NOTE: This parameter is both an INPUT AND OUTPUT parameter to this function
|
||||
//
|
||||
// OUTPUTS:
|
||||
// return value: The actual integral value for all 8 lanes which is defined by the spec as
|
||||
// the sum of all channel values to the left of a given pixel plus the result
|
||||
// written to the line directly above the current line.
|
||||
// accumulator: This is an input and and output. This parameter should be left with the accumulated
|
||||
// sums for the current 8 lanes keeping all entries in the proper lane (do not shuffle it)
|
||||
//
|
||||
// Below here is the channel specific implementation
|
||||
//
|
||||
|
||||
//========================================
|
||||
// 2 Channel Integral Implementation
|
||||
//========================================
|
||||
template<>
|
||||
CV_ALWAYS_INLINE
|
||||
__m512d IntegralCalculator < 2 > ::calculate_integral(__m512i src_longs, const __m512d above_values, __m512i &accumulator)
|
||||
{
|
||||
__m512i carryover_idxs = _mm512_set_epi64(7, 6, 7, 6, 7, 6, 7, 6);
|
||||
|
||||
// Align data to prepare for the adds:
|
||||
// shifts data left by 3 and 6 qwords(lanes) and gets rolling sum in all lanes
|
||||
// Vertical LANES : 76543210
|
||||
// src_longs : HGFEDCBA
|
||||
// shift2lanes : + FEDCBA
|
||||
// shift4lanes : + DCBA
|
||||
// shift6lanes : + BA
|
||||
// carry_over_idxs : + 76767676 (index position of result from previous iteration)
|
||||
// = integral
|
||||
__m512i shift2lanes = _mm512_maskz_expand_epi64(0xFC, src_longs);
|
||||
__m512i shift4lanes = _mm512_maskz_expand_epi64(0xF0, src_longs);
|
||||
__m512i shift6lanes = _mm512_maskz_expand_epi64(0xC0, src_longs);
|
||||
__m512i carry_over = _mm512_permutex2var_epi64(accumulator, carryover_idxs, accumulator);
|
||||
|
||||
// Add all values in tree form for perf ((0+2) + (4+6))
|
||||
__m512i sum_shift_02 = _mm512_add_epi64(src_longs, shift2lanes);
|
||||
__m512i sum_shift_46 = _mm512_add_epi64(shift4lanes, shift6lanes);
|
||||
__m512i sum_all = _mm512_add_epi64(sum_shift_02, sum_shift_46);
|
||||
accumulator = _mm512_add_epi64(sum_all, carry_over);
|
||||
|
||||
// Convert to packed double and add to the line above to get the true integral value
|
||||
__m512d accumulator_pd = _mm512_cvtepu64_pd(accumulator);
|
||||
__m512d integral_pd = _mm512_add_pd(accumulator_pd, above_values);
|
||||
return integral_pd;
|
||||
}
|
||||
|
||||
//========================================
|
||||
// 3 Channel Integral Implementation
|
||||
//========================================
|
||||
template<>
|
||||
CV_ALWAYS_INLINE
|
||||
__m512d IntegralCalculator < 3 > ::calculate_integral(__m512i src_longs, const __m512d above_values, __m512i &accumulator)
|
||||
{
|
||||
__m512i carryover_idxs = _mm512_set_epi64(6, 5, 7, 6, 5, 7, 6, 5);
|
||||
|
||||
// Align data to prepare for the adds:
|
||||
// shifts data left by 3 and 6 qwords(lanes) and gets rolling sum in all lanes
|
||||
// Vertical LANES: 76543210
|
||||
// src_longs : HGFEDCBA
|
||||
// shit3lanes : + EDCBA
|
||||
// shift6lanes : + BA
|
||||
// carry_over_idxs : + 65765765 (index position of result from previous iteration)
|
||||
// = integral
|
||||
__m512i shift3lanes = _mm512_maskz_expand_epi64(0xF8, src_longs);
|
||||
__m512i shift6lanes = _mm512_maskz_expand_epi64(0xC0, src_longs);
|
||||
__m512i carry_over = _mm512_permutex2var_epi64(accumulator, carryover_idxs, accumulator);
|
||||
|
||||
// Do the adds in tree form
|
||||
__m512i sum_shift_03 = _mm512_add_epi64(src_longs, shift3lanes);
|
||||
__m512i sum_rest = _mm512_add_epi64(shift6lanes, carry_over);
|
||||
accumulator = _mm512_add_epi64(sum_shift_03, sum_rest);
|
||||
|
||||
// Convert to packed double and add to the line above to get the true integral value
|
||||
__m512d accumulator_pd = _mm512_cvtepu64_pd(accumulator);
|
||||
__m512d integral_pd = _mm512_add_pd(accumulator_pd, above_values);
|
||||
return integral_pd;
|
||||
}
|
||||
|
||||
|
||||
//========================================
|
||||
// 4 Channel Integral Implementation
|
||||
//========================================
|
||||
template<>
|
||||
CV_ALWAYS_INLINE
|
||||
__m512d IntegralCalculator < 4 > ::calculate_integral(__m512i src_longs, const __m512d above_values, __m512i &accumulator)
|
||||
{
|
||||
__m512i carryover_idxs = _mm512_set_epi64(7, 6, 5, 4, 7, 6, 5, 4);
|
||||
|
||||
// Align data to prepare for the adds:
|
||||
// shifts data left by 3 and 6 qwords(lanes) and gets rolling sum in all lanes
|
||||
// Vertical LANES: 76543210
|
||||
// src_longs : HGFEDCBA
|
||||
// shit4lanes : + DCBA
|
||||
// carry_over_idxs : + 76547654 (index position of result from previous iteration)
|
||||
// = integral
|
||||
__m512i shifted4lanes = _mm512_maskz_expand_epi64(0xF0, src_longs);
|
||||
__m512i carry_over = _mm512_permutex2var_epi64(accumulator, carryover_idxs, accumulator);
|
||||
|
||||
// Add data pixels and carry over from last iteration
|
||||
__m512i sum_shift_04 = _mm512_add_epi64(src_longs, shifted4lanes);
|
||||
accumulator = _mm512_add_epi64(sum_shift_04, carry_over);
|
||||
|
||||
// Convert to packed double and add to the line above to get the true integral value
|
||||
__m512d accumulator_pd = _mm512_cvtepu64_pd(accumulator);
|
||||
__m512d integral_pd = _mm512_add_pd(accumulator_pd, above_values);
|
||||
return integral_pd;
|
||||
}
|
||||
|
||||
|
||||
} // end of anonymous namespace
|
||||
|
||||
namespace opt_AVX512_SKX {
|
||||
@@ -253,8 +375,22 @@ void calculate_integral_avx512(const uchar *src, size_t _srcstep,
|
||||
double *sqsum, size_t _sqsumstep,
|
||||
int width, int height, int cn)
|
||||
{
|
||||
IntegralCalculator_3Channel calculator;
|
||||
calculator.calculate_integral_avx512(src, _srcstep, sum, _sumstep, sqsum, _sqsumstep, width, height, cn);
|
||||
switch(cn){
|
||||
case 2: {
|
||||
IntegralCalculator<2> calculator;
|
||||
calculator.calculate_integral_avx512(src, _srcstep, sum, _sumstep, sqsum, _sqsumstep, width, height);
|
||||
break;
|
||||
}
|
||||
case 3: {
|
||||
IntegralCalculator<3> calculator;
|
||||
calculator.calculate_integral_avx512(src, _srcstep, sum, _sumstep, sqsum, _sqsumstep, width, height);
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
IntegralCalculator<4> calculator;
|
||||
calculator.calculate_integral_avx512(src, _srcstep, sum, _sumstep, sqsum, _sqsumstep, width, height);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -46,7 +46,6 @@
|
||||
#include "opencv2/core/hal/intrin.hpp"
|
||||
#include "sumpixels.hpp"
|
||||
|
||||
|
||||
namespace cv
|
||||
{
|
||||
|
||||
@@ -77,8 +76,8 @@ struct Integral_SIMD<uchar, double, double> {
|
||||
{
|
||||
#if CV_TRY_AVX512_SKX
|
||||
CV_UNUSED(_tiltedstep);
|
||||
// TODO: Add support for 1,2, and 4 channels
|
||||
if (CV_CPU_HAS_SUPPORT_AVX512_SKX && !tilted && cn == 3){
|
||||
// TODO: Add support for 1 channel input (WIP)
|
||||
if (CV_CPU_HAS_SUPPORT_AVX512_SKX && !tilted && ((cn >= 2) && (cn <= 4))){
|
||||
opt_AVX512_SKX::calculate_integral_avx512(src, _srcstep, sum, _sumstep,
|
||||
sqsum, _sqsumstep, width, height, cn);
|
||||
return true;
|
||||
|
||||
@@ -1667,12 +1667,11 @@ void CV_IntegralTest::get_test_array_types_and_sizes( int test_case_idx,
|
||||
{
|
||||
RNG& rng = ts->get_rng();
|
||||
int depth = cvtest::randInt(rng) % 2, sum_depth;
|
||||
int cn = cvtest::randInt(rng) % 3 + 1;
|
||||
int cn = cvtest::randInt(rng) % 4 + 1;
|
||||
cvtest::ArrayTest::get_test_array_types_and_sizes( test_case_idx, sizes, types );
|
||||
Size sum_size;
|
||||
|
||||
depth = depth == 0 ? CV_8U : CV_32F;
|
||||
cn += cn == 2;
|
||||
int b = (cvtest::randInt(rng) & 1) != 0;
|
||||
sum_depth = depth == CV_8U && b ? CV_32S : b ? CV_32F : CV_64F;
|
||||
|
||||
|
||||
@@ -110,6 +110,15 @@ public class JavaCamera2View extends CameraBridgeViewBase {
|
||||
if (mCameraID != null) {
|
||||
Log.i(LOGTAG, "Opening camera: " + mCameraID);
|
||||
manager.openCamera(mCameraID, mStateCallback, mBackgroundHandler);
|
||||
} else { // make JavaCamera2View behaves in the same way as JavaCameraView
|
||||
Log.i(LOGTAG, "Trying to open camera with the value (" + mCameraIndex + ")");
|
||||
if (mCameraIndex < camList.length) {
|
||||
mCameraID = camList[mCameraIndex];
|
||||
manager.openCamera(mCameraID, mStateCallback, mBackgroundHandler);
|
||||
} else {
|
||||
// CAMERA_DISCONNECTED is used when the camera id is no longer valid
|
||||
throw new CameraAccessException(CameraAccessException.CAMERA_DISCONNECTED);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
} catch (CameraAccessException e) {
|
||||
|
||||
@@ -1708,7 +1708,7 @@ static PyMethodDef special_methods[] = {
|
||||
struct ConstDef
|
||||
{
|
||||
const char * name;
|
||||
long val;
|
||||
long long val;
|
||||
};
|
||||
|
||||
static void init_submodule(PyObject * root, const char * name, PyMethodDef * methods, ConstDef * consts)
|
||||
@@ -1747,7 +1747,7 @@ static void init_submodule(PyObject * root, const char * name, PyMethodDef * met
|
||||
}
|
||||
for (ConstDef * c = consts; c->name != NULL; ++c)
|
||||
{
|
||||
PyDict_SetItemString(d, c->name, PyInt_FromLong(c->val));
|
||||
PyDict_SetItemString(d, c->name, PyLong_FromLongLong(c->val));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -265,6 +265,19 @@ enum
|
||||
MOTION_HOMOGRAPHY = 3
|
||||
};
|
||||
|
||||
/** @brief Computes the Enhanced Correlation Coefficient value between two images @cite EP08 .
|
||||
|
||||
@param templateImage single-channel template image; CV_8U or CV_32F array.
|
||||
@param inputImage single-channel input image to be warped to provide an image similar to
|
||||
templateImage, same type as templateImage.
|
||||
@param inputMask An optional mask to indicate valid values of inputImage.
|
||||
|
||||
@sa
|
||||
findTransformECC
|
||||
*/
|
||||
|
||||
CV_EXPORTS_W double computeECC(InputArray templateImage, InputArray inputImage, InputArray inputMask = noArray());
|
||||
|
||||
/** @example samples/cpp/image_alignment.cpp
|
||||
An example using the image alignment ECC algorithm
|
||||
*/
|
||||
@@ -273,7 +286,7 @@ An example using the image alignment ECC algorithm
|
||||
|
||||
@param templateImage single-channel template image; CV_8U or CV_32F array.
|
||||
@param inputImage single-channel input image which should be warped with the final warpMatrix in
|
||||
order to provide an image similar to templateImage, same type as temlateImage.
|
||||
order to provide an image similar to templateImage, same type as templateImage.
|
||||
@param warpMatrix floating-point \f$2\times 3\f$ or \f$3\times 3\f$ mapping matrix (warp).
|
||||
@param motionType parameter, specifying the type of motion:
|
||||
- **MOTION_TRANSLATION** sets a translational motion model; warpMatrix is \f$2\times 3\f$ with
|
||||
@@ -290,6 +303,7 @@ criteria.epsilon defines the threshold of the increment in the correlation coeff
|
||||
iterations (a negative criteria.epsilon makes criteria.maxcount the only termination criterion).
|
||||
Default values are shown in the declaration above.
|
||||
@param inputMask An optional mask to indicate valid values of inputImage.
|
||||
@param gaussFiltSize An optional value indicating size of gaussian blur filter; (DEFAULT: 5)
|
||||
|
||||
The function estimates the optimum transformation (warpMatrix) with respect to ECC criterion
|
||||
(@cite EP08), that is
|
||||
@@ -317,12 +331,19 @@ sample image_alignment.cpp that demonstrates the use of the function. Note that
|
||||
an exception if algorithm does not converges.
|
||||
|
||||
@sa
|
||||
estimateAffine2D, estimateAffinePartial2D, findHomography
|
||||
computeECC, estimateAffine2D, estimateAffinePartial2D, findHomography
|
||||
*/
|
||||
CV_EXPORTS_W double findTransformECC( InputArray templateImage, InputArray inputImage,
|
||||
InputOutputArray warpMatrix, int motionType = MOTION_AFFINE,
|
||||
TermCriteria criteria = TermCriteria(TermCriteria::COUNT+TermCriteria::EPS, 50, 0.001),
|
||||
InputArray inputMask = noArray());
|
||||
InputOutputArray warpMatrix, int motionType,
|
||||
TermCriteria criteria,
|
||||
InputArray inputMask, int gaussFiltSize);
|
||||
|
||||
/** @overload */
|
||||
CV_EXPORTS
|
||||
double findTransformECC(InputArray templateImage, InputArray inputImage,
|
||||
InputOutputArray warpMatrix, int motionType = MOTION_AFFINE,
|
||||
TermCriteria criteria = TermCriteria(TermCriteria::COUNT+TermCriteria::EPS, 50, 0.001),
|
||||
InputArray inputMask = noArray());
|
||||
|
||||
/** @example samples/cpp/kalman.cpp
|
||||
An example using the standard Kalman filter
|
||||
|
||||
@@ -309,16 +309,48 @@ static void update_warping_matrix_ECC (Mat& map_matrix, const Mat& update, const
|
||||
}
|
||||
|
||||
|
||||
/** Function that computes enhanced corelation coefficient from Georgios et.al. 2008
|
||||
* See https://github.com/opencv/opencv/issues/12432
|
||||
*/
|
||||
double cv::computeECC(InputArray templateImage, InputArray inputImage, InputArray inputMask)
|
||||
{
|
||||
CV_Assert(!templateImage.empty());
|
||||
CV_Assert(!inputImage.empty());
|
||||
|
||||
if( ! (templateImage.type()==inputImage.type()))
|
||||
CV_Error( Error::StsUnmatchedFormats, "Both input images must have the same data type" );
|
||||
|
||||
Scalar meanTemplate, sdTemplate;
|
||||
|
||||
int active_pixels = inputMask.empty() ? templateImage.size().area() : countNonZero(inputMask);
|
||||
|
||||
meanStdDev(templateImage, meanTemplate, sdTemplate, inputMask);
|
||||
Mat templateImage_zeromean = Mat::zeros(templateImage.size(), templateImage.type());
|
||||
subtract(templateImage, meanTemplate, templateImage_zeromean, inputMask);
|
||||
double templateImagenorm = std::sqrt(active_pixels*sdTemplate.val[0]*sdTemplate.val[0]);
|
||||
|
||||
Scalar meanInput, sdInput;
|
||||
|
||||
Mat inputImage_zeromean = Mat::zeros(inputImage.size(), inputImage.type());
|
||||
meanStdDev(inputImage, meanInput, sdInput, inputMask);
|
||||
subtract(inputImage, meanInput, inputImage_zeromean, inputMask);
|
||||
double inputImagenorm = std::sqrt(active_pixels*sdInput.val[0]*sdInput.val[0]);
|
||||
|
||||
return templateImage_zeromean.dot(inputImage_zeromean)/(templateImagenorm*inputImagenorm);
|
||||
}
|
||||
|
||||
|
||||
double cv::findTransformECC(InputArray templateImage,
|
||||
InputArray inputImage,
|
||||
InputOutputArray warpMatrix,
|
||||
int motionType,
|
||||
TermCriteria criteria,
|
||||
InputArray inputMask)
|
||||
InputArray inputMask,
|
||||
int gaussFiltSize)
|
||||
{
|
||||
|
||||
|
||||
Mat src = templateImage.getMat();//template iamge
|
||||
Mat src = templateImage.getMat();//template image
|
||||
Mat dst = inputImage.getMat(); //input image (to be warped)
|
||||
Mat map = warpMatrix.getMat(); //warp (transformation)
|
||||
|
||||
@@ -416,11 +448,11 @@ double cv::findTransformECC(InputArray templateImage,
|
||||
|
||||
//gaussian filtering is optional
|
||||
src.convertTo(templateFloat, templateFloat.type());
|
||||
GaussianBlur(templateFloat, templateFloat, Size(5, 5), 0, 0);
|
||||
GaussianBlur(templateFloat, templateFloat, Size(gaussFiltSize, gaussFiltSize), 0, 0);
|
||||
|
||||
Mat preMaskFloat;
|
||||
preMask.convertTo(preMaskFloat, CV_32F);
|
||||
GaussianBlur(preMaskFloat, preMaskFloat, Size(5, 5), 0, 0);
|
||||
GaussianBlur(preMaskFloat, preMaskFloat, Size(gaussFiltSize, gaussFiltSize), 0, 0);
|
||||
// Change threshold.
|
||||
preMaskFloat *= (0.5/0.95);
|
||||
// Rounding conversion.
|
||||
@@ -428,7 +460,7 @@ double cv::findTransformECC(InputArray templateImage,
|
||||
preMask.convertTo(preMaskFloat, preMaskFloat.type());
|
||||
|
||||
dst.convertTo(imageFloat, imageFloat.type());
|
||||
GaussianBlur(imageFloat, imageFloat, Size(5, 5), 0, 0);
|
||||
GaussianBlur(imageFloat, imageFloat, Size(gaussFiltSize, gaussFiltSize), 0, 0);
|
||||
|
||||
// needed matrices for gradients and warped gradients
|
||||
Mat gradientX = Mat::zeros(hd, wd, CV_32FC1);
|
||||
@@ -557,5 +589,13 @@ double cv::findTransformECC(InputArray templateImage,
|
||||
return rho;
|
||||
}
|
||||
|
||||
double cv::findTransformECC(InputArray templateImage, InputArray inputImage,
|
||||
InputOutputArray warpMatrix, int motionType,
|
||||
TermCriteria criteria,
|
||||
InputArray inputMask)
|
||||
{
|
||||
// Use default value of 5 for gaussFiltSize to maintain backward compatibility.
|
||||
return findTransformECC(templateImage, inputImage, warpMatrix, motionType, criteria, inputMask, 5);
|
||||
}
|
||||
|
||||
/* End of file. */
|
||||
|
||||
@@ -95,7 +95,6 @@ double CV_ECC_BaseTest::computeRMS(const Mat& mat1, const Mat& mat2){
|
||||
return sqrt(errorMat.dot(errorMat)/(mat1.rows*mat1.cols));
|
||||
}
|
||||
|
||||
|
||||
class CV_ECC_Test_Translation : public CV_ECC_BaseTest
|
||||
{
|
||||
public:
|
||||
@@ -464,6 +463,22 @@ bool CV_ECC_Test_Mask::testMask(int from)
|
||||
return false;
|
||||
}
|
||||
|
||||
// Test with non-default gaussian blur.
|
||||
findTransformECC(warpedImage, testImg, mapTranslation, 0,
|
||||
TermCriteria(TermCriteria::COUNT+TermCriteria::EPS, ECC_iterations, ECC_epsilon), mask, 1);
|
||||
|
||||
if (!isMapCorrect(mapTranslation)){
|
||||
ts->set_failed_test_info(cvtest::TS::FAIL_INVALID_OUTPUT);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (computeRMS(mapTranslation, translationGround)>MAX_RMS_ECC){
|
||||
ts->set_failed_test_info(cvtest::TS::FAIL_BAD_ACCURACY);
|
||||
ts->printf( ts->LOG, "RMS = %f",
|
||||
computeRMS(mapTranslation, translationGround));
|
||||
return false;
|
||||
}
|
||||
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -476,6 +491,16 @@ void CV_ECC_Test_Mask::run(int from)
|
||||
ts->set_failed_test_info(cvtest::TS::OK);
|
||||
}
|
||||
|
||||
TEST(Video_ECC_Test_Compute, accuracy)
|
||||
{
|
||||
Mat testImg = (Mat_<float>(3, 3) << 1, 0, 0, 1, 0, 0, 1, 0, 0);
|
||||
Mat warpedImage = (Mat_<float>(3, 3) << 0, 1, 0, 0, 1, 0, 0, 1, 0);
|
||||
Mat_<unsigned char> mask = Mat_<unsigned char>::ones(testImg.rows, testImg.cols);
|
||||
double ecc = computeECC(warpedImage, testImg, mask);
|
||||
|
||||
EXPECT_NEAR(ecc, -0.5f, 1e-5f);
|
||||
}
|
||||
|
||||
TEST(Video_ECC_Translation, accuracy) { CV_ECC_Test_Translation test; test.safe_run();}
|
||||
TEST(Video_ECC_Euclidean, accuracy) { CV_ECC_Test_Euclidean test; test.safe_run(); }
|
||||
TEST(Video_ECC_Affine, accuracy) { CV_ECC_Test_Affine test; test.safe_run(); }
|
||||
|
||||
@@ -226,6 +226,7 @@ make & enjoy!
|
||||
#include <assert.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/ioctl.h>
|
||||
#include <limits>
|
||||
|
||||
#ifdef HAVE_CAMV4L2
|
||||
#include <asm/types.h> /* for videodev2.h */
|
||||
@@ -538,7 +539,9 @@ bool CvCaptureCAM_V4L::convertableToRgb() const
|
||||
|
||||
void CvCaptureCAM_V4L::v4l2_create_frame()
|
||||
{
|
||||
CvSize size = {form.fmt.pix.width, form.fmt.pix.height};
|
||||
CV_Assert(form.fmt.pix.width <= (uint)std::numeric_limits<int>::max());
|
||||
CV_Assert(form.fmt.pix.height <= (uint)std::numeric_limits<int>::max());
|
||||
CvSize size = {(int)form.fmt.pix.width, (int)form.fmt.pix.height};
|
||||
int channels = 3;
|
||||
int depth = IPL_DEPTH_8U;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user