diff --git a/modules/core/CMakeLists.txt b/modules/core/CMakeLists.txt index a51d2bd280..d5ce207b12 100644 --- a/modules/core/CMakeLists.txt +++ b/modules/core/CMakeLists.txt @@ -1,7 +1,7 @@ set(the_description "The Core Functionality") ocv_add_dispatched_file(mathfuncs_core SSE2 AVX AVX2 LASX) -ocv_add_dispatched_file(stat SSE4_2 AVX2 LASX) +ocv_add_dispatched_file(stat SSE4_2 AVX2 AVX512_SKX AVX512_ICL LASX) ocv_add_dispatched_file(arithm SSE2 SSE4_1 AVX2 VSX3 LASX) ocv_add_dispatched_file(convert SSE2 AVX2 VSX3 LASX) ocv_add_dispatched_file(convert_scale SSE2 AVX2 LASX) @@ -14,7 +14,7 @@ ocv_add_dispatched_file(minmax SSE2 SSE4_1 AVX2 AVX512_SKX AVX512_ICL LASX) ocv_add_dispatched_file(split SSE2 AVX2 LASX) ocv_add_dispatched_file(sum SSE2 AVX2 LASX) ocv_add_dispatched_file(reduce SSE2 SSSE3 AVX2 NEON_DOTPROD) -ocv_add_dispatched_file(norm SSE2 SSE4_1 AVX AVX2 NEON_DOTPROD LASX) +ocv_add_dispatched_file(norm SSE2 SSE4_1 AVX AVX2 AVX512_SKX AVX512_ICL NEON_DOTPROD LASX) ocv_add_dispatched_file(lut AVX512_ICL) # dispatching for accuracy tests diff --git a/modules/core/src/norm.dispatch.cpp b/modules/core/src/norm.dispatch.cpp index a20eaf0824..45f51d1b5e 100644 --- a/modules/core/src/norm.dispatch.cpp +++ b/modules/core/src/norm.dispatch.cpp @@ -1,6 +1,8 @@ // This file is part of OpenCV project. // It is subject to the license terms in the LICENSE file found in the top-level directory // of this distribution and at http://opencv.org/license.html +// +// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved. #include "precomp.hpp" @@ -155,20 +157,27 @@ float normL2Sqr_(const float* a, const float* b, int n) { int j = 0; float d = 0.f; #if (CV_SIMD || CV_SIMD_SCALABLE) + const int vl = VTraits::vlanes(); v_float32 v_d0 = vx_setzero_f32(), v_d1 = vx_setzero_f32(); v_float32 v_d2 = vx_setzero_f32(), v_d3 = vx_setzero_f32(); - for (; j <= n - 4 * VTraits::vlanes(); j += 4 * VTraits::vlanes()) + for (; j <= n - 4 * vl; j += 4 * vl) { v_float32 t0 = v_sub(vx_load(a + j), vx_load(b + j)); - v_float32 t1 = v_sub(vx_load(a + j + VTraits::vlanes()), vx_load(b + j + VTraits::vlanes())); + v_float32 t1 = v_sub(vx_load(a + j + vl), vx_load(b + j + vl)); v_d0 = v_muladd(t0, t0, v_d0); - v_float32 t2 = v_sub(vx_load(a + j + 2 * VTraits::vlanes()), vx_load(b + j + 2 * VTraits::vlanes())); + v_float32 t2 = v_sub(vx_load(a + j + 2 * vl), vx_load(b + j + 2 * vl)); v_d1 = v_muladd(t1, t1, v_d1); - v_float32 t3 = v_sub(vx_load(a + j + 3 * VTraits::vlanes()), vx_load(b + j + 3 * VTraits::vlanes())); + v_float32 t3 = v_sub(vx_load(a + j + 3 * vl), vx_load(b + j + 3 * vl)); v_d2 = v_muladd(t2, t2, v_d2); v_d3 = v_muladd(t3, t3, v_d3); } - d = v_reduce_sum(v_add(v_add(v_add(v_d0, v_d1), v_d2), v_d3)); + v_d0 = v_add(v_add(v_d0, v_d1), v_add(v_d2, v_d3)); + for (; j <= n - vl; j += vl) + { + v_float32 t0 = v_sub(vx_load(a + j), vx_load(b + j)); + v_d0 = v_muladd(t0, t0, v_d0); + } + d = v_reduce_sum(v_d0); #endif for( ; j < n; j++ ) { @@ -183,16 +192,20 @@ float normL1_(const float* a, const float* b, int n) { int j = 0; float d = 0.f; #if (CV_SIMD || CV_SIMD_SCALABLE) + const int vl = VTraits::vlanes(); v_float32 v_d0 = vx_setzero_f32(), v_d1 = vx_setzero_f32(); v_float32 v_d2 = vx_setzero_f32(), v_d3 = vx_setzero_f32(); - for (; j <= n - 4 * VTraits::vlanes(); j += 4 * VTraits::vlanes()) + for (; j <= n - 4 * vl; j += 4 * vl) { v_d0 = v_add(v_d0, v_absdiff(vx_load(a + j), vx_load(b + j))); - v_d1 = v_add(v_d1, v_absdiff(vx_load(a + j + VTraits::vlanes()), vx_load(b + j + VTraits::vlanes()))); - v_d2 = v_add(v_d2, v_absdiff(vx_load(a + j + 2 * VTraits::vlanes()), vx_load(b + j + 2 * VTraits::vlanes()))); - v_d3 = v_add(v_d3, v_absdiff(vx_load(a + j + 3 * VTraits::vlanes()), vx_load(b + j + 3 * VTraits::vlanes()))); + v_d1 = v_add(v_d1, v_absdiff(vx_load(a + j + vl), vx_load(b + j + vl))); + v_d2 = v_add(v_d2, v_absdiff(vx_load(a + j + 2 * vl), vx_load(b + j + 2 * vl))); + v_d3 = v_add(v_d3, v_absdiff(vx_load(a + j + 3 * vl), vx_load(b + j + 3 * vl))); } - d = v_reduce_sum(v_add(v_add(v_add(v_d0, v_d1), v_d2), v_d3)); + v_d0 = v_add(v_add(v_d0, v_d1), v_add(v_d2, v_d3)); + for (; j <= n - vl; j += vl) + v_d0 = v_add(v_d0, v_absdiff(vx_load(a + j), vx_load(b + j))); + d = v_reduce_sum(v_d0); #endif for( ; j < n; j++ ) d += std::abs(a[j] - b[j]); @@ -203,11 +216,14 @@ int normL1_(const uchar* a, const uchar* b, int n) { int j = 0, d = 0; #if (CV_SIMD || CV_SIMD_SCALABLE) - for (; j <= n - 4 * VTraits::vlanes(); j += 4 * VTraits::vlanes()) + const int vl = VTraits::vlanes(); + for (; j <= n - 4 * vl; j += 4 * vl) d += v_reduce_sad(vx_load(a + j), vx_load(b + j)) + - v_reduce_sad(vx_load(a + j + VTraits::vlanes()), vx_load(b + j + VTraits::vlanes())) + - v_reduce_sad(vx_load(a + j + 2 * VTraits::vlanes()), vx_load(b + j + 2 * VTraits::vlanes())) + - v_reduce_sad(vx_load(a + j + 3 * VTraits::vlanes()), vx_load(b + j + 3 * VTraits::vlanes())); + v_reduce_sad(vx_load(a + j + vl), vx_load(b + j + vl)) + + v_reduce_sad(vx_load(a + j + 2 * vl), vx_load(b + j + 2 * vl)) + + v_reduce_sad(vx_load(a + j + 3 * vl), vx_load(b + j + 3 * vl)); + for (; j <= n - vl; j += vl) + d += v_reduce_sad(vx_load(a + j), vx_load(b + j)); #endif for( ; j < n; j++ ) d += std::abs(a[j] - b[j]); diff --git a/modules/core/src/norm.simd.hpp b/modules/core/src/norm.simd.hpp index e5a2e657e7..6fc465eec8 100644 --- a/modules/core/src/norm.simd.hpp +++ b/modules/core/src/norm.simd.hpp @@ -3,6 +3,7 @@ // of this distribution and at http://opencv.org/license.html // // Copyright (C) 2025, SpaceMIT Inc., all rights reserved. +// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved. // Third party copyrights are property of their respective owners. #include "precomp.hpp" @@ -1462,21 +1463,39 @@ struct MaskedNormInf_SIMD { result = std::max(result, (int)src[i]); } } + else if (cn == 4 && len >= VTraits::vlanes()) { + const int vstep = VTraits::vlanes(); + v_uint8 acc = vx_setzero_u8(); + int i = 0; + for (;;) { + if (i > len - vstep) + i = len - vstep; + v_uint8 c0, c1, c2, c3; + v_load_deinterleave(src + i * 4, c0, c1, c2, c3); + v_uint8 pmax = v_max(v_max(c0, c1), v_max(c2, c3)); + v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8()); + acc = v_max(acc, v_and(pmax, m)); + if (i >= len - vstep) + break; + i += vstep; + } + result = (int)v_reduce_max(acc); + } + else if (cn == 4) { + // len < one vector: pure scalar + for (int i = 0; i < len; i++) { + if (mask[i]) { + const uchar* elem = src + i * 4; + result = std::max(result, (int)std::max(std::max(elem[0], elem[1]), + std::max(elem[2], elem[3]))); + } + } + } else { for (int i = 0; i < len; i++) { if (mask[i]) { const uchar* elem = src + i * cn; - int k = 0; - const int vstep = VTraits::vlanes(); - v_uint8 acc = vx_setzero_u8(); - - for (; k <= cn - vstep; k += vstep) { - acc = v_max(acc, vx_load(elem + k)); - } - - result = std::max(result, (int)v_reduce_max(acc)); - - for (; k < cn; k++) + for (int k = 0; k < cn; k++) result = std::max(result, (int)elem[k]); } } @@ -1508,22 +1527,30 @@ struct MaskedNormL1_SIMD { } } else { - for (int i = 0; i < len; i++) { - if (mask[i]) { - const uchar* elem = src + i * cn; - int k = 0; - const int vstep = VTraits::vlanes() / 4; - v_uint32 acc = vx_setzero_u32(); - - for (; k <= cn - vstep; k += vstep) { - v_uint32 s = vx_load_expand_q(elem + k); - acc = v_add(acc, s); + const int vstep = VTraits::vlanes() / 4; + if (cn >= vstep) { + for (int i = 0; i < len; i++) { + if (mask[i]) { + const uchar* elem = src + i * cn; + int k = 0; + v_uint32 acc = vx_setzero_u32(); + for (; k <= cn - vstep; k += vstep) { + v_uint32 s = vx_load_expand_q(elem + k); + acc = v_add(acc, s); + } + result += (int)v_reduce_sum(acc); + for (; k < cn; k++) + result += elem[k]; + } + } + } + else { + for (int i = 0; i < len; i++) { + if (mask[i]) { + const uchar* elem = src + i * cn; + for (int k = 0; k < cn; k++) + result += elem[k]; } - - result += (int)v_reduce_sum(acc); - - for (; k < cn; k++) - result += elem[k]; } } } @@ -1738,6 +1765,428 @@ normL2_(const T* src, const uchar* mask, ST* _result, int len, int cn) { return 0; } +#if (CV_SIMD || CV_SIMD_SCALABLE) +inline v_uint8 v_diffmask(const v_uint8& d, const uchar* m) { return v_and(d, v_gt(vx_load(m), vx_setzero_u8())); } +inline v_uint16 v_diffmask(const v_uint16& d, const uchar* m) { return v_and(d, v_gt(vx_load_expand(m), vx_setzero_u16())); } +inline v_uint32 v_diffmask(const v_uint32& d, const uchar* m) { return v_and(d, v_gt(vx_load_expand_q(m), vx_setzero_u32())); } +inline v_float32 v_diffmask(const v_float32& d, const uchar* m) { + v_uint32 cm = v_gt(vx_load_expand_q(m), vx_setzero_u32()); + return v_reinterpret_as_f32(v_and(v_reinterpret_as_u32(d), cm)); +} +#endif + +template +struct MaskedNormDiffInf_SIMD { + inline ST operator()(const T* s1, const T* s2, const uchar* mask, int len, int cn) const { + ST result = 0; +#if (CV_SIMD || CV_SIMD_SCALABLE) + const int vstep = VTraits::vlanes(); + if (cn == 1 && len >= vstep) { + auto acc = v_diffmask(v_absdiff(vx_load(s1), vx_load(s2)), mask); + int i = vstep; + for (; i <= len - vstep; i += vstep) + acc = v_max(acc, v_diffmask(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), mask + i)); + result = (ST)v_reduce_max(acc); + for (; i < len; i++) if (mask[i]) result = std::max(result, (ST)std::abs((double)s1[i] - (double)s2[i])); + return result; + } +#endif + for (int i = 0; i < len; i++) if (mask[i]) { + const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn; + for (int k = 0; k < cn; k++) result = std::max(result, (ST)std::abs((double)e1[k] - (double)e2[k])); + } + return result; + } +}; +template +struct MaskedNormDiffL1_SIMD { + inline ST operator()(const T* s1, const T* s2, const uchar* mask, int len, int cn) const { + ST s = 0; + for (int i = 0; i < len; i++) if (mask[i]) { + const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn; + for (int k = 0; k < cn; k++) s += std::abs(e1[k] - e2[k]); + } + return s; + } +}; +template +struct MaskedNormDiffL2_SIMD { + inline ST operator()(const T* s1, const T* s2, const uchar* mask, int len, int cn) const { + ST s = 0; + for (int i = 0; i < len; i++) if (mask[i]) { + const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn; + for (int k = 0; k < cn; k++) { ST v = (ST)e1[k] - (ST)e2[k]; s += v*v; } + } + return s; + } +}; + +#if (CV_SIMD || CV_SIMD_SCALABLE) +// Shared scalar cn>1 fallback for masked norm-diff (RT = accumulator type). +// Used by every SIMD specialization's multi-channel branch. +template +static inline RT maskedNormDiffL1Tail(const T* s1, const T* s2, const uchar* mask, int len, int cn, RT result) { + for (int i = 0; i < len; i++) if (mask[i]) { + const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn; + for (int k = 0; k < cn; k++) result += (RT)std::abs((RT)e1[k] - (RT)e2[k]); + } + return result; +} +template +static inline RT maskedNormDiffL2Tail(const T* s1, const T* s2, const uchar* mask, int len, int cn, RT result) { + for (int i = 0; i < len; i++) if (mask[i]) { + const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn; + for (int k = 0; k < cn; k++) { RT v = (RT)e1[k] - (RT)e2[k]; result += v*v; } + } + return result; +} + +// Shared 8-bit cn==1 masked kernels (uchar/schar): v_absdiff() yields v_uint8 +// for both, so a single typename-T body serves both depths. +template +static inline int maskedNormDiffL1_8(const T* s1, const T* s2, const uchar* mask, int len) { + int i = 0; const int vstep = VTraits::vlanes(); + const v_uint8 one = vx_setall_u8(1); + v_uint32 acc = vx_setzero_u32(); + for (; i <= len - vstep; i += vstep) { + v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8()); + v_uint8 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m); + acc = v_dotprod_expand_fast(ad, one, acc); // sum of masked |diff| + } + int result = (int)v_reduce_sum(acc); + for (; i < len; i++) if (mask[i]) result += std::abs((int)s1[i] - (int)s2[i]); + return result; +} +template +static inline int maskedNormDiffL2_8(const T* s1, const T* s2, const uchar* mask, int len) { + int i = 0; const int vstep = VTraits::vlanes(); + v_uint32 acc = vx_setzero_u32(); + for (; i <= len - vstep; i += vstep) { + v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8()); + v_uint8 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m); + acc = v_dotprod_expand_fast(ad, ad, acc); // sum of masked diff^2 + } + int result = (int)v_reduce_sum(acc); + for (; i < len; i++) if (mask[i]) { int v = (int)s1[i] - (int)s2[i]; result += v*v; } + return result; +} + +template<> +struct MaskedNormDiffInf_SIMD { + int operator()(const uchar* s1, const uchar* s2, const uchar* mask, int len, int cn) const { + int result = 0; + const int vstep = VTraits::vlanes(); + if (cn == 1) { + int i = 0; + v_uint8 acc = vx_setzero_u8(); + for (; i <= len - vstep; i += vstep) { + v_uint8 ad = v_absdiff(vx_load(s1 + i), vx_load(s2 + i)); + v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8()); + acc = v_max(acc, v_and(ad, m)); + } + result = (int)v_reduce_max(acc); + for (; i < len; i++) if (mask[i]) result = std::max(result, std::abs((int)s1[i] - (int)s2[i])); + } + else if (cn == 4 && len >= vstep) { + v_uint8 acc = vx_setzero_u8(); + int i = 0; + for (;;) { + if (i > len - vstep) i = len - vstep; // back-step (max is idempotent) + v_uint8 a0,a1,a2,a3,b0,b1,b2,b3; + v_load_deinterleave(s1 + i*4, a0,a1,a2,a3); + v_load_deinterleave(s2 + i*4, b0,b1,b2,b3); + v_uint8 ad = v_max(v_max(v_absdiff(a0,b0), v_absdiff(a1,b1)), + v_max(v_absdiff(a2,b2), v_absdiff(a3,b3))); + v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8()); + acc = v_max(acc, v_and(ad, m)); + if (i >= len - vstep) break; + i += vstep; + } + result = (int)v_reduce_max(acc); + } + else { + for (int i = 0; i < len; i++) if (mask[i]) { + const uchar* e1 = s1 + i*cn; const uchar* e2 = s2 + i*cn; + for (int k = 0; k < cn; k++) result = std::max(result, std::abs((int)e1[k] - (int)e2[k])); + } + } + return result; + } +}; + +template<> +struct MaskedNormDiffInf_SIMD { + int operator()(const int* s1, const int* s2, const uchar* mask, int len, int cn) const { + int result = 0; + const int vstep = VTraits::vlanes(); + if (cn == 1 && len >= vstep) { + // Use wrapping int subtraction (v_abs(v_sub)) to match the non-masked + // NormDiffInf_SIMD kernel. v_absdiff would compute the true + // unsigned |a-b| which can exceed INT_MAX and overflow on the cast to int. + v_uint32 acc = vx_setzero_u32(); + int i = 0; + for (; i <= len - vstep; i += vstep) { + v_uint32 ad = v_abs(v_sub(vx_load(s1 + i), vx_load(s2 + i))); + v_uint32 m = v_gt(vx_load_expand_q(mask + i), vx_setzero_u32()); + acc = v_max(acc, v_and(ad, m)); + } + result = (int)v_reduce_max(acc); + for (; i < len; i++) if (mask[i]) result = std::max(result, (int)std::abs(s1[i] - s2[i])); + return result; + } + for (int i = 0; i < len; i++) if (mask[i]) { + const int* e1 = s1 + i*cn; const int* e2 = s2 + i*cn; + for (int k = 0; k < cn; k++) result = std::max(result, (int)std::abs(e1[k] - e2[k])); + } + return result; + } +}; + +template<> +struct MaskedNormDiffL1_SIMD { + int operator()(const uchar* s1, const uchar* s2, const uchar* mask, int len, int cn) const { + if (cn == 1) return maskedNormDiffL1_8(s1, s2, mask, len); + if (cn == 4) { + int result = 0; + int i = 0; + const int vstep = VTraits::vlanes(); + const v_uint8 one = vx_setall_u8(1); + v_uint32 acc = vx_setzero_u32(); + for (; i <= len - vstep; i += vstep) { + v_uint8 a0,a1,a2,a3,b0,b1,b2,b3; + v_load_deinterleave(s1 + i*4, a0,a1,a2,a3); + v_load_deinterleave(s2 + i*4, b0,b1,b2,b3); + v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8()); + acc = v_dotprod_expand_fast(v_and(v_absdiff(a0,b0), m), one, acc); + acc = v_dotprod_expand_fast(v_and(v_absdiff(a1,b1), m), one, acc); + acc = v_dotprod_expand_fast(v_and(v_absdiff(a2,b2), m), one, acc); + acc = v_dotprod_expand_fast(v_and(v_absdiff(a3,b3), m), one, acc); + } + result = (int)v_reduce_sum(acc); + for (; i < len; i++) if (mask[i]) { + const uchar* e1 = s1 + i*4; const uchar* e2 = s2 + i*4; + for (int k = 0; k < 4; k++) result += std::abs((int)e1[k] - (int)e2[k]); + } + return result; + } + return maskedNormDiffL1Tail(s1, s2, mask, len, cn, 0); + } +}; + +template<> +struct MaskedNormDiffL2_SIMD { + int operator()(const uchar* s1, const uchar* s2, const uchar* mask, int len, int cn) const { + if (cn == 1) return maskedNormDiffL2_8(s1, s2, mask, len); + if (cn == 4) { + int result = 0; + int i = 0; + const int vstep = VTraits::vlanes(); + v_uint32 acc = vx_setzero_u32(); + for (; i <= len - vstep; i += vstep) { + v_uint8 a0,a1,a2,a3,b0,b1,b2,b3; + v_load_deinterleave(s1 + i*4, a0,a1,a2,a3); + v_load_deinterleave(s2 + i*4, b0,b1,b2,b3); + v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8()); + v_uint8 d0 = v_and(v_absdiff(a0,b0), m), d1 = v_and(v_absdiff(a1,b1), m); + v_uint8 d2 = v_and(v_absdiff(a2,b2), m), d3 = v_and(v_absdiff(a3,b3), m); + acc = v_dotprod_expand_fast(d0, d0, acc); + acc = v_dotprod_expand_fast(d1, d1, acc); + acc = v_dotprod_expand_fast(d2, d2, acc); + acc = v_dotprod_expand_fast(d3, d3, acc); + } + result = (int)v_reduce_sum(acc); + for (; i < len; i++) if (mask[i]) { + const uchar* e1 = s1 + i*4; const uchar* e2 = s2 + i*4; + for (int k = 0; k < 4; k++) { int v = (int)e1[k] - (int)e2[k]; result += v*v; } + } + return result; + } + return maskedNormDiffL2Tail(s1, s2, mask, len, cn, 0); + } +}; + +template<> +struct MaskedNormDiffInf_SIMD { + inline double operator()(const double* s1, const double* s2, const uchar* mask, int len, int cn) const { + double result = 0.0; + for (int i = 0; i < len; i++) if (mask[i]) { + const double* e1 = s1 + i*cn; const double* e2 = s2 + i*cn; + for (int k = 0; k < cn; k++) result = std::max(result, std::abs(e1[k] - e2[k])); + } + return result; + } +}; + +#if CV_SIMD_64F || CV_SIMD_SCALABLE_64F +template<> +struct MaskedNormDiffL1_SIMD { + double operator()(const int* s1, const int* s2, const uchar* mask, int len, int cn) const { + double result = 0.0; + if (cn == 1) { + int i = 0; + const int vstep = VTraits::vlanes(); + v_float64 acc = vx_setzero_f64(); + for (; i <= len - vstep; i += vstep) { + v_uint32 m = v_gt(vx_load_expand_q(mask + i), vx_setzero_u32()); + v_uint32 ad = v_and(v_abs(v_sub(vx_load(s1 + i), vx_load(s2 + i))), m); + v_int32 adi = v_reinterpret_as_s32(ad); + acc = v_add(acc, v_cvt_f64(adi)); + acc = v_add(acc, v_cvt_f64_high(adi)); + } + result = v_reduce_sum(acc); + // Use wrapping int subtraction to match the SIMD body (v_abs(v_sub)) + // and the non-masked NormDiffL1_SIMD kernel. + for (; i < len; i++) if (mask[i]) result += std::abs(s1[i] - s2[i]); + } + else { + result = maskedNormDiffL1Tail(s1, s2, mask, len, cn, 0.0); + } + return result; + } +}; + +// Masked L1/L2 SIMD for the remaining depths (cn==1). Without these the masked +// norm-diff for 8s/16u/16s/32s/32f fell back to the scalar base template, which +// the compiler autovectorizes poorly once AVX-512 dispatch is enabled. +template<> struct MaskedNormDiffL1_SIMD { + int operator()(const schar* s1, const schar* s2, const uchar* mask, int len, int cn) const { + if (cn == 1) return maskedNormDiffL1_8(s1, s2, mask, len); + return maskedNormDiffL1Tail(s1, s2, mask, len, cn, 0); + } +}; + +// Shared 16-bit masked L1 kernel (ushort/short): v_absdiff() yields v_uint16 +// for both, so a single typename-T body serves both depths. +template +static inline int maskedNormDiffL1_16(const T* s1, const T* s2, const uchar* mask, int len, int cn) { + int result = 0; + if (cn == 1) { + int i = 0; const int vstep = VTraits::vlanes(); + v_uint32 acc = vx_setzero_u32(); + for (; i <= len - vstep; i += vstep) { + v_uint16 m = v_gt(vx_load_expand(mask + i), vx_setzero_u16()); + v_uint16 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m); + v_uint32 lo, hi; v_expand(ad, lo, hi); + acc = v_add(acc, v_add(lo, hi)); + } + result = (int)v_reduce_sum(acc); + for (; i < len; i++) if (mask[i]) result += std::abs((int)s1[i] - (int)s2[i]); + } else { + result = maskedNormDiffL1Tail(s1, s2, mask, len, cn, 0); + } + return result; +} +template<> struct MaskedNormDiffL1_SIMD { + int operator()(const ushort* s1, const ushort* s2, const uchar* mask, int len, int cn) const + { return maskedNormDiffL1_16(s1, s2, mask, len, cn); } }; +template<> struct MaskedNormDiffL1_SIMD { + int operator()(const short* s1, const short* s2, const uchar* mask, int len, int cn) const + { return maskedNormDiffL1_16(s1, s2, mask, len, cn); } }; + +template<> +struct MaskedNormDiffL1_SIMD { + double operator()(const float* s1, const float* s2, const uchar* mask, int len, int cn) const { + double result = 0.0; + if (cn == 1) { + int i = 0; const int vstep = VTraits::vlanes(); + v_float64 acc0 = vx_setzero_f64(), acc1 = vx_setzero_f64(); + for (; i <= len - vstep; i += vstep) { + v_float32 m = v_reinterpret_as_f32(v_gt(vx_load_expand_q(mask + i), vx_setzero_u32())); + v_float32 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m); + acc0 = v_add(acc0, v_cvt_f64(ad)); + acc1 = v_add(acc1, v_cvt_f64_high(ad)); + } + result = v_reduce_sum(v_add(acc0, acc1)); + for (; i < len; i++) if (mask[i]) result += std::abs((double)s1[i] - (double)s2[i]); + } else { + result = maskedNormDiffL1Tail(s1, s2, mask, len, cn, 0.0); + } + return result; + } +}; + +template<> struct MaskedNormDiffL2_SIMD { + int operator()(const schar* s1, const schar* s2, const uchar* mask, int len, int cn) const { + if (cn == 1) return maskedNormDiffL2_8(s1, s2, mask, len); + return maskedNormDiffL2Tail(s1, s2, mask, len, cn, 0); + } +}; + +// Shared 16-bit masked L2 kernel (ushort/short): v_absdiff() yields v_uint16 +// for both, so a single typename-T body serves both depths. +template +static inline double maskedNormDiffL2_16(const T* s1, const T* s2, const uchar* mask, int len, int cn) { + double result = 0.0; + if (cn == 1) { + int i = 0; const int vstep = VTraits::vlanes(); + v_float64 acc = vx_setzero_f64(); + for (; i <= len - vstep; i += vstep) { + v_uint16 m = v_gt(vx_load_expand(mask + i), vx_setzero_u16()); + v_uint16 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m); + v_uint64 u = v_dotprod_expand_fast(ad, ad); + acc = v_add(acc, v_cvt_f64(v_reinterpret_as_s64(u))); + } + result = v_reduce_sum(acc); + for (; i < len; i++) if (mask[i]) { double v = (double)s1[i] - (double)s2[i]; result += v*v; } + } else { + result = maskedNormDiffL2Tail(s1, s2, mask, len, cn, 0.0); + } + return result; +} +template<> struct MaskedNormDiffL2_SIMD { + double operator()(const ushort* s1, const ushort* s2, const uchar* mask, int len, int cn) const + { return maskedNormDiffL2_16(s1, s2, mask, len, cn); } }; +template<> struct MaskedNormDiffL2_SIMD { + double operator()(const short* s1, const short* s2, const uchar* mask, int len, int cn) const + { return maskedNormDiffL2_16(s1, s2, mask, len, cn); } }; + +template<> +struct MaskedNormDiffL2_SIMD { + double operator()(const int* s1, const int* s2, const uchar* mask, int len, int cn) const { + double result = 0.0; + if (cn == 1) { + int i = 0; const int vstep = VTraits::vlanes(); + v_float64 r0 = vx_setzero_f64(), r1 = vx_setzero_f64(); + for (; i <= len - vstep; i += vstep) { + v_uint32 m = v_gt(vx_load_expand_q(mask + i), vx_setzero_u32()); + v_uint32 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m); + v_uint64 e0, e1; v_expand(ad, e0, e1); + v_float64 f0 = v_cvt_f64(v_reinterpret_as_s64(e0)), f1 = v_cvt_f64(v_reinterpret_as_s64(e1)); + r0 = v_fma(f0, f0, r0); r1 = v_fma(f1, f1, r1); + } + result = v_reduce_sum(v_add(r0, r1)); + for (; i < len; i++) if (mask[i]) { double v = (double)s1[i] - (double)s2[i]; result += v*v; } + } else { + result = maskedNormDiffL2Tail(s1, s2, mask, len, cn, 0.0); + } + return result; + } +}; + +template<> +struct MaskedNormDiffL2_SIMD { + double operator()(const float* s1, const float* s2, const uchar* mask, int len, int cn) const { + double result = 0.0; + if (cn == 1) { + int i = 0; const int vstep = VTraits::vlanes(); + v_float64 r0 = vx_setzero_f64(), r1 = vx_setzero_f64(); + for (; i <= len - vstep; i += vstep) { + v_float32 m = v_reinterpret_as_f32(v_gt(vx_load_expand_q(mask + i), vx_setzero_u32())); + v_float32 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m); + v_float64 f0 = v_cvt_f64(ad), f1 = v_cvt_f64_high(ad); + r0 = v_fma(f0, f0, r0); r1 = v_fma(f1, f1, r1); + } + result = v_reduce_sum(v_add(r0, r1)); + for (; i < len; i++) if (mask[i]) { double v = (double)s1[i] - (double)s2[i]; result += v*v; } + } else { + result = maskedNormDiffL2Tail(s1, s2, mask, len, cn, 0.0); + } + return result; + } +}; +#endif +#endif + template int normDiffInf_(const T* src1, const T* src2, const uchar* mask, ST* _result, int len, int cn) { ST result = *_result; @@ -1745,13 +2194,8 @@ normDiffInf_(const T* src1, const T* src2, const uchar* mask, ST* _result, int l NormDiffInf_SIMD op; result = std::max(result, op(src1, src2, len*cn)); } else { - for( int i = 0; i < len; i++, src1 += cn, src2 += cn ) { - if( mask[i] ) { - for( int k = 0; k < cn; k++ ) { - result = std::max(result, (ST)std::abs(src1[k] - src2[k])); - } - } - } + MaskedNormDiffInf_SIMD op; + result = std::max(result, op(src1, src2, mask, len, cn)); } *_result = result; return 0; @@ -1765,13 +2209,8 @@ normDiffL1_(const T* src1, const T* src2, const uchar* mask, ST* _result, int le result += op(src1, src2, len*cn); } else { - for( int i = 0; i < len; i++, src1 += cn, src2 += cn ) { - if( mask[i] ) { - for( int k = 0; k < cn; k++ ) { - result += std::abs(src1[k] - src2[k]); - } - } - } + MaskedNormDiffL1_SIMD op; + result += op(src1, src2, mask, len, cn); } *_result = result; return 0; @@ -1784,14 +2223,8 @@ normDiffL2_(const T* src1, const T* src2, const uchar* mask, ST* _result, int le NormDiffL2_SIMD op; result += op(src1, src2, len*cn); } else { - for( int i = 0; i < len; i++, src1 += cn, src2 += cn ) { - if( mask[i] ) { - for( int k = 0; k < cn; k++ ) { - ST v = (ST)src1[k] - (ST)src2[k]; - result += v*v; - } - } - } + MaskedNormDiffL2_SIMD op; + result += op(src1, src2, mask, len, cn); } *_result = result; return 0; diff --git a/modules/core/src/stat.dispatch.cpp b/modules/core/src/stat.dispatch.cpp index 08275fac50..9df94a4efb 100644 --- a/modules/core/src/stat.dispatch.cpp +++ b/modules/core/src/stat.dispatch.cpp @@ -1,6 +1,8 @@ // This file is part of OpenCV project. // It is subject to the license terms in the LICENSE file found in the top-level directory // of this distribution and at http://opencv.org/license.html. +// +// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved. #include "precomp.hpp" @@ -9,20 +11,26 @@ namespace cv { namespace hal { +// Resolve the SIMD implementation ONCE via a cached function pointer. normHamming +// is called per-descriptor in hot matcher loops on short (32/64-byte) descriptors, +// where a per-call dispatch chain + CV_INSTRUMENT_REGION dominate the cost. +static NormHammingFunc getNormHammingFunc() { + CV_CPU_DISPATCH(getNormHammingFunc, (), CV_CPU_DISPATCH_MODES_ALL); +} +static NormHammingDiffFunc getNormHammingDiffFunc() { + CV_CPU_DISPATCH(getNormHammingDiffFunc, (), CV_CPU_DISPATCH_MODES_ALL); +} + int normHamming(const uchar* a, int n) { - CV_INSTRUMENT_REGION(); - - CV_CPU_DISPATCH(normHamming, (a, n), - CV_CPU_DISPATCH_MODES_ALL); + static const NormHammingFunc fn = getNormHammingFunc(); + return fn(a, n); } int normHamming(const uchar* a, const uchar* b, int n) { - CV_INSTRUMENT_REGION(); - - CV_CPU_DISPATCH(normHamming, (a, b, n), - CV_CPU_DISPATCH_MODES_ALL); + static const NormHammingDiffFunc fn = getNormHammingDiffFunc(); + return fn(a, b, n); } }} //cv::hal diff --git a/modules/core/src/stat.simd.hpp b/modules/core/src/stat.simd.hpp index e363313e5b..0322c6e4e2 100644 --- a/modules/core/src/stat.simd.hpp +++ b/modules/core/src/stat.simd.hpp @@ -1,6 +1,8 @@ // This file is part of OpenCV project. // It is subject to the license terms in the LICENSE file found in the top-level directory // of this distribution and at http://opencv.org/license.html. +// +// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved. #include "opencv2/core/hal/intrin.hpp" @@ -8,12 +10,18 @@ namespace cv { namespace hal { extern const uchar popCountTable[256]; +typedef int (*NormHammingFunc)(const uchar*, int); +typedef int (*NormHammingDiffFunc)(const uchar*, const uchar*, int); + CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN // forward declarations int normHamming(const uchar* a, int n); int normHamming(const uchar* a, const uchar* b, int n); +NormHammingFunc getNormHammingFunc(); +NormHammingDiffFunc getNormHammingDiffFunc(); + #ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY #if CV_AVX2 @@ -125,6 +133,17 @@ int normHamming(const uchar* a, const uchar* b, int n) return result; } +NormHammingFunc getNormHammingFunc() +{ + NormHammingFunc f = &normHamming; // disambiguate the (a,n) overload + return f; +} +NormHammingDiffFunc getNormHammingDiffFunc() +{ + NormHammingDiffFunc f = &normHamming; // disambiguate the (a,b,n) overload + return f; +} + #endif // CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY CV_CPU_OPTIMIZATION_NAMESPACE_END diff --git a/modules/features2d/perf/perf_matcher.cpp b/modules/features2d/perf/perf_matcher.cpp new file mode 100644 index 0000000000..fe8a892a8a --- /dev/null +++ b/modules/features2d/perf/perf_matcher.cpp @@ -0,0 +1,91 @@ +// This file is part of OpenCV project. +// It is subject to the license terms in the LICENSE file found in the top-level directory +// of this distribution and at http://opencv.org/license.html. +// +// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved. + +// Brute-force descriptor matching perf tests. These exercise the core distance +// kernels (hal::normL2Sqr_ / normL1_ / normHamming via cv::batchDistance) that +// back BFMatcher, which is how float descriptors (SIFT 128-d, SURF 64-d) and +// binary descriptors (ORB 32-byte) are matched. + +#include "perf_precomp.hpp" + +namespace opencv_test +{ +using namespace perf; + +// (descriptor dimension, query/train descriptor count) +typedef tuple Dim_Count_t; +typedef TestBaseWithParam DescriptorMatcherFixture; + +// Float descriptors matched with L2 (SIFT=128, SURF=64) — uses hal::normL2Sqr_. +PERF_TEST_P(DescriptorMatcherFixture, bfmatch_knn_L2_float, + testing::Combine( + testing::Values(64, 128), // SURF, SIFT descriptor sizes + testing::Values(512, 1000) + )) +{ + const int dim = get<0>(GetParam()); + const int count = get<1>(GetParam()); + + Mat query(count, dim, CV_32F); + Mat train(count, dim, CV_32F); + declare.in(query, train, WARMUP_RNG); + declare.time(60); + + BFMatcher matcher(NORM_L2, false); + std::vector > matches; + + TEST_CYCLE() matcher.knnMatch(query, train, matches, 2); + + SANITY_CHECK_NOTHING(); +} + +// Float descriptors matched with L1 — uses hal::normL1_. +PERF_TEST_P(DescriptorMatcherFixture, bfmatch_knn_L1_float, + testing::Combine( + testing::Values(64, 128), + testing::Values(512, 1000) + )) +{ + const int dim = get<0>(GetParam()); + const int count = get<1>(GetParam()); + + Mat query(count, dim, CV_32F); + Mat train(count, dim, CV_32F); + declare.in(query, train, WARMUP_RNG); + declare.time(60); + + BFMatcher matcher(NORM_L1, false); + std::vector > matches; + + TEST_CYCLE() matcher.knnMatch(query, train, matches, 2); + + SANITY_CHECK_NOTHING(); +} + +// Binary descriptors matched with Hamming (ORB/BRISK=32 bytes) — uses hal::normHamming. +PERF_TEST_P(DescriptorMatcherFixture, bfmatch_knn_Hamming_binary, + testing::Combine( + testing::Values(32, 64), // ORB (32), BRISK/FREAK (64) byte sizes + testing::Values(512, 1000) + )) +{ + const int bytes = get<0>(GetParam()); + const int count = get<1>(GetParam()); + + Mat query(count, bytes, CV_8U); + Mat train(count, bytes, CV_8U); + declare.in(query, train, WARMUP_RNG); + declare.time(60); + + BFMatcher matcher(NORM_HAMMING, false); + std::vector > matches; + + TEST_CYCLE() matcher.knnMatch(query, train, matches, 2); + + SANITY_CHECK_NOTHING(); +} + +} // namespace