mirror of
https://github.com/opencv/opencv.git
synced 2026-07-21 19:33:03 +04:00
Compare commits
20 Commits
5.x
...
6f402272e7
| Author | SHA1 | Date | |
|---|---|---|---|
| 6f402272e7 | |||
| 0654a42e19 | |||
| f7dd7df170 | |||
| 7aa163b83f | |||
| c47541acbd | |||
| b5fe8dfef3 | |||
| 4d1e206f5e | |||
| 2e778c52c1 | |||
| 6b640b424c | |||
| 06574736b3 | |||
| 4fe51e51e0 | |||
| 335abd236f | |||
| 46d1b6c99d | |||
| e6b1ef272e | |||
| 1bea199e00 | |||
| 42cd88d737 | |||
| 4c3895e96c | |||
| 2c14cc1897 | |||
| 13fb140932 | |||
| 9d66a589b4 |
Vendored
+6
-6
@@ -1,9 +1,9 @@
|
||||
# Binaries branch name: ffmpeg/4.x_20251226
|
||||
# Binaries were created for OpenCV: cff7581175d2abfc6aef2e4f04f482e258b5c864
|
||||
ocv_update(FFMPEG_BINARIES_COMMIT "d82ad9a54a7b42a1648a9cae8fed5c2f20ea396c")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "47730de2286110b0d1250ff9cf50ce56")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "3248b4663ffef770cdb54ec8b9d16a28")
|
||||
ocv_update(FFMPEG_FILE_HASH_CMAKE "8862c87496e2e8c375965e1277dee1c7")
|
||||
# Binaries branch name: ffmpeg/4.x_20260715
|
||||
# Binaries were created for OpenCV: 6b640b424c516d27217700392a7ed6a362e040a5
|
||||
ocv_update(FFMPEG_BINARIES_COMMIT "bd9418020a5c342be979c56d6e6434261959d3af")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "31968b434799d3dd56969b15aac3efa8")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "84757ed0f16ddedce99227529d3165e9")
|
||||
ocv_update(FFMPEG_FILE_HASH_CMAKE "e09efc33312d1173be8a9446f3b088fe")
|
||||
|
||||
function(download_win_ffmpeg script_var)
|
||||
set(${script_var} "" PARENT_SCOPE)
|
||||
|
||||
Vendored
+5
-5
@@ -2,7 +2,7 @@ function(download_ippicv root_var)
|
||||
set(${root_var} "" PARENT_SCOPE)
|
||||
|
||||
# Commit SHA in the opencv_3rdparty repo
|
||||
set(IPPICV_COMMIT "406d398c436d0465c8e53dd432d9ecd9301d5f4a")
|
||||
set(IPPICV_COMMIT "8338862a733cb3980d8b51d8e14917fe0e695f71")
|
||||
# Define actual ICV versions
|
||||
if(APPLE)
|
||||
set(IPPICV_COMMIT "0cc4aa06bf2bef4b05d237c69a5a96b9cd0cb85a")
|
||||
@@ -14,8 +14,8 @@ function(download_ippicv root_var)
|
||||
set(OPENCV_ICV_PLATFORM "linux")
|
||||
set(OPENCV_ICV_PACKAGE_SUBDIR "ippicv_lnx")
|
||||
if(X86_64)
|
||||
set(OPENCV_ICV_NAME "ippicv_2026.0.0_lnx_intel64_20260327_general.tgz")
|
||||
set(OPENCV_ICV_HASH "9a3ee0c5c3c02102faa422d60bfd1f4a")
|
||||
set(OPENCV_ICV_NAME "ippicv_2026.0.0_lnx_intel64_20260630_general.tgz")
|
||||
set(OPENCV_ICV_HASH "a77e60db544e07a126ae98f5ffe83be1")
|
||||
else()
|
||||
if(ANDROID)
|
||||
set(IPPICV_COMMIT "c7c6d527dde5fee7cb914ee9e4e20f7436aab3a1")
|
||||
@@ -31,8 +31,8 @@ function(download_ippicv root_var)
|
||||
set(OPENCV_ICV_PLATFORM "windows")
|
||||
set(OPENCV_ICV_PACKAGE_SUBDIR "ippicv_win")
|
||||
if(X86_64)
|
||||
set(OPENCV_ICV_NAME "ippicv_2026.0.0_win_intel64_20260327_general.zip")
|
||||
set(OPENCV_ICV_HASH "73bc67cd5e4c8da706fa88fe84630231")
|
||||
set(OPENCV_ICV_NAME "ippicv_2026.0.0_win_intel64_20260630_general.zip")
|
||||
set(OPENCV_ICV_HASH "d81c8b7d40da2867df82f0077a40afa1")
|
||||
else()
|
||||
set(IPPICV_COMMIT "7f55c0c26be418d494615afca15218566775c725")
|
||||
set(OPENCV_ICV_NAME "ippicv_2021.12.0_win_ia32_20240425_general.zip")
|
||||
|
||||
@@ -26,6 +26,7 @@ add_library(ipphal STATIC
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/canny_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/threshold_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/distancetransform_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/histogram_ipp.cpp"
|
||||
)
|
||||
|
||||
#TODO: HAVE_IPP_ICV and HAVE_IPP_IW added as private macro till OpenCV itself is
|
||||
|
||||
@@ -154,6 +154,11 @@ int ipp_hal_distanceTransform(const uchar* src_data, size_t src_step, uchar* dst
|
||||
#undef cv_hal_distanceTransform
|
||||
#define cv_hal_distanceTransform ipp_hal_distanceTransform
|
||||
|
||||
int ipp_hal_calcHist(const uchar* src_data, size_t src_step, int src_type, int src_width, int src_height,
|
||||
float* hist_data, int hist_size, const float** ranges, bool uniform, bool accumulate);
|
||||
#undef cv_hal_calcHist
|
||||
#define cv_hal_calcHist ipp_hal_calcHist
|
||||
|
||||
#endif // IPP_VERSION_X100 >= 700
|
||||
|
||||
#define IPP_DISABLE_PERF_CANNY_MT 1 // cv::Canny OpenCV MT performance is better
|
||||
|
||||
@@ -0,0 +1,223 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
// Copyright (C) 2026, BigVision LLC, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include <opencv2/core/utils/tls.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#if IPP_VERSION_X100 >= 700
|
||||
|
||||
#define IPP_HISTOGRAM_PARALLEL 1
|
||||
|
||||
namespace cv { namespace ipp { unsigned long long getIppTopFeatures(); } }
|
||||
|
||||
using namespace cv;
|
||||
|
||||
namespace {
|
||||
|
||||
typedef IppStatus(CV_STDCALL * IppiHistogram_C1)(const void* pSrc, int srcStep,
|
||||
IppiSize roiSize, Ipp32u* pHist, const IppiHistogramSpec* pSpec, Ipp8u* pBuffer);
|
||||
|
||||
static IppiHistogram_C1 getIppiHistogramFunction_C1(int type)
|
||||
{
|
||||
IppiHistogram_C1 ippFunction =
|
||||
(type == CV_8UC1) ? (IppiHistogram_C1)ippiHistogram_8u_C1R :
|
||||
(type == CV_16UC1) ? (IppiHistogram_C1)ippiHistogram_16u_C1R :
|
||||
(type == CV_32FC1) ? (IppiHistogram_C1)ippiHistogram_32f_C1R :
|
||||
NULL;
|
||||
|
||||
return ippFunction;
|
||||
}
|
||||
|
||||
class ipp_calcHistParallelTLS
|
||||
{
|
||||
public:
|
||||
ipp_calcHistParallelTLS() {}
|
||||
|
||||
IppAutoBuffer<IppiHistogramSpec> spec;
|
||||
IppAutoBuffer<Ipp8u> buffer;
|
||||
IppAutoBuffer<Ipp32u> thist;
|
||||
};
|
||||
|
||||
class ipp_calcHistParallel: public ParallelLoopBody
|
||||
{
|
||||
public:
|
||||
ipp_calcHistParallel(const Mat &src, Mat &hist, Ipp32s histSize, const float *ranges, bool uniform, bool &ok):
|
||||
ParallelLoopBody(), m_src(src), m_hist(hist), m_ok(ok)
|
||||
{
|
||||
ok = true;
|
||||
|
||||
m_uniform = uniform;
|
||||
m_ranges = ranges;
|
||||
m_histSize = histSize;
|
||||
m_type = ippiGetDataType(src.type());
|
||||
m_levelsNum = histSize+1;
|
||||
ippiHistogram_C1 = getIppiHistogramFunction_C1(src.type());
|
||||
m_fullRoi = ippiSize(src.size());
|
||||
m_bufferSize = 0;
|
||||
m_specSize = 0;
|
||||
if(!ippiHistogram_C1)
|
||||
{
|
||||
ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(ippiHistogramGetBufferSize(m_type, m_fullRoi, &m_levelsNum, 1, 1, &m_specSize, &m_bufferSize) < 0)
|
||||
{
|
||||
ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
hist.setTo(0);
|
||||
}
|
||||
|
||||
virtual void operator() (const Range & range) const CV_OVERRIDE
|
||||
{
|
||||
if(!m_ok)
|
||||
return;
|
||||
|
||||
ipp_calcHistParallelTLS *pTls = m_tls.get();
|
||||
|
||||
IppiSize roi = {m_src.cols, range.end - range.start };
|
||||
bool mtLoop = false;
|
||||
if(m_fullRoi.height != roi.height)
|
||||
mtLoop = true;
|
||||
|
||||
if(!pTls->spec)
|
||||
{
|
||||
pTls->spec.allocate(m_specSize);
|
||||
if(!pTls->spec.get())
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
pTls->buffer.allocate(m_bufferSize);
|
||||
if(!pTls->buffer.get() && m_bufferSize)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(m_uniform)
|
||||
{
|
||||
if(ippiHistogramUniformInit(m_type, (Ipp32f*)&m_ranges[0], (Ipp32f*)&m_ranges[1], (Ipp32s*)&m_levelsNum, 1, pTls->spec) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if(ippiHistogramInit(m_type, (const Ipp32f**)&m_ranges, (Ipp32s*)&m_levelsNum, 1, pTls->spec) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
pTls->thist.allocate(m_histSize*sizeof(Ipp32u));
|
||||
}
|
||||
|
||||
if(CV_INSTRUMENT_FUN_IPP(ippiHistogram_C1, m_src.ptr(range.start), (int)m_src.step, roi, pTls->thist, pTls->spec, pTls->buffer) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(mtLoop)
|
||||
{
|
||||
for(int i = 0; i < m_histSize; i++)
|
||||
CV_XADD((int*)(m_hist.ptr(i)), *(int*)((Ipp32u*)pTls->thist + i));
|
||||
}
|
||||
else
|
||||
ippiCopy_32s_C1R((Ipp32s*)pTls->thist.get(), sizeof(Ipp32u), (Ipp32s*)m_hist.ptr(), (int)m_hist.step, ippiSize(1, m_histSize));
|
||||
}
|
||||
|
||||
private:
|
||||
const Mat &m_src;
|
||||
Mat &m_hist;
|
||||
Ipp32s m_histSize;
|
||||
const float *m_ranges;
|
||||
bool m_uniform;
|
||||
|
||||
IppiHistogram_C1 ippiHistogram_C1;
|
||||
IppiSize m_fullRoi;
|
||||
IppDataType m_type;
|
||||
Ipp32s m_levelsNum;
|
||||
int m_bufferSize;
|
||||
int m_specSize;
|
||||
|
||||
mutable Mutex m_syncMutex;
|
||||
TLSData<ipp_calcHistParallelTLS> m_tls;
|
||||
|
||||
volatile bool &m_ok;
|
||||
const ipp_calcHistParallel & operator = (const ipp_calcHistParallel & );
|
||||
};
|
||||
|
||||
static bool ipp_calchist(const Mat &image, Mat &hist, int histSize, const float** ranges, bool uniform, bool accumulate)
|
||||
{
|
||||
#if IPP_VERSION_X100 < 201801
|
||||
// No SSE42 optimization for uniform 32f
|
||||
if(uniform && image.depth() == CV_32F && cv::ipp::getIppTopFeatures() == ippCPUID_SSE42)
|
||||
return false;
|
||||
#endif
|
||||
|
||||
// IPP_DISABLE_HISTOGRAM - https://github.com/opencv/opencv/issues/11544
|
||||
// and https://github.com/opencv/opencv/issues/21595
|
||||
if ((uniform && (ranges[0][1] - ranges[0][0]) != histSize) || abs(ranges[0][0]) != cvFloor(ranges[0][0]))
|
||||
return false;
|
||||
|
||||
Mat ihist = hist;
|
||||
if(accumulate)
|
||||
ihist.create(1, &histSize, CV_32S);
|
||||
|
||||
bool ok = true;
|
||||
int threads = ippiSuggestThreadsNum(image.cols, image.rows, image.elemSize(), (1+((double)ihist.total()/image.total()))*2);
|
||||
Range range(0, image.rows);
|
||||
ipp_calcHistParallel invoker(image, ihist, histSize, ranges[0], uniform, ok);
|
||||
if(!ok)
|
||||
return false;
|
||||
|
||||
if(IPP_HISTOGRAM_PARALLEL && threads > 1)
|
||||
parallel_for_(range, invoker, threads*2);
|
||||
else
|
||||
invoker(range);
|
||||
|
||||
if(ok)
|
||||
{
|
||||
if(accumulate)
|
||||
{
|
||||
IppiSize histRoi = ippiSize(1, histSize);
|
||||
IppAutoBuffer<Ipp32f> fhist(histSize*sizeof(Ipp32f));
|
||||
CV_INSTRUMENT_FUN_IPP(ippiConvert_32s32f_C1R, (Ipp32s*)ihist.ptr(), (int)ihist.step, (Ipp32f*)fhist, sizeof(Ipp32f), histRoi);
|
||||
CV_INSTRUMENT_FUN_IPP(ippiAdd_32f_C1IR, (Ipp32f*)fhist, sizeof(Ipp32f), (Ipp32f*)hist.ptr(), (int)hist.step, histRoi);
|
||||
}
|
||||
else
|
||||
CV_INSTRUMENT_FUN_IPP(ippiConvert_32s32f_C1R, (Ipp32s*)ihist.ptr(), (int)ihist.step, (Ipp32f*)hist.ptr(), (int)hist.step, ippiSize(1, histSize));
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
int ipp_hal_calcHist(const uchar* src_data, size_t src_step, int src_type, int src_width, int src_height,
|
||||
float* hist_data, int hist_size, const float** ranges, bool uniform, bool accumulate)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
Mat image(src_height, src_width, src_type, (void*)src_data, src_step);
|
||||
Mat hist(1, &hist_size, CV_32F, (void*)hist_data);
|
||||
|
||||
if(ipp_calchist(image, hist, hist_size, ranges, uniform, accumulate))
|
||||
return CV_HAL_ERROR_OK;
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -120,4 +120,30 @@ static inline int ippiSuggestRowThreadsNum(const ::ipp::IwiImage &image, size_t
|
||||
}
|
||||
#endif
|
||||
|
||||
#if IPP_VERSION_X100 >= 201700
|
||||
#define CV_IPP_MALLOC(SIZE) ippMalloc_L(SIZE)
|
||||
#else
|
||||
#define CV_IPP_MALLOC(SIZE) ippMalloc((int)SIZE)
|
||||
#endif
|
||||
|
||||
template<typename T>
|
||||
class IppAutoBuffer
|
||||
{
|
||||
public:
|
||||
IppAutoBuffer() { m_size = 0; m_pBuffer = NULL; }
|
||||
explicit IppAutoBuffer(size_t size) { m_size = 0; m_pBuffer = NULL; allocate(size); }
|
||||
~IppAutoBuffer() { deallocate(); }
|
||||
T* allocate(size_t size) { if(m_size < size) { deallocate(); m_pBuffer = (T*)CV_IPP_MALLOC(size); m_size = size; } return m_pBuffer; }
|
||||
void deallocate() { if(m_pBuffer) { ippFree(m_pBuffer); m_pBuffer = NULL; } m_size = 0; }
|
||||
inline T* get() { return (T*)m_pBuffer;}
|
||||
inline operator T* () { return (T*)m_pBuffer;}
|
||||
inline operator const T* () const { return (const T*)m_pBuffer;}
|
||||
private:
|
||||
IppAutoBuffer(IppAutoBuffer &) {}
|
||||
IppAutoBuffer& operator =(const IppAutoBuffer &) {return *this;}
|
||||
|
||||
size_t m_size;
|
||||
T* m_pBuffer;
|
||||
};
|
||||
|
||||
#endif //__PRECOMP_IPP_HPP__
|
||||
|
||||
@@ -939,7 +939,7 @@ inline scalartype v_reduce_sum(const _Tpvec& a) \
|
||||
}
|
||||
OPENCV_HAL_IMPL_RVV_REDUCE_SUM_FP(v_float32, v_float32, vfloat32m1_t, float, f32, VTraits<v_float32>::vlanes())
|
||||
#if CV_SIMD_SCALABLE_64F
|
||||
OPENCV_HAL_IMPL_RVV_REDUCE_SUM_FP(v_float64, v_float64, vfloat64m1_t, float, f64, VTraits<v_float64>::vlanes())
|
||||
OPENCV_HAL_IMPL_RVV_REDUCE_SUM_FP(v_float64, v_float64, vfloat64m1_t, double, f64, VTraits<v_float64>::vlanes())
|
||||
#endif
|
||||
|
||||
#define OPENCV_HAL_IMPL_RVV_REDUCE(_Tpvec, _nTpvec, func, scalartype, suffix, vl, red) \
|
||||
|
||||
@@ -6,9 +6,9 @@
|
||||
#define OPENCV_VERSION_HPP
|
||||
|
||||
#define CV_VERSION_MAJOR 4
|
||||
#define CV_VERSION_MINOR 14
|
||||
#define CV_VERSION_MINOR 15
|
||||
#define CV_VERSION_REVISION 0
|
||||
#define CV_VERSION_STATUS "-pre"
|
||||
#define CV_VERSION_STATUS "-dev"
|
||||
|
||||
#define CVAUX_STR_EXP(__A) #__A
|
||||
#define CVAUX_STR(__A) CVAUX_STR_EXP(__A)
|
||||
|
||||
@@ -1202,6 +1202,9 @@ typedef TestBaseWithParam<ReduceMinMaxParams> ReduceMinMaxFixture;
|
||||
OCL_PERF_TEST_P(ReduceMinMaxFixture, Reduce,
|
||||
::testing::Combine(OCL_TEST_SIZES,
|
||||
OCL_PERF_ENUM(std::make_pair<MatType, MatType>(CV_8UC1, CV_8UC1),
|
||||
std::make_pair<MatType, MatType>(CV_8UC4, CV_8UC4),
|
||||
std::make_pair<MatType, MatType>(CV_32FC1, CV_32FC1),
|
||||
std::make_pair<MatType, MatType>(CV_32FC3, CV_32FC3),
|
||||
std::make_pair<MatType, MatType>(CV_32FC4, CV_32FC4)),
|
||||
OCL_PERF_ENUM(0, 1),
|
||||
ReduceMinMaxOp::all()))
|
||||
|
||||
@@ -338,9 +338,13 @@ cv::Mat cv::Mat::cross(InputArray _m) const
|
||||
namespace cv
|
||||
{
|
||||
|
||||
typedef void (*ReduceSumFunc)(const Mat& src, Mat& dst);
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
|
||||
template <typename T, typename WT, typename Op>
|
||||
struct ReduceR_SIMD
|
||||
@@ -351,7 +355,6 @@ struct ReduceR_SIMD
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<typename T, typename ST, typename WT, class Op, class OpInit>
|
||||
class ReduceR_Invoker : public ParallelLoopBody
|
||||
{
|
||||
@@ -471,8 +474,6 @@ reduceC_( const Mat& srcmat, Mat& dstmat)
|
||||
parallel_for_(Range(0, srcmat.size().height), body);
|
||||
}
|
||||
|
||||
typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
|
||||
}
|
||||
|
||||
#define reduceSumR8u32s reduceR_<uchar, int, OpAdd<int>, OpNop<int> >
|
||||
@@ -818,9 +819,9 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
{
|
||||
if( op == REDUCE_SUM )
|
||||
{
|
||||
ReduceSumFunc simd_func = getReduceRSumFunc(sdepth, ddepth);
|
||||
ReduceFunc simd_func = getReduceRSumFunc(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = (ReduceFunc)simd_func;
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32S)
|
||||
func = reduceSumR8u32s;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32F)
|
||||
@@ -896,9 +897,11 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
{
|
||||
if(op == REDUCE_SUM)
|
||||
{
|
||||
ReduceSumFunc simd_func = getReduceCSumFunc(sdepth, ddepth);
|
||||
ReduceFunc simd_func = op0 == REDUCE_AVG
|
||||
? getReduceCAvgFunc(sdepth, ddepth)
|
||||
: getReduceCSumFunc(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = (ReduceFunc)simd_func;
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32S)
|
||||
func = reduceSumC8u32s;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32F)
|
||||
@@ -922,7 +925,10 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
}
|
||||
else if(op == REDUCE_MAX)
|
||||
{
|
||||
if(sdepth == CV_8U && ddepth == CV_8U)
|
||||
ReduceFunc simd_func = getReduceCMaxFunc(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_8U)
|
||||
func = reduceMaxC8u;
|
||||
else if(sdepth == CV_16U && ddepth == CV_16U)
|
||||
func = reduceMaxC16u;
|
||||
@@ -935,7 +941,10 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
}
|
||||
else if(op == REDUCE_MIN)
|
||||
{
|
||||
if(sdepth == CV_8U && ddepth == CV_8U)
|
||||
ReduceFunc simd_func = getReduceCMinFunc(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_8U)
|
||||
func = reduceMinC8u;
|
||||
else if(sdepth == CV_16U && ddepth == CV_16U)
|
||||
func = reduceMinC16u;
|
||||
@@ -948,7 +957,10 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
}
|
||||
else if(op == REDUCE_SUM2)
|
||||
{
|
||||
if(sdepth == CV_8U && ddepth == CV_32S)
|
||||
ReduceFunc simd_func = getReduceCSum2Func(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32S)
|
||||
func = reduceSum2C8u32s;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32F)
|
||||
func = reduceSum2C8u32f;
|
||||
@@ -1038,7 +1050,6 @@ template<typename T> static void sort_( const Mat& src, Mat& dst, int flags )
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
#if defined(HAVE_IPP) && !IPP_DISABLE_SORT
|
||||
typedef IppStatus (CV_STDCALL *IppSortFunc)(void *pSrcDst, int len, Ipp8u *pBuffer);
|
||||
|
||||
|
||||
@@ -9,18 +9,50 @@
|
||||
|
||||
namespace cv {
|
||||
|
||||
typedef void (*ReduceSumFunc)(const Mat& src, Mat& dst);
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCSumFunc, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth)
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCAvgFunc, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCMaxFunc, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCMinFunc, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCSum2Func, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceRSumFunc, (sdepth, ddepth),
|
||||
|
||||
@@ -7,12 +7,20 @@
|
||||
namespace cv {
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
typedef void (*ReduceSumFunc)(const Mat& src, Mat& dst);
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
#include "reduce_c.simd.hpp"
|
||||
#endif
|
||||
|
||||
// =====================================================================
|
||||
// Col reduce SUM (dim=1): sum each row into cn output values
|
||||
// =====================================================================
|
||||
@@ -1089,7 +1097,7 @@ static void reduceRowSum_64f64f(const Mat& srcmat, Mat& dstmat)
|
||||
// Dispatchers
|
||||
// =====================================================================
|
||||
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_32S) return reduceColSum_8u32s;
|
||||
@@ -1108,7 +1116,68 @@ ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth)
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth)
|
||||
{
|
||||
return getReduceCSumFunc(sdepth, ddepth);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_8U) return reduceColMax_8u;
|
||||
if (sdepth == CV_16U && ddepth == CV_16U) return reduceColMax_16u;
|
||||
if (sdepth == CV_16S && ddepth == CV_16S) return reduceColMax_16s;
|
||||
if (sdepth == CV_32F && ddepth == CV_32F) return reduceColMax_32f;
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
if (sdepth == CV_64F && ddepth == CV_64F) return reduceColMax_64f;
|
||||
#endif
|
||||
#else
|
||||
CV_UNUSED(sdepth);
|
||||
CV_UNUSED(ddepth);
|
||||
#endif
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_8U) return reduceColMin_8u;
|
||||
if (sdepth == CV_16U && ddepth == CV_16U) return reduceColMin_16u;
|
||||
if (sdepth == CV_16S && ddepth == CV_16S) return reduceColMin_16s;
|
||||
if (sdepth == CV_32F && ddepth == CV_32F) return reduceColMin_32f;
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
if (sdepth == CV_64F && ddepth == CV_64F) return reduceColMin_64f;
|
||||
#endif
|
||||
#else
|
||||
CV_UNUSED(sdepth);
|
||||
CV_UNUSED(ddepth);
|
||||
#endif
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_32S) return reduceColSum2_8u32s;
|
||||
if (sdepth == CV_8U && ddepth == CV_32F) return reduceColSum2_8u32f;
|
||||
if (sdepth == CV_8U && ddepth == CV_64F) return reduceColSum2_8u64f;
|
||||
if (sdepth == CV_16U && ddepth == CV_32F) return reduceColSum2_16u32f;
|
||||
if (sdepth == CV_16S && ddepth == CV_32F) return reduceColSum2_16s32f;
|
||||
if (sdepth == CV_32F && ddepth == CV_32F) return reduceColSum2_32f32f;
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
if (sdepth == CV_16U && ddepth == CV_64F) return reduceColSum2_16u64f;
|
||||
if (sdepth == CV_16S && ddepth == CV_64F) return reduceColSum2_16s64f;
|
||||
if (sdepth == CV_32F && ddepth == CV_64F) return reduceColSum2_32f64f;
|
||||
if (sdepth == CV_64F && ddepth == CV_64F) return reduceColSum2_64f64f;
|
||||
#endif
|
||||
#else
|
||||
CV_UNUSED(sdepth);
|
||||
CV_UNUSED(ddepth);
|
||||
#endif
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_32S) return reduceRowSum_8u32s;
|
||||
|
||||
@@ -0,0 +1,539 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#include "reduce_c_generic.hpp"
|
||||
|
||||
#if CV_RVV
|
||||
#include "reduce_c_rvv.hpp"
|
||||
#endif
|
||||
|
||||
#if CV_NEON
|
||||
#include "reduce_c_neon.hpp"
|
||||
#endif
|
||||
|
||||
#if CV_AVX2
|
||||
#include "reduce_c_avx2.hpp"
|
||||
#endif
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax8uC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax8uC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax8uC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_8uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax8uC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax8uC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax8uC3<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_8uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax8uC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax8uC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax8uC4<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_8uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cn = srcmat.channels();
|
||||
if (cn == 1)
|
||||
reduceColMinMax_8uC1<isMax>(srcmat, dstmat);
|
||||
else if (cn == 3)
|
||||
reduceColMinMax_8uC3<isMax>(srcmat, dstmat);
|
||||
else if (cn == 4)
|
||||
reduceColMinMax_8uC4<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_8uFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_8u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_8u<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_8u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_8u<false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16uC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16uC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16uC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16uC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16uC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16uC4<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16uC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16uC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16uC3<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() == 1)
|
||||
reduceColMinMax_16uC1<isMax>(srcmat, dstmat);
|
||||
else if (srcmat.channels() == 3)
|
||||
reduceColMinMax_16uC3<isMax>(srcmat, dstmat);
|
||||
else if (srcmat.channels() == 4)
|
||||
reduceColMinMax_16uC4<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_16uFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_16u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_16u<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_16u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_16u<false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16sC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16sC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16sC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16sC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16sFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16sC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16sC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16sC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16sC4<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16sFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16sC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16sC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16sC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16sC3<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16sFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16s(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() == 1)
|
||||
reduceColMinMax_16sC1<isMax>(srcmat, dstmat);
|
||||
else if (srcmat.channels() == 3)
|
||||
reduceColMinMax_16sC3<isMax>(srcmat, dstmat);
|
||||
else if (srcmat.channels() == 4)
|
||||
reduceColMinMax_16sC4<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_16sFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_16s(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_16s<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_16s(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_16s<false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax32fC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax32fC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax32fC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_32fFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32fC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax32fC3<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_32fFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax32fC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax32fC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax32fC4<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_32fFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cn = srcmat.channels();
|
||||
if (cn == 1)
|
||||
reduceColMinMax_32fC1<isMax>(srcmat, dstmat);
|
||||
else if (cn == 3)
|
||||
reduceColMinMax_32fC3<isMax>(srcmat, dstmat);
|
||||
else if (cn == 4)
|
||||
reduceColMinMax_32fC4<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_32fFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_32f<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_32f<false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::minMax64fC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::minMax64fC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax64fC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_64fFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() == 1)
|
||||
reduceColMinMax_64fC1<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_64fFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_64f<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_64f<false>(srcmat, dstmat);
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_8uC1<DT>(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_8uC1<DT>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_8uC1<DT>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_8uFallback<DT>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_8uC3<DT>(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_8uC3<DT>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_8uC3<DT>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_8uFallback<DT>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_8uC4<DT>(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_8uC4<DT>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_8uC4<DT>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_8uFallback<DT>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cn = srcmat.channels();
|
||||
if (cn == 1)
|
||||
reduceColSum2_8uC1<DT>(srcmat, dstmat);
|
||||
else if (cn == 3)
|
||||
reduceColSum2_8uC3<DT>(srcmat, dstmat);
|
||||
else if (cn == 4)
|
||||
reduceColSum2_8uC4<DT>(srcmat, dstmat);
|
||||
else
|
||||
reduceColSum2_8uFallback<DT>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_8u32s(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColSum2_8u<int>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_8u32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColSum2_8u<float>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_8u64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColSum2_8u<double>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_16u32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_16u32fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_16u32fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_16u32fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_16u32fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_16u32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_16s32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_16s32fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_16s32fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_16s32fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_16s32fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_16s32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_32fC1(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::sum2_32fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_32fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_32f32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32fC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_32fC3(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_32f32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_32fC4(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::sum2_32fC4(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_32fC4(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_32f32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cn = srcmat.channels();
|
||||
if (cn == 1)
|
||||
reduceColSum2_32f32fC1(srcmat, dstmat);
|
||||
else if (cn == 3)
|
||||
reduceColSum2_32f32fC3(srcmat, dstmat);
|
||||
else if (cn == 4)
|
||||
reduceColSum2_32f32fC4(srcmat, dstmat);
|
||||
else
|
||||
reduceColSum2_32f32fFallback(srcmat, dstmat);
|
||||
}
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
static void reduceColSum2_16u64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_16u64fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::sum2_16u64fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_16u64fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_16u64fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_16u64fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_16s64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_16s64fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::sum2_16s64fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_16s64fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_16s64fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_16s64fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_32f64fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::sum2_32f64fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_32f64fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_32f64fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_32f64fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_64f64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_64f64fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::sum2_64f64fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_64f64fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_64f64fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_64f64fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,989 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
namespace reduce_c_avx2
|
||||
{
|
||||
|
||||
// Optimized ReduceC support in this backend:
|
||||
//
|
||||
// | Input -> output type/channel | SUM | AVG | MIN | MAX | SUM2 |
|
||||
// |------------------------------------|:---:|:---:|:---:|:---:|:----:|
|
||||
// | 8UC1/C3/C4 -> 8UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 8UC1/C3/C4 -> 32SC1/C3/C4 | x | x | - | - | x |
|
||||
// | 8UC1/C3/C4 -> 32FC1/C3/C4 | x | x | - | - | x |
|
||||
// | 8UC1/C3/C4 -> 64FC1/C3/C4 | - | - | - | - | x |
|
||||
// | 16UC1/C3/C4 -> 16UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16UC1 -> 32FC1 | x | x | - | - | x |
|
||||
// | 16UC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16UC1 -> 64FC1 | - | - | - | - | x |
|
||||
// | 16UC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 16SC1/C3/C4 -> 16SC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16SC1 -> 32FC1 | x | x | - | - | x |
|
||||
// | 16SC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16SC1 -> 64FC1 | - | - | - | - | x |
|
||||
// | 16SC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 32FC1 -> 32FC1 | x | x | x | x | x |
|
||||
// | 32FC3 -> 32FC3 | x | x | - | - | - |
|
||||
// | 32FC4 -> 32FC4 | x | x | x | x | x |
|
||||
// | 32FC1 -> 64FC1 | x | x | - | - | x |
|
||||
// | 32FC3/C4 -> 64FC3/C4 | x | x | - | - | - |
|
||||
// | 64FC1 -> 64FC1 | x | x | x | x | x |
|
||||
// | 64FC3/C4 -> 64FC3/C4 | x | x | - | - | - |
|
||||
//
|
||||
// 'x' in SUM/AVG denotes the existing shared universal-intrinsics kernel; 'x'
|
||||
// in MIN/MAX/SUM2 denotes a native AVX2 kernel. For legal MIN/MAX/SUM2
|
||||
// combinations marked '-', and for other channel counts, dispatch uses the
|
||||
// shared generic fallback.
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
uchar result = isMax ? 0 : UCHAR_MAX;
|
||||
int x = 0;
|
||||
__m256i acc = _mm256_set1_epi8((char)result);
|
||||
for (; x <= cols - 32; x += 32)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
acc = isMax ? _mm256_max_epu8(acc, v) : _mm256_min_epu8(acc, v);
|
||||
}
|
||||
uchar lanes[32];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc);
|
||||
for (int i = 0; i < 32; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dst[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort result = isMax ? 0 : USHRT_MAX;
|
||||
int x = 0;
|
||||
__m256i acc = _mm256_set1_epi16((short)result);
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
acc = isMax ? _mm256_max_epu16(acc, v) : _mm256_min_epu16(acc, v);
|
||||
}
|
||||
ushort lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc);
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<ushort>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short result = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
int x = 0;
|
||||
__m256i acc = _mm256_set1_epi16(result);
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
acc = isMax ? _mm256_max_epi16(acc, v) : _mm256_min_epi16(acc, v);
|
||||
}
|
||||
short lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc);
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<short>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
const __m256i accInit = _mm256_set1_epi16((short)initial);
|
||||
const __m256i validMask = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(-1, -1, -1, -1, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0));
|
||||
const __m256i masks[4] = {
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(0, 1, 8, 9, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(2, 3, 10, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(4, 5, 12, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(6, 7, 14, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1))
|
||||
};
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort* dst = dstmat.ptr<ushort>(y);
|
||||
__m256i acc[4] = {accInit, accInit, accInit, accInit};
|
||||
int x = 0;
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 4));
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
__m256i vc = _mm256_shuffle_epi8(v, masks[c]);
|
||||
vc = _mm256_blendv_epi8(accInit, vc, validMask);
|
||||
acc[c] = isMax ? _mm256_max_epu16(acc[c], vc)
|
||||
: _mm256_min_epu16(acc[c], vc);
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
ushort lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc[c]);
|
||||
ushort result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 4 + c])
|
||||
: std::min(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
const __m256i accInit = _mm256_set1_epi16(initial);
|
||||
const __m256i validMask = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(-1, -1, -1, -1, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0));
|
||||
const __m256i masks[4] = {
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(0, 1, 8, 9, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(2, 3, 10, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(4, 5, 12, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(6, 7, 14, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1))
|
||||
};
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short* dst = dstmat.ptr<short>(y);
|
||||
__m256i acc[4] = {accInit, accInit, accInit, accInit};
|
||||
int x = 0;
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 4));
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
__m256i vc = _mm256_shuffle_epi8(v, masks[c]);
|
||||
vc = _mm256_blendv_epi8(accInit, vc, validMask);
|
||||
acc[c] = isMax ? _mm256_max_epi16(acc[c], vc)
|
||||
: _mm256_min_epi16(acc[c], vc);
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
short lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc[c]);
|
||||
short result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 4 + c])
|
||||
: std::min(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename T, bool isMax, bool isSigned>
|
||||
static void minMax16C3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const T initial = isMax ? std::numeric_limits<T>::lowest() : std::numeric_limits<T>::max();
|
||||
const __m256i accInit = _mm256_set1_epi16((short)initial);
|
||||
const __m256i masks[3] = {
|
||||
_mm256_setr_epi8(0, 1, 6, 7, 12, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
2, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1),
|
||||
_mm256_setr_epi8(2, 3, 8, 9, 14, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
4, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1),
|
||||
_mm256_setr_epi8(4, 5, 10, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 1, 6, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)
|
||||
};
|
||||
const __m256i validMasks[3] = {
|
||||
_mm256_setr_epi8(-1, -1, -1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
-1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0),
|
||||
_mm256_setr_epi8(-1, -1, -1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
-1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0),
|
||||
_mm256_setr_epi8(-1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
-1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0)
|
||||
};
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const T* src = srcmat.ptr<T>(y);
|
||||
T* dst = dstmat.ptr<T>(y);
|
||||
__m256i acc[3] = {accInit, accInit, accInit};
|
||||
int x = 0;
|
||||
for (; x <= cols - 6; x += 4)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 3));
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
__m256i vc = _mm256_shuffle_epi8(v, masks[c]);
|
||||
vc = _mm256_blendv_epi8(accInit, vc, validMasks[c]);
|
||||
if (isSigned)
|
||||
acc[c] = isMax ? _mm256_max_epi16(acc[c], vc) : _mm256_min_epi16(acc[c], vc);
|
||||
else
|
||||
acc[c] = isMax ? _mm256_max_epu16(acc[c], vc) : _mm256_min_epu16(acc[c], vc);
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
T lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc[c]);
|
||||
T result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 3 + c])
|
||||
: std::min(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
minMax16C3<ushort, isMax, false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
minMax16C3<short, isMax, true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
const __m256i accInit = _mm256_set1_epi8((char)initial);
|
||||
const __m256i mask0 = _mm256_setr_epi8(
|
||||
0, 3, 6, 9, 12, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
2, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i mask1 = _mm256_setr_epi8(
|
||||
1, 4, 7, 10, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 3, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i mask2 = _mm256_setr_epi8(
|
||||
2, 5, 8, 11, 14, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
1, 4, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i invalid0 = _mm256_setr_epi8(
|
||||
0, 0, 0, 0, 0, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i invalid12 = _mm256_setr_epi8(
|
||||
0, 0, 0, 0, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 0, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
__m256i acc0 = accInit, acc1 = accInit, acc2 = accInit;
|
||||
int x = 0;
|
||||
for (; x <= cols - 11; x += 8)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 3));
|
||||
__m256i v0 = _mm256_shuffle_epi8(v, mask0);
|
||||
__m256i v1 = _mm256_shuffle_epi8(v, mask1);
|
||||
__m256i v2 = _mm256_shuffle_epi8(v, mask2);
|
||||
if (!isMax)
|
||||
{
|
||||
v0 = _mm256_or_si256(v0, invalid0);
|
||||
v1 = _mm256_or_si256(v1, invalid12);
|
||||
v2 = _mm256_or_si256(v2, invalid12);
|
||||
}
|
||||
acc0 = isMax ? _mm256_max_epu8(acc0, v0) : _mm256_min_epu8(acc0, v0);
|
||||
acc1 = isMax ? _mm256_max_epu8(acc1, v1) : _mm256_min_epu8(acc1, v1);
|
||||
acc2 = isMax ? _mm256_max_epu8(acc2, v2) : _mm256_min_epu8(acc2, v2);
|
||||
}
|
||||
|
||||
__m256i accs[3] = {acc0, acc1, acc2};
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
uchar lanes[32];
|
||||
_mm256_storeu_si256((__m256i*)lanes, accs[c]);
|
||||
uchar result = initial;
|
||||
for (int i = 0; i < 32; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
const __m256i accInit = _mm256_set1_epi8((char)initial);
|
||||
const __m256i maskInvalid = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(0, 0, 0, 0, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask0 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(0, 4, 8, 12, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask1 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(1, 5, 9, 13, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask2 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(2, 6, 10, 14, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask3 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(3, 7, 11, 15, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
__m256i acc0 = accInit, acc1 = accInit, acc2 = accInit, acc3 = accInit;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 4));
|
||||
__m256i v0 = _mm256_shuffle_epi8(v, mask0);
|
||||
__m256i v1 = _mm256_shuffle_epi8(v, mask1);
|
||||
__m256i v2 = _mm256_shuffle_epi8(v, mask2);
|
||||
__m256i v3 = _mm256_shuffle_epi8(v, mask3);
|
||||
if (!isMax)
|
||||
{
|
||||
v0 = _mm256_or_si256(v0, maskInvalid);
|
||||
v1 = _mm256_or_si256(v1, maskInvalid);
|
||||
v2 = _mm256_or_si256(v2, maskInvalid);
|
||||
v3 = _mm256_or_si256(v3, maskInvalid);
|
||||
}
|
||||
acc0 = isMax ? _mm256_max_epu8(acc0, v0) : _mm256_min_epu8(acc0, v0);
|
||||
acc1 = isMax ? _mm256_max_epu8(acc1, v1) : _mm256_min_epu8(acc1, v1);
|
||||
acc2 = isMax ? _mm256_max_epu8(acc2, v2) : _mm256_min_epu8(acc2, v2);
|
||||
acc3 = isMax ? _mm256_max_epu8(acc3, v3) : _mm256_min_epu8(acc3, v3);
|
||||
}
|
||||
|
||||
__m256i accs[4] = {acc0, acc1, acc2, acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
uchar lanes[32];
|
||||
_mm256_storeu_si256((__m256i*)lanes, accs[c]);
|
||||
uchar result = initial;
|
||||
for (int i = 0; i < 32; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = src[0];
|
||||
int x = 0;
|
||||
__m256 acc = _mm256_set1_ps(result);
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x);
|
||||
acc = isMax ? _mm256_max_ps(acc, v) : _mm256_min_ps(acc, v);
|
||||
}
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const float initial = isMax ? std::numeric_limits<float>::lowest() : std::numeric_limits<float>::max();
|
||||
const __m256 accInit = _mm256_set1_ps(initial);
|
||||
const __m256 validMask = _mm256_castsi256_ps(_mm256_setr_epi32(-1, -1, 0, 0, 0, 0, 0, 0));
|
||||
const __m256i idx0 = _mm256_setr_epi32(0, 4, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx1 = _mm256_setr_epi32(1, 5, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx2 = _mm256_setr_epi32(2, 6, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx3 = _mm256_setr_epi32(3, 7, 0, 0, 0, 0, 0, 0);
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
__m256 acc0 = accInit, acc1 = accInit, acc2 = accInit, acc3 = accInit;
|
||||
int x = 0;
|
||||
for (; x <= cols - 2; x += 2)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x * 4);
|
||||
__m256 v0 = _mm256_blendv_ps(accInit, _mm256_permutevar8x32_ps(v, idx0), validMask);
|
||||
__m256 v1 = _mm256_blendv_ps(accInit, _mm256_permutevar8x32_ps(v, idx1), validMask);
|
||||
__m256 v2 = _mm256_blendv_ps(accInit, _mm256_permutevar8x32_ps(v, idx2), validMask);
|
||||
__m256 v3 = _mm256_blendv_ps(accInit, _mm256_permutevar8x32_ps(v, idx3), validMask);
|
||||
acc0 = isMax ? _mm256_max_ps(acc0, v0) : _mm256_min_ps(acc0, v0);
|
||||
acc1 = isMax ? _mm256_max_ps(acc1, v1) : _mm256_min_ps(acc1, v1);
|
||||
acc2 = isMax ? _mm256_max_ps(acc2, v2) : _mm256_min_ps(acc2, v2);
|
||||
acc3 = isMax ? _mm256_max_ps(acc3, v3) : _mm256_min_ps(acc3, v3);
|
||||
}
|
||||
|
||||
__m256 accs[4] = {acc0, acc1, acc2, acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, accs[c]);
|
||||
float result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const double initial = isMax ? std::numeric_limits<double>::lowest()
|
||||
: std::numeric_limits<double>::max();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
double result = initial;
|
||||
int x = 0;
|
||||
__m256d acc0 = _mm256_set1_pd(initial);
|
||||
__m256d acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256d v0 = _mm256_loadu_pd(src + x);
|
||||
__m256d v1 = _mm256_loadu_pd(src + x + 4);
|
||||
__m256d v2 = _mm256_loadu_pd(src + x + 8);
|
||||
__m256d v3 = _mm256_loadu_pd(src + x + 12);
|
||||
acc0 = isMax ? _mm256_max_pd(acc0, v0) : _mm256_min_pd(acc0, v0);
|
||||
acc1 = isMax ? _mm256_max_pd(acc1, v1) : _mm256_min_pd(acc1, v1);
|
||||
acc2 = isMax ? _mm256_max_pd(acc2, v2) : _mm256_min_pd(acc2, v2);
|
||||
acc3 = isMax ? _mm256_max_pd(acc3, v3) : _mm256_min_pd(acc3, v3);
|
||||
}
|
||||
acc0 = isMax ? _mm256_max_pd(acc0, acc1) : _mm256_min_pd(acc0, acc1);
|
||||
acc2 = isMax ? _mm256_max_pd(acc2, acc3) : _mm256_min_pd(acc2, acc3);
|
||||
acc0 = isMax ? _mm256_max_pd(acc0, acc2) : _mm256_min_pd(acc0, acc2);
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t result = 0;
|
||||
int x = 0;
|
||||
__m256i acc = _mm256_setzero_si256();
|
||||
for (; x <= cols - 32; x += 32)
|
||||
{
|
||||
__m256i bytes = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m128i lo = _mm256_castsi256_si128(bytes);
|
||||
__m128i hi = _mm256_extracti128_si256(bytes, 1);
|
||||
__m256i lo16 = _mm256_cvtepu8_epi16(lo);
|
||||
__m256i hi16 = _mm256_cvtepu8_epi16(hi);
|
||||
acc = _mm256_add_epi32(acc, _mm256_madd_epi16(lo16, lo16));
|
||||
acc = _mm256_add_epi32(acc, _mm256_madd_epi16(hi16, hi16));
|
||||
}
|
||||
uint32_t lanes[8];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
result += (uint32_t)src[x] * src[x];
|
||||
dst[0] = (DT)(int32_t)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
__m256 acc0 = _mm256_setzero_ps(), acc1 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m256 v0 = _mm256_cvtepi32_ps(_mm256_cvtepu16_epi32(_mm256_castsi256_si128(v)));
|
||||
__m256 v1 = _mm256_cvtepi32_ps(_mm256_cvtepu16_epi32(_mm256_extracti128_si256(v, 1)));
|
||||
acc0 = _mm256_add_ps(acc0, _mm256_mul_ps(v0, v0));
|
||||
acc1 = _mm256_add_ps(acc1, _mm256_mul_ps(v1, v1));
|
||||
}
|
||||
acc0 = _mm256_add_ps(acc0, acc1);
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, acc0);
|
||||
float result = 0;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
float value = (float)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
__m256 acc0 = _mm256_setzero_ps(), acc1 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m256 v0 = _mm256_cvtepi32_ps(_mm256_cvtepi16_epi32(_mm256_castsi256_si128(v)));
|
||||
__m256 v1 = _mm256_cvtepi32_ps(_mm256_cvtepi16_epi32(_mm256_extracti128_si256(v, 1)));
|
||||
acc0 = _mm256_add_ps(acc0, _mm256_mul_ps(v0, v0));
|
||||
acc1 = _mm256_add_ps(acc1, _mm256_mul_ps(v1, v1));
|
||||
}
|
||||
acc0 = _mm256_add_ps(acc0, acc1);
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, acc0);
|
||||
float result = 0;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
float value = (float)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
__m256d acc0 = _mm256_setzero_pd(), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m256i v0 = _mm256_cvtepu16_epi32(_mm256_castsi256_si128(v));
|
||||
__m256i v1 = _mm256_cvtepu16_epi32(_mm256_extracti128_si256(v, 1));
|
||||
__m256d d0 = _mm256_cvtepi32_pd(_mm256_castsi256_si128(v0));
|
||||
__m256d d1 = _mm256_cvtepi32_pd(_mm256_extracti128_si256(v0, 1));
|
||||
__m256d d2 = _mm256_cvtepi32_pd(_mm256_castsi256_si128(v1));
|
||||
__m256d d3 = _mm256_cvtepi32_pd(_mm256_extracti128_si256(v1, 1));
|
||||
acc0 = _mm256_add_pd(acc0, _mm256_mul_pd(d0, d0));
|
||||
acc1 = _mm256_add_pd(acc1, _mm256_mul_pd(d1, d1));
|
||||
acc2 = _mm256_add_pd(acc2, _mm256_mul_pd(d2, d2));
|
||||
acc3 = _mm256_add_pd(acc3, _mm256_mul_pd(d3, d3));
|
||||
}
|
||||
acc0 = _mm256_add_pd(_mm256_add_pd(acc0, acc1), _mm256_add_pd(acc2, acc3));
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1] + lanes[2] + lanes[3];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
double value = (double)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
__m256d acc0 = _mm256_setzero_pd(), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m256i v0 = _mm256_cvtepi16_epi32(_mm256_castsi256_si128(v));
|
||||
__m256i v1 = _mm256_cvtepi16_epi32(_mm256_extracti128_si256(v, 1));
|
||||
__m256d d0 = _mm256_cvtepi32_pd(_mm256_castsi256_si128(v0));
|
||||
__m256d d1 = _mm256_cvtepi32_pd(_mm256_extracti128_si256(v0, 1));
|
||||
__m256d d2 = _mm256_cvtepi32_pd(_mm256_castsi256_si128(v1));
|
||||
__m256d d3 = _mm256_cvtepi32_pd(_mm256_extracti128_si256(v1, 1));
|
||||
acc0 = _mm256_add_pd(acc0, _mm256_mul_pd(d0, d0));
|
||||
acc1 = _mm256_add_pd(acc1, _mm256_mul_pd(d1, d1));
|
||||
acc2 = _mm256_add_pd(acc2, _mm256_mul_pd(d2, d2));
|
||||
acc3 = _mm256_add_pd(acc3, _mm256_mul_pd(d3, d3));
|
||||
}
|
||||
acc0 = _mm256_add_pd(_mm256_add_pd(acc0, acc1), _mm256_add_pd(acc2, acc3));
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1] + lanes[2] + lanes[3];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
double value = (double)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
__m256d acc0 = _mm256_setzero_pd(), acc1 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x);
|
||||
__m256d v0 = _mm256_cvtps_pd(_mm256_castps256_ps128(v));
|
||||
__m256d v1 = _mm256_cvtps_pd(_mm256_extractf128_ps(v, 1));
|
||||
acc0 = _mm256_add_pd(acc0, _mm256_mul_pd(v0, v0));
|
||||
acc1 = _mm256_add_pd(acc1, _mm256_mul_pd(v1, v1));
|
||||
}
|
||||
acc0 = _mm256_add_pd(acc0, acc1);
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1] + lanes[2] + lanes[3];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
double value = (double)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_64f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
__m256d acc0 = _mm256_setzero_pd(), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256d v0 = _mm256_loadu_pd(src + x);
|
||||
__m256d v1 = _mm256_loadu_pd(src + x + 4);
|
||||
__m256d v2 = _mm256_loadu_pd(src + x + 8);
|
||||
__m256d v3 = _mm256_loadu_pd(src + x + 12);
|
||||
acc0 = _mm256_add_pd(acc0, _mm256_mul_pd(v0, v0));
|
||||
acc1 = _mm256_add_pd(acc1, _mm256_mul_pd(v1, v1));
|
||||
acc2 = _mm256_add_pd(acc2, _mm256_mul_pd(v2, v2));
|
||||
acc3 = _mm256_add_pd(acc3, _mm256_mul_pd(v3, v3));
|
||||
}
|
||||
acc0 = _mm256_add_pd(_mm256_add_pd(acc0, acc1), _mm256_add_pd(acc2, acc3));
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1] + lanes[2] + lanes[3];
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const __m256i mask0 = _mm256_setr_epi8(
|
||||
0, 3, 6, 9, 12, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
2, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i mask1 = _mm256_setr_epi8(
|
||||
1, 4, 7, 10, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 3, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i mask2 = _mm256_setr_epi8(
|
||||
2, 5, 8, 11, 14, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
1, 4, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
__m256i acc0 = _mm256_setzero_si256();
|
||||
__m256i acc1 = _mm256_setzero_si256();
|
||||
__m256i acc2 = _mm256_setzero_si256();
|
||||
int x = 0;
|
||||
for (; x <= cols - 11; x += 8)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 3));
|
||||
__m256i channels[3] = {
|
||||
_mm256_shuffle_epi8(v, mask0),
|
||||
_mm256_shuffle_epi8(v, mask1),
|
||||
_mm256_shuffle_epi8(v, mask2)
|
||||
};
|
||||
__m256i* accs[3] = {&acc0, &acc1, &acc2};
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
__m128i lo = _mm256_castsi256_si128(channels[c]);
|
||||
__m128i hi = _mm256_extracti128_si256(channels[c], 1);
|
||||
__m256i lo16 = _mm256_cvtepu8_epi16(lo);
|
||||
__m256i hi16 = _mm256_cvtepu8_epi16(hi);
|
||||
*accs[c] = _mm256_add_epi32(*accs[c], _mm256_madd_epi16(lo16, lo16));
|
||||
*accs[c] = _mm256_add_epi32(*accs[c], _mm256_madd_epi16(hi16, hi16));
|
||||
}
|
||||
}
|
||||
|
||||
__m256i accs[3] = {acc0, acc1, acc2};
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
uint32_t lanes[8];
|
||||
uint32_t result = 0;
|
||||
_mm256_storeu_si256((__m256i*)lanes, accs[c]);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (int i = x; i < cols; i++)
|
||||
{
|
||||
uint32_t value = src[i * 3 + c];
|
||||
result += value * value;
|
||||
}
|
||||
dst[c] = (DT)(int32_t)result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const __m256i mask0 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(0, 4, 8, 12, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask1 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(1, 5, 9, 13, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask2 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(2, 6, 10, 14, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask3 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(3, 7, 11, 15, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
__m256i acc0 = _mm256_setzero_si256();
|
||||
__m256i acc1 = _mm256_setzero_si256();
|
||||
__m256i acc2 = _mm256_setzero_si256();
|
||||
__m256i acc3 = _mm256_setzero_si256();
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 4));
|
||||
__m256i channels[4] = {
|
||||
_mm256_shuffle_epi8(v, mask0),
|
||||
_mm256_shuffle_epi8(v, mask1),
|
||||
_mm256_shuffle_epi8(v, mask2),
|
||||
_mm256_shuffle_epi8(v, mask3)
|
||||
};
|
||||
__m256i* accs[4] = {&acc0, &acc1, &acc2, &acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
__m128i lo = _mm256_castsi256_si128(channels[c]);
|
||||
__m128i hi = _mm256_extracti128_si256(channels[c], 1);
|
||||
__m256i lo16 = _mm256_cvtepu8_epi16(lo);
|
||||
__m256i hi16 = _mm256_cvtepu8_epi16(hi);
|
||||
*accs[c] = _mm256_add_epi32(*accs[c], _mm256_madd_epi16(lo16, lo16));
|
||||
*accs[c] = _mm256_add_epi32(*accs[c], _mm256_madd_epi16(hi16, hi16));
|
||||
}
|
||||
}
|
||||
|
||||
__m256i accs[4] = {acc0, acc1, acc2, acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
uint32_t lanes[8];
|
||||
uint32_t result = 0;
|
||||
_mm256_storeu_si256((__m256i*)lanes, accs[c]);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (int i = x; i < cols; i++)
|
||||
{
|
||||
uint32_t value = src[i * 4 + c];
|
||||
result += value * value;
|
||||
}
|
||||
dst[c] = (DT)(int32_t)result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = 0;
|
||||
int x = 0;
|
||||
__m256 acc = _mm256_setzero_ps();
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x);
|
||||
acc = _mm256_add_ps(acc, _mm256_mul_ps(v, v));
|
||||
}
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const __m256 zero = _mm256_setzero_ps();
|
||||
const __m256 validMask = _mm256_castsi256_ps(_mm256_setr_epi32(-1, -1, 0, 0, 0, 0, 0, 0));
|
||||
const __m256i idx0 = _mm256_setr_epi32(0, 4, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx1 = _mm256_setr_epi32(1, 5, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx2 = _mm256_setr_epi32(2, 6, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx3 = _mm256_setr_epi32(3, 7, 0, 0, 0, 0, 0, 0);
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
__m256 acc0 = zero, acc1 = zero, acc2 = zero, acc3 = zero;
|
||||
int x = 0;
|
||||
for (; x <= cols - 2; x += 2)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x * 4);
|
||||
__m256 v0 = _mm256_blendv_ps(zero, _mm256_permutevar8x32_ps(v, idx0), validMask);
|
||||
__m256 v1 = _mm256_blendv_ps(zero, _mm256_permutevar8x32_ps(v, idx1), validMask);
|
||||
__m256 v2 = _mm256_blendv_ps(zero, _mm256_permutevar8x32_ps(v, idx2), validMask);
|
||||
__m256 v3 = _mm256_blendv_ps(zero, _mm256_permutevar8x32_ps(v, idx3), validMask);
|
||||
acc0 = _mm256_add_ps(acc0, _mm256_mul_ps(v0, v0));
|
||||
acc1 = _mm256_add_ps(acc1, _mm256_mul_ps(v1, v1));
|
||||
acc2 = _mm256_add_ps(acc2, _mm256_mul_ps(v2, v2));
|
||||
acc3 = _mm256_add_ps(acc3, _mm256_mul_ps(v3, v3));
|
||||
}
|
||||
|
||||
__m256 accs[4] = {acc0, acc1, acc2, acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
float lanes[8];
|
||||
float result = 0;
|
||||
_mm256_storeu_ps(lanes, accs[c]);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (int i = x; i < cols; i++)
|
||||
result += src[i * 4 + c] * src[i * 4 + c];
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
} // namespace reduce_c_avx2
|
||||
@@ -0,0 +1,701 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
template<typename T, typename VT>
|
||||
static inline VT vx_load_strided(const T *ptr, size_t step)
|
||||
{
|
||||
constexpr int nlanes = VTraits<VT>::max_nlanes;
|
||||
T buf[nlanes];
|
||||
for (int i = 0; i < VTraits<VT>::vlanes(); i++)
|
||||
{
|
||||
buf[i] = *ptr;
|
||||
ptr += step;
|
||||
}
|
||||
return vx_load(buf);
|
||||
}
|
||||
|
||||
#if CV_RVV
|
||||
template<>
|
||||
inline v_uint8 vx_load_strided(const uchar *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse8_v_u8m2(ptr, step * sizeof(uchar), __riscv_vsetvlmax_e8m2());
|
||||
}
|
||||
template<>
|
||||
inline v_uint16 vx_load_strided(const ushort *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse16_v_u16m2(ptr, step * sizeof(ushort), __riscv_vsetvlmax_e16m2());
|
||||
}
|
||||
template<>
|
||||
inline v_int16 vx_load_strided(const short *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse16_v_i16m2(ptr, step * sizeof(short), __riscv_vsetvlmax_e16m2());
|
||||
}
|
||||
template<>
|
||||
inline v_int32 vx_load_strided(const int *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse32_v_i32m2(ptr, step * sizeof(int), __riscv_vsetvlmax_e32m2());
|
||||
}
|
||||
template<>
|
||||
inline v_float32 vx_load_strided(const float *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse32_v_f32m2(ptr, step * sizeof(float), __riscv_vsetvlmax_e32m2());
|
||||
}
|
||||
template<>
|
||||
inline v_float64 vx_load_strided(const double *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse64_v_f64m2(ptr, step * sizeof(double), __riscv_vsetvlmax_e64m2());
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
template<typename stype, typename itype>
|
||||
struct ReduceOpAddSqr
|
||||
{
|
||||
using v_stype = stype;
|
||||
using v_itype = itype;
|
||||
static const int vlanes;
|
||||
static inline stype load(const stype *ptr, size_t step) { (void)step; return *ptr; }
|
||||
static inline itype init() { return (itype)0; }
|
||||
static inline itype reduce(const itype &val) { return val; }
|
||||
inline itype operator()(const itype &a, const stype &b) const { return a + (itype)b * (itype)b; }
|
||||
};
|
||||
template<typename stype, typename itype>
|
||||
const int ReduceOpAddSqr<stype, itype>::vlanes = 1;
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
template<typename stype, typename itype>
|
||||
struct ReduceVecOpAddSqr;
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<uchar, int>
|
||||
{
|
||||
using stype = uchar;
|
||||
using itype = int;
|
||||
using v_stype = v_uint8;
|
||||
using v_itype = v_int32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_s32(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_uint16 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
|
||||
v_uint32 s00, s01;
|
||||
v_expand(b0, s00, s01);
|
||||
s00 = v_add(s00, s01);
|
||||
v_uint32 s10, s11;
|
||||
v_expand(b1, s10, s11);
|
||||
s10 = v_add(s10, s11);
|
||||
return v_add(a, v_reinterpret_as_s32(v_add(s00, s10)));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<uchar, int>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<ushort, float>
|
||||
{
|
||||
using stype = ushort;
|
||||
using itype = float;
|
||||
using v_stype = v_uint16;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f32(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_uint32 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
v_int32 sb = v_reinterpret_as_s32(v_add(b0, b1));
|
||||
return v_add(a, v_cvt_f32(sb));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<ushort, float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<short, float>
|
||||
{
|
||||
using stype = short;
|
||||
using itype = float;
|
||||
using v_stype = v_int16;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f32(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_int32 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
v_int32 sb = v_add(b0, b1);
|
||||
return v_add(a, v_cvt_f32(sb));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<short, float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<float, float>
|
||||
{
|
||||
using stype = float;
|
||||
using itype = float;
|
||||
using v_stype = v_float32;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f32(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_add(a, v_mul(b, b)); }
|
||||
};
|
||||
const int ReduceVecOpAddSqr<float, float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<short, double>
|
||||
{
|
||||
using stype = short;
|
||||
using itype = double;
|
||||
using v_stype = v_int16;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f64(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_int32 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
v_int32 sb = v_add(b0, b1);
|
||||
return v_add(a, v_add(v_cvt_f64(sb), v_cvt_f64_high(sb)));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<short, double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<ushort, double>
|
||||
{
|
||||
using stype = ushort;
|
||||
using itype = double;
|
||||
using v_stype = v_uint16;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f64(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_uint32 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
v_int32 sb = v_reinterpret_as_s32(v_add(b0, b1));
|
||||
return v_add(a, v_add(v_cvt_f64(sb), v_cvt_f64_high(sb)));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<ushort, double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<float, double>
|
||||
{
|
||||
using stype = float;
|
||||
using itype = double;
|
||||
using v_stype = v_float32;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f64(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_itype b0 = v_cvt_f64(b), b1 = v_cvt_f64_high(b);
|
||||
return v_add(a, v_add(v_mul(b0, b0), v_mul(b1, b1)));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<float, double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<double, double>
|
||||
{
|
||||
using stype = double;
|
||||
using itype = double;
|
||||
using v_stype = v_float64;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f64(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_add(a, v_mul(b, b)); }
|
||||
};
|
||||
const int ReduceVecOpAddSqr<double, double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
template<typename stype>
|
||||
struct ReduceOpMax
|
||||
{
|
||||
using v_stype = stype;
|
||||
using v_itype = stype;
|
||||
static const int vlanes;
|
||||
static inline stype load(const stype *ptr, size_t step) { (void)step; return *ptr; }
|
||||
static inline stype init() { return std::numeric_limits<stype>::lowest(); }
|
||||
static inline stype reduce(const stype &val) { return val; }
|
||||
inline stype operator()(const stype &a, const stype &b) const { return std::max(a, b); }
|
||||
};
|
||||
template<typename stype>
|
||||
const int ReduceOpMax<stype>::vlanes = 1;
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
template<typename stype>
|
||||
struct ReduceVecOpMax;
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<uchar>
|
||||
{
|
||||
using stype = uchar;
|
||||
using itype = uchar;
|
||||
using v_stype = v_uint8;
|
||||
using v_itype = v_uint8;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_u8(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_max(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<uchar>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<ushort>
|
||||
{
|
||||
using stype = ushort;
|
||||
using itype = ushort;
|
||||
using v_stype = v_uint16;
|
||||
using v_itype = v_uint16;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_u16(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_max(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<ushort>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<short>
|
||||
{
|
||||
using stype = short;
|
||||
using itype = short;
|
||||
using v_stype = v_int16;
|
||||
using v_itype = v_int16;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_s16(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_max(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<short>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<float>
|
||||
{
|
||||
using stype = float;
|
||||
using itype = float;
|
||||
using v_stype = v_float32;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_f32(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_max(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<double>
|
||||
{
|
||||
using stype = double;
|
||||
using itype = double;
|
||||
using v_stype = v_float64;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_f64(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val)
|
||||
{
|
||||
constexpr int nlanes = VTraits<v_itype>::max_nlanes;
|
||||
itype buf[nlanes];
|
||||
vx_store(buf, val);
|
||||
itype m = buf[0];
|
||||
for (int i = 1; i < VTraits<v_itype>::vlanes(); i++)
|
||||
{
|
||||
if (m < buf[i]) m = buf[i];
|
||||
}
|
||||
return m;
|
||||
}
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
template<typename stype>
|
||||
struct ReduceOpMin
|
||||
{
|
||||
using v_stype = stype;
|
||||
using v_itype = stype;
|
||||
static const int vlanes;
|
||||
static inline stype load(const stype *ptr, size_t step) { (void)step; return *ptr; }
|
||||
static inline stype init() { return std::numeric_limits<stype>::max(); }
|
||||
static inline stype reduce(const stype &val) { return (stype)val; }
|
||||
inline stype operator()(const stype &a, const stype &b) const { return std::min(a, b); }
|
||||
};
|
||||
template<typename stype>
|
||||
const int ReduceOpMin<stype>::vlanes = 1;
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
template<typename stype>
|
||||
struct ReduceVecOpMin;
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<uchar>
|
||||
{
|
||||
using stype = uchar;
|
||||
using itype = uchar;
|
||||
using v_stype = v_uint8;
|
||||
using v_itype = v_uint8;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_u8(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_min(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<uchar>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<ushort>
|
||||
{
|
||||
using stype = ushort;
|
||||
using itype = ushort;
|
||||
using v_stype = v_uint16;
|
||||
using v_itype = v_uint16;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_u16(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_min(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<ushort>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<short>
|
||||
{
|
||||
using stype = short;
|
||||
using itype = short;
|
||||
using v_stype = v_int16;
|
||||
using v_itype = v_int16;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_s16(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_min(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<short>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<float>
|
||||
{
|
||||
using stype = float;
|
||||
using itype = float;
|
||||
using v_stype = v_float32;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_f32(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_min(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<double>
|
||||
{
|
||||
using stype = double;
|
||||
using itype = double;
|
||||
using v_stype = v_float64;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_f64(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val)
|
||||
{
|
||||
constexpr int nlanes = VTraits<v_itype>::max_nlanes;
|
||||
itype buf[nlanes];
|
||||
vx_store(buf, val);
|
||||
itype m = buf[0];
|
||||
for (int i = 1; i < VTraits<v_itype>::vlanes(); i++)
|
||||
{
|
||||
if (m > buf[i]) m = buf[i];
|
||||
}
|
||||
return m;
|
||||
}
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
|
||||
using ReduceOpAddSqr_8U32S = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceOpAddSqr_8U32F = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceOpAddSqr_8U64F = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceOpAddSqr_16U32F = ReduceOpAddSqr<ushort, float>;
|
||||
using ReduceOpAddSqr_16U64F = ReduceOpAddSqr<ushort, double>;
|
||||
using ReduceOpAddSqr_16S32F = ReduceOpAddSqr<short, float>;
|
||||
using ReduceOpAddSqr_16S64F = ReduceOpAddSqr<short, double>;
|
||||
using ReduceOpAddSqr_32F32F = ReduceOpAddSqr<float, float>;
|
||||
using ReduceOpAddSqr_32F64F = ReduceOpAddSqr<float, double>;
|
||||
using ReduceOpAddSqr_64F64F = ReduceOpAddSqr<double, double>;
|
||||
|
||||
using ReduceOpMax_8U = ReduceOpMax<uchar>;
|
||||
using ReduceOpMax_16U = ReduceOpMax<ushort>;
|
||||
using ReduceOpMax_16S = ReduceOpMax<short>;
|
||||
using ReduceOpMax_32F = ReduceOpMax<float>;
|
||||
using ReduceOpMax_64F = ReduceOpMax<double>;
|
||||
|
||||
using ReduceOpMin_8U = ReduceOpMin<uchar>;
|
||||
using ReduceOpMin_16U = ReduceOpMin<ushort>;
|
||||
using ReduceOpMin_16S = ReduceOpMin<short>;
|
||||
using ReduceOpMin_32F = ReduceOpMin<float>;
|
||||
using ReduceOpMin_64F = ReduceOpMin<double>;
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
using ReduceVecOpAddSqr_8U32S = ReduceVecOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_8U32F = ReduceVecOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_8U64F = ReduceVecOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_16U32F = ReduceVecOpAddSqr<ushort, float>;
|
||||
using ReduceVecOpAddSqr_16S32F = ReduceVecOpAddSqr<short, float>;
|
||||
using ReduceVecOpAddSqr_32F32F = ReduceVecOpAddSqr<float, float>;
|
||||
using ReduceVecOpMax_8U = ReduceVecOpMax<uchar>;
|
||||
using ReduceVecOpMax_16U = ReduceVecOpMax<ushort>;
|
||||
using ReduceVecOpMax_16S = ReduceVecOpMax<short>;
|
||||
using ReduceVecOpMax_32F = ReduceVecOpMax<float>;
|
||||
using ReduceVecOpMin_8U = ReduceVecOpMin<uchar>;
|
||||
using ReduceVecOpMin_16U = ReduceVecOpMin<ushort>;
|
||||
using ReduceVecOpMin_16S = ReduceVecOpMin<short>;
|
||||
using ReduceVecOpMin_32F = ReduceVecOpMin<float>;
|
||||
|
||||
#else
|
||||
|
||||
using ReduceVecOpAddSqr_8U32S = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_8U32F = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_8U64F = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_16U32F = ReduceOpAddSqr<ushort, float>;
|
||||
using ReduceVecOpAddSqr_16S32F = ReduceOpAddSqr<short, float>;
|
||||
using ReduceVecOpAddSqr_32F32F = ReduceOpAddSqr<float, float>;
|
||||
using ReduceVecOpMax_8U = ReduceOpMax<uchar>;
|
||||
using ReduceVecOpMax_16U = ReduceOpMax<ushort>;
|
||||
using ReduceVecOpMax_16S = ReduceOpMax<short>;
|
||||
using ReduceVecOpMax_32F = ReduceOpMax<float>;
|
||||
using ReduceVecOpMin_8U = ReduceOpMin<uchar>;
|
||||
using ReduceVecOpMin_16U = ReduceOpMin<ushort>;
|
||||
using ReduceVecOpMin_16S = ReduceOpMin<short>;
|
||||
using ReduceVecOpMin_32F = ReduceOpMin<float>;
|
||||
|
||||
#endif
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
using ReduceVecOpAddSqr_16U64F = ReduceVecOpAddSqr<ushort, double>;
|
||||
using ReduceVecOpAddSqr_16S64F = ReduceVecOpAddSqr<short, double>;
|
||||
using ReduceVecOpAddSqr_32F64F = ReduceVecOpAddSqr<float, double>;
|
||||
using ReduceVecOpAddSqr_64F64F = ReduceVecOpAddSqr<double, double>;
|
||||
using ReduceVecOpMax_64F = ReduceVecOpMax<double>;
|
||||
using ReduceVecOpMin_64F = ReduceVecOpMin<double>;
|
||||
|
||||
#else
|
||||
|
||||
using ReduceVecOpAddSqr_16U64F = ReduceOpAddSqr<ushort, double>;
|
||||
using ReduceVecOpAddSqr_16S64F = ReduceOpAddSqr<short, double>;
|
||||
using ReduceVecOpAddSqr_32F64F = ReduceOpAddSqr<float, double>;
|
||||
using ReduceVecOpAddSqr_64F64F = ReduceOpAddSqr<double, double>;
|
||||
using ReduceVecOpMax_64F = ReduceOpMax<double>;
|
||||
using ReduceVecOpMin_64F = ReduceOpMin<double>;
|
||||
|
||||
#endif
|
||||
|
||||
template<typename T, typename ST, class Op, class VecOp>
|
||||
class ReduceC_Invoker : public ParallelLoopBody
|
||||
{
|
||||
using WT = typename Op::v_itype;
|
||||
using VT = typename VecOp::v_itype;
|
||||
public:
|
||||
ReduceC_Invoker(const Mat& aSrcmat, Mat& aDstmat, Op& aOp, VecOp& aVop)
|
||||
:srcmat(aSrcmat),dstmat(aDstmat),op(aOp),vop(aVop)
|
||||
{
|
||||
}
|
||||
void operator()(const Range& range) const CV_OVERRIDE
|
||||
{
|
||||
int channels = srcmat.channels();
|
||||
int width = srcmat.cols;
|
||||
|
||||
const int nlanes = VecOp::vlanes;
|
||||
|
||||
for (int h = range.start; h < range.end; h++)
|
||||
{
|
||||
const T *srcrow = srcmat.ptr<T>(h);
|
||||
ST *dst = dstmat.ptr<ST>(h);
|
||||
for (int cn = 0; cn < channels; cn++)
|
||||
{
|
||||
const T *src = srcrow + cn;
|
||||
VT vbuf = vop.init();
|
||||
int w = 0;
|
||||
for (; w <= width - nlanes; w += nlanes)
|
||||
{
|
||||
vbuf = vop(vbuf, vop.load(src+w*channels, channels));
|
||||
}
|
||||
WT wbuf = vop.reduce(vbuf);
|
||||
for (; w < width; w++)
|
||||
{
|
||||
wbuf = op(wbuf, op.load(src+w*channels, channels));
|
||||
}
|
||||
dst[cn] = (ST)op.reduce(wbuf);
|
||||
}
|
||||
}
|
||||
}
|
||||
private:
|
||||
const Mat& srcmat;
|
||||
Mat& dstmat;
|
||||
Op& op;
|
||||
VecOp& vop;
|
||||
};
|
||||
|
||||
template<typename T, typename ST, class Op, class VecOp> static void
|
||||
reduceColGeneric(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
Op op;
|
||||
VecOp vop;
|
||||
|
||||
ReduceC_Invoker<T, ST, Op, VecOp> body(srcmat, dstmat, op, vop);
|
||||
parallel_for_(Range(0, srcmat.size().height), body);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static inline uchar reduceScalarMinMax(uchar a, uchar b)
|
||||
{
|
||||
return isMax ? std::max(a, b) : std::min(a, b);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8uFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<uchar, uchar, ReduceOpMax_8U, ReduceVecOpMax_8U>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<uchar, uchar, ReduceOpMin_8U, ReduceVecOpMin_8U>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16uFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<ushort, ushort, ReduceOpMax_16U, ReduceVecOpMax_16U>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<ushort, ushort, ReduceOpMin_16U, ReduceVecOpMin_16U>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16sFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<short, short, ReduceOpMax_16S, ReduceVecOpMax_16S>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<short, short, ReduceOpMin_16S, ReduceVecOpMin_16S>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static inline float reduceScalarMinMax(float a, float b)
|
||||
{
|
||||
return isMax ? std::max(a, b) : std::min(a, b);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<float, float, ReduceOpMax_32F, ReduceVecOpMax_32F>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<float, float, ReduceOpMin_32F, ReduceVecOpMin_32F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<double, double, ReduceOpMax_64F, ReduceVecOpMax_64F>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<double, double, ReduceOpMin_64F, ReduceVecOpMin_64F>(srcmat, dstmat);
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8uFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (std::is_same<DT, int>::value)
|
||||
reduceColGeneric<uchar, int, ReduceOpAddSqr_8U32S, ReduceVecOpAddSqr_8U32S>(srcmat, dstmat);
|
||||
else if (std::is_same<DT, float>::value)
|
||||
reduceColGeneric<uchar, float, ReduceOpAddSqr_8U32F, ReduceVecOpAddSqr_8U32F>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<uchar, double, ReduceOpAddSqr_8U64F, ReduceVecOpAddSqr_8U64F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_16u32fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<ushort, float, ReduceOpAddSqr_16U32F, ReduceVecOpAddSqr_16U32F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_16s32fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<short, float, ReduceOpAddSqr_16S32F, ReduceVecOpAddSqr_16S32F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<float, float, ReduceOpAddSqr_32F32F, ReduceVecOpAddSqr_32F32F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
static void reduceColSum2_16u64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<ushort, double, ReduceOpAddSqr_16U64F, ReduceVecOpAddSqr_16U64F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_16s64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<short, double, ReduceOpAddSqr_16S64F, ReduceVecOpAddSqr_16S64F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<float, double, ReduceOpAddSqr_32F64F, ReduceVecOpAddSqr_32F64F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_64f64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<double, double, ReduceOpAddSqr_64F64F, ReduceVecOpAddSqr_64F64F>(srcmat, dstmat);
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,827 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
namespace reduce_c_neon
|
||||
{
|
||||
|
||||
// Optimized ReduceC support in this backend:
|
||||
//
|
||||
// | Input -> output type/channel | SUM | AVG | MIN | MAX | SUM2 |
|
||||
// |------------------------------------|:---:|:---:|:---:|:---:|:----:|
|
||||
// | 8UC1/C3/C4 -> 8UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 8UC1/C3/C4 -> 32SC1/C3/C4 | x | x | - | - | x* |
|
||||
// | 8UC1/C3/C4 -> 32FC1/C3/C4 | x | x | - | - | x* |
|
||||
// | 8UC1/C3/C4 -> 64FC1/C3/C4 | - | - | - | - | x* |
|
||||
// | 16UC1/C3/C4 -> 16UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16UC1 -> 32FC1 | x | x | - | - | x* |
|
||||
// | 16UC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16UC1 -> 64FC1 | - | - | - | - | x* |
|
||||
// | 16UC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 16SC1/C3/C4 -> 16SC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16SC1 -> 32FC1 | x | x | - | - | x* |
|
||||
// | 16SC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16SC1 -> 64FC1 | - | - | - | - | x* |
|
||||
// | 16SC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 32FC1 -> 32FC1 | x | x | x | x | x |
|
||||
// | 32FC3 -> 32FC3 | x | x | - | - | - |
|
||||
// | 32FC4 -> 32FC4 | x | x | x | x | x |
|
||||
// | 32FC1 -> 64FC1 | x* | x* | - | - | x* |
|
||||
// | 32FC3/C4 -> 64FC3/C4 | x* | x* | - | - | - |
|
||||
// | 64FC1 -> 64FC1 | x* | x* | x* | x* | x* |
|
||||
// | 64FC3/C4 -> 64FC3/C4 | x* | x* | - | - | - |
|
||||
//
|
||||
// 'x' in SUM/AVG denotes the existing shared universal-intrinsics kernel; 'x'
|
||||
// in MIN/MAX/SUM2 denotes a native NEON kernel. '*' requires AArch64. For legal
|
||||
// MIN/MAX/SUM2 combinations marked '-', and for other channel counts, dispatch
|
||||
// uses the shared generic fallback.
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
uchar result = isMax ? 0 : UCHAR_MAX;
|
||||
int x = 0;
|
||||
uint8x16_t acc = vdupq_n_u8(result);
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16_t v = vld1q_u8(src + x);
|
||||
acc = isMax ? vmaxq_u8(acc, v) : vminq_u8(acc, v);
|
||||
}
|
||||
uchar lanes[16];
|
||||
vst1q_u8(lanes, acc);
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dst[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort result = isMax ? 0 : USHRT_MAX;
|
||||
int x = 0;
|
||||
uint16x8_t acc = vdupq_n_u16(result);
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
uint16x8_t v = vld1q_u16(src + x);
|
||||
acc = isMax ? vmaxq_u16(acc, v) : vminq_u16(acc, v);
|
||||
}
|
||||
ushort lanes[8];
|
||||
vst1q_u16(lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<ushort>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short result = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
int x = 0;
|
||||
int16x8_t acc = vdupq_n_s16(result);
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
int16x8_t v = vld1q_s16(src + x);
|
||||
acc = isMax ? vmaxq_s16(acc, v) : vminq_s16(acc, v);
|
||||
}
|
||||
short lanes[8];
|
||||
vst1q_s16(lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<short>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort* dst = dstmat.ptr<ushort>(y);
|
||||
uint16x8x4_t acc = {{
|
||||
vdupq_n_u16(initial), vdupq_n_u16(initial),
|
||||
vdupq_n_u16(initial), vdupq_n_u16(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
uint16x8x4_t v = vld4q_u16(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = isMax ? vmaxq_u16(acc.val[c], v.val[c])
|
||||
: vminq_u16(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
ushort lanes[8];
|
||||
vst1q_u16(lanes, acc.val[c]);
|
||||
ushort result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 4 + c])
|
||||
: std::min(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short* dst = dstmat.ptr<short>(y);
|
||||
int16x8x4_t acc = {{
|
||||
vdupq_n_s16(initial), vdupq_n_s16(initial),
|
||||
vdupq_n_s16(initial), vdupq_n_s16(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
int16x8x4_t v = vld4q_s16(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = isMax ? vmaxq_s16(acc.val[c], v.val[c])
|
||||
: vminq_s16(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
short lanes[8];
|
||||
vst1q_s16(lanes, acc.val[c]);
|
||||
short result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 4 + c])
|
||||
: std::min(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort* dst = dstmat.ptr<ushort>(y);
|
||||
uint16x8x3_t acc = {{vdupq_n_u16(initial), vdupq_n_u16(initial), vdupq_n_u16(initial)}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
uint16x8x3_t v = vld3q_u16(src + x * 3);
|
||||
for (int c = 0; c < 3; c++)
|
||||
acc.val[c] = isMax ? vmaxq_u16(acc.val[c], v.val[c])
|
||||
: vminq_u16(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
ushort lanes[8];
|
||||
vst1q_u16(lanes, acc.val[c]);
|
||||
ushort result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 3 + c])
|
||||
: std::min(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short* dst = dstmat.ptr<short>(y);
|
||||
int16x8x3_t acc = {{vdupq_n_s16(initial), vdupq_n_s16(initial), vdupq_n_s16(initial)}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
int16x8x3_t v = vld3q_s16(src + x * 3);
|
||||
for (int c = 0; c < 3; c++)
|
||||
acc.val[c] = isMax ? vmaxq_s16(acc.val[c], v.val[c])
|
||||
: vminq_s16(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
short lanes[8];
|
||||
vst1q_s16(lanes, acc.val[c]);
|
||||
short result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 3 + c])
|
||||
: std::min(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
uint8x16x3_t acc = {{
|
||||
vdupq_n_u8(initial), vdupq_n_u8(initial), vdupq_n_u8(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16x3_t v = vld3q_u8(src + x * 3);
|
||||
for (int c = 0; c < 3; c++)
|
||||
acc.val[c] = isMax ? vmaxq_u8(acc.val[c], v.val[c])
|
||||
: vminq_u8(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
uchar lanes[16];
|
||||
vst1q_u8(lanes, acc.val[c]);
|
||||
uchar result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
uint8x16x4_t acc = {{
|
||||
vdupq_n_u8(initial), vdupq_n_u8(initial),
|
||||
vdupq_n_u8(initial), vdupq_n_u8(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16x4_t v = vld4q_u8(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = isMax ? vmaxq_u8(acc.val[c], v.val[c])
|
||||
: vminq_u8(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
uchar lanes[16];
|
||||
vst1q_u8(lanes, acc.val[c]);
|
||||
uchar result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = src[0];
|
||||
int x = 0;
|
||||
float32x4_t acc = vdupq_n_f32(result);
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
float32x4_t v = vld1q_f32(src + x);
|
||||
acc = isMax ? vmaxq_f32(acc, v) : vminq_f32(acc, v);
|
||||
}
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const float initial = isMax ? std::numeric_limits<float>::lowest() : std::numeric_limits<float>::max();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
float32x4x4_t acc = {{
|
||||
vdupq_n_f32(initial), vdupq_n_f32(initial),
|
||||
vdupq_n_f32(initial), vdupq_n_f32(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
float32x4x4_t v = vld4q_f32(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = isMax ? vmaxq_f32(acc.val[c], v.val[c])
|
||||
: vminq_f32(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc.val[c]);
|
||||
float result = initial;
|
||||
for (int i = 0; i < 4; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
template<bool isMax>
|
||||
static void minMax64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const double initial = isMax ? std::numeric_limits<double>::lowest()
|
||||
: std::numeric_limits<double>::max();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
double result = initial;
|
||||
int x = 0;
|
||||
float64x2_t acc0 = vdupq_n_f64(initial);
|
||||
float64x2_t acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
float64x2_t v0 = vld1q_f64(src + x);
|
||||
float64x2_t v1 = vld1q_f64(src + x + 2);
|
||||
float64x2_t v2 = vld1q_f64(src + x + 4);
|
||||
float64x2_t v3 = vld1q_f64(src + x + 6);
|
||||
acc0 = isMax ? vmaxq_f64(acc0, v0) : vminq_f64(acc0, v0);
|
||||
acc1 = isMax ? vmaxq_f64(acc1, v1) : vminq_f64(acc1, v1);
|
||||
acc2 = isMax ? vmaxq_f64(acc2, v2) : vminq_f64(acc2, v2);
|
||||
acc3 = isMax ? vmaxq_f64(acc3, v3) : vminq_f64(acc3, v3);
|
||||
}
|
||||
acc0 = isMax ? vmaxq_f64(acc0, acc1) : vminq_f64(acc0, acc1);
|
||||
acc2 = isMax ? vmaxq_f64(acc2, acc3) : vminq_f64(acc2, acc3);
|
||||
acc0 = isMax ? vmaxq_f64(acc0, acc2) : vminq_f64(acc0, acc2);
|
||||
double lanes[2];
|
||||
vst1q_f64(lanes, acc0);
|
||||
for (int i = 0; i < 2; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
static inline uint32_t reduceSum2_8u_NEON(uint8x16_t v)
|
||||
{
|
||||
uint16x8_t lo = vmull_u8(vget_low_u8(v), vget_low_u8(v));
|
||||
uint16x8_t hi = vmull_u8(vget_high_u8(v), vget_high_u8(v));
|
||||
return vaddvq_u32(vpaddlq_u16(lo)) + vaddvq_u32(vpaddlq_u16(hi));
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t result = 0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
result += reduceSum2_8u_NEON(vld1q_u8(src + x));
|
||||
for (; x < cols; x++)
|
||||
result += (uint32_t)src[x] * src[x];
|
||||
dst[0] = (DT)(int32_t)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
float result = 0;
|
||||
float32x4_t acc0 = vdupq_n_f32(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint16x8_t v01 = vld1q_u16(src + x);
|
||||
uint16x8_t v23 = vld1q_u16(src + x + 8);
|
||||
float32x4_t v0 = vcvtq_f32_u32(vmovl_u16(vget_low_u16(v01)));
|
||||
float32x4_t v1 = vcvtq_f32_u32(vmovl_u16(vget_high_u16(v01)));
|
||||
float32x4_t v2 = vcvtq_f32_u32(vmovl_u16(vget_low_u16(v23)));
|
||||
float32x4_t v3 = vcvtq_f32_u32(vmovl_u16(vget_high_u16(v23)));
|
||||
acc0 = vmlaq_f32(acc0, v0, v0);
|
||||
acc1 = vmlaq_f32(acc1, v1, v1);
|
||||
acc2 = vmlaq_f32(acc2, v2, v2);
|
||||
acc3 = vmlaq_f32(acc3, v3, v3);
|
||||
}
|
||||
acc0 = vaddq_f32(vaddq_f32(acc0, acc1), vaddq_f32(acc2, acc3));
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc0);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
float value = (float)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
float result = 0;
|
||||
float32x4_t acc0 = vdupq_n_f32(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
int16x8_t v01 = vld1q_s16(src + x);
|
||||
int16x8_t v23 = vld1q_s16(src + x + 8);
|
||||
float32x4_t v0 = vcvtq_f32_s32(vmovl_s16(vget_low_s16(v01)));
|
||||
float32x4_t v1 = vcvtq_f32_s32(vmovl_s16(vget_high_s16(v01)));
|
||||
float32x4_t v2 = vcvtq_f32_s32(vmovl_s16(vget_low_s16(v23)));
|
||||
float32x4_t v3 = vcvtq_f32_s32(vmovl_s16(vget_high_s16(v23)));
|
||||
acc0 = vmlaq_f32(acc0, v0, v0);
|
||||
acc1 = vmlaq_f32(acc1, v1, v1);
|
||||
acc2 = vmlaq_f32(acc2, v2, v2);
|
||||
acc3 = vmlaq_f32(acc3, v3, v3);
|
||||
}
|
||||
acc0 = vaddq_f32(vaddq_f32(acc0, acc1), vaddq_f32(acc2, acc3));
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc0);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
float value = (float)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
uint64x2_t acc0 = vdupq_n_u64(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
uint16x8_t v = vld1q_u16(src + x);
|
||||
uint32x4_t sq0 = vmull_u16(vget_low_u16(v), vget_low_u16(v));
|
||||
uint32x4_t sq1 = vmull_u16(vget_high_u16(v), vget_high_u16(v));
|
||||
acc0 = vaddq_u64(acc0, vmovl_u32(vget_low_u32(sq0)));
|
||||
acc1 = vaddq_u64(acc1, vmovl_high_u32(sq0));
|
||||
acc2 = vaddq_u64(acc2, vmovl_u32(vget_low_u32(sq1)));
|
||||
acc3 = vaddq_u64(acc3, vmovl_high_u32(sq1));
|
||||
}
|
||||
acc0 = vaddq_u64(vaddq_u64(acc0, acc1), vaddq_u64(acc2, acc3));
|
||||
uint64_t lanes[2];
|
||||
vst1q_u64(lanes, acc0);
|
||||
uint64_t result = lanes[0] + lanes[1];
|
||||
for (; x < cols; x++)
|
||||
result += (uint64_t)src[x] * src[x];
|
||||
dstmat.ptr<double>(y)[0] = (double)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
uint64x2_t acc0 = vdupq_n_u64(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
int16x8_t v = vld1q_s16(src + x);
|
||||
int32x4_t sq0 = vmull_s16(vget_low_s16(v), vget_low_s16(v));
|
||||
int32x4_t sq1 = vmull_s16(vget_high_s16(v), vget_high_s16(v));
|
||||
uint32x4_t usq0 = vreinterpretq_u32_s32(sq0);
|
||||
uint32x4_t usq1 = vreinterpretq_u32_s32(sq1);
|
||||
acc0 = vaddq_u64(acc0, vmovl_u32(vget_low_u32(usq0)));
|
||||
acc1 = vaddq_u64(acc1, vmovl_high_u32(usq0));
|
||||
acc2 = vaddq_u64(acc2, vmovl_u32(vget_low_u32(usq1)));
|
||||
acc3 = vaddq_u64(acc3, vmovl_high_u32(usq1));
|
||||
}
|
||||
acc0 = vaddq_u64(vaddq_u64(acc0, acc1), vaddq_u64(acc2, acc3));
|
||||
uint64_t lanes[2];
|
||||
vst1q_u64(lanes, acc0);
|
||||
uint64_t result = lanes[0] + lanes[1];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
int64_t value = src[x];
|
||||
result += (uint64_t)(value * value);
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = (double)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float64x2_t acc0 = vdupq_n_f64(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
float32x4_t v01 = vld1q_f32(src + x);
|
||||
float32x4_t v23 = vld1q_f32(src + x + 4);
|
||||
float64x2_t v0 = vcvt_f64_f32(vget_low_f32(v01));
|
||||
float64x2_t v1 = vcvt_high_f64_f32(v01);
|
||||
float64x2_t v2 = vcvt_f64_f32(vget_low_f32(v23));
|
||||
float64x2_t v3 = vcvt_high_f64_f32(v23);
|
||||
acc0 = vmlaq_f64(acc0, v0, v0);
|
||||
acc1 = vmlaq_f64(acc1, v1, v1);
|
||||
acc2 = vmlaq_f64(acc2, v2, v2);
|
||||
acc3 = vmlaq_f64(acc3, v3, v3);
|
||||
}
|
||||
acc0 = vaddq_f64(vaddq_f64(acc0, acc1), vaddq_f64(acc2, acc3));
|
||||
double lanes[2];
|
||||
vst1q_f64(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
double value = (double)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_64f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
float64x2_t acc0 = vdupq_n_f64(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
float64x2_t v0 = vld1q_f64(src + x);
|
||||
float64x2_t v1 = vld1q_f64(src + x + 2);
|
||||
float64x2_t v2 = vld1q_f64(src + x + 4);
|
||||
float64x2_t v3 = vld1q_f64(src + x + 6);
|
||||
acc0 = vmlaq_f64(acc0, v0, v0);
|
||||
acc1 = vmlaq_f64(acc1, v1, v1);
|
||||
acc2 = vmlaq_f64(acc2, v2, v2);
|
||||
acc3 = vmlaq_f64(acc3, v3, v3);
|
||||
}
|
||||
acc0 = vaddq_f64(vaddq_f64(acc0, acc1), vaddq_f64(acc2, acc3));
|
||||
double lanes[2];
|
||||
vst1q_f64(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1];
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t results[3] = {0, 0, 0};
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16x3_t v = vld3q_u8(src + x * 3);
|
||||
for (int c = 0; c < 3; c++)
|
||||
results[c] += reduceSum2_8u_NEON(v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
for (int i = x; i < cols; i++)
|
||||
{
|
||||
uint32_t value = src[i * 3 + c];
|
||||
results[c] += value * value;
|
||||
}
|
||||
dst[c] = (DT)(int32_t)results[c];
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t results[4] = {0, 0, 0, 0};
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16x4_t v = vld4q_u8(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
results[c] += reduceSum2_8u_NEON(v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
for (int i = x; i < cols; i++)
|
||||
{
|
||||
uint32_t value = src[i * 4 + c];
|
||||
results[c] += value * value;
|
||||
}
|
||||
dst[c] = (DT)(int32_t)results[c];
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
static void sum2_32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = 0;
|
||||
int x = 0;
|
||||
float32x4_t acc = vdupq_n_f32(0);
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
float32x4_t v = vld1q_f32(src + x);
|
||||
acc = vaddq_f32(acc, vmulq_f32(v, v));
|
||||
}
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
float32x4x4_t acc = {{
|
||||
vdupq_n_f32(0), vdupq_n_f32(0),
|
||||
vdupq_n_f32(0), vdupq_n_f32(0)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
float32x4x4_t v = vld4q_f32(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = vmlaq_f32(acc.val[c], v.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
float lanes[4];
|
||||
float result = 0;
|
||||
vst1q_f32(lanes, acc.val[c]);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result += lanes[i];
|
||||
for (int i = x; i < cols; i++)
|
||||
result += src[i * 4 + c] * src[i * 4 + c];
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
} // namespace reduce_c_neon
|
||||
@@ -0,0 +1,866 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
namespace reduce_c_rvv
|
||||
{
|
||||
|
||||
// Optimized ReduceC support in this backend:
|
||||
//
|
||||
// | Input -> output type/channel | SUM | AVG | MIN | MAX | SUM2 |
|
||||
// |------------------------------------|:---:|:---:|:---:|:---:|:----:|
|
||||
// | 8UC1/C3/C4 -> 8UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 8UC1/C3/C4 -> 32SC1/C3/C4 | x | x | - | - | x |
|
||||
// | 8UC1/C3/C4 -> 32FC1/C3/C4 | x | x | - | - | x |
|
||||
// | 8UC1/C3/C4 -> 64FC1/C3/C4 | - | - | - | - | x |
|
||||
// | 16UC1/C3/C4 -> 16UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16UC1 -> 32FC1 | x | x | - | - | x |
|
||||
// | 16UC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16UC1 -> 64FC1 | - | - | - | - | x* |
|
||||
// | 16UC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 16SC1/C3/C4 -> 16SC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16SC1 -> 32FC1 | x | x | - | - | x |
|
||||
// | 16SC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16SC1 -> 64FC1 | - | - | - | - | x* |
|
||||
// | 16SC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 32FC1/C3/C4 -> 32FC1/C3/C4 | x | x | x | x | x |
|
||||
// | 32FC1 -> 64FC1 | x* | x* | - | - | x* |
|
||||
// | 32FC3/C4 -> 64FC3/C4 | x* | x* | - | - | - |
|
||||
// | 64FC1 -> 64FC1 | x* | x* | x* | x* | x* |
|
||||
// | 64FC3/C4 -> 64FC3/C4 | x* | x* | - | - | - |
|
||||
//
|
||||
// 'x' in SUM/AVG denotes the existing shared universal-intrinsics kernel; 'x'
|
||||
// in MIN/MAX/SUM2 denotes a native RVV kernel. '*' requires
|
||||
// CV_SIMD_SCALABLE_64F. For legal MIN/MAX/SUM2 combinations marked '-', and for
|
||||
// other channel counts, dispatch uses the shared generic fallback.
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
uchar result = isMax ? 0 : UCHAR_MAX;
|
||||
int x = 0;
|
||||
const int vlmax = __riscv_vsetvlmax_e8m8();
|
||||
vuint8m8_t acc = __riscv_vmv_v_x_u8m8(result, vlmax);
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e8m8(cols - x);
|
||||
vuint8m8_t v = __riscv_vle8_v_u8m8(src + x, vl);
|
||||
acc = isMax ? __riscv_vmaxu_tu(acc, acc, v, vl)
|
||||
: __riscv_vminu_tu(acc, acc, v, vl);
|
||||
x += vl;
|
||||
}
|
||||
vuint8m1_t seed = __riscv_vmv_s_x_u8m1(result, __riscv_vsetvlmax_e8m1());
|
||||
vuint8m1_t reduced = isMax ? __riscv_vredmaxu(acc, seed, vlmax)
|
||||
: __riscv_vredminu(acc, seed, vlmax);
|
||||
result = (uchar)__riscv_vmv_x(reduced);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dst[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
const int vlmax = __riscv_vsetvlmax_e16m8();
|
||||
vuint16m8_t acc = __riscv_vmv_v_x_u16m8(initial, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m8(cols - x);
|
||||
vuint16m8_t v = __riscv_vle16_v_u16m8(src + x, vl);
|
||||
acc = isMax ? __riscv_vmaxu_tu(acc, acc, v, vl)
|
||||
: __riscv_vminu_tu(acc, acc, v, vl);
|
||||
x += vl;
|
||||
}
|
||||
vuint16m1_t seed = __riscv_vmv_s_x_u16m1(initial, __riscv_vsetvlmax_e16m1());
|
||||
vuint16m1_t reduced = isMax ? __riscv_vredmaxu(acc, seed, vlmax)
|
||||
: __riscv_vredminu(acc, seed, vlmax);
|
||||
dstmat.ptr<ushort>(y)[0] = (ushort)__riscv_vmv_x(reduced);
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
const int vlmax = __riscv_vsetvlmax_e16m8();
|
||||
vint16m8_t acc = __riscv_vmv_v_x_i16m8(initial, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m8(cols - x);
|
||||
vint16m8_t v = __riscv_vle16_v_i16m8(src + x, vl);
|
||||
acc = isMax ? __riscv_vmax_tu(acc, acc, v, vl)
|
||||
: __riscv_vmin_tu(acc, acc, v, vl);
|
||||
x += vl;
|
||||
}
|
||||
vint16m1_t seed = __riscv_vmv_s_x_i16m1(initial, __riscv_vsetvlmax_e16m1());
|
||||
vint16m1_t reduced = isMax ? __riscv_vredmax(acc, seed, vlmax)
|
||||
: __riscv_vredmin(acc, seed, vlmax);
|
||||
dstmat.ptr<short>(y)[0] = (short)__riscv_vmv_x(reduced);
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
const int vlmax = __riscv_vsetvlmax_e16m2();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort* dst = dstmat.ptr<ushort>(y);
|
||||
vuint16m2_t acc0 = __riscv_vmv_v_x_u16m2(initial, vlmax);
|
||||
vuint16m2_t acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m2(cols - x);
|
||||
vuint16m2x4_t v = __riscv_vlseg4e16_v_u16m2x4(src + x * 4, vl);
|
||||
vuint16m2_t v0 = __riscv_vget_v_u16m2x4_u16m2(v, 0);
|
||||
vuint16m2_t v1 = __riscv_vget_v_u16m2x4_u16m2(v, 1);
|
||||
vuint16m2_t v2 = __riscv_vget_v_u16m2x4_u16m2(v, 2);
|
||||
vuint16m2_t v3 = __riscv_vget_v_u16m2x4_u16m2(v, 3);
|
||||
acc0 = isMax ? __riscv_vmaxu_tu(acc0, acc0, v0, vl)
|
||||
: __riscv_vminu_tu(acc0, acc0, v0, vl);
|
||||
acc1 = isMax ? __riscv_vmaxu_tu(acc1, acc1, v1, vl)
|
||||
: __riscv_vminu_tu(acc1, acc1, v1, vl);
|
||||
acc2 = isMax ? __riscv_vmaxu_tu(acc2, acc2, v2, vl)
|
||||
: __riscv_vminu_tu(acc2, acc2, v2, vl);
|
||||
acc3 = isMax ? __riscv_vmaxu_tu(acc3, acc3, v3, vl)
|
||||
: __riscv_vminu_tu(acc3, acc3, v3, vl);
|
||||
x += vl;
|
||||
}
|
||||
vuint16m1_t seed = __riscv_vmv_s_x_u16m1(initial, __riscv_vsetvlmax_e16m1());
|
||||
dst[0] = (ushort)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc0, seed, vlmax) : __riscv_vredminu(acc0, seed, vlmax));
|
||||
dst[1] = (ushort)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc1, seed, vlmax) : __riscv_vredminu(acc1, seed, vlmax));
|
||||
dst[2] = (ushort)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc2, seed, vlmax) : __riscv_vredminu(acc2, seed, vlmax));
|
||||
dst[3] = (ushort)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc3, seed, vlmax) : __riscv_vredminu(acc3, seed, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
const int vlmax = __riscv_vsetvlmax_e16m2();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short* dst = dstmat.ptr<short>(y);
|
||||
vint16m2_t acc0 = __riscv_vmv_v_x_i16m2(initial, vlmax);
|
||||
vint16m2_t acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m2(cols - x);
|
||||
vint16m2x4_t v = __riscv_vlseg4e16_v_i16m2x4(src + x * 4, vl);
|
||||
vint16m2_t v0 = __riscv_vget_v_i16m2x4_i16m2(v, 0);
|
||||
vint16m2_t v1 = __riscv_vget_v_i16m2x4_i16m2(v, 1);
|
||||
vint16m2_t v2 = __riscv_vget_v_i16m2x4_i16m2(v, 2);
|
||||
vint16m2_t v3 = __riscv_vget_v_i16m2x4_i16m2(v, 3);
|
||||
acc0 = isMax ? __riscv_vmax_tu(acc0, acc0, v0, vl)
|
||||
: __riscv_vmin_tu(acc0, acc0, v0, vl);
|
||||
acc1 = isMax ? __riscv_vmax_tu(acc1, acc1, v1, vl)
|
||||
: __riscv_vmin_tu(acc1, acc1, v1, vl);
|
||||
acc2 = isMax ? __riscv_vmax_tu(acc2, acc2, v2, vl)
|
||||
: __riscv_vmin_tu(acc2, acc2, v2, vl);
|
||||
acc3 = isMax ? __riscv_vmax_tu(acc3, acc3, v3, vl)
|
||||
: __riscv_vmin_tu(acc3, acc3, v3, vl);
|
||||
x += vl;
|
||||
}
|
||||
vint16m1_t seed = __riscv_vmv_s_x_i16m1(initial, __riscv_vsetvlmax_e16m1());
|
||||
dst[0] = (short)__riscv_vmv_x(isMax ? __riscv_vredmax(acc0, seed, vlmax) : __riscv_vredmin(acc0, seed, vlmax));
|
||||
dst[1] = (short)__riscv_vmv_x(isMax ? __riscv_vredmax(acc1, seed, vlmax) : __riscv_vredmin(acc1, seed, vlmax));
|
||||
dst[2] = (short)__riscv_vmv_x(isMax ? __riscv_vredmax(acc2, seed, vlmax) : __riscv_vredmin(acc2, seed, vlmax));
|
||||
dst[3] = (short)__riscv_vmv_x(isMax ? __riscv_vredmax(acc3, seed, vlmax) : __riscv_vredmin(acc3, seed, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
const int vlmax = __riscv_vsetvlmax_e16m2();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort* dst = dstmat.ptr<ushort>(y);
|
||||
vuint16m2_t acc0 = __riscv_vmv_v_x_u16m2(initial, vlmax);
|
||||
vuint16m2_t acc1 = acc0, acc2 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m2(cols - x);
|
||||
vuint16m2x3_t v = __riscv_vlseg3e16_v_u16m2x3(src + x * 3, vl);
|
||||
vuint16m2_t v0 = __riscv_vget_v_u16m2x3_u16m2(v, 0);
|
||||
vuint16m2_t v1 = __riscv_vget_v_u16m2x3_u16m2(v, 1);
|
||||
vuint16m2_t v2 = __riscv_vget_v_u16m2x3_u16m2(v, 2);
|
||||
acc0 = isMax ? __riscv_vmaxu_tu(acc0, acc0, v0, vl) : __riscv_vminu_tu(acc0, acc0, v0, vl);
|
||||
acc1 = isMax ? __riscv_vmaxu_tu(acc1, acc1, v1, vl) : __riscv_vminu_tu(acc1, acc1, v1, vl);
|
||||
acc2 = isMax ? __riscv_vmaxu_tu(acc2, acc2, v2, vl) : __riscv_vminu_tu(acc2, acc2, v2, vl);
|
||||
x += vl;
|
||||
}
|
||||
vuint16m1_t seed = __riscv_vmv_s_x_u16m1(initial, __riscv_vsetvlmax_e16m1());
|
||||
dst[0] = (ushort)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc0, seed, vlmax) : __riscv_vredminu(acc0, seed, vlmax));
|
||||
dst[1] = (ushort)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc1, seed, vlmax) : __riscv_vredminu(acc1, seed, vlmax));
|
||||
dst[2] = (ushort)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc2, seed, vlmax) : __riscv_vredminu(acc2, seed, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
const int vlmax = __riscv_vsetvlmax_e16m2();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short* dst = dstmat.ptr<short>(y);
|
||||
vint16m2_t acc0 = __riscv_vmv_v_x_i16m2(initial, vlmax);
|
||||
vint16m2_t acc1 = acc0, acc2 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m2(cols - x);
|
||||
vint16m2x3_t v = __riscv_vlseg3e16_v_i16m2x3(src + x * 3, vl);
|
||||
vint16m2_t v0 = __riscv_vget_v_i16m2x3_i16m2(v, 0);
|
||||
vint16m2_t v1 = __riscv_vget_v_i16m2x3_i16m2(v, 1);
|
||||
vint16m2_t v2 = __riscv_vget_v_i16m2x3_i16m2(v, 2);
|
||||
acc0 = isMax ? __riscv_vmax_tu(acc0, acc0, v0, vl) : __riscv_vmin_tu(acc0, acc0, v0, vl);
|
||||
acc1 = isMax ? __riscv_vmax_tu(acc1, acc1, v1, vl) : __riscv_vmin_tu(acc1, acc1, v1, vl);
|
||||
acc2 = isMax ? __riscv_vmax_tu(acc2, acc2, v2, vl) : __riscv_vmin_tu(acc2, acc2, v2, vl);
|
||||
x += vl;
|
||||
}
|
||||
vint16m1_t seed = __riscv_vmv_s_x_i16m1(initial, __riscv_vsetvlmax_e16m1());
|
||||
dst[0] = (short)__riscv_vmv_x(isMax ? __riscv_vredmax(acc0, seed, vlmax) : __riscv_vredmin(acc0, seed, vlmax));
|
||||
dst[1] = (short)__riscv_vmv_x(isMax ? __riscv_vredmax(acc1, seed, vlmax) : __riscv_vredmin(acc1, seed, vlmax));
|
||||
dst[2] = (short)__riscv_vmv_x(isMax ? __riscv_vredmax(acc2, seed, vlmax) : __riscv_vredmin(acc2, seed, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
const int vlmax = __riscv_vsetvlmax_e8m1();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
vuint8m1_t acc0 = __riscv_vmv_v_x_u8m1(initial, vlmax);
|
||||
vuint8m1_t acc1 = acc0, acc2 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e8m1(cols - x);
|
||||
vuint8m1x3_t v = __riscv_vlseg3e8_v_u8m1x3(src + x * 3, vl);
|
||||
vuint8m1_t v0 = __riscv_vget_v_u8m1x3_u8m1(v, 0);
|
||||
vuint8m1_t v1 = __riscv_vget_v_u8m1x3_u8m1(v, 1);
|
||||
vuint8m1_t v2 = __riscv_vget_v_u8m1x3_u8m1(v, 2);
|
||||
acc0 = isMax ? __riscv_vmaxu_tu(acc0, acc0, v0, vl)
|
||||
: __riscv_vminu_tu(acc0, acc0, v0, vl);
|
||||
acc1 = isMax ? __riscv_vmaxu_tu(acc1, acc1, v1, vl)
|
||||
: __riscv_vminu_tu(acc1, acc1, v1, vl);
|
||||
acc2 = isMax ? __riscv_vmaxu_tu(acc2, acc2, v2, vl)
|
||||
: __riscv_vminu_tu(acc2, acc2, v2, vl);
|
||||
x += vl;
|
||||
}
|
||||
vuint8m1_t seed = __riscv_vmv_s_x_u8m1(initial, vlmax);
|
||||
dst[0] = (uchar)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc0, seed, vlmax)
|
||||
: __riscv_vredminu(acc0, seed, vlmax));
|
||||
dst[1] = (uchar)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc1, seed, vlmax)
|
||||
: __riscv_vredminu(acc1, seed, vlmax));
|
||||
dst[2] = (uchar)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc2, seed, vlmax)
|
||||
: __riscv_vredminu(acc2, seed, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
const int vlmax = __riscv_vsetvlmax_e8m1();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
vuint8m1_t acc0 = __riscv_vmv_v_x_u8m1(initial, vlmax);
|
||||
vuint8m1_t acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e8m1(cols - x);
|
||||
vuint8m1x4_t v = __riscv_vlseg4e8_v_u8m1x4(src + x * 4, vl);
|
||||
vuint8m1_t v0 = __riscv_vget_v_u8m1x4_u8m1(v, 0);
|
||||
vuint8m1_t v1 = __riscv_vget_v_u8m1x4_u8m1(v, 1);
|
||||
vuint8m1_t v2 = __riscv_vget_v_u8m1x4_u8m1(v, 2);
|
||||
vuint8m1_t v3 = __riscv_vget_v_u8m1x4_u8m1(v, 3);
|
||||
acc0 = isMax ? __riscv_vmaxu_tu(acc0, acc0, v0, vl)
|
||||
: __riscv_vminu_tu(acc0, acc0, v0, vl);
|
||||
acc1 = isMax ? __riscv_vmaxu_tu(acc1, acc1, v1, vl)
|
||||
: __riscv_vminu_tu(acc1, acc1, v1, vl);
|
||||
acc2 = isMax ? __riscv_vmaxu_tu(acc2, acc2, v2, vl)
|
||||
: __riscv_vminu_tu(acc2, acc2, v2, vl);
|
||||
acc3 = isMax ? __riscv_vmaxu_tu(acc3, acc3, v3, vl)
|
||||
: __riscv_vminu_tu(acc3, acc3, v3, vl);
|
||||
x += vl;
|
||||
}
|
||||
vuint8m1_t seed = __riscv_vmv_s_x_u8m1(initial, vlmax);
|
||||
dst[0] = (uchar)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc0, seed, vlmax)
|
||||
: __riscv_vredminu(acc0, seed, vlmax));
|
||||
dst[1] = (uchar)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc1, seed, vlmax)
|
||||
: __riscv_vredminu(acc1, seed, vlmax));
|
||||
dst[2] = (uchar)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc2, seed, vlmax)
|
||||
: __riscv_vredminu(acc2, seed, vlmax));
|
||||
dst[3] = (uchar)__riscv_vmv_x(isMax ? __riscv_vredmaxu(acc3, seed, vlmax)
|
||||
: __riscv_vredminu(acc3, seed, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = src[0];
|
||||
int x = 0;
|
||||
const int vlmax = __riscv_vsetvlmax_e32m8();
|
||||
vfloat32m8_t acc = __riscv_vfmv_v_f_f32m8(result, vlmax);
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e32m8(cols - x);
|
||||
vfloat32m8_t v = __riscv_vle32_v_f32m8(src + x, vl);
|
||||
acc = isMax ? __riscv_vfmax_tu(acc, acc, v, vl)
|
||||
: __riscv_vfmin_tu(acc, acc, v, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat32m1_t seed = __riscv_vfmv_s_f_f32m1(result, __riscv_vsetvlmax_e32m1());
|
||||
vfloat32m1_t reduced = isMax ? __riscv_vfredmax(acc, seed, vlmax)
|
||||
: __riscv_vfredmin(acc, seed, vlmax);
|
||||
result = __riscv_vfmv_f(reduced);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const float initial = isMax ? std::numeric_limits<float>::lowest() : std::numeric_limits<float>::max();
|
||||
const int vlmax = __riscv_vsetvlmax_e32m2();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
vfloat32m2_t acc0 = __riscv_vfmv_v_f_f32m2(initial, vlmax);
|
||||
vfloat32m2_t acc1 = acc0, acc2 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e32m2(cols - x);
|
||||
vfloat32m2x3_t v = __riscv_vlseg3e32_v_f32m2x3(src + x * 3, vl);
|
||||
vfloat32m2_t v0 = __riscv_vget_v_f32m2x3_f32m2(v, 0);
|
||||
vfloat32m2_t v1 = __riscv_vget_v_f32m2x3_f32m2(v, 1);
|
||||
vfloat32m2_t v2 = __riscv_vget_v_f32m2x3_f32m2(v, 2);
|
||||
acc0 = isMax ? __riscv_vfmax_tu(acc0, acc0, v0, vl)
|
||||
: __riscv_vfmin_tu(acc0, acc0, v0, vl);
|
||||
acc1 = isMax ? __riscv_vfmax_tu(acc1, acc1, v1, vl)
|
||||
: __riscv_vfmin_tu(acc1, acc1, v1, vl);
|
||||
acc2 = isMax ? __riscv_vfmax_tu(acc2, acc2, v2, vl)
|
||||
: __riscv_vfmin_tu(acc2, acc2, v2, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat32m1_t seed = __riscv_vfmv_s_f_f32m1(initial, __riscv_vsetvlmax_e32m1());
|
||||
dst[0] = __riscv_vfmv_f(isMax ? __riscv_vfredmax(acc0, seed, vlmax)
|
||||
: __riscv_vfredmin(acc0, seed, vlmax));
|
||||
dst[1] = __riscv_vfmv_f(isMax ? __riscv_vfredmax(acc1, seed, vlmax)
|
||||
: __riscv_vfredmin(acc1, seed, vlmax));
|
||||
dst[2] = __riscv_vfmv_f(isMax ? __riscv_vfredmax(acc2, seed, vlmax)
|
||||
: __riscv_vfredmin(acc2, seed, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const float initial = isMax ? std::numeric_limits<float>::lowest() : std::numeric_limits<float>::max();
|
||||
const int vlmax = __riscv_vsetvlmax_e32m2();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
vfloat32m2_t acc0 = __riscv_vfmv_v_f_f32m2(initial, vlmax);
|
||||
vfloat32m2_t acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e32m2(cols - x);
|
||||
vfloat32m2x4_t v = __riscv_vlseg4e32_v_f32m2x4(src + x * 4, vl);
|
||||
vfloat32m2_t v0 = __riscv_vget_v_f32m2x4_f32m2(v, 0);
|
||||
vfloat32m2_t v1 = __riscv_vget_v_f32m2x4_f32m2(v, 1);
|
||||
vfloat32m2_t v2 = __riscv_vget_v_f32m2x4_f32m2(v, 2);
|
||||
vfloat32m2_t v3 = __riscv_vget_v_f32m2x4_f32m2(v, 3);
|
||||
acc0 = isMax ? __riscv_vfmax_tu(acc0, acc0, v0, vl)
|
||||
: __riscv_vfmin_tu(acc0, acc0, v0, vl);
|
||||
acc1 = isMax ? __riscv_vfmax_tu(acc1, acc1, v1, vl)
|
||||
: __riscv_vfmin_tu(acc1, acc1, v1, vl);
|
||||
acc2 = isMax ? __riscv_vfmax_tu(acc2, acc2, v2, vl)
|
||||
: __riscv_vfmin_tu(acc2, acc2, v2, vl);
|
||||
acc3 = isMax ? __riscv_vfmax_tu(acc3, acc3, v3, vl)
|
||||
: __riscv_vfmin_tu(acc3, acc3, v3, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat32m1_t seed = __riscv_vfmv_s_f_f32m1(initial, __riscv_vsetvlmax_e32m1());
|
||||
dst[0] = __riscv_vfmv_f(isMax ? __riscv_vfredmax(acc0, seed, vlmax)
|
||||
: __riscv_vfredmin(acc0, seed, vlmax));
|
||||
dst[1] = __riscv_vfmv_f(isMax ? __riscv_vfredmax(acc1, seed, vlmax)
|
||||
: __riscv_vfredmin(acc1, seed, vlmax));
|
||||
dst[2] = __riscv_vfmv_f(isMax ? __riscv_vfredmax(acc2, seed, vlmax)
|
||||
: __riscv_vfredmin(acc2, seed, vlmax));
|
||||
dst[3] = __riscv_vfmv_f(isMax ? __riscv_vfredmax(acc3, seed, vlmax)
|
||||
: __riscv_vfredmin(acc3, seed, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
#if CV_SIMD_SCALABLE_64F
|
||||
template<bool isMax>
|
||||
static void minMax64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const double initial = isMax ? std::numeric_limits<double>::lowest()
|
||||
: std::numeric_limits<double>::max();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
const int vlmax = __riscv_vsetvlmax_e64m8();
|
||||
vfloat64m8_t acc = __riscv_vfmv_v_f_f64m8(initial, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e64m8(cols - x);
|
||||
vfloat64m8_t v = __riscv_vle64_v_f64m8(src + x, vl);
|
||||
acc = isMax ? __riscv_vfmax_tu(acc, acc, v, vl)
|
||||
: __riscv_vfmin_tu(acc, acc, v, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat64m1_t seed = __riscv_vfmv_s_f_f64m1(initial, __riscv_vsetvlmax_e64m1());
|
||||
vfloat64m1_t reduced = isMax ? __riscv_vfredmax(acc, seed, vlmax)
|
||||
: __riscv_vfredmin(acc, seed, vlmax);
|
||||
dstmat.ptr<double>(y)[0] = __riscv_vfmv_f(reduced);
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t result = 0;
|
||||
int x = 0;
|
||||
vuint32m1_t acc = __riscv_vmv_v_x_u32m1(0, __riscv_vsetvlmax_e32m1());
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e8m4(cols - x);
|
||||
vuint8m4_t v = __riscv_vle8_v_u8m4(src + x, vl);
|
||||
acc = __riscv_vwredsumu(__riscv_vwmulu(v, v, vl), acc, vl);
|
||||
x += vl;
|
||||
}
|
||||
result = (uint32_t)__riscv_vmv_x(acc);
|
||||
for (; x < cols; x++)
|
||||
result += (uint32_t)src[x] * src[x];
|
||||
dst[0] = (DT)(int32_t)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
const int vlmax = __riscv_vsetvlmax_e32m8();
|
||||
vfloat32m8_t acc = __riscv_vfmv_v_f_f32m8(0, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m4(cols - x);
|
||||
vuint16m4_t v = __riscv_vle16_v_u16m4(src + x, vl);
|
||||
vfloat32m8_t vf = __riscv_vfwcvt_f_xu_v_f32m8(v, vl);
|
||||
acc = __riscv_vfmacc_tu(acc, vf, vf, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat32m1_t zero = __riscv_vfmv_s_f_f32m1(0, __riscv_vsetvlmax_e32m1());
|
||||
dstmat.ptr<float>(y)[0] = __riscv_vfmv_f(__riscv_vfredusum(acc, zero, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
const int vlmax = __riscv_vsetvlmax_e32m8();
|
||||
vfloat32m8_t acc = __riscv_vfmv_v_f_f32m8(0, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m4(cols - x);
|
||||
vint16m4_t v = __riscv_vle16_v_i16m4(src + x, vl);
|
||||
vfloat32m8_t vf = __riscv_vfwcvt_f_x_v_f32m8(v, vl);
|
||||
acc = __riscv_vfmacc_tu(acc, vf, vf, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat32m1_t zero = __riscv_vfmv_s_f_f32m1(0, __riscv_vsetvlmax_e32m1());
|
||||
dstmat.ptr<float>(y)[0] = __riscv_vfmv_f(__riscv_vfredusum(acc, zero, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
#if CV_SIMD_SCALABLE_64F
|
||||
static void sum2_16u64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
const int vlmax = __riscv_vsetvlmax_e64m8();
|
||||
vfloat64m8_t acc = __riscv_vfmv_v_f_f64m8(0, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m2(cols - x);
|
||||
vuint16m2_t v = __riscv_vle16_v_u16m2(src + x, vl);
|
||||
vfloat32m4_t vf = __riscv_vfwcvt_f_xu_v_f32m4(v, vl);
|
||||
vfloat64m8_t vd = __riscv_vfwcvt_f_f_v_f64m8(vf, vl);
|
||||
acc = __riscv_vfmacc_tu(acc, vd, vd, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat64m1_t zero = __riscv_vfmv_s_f_f64m1(0, __riscv_vsetvlmax_e64m1());
|
||||
dstmat.ptr<double>(y)[0] = __riscv_vfmv_f(__riscv_vfredusum(acc, zero, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
const int vlmax = __riscv_vsetvlmax_e64m8();
|
||||
vfloat64m8_t acc = __riscv_vfmv_v_f_f64m8(0, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e16m2(cols - x);
|
||||
vint16m2_t v = __riscv_vle16_v_i16m2(src + x, vl);
|
||||
vfloat32m4_t vf = __riscv_vfwcvt_f_x_v_f32m4(v, vl);
|
||||
vfloat64m8_t vd = __riscv_vfwcvt_f_f_v_f64m8(vf, vl);
|
||||
acc = __riscv_vfmacc_tu(acc, vd, vd, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat64m1_t zero = __riscv_vfmv_s_f_f64m1(0, __riscv_vsetvlmax_e64m1());
|
||||
dstmat.ptr<double>(y)[0] = __riscv_vfmv_f(__riscv_vfredusum(acc, zero, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
const int vlmax = __riscv_vsetvlmax_e64m8();
|
||||
vfloat64m8_t acc = __riscv_vfmv_v_f_f64m8(0, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e32m4(cols - x);
|
||||
vfloat32m4_t v = __riscv_vle32_v_f32m4(src + x, vl);
|
||||
vfloat64m8_t vd = __riscv_vfwcvt_f_f_v_f64m8(v, vl);
|
||||
acc = __riscv_vfmacc_tu(acc, vd, vd, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat64m1_t zero = __riscv_vfmv_s_f_f64m1(0, __riscv_vsetvlmax_e64m1());
|
||||
dstmat.ptr<double>(y)[0] = __riscv_vfmv_f(__riscv_vfredusum(acc, zero, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_64f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
const int vlmax = __riscv_vsetvlmax_e64m8();
|
||||
vfloat64m8_t acc = __riscv_vfmv_v_f_f64m8(0, vlmax);
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e64m8(cols - x);
|
||||
vfloat64m8_t v = __riscv_vle64_v_f64m8(src + x, vl);
|
||||
acc = __riscv_vfmacc_tu(acc, v, v, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat64m1_t zero = __riscv_vfmv_s_f_f64m1(0, __riscv_vsetvlmax_e64m1());
|
||||
dstmat.ptr<double>(y)[0] = __riscv_vfmv_f(__riscv_vfredusum(acc, zero, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
vuint32m1_t acc0 = __riscv_vmv_v_x_u32m1(0, __riscv_vsetvlmax_e32m1());
|
||||
vuint32m1_t acc1 = acc0, acc2 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e8m1(cols - x);
|
||||
vuint8m1x3_t v = __riscv_vlseg3e8_v_u8m1x3(src + x * 3, vl);
|
||||
vuint8m1_t v0 = __riscv_vget_v_u8m1x3_u8m1(v, 0);
|
||||
vuint8m1_t v1 = __riscv_vget_v_u8m1x3_u8m1(v, 1);
|
||||
vuint8m1_t v2 = __riscv_vget_v_u8m1x3_u8m1(v, 2);
|
||||
acc0 = __riscv_vwredsumu(__riscv_vwmulu(v0, v0, vl), acc0, vl);
|
||||
acc1 = __riscv_vwredsumu(__riscv_vwmulu(v1, v1, vl), acc1, vl);
|
||||
acc2 = __riscv_vwredsumu(__riscv_vwmulu(v2, v2, vl), acc2, vl);
|
||||
x += vl;
|
||||
}
|
||||
dst[0] = (DT)(int32_t)(uint32_t)__riscv_vmv_x(acc0);
|
||||
dst[1] = (DT)(int32_t)(uint32_t)__riscv_vmv_x(acc1);
|
||||
dst[2] = (DT)(int32_t)(uint32_t)__riscv_vmv_x(acc2);
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
vuint32m1_t acc0 = __riscv_vmv_v_x_u32m1(0, __riscv_vsetvlmax_e32m1());
|
||||
vuint32m1_t acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e8m1(cols - x);
|
||||
vuint8m1x4_t v = __riscv_vlseg4e8_v_u8m1x4(src + x * 4, vl);
|
||||
vuint8m1_t v0 = __riscv_vget_v_u8m1x4_u8m1(v, 0);
|
||||
vuint8m1_t v1 = __riscv_vget_v_u8m1x4_u8m1(v, 1);
|
||||
vuint8m1_t v2 = __riscv_vget_v_u8m1x4_u8m1(v, 2);
|
||||
vuint8m1_t v3 = __riscv_vget_v_u8m1x4_u8m1(v, 3);
|
||||
acc0 = __riscv_vwredsumu(__riscv_vwmulu(v0, v0, vl), acc0, vl);
|
||||
acc1 = __riscv_vwredsumu(__riscv_vwmulu(v1, v1, vl), acc1, vl);
|
||||
acc2 = __riscv_vwredsumu(__riscv_vwmulu(v2, v2, vl), acc2, vl);
|
||||
acc3 = __riscv_vwredsumu(__riscv_vwmulu(v3, v3, vl), acc3, vl);
|
||||
x += vl;
|
||||
}
|
||||
dst[0] = (DT)(int32_t)(uint32_t)__riscv_vmv_x(acc0);
|
||||
dst[1] = (DT)(int32_t)(uint32_t)__riscv_vmv_x(acc1);
|
||||
dst[2] = (DT)(int32_t)(uint32_t)__riscv_vmv_x(acc2);
|
||||
dst[3] = (DT)(int32_t)(uint32_t)__riscv_vmv_x(acc3);
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = 0;
|
||||
int x = 0;
|
||||
const int vlmax = __riscv_vsetvlmax_e32m8();
|
||||
vfloat32m8_t acc = __riscv_vfmv_v_f_f32m8(0, vlmax);
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e32m8(cols - x);
|
||||
vfloat32m8_t v = __riscv_vle32_v_f32m8(src + x, vl);
|
||||
acc = __riscv_vfmacc_tu(acc, v, v, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat32m1_t zero = __riscv_vfmv_s_f_f32m1(0, __riscv_vsetvlmax_e32m1());
|
||||
result = __riscv_vfmv_f(__riscv_vfredusum(acc, zero, vlmax));
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const int vlmax = __riscv_vsetvlmax_e32m2();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
vfloat32m2_t acc0 = __riscv_vfmv_v_f_f32m2(0, vlmax);
|
||||
vfloat32m2_t acc1 = acc0, acc2 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e32m2(cols - x);
|
||||
vfloat32m2x3_t v = __riscv_vlseg3e32_v_f32m2x3(src + x * 3, vl);
|
||||
vfloat32m2_t v0 = __riscv_vget_v_f32m2x3_f32m2(v, 0);
|
||||
vfloat32m2_t v1 = __riscv_vget_v_f32m2x3_f32m2(v, 1);
|
||||
vfloat32m2_t v2 = __riscv_vget_v_f32m2x3_f32m2(v, 2);
|
||||
acc0 = __riscv_vfmacc_tu(acc0, v0, v0, vl);
|
||||
acc1 = __riscv_vfmacc_tu(acc1, v1, v1, vl);
|
||||
acc2 = __riscv_vfmacc_tu(acc2, v2, v2, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat32m1_t zero = __riscv_vfmv_s_f_f32m1(0, __riscv_vsetvlmax_e32m1());
|
||||
dst[0] = __riscv_vfmv_f(__riscv_vfredusum(acc0, zero, vlmax));
|
||||
dst[1] = __riscv_vfmv_f(__riscv_vfredusum(acc1, zero, vlmax));
|
||||
dst[2] = __riscv_vfmv_f(__riscv_vfredusum(acc2, zero, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const int vlmax = __riscv_vsetvlmax_e32m2();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
vfloat32m2_t acc0 = __riscv_vfmv_v_f_f32m2(0, vlmax);
|
||||
vfloat32m2_t acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x < cols; )
|
||||
{
|
||||
const int vl = __riscv_vsetvl_e32m2(cols - x);
|
||||
vfloat32m2x4_t v = __riscv_vlseg4e32_v_f32m2x4(src + x * 4, vl);
|
||||
vfloat32m2_t v0 = __riscv_vget_v_f32m2x4_f32m2(v, 0);
|
||||
vfloat32m2_t v1 = __riscv_vget_v_f32m2x4_f32m2(v, 1);
|
||||
vfloat32m2_t v2 = __riscv_vget_v_f32m2x4_f32m2(v, 2);
|
||||
vfloat32m2_t v3 = __riscv_vget_v_f32m2x4_f32m2(v, 3);
|
||||
acc0 = __riscv_vfmacc_tu(acc0, v0, v0, vl);
|
||||
acc1 = __riscv_vfmacc_tu(acc1, v1, v1, vl);
|
||||
acc2 = __riscv_vfmacc_tu(acc2, v2, v2, vl);
|
||||
acc3 = __riscv_vfmacc_tu(acc3, v3, v3, vl);
|
||||
x += vl;
|
||||
}
|
||||
vfloat32m1_t zero = __riscv_vfmv_s_f_f32m1(0, __riscv_vsetvlmax_e32m1());
|
||||
dst[0] = __riscv_vfmv_f(__riscv_vfredusum(acc0, zero, vlmax));
|
||||
dst[1] = __riscv_vfmv_f(__riscv_vfredusum(acc1, zero, vlmax));
|
||||
dst[2] = __riscv_vfmv_f(__riscv_vfredusum(acc2, zero, vlmax));
|
||||
dst[3] = __riscv_vfmv_f(__riscv_vfredusum(acc3, zero, vlmax));
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
} // namespace reduce_c_rvv
|
||||
@@ -170,7 +170,7 @@ int Core_ReduceTest::checkOp( const Mat& src, int dstType, int opType, const Mat
|
||||
getMatTypeStr( dstType, dstTypeStr );
|
||||
const char* dimStr = dim == 0 ? "ROWS" : "COLS";
|
||||
|
||||
snprintf( msg, sizeof(msg), "bad accuracy with srcType = %s, dstType = %s, opType = %s, dim = %s",
|
||||
snprintf( msg, sizeof(msg), "bad accuracy with srcType = %s, dstType = %s, opType = %s, dim = %s\n",
|
||||
srcTypeStr.c_str(), dstTypeStr.c_str(), opTypeStr, dimStr );
|
||||
ts->printf( cvtest::TS::LOG, msg );
|
||||
return cvtest::TS::FAIL_BAD_ACCURACY;
|
||||
|
||||
@@ -6,6 +6,8 @@
|
||||
#include "npy_blob.hpp"
|
||||
#include <opencv2/dnn/shape_utils.hpp>
|
||||
#include <opencv2/dnn/all_layers.hpp>
|
||||
#include <iostream>
|
||||
|
||||
namespace opencv_test { namespace {
|
||||
|
||||
testing::internal::ParamGenerator< tuple<Backend, Target> > dnnBackendsAndTargetsInt8()
|
||||
@@ -34,6 +36,7 @@ public:
|
||||
int numInps = 1, int numOuts = 1, bool useCaffeModel = false,
|
||||
bool useCommonInputBlob = true, bool hasText = false, bool perChannel = true)
|
||||
{
|
||||
std::cout << "Testning layer " << basename << std::endl;
|
||||
CV_Assert_N(numInps >= 1, numInps <= 10, numOuts >= 1, numOuts <= 10);
|
||||
std::vector<Mat> inps(numInps), inps_int8(numInps);
|
||||
std::vector<Mat> refs(numOuts), outs_int8(numOuts), outs_dequantized(numOuts);
|
||||
@@ -239,7 +242,10 @@ TEST_P(Test_Int8_layers, MaxPooling)
|
||||
|
||||
TEST_P(Test_Int8_layers, Reduce)
|
||||
{
|
||||
testLayer("reduce_mean", "TensorFlow", 0.0005, 0.0014);
|
||||
// Test fails on some CI hosts
|
||||
if (backend != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
testLayer("reduce_mean", "TensorFlow", 0.0005, 0.0014);
|
||||
|
||||
testLayer("reduce_mean", "ONNX", 0.00062, 0.0014);
|
||||
testLayer("reduce_mean_axis1", "ONNX", 0.00032, 0.0007);
|
||||
testLayer("reduce_mean_axis2", "ONNX", 0.00033, 0.001);
|
||||
|
||||
@@ -328,6 +328,42 @@ TEST(Features2d_FLANN_Composite, regression) { CV_FlannCompositeIndexTest test;
|
||||
TEST(Features2d_FLANN_Auto, regression) { CV_FlannAutotunedIndexTest test; test.safe_run(); }
|
||||
TEST(Features2d_FLANN_Saved, regression) { CV_FlannSavedIndexTest test; test.safe_run(); }
|
||||
|
||||
// A saved KD-tree index whose serialized leaf node carries a point index
|
||||
// outside the dataset must be rejected on load. Before the added validation
|
||||
// the malformed index loaded silently and the out-of-range index was
|
||||
// dereferenced during search (heap out-of-bounds access).
|
||||
TEST(Features2d_FLANN_KDTree, load_rejects_out_of_range_leaf_index)
|
||||
{
|
||||
Mat features(1, 4, CV_32F);
|
||||
features.at<float>(0, 0) = 1.f; features.at<float>(0, 1) = 2.f;
|
||||
features.at<float>(0, 2) = 3.f; features.at<float>(0, 3) = 4.f;
|
||||
|
||||
const String filename = tempfile();
|
||||
{
|
||||
Index index(features, KDTreeIndexParams(1));
|
||||
index.save(filename);
|
||||
}
|
||||
|
||||
// Overwrite the single leaf node's divfeat (the first int of the last
|
||||
// serialized Node record) with an index far outside the 1-row dataset.
|
||||
{
|
||||
FILE* f = fopen(filename.c_str(), "r+b");
|
||||
ASSERT_TRUE(f != NULL);
|
||||
ASSERT_EQ(0, fseek(f, 0, SEEK_END));
|
||||
const long node_size = (long)(sizeof(int) + sizeof(float) + 2 * sizeof(void*));
|
||||
const long size = ftell(f);
|
||||
ASSERT_GT(size, node_size);
|
||||
ASSERT_EQ(0, fseek(f, size - node_size, SEEK_SET));
|
||||
const int out_of_range = 1 << 28;
|
||||
ASSERT_EQ((size_t)1, fwrite(&out_of_range, sizeof(int), 1, f));
|
||||
fclose(f);
|
||||
}
|
||||
|
||||
Index loaded;
|
||||
EXPECT_THROW(loaded.load(features, filename), cv::Exception);
|
||||
remove(filename.c_str());
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
}} // namespace
|
||||
|
||||
@@ -266,11 +266,25 @@ private:
|
||||
{
|
||||
tree = pool_.allocate<Node>();
|
||||
load_value(stream, *tree);
|
||||
if (tree->child1!=NULL) {
|
||||
load_tree(stream, tree->child1);
|
||||
if (tree->child1!=NULL || tree->child2!=NULL) {
|
||||
// Internal node: divfeat is the split dimension and is used to index
|
||||
// the query vector during search, so it must be a valid dimension.
|
||||
if (tree->divfeat < 0 || (size_t)tree->divfeat >= veclen_) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree index: split dimension is out of range");
|
||||
}
|
||||
if (tree->child1!=NULL) {
|
||||
load_tree(stream, tree->child1);
|
||||
}
|
||||
if (tree->child2!=NULL) {
|
||||
load_tree(stream, tree->child2);
|
||||
}
|
||||
}
|
||||
if (tree->child2!=NULL) {
|
||||
load_tree(stream, tree->child2);
|
||||
else {
|
||||
// Leaf node: divfeat is a dataset point index dereferenced during
|
||||
// search, so it must fall inside the dataset.
|
||||
if (tree->divfeat < 0 || (size_t)tree->divfeat >= size_) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree index: leaf feature index is out of range");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -158,12 +158,33 @@ public:
|
||||
{
|
||||
load_value(stream, size_);
|
||||
load_value(stream, dim_);
|
||||
// The dataset the index is attached to is fixed by the caller, so a
|
||||
// saved index whose stored size/dim disagree with it is malformed.
|
||||
if (size_ != dataset_.rows || dim_ != dataset_.cols) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree(single) index: saved dataset dimensions do not match");
|
||||
}
|
||||
load_value(stream, root_bbox_);
|
||||
if (root_bbox_.size() != dim_) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree(single) index: bounding box has wrong length");
|
||||
}
|
||||
load_value(stream, reorder_);
|
||||
load_value(stream, leaf_max_size_);
|
||||
load_value(stream, vind_);
|
||||
// vind_ holds one dataset point index per row and every entry is
|
||||
// dereferenced during search.
|
||||
if (vind_.size() != size_) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree(single) index: index permutation has wrong length");
|
||||
}
|
||||
for (size_t i = 0; i < vind_.size(); ++i) {
|
||||
if (vind_[i] < 0 || (size_t)vind_[i] >= size_) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree(single) index: point index is out of range");
|
||||
}
|
||||
}
|
||||
if (reorder_) {
|
||||
load_value(stream, data_);
|
||||
if (data_.rows != size_ || data_.cols != dim_) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree(single) index: reordered data has wrong shape");
|
||||
}
|
||||
}
|
||||
else {
|
||||
data_ = dataset_;
|
||||
@@ -303,11 +324,25 @@ private:
|
||||
{
|
||||
tree = pool_.allocate<Node>();
|
||||
load_value(stream, *tree);
|
||||
if (tree->child1!=NULL) {
|
||||
load_tree(stream, tree->child1);
|
||||
if (tree->child1!=NULL || tree->child2!=NULL) {
|
||||
// Internal node: divfeat is the split dimension, used to index the
|
||||
// query vector and the per-dimension distance array during search.
|
||||
if (tree->divfeat < 0 || (size_t)tree->divfeat >= dim_) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree(single) index: split dimension is out of range");
|
||||
}
|
||||
if (tree->child1!=NULL) {
|
||||
load_tree(stream, tree->child1);
|
||||
}
|
||||
if (tree->child2!=NULL) {
|
||||
load_tree(stream, tree->child2);
|
||||
}
|
||||
}
|
||||
if (tree->child2!=NULL) {
|
||||
load_tree(stream, tree->child2);
|
||||
else {
|
||||
// Leaf node: [left, right) is a range of point slots dereferenced
|
||||
// during search, so it must stay inside the dataset.
|
||||
if (tree->left < 0 || tree->right < tree->left || (size_t)tree->right > size_) {
|
||||
FLANN_THROW(cv::Error::StsParseError, "FLANN kd-tree(single) index: leaf point range is out of range");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -77,7 +77,7 @@ TEST_P(GStreamerSourceTest, AccuracyTest)
|
||||
|
||||
EXPECT_FALSE(ccomp.running());
|
||||
|
||||
EXPECT_EQ(streamLength, framesCount);
|
||||
EXPECT_NEAR(streamLength, framesCount, 1);
|
||||
}
|
||||
|
||||
TEST_P(GStreamerSourceTest, TimestampsTest)
|
||||
@@ -124,12 +124,12 @@ TEST_P(GStreamerSourceTest, TimestampsTest)
|
||||
EXPECT_FALSE(ccomp.running());
|
||||
|
||||
EXPECT_EQ(0L, allSeqIds.front());
|
||||
EXPECT_EQ(int64_t(streamLength) - 1, allSeqIds.back());
|
||||
EXPECT_EQ(streamLength, allSeqIds.size());
|
||||
EXPECT_NEAR(int64_t(streamLength) - 1, allSeqIds.back(), 1);
|
||||
EXPECT_NEAR(streamLength, allSeqIds.size(), 1);
|
||||
EXPECT_TRUE(std::is_sorted(allSeqIds.begin(), allSeqIds.end()));
|
||||
EXPECT_EQ(allSeqIds.size(), std::set<int64_t>(allSeqIds.begin(), allSeqIds.end()).size());
|
||||
|
||||
EXPECT_EQ(streamLength, allTimestamps.size());
|
||||
EXPECT_NEAR(streamLength, allTimestamps.size(), 1);
|
||||
EXPECT_TRUE(std::is_sorted(allTimestamps.begin(), allTimestamps.end()));
|
||||
}
|
||||
|
||||
@@ -270,7 +270,7 @@ TEST_P(GStreamerSourceTest, GFrameTest)
|
||||
|
||||
EXPECT_FALSE(ccomp.running());
|
||||
|
||||
EXPECT_EQ(streamLength, framesCount);
|
||||
EXPECT_NEAR(streamLength, framesCount, 1);
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -11,6 +11,8 @@ ocv_add_dispatched_file(morph SSE2 SSE4_1 AVX2)
|
||||
ocv_add_dispatched_file(smooth SSE2 SSE4_1 AVX2 AVX512_ICL)
|
||||
ocv_add_dispatched_file(sumpixels SSE2 AVX2 AVX512_SKX)
|
||||
ocv_add_dispatched_file(equalize_hist AVX512_ICL)
|
||||
ocv_add_dispatched_file(imgwarp SSE4_1 AVX2 AVX512_SKX AVX512_ICL)
|
||||
ocv_add_dispatched_file(pyramids_avx512_vbmi AVX512_ICL)
|
||||
ocv_define_module(imgproc opencv_core WRAP java objc python js)
|
||||
|
||||
if(OPENCV_CORE_EXCLUDE_C_API)
|
||||
|
||||
@@ -688,152 +688,6 @@ calcHist_8u( std::vector<uchar*>& _ptrs, const std::vector<int>& _deltas,
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
|
||||
typedef IppStatus(CV_STDCALL * IppiHistogram_C1)(const void* pSrc, int srcStep,
|
||||
IppiSize roiSize, Ipp32u* pHist, const IppiHistogramSpec* pSpec, Ipp8u* pBuffer);
|
||||
|
||||
static IppiHistogram_C1 getIppiHistogramFunction_C1(int type)
|
||||
{
|
||||
IppiHistogram_C1 ippFunction =
|
||||
(type == CV_8UC1) ? (IppiHistogram_C1)ippiHistogram_8u_C1R :
|
||||
(type == CV_16UC1) ? (IppiHistogram_C1)ippiHistogram_16u_C1R :
|
||||
(type == CV_32FC1) ? (IppiHistogram_C1)ippiHistogram_32f_C1R :
|
||||
NULL;
|
||||
|
||||
return ippFunction;
|
||||
}
|
||||
|
||||
class ipp_calcHistParallelTLS
|
||||
{
|
||||
public:
|
||||
ipp_calcHistParallelTLS() {}
|
||||
|
||||
IppAutoBuffer<IppiHistogramSpec> spec;
|
||||
IppAutoBuffer<Ipp8u> buffer;
|
||||
IppAutoBuffer<Ipp32u> thist;
|
||||
};
|
||||
|
||||
class ipp_calcHistParallel: public ParallelLoopBody
|
||||
{
|
||||
public:
|
||||
ipp_calcHistParallel(const Mat &src, Mat &hist, Ipp32s histSize, const float *ranges, bool uniform, bool &ok):
|
||||
ParallelLoopBody(), m_src(src), m_hist(hist), m_ok(ok)
|
||||
{
|
||||
ok = true;
|
||||
|
||||
m_uniform = uniform;
|
||||
m_ranges = ranges;
|
||||
m_histSize = histSize;
|
||||
m_type = ippiGetDataType(src.type());
|
||||
m_levelsNum = histSize+1;
|
||||
ippiHistogram_C1 = getIppiHistogramFunction_C1(src.type());
|
||||
m_fullRoi = ippiSize(src.size());
|
||||
m_bufferSize = 0;
|
||||
m_specSize = 0;
|
||||
if(!ippiHistogram_C1)
|
||||
{
|
||||
ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(ippiHistogramGetBufferSize(m_type, m_fullRoi, &m_levelsNum, 1, 1, &m_specSize, &m_bufferSize) < 0)
|
||||
{
|
||||
ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
hist.setTo(0);
|
||||
}
|
||||
|
||||
virtual void operator() (const Range & range) const CV_OVERRIDE
|
||||
{
|
||||
CV_INSTRUMENT_REGION_IPP();
|
||||
|
||||
if(!m_ok)
|
||||
return;
|
||||
|
||||
ipp_calcHistParallelTLS *pTls = m_tls.get();
|
||||
|
||||
IppiSize roi = {m_src.cols, range.end - range.start };
|
||||
bool mtLoop = false;
|
||||
if(m_fullRoi.height != roi.height)
|
||||
mtLoop = true;
|
||||
|
||||
if(!pTls->spec)
|
||||
{
|
||||
pTls->spec.allocate(m_specSize);
|
||||
if(!pTls->spec.get())
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
pTls->buffer.allocate(m_bufferSize);
|
||||
if(!pTls->buffer.get() && m_bufferSize)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(m_uniform)
|
||||
{
|
||||
if(ippiHistogramUniformInit(m_type, (Ipp32f*)&m_ranges[0], (Ipp32f*)&m_ranges[1], (Ipp32s*)&m_levelsNum, 1, pTls->spec) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if(ippiHistogramInit(m_type, (const Ipp32f**)&m_ranges, (Ipp32s*)&m_levelsNum, 1, pTls->spec) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
pTls->thist.allocate(m_histSize*sizeof(Ipp32u));
|
||||
}
|
||||
|
||||
if(CV_INSTRUMENT_FUN_IPP(ippiHistogram_C1, m_src.ptr(range.start), (int)m_src.step, roi, pTls->thist, pTls->spec, pTls->buffer) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(mtLoop)
|
||||
{
|
||||
for(int i = 0; i < m_histSize; i++)
|
||||
CV_XADD((int*)(m_hist.ptr(i)), *(int*)((Ipp32u*)pTls->thist + i));
|
||||
}
|
||||
else
|
||||
ippiCopy_32s_C1R((Ipp32s*)pTls->thist.get(), sizeof(Ipp32u), (Ipp32s*)m_hist.ptr(), (int)m_hist.step, ippiSize(1, m_histSize));
|
||||
}
|
||||
|
||||
private:
|
||||
const Mat &m_src;
|
||||
Mat &m_hist;
|
||||
Ipp32s m_histSize;
|
||||
const float *m_ranges;
|
||||
bool m_uniform;
|
||||
|
||||
IppiHistogram_C1 ippiHistogram_C1;
|
||||
IppiSize m_fullRoi;
|
||||
IppDataType m_type;
|
||||
Ipp32s m_levelsNum;
|
||||
int m_bufferSize;
|
||||
int m_specSize;
|
||||
|
||||
mutable Mutex m_syncMutex;
|
||||
TLSData<ipp_calcHistParallelTLS> m_tls;
|
||||
|
||||
volatile bool &m_ok;
|
||||
const ipp_calcHistParallel & operator = (const ipp_calcHistParallel & );
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
}
|
||||
|
||||
#ifdef HAVE_OPENVX
|
||||
@@ -894,58 +748,6 @@ namespace cv
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#define IPP_HISTOGRAM_PARALLEL 1
|
||||
namespace cv
|
||||
{
|
||||
static bool ipp_calchist(const Mat &image, Mat &hist, int histSize, const float** ranges, bool uniform, bool accumulate)
|
||||
{
|
||||
CV_INSTRUMENT_REGION_IPP();
|
||||
|
||||
#if IPP_VERSION_X100 < 201801
|
||||
// No SSE42 optimization for uniform 32f
|
||||
if(uniform && image.depth() == CV_32F && cv::ipp::getIppTopFeatures() == ippCPUID_SSE42)
|
||||
return false;
|
||||
#endif
|
||||
|
||||
// IPP_DISABLE_HISTOGRAM - https://github.com/opencv/opencv/issues/11544
|
||||
// and https://github.com/opencv/opencv/issues/21595
|
||||
if ((uniform && (ranges[0][1] - ranges[0][0]) != histSize) || abs(ranges[0][0]) != cvFloor(ranges[0][0]))
|
||||
return false;
|
||||
|
||||
Mat ihist = hist;
|
||||
if(accumulate)
|
||||
ihist.create(1, &histSize, CV_32S);
|
||||
|
||||
bool ok = true;
|
||||
int threads = ippiSuggestThreadsNum(image, (1+((double)ihist.total()/image.total()))*2);
|
||||
Range range(0, image.rows);
|
||||
ipp_calcHistParallel invoker(image, ihist, histSize, ranges[0], uniform, ok);
|
||||
if(!ok)
|
||||
return false;
|
||||
|
||||
if(IPP_HISTOGRAM_PARALLEL && threads > 1)
|
||||
parallel_for_(range, invoker, threads*2);
|
||||
else
|
||||
invoker(range);
|
||||
|
||||
if(ok)
|
||||
{
|
||||
if(accumulate)
|
||||
{
|
||||
IppiSize histRoi = ippiSize(1, histSize);
|
||||
IppAutoBuffer<Ipp32f> fhist(histSize*sizeof(Ipp32f));
|
||||
CV_INSTRUMENT_FUN_IPP(ippiConvert_32s32f_C1R, (Ipp32s*)ihist.ptr(), (int)ihist.step, (Ipp32f*)fhist, sizeof(Ipp32f), histRoi);
|
||||
CV_INSTRUMENT_FUN_IPP(ippiAdd_32f_C1IR, (Ipp32f*)fhist, sizeof(Ipp32f), (Ipp32f*)hist.ptr(), (int)hist.step, histRoi);
|
||||
}
|
||||
else
|
||||
CV_INSTRUMENT_FUN_IPP(ippiConvert_32s32f_C1R, (Ipp32s*)ihist.ptr(), (int)ihist.step, (Ipp32f*)hist.ptr(), (int)hist.step, ippiSize(1, histSize));
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
void cv::calcHist( const Mat* images, int nimages, const int* channels,
|
||||
InputArray _mask, OutputArray _hist, int dims, const int* histSize,
|
||||
const float** ranges, bool uniform, bool accumulate )
|
||||
@@ -973,11 +775,6 @@ void cv::calcHist( const Mat* images, int nimages, const int* channels,
|
||||
if(histdata != hist.data)
|
||||
accumulate = false;
|
||||
|
||||
CV_IPP_RUN(
|
||||
nimages == 1 && dims == 1 && channels && channels[0] == 0
|
||||
&& _mask.empty() && images[0].dims <= 2 && ranges && ranges[0],
|
||||
ipp_calchist(images[0], hist, histSize[0], ranges, uniform, accumulate));
|
||||
|
||||
if (nimages == 1 && dims == 1 && channels && channels[0] == 0 && _mask.empty() && images[0].dims <= 2 && ranges && ranges[0]) {
|
||||
CALL_HAL(calcHist, cv_hal_calcHist, images[0].data, images[0].step, images[0].type(), images[0].cols, images[0].rows,
|
||||
hist.ptr<float>(), histSize[0], ranges, uniform, accumulate);
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
// Copyright (C) 2000-2008, Intel Corporation, all rights reserved.
|
||||
// Copyright (C) 2009, Willow Garage Inc., all rights reserved.
|
||||
// Copyright (C) 2014-2015, Itseez Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without modification,
|
||||
@@ -95,4 +96,7 @@ int warpAffineBlockline(int *adelta, int *bdelta, short* xy, short* alpha, int X
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
#include "imgwarp.simd.hpp"
|
||||
|
||||
/* End of file. */
|
||||
|
||||
+219
-40
@@ -13,6 +13,7 @@
|
||||
// Copyright (C) 2000-2008, Intel Corporation, all rights reserved.
|
||||
// Copyright (C) 2009, Willow Garage Inc., all rights reserved.
|
||||
// Copyright (C) 2014-2015, Itseez Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without modification,
|
||||
@@ -55,6 +56,9 @@
|
||||
#include "opencv2/core/softfloat.hpp"
|
||||
#include "imgwarp.hpp"
|
||||
|
||||
#include "imgwarp.simd.hpp"
|
||||
#include "imgwarp.simd_declarations.hpp" // defines CV_CPU_DISPATCH_MODES_ALL=AVX512_ICL,...,BASELINE based on CMakeLists.txt content
|
||||
|
||||
using namespace cv;
|
||||
|
||||
namespace cv
|
||||
@@ -611,6 +615,81 @@ template<bool isRelative> using RemapVec_8u = RemapNoVec<isRelative>;
|
||||
|
||||
#endif
|
||||
|
||||
template<typename T, typename AT>
|
||||
struct RemapBilinearVecC1
|
||||
{
|
||||
int operator()(const T*, size_t, T*, const short*, const ushort*,
|
||||
const AT*, int, int, int) const { return 0; }
|
||||
};
|
||||
|
||||
template<>
|
||||
struct RemapBilinearVecC1<float, float>
|
||||
{
|
||||
int operator()(const float* S0, size_t sstep, float* D, const short* XY,
|
||||
const ushort* FXY, const float* wtab, int dx, int X1, int off_y) const
|
||||
{
|
||||
CV_CPU_DISPATCH(remapBilinearC1_simd,
|
||||
(CV_32F, (const uchar*)S0, sstep, (uchar*)D, XY, FXY, wtab, dx, X1, off_y),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct RemapBilinearVecC1<ushort, float>
|
||||
{
|
||||
int operator()(const ushort* S0, size_t sstep, ushort* D, const short* XY,
|
||||
const ushort* FXY, const float* wtab, int dx, int X1, int off_y) const
|
||||
{
|
||||
CV_CPU_DISPATCH(remapBilinearC1_simd,
|
||||
(CV_16U, (const uchar*)S0, sstep, (uchar*)D, XY, FXY, wtab, dx, X1, off_y),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct RemapBilinearVecC1<short, float>
|
||||
{
|
||||
int operator()(const short* S0, size_t sstep, short* D, const short* XY,
|
||||
const ushort* FXY, const float* wtab, int dx, int X1, int off_y) const
|
||||
{
|
||||
CV_CPU_DISPATCH(remapBilinearC1_simd,
|
||||
(CV_16S, (const uchar*)S0, sstep, (uchar*)D, XY, FXY, wtab, dx, X1, off_y),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
};
|
||||
|
||||
static inline int remapBilinearSameRun( const short* XY, int dx, int end,
|
||||
unsigned width1, unsigned height1, bool inl )
|
||||
{
|
||||
int n = 0;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
const int span = VTraits<v_int16>::vlanes();
|
||||
const v_int16 vw = vx_setall_s16((short)std::min<unsigned>(width1, 0x7fff));
|
||||
const v_int16 vh = vx_setall_s16((short)std::min<unsigned>(height1, 0x7fff));
|
||||
const v_int16 vm1 = vx_setall_s16(-1);
|
||||
for( ; dx + n + span <= end; n += span )
|
||||
{
|
||||
v_int16 sx, sy;
|
||||
v_load_deinterleave(XY + (dx + n) * 2, sx, sy);
|
||||
// in-bounds: 0 <= sx < width1 && 0 <= sy < height1
|
||||
v_int16 inb = v_and(v_and(v_gt(sx, vm1), v_lt(sx, vw)),
|
||||
v_and(v_gt(sy, vm1), v_lt(sy, vh)));
|
||||
const bool allSame = inl ? v_check_all(inb) : !v_check_any(inb);
|
||||
if( !allSame )
|
||||
break;
|
||||
}
|
||||
vx_cleanup();
|
||||
#endif
|
||||
for( ; dx + n < end; n++ )
|
||||
{
|
||||
const int sx = XY[(dx + n) * 2], sy = XY[(dx + n) * 2 + 1];
|
||||
const bool ib = (unsigned)sx < width1 && (unsigned)sy < height1;
|
||||
if( ib != inl )
|
||||
break;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
template<class CastOp, class VecOp, typename AT, bool isRelative>
|
||||
static void remapBilinear( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
const Mat& _fxy, const void* _wtab,
|
||||
@@ -647,6 +726,12 @@ static void remapBilinear( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
const int off_y = (isRelative ? (_offset.y+dy) : 0);
|
||||
for(int dx = 0; dx <= dsize.width; dx++ )
|
||||
{
|
||||
if( !isRelative && dx < dsize.width )
|
||||
{
|
||||
int n = remapBilinearSameRun(XY, dx, dsize.width, width1, height1, prevInlier);
|
||||
if( n > 0 )
|
||||
dx += n - 1;
|
||||
}
|
||||
bool curInlier = dx < dsize.width ?
|
||||
(unsigned)XY[dx*2]+(isRelative ? (_offset.x+dx) : 0) < width1 &&
|
||||
(unsigned)XY[dx*2+1]+off_y < height1 : !prevInlier;
|
||||
@@ -667,6 +752,11 @@ static void remapBilinear( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
|
||||
if( cn == 1 )
|
||||
{
|
||||
if( !isRelative )
|
||||
{
|
||||
int n = RemapBilinearVecC1<T, AT>()(S0, sstep, D, XY, FXY, wtab, dx, X1, off_y);
|
||||
D += n; dx += n;
|
||||
}
|
||||
for( ; dx < X1; dx++, D++ )
|
||||
{
|
||||
int sx = XY[dx*2]+(isRelative ? (_offset.x+dx) : 0), sy = XY[dx*2+1]+off_y;
|
||||
@@ -843,6 +933,53 @@ static void remapBilinear( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
}
|
||||
|
||||
|
||||
// Dispatch shim for the single-channel non-relative bicubic in-bounds fast path (32F only).
|
||||
template<typename T, typename AT>
|
||||
struct RemapBicubicVecC1
|
||||
{
|
||||
int operator()(const T*, size_t, T*, const short*, const ushort*, const AT*,
|
||||
int, int, unsigned, unsigned, int) const { return 0; }
|
||||
};
|
||||
|
||||
template<>
|
||||
struct RemapBicubicVecC1<float, float>
|
||||
{
|
||||
int operator()(const float* S0, size_t sstep, float* D, const short* XY,
|
||||
const ushort* FXY, const float* wtab, int dx, int dwidth, unsigned width1,
|
||||
unsigned height1, int off_y) const
|
||||
{
|
||||
CV_CPU_DISPATCH(remapBicubicC1wp_simd,
|
||||
(CV_32F, (const uchar*)S0, sstep, (uchar*)D, XY, FXY, wtab, dx, dwidth, width1, height1, off_y),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct RemapBicubicVecC1<ushort, float>
|
||||
{
|
||||
int operator()(const ushort* S0, size_t sstep, ushort* D, const short* XY,
|
||||
const ushort* FXY, const float* wtab, int dx, int dwidth, unsigned width1,
|
||||
unsigned height1, int off_y) const
|
||||
{
|
||||
CV_CPU_DISPATCH(remapBicubicC1wp_simd,
|
||||
(CV_16U, (const uchar*)S0, sstep, (uchar*)D, XY, FXY, wtab, dx, dwidth, width1, height1, off_y),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct RemapBicubicVecC1<short, float>
|
||||
{
|
||||
int operator()(const short* S0, size_t sstep, short* D, const short* XY,
|
||||
const ushort* FXY, const float* wtab, int dx, int dwidth, unsigned width1,
|
||||
unsigned height1, int off_y) const
|
||||
{
|
||||
CV_CPU_DISPATCH(remapBicubicC1wp_simd,
|
||||
(CV_16S, (const uchar*)S0, sstep, (uchar*)D, XY, FXY, wtab, dx, dwidth, width1, height1, off_y),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
};
|
||||
|
||||
template<class CastOp, typename AT, int ONE, bool isRelative>
|
||||
static void remapBicubic( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
const Mat& _fxy, const void* _wtab,
|
||||
@@ -879,6 +1016,12 @@ static void remapBicubic( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
const int off_y = isRelative ? (_offset.y+dy) : 0;
|
||||
for(int dx = 0; dx < dsize.width; dx++, D += cn )
|
||||
{
|
||||
if( cn == 1 && !isRelative )
|
||||
{
|
||||
int n = RemapBicubicVecC1<T, AT>()(S0, sstep, D, XY, FXY, wtab, dx,
|
||||
dsize.width, width1, height1, off_y);
|
||||
if( n > 0 ) { D += (n - 1)*cn; dx += n - 1; continue; }
|
||||
}
|
||||
const int off_x = isRelative ? (_offset.x+dx) : 0;
|
||||
int sx = XY[dx*2]-1+off_x, sy = XY[dx*2+1]-1+off_y;
|
||||
const AT* w = wtab + FXY[dx]*16;
|
||||
@@ -948,6 +1091,33 @@ static void remapBicubic( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
}
|
||||
|
||||
|
||||
template<typename T, typename AT>
|
||||
struct RemapLanczos4VecC1
|
||||
{
|
||||
int operator()(const T*, size_t, T*, const short*, const ushort*, const AT*,
|
||||
int, int, unsigned, unsigned, int) const { return 0; }
|
||||
};
|
||||
|
||||
#define CV_REMAP_LANCZOS4_SHIM(T, DEPTH) \
|
||||
template<> struct RemapLanczos4VecC1<T, float> \
|
||||
{ \
|
||||
int operator()(const T* S0, size_t sstep, T* D, const short* XY, \
|
||||
const ushort* FXY, const float* wtab, int dx, int dwidth, \
|
||||
unsigned width1, unsigned height1, int off_y) const \
|
||||
{ \
|
||||
CV_CPU_DISPATCH(remapLanczos4C1_simd, \
|
||||
(DEPTH, (const uchar*)S0, sstep, (uchar*)D, XY, FXY, wtab, \
|
||||
dx, dwidth, width1, height1, off_y), \
|
||||
CV_CPU_DISPATCH_MODES_ALL); \
|
||||
} \
|
||||
};
|
||||
// 32F is intentionally not shimmed: its vectorized accumulation deviates beyond
|
||||
// the float accuracy tolerance, so it stays on the scalar loop. Emitting a shim
|
||||
// would add a per-pixel dispatch call that returns 0 and slows the scalar path.
|
||||
CV_REMAP_LANCZOS4_SHIM(ushort, CV_16U)
|
||||
CV_REMAP_LANCZOS4_SHIM(short, CV_16S)
|
||||
#undef CV_REMAP_LANCZOS4_SHIM
|
||||
|
||||
template<class CastOp, typename AT, int ONE, bool isRelative>
|
||||
static void remapLanczos4( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
const Mat& _fxy, const void* _wtab,
|
||||
@@ -984,6 +1154,12 @@ static void remapLanczos4( const Mat& _src, Mat& _dst, const Mat& _xy,
|
||||
const int off_y = isRelative ? (_offset.y+dy) : 0;
|
||||
for(int dx = 0; dx < dsize.width; dx++, D += cn )
|
||||
{
|
||||
if( cn == 1 && !isRelative )
|
||||
{
|
||||
int n = RemapLanczos4VecC1<T, AT>()(S0, sstep, D, XY, FXY, wtab, dx,
|
||||
dsize.width, width1, height1, off_y);
|
||||
if( n > 0 ) { D += (n - 1)*cn; dx += n - 1; continue; }
|
||||
}
|
||||
const int off_x = isRelative ? (_offset.x+dx) : 0;
|
||||
int sx = XY[dx*2]-3+off_x, sy = XY[dx*2+1]-3+off_y;
|
||||
const AT* w = wtab + FXY[dx]*64;
|
||||
@@ -1131,21 +1307,21 @@ public:
|
||||
const float* sY = m2->ptr<float>(y+y1) + x;
|
||||
x1 = 0;
|
||||
|
||||
#if CV_SIMD128
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
{
|
||||
int span = VTraits<v_float32x4>::vlanes();
|
||||
int span = VTraits<v_float32>::vlanes();
|
||||
for( ; x1 <= bcols - span * 2; x1 += span * 2 )
|
||||
{
|
||||
v_int32x4 ix0 = v_round(v_load(sX + x1));
|
||||
v_int32x4 iy0 = v_round(v_load(sY + x1));
|
||||
v_int32x4 ix1 = v_round(v_load(sX + x1 + span));
|
||||
v_int32x4 iy1 = v_round(v_load(sY + x1 + span));
|
||||
v_int32 ix0 = v_round(vx_load(sX + x1));
|
||||
v_int32 iy0 = v_round(vx_load(sY + x1));
|
||||
v_int32 ix1 = v_round(vx_load(sX + x1 + span));
|
||||
v_int32 iy1 = v_round(vx_load(sY + x1 + span));
|
||||
|
||||
v_int16x8 dx, dy;
|
||||
dx = v_pack(ix0, ix1);
|
||||
dy = v_pack(iy0, iy1);
|
||||
v_int16 dx = v_pack(ix0, ix1);
|
||||
v_int16 dy = v_pack(iy0, iy1);
|
||||
v_store_interleave(XY + x1 * 2, dx, dy);
|
||||
}
|
||||
vx_cleanup();
|
||||
}
|
||||
#endif
|
||||
for( ; x1 < bcols; x1++ )
|
||||
@@ -1172,12 +1348,13 @@ public:
|
||||
const ushort* sA = m2->ptr<ushort>(y+y1) + x;
|
||||
x1 = 0;
|
||||
|
||||
#if CV_SIMD128
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
{
|
||||
v_uint16x8 v_scale = v_setall_u16(INTER_TAB_SIZE2 - 1);
|
||||
int span = VTraits<v_uint16x8>::vlanes();
|
||||
v_uint16 v_scale = vx_setall_u16(INTER_TAB_SIZE2 - 1);
|
||||
int span = VTraits<v_uint16>::vlanes();
|
||||
for( ; x1 <= bcols - span; x1 += span )
|
||||
v_store((unsigned short*)(A + x1), v_and(v_load(sA + x1), v_scale));
|
||||
v_store((unsigned short*)(A + x1), v_and(vx_load(sA + x1), v_scale));
|
||||
vx_cleanup();
|
||||
}
|
||||
#endif
|
||||
for( ; x1 < bcols; x1++ )
|
||||
@@ -1189,26 +1366,27 @@ public:
|
||||
const float* sY = m2->ptr<float>(y+y1) + x;
|
||||
|
||||
x1 = 0;
|
||||
#if CV_SIMD128
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
{
|
||||
v_float32x4 v_scale = v_setall_f32((float)INTER_TAB_SIZE);
|
||||
v_int32x4 v_scale2 = v_setall_s32(INTER_TAB_SIZE - 1);
|
||||
int span = VTraits<v_float32x4>::vlanes();
|
||||
v_float32 v_scale = vx_setall_f32((float)INTER_TAB_SIZE);
|
||||
v_int32 v_scale2 = vx_setall_s32(INTER_TAB_SIZE - 1);
|
||||
int span = VTraits<v_float32>::vlanes();
|
||||
for( ; x1 <= bcols - span * 2; x1 += span * 2 )
|
||||
{
|
||||
v_int32x4 v_sx0 = v_round(v_mul(v_scale, v_load(sX + x1)));
|
||||
v_int32x4 v_sy0 = v_round(v_mul(v_scale, v_load(sY + x1)));
|
||||
v_int32x4 v_sx1 = v_round(v_mul(v_scale, v_load(sX + x1 + span)));
|
||||
v_int32x4 v_sy1 = v_round(v_mul(v_scale, v_load(sY + x1 + span)));
|
||||
v_uint16x8 v_sx8 = v_reinterpret_as_u16(v_pack(v_and(v_sx0, v_scale2), v_and(v_sx1, v_scale2)));
|
||||
v_uint16x8 v_sy8 = v_reinterpret_as_u16(v_pack(v_and(v_sy0, v_scale2), v_and(v_sy1, v_scale2)));
|
||||
v_uint16x8 v_v = v_or(v_shl<INTER_BITS>(v_sy8), v_sx8);
|
||||
v_int32 v_sx0 = v_round(v_mul(v_scale, vx_load(sX + x1)));
|
||||
v_int32 v_sy0 = v_round(v_mul(v_scale, vx_load(sY + x1)));
|
||||
v_int32 v_sx1 = v_round(v_mul(v_scale, vx_load(sX + x1 + span)));
|
||||
v_int32 v_sy1 = v_round(v_mul(v_scale, vx_load(sY + x1 + span)));
|
||||
v_uint16 v_sx8 = v_reinterpret_as_u16(v_pack(v_and(v_sx0, v_scale2), v_and(v_sx1, v_scale2)));
|
||||
v_uint16 v_sy8 = v_reinterpret_as_u16(v_pack(v_and(v_sy0, v_scale2), v_and(v_sy1, v_scale2)));
|
||||
v_uint16 v_v = v_or(v_shl<INTER_BITS>(v_sy8), v_sx8);
|
||||
v_store(A + x1, v_v);
|
||||
|
||||
v_int16x8 v_d0 = v_pack(v_shr<INTER_BITS>(v_sx0), v_shr<INTER_BITS>(v_sx1));
|
||||
v_int16x8 v_d1 = v_pack(v_shr<INTER_BITS>(v_sy0), v_shr<INTER_BITS>(v_sy1));
|
||||
v_int16 v_d0 = v_pack(v_shr<INTER_BITS>(v_sx0), v_shr<INTER_BITS>(v_sx1));
|
||||
v_int16 v_d1 = v_pack(v_shr<INTER_BITS>(v_sy0), v_shr<INTER_BITS>(v_sy1));
|
||||
v_store_interleave(XY + (x1 << 1), v_d0, v_d1);
|
||||
}
|
||||
vx_cleanup();
|
||||
}
|
||||
#endif
|
||||
for( ; x1 < bcols; x1++ )
|
||||
@@ -1226,28 +1404,29 @@ public:
|
||||
const float* sXY = m1->ptr<float>(y+y1) + x*2;
|
||||
x1 = 0;
|
||||
|
||||
#if CV_SIMD128
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
{
|
||||
v_float32x4 v_scale = v_setall_f32((float)INTER_TAB_SIZE);
|
||||
v_int32x4 v_scale2 = v_setall_s32(INTER_TAB_SIZE - 1), v_scale3 = v_setall_s32(INTER_TAB_SIZE);
|
||||
int span = VTraits<v_float32x4>::vlanes();
|
||||
v_float32 v_scale = vx_setall_f32((float)INTER_TAB_SIZE);
|
||||
v_int32 v_scale2 = vx_setall_s32(INTER_TAB_SIZE - 1), v_scale3 = vx_setall_s32(INTER_TAB_SIZE);
|
||||
int span = VTraits<v_float32>::vlanes();
|
||||
for( ; x1 <= bcols - span * 2; x1 += span * 2 )
|
||||
{
|
||||
v_float32x4 v_fx, v_fy;
|
||||
v_float32 v_fx, v_fy;
|
||||
v_load_deinterleave(sXY + (x1 << 1), v_fx, v_fy);
|
||||
v_int32x4 v_sx0 = v_round(v_mul(v_fx, v_scale));
|
||||
v_int32x4 v_sy0 = v_round(v_mul(v_fy, v_scale));
|
||||
v_int32 v_sx0 = v_round(v_mul(v_fx, v_scale));
|
||||
v_int32 v_sy0 = v_round(v_mul(v_fy, v_scale));
|
||||
v_load_deinterleave(sXY + ((x1 + span) << 1), v_fx, v_fy);
|
||||
v_int32x4 v_sx1 = v_round(v_mul(v_fx, v_scale));
|
||||
v_int32x4 v_sy1 = v_round(v_mul(v_fy, v_scale));
|
||||
v_int32x4 v_v0 = v_muladd(v_scale3, (v_and(v_sy0, v_scale2)), (v_and(v_sx0, v_scale2)));
|
||||
v_int32x4 v_v1 = v_muladd(v_scale3, (v_and(v_sy1, v_scale2)), (v_and(v_sx1, v_scale2)));
|
||||
v_uint16x8 v_v8 = v_reinterpret_as_u16(v_pack(v_v0, v_v1));
|
||||
v_int32 v_sx1 = v_round(v_mul(v_fx, v_scale));
|
||||
v_int32 v_sy1 = v_round(v_mul(v_fy, v_scale));
|
||||
v_int32 v_v0 = v_muladd(v_scale3, (v_and(v_sy0, v_scale2)), (v_and(v_sx0, v_scale2)));
|
||||
v_int32 v_v1 = v_muladd(v_scale3, (v_and(v_sy1, v_scale2)), (v_and(v_sx1, v_scale2)));
|
||||
v_uint16 v_v8 = v_reinterpret_as_u16(v_pack(v_v0, v_v1));
|
||||
v_store(A + x1, v_v8);
|
||||
v_int16x8 v_dx = v_pack(v_shr<INTER_BITS>(v_sx0), v_shr<INTER_BITS>(v_sx1));
|
||||
v_int16x8 v_dy = v_pack(v_shr<INTER_BITS>(v_sy0), v_shr<INTER_BITS>(v_sy1));
|
||||
v_int16 v_dx = v_pack(v_shr<INTER_BITS>(v_sx0), v_shr<INTER_BITS>(v_sx1));
|
||||
v_int16 v_dy = v_pack(v_shr<INTER_BITS>(v_sy0), v_shr<INTER_BITS>(v_sy1));
|
||||
v_store_interleave(XY + (x1 << 1), v_dx, v_dy);
|
||||
}
|
||||
vx_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
@@ -0,0 +1,478 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
//
|
||||
// Copyright (C) 2026, Advanced Micro Devices, all rights reserved.
|
||||
|
||||
#include "opencv2/core/hal/intrin.hpp"
|
||||
|
||||
namespace cv {
|
||||
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
int remapBilinearC1_simd(int depth, const uchar* S0, size_t sstep, uchar* D,
|
||||
const short* XY, const ushort* FXY, const float* wtab,
|
||||
int dx, int X1, int off_y);
|
||||
|
||||
int remapBicubicC1_simd(int depth, const uchar* S0, size_t sstep, uchar* D,
|
||||
const short* XY, const ushort* FXY,
|
||||
int dx, int dwidth, unsigned width1, unsigned height1, int off_y);
|
||||
|
||||
int remapLanczos4C1_simd(int depth, const uchar* S0, size_t sstep, uchar* D,
|
||||
const short* XY, const ushort* FXY, const float* wtab,
|
||||
int dx, int dwidth, unsigned width1, unsigned height1, int off_y);
|
||||
|
||||
|
||||
int remapBicubicC1wp_simd(int depth, const uchar* S0, size_t sstep, uchar* D,
|
||||
const short* XY, const ushort* FXY, const float* wtab,
|
||||
int dx, int dwidth, unsigned width1, unsigned height1, int off_y);
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
static inline v_float32 remapGatherF32(const float* base, const int* ofs)
|
||||
{
|
||||
float CV_DECL_ALIGNED(CV_SIMD_WIDTH) buf[VTraits<v_float32>::max_nlanes];
|
||||
const int n = VTraits<v_float32>::vlanes();
|
||||
for (int k = 0; k < n; k++)
|
||||
buf[k] = base[ofs[k]];
|
||||
return vx_load(buf);
|
||||
}
|
||||
|
||||
static inline v_float32 remapGatherF32(const ushort* base, const int* ofs)
|
||||
{
|
||||
float CV_DECL_ALIGNED(CV_SIMD_WIDTH) buf[VTraits<v_float32>::max_nlanes];
|
||||
const int n = VTraits<v_float32>::vlanes();
|
||||
for (int k = 0; k < n; k++)
|
||||
buf[k] = (float)base[ofs[k]];
|
||||
return vx_load(buf);
|
||||
}
|
||||
|
||||
static inline v_float32 remapGatherF32(const short* base, const int* ofs)
|
||||
{
|
||||
float CV_DECL_ALIGNED(CV_SIMD_WIDTH) buf[VTraits<v_float32>::max_nlanes];
|
||||
const int n = VTraits<v_float32>::vlanes();
|
||||
for (int k = 0; k < n; k++)
|
||||
buf[k] = (float)base[ofs[k]];
|
||||
return vx_load(buf);
|
||||
}
|
||||
|
||||
static CV_ALWAYS_INLINE void remapCorners(const short* S0, size_t sstep, const int* ofs,
|
||||
v_float32& s0, v_float32& s1,
|
||||
v_float32& s2, v_float32& s3)
|
||||
{
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) topbuf[VTraits<v_float32>::max_nlanes];
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) botbuf[VTraits<v_float32>::max_nlanes];
|
||||
const int n = VTraits<v_float32>::vlanes();
|
||||
for (int k = 0; k < n; k++)
|
||||
{
|
||||
const short* p = S0 + ofs[k];
|
||||
int t, b;
|
||||
memcpy(&t, p, sizeof(t));
|
||||
memcpy(&b, p + sstep, sizeof(b));
|
||||
topbuf[k] = t; botbuf[k] = b;
|
||||
}
|
||||
v_int32 top = vx_load(topbuf), bot = vx_load(botbuf);
|
||||
s0 = v_cvt_f32(v_shr<16>(v_shl<16>(top))); // low 16 bits (sign-extended)
|
||||
s1 = v_cvt_f32(v_shr<16>(top)); // high 16 bits (sign-extended)
|
||||
s2 = v_cvt_f32(v_shr<16>(v_shl<16>(bot)));
|
||||
s3 = v_cvt_f32(v_shr<16>(bot));
|
||||
}
|
||||
|
||||
static CV_ALWAYS_INLINE void remapCorners(const ushort* S0, size_t sstep, const int* ofs,
|
||||
v_float32& s0, v_float32& s1,
|
||||
v_float32& s2, v_float32& s3)
|
||||
{
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) topbuf[VTraits<v_float32>::max_nlanes];
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) botbuf[VTraits<v_float32>::max_nlanes];
|
||||
const int n = VTraits<v_float32>::vlanes();
|
||||
for (int k = 0; k < n; k++)
|
||||
{
|
||||
const ushort* p = S0 + ofs[k];
|
||||
int t, b;
|
||||
memcpy(&t, p, sizeof(t));
|
||||
memcpy(&b, p + sstep, sizeof(b));
|
||||
topbuf[k] = t; botbuf[k] = b;
|
||||
}
|
||||
const v_uint32 lo16 = vx_setall_u32(0xffff);
|
||||
v_uint32 top = v_reinterpret_as_u32(vx_load(topbuf));
|
||||
v_uint32 bot = v_reinterpret_as_u32(vx_load(botbuf));
|
||||
s0 = v_cvt_f32(v_reinterpret_as_s32(v_and(top, lo16))); // low 16 (zero-ext)
|
||||
s1 = v_cvt_f32(v_reinterpret_as_s32(v_shr<16>(top))); // high 16 (zero-ext)
|
||||
s2 = v_cvt_f32(v_reinterpret_as_s32(v_and(bot, lo16)));
|
||||
s3 = v_cvt_f32(v_reinterpret_as_s32(v_shr<16>(bot)));
|
||||
}
|
||||
|
||||
static inline void remapStoreC1(float* D, const v_float32& res)
|
||||
{
|
||||
v_store(D, res);
|
||||
}
|
||||
|
||||
static inline void remapStoreC1(ushort* D, const v_float32& res)
|
||||
{
|
||||
v_pack_u_store(D, v_round(res));
|
||||
}
|
||||
|
||||
static inline void remapStoreC1(short* D, const v_float32& res)
|
||||
{
|
||||
v_pack_store(D, v_round(res));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static int remapBilinearC1_run(const T* S0, size_t sstep, T* D,
|
||||
const short* XY, const ushort* FXY,
|
||||
const float* wtab, int dx, int X1, int off_y)
|
||||
{
|
||||
CV_UNUSED(wtab);
|
||||
const int vlanes = VTraits<v_float32>::vlanes();
|
||||
const int dx0 = dx;
|
||||
const v_float32 vone = vx_setall_f32(1.f);
|
||||
const v_float32 vscale = vx_setall_f32(1.f / INTER_TAB_SIZE);
|
||||
const v_int32 vmask = vx_setall_s32(INTER_TAB_SIZE - 1);
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) ofs[VTraits<v_float32>::max_nlanes];
|
||||
for( ; dx <= X1 - vlanes; dx += vlanes )
|
||||
{
|
||||
for( int k = 0; k < vlanes; k++ )
|
||||
{
|
||||
const int sx = XY[(dx + k) * 2];
|
||||
const int sy = XY[(dx + k) * 2 + 1] + off_y;
|
||||
ofs[k] = sy * (int)sstep + sx;
|
||||
}
|
||||
v_float32 s0, s1, s2, s3;
|
||||
remapCorners(S0, sstep, ofs, s0, s1, s2, s3);
|
||||
|
||||
v_int32 fxy = v_reinterpret_as_s32(vx_load_expand(FXY + dx));
|
||||
v_float32 fx = v_mul(v_cvt_f32(v_and(fxy, vmask)), vscale);
|
||||
v_float32 fy = v_mul(v_cvt_f32(v_shr<INTER_BITS>(fxy)), vscale);
|
||||
v_float32 cx0 = v_sub(vone, fx);
|
||||
v_float32 cy0 = v_sub(vone, fy);
|
||||
v_float32 w0 = v_mul(cx0, cy0);
|
||||
v_float32 w1 = v_mul(fx, cy0);
|
||||
v_float32 w2 = v_mul(cx0, fy);
|
||||
v_float32 w3 = v_mul(fx, fy);
|
||||
|
||||
v_float32 res = v_fma(s0, w0, v_fma(s1, w1, v_fma(s2, w2, v_mul(s3, w3))));
|
||||
remapStoreC1(D + (dx - dx0), res);
|
||||
}
|
||||
vx_cleanup();
|
||||
return dx - dx0;
|
||||
}
|
||||
|
||||
static int remapBilinearF32_run(const float* S0, size_t sstep, float* D,
|
||||
const short* XY, const ushort* FXY,
|
||||
const float* wtab, int dx, int X1, int off_y)
|
||||
{
|
||||
CV_UNUSED(wtab);
|
||||
const int vlanes = VTraits<v_float32>::vlanes();
|
||||
const int dx0 = dx;
|
||||
const float* S1 = S0 + sstep;
|
||||
const v_float32 vone = vx_setall_f32(1.f);
|
||||
const v_float32 vscale = vx_setall_f32(1.f / INTER_TAB_SIZE);
|
||||
const v_int32 vmask = vx_setall_s32(INTER_TAB_SIZE - 1);
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) ofs[VTraits<v_float32>::max_nlanes];
|
||||
for( ; dx <= X1 - vlanes; dx += vlanes )
|
||||
{
|
||||
for( int k = 0; k < vlanes; k++ )
|
||||
{
|
||||
const int sx = XY[(dx + k) * 2];
|
||||
const int sy = XY[(dx + k) * 2 + 1] + off_y;
|
||||
ofs[k] = sy * (int)sstep + sx;
|
||||
}
|
||||
v_float32 s0 = remapGatherF32(S0, ofs);
|
||||
v_float32 s1 = remapGatherF32(S0 + 1, ofs);
|
||||
v_float32 s2 = remapGatherF32(S1, ofs);
|
||||
v_float32 s3 = remapGatherF32(S1 + 1, ofs);
|
||||
|
||||
v_int32 fxy = v_reinterpret_as_s32(vx_load_expand(FXY + dx));
|
||||
v_float32 fx = v_mul(v_cvt_f32(v_and(fxy, vmask)), vscale);
|
||||
v_float32 fy = v_mul(v_cvt_f32(v_shr<INTER_BITS>(fxy)), vscale);
|
||||
v_float32 cx0 = v_sub(vone, fx);
|
||||
v_float32 cy0 = v_sub(vone, fy);
|
||||
v_float32 w0 = v_mul(cx0, cy0);
|
||||
v_float32 w1 = v_mul(fx, cy0);
|
||||
v_float32 w2 = v_mul(cx0, fy);
|
||||
v_float32 w3 = v_mul(fx, fy);
|
||||
|
||||
v_float32 res = v_fma(s0, w0, v_fma(s1, w1, v_fma(s2, w2, v_mul(s3, w3))));
|
||||
v_store(D + (dx - dx0), res);
|
||||
}
|
||||
vx_cleanup();
|
||||
return dx - dx0;
|
||||
}
|
||||
|
||||
// Evaluate the four cubic interpolation coefficients for a vector of fractional
|
||||
// positions, matching interpolateCubic() (A = -0.75). The strict remap test
|
||||
// tolerates an absolute error of 1.0 for bicubic, so FMA contraction is fine.
|
||||
static inline void interpolateCubicV(const v_float32& x,
|
||||
v_float32& c0, v_float32& c1,
|
||||
v_float32& c2, v_float32& c3)
|
||||
{
|
||||
const v_float32 A = vx_setall_f32(-0.75f);
|
||||
const v_float32 A5 = vx_setall_f32(-3.75f); // 5*A
|
||||
const v_float32 A8 = vx_setall_f32(-6.0f); // 8*A
|
||||
const v_float32 A4 = vx_setall_f32(-3.0f); // 4*A
|
||||
const v_float32 Ap2 = vx_setall_f32(1.25f); // A+2
|
||||
const v_float32 Ap3 = vx_setall_f32(2.25f); // A+3
|
||||
const v_float32 one = vx_setall_f32(1.f);
|
||||
|
||||
v_float32 xp1 = v_add(x, one);
|
||||
c0 = v_sub(v_mul(v_add(v_mul(v_sub(v_mul(A, xp1), A5), xp1), A8), xp1), A4);
|
||||
c1 = v_add(v_mul(v_mul(v_sub(v_mul(Ap2, x), Ap3), x), x), one);
|
||||
v_float32 u = v_sub(one, x);
|
||||
c2 = v_add(v_mul(v_mul(v_sub(v_mul(Ap2, u), Ap3), u), u), one);
|
||||
c3 = v_sub(v_sub(v_sub(one, c0), c1), c2);
|
||||
}
|
||||
|
||||
// Select one of four vectors by runtime index. Used instead of an array of
|
||||
// vector types, which is not valid for sizeless RVV vector types.
|
||||
static inline v_float32 selectV4(int i, const v_float32& a, const v_float32& b,
|
||||
const v_float32& c, const v_float32& d)
|
||||
{
|
||||
return i == 0 ? a : (i == 1 ? b : (i == 2 ? c : d));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static int remapBicubicC1_run(const T* S0, size_t sstep, T* D, const short* XY,
|
||||
const ushort* FXY, int dx, int dwidth,
|
||||
unsigned width1, unsigned height1, int off_y)
|
||||
{
|
||||
const int vlanes = VTraits<v_float32>::vlanes();
|
||||
const int dx0 = dx;
|
||||
const v_float32 vscale = vx_setall_f32(1.f / INTER_TAB_SIZE);
|
||||
const v_int32 vmask = vx_setall_s32(INTER_TAB_SIZE - 1);
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) ofs[VTraits<v_float32>::max_nlanes];
|
||||
float CV_DECL_ALIGNED(CV_SIMD_WIDTH) buf[VTraits<v_float32>::max_nlanes];
|
||||
for( ; dx <= dwidth - vlanes; dx += vlanes )
|
||||
{
|
||||
bool allIn = true;
|
||||
for( int k = 0; k < vlanes; k++ )
|
||||
{
|
||||
const unsigned sx = (unsigned)(XY[(dx + k) * 2] - 1);
|
||||
const unsigned sy = (unsigned)(XY[(dx + k) * 2 + 1] - 1 + off_y);
|
||||
if( sx >= width1 || sy >= height1 ) { allIn = false; break; }
|
||||
ofs[k] = (int)((XY[(dx + k) * 2 + 1] - 1 + off_y) * (int)sstep
|
||||
+ (XY[(dx + k) * 2] - 1));
|
||||
}
|
||||
if( !allIn )
|
||||
break;
|
||||
|
||||
v_int32 fxy = v_reinterpret_as_s32(vx_load_expand(FXY + dx));
|
||||
v_float32 fx = v_mul(v_cvt_f32(v_and(fxy, vmask)), vscale);
|
||||
v_float32 fy = v_mul(v_cvt_f32(v_shr<INTER_BITS>(fxy)), vscale);
|
||||
v_float32 vx0, vx1, vx2, vx3, vy0, vy1, vy2, vy3;
|
||||
interpolateCubicV(fx, vx0, vx1, vx2, vx3);
|
||||
interpolateCubicV(fy, vy0, vy1, vy2, vy3);
|
||||
|
||||
v_float32 acc = vx_setzero_f32();
|
||||
for( int r = 0; r < 4; r++ )
|
||||
{
|
||||
const int roff = r * (int)sstep;
|
||||
const v_float32 vyr = selectV4(r, vy0, vy1, vy2, vy3);
|
||||
for( int c = 0; c < 4; c++ )
|
||||
{
|
||||
for( int k = 0; k < vlanes; k++ )
|
||||
buf[k] = (float)S0[ofs[k] + roff + c];
|
||||
acc = v_fma(vx_load(buf), v_mul(vyr, selectV4(c, vx0, vx1, vx2, vx3)), acc);
|
||||
}
|
||||
}
|
||||
remapStoreC1(D + (dx - dx0), acc);
|
||||
}
|
||||
vx_cleanup();
|
||||
return dx - dx0;
|
||||
}
|
||||
|
||||
#if CV_SIMD128
|
||||
static inline void remapLoad8(const float* S, v_float32x4& lo, v_float32x4& hi)
|
||||
{
|
||||
lo = v_load(S); hi = v_load(S + 4);
|
||||
}
|
||||
static inline void remapLoad8(const ushort* S, v_float32x4& lo, v_float32x4& hi)
|
||||
{
|
||||
v_uint16x8 v = v_load(S);
|
||||
v_uint32x4 a, b; v_expand(v, a, b);
|
||||
lo = v_cvt_f32(v_reinterpret_as_s32(a));
|
||||
hi = v_cvt_f32(v_reinterpret_as_s32(b));
|
||||
}
|
||||
static inline void remapLoad8(const short* S, v_float32x4& lo, v_float32x4& hi)
|
||||
{
|
||||
v_int16x8 v = v_load(S);
|
||||
v_int32x4 a, b; v_expand(v, a, b);
|
||||
lo = v_cvt_f32(a); hi = v_cvt_f32(b);
|
||||
}
|
||||
|
||||
static inline void remapStoreScalar(float* D, float v) { *D = v; }
|
||||
static inline void remapStoreScalar(ushort* D, float v) { *D = saturate_cast<ushort>(v); }
|
||||
static inline void remapStoreScalar(short* D, float v) { *D = saturate_cast<short>(v); }
|
||||
|
||||
static inline v_float32x4 remapLoad4(const float* S) { return v_load(S); }
|
||||
static inline v_float32x4 remapLoad4(const ushort* S) { return v_cvt_f32(v_reinterpret_as_s32(v_load_expand(S))); }
|
||||
static inline v_float32x4 remapLoad4(const short* S) { return v_cvt_f32(v_load_expand(S)); }
|
||||
|
||||
template<typename T>
|
||||
static int remapBicubicC1wp_run(const T* S0, size_t sstep, T* D, const short* XY,
|
||||
const ushort* FXY, const float* wtab, int dx,
|
||||
int dwidth, unsigned width1, unsigned height1, int off_y)
|
||||
{
|
||||
const int dx0 = dx;
|
||||
for( ; dx < dwidth; dx++ )
|
||||
{
|
||||
const unsigned sx = (unsigned)(XY[dx * 2] - 1);
|
||||
const unsigned sy = (unsigned)(XY[dx * 2 + 1] - 1 + off_y);
|
||||
if( sx >= width1 || sy >= height1 )
|
||||
break;
|
||||
const float* w = wtab + FXY[dx] * 16;
|
||||
const T* S = S0 + (size_t)sy * sstep + sx;
|
||||
v_float32x4 acc = v_setzero_f32();
|
||||
for( int r = 0; r < 4; r++, S += sstep, w += 4 )
|
||||
acc = v_fma(remapLoad4(S), v_load(w), acc);
|
||||
remapStoreScalar(D + (dx - dx0), v_reduce_sum(acc));
|
||||
}
|
||||
vx_cleanup();
|
||||
return dx - dx0;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static int remapLanczos4C1_run(const T* S0, size_t sstep, T* D, const short* XY,
|
||||
const ushort* FXY, const float* wtab, int dx,
|
||||
int dwidth, unsigned width1, unsigned height1, int off_y)
|
||||
{
|
||||
const int dx0 = dx;
|
||||
for( ; dx < dwidth; dx++ )
|
||||
{
|
||||
const unsigned sx = (unsigned)(XY[dx * 2] - 3);
|
||||
const unsigned sy = (unsigned)(XY[dx * 2 + 1] - 3 + off_y);
|
||||
if( sx >= width1 || sy >= height1 )
|
||||
break;
|
||||
const float* w = wtab + FXY[dx] * 64;
|
||||
const T* S = S0 + (size_t)sy * sstep + sx;
|
||||
v_float32x4 acc = v_setzero_f32();
|
||||
for( int r = 0; r < 8; r++, S += sstep, w += 8 )
|
||||
{
|
||||
v_float32x4 s_lo, s_hi;
|
||||
remapLoad8(S, s_lo, s_hi);
|
||||
acc = v_fma(s_lo, v_load(w), acc);
|
||||
acc = v_fma(s_hi, v_load(w + 4), acc);
|
||||
}
|
||||
remapStoreScalar(D + (dx - dx0), v_reduce_sum(acc));
|
||||
}
|
||||
vx_cleanup();
|
||||
return dx - dx0;
|
||||
}
|
||||
#endif // CV_SIMD128
|
||||
|
||||
#endif // CV_SIMD
|
||||
|
||||
int remapBilinearC1_simd(int depth, const uchar* S0, size_t sstep, uchar* D,
|
||||
const short* XY, const ushort* FXY, const float* wtab,
|
||||
int dx, int X1, int off_y)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
switch (depth)
|
||||
{
|
||||
case CV_32F:
|
||||
return remapBilinearF32_run((const float*)S0, sstep, (float*)D,
|
||||
XY, FXY, wtab, dx, X1, off_y);
|
||||
case CV_16U:
|
||||
return remapBilinearC1_run<ushort>((const ushort*)S0, sstep, (ushort*)D,
|
||||
XY, FXY, wtab, dx, X1, off_y);
|
||||
case CV_16S:
|
||||
return remapBilinearC1_run<short>((const short*)S0, sstep, (short*)D,
|
||||
XY, FXY, wtab, dx, X1, off_y);
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
CV_UNUSED(depth); CV_UNUSED(S0); CV_UNUSED(sstep); CV_UNUSED(D);
|
||||
CV_UNUSED(XY); CV_UNUSED(FXY); CV_UNUSED(wtab);
|
||||
CV_UNUSED(dx); CV_UNUSED(X1); CV_UNUSED(off_y);
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
int remapBicubicC1_simd(int depth, const uchar* S0, size_t sstep, uchar* D,
|
||||
const short* XY, const ushort* FXY,
|
||||
int dx, int dwidth, unsigned width1, unsigned height1, int off_y)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
switch (depth)
|
||||
{
|
||||
case CV_32F:
|
||||
return remapBicubicC1_run<float>((const float*)S0, sstep, (float*)D,
|
||||
XY, FXY, dx, dwidth, width1, height1, off_y);
|
||||
case CV_16U:
|
||||
return remapBicubicC1_run<ushort>((const ushort*)S0, sstep, (ushort*)D,
|
||||
XY, FXY, dx, dwidth, width1, height1, off_y);
|
||||
case CV_16S:
|
||||
return remapBicubicC1_run<short>((const short*)S0, sstep, (short*)D,
|
||||
XY, FXY, dx, dwidth, width1, height1, off_y);
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
CV_UNUSED(depth); CV_UNUSED(S0); CV_UNUSED(sstep); CV_UNUSED(D);
|
||||
CV_UNUSED(XY); CV_UNUSED(FXY); CV_UNUSED(dx); CV_UNUSED(dwidth);
|
||||
CV_UNUSED(width1); CV_UNUSED(height1); CV_UNUSED(off_y);
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
int remapLanczos4C1_simd(int depth, const uchar* S0, size_t sstep, uchar* D,
|
||||
const short* XY, const ushort* FXY, const float* wtab,
|
||||
int dx, int dwidth, unsigned width1, unsigned height1, int off_y)
|
||||
{
|
||||
#if CV_SIMD128
|
||||
switch (depth)
|
||||
{
|
||||
// CV_32F is intentionally omitted: the vectorized 8x8 accumulation reorders
|
||||
// the 64-tap sum, so on the tight 1e-3 float tolerance it diverges from the
|
||||
// scalar path (used for relative maps). 32F lanczos4 stays on the scalar loop.
|
||||
case CV_16U:
|
||||
return remapLanczos4C1_run<ushort>((const ushort*)S0, sstep, (ushort*)D,
|
||||
XY, FXY, wtab, dx, dwidth, width1, height1, off_y);
|
||||
case CV_16S:
|
||||
return remapLanczos4C1_run<short>((const short*)S0, sstep, (short*)D,
|
||||
XY, FXY, wtab, dx, dwidth, width1, height1, off_y);
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
CV_UNUSED(depth); CV_UNUSED(S0); CV_UNUSED(sstep); CV_UNUSED(D);
|
||||
CV_UNUSED(XY); CV_UNUSED(FXY); CV_UNUSED(wtab); CV_UNUSED(dx); CV_UNUSED(dwidth);
|
||||
CV_UNUSED(width1); CV_UNUSED(height1); CV_UNUSED(off_y);
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
int remapBicubicC1wp_simd(int depth, const uchar* S0, size_t sstep, uchar* D,
|
||||
const short* XY, const ushort* FXY, const float* wtab,
|
||||
int dx, int dwidth, unsigned width1, unsigned height1, int off_y)
|
||||
{
|
||||
#if CV_SIMD128
|
||||
switch (depth)
|
||||
{
|
||||
case CV_32F:
|
||||
return remapBicubicC1wp_run<float>((const float*)S0, sstep, (float*)D,
|
||||
XY, FXY, wtab, dx, dwidth, width1, height1, off_y);
|
||||
case CV_16U:
|
||||
return remapBicubicC1wp_run<ushort>((const ushort*)S0, sstep, (ushort*)D,
|
||||
XY, FXY, wtab, dx, dwidth, width1, height1, off_y);
|
||||
case CV_16S:
|
||||
return remapBicubicC1wp_run<short>((const short*)S0, sstep, (short*)D,
|
||||
XY, FXY, wtab, dx, dwidth, width1, height1, off_y);
|
||||
default:
|
||||
return 0;
|
||||
}
|
||||
#else
|
||||
CV_UNUSED(depth); CV_UNUSED(S0); CV_UNUSED(sstep); CV_UNUSED(D);
|
||||
CV_UNUSED(XY); CV_UNUSED(FXY); CV_UNUSED(wtab); CV_UNUSED(dx); CV_UNUSED(dwidth);
|
||||
CV_UNUSED(width1); CV_UNUSED(height1); CV_UNUSED(off_y);
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
#endif // CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_END
|
||||
|
||||
} // namespace cv
|
||||
@@ -13,6 +13,7 @@
|
||||
// Copyright (C) 2000-2008, Intel Corporation, all rights reserved.
|
||||
// Copyright (C) 2009, Willow Garage Inc., all rights reserved.
|
||||
// Copyright (C) 2014-2015, Itseez Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without modification,
|
||||
@@ -466,4 +467,7 @@ Ptr<WarpPerspectiveLine_SSE4> WarpPerspectiveLine_SSE4::getImpl(const double *M)
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
#include "imgwarp.simd.hpp"
|
||||
|
||||
/* End of file. */
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
// Copyright (C) 2000-2008, Intel Corporation, all rights reserved.
|
||||
// Copyright (C) 2009, Willow Garage Inc., all rights reserved.
|
||||
// Copyright (C) 2014-2015, Itseez Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without modification,
|
||||
@@ -54,6 +55,12 @@ template<typename T, int shift> struct FixPtCast
|
||||
typedef T rtype;
|
||||
rtype operator ()(type1 arg) const { return (T)((arg + (1 << (shift-1))) >> shift); }
|
||||
};
|
||||
template<typename T, int shift> struct FixPtUcharCast
|
||||
{
|
||||
typedef ushort type1;
|
||||
typedef T rtype;
|
||||
rtype operator ()(type1 arg) const { return (T)((arg + (1 << (shift-1))) >> shift); }
|
||||
};
|
||||
|
||||
template<typename T, int shift> struct FltCast
|
||||
{
|
||||
@@ -83,6 +90,76 @@ template<typename T1, typename T2> int PyrUpVecV(T1**, T2**, int) { return 0; }
|
||||
template<typename T1, typename T2> int PyrUpVecVOneRow(T1**, T2*, int) { return 0; }
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
// Dispatched AVX-512 VBMI implementations
|
||||
int PyrDownVecH_uchar_ushort_1_dispatch(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_2_dispatch(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_3_dispatch(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_4_dispatch(const uchar* src, ushort* row, int width);
|
||||
|
||||
// uchar -> ushort intermediate storage implementations
|
||||
template<> int PyrDownVecH<uchar, ushort, 1>(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
int x = PyrDownVecH_uchar_ushort_1_dispatch(src, row, width);
|
||||
return x;
|
||||
}
|
||||
|
||||
template<> int PyrDownVecH<uchar, ushort, 2>(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
int x = PyrDownVecH_uchar_ushort_2_dispatch(src, row, width);
|
||||
return x;
|
||||
}
|
||||
|
||||
template<> int PyrDownVecH<uchar, ushort, 3>(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
int x = PyrDownVecH_uchar_ushort_3_dispatch(src, row, width);
|
||||
return x;
|
||||
}
|
||||
|
||||
template<> int PyrDownVecH<uchar, ushort, 4>(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
int x = PyrDownVecH_uchar_ushort_4_dispatch(src, row, width);
|
||||
return x;
|
||||
|
||||
}
|
||||
|
||||
template<> int PyrDownVecV<ushort, uchar>(ushort** src, uchar* dst, int width)
|
||||
{
|
||||
int x = 0;
|
||||
const ushort *row0 = src[0], *row1 = src[1], *row2 = src[2], *row3 = src[3], *row4 = src[4];
|
||||
|
||||
for( ; x <= width - VTraits<v_uint8>::vlanes(); x += VTraits<v_uint8>::vlanes() )
|
||||
{
|
||||
v_uint16 r0, r1, r2, r3, r4, t0, t1;
|
||||
r0 = vx_load(row0 + x);
|
||||
r1 = vx_load(row1 + x);
|
||||
r2 = vx_load(row2 + x);
|
||||
r3 = vx_load(row3 + x);
|
||||
r4 = vx_load(row4 + x);
|
||||
t0 = v_add(v_add(v_add(r0, r4), v_add(r2, r2)), v_shl<2>(v_add(v_add(r1, r3), r2)));
|
||||
r0 = vx_load(row0 + x + VTraits<v_uint16>::vlanes());
|
||||
r1 = vx_load(row1 + x + VTraits<v_uint16>::vlanes());
|
||||
r2 = vx_load(row2 + x + VTraits<v_uint16>::vlanes());
|
||||
r3 = vx_load(row3 + x + VTraits<v_uint16>::vlanes());
|
||||
r4 = vx_load(row4 + x + VTraits<v_uint16>::vlanes());
|
||||
t1 = v_add(v_add(v_add(r0, r4), v_add(r2, r2)), v_shl<2>(v_add(v_add(r1, r3), r2)));
|
||||
v_store(dst + x, v_rshr_pack<8>(t0, t1));
|
||||
}
|
||||
if (x <= width - VTraits<v_uint16>::vlanes())
|
||||
{
|
||||
v_uint16 r0, r1, r2, r3, r4, t0;
|
||||
r0 = vx_load(row0 + x);
|
||||
r1 = vx_load(row1 + x);
|
||||
r2 = vx_load(row2 + x);
|
||||
r3 = vx_load(row3 + x);
|
||||
r4 = vx_load(row4 + x);
|
||||
t0 = v_add(v_add(v_add(r0, r4), v_add(r2, r2)), v_shl<2>(v_add(v_add(r1, r3), r2)));
|
||||
v_rshr_pack_store<8>(dst + x, t0);
|
||||
x += VTraits<v_uint16>::vlanes();
|
||||
}
|
||||
vx_cleanup();
|
||||
|
||||
return x;
|
||||
}
|
||||
|
||||
template<> int PyrDownVecH<uchar, int, 1>(const uchar* src, int* row, int width)
|
||||
{
|
||||
@@ -1291,10 +1368,23 @@ void cv::pyrDown( InputArray _src, OutputArray _dst, const Size& _dsz, int borde
|
||||
{
|
||||
CALL_HAL(pyrDown, cv_hal_pyrdown, src.data, src.step, src.cols, src.rows, dst.data, dst.step, dst.cols, dst.rows, depth, src.channels(), borderType);
|
||||
}
|
||||
bool use_avx512vbmi = checkHardwareSupport(CPU_AVX_512VBMI);
|
||||
|
||||
PyrFunc func = 0;
|
||||
if( depth == CV_8U )
|
||||
func = pyrDown_< FixPtCast<uchar, 8> >;
|
||||
{
|
||||
if(use_avx512vbmi)
|
||||
{
|
||||
// intermediate storage in 16bit only improves when used along with AVX512_VBMI ISA.
|
||||
// AVX2: Usage with AVX2 has negligible measurable uplift in performance.
|
||||
// ARM: Using 16bit intermediate storage has shown 5 to 10% degradation on ARM platform.
|
||||
func = pyrDown_< FixPtUcharCast<uchar, 8> >;
|
||||
}
|
||||
else
|
||||
{
|
||||
func = pyrDown_< FixPtCast<uchar, 8> >;
|
||||
}
|
||||
}
|
||||
else if( depth == CV_16S )
|
||||
func = pyrDown_< FixPtCast<short, 8> >;
|
||||
else if( depth == CV_16U )
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#include "precomp.hpp"
|
||||
#include "opencv2/core/hal/intrin.hpp"
|
||||
|
||||
#include "pyramids_avx512_vbmi.simd.hpp"
|
||||
#include "pyramids_avx512_vbmi.simd_declarations.hpp" // defines CV_CPU_DISPATCH_MODES_ALL=AVX512_ICL,...,BASELINE
|
||||
|
||||
namespace cv {
|
||||
|
||||
// declared in pyramids.cpp
|
||||
int PyrDownVecH_uchar_ushort_1_dispatch(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_2_dispatch(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_3_dispatch(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_4_dispatch(const uchar* src, ushort* row, int width);
|
||||
|
||||
int PyrDownVecH_uchar_ushort_1_dispatch(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
CV_CPU_DISPATCH(PyrDownVecH_uchar_ushort_1_vbmi, (src, row, width),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
int PyrDownVecH_uchar_ushort_2_dispatch(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
CV_CPU_DISPATCH(PyrDownVecH_uchar_ushort_2_vbmi, (src, row, width),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
int PyrDownVecH_uchar_ushort_3_dispatch(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
CV_CPU_DISPATCH(PyrDownVecH_uchar_ushort_3_vbmi, (src, row, width),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
int PyrDownVecH_uchar_ushort_4_dispatch(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
CV_CPU_DISPATCH(PyrDownVecH_uchar_ushort_4_vbmi, (src, row, width),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
} // namespace cv
|
||||
@@ -0,0 +1,259 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#include "opencv2/core/hal/intrin.hpp"
|
||||
|
||||
namespace cv {
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
// AVX-512 VBMI vpermb-based PyrDownVecH for uchar->ushort.
|
||||
int PyrDownVecH_uchar_ushort_1_vbmi(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_2_vbmi(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_3_vbmi(const uchar* src, ushort* row, int width);
|
||||
int PyrDownVecH_uchar_ushort_4_vbmi(const uchar* src, ushort* row, int width);
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
int PyrDownVecH_uchar_ushort_1_vbmi(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
int x = 0;
|
||||
|
||||
#if !CV_AVX_512VBMI
|
||||
CV_UNUSED(src); CV_UNUSED(row); CV_UNUSED(width);
|
||||
#else
|
||||
// cn=1: 32 output pixels per iteration.
|
||||
// Source stride = 2 bytes/pixel. Tap k for pixel i: src[2*i + k], k=0..4.
|
||||
// Max source byte: 2*31 + 4 = 66. Fits in 2 zmm (128 bytes).
|
||||
__m512i vidx0 = _v512_set_epu8(
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
62,60,58,56,54,52,50,48,46,44,42,40,38,36,34,32,
|
||||
30,28,26,24,22,20,18,16,14,12,10, 8, 6, 4, 2, 0);
|
||||
__m512i vidx1 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(1));
|
||||
__m512i vidx2 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(2));
|
||||
__m512i vidx3 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(3));
|
||||
__m512i vidx4 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(4));
|
||||
__m512i v6 = _mm512_set1_epi16(6);
|
||||
for (; x <= width - 32; x += 32, src += 64, row += 32)
|
||||
{
|
||||
__m512i sA = _mm512_loadu_si512(src);
|
||||
__m512i sB = _mm512_loadu_si512(src + 64);
|
||||
|
||||
__m512i t0 = _mm512_permutex2var_epi8(sA, vidx0, sB);
|
||||
__m512i t1 = _mm512_permutex2var_epi8(sA, vidx1, sB);
|
||||
__m512i t2 = _mm512_permutex2var_epi8(sA, vidx2, sB);
|
||||
__m512i t3 = _mm512_permutex2var_epi8(sA, vidx3, sB);
|
||||
__m512i t4 = _mm512_permutex2var_epi8(sA, vidx4, sB);
|
||||
|
||||
__m512i w0 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t0));
|
||||
__m512i w1 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t1));
|
||||
__m512i w2 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t2));
|
||||
__m512i w3 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t3));
|
||||
__m512i w4 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t4));
|
||||
|
||||
__m512i res = _mm512_add_epi16(
|
||||
_mm512_add_epi16(_mm512_mullo_epi16(w2, v6),
|
||||
_mm512_slli_epi16(_mm512_add_epi16(w1, w3), 2)),
|
||||
_mm512_add_epi16(w0, w4));
|
||||
|
||||
_mm512_storeu_si512(row, res);
|
||||
}
|
||||
_mm256_zeroupper();
|
||||
#endif
|
||||
return x;
|
||||
}
|
||||
|
||||
int PyrDownVecH_uchar_ushort_2_vbmi(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
int x = 0;
|
||||
|
||||
#if !CV_AVX_512VBMI
|
||||
CV_UNUSED(src); CV_UNUSED(row); CV_UNUSED(width);
|
||||
#else
|
||||
// cn=2: 16 output pixels x 2 channels = 32 output ushorts per iteration.
|
||||
// Source stride = 4 bytes/pixel. Tap k for pixel i, ch j: src[4*i + 2*k + j].
|
||||
// Max source byte: 4*15 + 8 + 1 = 69. Fits in 2 zmm.
|
||||
__m512i vidx0 = _v512_set_epu8(
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
61,60,57,56,53,52,49,48,45,44,41,40,37,36,33,32,
|
||||
29,28,25,24,21,20,17,16,13,12, 9, 8, 5, 4, 1, 0);
|
||||
__m512i vidx1 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(2));
|
||||
__m512i vidx2 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(4));
|
||||
__m512i vidx3 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(6));
|
||||
__m512i vidx4 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(8));
|
||||
__m512i v6 = _mm512_set1_epi16(6);
|
||||
|
||||
for (; x <= width - 32; x += 32, src += 64, row += 32)
|
||||
{
|
||||
__m512i sA = _mm512_loadu_si512(src);
|
||||
__m512i sB = _mm512_loadu_si512(src + 64);
|
||||
|
||||
__m512i t0 = _mm512_permutex2var_epi8(sA, vidx0, sB);
|
||||
__m512i t1 = _mm512_permutex2var_epi8(sA, vidx1, sB);
|
||||
__m512i t2 = _mm512_permutex2var_epi8(sA, vidx2, sB);
|
||||
__m512i t3 = _mm512_permutex2var_epi8(sA, vidx3, sB);
|
||||
__m512i t4 = _mm512_permutex2var_epi8(sA, vidx4, sB);
|
||||
|
||||
__m512i w0 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t0));
|
||||
__m512i w1 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t1));
|
||||
__m512i w2 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t2));
|
||||
__m512i w3 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t3));
|
||||
__m512i w4 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t4));
|
||||
|
||||
__m512i res = _mm512_add_epi16(
|
||||
_mm512_add_epi16(_mm512_mullo_epi16(w2, v6),
|
||||
_mm512_slli_epi16(_mm512_add_epi16(w1, w3), 2)),
|
||||
_mm512_add_epi16(w0, w4));
|
||||
|
||||
_mm512_storeu_si512(row, res);
|
||||
}
|
||||
_mm256_zeroupper();
|
||||
#endif
|
||||
return x;
|
||||
}
|
||||
|
||||
int PyrDownVecH_uchar_ushort_3_vbmi(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
int x = 0;
|
||||
|
||||
#if !CV_AVX_512VBMI
|
||||
CV_UNUSED(src); CV_UNUSED(row); CV_UNUSED(width);
|
||||
#else
|
||||
// Each iteration: 16 output pixels × 3 channels = 48 output ushorts.
|
||||
// Source span: 16×6 + 4*3 = 108 bytes. Load 128 contiguous bytes in 2 zmm.
|
||||
__m512i vidx0 = _v512_set_epu8(
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
92,91,90, 86,85,84, 80,79,78, 74,73,72, 68,67,66, 62,
|
||||
61,60, 56,55,54, 50,49,48, 44,43,42, 38,37,36, 32,31,
|
||||
30, 26,25,24, 20,19,18, 14,13,12, 8, 7, 6, 2, 1, 0);
|
||||
__m512i vidx1 = _v512_set_epu8(
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
95,94,93, 89,88,87, 83,82,81, 77,76,75, 71,70,69, 65,
|
||||
64,63, 59,58,57, 53,52,51, 47,46,45, 41,40,39, 35,34,
|
||||
33, 29,28,27, 23,22,21, 17,16,15, 11,10, 9, 5, 4, 3);
|
||||
__m512i vidx2 = _v512_set_epu8(
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
98,97,96, 92,91,90, 86,85,84, 80,79,78, 74,73,72, 68,
|
||||
67,66, 62,61,60, 56,55,54, 50,49,48, 44,43,42, 38,37,
|
||||
36, 32,31,30, 26,25,24, 20,19,18, 14,13,12, 8, 7, 6);
|
||||
__m512i vidx3 = _v512_set_epu8(
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
101,100,99, 95,94,93, 89,88,87, 83,82,81, 77,76,75, 71,
|
||||
70,69, 65,64,63, 59,58,57, 53,52,51, 47,46,45, 41,40,
|
||||
39, 35,34,33, 29,28,27, 23,22,21, 17,16,15, 11,10, 9);
|
||||
__m512i vidx4 = _v512_set_epu8(
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
104,103,102, 98,97,96, 92,91,90, 86,85,84, 80,79,78, 74,
|
||||
73,72, 68,67,66, 62,61,60, 56,55,54, 50,49,48, 44,43,
|
||||
42, 38,37,36, 32,31,30, 26,25,24, 20,19,18, 14,13,12);
|
||||
|
||||
__m512i v6 = _mm512_set1_epi16(6);
|
||||
|
||||
//16 output pixels × 3 channels
|
||||
for (; x <= width - 48; x += 48, src += 96, row += 48)
|
||||
{
|
||||
__m512i sA = _mm512_loadu_si512(src);
|
||||
__m512i sB = _mm512_loadu_si512(src + 64);
|
||||
|
||||
__m512i t0 = _mm512_permutex2var_epi8(sA, vidx0, sB);
|
||||
__m512i t1 = _mm512_permutex2var_epi8(sA, vidx1, sB);
|
||||
__m512i t2 = _mm512_permutex2var_epi8(sA, vidx2, sB);
|
||||
__m512i t3 = _mm512_permutex2var_epi8(sA, vidx3, sB);
|
||||
__m512i t4 = _mm512_permutex2var_epi8(sA, vidx4, sB);
|
||||
|
||||
// Expand lower 32 bytes of each tap to u16 (first 32 of 48 bytes)
|
||||
__m512i t0_lo = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t0));
|
||||
__m512i t1_lo = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t1));
|
||||
__m512i t2_lo = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t2));
|
||||
__m512i t3_lo = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t3));
|
||||
__m512i t4_lo = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t4));
|
||||
|
||||
// Expand upper 16 bytes (bytes 32..47) to u16
|
||||
__m256i t0_hi = _mm256_cvtepu8_epi16(_mm512_extracti32x4_epi32(t0, 2));
|
||||
__m256i t1_hi = _mm256_cvtepu8_epi16(_mm512_extracti32x4_epi32(t1, 2));
|
||||
__m256i t2_hi = _mm256_cvtepu8_epi16(_mm512_extracti32x4_epi32(t2, 2));
|
||||
__m256i t3_hi = _mm256_cvtepu8_epi16(_mm512_extracti32x4_epi32(t3, 2));
|
||||
__m256i t4_hi = _mm256_cvtepu8_epi16(_mm512_extracti32x4_epi32(t4, 2));
|
||||
|
||||
// Compute low half (32xu16 values):
|
||||
__m512i sum13_lo = _mm512_add_epi16(t1_lo, t3_lo);
|
||||
__m512i res_lo = _mm512_add_epi16(
|
||||
_mm512_add_epi16(_mm512_mullo_epi16(t2_lo, v6),
|
||||
_mm512_slli_epi16(sum13_lo, 2)),
|
||||
_mm512_add_epi16(t0_lo, t4_lo));
|
||||
|
||||
// Compute high half (16xu16 values):
|
||||
__m256i v6_256 = _mm256_set1_epi16(6);
|
||||
__m256i sum13_hi = _mm256_add_epi16(t1_hi, t3_hi);
|
||||
__m256i res_hi = _mm256_add_epi16(
|
||||
_mm256_add_epi16(_mm256_mullo_epi16(t2_hi, v6_256),
|
||||
_mm256_slli_epi16(sum13_hi, 2)),
|
||||
_mm256_add_epi16(t0_hi, t4_hi));
|
||||
|
||||
_mm512_storeu_si512(row, res_lo); //32 ushort
|
||||
_mm256_storeu_si256((__m256i*)(row + 32), res_hi); //16 ushort
|
||||
}
|
||||
_mm256_zeroupper();
|
||||
#endif // CV_AVX_512VBMI
|
||||
|
||||
return x;
|
||||
}
|
||||
|
||||
int PyrDownVecH_uchar_ushort_4_vbmi(const uchar* src, ushort* row, int width)
|
||||
{
|
||||
int x = 0;
|
||||
|
||||
#if !CV_AVX_512VBMI
|
||||
CV_UNUSED(src); CV_UNUSED(row); CV_UNUSED(width);
|
||||
#else
|
||||
// cn=4: 8 output pixels x 4 channels = 32 output ushorts per iteration.
|
||||
// Source stride = 8 bytes/pixel. Tap k for pixel i, ch j: src[8*i + 4*k + j].
|
||||
// Max source byte: 8*7 + 4*4 + 3 = 75. Fits in 2 zmm.
|
||||
__m512i vidx0 = _v512_set_epu8(
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
59,58,57,56,51,50,49,48,43,42,41,40,35,34,33,32,
|
||||
27,26,25,24,19,18,17,16,11,10, 9, 8, 3, 2, 1, 0);
|
||||
__m512i vidx1 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(4));
|
||||
__m512i vidx2 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(8));
|
||||
__m512i vidx3 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(12));
|
||||
__m512i vidx4 = _mm512_add_epi8(vidx0, _mm512_set1_epi8(16));
|
||||
|
||||
__m512i v6 = _mm512_set1_epi16(6);
|
||||
|
||||
for (; x <= width - 32; x += 32, src += 64, row += 32)
|
||||
{
|
||||
__m512i sA = _mm512_loadu_si512(src);
|
||||
__m512i sB = _mm512_loadu_si512(src + 64);
|
||||
|
||||
__m512i t0 = _mm512_permutex2var_epi8(sA, vidx0, sB);
|
||||
__m512i t1 = _mm512_permutex2var_epi8(sA, vidx1, sB);
|
||||
__m512i t2 = _mm512_permutex2var_epi8(sA, vidx2, sB);
|
||||
__m512i t3 = _mm512_permutex2var_epi8(sA, vidx3, sB);
|
||||
__m512i t4 = _mm512_permutex2var_epi8(sA, vidx4, sB);
|
||||
|
||||
__m512i w0 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t0));
|
||||
__m512i w1 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t1));
|
||||
__m512i w2 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t2));
|
||||
__m512i w3 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t3));
|
||||
__m512i w4 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(t4));
|
||||
|
||||
__m512i res = _mm512_add_epi16(
|
||||
_mm512_add_epi16(_mm512_mullo_epi16(w2, v6),
|
||||
_mm512_slli_epi16(_mm512_add_epi16(w1, w3), 2)),
|
||||
_mm512_add_epi16(w0, w4));
|
||||
|
||||
_mm512_storeu_si512(row, res);
|
||||
}
|
||||
_mm256_zeroupper();
|
||||
#endif
|
||||
return x;
|
||||
}
|
||||
|
||||
#endif // CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_END
|
||||
} // namespace cv
|
||||
@@ -109,6 +109,8 @@ bool Dictionary::readDictionary(const cv::FileNode& fn) {
|
||||
int nMarkers = 0, _markerSize = 0;
|
||||
if (fn.empty() || !readParameter("nmarkers", nMarkers, fn) || !readParameter("markersize", _markerSize, fn))
|
||||
return false;
|
||||
if (_markerSize <= 0)
|
||||
return false;
|
||||
Mat bytes(0, 0, CV_8UC1), marker(_markerSize, _markerSize, CV_8UC1);
|
||||
std::string markerString;
|
||||
for (int i = 0; i < nMarkers; i++) {
|
||||
@@ -116,6 +118,8 @@ bool Dictionary::readDictionary(const cv::FileNode& fn) {
|
||||
ostr << i;
|
||||
if (!readParameter("marker_" + ostr.str(), markerString, fn))
|
||||
return false;
|
||||
if (markerString.size() != (size_t)_markerSize * _markerSize)
|
||||
return false;
|
||||
for (int j = 0; j < (int) markerString.size(); j++)
|
||||
marker.at<unsigned char>(j) = (markerString[j] == '0') ? 0 : 1;
|
||||
bytes.push_back(Dictionary::getByteListFromBits(marker));
|
||||
|
||||
@@ -1409,6 +1409,46 @@ TEST(CV_ArucoMultiDict, serialization)
|
||||
}
|
||||
|
||||
|
||||
TEST(CV_ArucoDictionary, readDictionary_oversized_marker)
|
||||
{
|
||||
// A marker string longer than markersize*markersize must be rejected instead of
|
||||
// being written past the markersize x markersize buffer in readDictionary.
|
||||
std::string serialized;
|
||||
{
|
||||
FileStorage fs_out(".json", FileStorage::WRITE + FileStorage::MEMORY);
|
||||
ASSERT_TRUE(fs_out.isOpened());
|
||||
fs_out << "nmarkers" << 1;
|
||||
fs_out << "markersize" << 5;
|
||||
fs_out << "marker_0" << std::string(2000, '1');
|
||||
serialized = fs_out.releaseAndGetString();
|
||||
}
|
||||
FileStorage fs_in(serialized, FileStorage::READ + FileStorage::MEMORY);
|
||||
ASSERT_TRUE(fs_in.isOpened());
|
||||
aruco::Dictionary dict;
|
||||
bool ok = true;
|
||||
ASSERT_NO_THROW(ok = dict.readDictionary(fs_in.root()));
|
||||
EXPECT_FALSE(ok);
|
||||
}
|
||||
|
||||
|
||||
TEST(CV_ArucoDictionary, readDictionary_roundtrip)
|
||||
{
|
||||
aruco::Dictionary dict = aruco::getPredefinedDictionary(aruco::DICT_5X5_50);
|
||||
std::string serialized;
|
||||
{
|
||||
FileStorage fs_out(".json", FileStorage::WRITE + FileStorage::MEMORY);
|
||||
ASSERT_TRUE(fs_out.isOpened());
|
||||
dict.writeDictionary(fs_out);
|
||||
serialized = fs_out.releaseAndGetString();
|
||||
}
|
||||
FileStorage fs_in(serialized, FileStorage::READ + FileStorage::MEMORY);
|
||||
ASSERT_TRUE(fs_in.isOpened());
|
||||
aruco::Dictionary loaded;
|
||||
ASSERT_TRUE(loaded.readDictionary(fs_in.root()));
|
||||
EXPECT_EQ(dict, loaded);
|
||||
}
|
||||
|
||||
|
||||
struct ArucoThreading: public testing::TestWithParam<aruco::CornerRefineMethod>
|
||||
{
|
||||
struct NumThreadsSetter {
|
||||
|
||||
@@ -845,8 +845,6 @@ TEST(videoio_ffmpeg, create_with_property_badarg)
|
||||
EXPECT_FALSE(cap.isOpened());
|
||||
}
|
||||
|
||||
// requires FFmpeg wrapper rebuild on Windows
|
||||
#ifndef _WIN32
|
||||
TEST(videoio_ffmpeg, open_with_format_cv8uc3)
|
||||
{
|
||||
if (!videoio_registry::hasBackend(CAP_FFMPEG))
|
||||
@@ -862,7 +860,6 @@ TEST(videoio_ffmpeg, open_with_format_cv8uc3)
|
||||
ASSERT_TRUE(cap.read(frame));
|
||||
EXPECT_EQ(frame.channels(), 3);
|
||||
}
|
||||
#endif
|
||||
|
||||
// related issue: https://github.com/opencv/opencv/issues/16821
|
||||
TEST(videoio_ffmpeg, DISABLED_open_from_web)
|
||||
@@ -1049,12 +1046,9 @@ inline static std::string videoio_ffmpeg_mismatch_name_printer(const testing::Te
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/**/, videoio_ffmpeg_channel_mismatch, testing::ValuesIn(mismatch_cases), videoio_ffmpeg_mismatch_name_printer);
|
||||
|
||||
#ifndef _WIN32
|
||||
|
||||
typedef tuple<string, string> AlphaChannelParams;
|
||||
typedef testing::TestWithParam< AlphaChannelParams > videoio_ffmpeg_alpha_channel;
|
||||
|
||||
// New feature in https://github.com/opencv/opencv/pull/28751 requires FFmpeg wrapper rebuild on Windows
|
||||
TEST_P(videoio_ffmpeg_alpha_channel, write_read)
|
||||
{
|
||||
if (!videoio_registry::hasBackend(CAP_FFMPEG))
|
||||
@@ -1116,7 +1110,6 @@ AlphaChannelParams alpha_params[] =
|
||||
};
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/**/, videoio_ffmpeg_alpha_channel, testing::ValuesIn(alpha_params));
|
||||
#endif
|
||||
|
||||
// related issue: https://github.com/opencv/opencv/issues/23088
|
||||
TEST(ffmpeg_cap_properties, set_pos_get_msec)
|
||||
|
||||
@@ -30,6 +30,12 @@ int main( int argc, char** argv )
|
||||
cout << "Usage: " << argv[0] << " <Input image>" << endl;
|
||||
return -1;
|
||||
}
|
||||
if (image.channels() != 3)
|
||||
{
|
||||
cout << "The tutorial expects 3 channel image as input!\n" << endl;
|
||||
cout << "Usage: " << argv[0] << " <Input image>" << endl;
|
||||
return -1;
|
||||
}
|
||||
//! [basic-linear-transform-load]
|
||||
|
||||
//! [basic-linear-transform-output]
|
||||
@@ -54,7 +60,7 @@ int main( int argc, char** argv )
|
||||
//! [basic-linear-transform-operation]
|
||||
for( int y = 0; y < image.rows; y++ ) {
|
||||
for( int x = 0; x < image.cols; x++ ) {
|
||||
for( int c = 0; c < image.channels(); c++ ) {
|
||||
for( int c = 0; c < 3; c++ ) {
|
||||
new_image.at<Vec3b>(y,x)[c] =
|
||||
saturate_cast<uchar>( alpha*image.at<Vec3b>(y,x)[c] + beta );
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user