mirror of
https://github.com/opencv/opencv.git
synced 2026-07-21 19:33:03 +04:00
Compare commits
172 Commits
next
...
6f402272e7
| Author | SHA1 | Date | |
|---|---|---|---|
| 6f402272e7 | |||
| 0654a42e19 | |||
| f7dd7df170 | |||
| 7aa163b83f | |||
| c47541acbd | |||
| b5fe8dfef3 | |||
| 4d1e206f5e | |||
| 2e778c52c1 | |||
| 6b640b424c | |||
| 06574736b3 | |||
| 4fe51e51e0 | |||
| 335abd236f | |||
| 46d1b6c99d | |||
| e6b1ef272e | |||
| 1bea199e00 | |||
| 42cd88d737 | |||
| 4c3895e96c | |||
| 2c14cc1897 | |||
| 13fb140932 | |||
| 7f9f6acb83 | |||
| 4533beab85 | |||
| 87af90b847 | |||
| d7e6e58652 | |||
| b21eae2b8b | |||
| b41dd0fdff | |||
| be4ef205b1 | |||
| 062a00bab3 | |||
| b022a9b321 | |||
| 7230c1ce08 | |||
| 818145f41d | |||
| 93cd614472 | |||
| 7507accf2e | |||
| 8eb2c8998c | |||
| 738739d7c2 | |||
| e978c193cf | |||
| b887e2fd3e | |||
| 0eb3778a1b | |||
| 9121ebe28b | |||
| d48bf69f65 | |||
| aa70c82ebe | |||
| c85397c3bb | |||
| 0488f6e942 | |||
| 51495995d2 | |||
| c344b7dcef | |||
| 0e6433327a | |||
| f6be96989e | |||
| e9289fc7c4 | |||
| 50beb24c6b | |||
| fcc3418cd7 | |||
| 80fda1da8c | |||
| cddfd62a18 | |||
| a9573e5ac0 | |||
| 9d66a589b4 | |||
| 24030ff3a2 | |||
| c77286749a | |||
| e6d0c0340b | |||
| 9460e30793 | |||
| b4164a4bc4 | |||
| 579a505734 | |||
| 83ee55fcda | |||
| 28a1d0dddb | |||
| 2f779c5c36 | |||
| 5b83b0c440 | |||
| a16c4a9fe2 | |||
| 2a47e2c1a5 | |||
| 37f3c7539b | |||
| 3c10324630 | |||
| c831290769 | |||
| a015333f04 | |||
| 5dbfbcdcac | |||
| edd89cfca5 | |||
| 15b5d28325 | |||
| dd541965e9 | |||
| da4a417b2a | |||
| 416e730a61 | |||
| 69d2303531 | |||
| b64a215e71 | |||
| 61bd5e7b55 | |||
| fe849ec08d | |||
| b864ee7335 | |||
| d4005e3cb1 | |||
| fbd4dc6ca0 | |||
| c0c9e5e76c | |||
| 8dcaac1ac4 | |||
| d6a187480f | |||
| 74a25df87e | |||
| 1aea68260a | |||
| 13eba3d842 | |||
| 84d900cfda | |||
| 2e80d30df4 | |||
| 5137676ea7 | |||
| 15367f7c52 | |||
| 577f395bee | |||
| 639fd5d07a | |||
| bc184b560e | |||
| af1e2232cd | |||
| d8f609c09f | |||
| 6678e48c12 | |||
| 2adeab8504 | |||
| 4eb3f8ecb6 | |||
| dceead809e | |||
| df0c9d7950 | |||
| c3aecdd531 | |||
| 10f4847bf6 | |||
| 5814ff8f9c | |||
| fc746f35b9 | |||
| 4f17d30997 | |||
| 60122e655c | |||
| 550251b3c2 | |||
| de90d1aec4 | |||
| 70a2c50b43 | |||
| d0925fa360 | |||
| bc18ec2ed3 | |||
| 853ee9b961 | |||
| c796023029 | |||
| 837f715eec | |||
| 4e56cf9ac8 | |||
| 414a0379d8 | |||
| 6eb0dc97f5 | |||
| 3def56d25a | |||
| 57081ac946 | |||
| 62bcd12f9c | |||
| 6a2d8d24da | |||
| d878b04c6b | |||
| 4208a89b3e | |||
| 41d48d6eeb | |||
| 0a4d6a849a | |||
| 017a5138d1 | |||
| afbfa275d8 | |||
| 6df9732c81 | |||
| 40d6727ea2 | |||
| 8e157165e8 | |||
| cfb3a8bbb7 | |||
| 194db6f5cb | |||
| 5aad95cb1d | |||
| 86433eac87 | |||
| 2906d4d73a | |||
| 96a71a8d83 | |||
| 19ec7ea28b | |||
| 12eaf9bd4c | |||
| cadcb7e4e0 | |||
| fa55ed2837 | |||
| 4dee742731 | |||
| a77474f3b1 | |||
| 590d8f6439 | |||
| aed41fdabe | |||
| f80d02e1c7 | |||
| ffa38e1b74 | |||
| 1780a86075 | |||
| 48b8099214 | |||
| 3d02d863f9 | |||
| 3d55d2fcef | |||
| d10138fa1c | |||
| 9780b3b329 | |||
| 63b59865d3 | |||
| b9a38df0e6 | |||
| fe61ae0e41 | |||
| 6f29af625b | |||
| 0397596e17 | |||
| 2a70226181 | |||
| 21356eed2b | |||
| 11b0bd15cf | |||
| c936dcaef7 | |||
| ab1aaa75aa | |||
| d9f89b028b | |||
| 517710fe56 | |||
| a99141acd7 | |||
| 1b047868dd | |||
| a7bb1d1e1e | |||
| aee03ffbb2 | |||
| d79d9d7a5b | |||
| b723ddd72b |
@@ -15,6 +15,9 @@ jobs:
|
||||
Linux-no-HAL:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-Linux-NoHAL.yaml@main
|
||||
|
||||
Linux-Apline:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-Linux-Alpine.yaml@main
|
||||
|
||||
Windows:
|
||||
uses: opencv/ci-gha-workflow/.github/workflows/OCV-PR-Windows.yaml@main
|
||||
with:
|
||||
|
||||
Vendored
+6
-6
@@ -1,9 +1,9 @@
|
||||
# Binaries branch name: ffmpeg/4.x_20251226
|
||||
# Binaries were created for OpenCV: cff7581175d2abfc6aef2e4f04f482e258b5c864
|
||||
ocv_update(FFMPEG_BINARIES_COMMIT "d82ad9a54a7b42a1648a9cae8fed5c2f20ea396c")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "47730de2286110b0d1250ff9cf50ce56")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "3248b4663ffef770cdb54ec8b9d16a28")
|
||||
ocv_update(FFMPEG_FILE_HASH_CMAKE "8862c87496e2e8c375965e1277dee1c7")
|
||||
# Binaries branch name: ffmpeg/4.x_20260715
|
||||
# Binaries were created for OpenCV: 6b640b424c516d27217700392a7ed6a362e040a5
|
||||
ocv_update(FFMPEG_BINARIES_COMMIT "bd9418020a5c342be979c56d6e6434261959d3af")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "31968b434799d3dd56969b15aac3efa8")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "84757ed0f16ddedce99227529d3165e9")
|
||||
ocv_update(FFMPEG_FILE_HASH_CMAKE "e09efc33312d1173be8a9446f3b088fe")
|
||||
|
||||
function(download_win_ffmpeg script_var)
|
||||
set(${script_var} "" PARENT_SCOPE)
|
||||
|
||||
Vendored
+5
-5
@@ -2,7 +2,7 @@ function(download_ippicv root_var)
|
||||
set(${root_var} "" PARENT_SCOPE)
|
||||
|
||||
# Commit SHA in the opencv_3rdparty repo
|
||||
set(IPPICV_COMMIT "406d398c436d0465c8e53dd432d9ecd9301d5f4a")
|
||||
set(IPPICV_COMMIT "8338862a733cb3980d8b51d8e14917fe0e695f71")
|
||||
# Define actual ICV versions
|
||||
if(APPLE)
|
||||
set(IPPICV_COMMIT "0cc4aa06bf2bef4b05d237c69a5a96b9cd0cb85a")
|
||||
@@ -14,8 +14,8 @@ function(download_ippicv root_var)
|
||||
set(OPENCV_ICV_PLATFORM "linux")
|
||||
set(OPENCV_ICV_PACKAGE_SUBDIR "ippicv_lnx")
|
||||
if(X86_64)
|
||||
set(OPENCV_ICV_NAME "ippicv_2026.0.0_lnx_intel64_20260327_general.tgz")
|
||||
set(OPENCV_ICV_HASH "9a3ee0c5c3c02102faa422d60bfd1f4a")
|
||||
set(OPENCV_ICV_NAME "ippicv_2026.0.0_lnx_intel64_20260630_general.tgz")
|
||||
set(OPENCV_ICV_HASH "a77e60db544e07a126ae98f5ffe83be1")
|
||||
else()
|
||||
if(ANDROID)
|
||||
set(IPPICV_COMMIT "c7c6d527dde5fee7cb914ee9e4e20f7436aab3a1")
|
||||
@@ -31,8 +31,8 @@ function(download_ippicv root_var)
|
||||
set(OPENCV_ICV_PLATFORM "windows")
|
||||
set(OPENCV_ICV_PACKAGE_SUBDIR "ippicv_win")
|
||||
if(X86_64)
|
||||
set(OPENCV_ICV_NAME "ippicv_2026.0.0_win_intel64_20260327_general.zip")
|
||||
set(OPENCV_ICV_HASH "73bc67cd5e4c8da706fa88fe84630231")
|
||||
set(OPENCV_ICV_NAME "ippicv_2026.0.0_win_intel64_20260630_general.zip")
|
||||
set(OPENCV_ICV_HASH "d81c8b7d40da2867df82f0077a40afa1")
|
||||
else()
|
||||
set(IPPICV_COMMIT "7f55c0c26be418d494615afca15218566775c725")
|
||||
set(OPENCV_ICV_NAME "ippicv_2021.12.0_win_ia32_20240425_general.zip")
|
||||
|
||||
-3
@@ -136,9 +136,6 @@ endif()
|
||||
|
||||
set(JPEG_LIB_VERSION 70)
|
||||
|
||||
# OpenCV
|
||||
set(JPEG_LIB_VERSION "${VERSION}-${JPEG_LIB_VERSION}" PARENT_SCOPE)
|
||||
|
||||
set(THREAD_LOCAL "") # WITH_TURBOJPEG is not used
|
||||
|
||||
add_definitions(-DNO_GETENV -DNO_PUTENV)
|
||||
|
||||
@@ -102,8 +102,23 @@ if(WITH_JPEG)
|
||||
macro(ocv_detect_jpeg_version header_file)
|
||||
if(NOT DEFINED JPEG_LIB_VERSION AND EXISTS "${header_file}")
|
||||
ocv_parse_header("${header_file}" JPEG_VERSION_LINES JPEG_LIB_VERSION)
|
||||
|
||||
if(DEFINED JPEG_LIB_VERSION)
|
||||
# Extract libjpeg-turbo version from the header file if JPEG_LIB_VERSION is found.
|
||||
file(STRINGS "${header_file}" JPEG_TURBO_VERSION_LINE REGEX "^#define[\t ]+LIBJPEG_TURBO_VERSION[\t ]")
|
||||
|
||||
if(JPEG_TURBO_VERSION_LINE)
|
||||
# Support both raw values (e.g., 3.1.2) and quoted strings (e.g., "3.1.2").
|
||||
string(REGEX REPLACE "^#define[\t ]+LIBJPEG_TURBO_VERSION[\t ]+\"?([^\"]+)\"?.*" "\\1" JPEG_TURBO_VERSION_STRING "${JPEG_TURBO_VERSION_LINE}")
|
||||
if(JPEG_TURBO_VERSION_STRING)
|
||||
string(STRIP "${JPEG_TURBO_VERSION_STRING}" JPEG_TURBO_VERSION_STRING)
|
||||
set(JPEG_LIB_VERSION "${JPEG_TURBO_VERSION_STRING}-${JPEG_LIB_VERSION}")
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
ocv_detect_jpeg_version("${JPEG_INCLUDE_DIR}/jpeglib.h")
|
||||
if(DEFINED CMAKE_CXX_LIBRARY_ARCHITECTURE)
|
||||
ocv_detect_jpeg_version("${JPEG_INCLUDE_DIR}/${CMAKE_CXX_LIBRARY_ARCHITECTURE}/jconfig.h")
|
||||
|
||||
@@ -96,7 +96,11 @@ else() # UNIX
|
||||
|
||||
endif()
|
||||
|
||||
ocv_update(CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${OPENCV_LIB_INSTALL_PATH}")
|
||||
if(APPLE_FRAMEWORK AND BUILD_SHARED_LIBS)
|
||||
ocv_update(CMAKE_INSTALL_RPATH "@executable_path/Frameworks;@loader_path/Frameworks;@executable_path/../${OPENCV_3P_LIB_INSTALL_PATH}")
|
||||
else()
|
||||
ocv_update(CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${OPENCV_LIB_INSTALL_PATH}")
|
||||
endif()
|
||||
set(CMAKE_INSTALL_RPATH_USE_LINK_PATH TRUE)
|
||||
|
||||
if(INSTALL_TO_MANGLED_PATHS)
|
||||
|
||||
@@ -1 +1,8 @@
|
||||
# empty
|
||||
# Additional flags for dynamic framework (macOS)
|
||||
if(APPLE_FRAMEWORK AND BUILD_SHARED_LIBS)
|
||||
string(APPEND CMAKE_MODULE_LINKER_FLAGS " -rpath @executable_path/Frameworks -rpath @loader_path/Frameworks")
|
||||
string(APPEND CMAKE_SHARED_LINKER_FLAGS " -rpath @executable_path/Frameworks -rpath @loader_path/Frameworks")
|
||||
string(APPEND CMAKE_EXE_LINKER_FLAGS " -rpath @executable_path/Frameworks -rpath @loader_path/Frameworks")
|
||||
set(CMAKE_SHARED_LIBRARY_RUNTIME_C_FLAG 1)
|
||||
set(CMAKE_INSTALL_NAME_DIR "@rpath")
|
||||
endif()
|
||||
|
||||
@@ -10,6 +10,14 @@ set(OpenCV_CUDNN_VERSION "@CUDNN_VERSION@")
|
||||
set(OpenCV_USE_CUDNN "@HAVE_CUDNN@")
|
||||
set(ENABLE_CUDA_FIRST_CLASS_LANGUAGE ON)
|
||||
|
||||
# By default OpenCV only requires the consumer's CUDA Toolkit to match the
|
||||
# major.minor version it was built with. The CUDA runtime API is stable within
|
||||
# a minor release and patch versions change frequently, so pinning the patch
|
||||
# component forces needless rebuilds (issue #26965). Set
|
||||
# OPENCV_STRONG_CUDA_VERSION_CHECK before find_package(OpenCV) to require an
|
||||
# exact match including the patch component.
|
||||
string(REGEX MATCH "^[0-9]+\\.[0-9]+" OpenCV_CUDA_VERSION_MAJOR_MINOR "${OpenCV_CUDA_VERSION}")
|
||||
|
||||
if(NOT CUDAToolkit_FOUND)
|
||||
if(NOT CMAKE_VERSION VERSION_LESS 3.18)
|
||||
if(UNIX AND NOT CMAKE_CUDA_COMPILER AND NOT CUDAToolkit_ROOT)
|
||||
@@ -17,15 +25,28 @@ if(NOT CUDAToolkit_FOUND)
|
||||
set(CUDA_PATH "/usr/local/cuda" CACHE INTERNAL "")
|
||||
set(ENV{CUDA_PATH} ${CUDA_PATH})
|
||||
endif()
|
||||
find_package(CUDAToolkit ${OpenCV_CUDA_VERSION} EXACT REQUIRED)
|
||||
if(OPENCV_STRONG_CUDA_VERSION_CHECK)
|
||||
find_package(CUDAToolkit ${OpenCV_CUDA_VERSION} EXACT REQUIRED)
|
||||
else()
|
||||
find_package(CUDAToolkit ${OpenCV_CUDA_VERSION_MAJOR_MINOR} REQUIRED)
|
||||
endif()
|
||||
else()
|
||||
message(FATAL_ERROR "Using OpenCV compiled with CUDA as first class language requires CMake \>= 3.18.")
|
||||
endif()
|
||||
else()
|
||||
if(CUDAToolkit_FOUND)
|
||||
set(CUDA_VERSION_STRING ${CUDAToolkit_VERSION})
|
||||
endif()
|
||||
if(NOT CUDA_VERSION_STRING VERSION_EQUAL OpenCV_CUDA_VERSION)
|
||||
message(FATAL_ERROR "OpenCV library was compiled with CUDA ${OpenCV_CUDA_VERSION} support. Please, use the same version or rebuild OpenCV with CUDA ${CUDA_VERSION_STRING}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(CUDAToolkit_FOUND)
|
||||
set(CUDA_VERSION_STRING ${CUDAToolkit_VERSION})
|
||||
endif()
|
||||
|
||||
string(REGEX MATCH "^[0-9]+\\.[0-9]+" CUDA_VERSION_STRING_MAJOR_MINOR "${CUDA_VERSION_STRING}")
|
||||
if(OPENCV_STRONG_CUDA_VERSION_CHECK)
|
||||
set(__ocv_cuda_found "${CUDA_VERSION_STRING}")
|
||||
set(__ocv_cuda_built "${OpenCV_CUDA_VERSION}")
|
||||
else()
|
||||
set(__ocv_cuda_found "${CUDA_VERSION_STRING_MAJOR_MINOR}")
|
||||
set(__ocv_cuda_built "${OpenCV_CUDA_VERSION_MAJOR_MINOR}")
|
||||
endif()
|
||||
if(NOT __ocv_cuda_found VERSION_EQUAL __ocv_cuda_built)
|
||||
message(FATAL_ERROR "OpenCV library was compiled with CUDA ${OpenCV_CUDA_VERSION} support. Please, use the same version or rebuild OpenCV with CUDA ${CUDA_VERSION_STRING}")
|
||||
endif()
|
||||
|
||||
@@ -105,7 +105,7 @@ let mat = new cv.Mat();
|
||||
let matVec = new cv.MatVector();
|
||||
// Push a Mat back into MatVector
|
||||
matVec.push_back(mat);
|
||||
// Get a Mat fom MatVector
|
||||
// Get a Mat from MatVector
|
||||
let cnt = matVec.get(0);
|
||||
mat.delete(); matVec.delete(); cnt.delete();
|
||||
@endcode
|
||||
|
||||
@@ -27,7 +27,7 @@ foreground object (Always try to keep foreground in white). So what it does? The
|
||||
through the image (as in 2D convolution). A pixel in the original image (either 1 or 0) will be
|
||||
considered 1 only if all the pixels under the kernel is 1, otherwise it is eroded (made to zero).
|
||||
|
||||
So what happends is that, all the pixels near boundary will be discarded depending upon the size of
|
||||
So what happens is that, all the pixels near boundary will be discarded depending upon the size of
|
||||
kernel. So the thickness or size of the foreground object decreases or simply white region decreases
|
||||
in the image. It is useful for removing small white noises (as we have seen in colorspace chapter),
|
||||
detach two connected objects etc.
|
||||
@@ -174,4 +174,4 @@ Try it
|
||||
<iframe src="../../js_morphological_ops_getStructuringElement.html" width="100%"
|
||||
onload="this.style.height=this.contentDocument.body.scrollHeight +'px';">
|
||||
</iframe>
|
||||
\endhtmlonly
|
||||
\endhtmlonly
|
||||
|
||||
+1
-1
@@ -1350,7 +1350,7 @@
|
||||
@article{TRUCO2026,
|
||||
title={TRUCO: A Scalable Lock-Free Algorithm for Parallel Contour Extraction},
|
||||
author={Mu\~noz-Salinas, Rafael and Romero-Ramírez, Francisco J. and Marín-Jiménez, Manuel J.},
|
||||
journal={Pattern Recognition under review},
|
||||
journal={To be published},
|
||||
year={2026}
|
||||
}
|
||||
|
||||
|
||||
@@ -81,7 +81,7 @@ pass in terms of square size).
|
||||
|
||||
So to find pattern in chess board, we can use the function, **cv.findChessboardCorners()**. We also
|
||||
need to pass what kind of pattern we are looking for, like 8x8 grid, 5x5 grid etc. In this example, we
|
||||
use 7x6 grid. (Normally a chess board has 8x8 squares and 7x7 internal corners). It returns the
|
||||
use 6x7 grid. (Normally a chess board has 8x8 squares and 7x7 internal corners). It returns the
|
||||
corner points and retval which will be True if pattern is obtained. These corners will be placed in
|
||||
an order (from left-to-right, top-to-bottom)
|
||||
|
||||
@@ -107,9 +107,11 @@ import glob
|
||||
# termination criteria
|
||||
criteria = (cv.TERM_CRITERIA_EPS + cv.TERM_CRITERIA_MAX_ITER, 30, 0.001)
|
||||
|
||||
# prepare object points, like (0,0,0), (1,0,0), (2,0,0) ....,(6,5,0)
|
||||
objp = np.zeros((6*7,3), np.float32)
|
||||
objp[:,:2] = np.mgrid[0:7,0:6].T.reshape(-1,2)
|
||||
# prepare object points, like (0,0,0), (1,0,0), (2,0,0) ....,(5,6,0)
|
||||
cols = 6
|
||||
rows = 7
|
||||
objp = np.zeros((cols*rows,3), np.float32)
|
||||
objp[:,:2] = np.mgrid[0:cols,0:rows].T.reshape(-1,2)
|
||||
|
||||
# Arrays to store object points and image points from all the images.
|
||||
objpoints = [] # 3d point in real world space
|
||||
@@ -122,7 +124,7 @@ for fname in images:
|
||||
gray = cv.cvtColor(img, cv.COLOR_BGR2GRAY)
|
||||
|
||||
# Find the chess board corners
|
||||
ret, corners = cv.findChessboardCorners(gray, (7,6), None)
|
||||
ret, corners = cv.findChessboardCorners(gray, (cols, rows), None)
|
||||
|
||||
# If found, add object points, image points (after refining them)
|
||||
if ret == True:
|
||||
@@ -132,7 +134,7 @@ for fname in images:
|
||||
imgpoints.append(corners2)
|
||||
|
||||
# Draw and display the corners
|
||||
cv.drawChessboardCorners(img, (7,6), corners2, ret)
|
||||
cv.drawChessboardCorners(img, (cols, rows), corners2, ret)
|
||||
cv.imshow('img', img)
|
||||
cv.waitKey(500)
|
||||
|
||||
|
||||
@@ -50,12 +50,14 @@ our X axis is drawn from (0,0,0) to (3,0,0), so for Y axis. For Z axis, it is dr
|
||||
(0,0,-3). Negative denotes it is drawn towards the camera.
|
||||
@code{.py}
|
||||
criteria = (cv.TERM_CRITERIA_EPS + cv.TERM_CRITERIA_MAX_ITER, 30, 0.001)
|
||||
objp = np.zeros((6*7,3), np.float32)
|
||||
objp[:,:2] = np.mgrid[0:7,0:6].T.reshape(-1,2)
|
||||
cols = 6
|
||||
rows = 7
|
||||
objp = np.zeros((cols*rows,3), np.float32)
|
||||
objp[:,:2] = np.mgrid[0:cols,0:rows].T.reshape(-1,2)
|
||||
|
||||
axis = np.float32([[3,0,0], [0,3,0], [0,0,-3]]).reshape(-1,3)
|
||||
@endcode
|
||||
Now, as usual, we load each image. Search for 7x6 grid. If found, we refine it with subcorner
|
||||
Now, as usual, we load each image. Search for 6x7 grid. If found, we refine it with subcorner
|
||||
pixels. Then to calculate the rotation and translation, we use the function,
|
||||
**cv.solvePnPRansac()**. Once we those transformation matrices, we use them to project our **axis
|
||||
points** to the image plane. In simple words, we find the points on image plane corresponding to
|
||||
@@ -65,7 +67,7 @@ to each of these points using our generateImage() function. Done !!!
|
||||
for fname in glob.glob('left*.jpg'):
|
||||
img = cv.imread(fname)
|
||||
gray = cv.cvtColor(img,cv.COLOR_BGR2GRAY)
|
||||
ret, corners = cv.findChessboardCorners(gray, (7,6),None)
|
||||
ret, corners = cv.findChessboardCorners(gray, (cols, rows), None)
|
||||
|
||||
if ret == True:
|
||||
corners2 = cv.cornerSubPix(gray,corners,(11,11),(-1,-1),criteria)
|
||||
|
||||
@@ -103,4 +103,4 @@ int main(void)
|
||||
|
||||
Limitation/Known problem
|
||||
------------------------
|
||||
- cv::moveWindow() is not implementated. ( See. https://github.com/opencv/opencv/issues/25478 )
|
||||
- cv::moveWindow() is not implemented. ( See. https://github.com/opencv/opencv/issues/25478 )
|
||||
|
||||
@@ -72,8 +72,8 @@ In order to use the Astra camera's depth sensor with OpenCV you should do the fo
|
||||
|
||||
@note The last tried version `2.3.0.86_202210111154_4c8f5aa4_beta6` does not work correctly with
|
||||
modern Linux, even after libusb rebuild as recommended by the instruction. The last know good
|
||||
configuration is version 2.3.0.63 (tested with Ubuntu 18.04 amd64). It's not provided officialy
|
||||
with the downloading page, but published by Orbbec technical suport on Orbbec community forum
|
||||
configuration is version 2.3.0.63 (tested with Ubuntu 18.04 amd64). It's not provided officially
|
||||
with the downloading page, but published by Orbbec technical support on Orbbec community forum
|
||||
[here](https://3dclub.orbbec3d.com/t/universal-download-thread-for-astra-series-cameras/622).
|
||||
|
||||
-# Now you can configure OpenCV with OpenNI support enabled by setting the `WITH_OPENNI2` flag in CMake.
|
||||
|
||||
+2
-2
@@ -63,7 +63,7 @@ Example code to generate features coordinates for calibration with symmetric gri
|
||||
}
|
||||
}
|
||||
```
|
||||
Example code to generate features corrdinates for calibration with asymmetic grid (object points):
|
||||
Example code to generate features coordinates for calibration with asymmetric grid (object points):
|
||||
```
|
||||
std::vector<cv::Point3f> objectPoints;
|
||||
for (int i = 0; i < boardSize.height; i++) {
|
||||
@@ -84,7 +84,7 @@ about ArUco pairs. In opposite to the previous pattern partially occluded board
|
||||
corners are labeled. The board is rotation invariant, but set of ArUco markers and their order
|
||||
should be known to detector apriori. It cannot detect ChAruco board with predefined size and random
|
||||
set of markers.
|
||||
Example code to generate features corrdinates for calibration (object points) for board size in units:
|
||||
Example code to generate features coordinates for calibration (object points) for board size in units:
|
||||
```
|
||||
std::vector<cv::Point3f> objectPoints;
|
||||
for (int i = 0; i < boardSize.height-1; ++i) {
|
||||
|
||||
@@ -64,25 +64,25 @@ We encourage you to add new algorithms to these APIs.
|
||||
|
||||
```
|
||||
crnn.onnx:
|
||||
url: https://drive.google.com/uc?export=dowload&id=1ooaLR-rkTl8jdpGy1DoQs0-X0lQsB6Fj
|
||||
url: https://drive.google.com/uc?export=download&id=1ooaLR-rkTl8jdpGy1DoQs0-X0lQsB6Fj
|
||||
sha: 270d92c9ccb670ada2459a25977e8deeaf8380d3,
|
||||
alphabet_36.txt: https://drive.google.com/uc?export=dowload&id=1oPOYx5rQRp8L6XQciUwmwhMCfX0KyO4b
|
||||
alphabet_36.txt: https://drive.google.com/uc?export=download&id=1oPOYx5rQRp8L6XQciUwmwhMCfX0KyO4b
|
||||
parameter setting: -rgb=0;
|
||||
description: The classification number of this model is 36 (0~9 + a~z).
|
||||
The training dataset is MJSynth.
|
||||
|
||||
crnn_cs.onnx:
|
||||
url: https://drive.google.com/uc?export=dowload&id=12diBsVJrS9ZEl6BNUiRp9s0xPALBS7kt
|
||||
url: https://drive.google.com/uc?export=download&id=12diBsVJrS9ZEl6BNUiRp9s0xPALBS7kt
|
||||
sha: a641e9c57a5147546f7a2dbea4fd322b47197cd5
|
||||
alphabet_94.txt: https://drive.google.com/uc?export=dowload&id=1oKXxXKusquimp7XY1mFvj9nwLzldVgBR
|
||||
alphabet_94.txt: https://drive.google.com/uc?export=download&id=1oKXxXKusquimp7XY1mFvj9nwLzldVgBR
|
||||
parameter setting: -rgb=1;
|
||||
description: The classification number of this model is 94 (0~9 + a~z + A~Z + punctuations).
|
||||
The training datasets are MJsynth and SynthText.
|
||||
|
||||
crnn_cs_CN.onnx:
|
||||
url: https://drive.google.com/uc?export=dowload&id=1is4eYEUKH7HR7Gl37Sw4WPXx6Ir8oQEG
|
||||
url: https://drive.google.com/uc?export=download&id=1is4eYEUKH7HR7Gl37Sw4WPXx6Ir8oQEG
|
||||
sha: 3940942b85761c7f240494cf662dcbf05dc00d14
|
||||
alphabet_3944.txt: https://drive.google.com/uc?export=dowload&id=18IZUUdNzJ44heWTndDO6NNfIpJMmN-ul
|
||||
alphabet_3944.txt: https://drive.google.com/uc?export=download&id=18IZUUdNzJ44heWTndDO6NNfIpJMmN-ul
|
||||
parameter setting: -rgb=1;
|
||||
description: The classification number of this model is 3944 (0~9 + a~z + A~Z + Chinese characters + special characters).
|
||||
The training dataset is ReCTS (https://rrc.cvc.uab.es/?ch=12).
|
||||
@@ -96,25 +96,25 @@ You can train more models by [CRNN](https://github.com/meijieru/crnn.pytorch), a
|
||||
|
||||
```
|
||||
- DB_IC15_resnet50.onnx:
|
||||
url: https://drive.google.com/uc?export=dowload&id=17_ABp79PlFt9yPCxSaarVc_DKTmrSGGf
|
||||
url: https://drive.google.com/uc?export=download&id=17_ABp79PlFt9yPCxSaarVc_DKTmrSGGf
|
||||
sha: bef233c28947ef6ec8c663d20a2b326302421fa3
|
||||
recommended parameter setting: -inputHeight=736, -inputWidth=1280;
|
||||
description: This model is trained on ICDAR2015, so it can only detect English text instances.
|
||||
|
||||
- DB_IC15_resnet18.onnx:
|
||||
url: https://drive.google.com/uc?export=dowload&id=1vY_KsDZZZb_svd5RT6pjyI8BS1nPbBSX
|
||||
url: https://drive.google.com/uc?export=download&id=1vY_KsDZZZb_svd5RT6pjyI8BS1nPbBSX
|
||||
sha: 19543ce09b2efd35f49705c235cc46d0e22df30b
|
||||
recommended parameter setting: -inputHeight=736, -inputWidth=1280;
|
||||
description: This model is trained on ICDAR2015, so it can only detect English text instances.
|
||||
|
||||
- DB_TD500_resnet50.onnx:
|
||||
url: https://drive.google.com/uc?export=dowload&id=19YWhArrNccaoSza0CfkXlA8im4-lAGsR
|
||||
url: https://drive.google.com/uc?export=download&id=19YWhArrNccaoSza0CfkXlA8im4-lAGsR
|
||||
sha: 1b4dd21a6baa5e3523156776970895bd3db6960a
|
||||
recommended parameter setting: -inputHeight=736, -inputWidth=736;
|
||||
description: This model is trained on MSRA-TD500, so it can detect both English and Chinese text instances.
|
||||
|
||||
- DB_TD500_resnet18.onnx:
|
||||
url: https://drive.google.com/uc?export=dowload&id=1sZszH3pEt8hliyBlTmB-iulxHP1dCQWV
|
||||
url: https://drive.google.com/uc?export=download&id=1sZszH3pEt8hliyBlTmB-iulxHP1dCQWV
|
||||
sha: 8a3700bdc13e00336a815fc7afff5dcc1ce08546
|
||||
recommended parameter setting: -inputHeight=736, -inputWidth=736;
|
||||
description: This model is trained on MSRA-TD500, so it can detect both English and Chinese text instances.
|
||||
@@ -133,11 +133,11 @@ This model is based on https://github.com/argman/EAST
|
||||
|
||||
```
|
||||
Text Recognition:
|
||||
url: https://drive.google.com/uc?export=dowload&id=1nMcEy68zDNpIlqAn6xCk_kYcUTIeSOtN
|
||||
url: https://drive.google.com/uc?export=download&id=1nMcEy68zDNpIlqAn6xCk_kYcUTIeSOtN
|
||||
sha: 89205612ce8dd2251effa16609342b69bff67ca3
|
||||
|
||||
Text Detection:
|
||||
url: https://drive.google.com/uc?export=dowload&id=149tAhIcvfCYeyufRoZ9tmc2mZDKE_XrF
|
||||
url: https://drive.google.com/uc?export=download&id=149tAhIcvfCYeyufRoZ9tmc2mZDKE_XrF
|
||||
sha: ced3c03fb7f8d9608169a913acf7e7b93e07109b
|
||||
```
|
||||
|
||||
|
||||
@@ -129,7 +129,7 @@ than YOLOX) in case it is needed. However, usually each YOLO repository has pred
|
||||
|
||||
#### Exporting YOLOv10 model
|
||||
|
||||
In oder to run YOLOv10 one needs to cut off postporcessing with dynamic shapes from torch and then convert it to ONNX. If someone is looking for on how to cut off the postprocessing, there is this [forked branch](https://github.com/Abdurrahheem/yolov10/tree/ash/opencv-export) from official YOLOv10. The forked branch cuts of the postprocessing by [returning output](https://github.com/Abdurrahheem/yolov10/blob/4fdaafd912c8891642bfbe85751ea66ec20f05ad/ultralytics/nn/modules/head.py#L522) of the model before postprocessing procedure itself. To convert torch model to ONNX follow this proceduce.
|
||||
In order to run YOLOv10 one needs to cut off postprocessing with dynamic shapes from torch and then convert it to ONNX. If someone is looking for on how to cut off the postprocessing, there is this [forked branch](https://github.com/Abdurrahheem/yolov10/tree/ash/opencv-export) from official YOLOv10. The forked branch cuts off the postprocessing by [returning output](https://github.com/Abdurrahheem/yolov10/blob/4fdaafd912c8891642bfbe85751ea66ec20f05ad/ultralytics/nn/modules/head.py#L522) of the model before postprocessing procedure itself. To convert torch model to ONNX follow this procedure.
|
||||
|
||||
@code{.bash}
|
||||
git clone git@github.com:Abdurrahheem/yolov10.git
|
||||
|
||||
@@ -46,14 +46,14 @@ Open your Doxyfile using your favorite text editor and search for the key
|
||||
`TAGFILES`. Change it as follows:
|
||||
|
||||
@code
|
||||
TAGFILES = ./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/4.13.0
|
||||
TAGFILES = ./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/4.14.0
|
||||
@endcode
|
||||
|
||||
If you had other definitions already, you can append the line using a `\`:
|
||||
|
||||
@code
|
||||
TAGFILES = ./docs/doxygen-tags/libstdc++.tag=https://gcc.gnu.org/onlinedocs/libstdc++/latest-doxygen \
|
||||
./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/4.13.0
|
||||
./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/4.14.0
|
||||
@endcode
|
||||
|
||||
Doxygen can now use the information from the tag file to link to the OpenCV
|
||||
|
||||
@@ -167,7 +167,7 @@ for `cv::aruco::Dictionary`. The data member of board classes are public and can
|
||||
To do so, you will need to use an external rendering engine library, such as OpenGL. The aruco module
|
||||
only provides the functionality to obtain the camera pose, i.e. the rotation and translation vectors,
|
||||
which is necessary to create the augmented reality effect. However, you will need to adapt the rotation
|
||||
and traslation vectors from the OpenCV format to the format accepted by your 3d rendering library.
|
||||
and translation vectors from the OpenCV format to the format accepted by your 3d rendering library.
|
||||
The original ArUco library contains examples of how to do it for OpenGL and Ogre3D.
|
||||
|
||||
|
||||
|
||||
+1
-1
@@ -12,7 +12,7 @@ It is similar to a ChArUco board in appearance, however they are conceptually di
|
||||
In both, ChArUco board and Diamond markers, their detection is based on the previous detected ArUco
|
||||
markers. In the ChArUco case, the used markers are selected by directly looking their identifiers. This means
|
||||
that if a marker (included in the board) is found on a image, it will be automatically assumed to belong to the board. Furthermore,
|
||||
if a marker board is found more than once in the image, it will produce an ambiguity since the system wont
|
||||
if a marker board is found more than once in the image, it will produce an ambiguity since the system won't
|
||||
be able to know which one should be used for the Board.
|
||||
|
||||
On the other hand, the detection of Diamond marker is not based on the identifiers. Instead, their detection
|
||||
|
||||
@@ -2285,7 +2285,7 @@ namespace CAROTENE_NS {
|
||||
f64 alpha, f64 beta);
|
||||
|
||||
/*
|
||||
Reduce matrix to a vector by calculatin given operation for each column
|
||||
Reduce matrix to a vector by calculating given operation for each column
|
||||
*/
|
||||
void reduceColSum(const Size2D &size,
|
||||
const u8 * srcBase, ptrdiff_t srcStride,
|
||||
|
||||
@@ -231,7 +231,7 @@ void accumulateSquare(const Size2D &size,
|
||||
internal::assertSupportedConfiguration();
|
||||
|
||||
#ifdef CAROTENE_NEON
|
||||
// this ugly contruction is needed to avoid:
|
||||
// this ugly construction is needed to avoid:
|
||||
// /usr/lib/gcc/arm-linux-gnueabihf/4.8/include/arm_neon.h:3581:59: error: argument must be a constant
|
||||
// return (int16x8_t)__builtin_neon_vshr_nv8hi (__a, __b, 1);
|
||||
|
||||
|
||||
@@ -524,7 +524,7 @@ inline void Canny3x3(const Size2D &size, s32 cn,
|
||||
|
||||
//i == 0
|
||||
normEstimator.firstRow(size, cn, srcBase, srcStride, dxBase, dxStride, dyBase, dyStride, mag_buf);
|
||||
// calculate magnitude and angle of gradient, perform non-maxima supression.
|
||||
// calculate magnitude and angle of gradient, perform non-maxima suppression.
|
||||
// fill the map with one of the following values:
|
||||
// 0 - the pixel might belong to an edge
|
||||
// 1 - the pixel can not belong to an edge
|
||||
|
||||
@@ -66,7 +66,7 @@ inline float32x2_t vrecp_f32(float32x2_t val)
|
||||
return reciprocal;
|
||||
}
|
||||
|
||||
// caclulate sqrt value
|
||||
// calculate sqrt value
|
||||
|
||||
inline float32x4_t vrsqrtq_f32(float32x4_t val)
|
||||
{
|
||||
|
||||
@@ -9,13 +9,24 @@ set(IPP_HAL_HEADERS
|
||||
CACHE INTERNAL "")
|
||||
|
||||
add_library(ipphal STATIC
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/math_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/mean_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/minmax_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/norm_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/cart_polar_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/transforms_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/warp_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/deriv_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/resize_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/box_filter_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/sum_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/color_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/filter_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/matchtemplate_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/canny_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/threshold_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/distancetransform_ipp.cpp"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/src/histogram_ipp.cpp"
|
||||
)
|
||||
|
||||
#TODO: HAVE_IPP_ICV and HAVE_IPP_IW added as private macro till OpenCV itself is
|
||||
|
||||
@@ -12,8 +12,10 @@
|
||||
int ipp_hal_meanStdDev(const uchar* src_data, size_t src_step, int width, int height, int src_type,
|
||||
double* mean_val, double* stddev_val, uchar* mask, size_t mask_step);
|
||||
|
||||
#undef cv_hal_meanStdDev
|
||||
#define cv_hal_meanStdDev ipp_hal_meanStdDev
|
||||
// IPP version is less efficient than implementation with universtal intrinsics
|
||||
// See https://github.com/opencv/opencv/pull/29339
|
||||
//#undef cv_hal_meanStdDev
|
||||
//#define cv_hal_meanStdDev ipp_hal_meanStdDev
|
||||
|
||||
int ipp_hal_minMaxIdxMaskStep(const uchar* src_data, size_t src_step, int width, int height, int depth,
|
||||
double* _minVal, double* _maxVal, int* _minIdx, int* _maxIdx, uchar* mask, size_t mask_step);
|
||||
@@ -74,6 +76,33 @@ int ipp_hal_transpose2d(const uchar* src_data, size_t src_step, uchar* dst_data,
|
||||
#undef cv_hal_transpose2d
|
||||
#define cv_hal_transpose2d ipp_hal_transpose2d
|
||||
|
||||
int ipp_hal_invSqrt32f(const float* src, float* dst, int len);
|
||||
int ipp_hal_invSqrt64f(const double* src, double* dst, int len);
|
||||
|
||||
#undef cv_hal_invSqrt32f
|
||||
#define cv_hal_invSqrt32f ipp_hal_invSqrt32f
|
||||
|
||||
#undef cv_hal_invSqrt64f
|
||||
#define cv_hal_invSqrt64f ipp_hal_invSqrt64f
|
||||
|
||||
int ipp_hal_exp32f(const float* src, float* dst, int len);
|
||||
int ipp_hal_exp64f(const double* src, double* dst, int len);
|
||||
|
||||
#undef cv_hal_exp32f
|
||||
#define cv_hal_exp32f ipp_hal_exp32f
|
||||
|
||||
#undef cv_hal_exp64f
|
||||
#define cv_hal_exp64f ipp_hal_exp64f
|
||||
|
||||
int ipp_hal_log32f(const float* src, float* dst, int len);
|
||||
int ipp_hal_log64f(const double* src, double* dst, int len);
|
||||
|
||||
#undef cv_hal_log32f
|
||||
#define cv_hal_log32f ipp_hal_log32f
|
||||
|
||||
#undef cv_hal_log64f
|
||||
#define cv_hal_log64f ipp_hal_log64f
|
||||
|
||||
//! @endcond
|
||||
|
||||
#endif
|
||||
|
||||
@@ -8,13 +8,35 @@
|
||||
#include <opencv2/core/base.hpp>
|
||||
#include "ipp_utils.hpp"
|
||||
|
||||
#if IPP_VERSION_X100 >= 810
|
||||
// Disabled in https://github.com/opencv/opencv/pull/13085 due large binary size
|
||||
#define DISABLE_IPP_BOX_FILTER 1
|
||||
|
||||
// IPP filter2D integration is disabled in main OpenCV; kept behind a macro like box filter
|
||||
#define DISABLE_IPP_FILTER2D 1
|
||||
// Too big difference compared to OpenCV FFT-based convolution, different results on masks > 7x7
|
||||
#define IPP_DISABLE_FILTER2D_BIG_MASK 1
|
||||
|
||||
#if IPP_VERSION_X100 >= 810
|
||||
#if defined(HAVE_IPP_IW)
|
||||
int ipp_hal_warpAffine(int src_type, const uchar *src_data, size_t src_step, int src_width, int src_height, uchar *dst_data, size_t dst_step, int dst_width,
|
||||
int dst_height, const double M[6], int interpolation, int borderType, const double borderValue[4]);
|
||||
#undef cv_hal_warpAffine
|
||||
#define cv_hal_warpAffine ipp_hal_warpAffine
|
||||
|
||||
int ipp_hal_sobel(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int src_depth, int dst_depth, int cn,
|
||||
int margin_left, int margin_top, int margin_right, int margin_bottom,
|
||||
int dx, int dy, int ksize, double scale, double delta, int border_type);
|
||||
#undef cv_hal_sobel
|
||||
#define cv_hal_sobel ipp_hal_sobel
|
||||
|
||||
int ipp_hal_scharr(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int src_depth, int dst_depth, int cn,
|
||||
int margin_left, int margin_top, int margin_right, int margin_bottom,
|
||||
int dx, int dy, double scale, double delta, int border_type);
|
||||
#undef cv_hal_scharr
|
||||
#define cv_hal_scharr ipp_hal_scharr
|
||||
|
||||
#endif
|
||||
|
||||
#if IPP_VERSION_X100 >= 202600
|
||||
@@ -26,6 +48,24 @@ int ipp_hal_warpPerspective(int src_type, const uchar *src_data, size_t src_step
|
||||
|
||||
#endif // IPP_VERSION_X100 >= 202600
|
||||
|
||||
#if defined(HAVE_IPP_IW)
|
||||
int ipp_hal_resize(int src_type, const uchar *src_data, size_t src_step, int src_width, int src_height,
|
||||
uchar *dst_data, size_t dst_step, int dst_width, int dst_height,
|
||||
double inv_scale_x, double inv_scale_y, int interpolation);
|
||||
#undef cv_hal_resize
|
||||
#define cv_hal_resize ipp_hal_resize
|
||||
#endif // HAVE_IPP_IW
|
||||
|
||||
#if defined(HAVE_IPP_IW) && !DISABLE_IPP_BOX_FILTER
|
||||
int ipp_hal_boxFilter(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int src_depth, int dst_depth, int cn,
|
||||
int margin_left, int margin_top, int margin_right, int margin_bottom,
|
||||
size_t ksize_width, size_t ksize_height, int anchor_x, int anchor_y,
|
||||
bool normalize, int border_type);
|
||||
#undef cv_hal_boxFilter
|
||||
#define cv_hal_boxFilter ipp_hal_boxFilter
|
||||
#endif // defined(HAVE_IPP_IW) && !DISABLE_IPP_BOX_FILTER
|
||||
|
||||
int ipp_hal_remap32f(int src_type, const uchar *src_data, size_t src_step, int src_width, int src_height,
|
||||
uchar *dst_data, size_t dst_step, int dst_width, int dst_height,
|
||||
float* mapx, size_t mapx_step, float* mapy, size_t mapy_step,
|
||||
@@ -33,6 +73,108 @@ int ipp_hal_remap32f(int src_type, const uchar *src_data, size_t src_step, int s
|
||||
#undef cv_hal_remap32f
|
||||
#define cv_hal_remap32f ipp_hal_remap32f
|
||||
|
||||
#if defined(HAVE_IPP_IW) && !DISABLE_IPP_FILTER2D
|
||||
int ipp_hal_filter2D(const uchar * src_data, size_t src_step, int src_type,
|
||||
uchar * dst_data, size_t dst_step, int dst_type,
|
||||
int width, int height, int full_width, int full_height,
|
||||
int offset_x, int offset_y,
|
||||
const uchar * kernel_data, size_t kernel_step, int kernel_type,
|
||||
int kernel_width, int kernel_height,
|
||||
int anchor_x, int anchor_y, double delta, int borderType,
|
||||
bool isSubmatrix, bool allowInplace);
|
||||
#undef cv_hal_filter_stateless
|
||||
#define cv_hal_filter_stateless ipp_hal_filter2D
|
||||
#endif // defined(HAVE_IPP_IW) && !DISABLE_IPP_FILTER2D
|
||||
|
||||
#endif //IPP_VERSION_X100 >= 810
|
||||
|
||||
#if IPP_VERSION_X100 >= 700
|
||||
|
||||
int ipp_hal_cvtBGRtoGray(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int scn, bool swapBlue);
|
||||
#undef cv_hal_cvtBGRtoGray
|
||||
#define cv_hal_cvtBGRtoGray ipp_hal_cvtBGRtoGray
|
||||
|
||||
int ipp_hal_cvtBGRtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int scn, int dcn, bool swapBlue);
|
||||
#undef cv_hal_cvtBGRtoBGR
|
||||
#define cv_hal_cvtBGRtoBGR ipp_hal_cvtBGRtoBGR
|
||||
|
||||
int ipp_hal_cvtGraytoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int dcn);
|
||||
#undef cv_hal_cvtGraytoBGR
|
||||
#define cv_hal_cvtGraytoBGR ipp_hal_cvtGraytoBGR
|
||||
|
||||
int ipp_hal_cvtBGRtoHSV(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int scn, bool swapBlue, bool isFullRange, bool isHSV);
|
||||
#undef cv_hal_cvtBGRtoHSV
|
||||
#define cv_hal_cvtBGRtoHSV ipp_hal_cvtBGRtoHSV
|
||||
|
||||
int ipp_hal_cvtHSVtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int dcn, bool swapBlue, bool isFullRange, bool isHSV);
|
||||
#undef cv_hal_cvtHSVtoBGR
|
||||
#define cv_hal_cvtHSVtoBGR ipp_hal_cvtHSVtoBGR
|
||||
|
||||
int ipp_hal_cvtBGRtoYUV(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int scn, bool swapBlue, bool isCbCr);
|
||||
#undef cv_hal_cvtBGRtoYUV
|
||||
#define cv_hal_cvtBGRtoYUV ipp_hal_cvtBGRtoYUV
|
||||
|
||||
int ipp_hal_cvtYUVtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int dcn, bool swapBlue, bool isCbCr);
|
||||
#undef cv_hal_cvtYUVtoBGR
|
||||
#define cv_hal_cvtYUVtoBGR ipp_hal_cvtYUVtoBGR
|
||||
|
||||
int ipp_hal_cvtBGRtoXYZ(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int scn, bool swapBlue);
|
||||
#undef cv_hal_cvtBGRtoXYZ
|
||||
#define cv_hal_cvtBGRtoXYZ ipp_hal_cvtBGRtoXYZ
|
||||
|
||||
int ipp_hal_cvtXYZtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int dcn, bool swapBlue);
|
||||
#undef cv_hal_cvtXYZtoBGR
|
||||
#define cv_hal_cvtXYZtoBGR ipp_hal_cvtXYZtoBGR
|
||||
|
||||
int ipp_hal_cvtBGRtoLab(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int scn, bool swapBlue, bool isLab, bool srgb);
|
||||
#undef cv_hal_cvtBGRtoLab
|
||||
#define cv_hal_cvtBGRtoLab ipp_hal_cvtBGRtoLab
|
||||
|
||||
int ipp_hal_cvtLabtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height, int depth, int dcn, bool swapBlue, bool isLab, bool srgb);
|
||||
#undef cv_hal_cvtLabtoBGR
|
||||
#define cv_hal_cvtLabtoBGR ipp_hal_cvtLabtoBGR
|
||||
|
||||
int ipp_hal_cvtRGBAtoMultipliedRGBA(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int width, int height);
|
||||
#undef cv_hal_cvtRGBAtoMultipliedRGBA
|
||||
#define cv_hal_cvtRGBAtoMultipliedRGBA ipp_hal_cvtRGBAtoMultipliedRGBA
|
||||
|
||||
int ipp_hal_matchTemplate(const uchar* src_data, size_t src_step, int src_width, int src_height,
|
||||
const uchar* templ_data, size_t templ_step, int templ_width, int templ_height,
|
||||
float* result_data, size_t result_step, int depth, int cn, int method);
|
||||
#undef cv_hal_matchTemplate
|
||||
#define cv_hal_matchTemplate ipp_hal_matchTemplate
|
||||
|
||||
int ipp_hal_threshold(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int cn, double thresh, double maxValue, int thresholdType);
|
||||
#undef cv_hal_threshold
|
||||
#define cv_hal_threshold ipp_hal_threshold
|
||||
|
||||
int ipp_hal_distanceTransform(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int dst_type, int dist_type, int mask_size);
|
||||
#undef cv_hal_distanceTransform
|
||||
#define cv_hal_distanceTransform ipp_hal_distanceTransform
|
||||
|
||||
int ipp_hal_calcHist(const uchar* src_data, size_t src_step, int src_type, int src_width, int src_height,
|
||||
float* hist_data, int hist_size, const float** ranges, bool uniform, bool accumulate);
|
||||
#undef cv_hal_calcHist
|
||||
#define cv_hal_calcHist ipp_hal_calcHist
|
||||
|
||||
#endif // IPP_VERSION_X100 >= 700
|
||||
|
||||
#define IPP_DISABLE_PERF_CANNY_MT 1 // cv::Canny OpenCV MT performance is better
|
||||
|
||||
#if defined(HAVE_IPP_IW)
|
||||
int ipp_hal_canny(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int cn,
|
||||
double lowThreshold, double highThreshold, int ksize, bool L2gradient);
|
||||
#undef cv_hal_canny
|
||||
#define cv_hal_canny ipp_hal_canny
|
||||
|
||||
int ipp_hal_canny_deriv(const short* dx_data, size_t dx_step, const short* dy_data, size_t dy_step,
|
||||
uchar* dst_data, size_t dst_step, int width, int height, int cn,
|
||||
double lowThreshold, double highThreshold, bool L2gradient);
|
||||
#undef cv_hal_canny_deriv
|
||||
#define cv_hal_canny_deriv ipp_hal_canny_deriv
|
||||
#endif
|
||||
|
||||
#endif //__IPP_HAL_IMGPROC_HPP__
|
||||
|
||||
@@ -0,0 +1,144 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, BigVision LLC, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#if defined(HAVE_IPP_IW) && !DISABLE_IPP_BOX_FILTER
|
||||
|
||||
namespace cv { namespace ipp { unsigned long long getIppTopFeatures(); } }
|
||||
|
||||
// Copied from core/private.hpp (gated by HAVE_IPP, which the plugin lacks).
|
||||
// boxGetIppBorderType: distinct name, adds the REFLECT_101 precomp's version drops.
|
||||
static inline IppiBorderType boxGetIppBorderType(int borderTypeNI)
|
||||
{
|
||||
return borderTypeNI == cv::BORDER_CONSTANT ? ippBorderConst :
|
||||
borderTypeNI == cv::BORDER_TRANSPARENT ? ippBorderTransp :
|
||||
borderTypeNI == cv::BORDER_REPLICATE ? ippBorderRepl :
|
||||
borderTypeNI == cv::BORDER_REFLECT_101 ? ippBorderMirror :
|
||||
(IppiBorderType)-1;
|
||||
}
|
||||
|
||||
static inline bool ippiCheckAnchor(int x, int y, int kernelWidth, int kernelHeight)
|
||||
{
|
||||
return (x == (kernelWidth - 1)/2 && y == (kernelHeight - 1)/2);
|
||||
}
|
||||
|
||||
static inline IppiBorderType ippiGetBorder(::ipp::IwiImage &image, int ocvBorderType, ::ipp::IwiBorderSize &borderSize)
|
||||
{
|
||||
int inMemFlags = 0;
|
||||
IppiBorderType border = boxGetIppBorderType(ocvBorderType & ~cv::BORDER_ISOLATED);
|
||||
if((int)border == -1)
|
||||
return (IppiBorderType)0;
|
||||
|
||||
if(!(ocvBorderType & cv::BORDER_ISOLATED))
|
||||
{
|
||||
if(image.m_inMemSize.left)
|
||||
{
|
||||
if(image.m_inMemSize.left >= borderSize.left)
|
||||
inMemFlags |= ippBorderInMemLeft;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.left = 0;
|
||||
if(image.m_inMemSize.top)
|
||||
{
|
||||
if(image.m_inMemSize.top >= borderSize.top)
|
||||
inMemFlags |= ippBorderInMemTop;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.top = 0;
|
||||
if(image.m_inMemSize.right)
|
||||
{
|
||||
if(image.m_inMemSize.right >= borderSize.right)
|
||||
inMemFlags |= ippBorderInMemRight;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.right = 0;
|
||||
if(image.m_inMemSize.bottom)
|
||||
{
|
||||
if(image.m_inMemSize.bottom >= borderSize.bottom)
|
||||
inMemFlags |= ippBorderInMemBottom;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.bottom = 0;
|
||||
}
|
||||
else
|
||||
borderSize.left = borderSize.right = borderSize.top = borderSize.bottom = 0;
|
||||
|
||||
return (IppiBorderType)(border | inMemFlags);
|
||||
}
|
||||
|
||||
int ipp_hal_boxFilter(const uchar* src_data, size_t src_step,
|
||||
uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int src_depth, int dst_depth, int cn,
|
||||
int margin_left, int margin_top, int margin_right, int margin_bottom,
|
||||
size_t ksize_width, size_t ksize_height,
|
||||
int anchor_x, int anchor_y,
|
||||
bool normalize, int border_type)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if IPP_VERSION_X100 < 201801
|
||||
// Problem with SSE42 optimization for 16s and some 8u modes
|
||||
if(cv::ipp::getIppTopFeatures() == ippCPUID_SSE42 &&
|
||||
(((src_depth == CV_16S || src_depth == CV_16U) && (cn == 3 || cn == 4)) ||
|
||||
(src_depth == CV_8U && cn == 3 && (ksize_width > 5 || ksize_height > 5))))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// Other optimizations has some degradations too
|
||||
if(((src_depth == CV_16S || src_depth == CV_16U) && cn == 4) ||
|
||||
(src_depth == CV_8U && cn == 1 && (ksize_width > 5 || ksize_height > 5)))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
#endif
|
||||
|
||||
if(!normalize)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(!ippiCheckAnchor(anchor_x, anchor_y, (int)ksize_width, (int)ksize_height))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
try
|
||||
{
|
||||
// raw-pointer equivalent of ippiGetImage: margins are the in-memory border
|
||||
::ipp::IwiBorderSize inMemBorder;
|
||||
inMemBorder.left = (IwSize)margin_left;
|
||||
inMemBorder.top = (IwSize)margin_top;
|
||||
inMemBorder.right = (IwSize)margin_right;
|
||||
inMemBorder.bottom = (IwSize)margin_bottom;
|
||||
|
||||
::ipp::IwiImage iwSrc, iwDst;
|
||||
iwSrc.Init(IwiSize{width, height}, ippiGetDataType(src_depth), cn,
|
||||
inMemBorder, (void*)src_data, IwSize(src_step));
|
||||
iwDst.Init(IwiSize{width, height}, ippiGetDataType(dst_depth), cn,
|
||||
::ipp::IwiBorderSize(), dst_data, IwSize(dst_step));
|
||||
|
||||
::ipp::IwiSize iwKSize{(int)ksize_width, (int)ksize_height};
|
||||
::ipp::IwiBorderSize borderSize(iwKSize);
|
||||
::ipp::IwiBorderType ippBorder(ippiGetBorder(iwSrc, border_type, borderSize));
|
||||
if(!ippBorder)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
CV_INSTRUMENT_FUN_IPP(::ipp::iwiFilterBox, iwSrc, iwDst, iwKSize, ::ipp::IwDefault(), ippBorder);
|
||||
}
|
||||
catch (const ::ipp::IwException &)
|
||||
{
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
#endif // defined(HAVE_IPP_IW) && !DISABLE_IPP_BOX_FILTER
|
||||
@@ -0,0 +1,95 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, BigVision LLC, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#ifdef HAVE_IPP_IW
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
int ipp_hal_canny(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int cn,
|
||||
double lowThreshold, double highThreshold, int ksize, bool L2gradient)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if IPP_DISABLE_PERF_CANNY_MT
|
||||
if(cv::getNumThreads() > 1)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
#endif
|
||||
|
||||
if(width <= 3 || height <= 3)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(cn != 1)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
IppiMaskSize kernel;
|
||||
if(ksize == 3)
|
||||
kernel = ippMskSize3x3;
|
||||
else if(ksize == 5)
|
||||
kernel = ippMskSize5x5;
|
||||
else
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
IppNormType norm = L2gradient ? ippNormL2 : ippNormL1;
|
||||
|
||||
try
|
||||
{
|
||||
::ipp::IwiImage iwSrc, iwDst;
|
||||
iwSrc.Init(IwiSize{width, height}, ipp8u, cn, ::ipp::IwiBorderSize(), (void*)src_data, IwSize(src_step));
|
||||
iwDst.Init(IwiSize{width, height}, ipp8u, cn, ::ipp::IwiBorderSize(), dst_data, IwSize(dst_step));
|
||||
|
||||
CV_INSTRUMENT_FUN_IPP(::ipp::iwiFilterCanny, iwSrc, iwDst, (float)lowThreshold, (float)highThreshold,
|
||||
::ipp::IwiFilterCannyParams(ippFilterSobel, kernel, norm), ippBorderRepl);
|
||||
}
|
||||
catch (const ::ipp::IwException &)
|
||||
{
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
int ipp_hal_canny_deriv(const short* dx_data, size_t dx_step, const short* dy_data, size_t dy_step,
|
||||
uchar* dst_data, size_t dst_step, int width, int height, int cn,
|
||||
double lowThreshold, double highThreshold, bool L2gradient)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if IPP_DISABLE_PERF_CANNY_MT
|
||||
if(cv::getNumThreads() > 1)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
#endif
|
||||
|
||||
if(width <= 3 || height <= 3)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(cn != 1)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
IppNormType norm = L2gradient ? ippNormL2 : ippNormL1;
|
||||
|
||||
try
|
||||
{
|
||||
::ipp::IwiImage iwSrcDx, iwSrcDy, iwDst;
|
||||
iwSrcDx.Init(IwiSize{width, height}, ipp16s, cn, ::ipp::IwiBorderSize(), (void*)dx_data, IwSize(dx_step));
|
||||
iwSrcDy.Init(IwiSize{width, height}, ipp16s, cn, ::ipp::IwiBorderSize(), (void*)dy_data, IwSize(dy_step));
|
||||
iwDst.Init(IwiSize{width, height}, ipp8u, cn, ::ipp::IwiBorderSize(), dst_data, IwSize(dst_step));
|
||||
|
||||
CV_INSTRUMENT_FUN_IPP(::ipp::iwiFilterCannyDeriv, iwSrcDx, iwSrcDy, iwDst, (float)lowThreshold, (float)highThreshold,
|
||||
::ipp::IwiFilterCannyDerivParams(norm));
|
||||
}
|
||||
catch (const ::ipp::IwException &)
|
||||
{
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,958 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
#include <opencv2/core.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#include <limits>
|
||||
|
||||
#if IPP_VERSION_X100 >= 700
|
||||
|
||||
#define MAX_IPP8u 255
|
||||
#define MAX_IPP16u 65535
|
||||
#define MAX_IPP32f 1.0
|
||||
|
||||
// moved from color_rgb.dispatch.cpp
|
||||
#define IPP_DISABLE_CVTCOLOR_GRAY2BGR_8UC3 1
|
||||
|
||||
// replicated from core/private.hpp
|
||||
#define IPP_DISABLE_RGB_HSV 1
|
||||
#define IPP_DISABLE_RGB_YUV 1
|
||||
#define IPP_DISABLE_YUV_RGB 1
|
||||
#define IPP_DISABLE_RGB_XYZ 1
|
||||
#define IPP_DISABLE_XYZ_RGB 1
|
||||
#define IPP_DISABLE_RGB_LAB 1
|
||||
#define IPP_DISABLE_LAB_RGB 1
|
||||
|
||||
namespace {
|
||||
|
||||
typedef IppStatus (CV_STDCALL* ippiGeneralFunc)(const void *, int, void *, int, IppiSize);
|
||||
typedef IppStatus (CV_STDCALL* ippiColor2GrayFunc)(const void *, int, void *, int, IppiSize, const Ipp32f *);
|
||||
typedef IppStatus (CV_STDCALL* ippiReorderFunc)(const void *, int, void *, int, IppiSize, const int *);
|
||||
|
||||
// BT.601 BGR->Y weights
|
||||
static const float B2YF = 0.114f;
|
||||
static const float G2YF = 0.587f;
|
||||
static const float R2YF = 0.299f;
|
||||
|
||||
template<typename _Tp> struct ColorChannel
|
||||
{
|
||||
static inline _Tp max() { return std::numeric_limits<_Tp>::max(); }
|
||||
};
|
||||
template<> struct ColorChannel<float>
|
||||
{
|
||||
static inline float max() { return 1.f; }
|
||||
};
|
||||
|
||||
template <typename Cvt>
|
||||
class CvtColorIPPLoop_Invoker : public cv::ParallelLoopBody
|
||||
{
|
||||
public:
|
||||
CvtColorIPPLoop_Invoker(const uchar * src_data_, size_t src_step_, uchar * dst_data_, size_t dst_step_,
|
||||
int width_, const Cvt& _cvt, bool *_ok) :
|
||||
ParallelLoopBody(), src_data(src_data_), src_step(src_step_), dst_data(dst_data_), dst_step(dst_step_),
|
||||
width(width_), cvt(_cvt), ok(_ok)
|
||||
{
|
||||
*ok = true;
|
||||
}
|
||||
|
||||
virtual void operator()(const cv::Range& range) const CV_OVERRIDE
|
||||
{
|
||||
const void *yS = src_data + src_step * range.start;
|
||||
void *yD = dst_data + dst_step * range.start;
|
||||
if( !cvt(yS, static_cast<int>(src_step), yD, static_cast<int>(dst_step), width, range.end - range.start) )
|
||||
*ok = false;
|
||||
else
|
||||
{
|
||||
CV_IMPL_ADD(CV_IMPL_IPP|CV_IMPL_MT);
|
||||
}
|
||||
}
|
||||
|
||||
private:
|
||||
const uchar * src_data;
|
||||
const size_t src_step;
|
||||
uchar * dst_data;
|
||||
const size_t dst_step;
|
||||
const int width;
|
||||
const Cvt& cvt;
|
||||
bool *ok;
|
||||
|
||||
const CvtColorIPPLoop_Invoker& operator= (const CvtColorIPPLoop_Invoker&);
|
||||
};
|
||||
|
||||
template <typename Cvt>
|
||||
bool CvtColorIPPLoop(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, const Cvt& cvt)
|
||||
{
|
||||
bool ok;
|
||||
cv::parallel_for_(cv::Range(0, height),
|
||||
CvtColorIPPLoop_Invoker<Cvt>(src_data, src_step, dst_data, dst_step, width, cvt, &ok),
|
||||
(width * height)/(double)(1<<16) );
|
||||
return ok;
|
||||
}
|
||||
|
||||
template <typename Cvt>
|
||||
bool CvtColorIPPLoopCopy(const uchar * src_data, size_t src_step, int src_type, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, const Cvt& cvt)
|
||||
{
|
||||
cv::Mat temp;
|
||||
cv::Mat src(cv::Size(width, height), src_type, const_cast<uchar*>(src_data), src_step);
|
||||
cv::Mat source = src;
|
||||
if( src_data == dst_data )
|
||||
{
|
||||
src.copyTo(temp);
|
||||
source = temp;
|
||||
}
|
||||
bool ok;
|
||||
cv::parallel_for_(cv::Range(0, source.rows),
|
||||
CvtColorIPPLoop_Invoker<Cvt>(source.data, source.step, dst_data, dst_step,
|
||||
source.cols, cvt, &ok),
|
||||
source.total()/(double)(1<<16) );
|
||||
return ok;
|
||||
}
|
||||
|
||||
struct IPPGeneralFunctor
|
||||
{
|
||||
IPPGeneralFunctor(ippiGeneralFunc _func) : ippiColorConvertGeneral(_func){}
|
||||
bool operator()(const void *src, int srcStep, void *dst, int dstStep, int cols, int rows) const
|
||||
{
|
||||
return ippiColorConvertGeneral ? CV_INSTRUMENT_FUN_IPP(ippiColorConvertGeneral, src, srcStep, dst, dstStep, ippiSize(cols, rows)) >= 0 : false;
|
||||
}
|
||||
private:
|
||||
ippiGeneralFunc ippiColorConvertGeneral;
|
||||
};
|
||||
|
||||
struct IPPReorderFunctor
|
||||
{
|
||||
IPPReorderFunctor(ippiReorderFunc _func, int _order0, int _order1, int _order2) : ippiColorConvertReorder(_func)
|
||||
{
|
||||
order[0] = _order0;
|
||||
order[1] = _order1;
|
||||
order[2] = _order2;
|
||||
order[3] = 3;
|
||||
}
|
||||
bool operator()(const void *src, int srcStep, void *dst, int dstStep, int cols, int rows) const
|
||||
{
|
||||
return ippiColorConvertReorder ? CV_INSTRUMENT_FUN_IPP(ippiColorConvertReorder, src, srcStep, dst, dstStep, ippiSize(cols, rows), order) >= 0 : false;
|
||||
}
|
||||
private:
|
||||
ippiReorderFunc ippiColorConvertReorder;
|
||||
int order[4];
|
||||
};
|
||||
|
||||
struct IPPReorderGeneralFunctor
|
||||
{
|
||||
IPPReorderGeneralFunctor(ippiReorderFunc _func1, ippiGeneralFunc _func2, int _order0, int _order1, int _order2, int _depth) :
|
||||
ippiColorConvertReorder(_func1), ippiColorConvertGeneral(_func2), depth(_depth)
|
||||
{
|
||||
order[0] = _order0;
|
||||
order[1] = _order1;
|
||||
order[2] = _order2;
|
||||
order[3] = 3;
|
||||
}
|
||||
bool operator()(const void *src, int srcStep, void *dst, int dstStep, int cols, int rows) const
|
||||
{
|
||||
if (ippiColorConvertReorder == 0 || ippiColorConvertGeneral == 0)
|
||||
return false;
|
||||
|
||||
cv::Mat temp;
|
||||
temp.create(rows, cols, CV_MAKETYPE(depth, 3));
|
||||
if(CV_INSTRUMENT_FUN_IPP(ippiColorConvertReorder, src, srcStep, temp.ptr(), (int)temp.step[0], ippiSize(cols, rows), order) < 0)
|
||||
return false;
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiColorConvertGeneral, temp.ptr(), (int)temp.step[0], dst, dstStep, ippiSize(cols, rows)) >= 0;
|
||||
}
|
||||
private:
|
||||
ippiReorderFunc ippiColorConvertReorder;
|
||||
ippiGeneralFunc ippiColorConvertGeneral;
|
||||
int order[4];
|
||||
int depth;
|
||||
};
|
||||
|
||||
struct IPPGeneralReorderFunctor
|
||||
{
|
||||
IPPGeneralReorderFunctor(ippiGeneralFunc _func1, ippiReorderFunc _func2, int _order0, int _order1, int _order2, int _depth) :
|
||||
ippiColorConvertGeneral(_func1), ippiColorConvertReorder(_func2), depth(_depth)
|
||||
{
|
||||
order[0] = _order0;
|
||||
order[1] = _order1;
|
||||
order[2] = _order2;
|
||||
order[3] = 3;
|
||||
}
|
||||
bool operator()(const void *src, int srcStep, void *dst, int dstStep, int cols, int rows) const
|
||||
{
|
||||
if (ippiColorConvertGeneral == 0 || ippiColorConvertReorder == 0)
|
||||
return false;
|
||||
|
||||
cv::Mat temp;
|
||||
temp.create(rows, cols, CV_MAKETYPE(depth, 3));
|
||||
if(CV_INSTRUMENT_FUN_IPP(ippiColorConvertGeneral, src, srcStep, temp.ptr(), (int)temp.step[0], ippiSize(cols, rows)) < 0)
|
||||
return false;
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiColorConvertReorder, temp.ptr(), (int)temp.step[0], dst, dstStep, ippiSize(cols, rows), order) >= 0;
|
||||
}
|
||||
private:
|
||||
ippiGeneralFunc ippiColorConvertGeneral;
|
||||
ippiReorderFunc ippiColorConvertReorder;
|
||||
int order[4];
|
||||
int depth;
|
||||
};
|
||||
|
||||
struct IPPColor2GrayFunctor
|
||||
{
|
||||
IPPColor2GrayFunctor(ippiColor2GrayFunc _func) : ippiColorToGray(_func)
|
||||
{
|
||||
coeffs[0] = B2YF;
|
||||
coeffs[1] = G2YF;
|
||||
coeffs[2] = R2YF;
|
||||
}
|
||||
bool operator()(const void *src, int srcStep, void *dst, int dstStep, int cols, int rows) const
|
||||
{
|
||||
return ippiColorToGray ? CV_INSTRUMENT_FUN_IPP(ippiColorToGray, src, srcStep, dst, dstStep, ippiSize(cols, rows), coeffs) >= 0 : false;
|
||||
}
|
||||
private:
|
||||
ippiColor2GrayFunc ippiColorToGray;
|
||||
Ipp32f coeffs[3];
|
||||
};
|
||||
|
||||
#if !IPP_DISABLE_CVTCOLOR_GRAY2BGR_8UC3
|
||||
static IppStatus ippiGrayToRGB_C1C3R(const Ipp8u* pSrc, int srcStep, Ipp8u* pDst, int dstStep, IppiSize roiSize)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiGrayToRGB_8u_C1C3R, pSrc, srcStep, pDst, dstStep, roiSize);
|
||||
}
|
||||
#endif
|
||||
static IppStatus ippiGrayToRGB_C1C3R(const Ipp16u* pSrc, int srcStep, Ipp16u* pDst, int dstStep, IppiSize roiSize)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiGrayToRGB_16u_C1C3R, pSrc, srcStep, pDst, dstStep, roiSize);
|
||||
}
|
||||
static IppStatus ippiGrayToRGB_C1C3R(const Ipp32f* pSrc, int srcStep, Ipp32f* pDst, int dstStep, IppiSize roiSize)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiGrayToRGB_32f_C1C3R, pSrc, srcStep, pDst, dstStep, roiSize);
|
||||
}
|
||||
|
||||
static IppStatus ippiGrayToRGB_C1C4R(const Ipp8u* pSrc, int srcStep, Ipp8u* pDst, int dstStep, IppiSize roiSize, Ipp8u aval)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiGrayToRGB_8u_C1C4R, pSrc, srcStep, pDst, dstStep, roiSize, aval);
|
||||
}
|
||||
static IppStatus ippiGrayToRGB_C1C4R(const Ipp16u* pSrc, int srcStep, Ipp16u* pDst, int dstStep, IppiSize roiSize, Ipp16u aval)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiGrayToRGB_16u_C1C4R, pSrc, srcStep, pDst, dstStep, roiSize, aval);
|
||||
}
|
||||
static IppStatus ippiGrayToRGB_C1C4R(const Ipp32f* pSrc, int srcStep, Ipp32f* pDst, int dstStep, IppiSize roiSize, Ipp32f aval)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiGrayToRGB_32f_C1C4R, pSrc, srcStep, pDst, dstStep, roiSize, aval);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
struct IPPGray2BGRFunctor
|
||||
{
|
||||
IPPGray2BGRFunctor(){}
|
||||
bool operator()(const void *src, int srcStep, void *dst, int dstStep, int cols, int rows) const
|
||||
{
|
||||
return ippiGrayToRGB_C1C3R((T*)src, srcStep, (T*)dst, dstStep, ippiSize(cols, rows)) >= 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct IPPGray2BGRAFunctor
|
||||
{
|
||||
IPPGray2BGRAFunctor()
|
||||
{
|
||||
alpha = ColorChannel<T>::max();
|
||||
}
|
||||
bool operator()(const void *src, int srcStep, void *dst, int dstStep, int cols, int rows) const
|
||||
{
|
||||
return ippiGrayToRGB_C1C4R((T*)src, srcStep, (T*)dst, dstStep, ippiSize(cols, rows), alpha) >= 0;
|
||||
}
|
||||
T alpha;
|
||||
};
|
||||
|
||||
static IppStatus CV_STDCALL ippiSwapChannels_8u_C3C4Rf(const Ipp8u* pSrc, int srcStep, Ipp8u* pDst, int dstStep,
|
||||
IppiSize roiSize, const int *dstOrder)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiSwapChannels_8u_C3C4R, pSrc, srcStep, pDst, dstStep, roiSize, dstOrder, MAX_IPP8u);
|
||||
}
|
||||
static IppStatus CV_STDCALL ippiSwapChannels_16u_C3C4Rf(const Ipp16u* pSrc, int srcStep, Ipp16u* pDst, int dstStep,
|
||||
IppiSize roiSize, const int *dstOrder)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiSwapChannels_16u_C3C4R, pSrc, srcStep, pDst, dstStep, roiSize, dstOrder, MAX_IPP16u);
|
||||
}
|
||||
static IppStatus CV_STDCALL ippiSwapChannels_32f_C3C4Rf(const Ipp32f* pSrc, int srcStep, Ipp32f* pDst, int dstStep,
|
||||
IppiSize roiSize, const int *dstOrder)
|
||||
{
|
||||
return CV_INSTRUMENT_FUN_IPP(ippiSwapChannels_32f_C3C4R, pSrc, srcStep, pDst, dstStep, roiSize, dstOrder, MAX_IPP32f);
|
||||
}
|
||||
|
||||
static const ippiColor2GrayFunc ippiColor2GrayC3Tab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiColor2GrayFunc)ippiColorToGray_8u_C3C1R, 0, (ippiColor2GrayFunc)ippiColorToGray_16u_C3C1R, 0,
|
||||
0, (ippiColor2GrayFunc)ippiColorToGray_32f_C3C1R, 0, 0
|
||||
};
|
||||
|
||||
static const ippiColor2GrayFunc ippiColor2GrayC4Tab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiColor2GrayFunc)ippiColorToGray_8u_AC4C1R, 0, (ippiColor2GrayFunc)ippiColorToGray_16u_AC4C1R, 0,
|
||||
0, (ippiColor2GrayFunc)ippiColorToGray_32f_AC4C1R, 0, 0
|
||||
};
|
||||
|
||||
static const ippiGeneralFunc ippiRGB2GrayC3Tab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiRGBToGray_8u_C3C1R, 0, (ippiGeneralFunc)ippiRGBToGray_16u_C3C1R, 0,
|
||||
0, (ippiGeneralFunc)ippiRGBToGray_32f_C3C1R, 0, 0
|
||||
};
|
||||
|
||||
static const ippiGeneralFunc ippiRGB2GrayC4Tab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiRGBToGray_8u_AC4C1R, 0, (ippiGeneralFunc)ippiRGBToGray_16u_AC4C1R, 0,
|
||||
0, (ippiGeneralFunc)ippiRGBToGray_32f_AC4C1R, 0, 0
|
||||
};
|
||||
|
||||
static const ippiReorderFunc ippiSwapChannelsC3C4RTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiReorderFunc)ippiSwapChannels_8u_C3C4Rf, 0, (ippiReorderFunc)ippiSwapChannels_16u_C3C4Rf, 0,
|
||||
0, (ippiReorderFunc)ippiSwapChannels_32f_C3C4Rf, 0, 0
|
||||
};
|
||||
|
||||
static const ippiGeneralFunc ippiCopyAC4C3RTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiCopy_8u_AC4C3R, 0, (ippiGeneralFunc)ippiCopy_16u_AC4C3R, 0,
|
||||
0, (ippiGeneralFunc)ippiCopy_32f_AC4C3R, 0, 0
|
||||
};
|
||||
|
||||
static const ippiReorderFunc ippiSwapChannelsC4C3RTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiReorderFunc)ippiSwapChannels_8u_C4C3R, 0, (ippiReorderFunc)ippiSwapChannels_16u_C4C3R, 0,
|
||||
0, (ippiReorderFunc)ippiSwapChannels_32f_C4C3R, 0, 0
|
||||
};
|
||||
|
||||
static const ippiReorderFunc ippiSwapChannelsC3RTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiReorderFunc)ippiSwapChannels_8u_C3R, 0, (ippiReorderFunc)ippiSwapChannels_16u_C3R, 0,
|
||||
0, (ippiReorderFunc)ippiSwapChannels_32f_C3R, 0, 0
|
||||
};
|
||||
|
||||
#if IPP_VERSION_X100 >= 810
|
||||
static const ippiReorderFunc ippiSwapChannelsC4RTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiReorderFunc)ippiSwapChannels_8u_C4R, 0, (ippiReorderFunc)ippiSwapChannels_16u_C4R, 0,
|
||||
0, (ippiReorderFunc)ippiSwapChannels_32f_C4R, 0, 0
|
||||
};
|
||||
#endif
|
||||
|
||||
#if !IPP_DISABLE_RGB_HSV
|
||||
static const ippiGeneralFunc ippiRGB2HSVTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiRGBToHSV_8u_C3R, 0, (ippiGeneralFunc)ippiRGBToHSV_16u_C3R, 0,
|
||||
0, 0, 0, 0
|
||||
};
|
||||
#endif
|
||||
|
||||
static const ippiGeneralFunc ippiHSV2RGBTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiHSVToRGB_8u_C3R, 0, (ippiGeneralFunc)ippiHSVToRGB_16u_C3R, 0,
|
||||
0, 0, 0, 0
|
||||
};
|
||||
|
||||
static const ippiGeneralFunc ippiRGB2HLSTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiRGBToHLS_8u_C3R, 0, (ippiGeneralFunc)ippiRGBToHLS_16u_C3R, 0,
|
||||
0, (ippiGeneralFunc)ippiRGBToHLS_32f_C3R, 0, 0
|
||||
};
|
||||
|
||||
static const ippiGeneralFunc ippiHLS2RGBTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiHLSToRGB_8u_C3R, 0, (ippiGeneralFunc)ippiHLSToRGB_16u_C3R, 0,
|
||||
0, (ippiGeneralFunc)ippiHLSToRGB_32f_C3R, 0, 0
|
||||
};
|
||||
|
||||
#if !IPP_DISABLE_RGB_XYZ
|
||||
static const ippiGeneralFunc ippiRGB2XYZTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiRGBToXYZ_8u_C3R, 0, (ippiGeneralFunc)ippiRGBToXYZ_16u_C3R, 0,
|
||||
0, (ippiGeneralFunc)ippiRGBToXYZ_32f_C3R, 0, 0
|
||||
};
|
||||
#endif
|
||||
|
||||
#if !IPP_DISABLE_XYZ_RGB
|
||||
static const ippiGeneralFunc ippiXYZ2RGBTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiXYZToRGB_8u_C3R, 0, (ippiGeneralFunc)ippiXYZToRGB_16u_C3R, 0,
|
||||
0, (ippiGeneralFunc)ippiXYZToRGB_32f_C3R, 0, 0
|
||||
};
|
||||
#endif
|
||||
|
||||
#if !IPP_DISABLE_RGB_LAB
|
||||
static const ippiGeneralFunc ippiRGBToLUVTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiRGBToLUV_8u_C3R, 0, (ippiGeneralFunc)ippiRGBToLUV_16u_C3R, 0,
|
||||
0, (ippiGeneralFunc)ippiRGBToLUV_32f_C3R, 0, 0
|
||||
};
|
||||
#endif
|
||||
|
||||
#if !IPP_DISABLE_LAB_RGB
|
||||
static const ippiGeneralFunc ippiLUVToRGBTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
(ippiGeneralFunc)ippiLUVToRGB_8u_C3R, 0, (ippiGeneralFunc)ippiLUVToRGB_16u_C3R, 0,
|
||||
0, (ippiGeneralFunc)ippiLUVToRGB_32f_C3R, 0, 0
|
||||
};
|
||||
#endif
|
||||
|
||||
} // namespace
|
||||
|
||||
int ipp_hal_cvtBGRtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int scn, int dcn, bool swapBlue)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
if(scn == 3 && dcn == 4 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderFunctor(ippiSwapChannelsC3C4RTab[depth], 0, 1, 2)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && dcn == 3 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiCopyAC4C3RTab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 3 && dcn == 4 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderFunctor(ippiSwapChannelsC3C4RTab[depth], 2, 1, 0)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && dcn == 3 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderFunctor(ippiSwapChannelsC4C3RTab[depth], 2, 1, 0)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 3 && dcn == 3 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, scn), dst_data, dst_step, width, height,
|
||||
IPPReorderFunctor(ippiSwapChannelsC3RTab[depth], 2, 1, 0)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
#if IPP_VERSION_X100 >= 810
|
||||
else if(scn == 4 && dcn == 4 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, scn), dst_data, dst_step, width, height,
|
||||
IPPReorderFunctor(ippiSwapChannelsC4RTab[depth], 2, 1, 0)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
#endif
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtGraytoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int dcn)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
bool ippres = false;
|
||||
if(dcn == 3)
|
||||
{
|
||||
if( depth == CV_8U )
|
||||
{
|
||||
#if !IPP_DISABLE_CVTCOLOR_GRAY2BGR_8UC3
|
||||
ippres = CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height, IPPGray2BGRFunctor<Ipp8u>());
|
||||
#endif
|
||||
}
|
||||
else if( depth == CV_16U )
|
||||
ippres = CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height, IPPGray2BGRFunctor<Ipp16u>());
|
||||
else
|
||||
ippres = CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height, IPPGray2BGRFunctor<Ipp32f>());
|
||||
}
|
||||
else if(dcn == 4)
|
||||
{
|
||||
if( depth == CV_8U )
|
||||
ippres = CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height, IPPGray2BGRAFunctor<Ipp8u>());
|
||||
else if( depth == CV_16U )
|
||||
ippres = CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height, IPPGray2BGRAFunctor<Ipp16u>());
|
||||
else
|
||||
ippres = CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height, IPPGray2BGRAFunctor<Ipp32f>());
|
||||
}
|
||||
|
||||
return ippres ? CV_HAL_ERROR_OK : CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtBGRtoGray(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int scn, bool swapBlue)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
// preserves original cvtBGRtoGray routing: only 32f was sent to IPP, other depths use the dispatch path
|
||||
if (depth != CV_32F)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(scn == 3 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPColor2GrayFunctor(ippiColor2GrayC3Tab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 3 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiRGB2GrayC3Tab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPColor2GrayFunctor(ippiColor2GrayC4Tab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiRGB2GrayC4Tab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtBGRtoHSV(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int scn, bool swapBlue, bool isFullRange, bool isHSV)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
if(depth == CV_8U && isFullRange)
|
||||
{
|
||||
if (isHSV)
|
||||
{
|
||||
#if !IPP_DISABLE_RGB_HSV
|
||||
if(scn == 3 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKE_TYPE(depth, scn), dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC3RTab[depth], ippiRGB2HSVTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth], ippiRGB2HSVTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth], ippiRGB2HSVTab[depth], 0, 1, 2, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
else
|
||||
{
|
||||
if(scn == 3 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKE_TYPE(depth, scn), dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC3RTab[depth], ippiRGB2HLSTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth], ippiRGB2HLSTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 3 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKE_TYPE(depth, scn), dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiRGB2HLSTab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth], ippiRGB2HLSTab[depth], 0, 1, 2, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtHSVtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int dcn, bool swapBlue, bool isFullRange, bool isHSV)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
if (depth == CV_8U && isFullRange)
|
||||
{
|
||||
if (isHSV)
|
||||
{
|
||||
if(dcn == 3 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, 3), dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiHSV2RGBTab[depth], ippiSwapChannelsC3RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(dcn == 4 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiHSV2RGBTab[depth], ippiSwapChannelsC3C4RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(dcn == 3 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, 3), dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiHSV2RGBTab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(dcn == 4 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiHSV2RGBTab[depth], ippiSwapChannelsC3C4RTab[depth], 0, 1, 2, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if(dcn == 3 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, 3), dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiHLS2RGBTab[depth], ippiSwapChannelsC3RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(dcn == 4 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiHLS2RGBTab[depth], ippiSwapChannelsC3C4RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(dcn == 3 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, 3), dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiHLS2RGBTab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(dcn == 4 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiHLS2RGBTab[depth], ippiSwapChannelsC3C4RTab[depth], 0, 1, 2, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtRGBAtoMultipliedRGBA(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor((ippiGeneralFunc)ippiAlphaPremul_8u_AC4R)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtBGRtoYUV(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int scn, bool swapBlue, bool isCbCr)
|
||||
{
|
||||
CV_UNUSED(src_data); CV_UNUSED(src_step); CV_UNUSED(dst_data); CV_UNUSED(dst_step); CV_UNUSED(width); CV_UNUSED(height); CV_UNUSED(depth); CV_UNUSED(scn); CV_UNUSED(swapBlue); CV_UNUSED(isCbCr);
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if !IPP_DISABLE_RGB_YUV
|
||||
if (scn == 3 && depth == CV_8U && swapBlue && !isCbCr)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor((ippiGeneralFunc)ippiRGBToYUV_8u_C3R)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 3 && depth == CV_8U && !swapBlue && !isCbCr)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC3RTab[depth],
|
||||
(ippiGeneralFunc)ippiRGBToYUV_8u_C3R, 2, 1, 0, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 4 && depth == CV_8U && swapBlue && !isCbCr)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth],
|
||||
(ippiGeneralFunc)ippiRGBToYUV_8u_C3R, 0, 1, 2, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 4 && depth == CV_8U && !swapBlue && !isCbCr)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth],
|
||||
(ippiGeneralFunc)ippiRGBToYUV_8u_C3R, 2, 1, 0, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
#endif
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtYUVtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int dcn, bool swapBlue, bool isCbCr)
|
||||
{
|
||||
CV_UNUSED(src_data); CV_UNUSED(src_step); CV_UNUSED(dst_data); CV_UNUSED(dst_step); CV_UNUSED(width); CV_UNUSED(height); CV_UNUSED(depth); CV_UNUSED(dcn); CV_UNUSED(swapBlue); CV_UNUSED(isCbCr);
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if !IPP_DISABLE_YUV_RGB
|
||||
if (dcn == 3 && depth == CV_8U && swapBlue && !isCbCr)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor((ippiGeneralFunc)ippiYUVToRGB_8u_C3R)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (dcn == 3 && depth == CV_8U && !swapBlue && !isCbCr)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor((ippiGeneralFunc)ippiYUVToRGB_8u_C3R,
|
||||
ippiSwapChannelsC3RTab[depth], 2, 1, 0, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (dcn == 4 && depth == CV_8U && swapBlue && !isCbCr)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor((ippiGeneralFunc)ippiYUVToRGB_8u_C3R,
|
||||
ippiSwapChannelsC3C4RTab[depth], 0, 1, 2, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (dcn == 4 && depth == CV_8U && !swapBlue && !isCbCr)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor((ippiGeneralFunc)ippiYUVToRGB_8u_C3R,
|
||||
ippiSwapChannelsC3C4RTab[depth], 2, 1, 0, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
#endif
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtBGRtoXYZ(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int scn, bool swapBlue)
|
||||
{
|
||||
CV_UNUSED(src_data); CV_UNUSED(src_step); CV_UNUSED(dst_data); CV_UNUSED(dst_step); CV_UNUSED(width); CV_UNUSED(height); CV_UNUSED(depth); CV_UNUSED(scn); CV_UNUSED(swapBlue);
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if !IPP_DISABLE_RGB_XYZ
|
||||
if(scn == 3 && depth != CV_32F && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, scn), dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC3RTab[depth], ippiRGB2XYZTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && depth != CV_32F && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth], ippiRGB2XYZTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 3 && depth != CV_32F && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, scn), dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiRGB2XYZTab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(scn == 4 && depth != CV_32F && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth], ippiRGB2XYZTab[depth], 0, 1, 2, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
#endif
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtXYZtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int dcn, bool swapBlue)
|
||||
{
|
||||
CV_UNUSED(src_data); CV_UNUSED(src_step); CV_UNUSED(dst_data); CV_UNUSED(dst_step); CV_UNUSED(width); CV_UNUSED(height); CV_UNUSED(depth); CV_UNUSED(dcn); CV_UNUSED(swapBlue);
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if !IPP_DISABLE_XYZ_RGB
|
||||
if(dcn == 3 && depth != CV_32F && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, 3), dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiXYZ2RGBTab[depth], ippiSwapChannelsC3RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(dcn == 4 && depth != CV_32F && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiXYZ2RGBTab[depth], ippiSwapChannelsC3C4RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
if(dcn == 3 && depth != CV_32F && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoopCopy(src_data, src_step, CV_MAKETYPE(depth, 3), dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiXYZ2RGBTab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(dcn == 4 && depth != CV_32F && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiXYZ2RGBTab[depth], ippiSwapChannelsC3C4RTab[depth], 0, 1, 2, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
#endif
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtBGRtoLab(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int scn, bool swapBlue, bool isLab, bool srgb)
|
||||
{
|
||||
CV_UNUSED(src_data); CV_UNUSED(src_step); CV_UNUSED(dst_data); CV_UNUSED(dst_step); CV_UNUSED(width); CV_UNUSED(height); CV_UNUSED(depth); CV_UNUSED(scn); CV_UNUSED(swapBlue); CV_UNUSED(isLab); CV_UNUSED(srgb);
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if !IPP_DISABLE_RGB_LAB
|
||||
if (!srgb)
|
||||
{
|
||||
if (isLab)
|
||||
{
|
||||
if (scn == 3 && depth == CV_8U && !swapBlue)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor((ippiGeneralFunc)ippiBGRToLab_8u_C3R)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 4 && depth == CV_8U && !swapBlue)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth],
|
||||
(ippiGeneralFunc)ippiBGRToLab_8u_C3R, 0, 1, 2, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 3 && depth == CV_8U && swapBlue) // slower than OpenCV
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC3RTab[depth],
|
||||
(ippiGeneralFunc)ippiBGRToLab_8u_C3R, 2, 1, 0, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 4 && depth == CV_8U && swapBlue) // slower than OpenCV
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth],
|
||||
(ippiGeneralFunc)ippiBGRToLab_8u_C3R, 2, 1, 0, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if (scn == 3 && swapBlue)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiRGBToLUVTab[depth])))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 4 && swapBlue)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth],
|
||||
ippiRGBToLUVTab[depth], 0, 1, 2, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 3 && !swapBlue)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC3RTab[depth],
|
||||
ippiRGBToLUVTab[depth], 2, 1, 0, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if (scn == 4 && !swapBlue)
|
||||
{
|
||||
if (CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPReorderGeneralFunctor(ippiSwapChannelsC4C3RTab[depth],
|
||||
ippiRGBToLUVTab[depth], 2, 1, 0, depth)))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_cvtLabtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int dcn, bool swapBlue, bool isLab, bool srgb)
|
||||
{
|
||||
CV_UNUSED(src_data); CV_UNUSED(src_step); CV_UNUSED(dst_data); CV_UNUSED(dst_step); CV_UNUSED(width); CV_UNUSED(height); CV_UNUSED(depth); CV_UNUSED(dcn); CV_UNUSED(swapBlue); CV_UNUSED(isLab); CV_UNUSED(srgb);
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
#if !IPP_DISABLE_LAB_RGB
|
||||
if (!srgb)
|
||||
{
|
||||
if (isLab)
|
||||
{
|
||||
if( dcn == 3 && depth == CV_8U && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor((ippiGeneralFunc)ippiLabToBGR_8u_C3R)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if( dcn == 4 && depth == CV_8U && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor((ippiGeneralFunc)ippiLabToBGR_8u_C3R,
|
||||
ippiSwapChannelsC3C4RTab[depth], 0, 1, 2, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
if( dcn == 3 && depth == CV_8U && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor((ippiGeneralFunc)ippiLabToBGR_8u_C3R,
|
||||
ippiSwapChannelsC3RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if( dcn == 4 && depth == CV_8U && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor((ippiGeneralFunc)ippiLabToBGR_8u_C3R,
|
||||
ippiSwapChannelsC3C4RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if( dcn == 3 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralFunctor(ippiLUVToRGBTab[depth])) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if( dcn == 4 && swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiLUVToRGBTab[depth],
|
||||
ippiSwapChannelsC3C4RTab[depth], 0, 1, 2, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
if( dcn == 3 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiLUVToRGBTab[depth],
|
||||
ippiSwapChannelsC3RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if( dcn == 4 && !swapBlue)
|
||||
{
|
||||
if( CvtColorIPPLoop(src_data, src_step, dst_data, dst_step, width, height,
|
||||
IPPGeneralReorderFunctor(ippiLUVToRGBTab[depth],
|
||||
ippiSwapChannelsC3C4RTab[depth], 2, 1, 0, depth)) )
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
#endif // IPP_VERSION_X100 >= 700
|
||||
@@ -0,0 +1,247 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, Intel Corporation, all rights reserved.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include <opencv2/core/base.hpp>
|
||||
|
||||
#if IPP_VERSION_X100 >= 810
|
||||
|
||||
#ifdef HAVE_IPP_IW
|
||||
#include "iw++/iw.hpp"
|
||||
|
||||
static inline IppiMaskSize ippiGetMaskSize(int kx, int ky)
|
||||
{
|
||||
return (kx == 1 && ky == 3) ? ippMskSize1x3 :
|
||||
(kx == 1 && ky == 5) ? ippMskSize1x5 :
|
||||
(kx == 3 && ky == 1) ? ippMskSize3x1 :
|
||||
(kx == 3 && ky == 3) ? ippMskSize3x3 :
|
||||
(kx == 5 && ky == 1) ? ippMskSize5x1 :
|
||||
(kx == 5 && ky == 5) ? ippMskSize5x5 :
|
||||
(IppiMaskSize)-1;
|
||||
}
|
||||
|
||||
static inline IwiDerivativeType ippiGetDerivType(int dx, int dy, bool nvert)
|
||||
{
|
||||
return (dx == 1 && dy == 0) ? ((nvert)?iwiDerivNVerFirst:iwiDerivVerFirst) :
|
||||
(dx == 0 && dy == 1) ? iwiDerivHorFirst :
|
||||
(dx == 2 && dy == 0) ? iwiDerivVerSecond :
|
||||
(dx == 0 && dy == 2) ? iwiDerivHorSecond :
|
||||
(IwiDerivativeType)-1;
|
||||
}
|
||||
|
||||
// Build an IwiImage from a raw buffer, encoding the in-memory border (the OpenCV
|
||||
// margins around the ROI) so that BORDER_*-with-physical-pixels works as in core.
|
||||
// TODO: promote to precomp_ipp.hpp and unify with the raw-pointer ippiGetImage in
|
||||
// transforms_ipp.cpp once more filter HALs share it
|
||||
static inline ::ipp::IwiImage ippiGetImage(int depth, int channels, const uchar* data, size_t step,
|
||||
int width, int height,
|
||||
int margin_left, int margin_top, int margin_right, int margin_bottom)
|
||||
{
|
||||
::ipp::IwiImage image;
|
||||
::ipp::IwiBorderSize inMemBorder((IwSize)margin_left, (IwSize)margin_top, (IwSize)margin_right, (IwSize)margin_bottom);
|
||||
image.Init({width, height}, ippiGetDataType(depth), channels, inMemBorder, (void*)data, step);
|
||||
return image;
|
||||
}
|
||||
|
||||
// Translate the OpenCV border type into an IPP border, accounting for in-memory pixels.
|
||||
// Returns (IppiBorderType)0 on unsupported configuration.
|
||||
// TODO: promote to precomp_ipp.hpp once shared by other filter HALs (box/gaussian/bilateral/
|
||||
// sepFilter etc.). Note the shared ippiGetBorderType there does not map BORDER_REFLECT_101;
|
||||
// widening it would also affect warp_ipp.cpp, so keep this filter-specific mapping separate
|
||||
// (or add a dedicated filter border helper) when consolidating.
|
||||
static inline IppiBorderType ippiGetBorder(::ipp::IwiImage &image, int ocvBorderType, ::ipp::IwiBorderSize &borderSize)
|
||||
{
|
||||
int inMemFlags = 0;
|
||||
int borderTypeNI = ocvBorderType & ~cv::BORDER_ISOLATED;
|
||||
IppiBorderType border = borderTypeNI == cv::BORDER_CONSTANT ? ippBorderConst :
|
||||
borderTypeNI == cv::BORDER_TRANSPARENT ? ippBorderTransp :
|
||||
borderTypeNI == cv::BORDER_REPLICATE ? ippBorderRepl :
|
||||
borderTypeNI == cv::BORDER_REFLECT_101 ? ippBorderMirror :
|
||||
(IppiBorderType)-1;
|
||||
if((int)border == -1)
|
||||
return (IppiBorderType)0;
|
||||
|
||||
if(!(ocvBorderType & cv::BORDER_ISOLATED))
|
||||
{
|
||||
if(image.m_inMemSize.left)
|
||||
{
|
||||
if(image.m_inMemSize.left >= borderSize.left)
|
||||
inMemFlags |= ippBorderInMemLeft;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.left = 0;
|
||||
if(image.m_inMemSize.top)
|
||||
{
|
||||
if(image.m_inMemSize.top >= borderSize.top)
|
||||
inMemFlags |= ippBorderInMemTop;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.top = 0;
|
||||
if(image.m_inMemSize.right)
|
||||
{
|
||||
if(image.m_inMemSize.right >= borderSize.right)
|
||||
inMemFlags |= ippBorderInMemRight;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.right = 0;
|
||||
if(image.m_inMemSize.bottom)
|
||||
{
|
||||
if(image.m_inMemSize.bottom >= borderSize.bottom)
|
||||
inMemFlags |= ippBorderInMemBottom;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.bottom = 0;
|
||||
}
|
||||
else
|
||||
borderSize.left = borderSize.right = borderSize.top = borderSize.bottom = 0;
|
||||
|
||||
return (IppiBorderType)(border|inMemFlags);
|
||||
}
|
||||
|
||||
// Shared worker for Sobel (useScharr == false) and Scharr (useScharr == true).
|
||||
static int ipp_Deriv(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int src_depth, int dst_depth, int cn,
|
||||
int margin_left, int margin_top, int margin_right, int margin_bottom,
|
||||
int dx, int dy, int ksize, double scale, double delta, int borderType,
|
||||
bool useScharr)
|
||||
{
|
||||
IppDataType srcType = ippiGetDataType(src_depth);
|
||||
IppDataType dstType = ippiGetDataType(dst_depth);
|
||||
bool useScale = false;
|
||||
|
||||
if(cn < 1 || cn > 4)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
if(src_depth < 0 || src_depth >= CV_DEPTH_MAX || dst_depth < 0 || dst_depth >= CV_DEPTH_MAX)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// Supported (source depth -> destination depth) x channels combinations.
|
||||
// Index order: [src_depth][dst_depth][channel-1]; depths are 8U,8S,16U,16S,32S,32F,64F.
|
||||
// IPP iwiFilterSobel/iwiFilterScharr provide only single-channel kernels for
|
||||
// 8u->16s, 16s->16s and 32f->32f; 8u->8u is realized as an 8u->16s filter plus a 16s->8u scale.
|
||||
// 8u->32f and 16s->32f are intentionally left disabled: IPP would need an extra full-image
|
||||
// conversion pass there and is slower than OpenCV's fused sepFilter2D.
|
||||
/* dst: 8U 8S 16U 16S 32S 32F 64F */
|
||||
#if defined(IPP_CALLS_ENFORCED)
|
||||
const char impl[CV_DEPTH_MAX][CV_DEPTH_MAX][4] = {
|
||||
/* src 8U */ {{1,0,0,0},{0,0,0,0},{0,0,0,0},{1,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 8S */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 16U */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 16S */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{1,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 32S */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 32F */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{1,0,0,0},{0,0,0,0}},
|
||||
/* src 64F */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}}};
|
||||
#else
|
||||
const char impl[CV_DEPTH_MAX][CV_DEPTH_MAX][4] = {
|
||||
/* src 8U */ {{1,0,0,0},{0,0,0,0},{0,0,0,0},{1,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 8S */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 16U */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 16S */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{1,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 32S */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}},
|
||||
/* src 32F */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{1,0,0,0},{0,0,0,0}},
|
||||
/* src 64F */ {{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0},{0,0,0,0}}};
|
||||
#endif // IPP_CALLS_ENFORCED
|
||||
if(impl[src_depth][dst_depth][cn-1] == 0)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(fabs(delta) > FLT_EPSILON || fabs(scale-1) > FLT_EPSILON)
|
||||
useScale = true;
|
||||
|
||||
// cv::Sobel accepts ksize == FILTER_SCHARR (-1, i.e. ksize <= 0) which selects the 3x3
|
||||
// Scharr derivative, so the Sobel entry point may legitimately request a Scharr filter.
|
||||
if(ksize <= 0)
|
||||
{
|
||||
ksize = 3;
|
||||
useScharr = true;
|
||||
}
|
||||
|
||||
IppiMaskSize maskSize = ippiGetMaskSize(ksize, ksize);
|
||||
if((int)maskSize < 0)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
#if IPP_VERSION_X100 <= 201703
|
||||
// Bug with mirror wrap
|
||||
if(borderType == cv::BORDER_REFLECT_101 && (ksize/2+1 > width || ksize/2+1 > height))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
#endif
|
||||
|
||||
IwiDerivativeType derivType = ippiGetDerivType(dx, dy, (useScharr)?false:true);
|
||||
if((int)derivType < 0)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
try
|
||||
{
|
||||
::ipp::IwiImage iwSrc = ippiGetImage(src_depth, cn, src_data, src_step, width, height,
|
||||
margin_left, margin_top, margin_right, margin_bottom);
|
||||
::ipp::IwiImage iwDst = ippiGetImage(dst_depth, cn, dst_data, dst_step, width, height,
|
||||
0, 0, 0, 0);
|
||||
::ipp::IwiImage iwSrcProc = iwSrc;
|
||||
::ipp::IwiImage iwDstProc = iwDst;
|
||||
::ipp::IwiBorderSize borderSize(maskSize);
|
||||
::ipp::IwiBorderType ippBorder(ippiGetBorder(iwSrc, borderType, borderSize));
|
||||
if(!ippBorder)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// IPP needs an extra iwiScale pass for 32f output with scale/delta, slower than OpenCV's fused approach
|
||||
if(useScale && dstType == ipp32f)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(srcType == ipp8u && dstType == ipp8u)
|
||||
{
|
||||
iwDstProc.Alloc(iwDst.m_size, ipp16s, cn);
|
||||
useScale = true;
|
||||
}
|
||||
|
||||
if(useScharr)
|
||||
CV_INSTRUMENT_FUN_IPP(::ipp::iwiFilterScharr, iwSrcProc, iwDstProc, derivType, maskSize, ::ipp::IwDefault(), ippBorder);
|
||||
else
|
||||
CV_INSTRUMENT_FUN_IPP(::ipp::iwiFilterSobel, iwSrcProc, iwDstProc, derivType, maskSize, ::ipp::IwDefault(), ippBorder);
|
||||
|
||||
if(useScale)
|
||||
CV_INSTRUMENT_FUN_IPP(::ipp::iwiScale, iwDstProc, iwDst, scale, delta, ::ipp::IwiScaleParams(ippAlgHintFast));
|
||||
}
|
||||
catch (const ::ipp::IwException &)
|
||||
{
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
int ipp_hal_sobel(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int src_depth, int dst_depth, int cn,
|
||||
int margin_left, int margin_top, int margin_right, int margin_bottom,
|
||||
int dx, int dy, int ksize, double scale, double delta, int border_type)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
return ipp_Deriv(src_data, src_step, dst_data, dst_step, width, height, src_depth, dst_depth, cn,
|
||||
margin_left, margin_top, margin_right, margin_bottom,
|
||||
dx, dy, ksize, scale, delta, border_type, false);
|
||||
}
|
||||
|
||||
int ipp_hal_scharr(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int src_depth, int dst_depth, int cn,
|
||||
int margin_left, int margin_top, int margin_right, int margin_bottom,
|
||||
int dx, int dy, double scale, double delta, int border_type)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
return ipp_Deriv(src_data, src_step, dst_data, dst_step, width, height, src_depth, dst_depth, cn,
|
||||
margin_left, margin_top, margin_right, margin_bottom,
|
||||
dx, dy, 0, scale, delta, border_type, true);
|
||||
}
|
||||
|
||||
#endif // HAVE_IPP_IW
|
||||
|
||||
#endif // IPP_VERSION_X100 >= 810
|
||||
@@ -0,0 +1,148 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, BigVision LLC, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include <opencv2/imgproc.hpp>
|
||||
|
||||
#include <climits>
|
||||
#include <cmath>
|
||||
|
||||
#if IPP_VERSION_X100 >= 700
|
||||
|
||||
// Mirrors CV_IPP_MALLOC from modules/core/include/opencv2/core/private.hpp: ippicv (IPP >= 2017)
|
||||
// exposes only the 64-bit ippMalloc_L.
|
||||
#if IPP_VERSION_X100 >= 201700
|
||||
#define IPP_HAL_MALLOC(SIZE) ippMalloc_L(SIZE)
|
||||
#else
|
||||
#define IPP_HAL_MALLOC(SIZE) ippMalloc((int)(SIZE))
|
||||
#endif
|
||||
|
||||
#ifndef IPP_DISABLE_PERF_TRUE_DIST_MT
|
||||
#define IPP_DISABLE_PERF_TRUE_DIST_MT 1
|
||||
#endif
|
||||
|
||||
// Fixed distance-transform mask weights, mirroring getDistanceTransformMask() in
|
||||
// modules/imgproc/src/distransform.cpp. maskType == distTypeIndex + mask_size*10.
|
||||
static bool ippGetDistanceTransformMask(int dist_type, int mask_size, float *metrics)
|
||||
{
|
||||
int maskType = (dist_type == cv::DIST_C ? 0 : dist_type == cv::DIST_L1 ? 1 : 2) + mask_size * 10;
|
||||
switch (maskType)
|
||||
{
|
||||
case 30: metrics[0] = 1.0f; metrics[1] = 1.0f; break;
|
||||
case 31: metrics[0] = 1.0f; metrics[1] = 2.0f; break;
|
||||
case 32: metrics[0] = 0.955f; metrics[1] = 1.3693f; break;
|
||||
case 50: metrics[0] = 1.0f; metrics[1] = 1.0f; metrics[2] = 2.0f; break;
|
||||
case 51: metrics[0] = 1.0f; metrics[1] = 2.0f; metrics[2] = 3.0f; break;
|
||||
case 52: metrics[0] = 1.0f; metrics[1] = 1.4f; metrics[2] = 2.1969f; break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
int ipp_hal_distanceTransform(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int dst_type, int dist_type, int mask_size)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
IppiSize roi = { width, height };
|
||||
|
||||
// DIST_L1 with 8-bit output: fixed L1 / 3x3 metrics.
|
||||
if (dst_type == CV_8U)
|
||||
{
|
||||
Ipp32s pMetrics[2] = { 1, 2 }; // L1, 3x3 mask
|
||||
if (CV_INSTRUMENT_FUN_IPP(ippiDistanceTransform_3x3_8u_C1R, src_data, (int)src_step,
|
||||
dst_data, (int)dst_step, roi, pMetrics) >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
// All remaining paths produce a 32-bit float distance map.
|
||||
if (dst_type != CV_32F)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if (mask_size == cv::DIST_MASK_PRECISE)
|
||||
{
|
||||
// 4097 can't square into a float.
|
||||
#if IPP_DISABLE_PERF_TRUE_DIST_MT
|
||||
if (!((cv::getNumThreads() <= 1 || ((size_t)width * height < (size_t)(1 << 14))) &&
|
||||
height < 4097 && width < 4097))
|
||||
{
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
#endif
|
||||
|
||||
int bufSize = 0;
|
||||
if (ippiTrueDistanceTransformGetBufferSize_8u32f_C1R(roi, &bufSize) < 0)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
Ipp8u *pBuffer = (Ipp8u *)IPP_HAL_MALLOC(bufSize);
|
||||
if (!pBuffer)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
IppStatus status = CV_INSTRUMENT_FUN_IPP(ippiTrueDistanceTransform_8u32f_C1R, src_data, (int)src_step,
|
||||
(Ipp32f *)dst_data, (int)dst_step, roi, pBuffer);
|
||||
ippFree(pBuffer);
|
||||
if (status < 0)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// https://github.com/opencv/opencv/issues/24082
|
||||
// There is probably a rounding issue that leads to non-deterministic behavior
|
||||
// between runs on positions closer to zeros by x-axis in straight direction.
|
||||
// As a workaround, we detect the distances that expected to be exact
|
||||
// number of pixels and round manually.
|
||||
static const float correctionDiff = 1.0f / (1 << 11);
|
||||
for (int i = 0; i < height; ++i)
|
||||
{
|
||||
float* row = (float*)(dst_data + (size_t)i * dst_step);
|
||||
for (int j = 0; j < width; ++j)
|
||||
{
|
||||
float rounded = static_cast<float>(cvRound(row[j]));
|
||||
if (std::fabs(row[j] - rounded) <= correctionDiff)
|
||||
row[j] = rounded;
|
||||
}
|
||||
}
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
// DIST_MASK_3 / DIST_MASK_5 chamfer distances.
|
||||
if (mask_size != cv::DIST_MASK_3 && mask_size != cv::DIST_MASK_5)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// IPP uses 32-bit signed intermediates; bail out when the image is too large.
|
||||
bool has_int_overflow = (int64)width * height >= INT_MAX;
|
||||
if (has_int_overflow)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
float metrics[5] = {0};
|
||||
if (!ippGetDistanceTransformMask(dist_type, mask_size, metrics))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if (mask_size == cv::DIST_MASK_3)
|
||||
{
|
||||
if (CV_INSTRUMENT_FUN_IPP(ippiDistanceTransform_3x3_8u32f_C1R, src_data, (int)src_step,
|
||||
(Ipp32f *)dst_data, (int)dst_step, roi, metrics) >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
else // DIST_MASK_5
|
||||
{
|
||||
if (CV_INSTRUMENT_FUN_IPP(ippiDistanceTransform_5x5_8u32f_C1R, src_data, (int)src_step,
|
||||
(Ipp32f *)dst_data, (int)dst_step, roi, metrics) >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
#endif // IPP_VERSION_X100 >= 700
|
||||
@@ -0,0 +1,157 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, BigVision LLC, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#include <cmath>
|
||||
#include <cfloat>
|
||||
|
||||
#if defined(HAVE_IPP_IW) && !DISABLE_IPP_FILTER2D
|
||||
|
||||
// Copied from core/private.hpp (gated by HAVE_IPP, which the plugin lacks).
|
||||
// filterGetIppBorderType: distinct name; adds the BORDER_REFLECT_101 -> ippBorderMirror
|
||||
// case that precomp_ipp.hpp's ippiGetBorderType omits.
|
||||
static inline IppiBorderType filterGetIppBorderType(int borderTypeNI)
|
||||
{
|
||||
return borderTypeNI == cv::BORDER_CONSTANT ? ippBorderConst :
|
||||
borderTypeNI == cv::BORDER_TRANSPARENT ? ippBorderTransp :
|
||||
borderTypeNI == cv::BORDER_REPLICATE ? ippBorderRepl :
|
||||
borderTypeNI == cv::BORDER_REFLECT_101 ? ippBorderMirror :
|
||||
(IppiBorderType)-1;
|
||||
}
|
||||
|
||||
static inline bool ippiCheckAnchor(int x, int y, int kernelWidth, int kernelHeight)
|
||||
{
|
||||
return (x == (kernelWidth - 1)/2 && y == (kernelHeight - 1)/2);
|
||||
}
|
||||
|
||||
static inline IppiBorderType ippiGetBorder(::ipp::IwiImage &image, int ocvBorderType, ::ipp::IwiBorderSize &borderSize)
|
||||
{
|
||||
int inMemFlags = 0;
|
||||
IppiBorderType border = filterGetIppBorderType(ocvBorderType & ~cv::BORDER_ISOLATED);
|
||||
if((int)border == -1)
|
||||
return (IppiBorderType)0;
|
||||
|
||||
if(!(ocvBorderType & cv::BORDER_ISOLATED))
|
||||
{
|
||||
if(image.m_inMemSize.left)
|
||||
{
|
||||
if(image.m_inMemSize.left >= borderSize.left)
|
||||
inMemFlags |= ippBorderInMemLeft;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.left = 0;
|
||||
if(image.m_inMemSize.top)
|
||||
{
|
||||
if(image.m_inMemSize.top >= borderSize.top)
|
||||
inMemFlags |= ippBorderInMemTop;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.top = 0;
|
||||
if(image.m_inMemSize.right)
|
||||
{
|
||||
if(image.m_inMemSize.right >= borderSize.right)
|
||||
inMemFlags |= ippBorderInMemRight;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.right = 0;
|
||||
if(image.m_inMemSize.bottom)
|
||||
{
|
||||
if(image.m_inMemSize.bottom >= borderSize.bottom)
|
||||
inMemFlags |= ippBorderInMemBottom;
|
||||
else
|
||||
return (IppiBorderType)0;
|
||||
}
|
||||
else
|
||||
borderSize.bottom = 0;
|
||||
}
|
||||
else
|
||||
borderSize.left = borderSize.right = borderSize.top = borderSize.bottom = 0;
|
||||
|
||||
return (IppiBorderType)(border | inMemFlags);
|
||||
}
|
||||
|
||||
int ipp_hal_filter2D(const uchar * src_data, size_t src_step, int src_type,
|
||||
uchar * dst_data, size_t dst_step, int dst_type,
|
||||
int width, int height, int full_width, int full_height,
|
||||
int offset_x, int offset_y,
|
||||
const uchar * kernel_data, size_t kernel_step, int kernel_type,
|
||||
int kernel_width, int kernel_height,
|
||||
int anchor_x, int anchor_y, double delta, int borderType,
|
||||
bool isSubmatrix, bool allowInplace)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
IppDataType type = ippiGetDataType(CV_MAT_DEPTH(src_type));
|
||||
int channels = CV_MAT_CN(src_type);
|
||||
|
||||
CV_UNUSED(isSubmatrix);
|
||||
CV_UNUSED(allowInplace);
|
||||
|
||||
#if IPP_VERSION_X100 >= 201700 && IPP_VERSION_X100 <= 201702 // IPP bug with 1x1 kernel
|
||||
if(kernel_width == 1 && kernel_height == 1)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
#endif
|
||||
|
||||
#if IPP_DISABLE_FILTER2D_BIG_MASK
|
||||
// Too big difference compared to OpenCV FFT-based convolution
|
||||
if(kernel_type == CV_32FC1 && (type == ipp16s || type == ipp16u) && (kernel_width > 7 || kernel_height > 7))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// Poor optimization for big kernels
|
||||
if(kernel_width > 7 || kernel_height > 7)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
#endif
|
||||
|
||||
if(src_data == dst_data)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(src_type != dst_type)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(kernel_type != CV_16SC1 && kernel_type != CV_32FC1)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// TODO: Implement offset for 8u, 16u
|
||||
if(std::fabs(delta) >= DBL_EPSILON)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(!ippiCheckAnchor(anchor_x, anchor_y, kernel_width, kernel_height))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
try
|
||||
{
|
||||
::ipp::IwiSize iwSize(width, height);
|
||||
::ipp::IwiSize kernelSize(kernel_width, kernel_height);
|
||||
::ipp::IwiImage iwKernel(ippiSize(kernel_width, kernel_height), ippiGetDataType(CV_MAT_DEPTH(kernel_type)), CV_MAT_CN(kernel_type), 0, (void*)kernel_data, kernel_step);
|
||||
::ipp::IwiImage iwSrc(iwSize, type, channels, ::ipp::IwiBorderSize(offset_x, offset_y, full_width-offset_x-width, full_height-offset_y-height), (void*)src_data, src_step);
|
||||
::ipp::IwiImage iwDst(iwSize, type, channels, ::ipp::IwiBorderSize(offset_x, offset_y, full_width-offset_x-width, full_height-offset_y-height), (void*)dst_data, dst_step);
|
||||
|
||||
::ipp::IwiBorderSize iwBorderSize = ::ipp::iwiSizeToBorderSize(kernelSize);
|
||||
::ipp::IwiBorderType iwBorderType = ippiGetBorder(iwSrc, borderType, iwBorderSize);
|
||||
if(!iwBorderType)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
CV_INSTRUMENT_FUN_IPP(::ipp::iwiFilter, iwSrc, iwDst, iwKernel, ::ipp::IwiFilterParams(1, 0, ippAlgHintNone, ippRndFinancial), iwBorderType);
|
||||
}
|
||||
catch (const ::ipp::IwException &)
|
||||
{
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
#endif // defined(HAVE_IPP_IW) && !DISABLE_IPP_FILTER2D
|
||||
@@ -0,0 +1,223 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
// Copyright (C) 2026, BigVision LLC, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include <opencv2/core/utils/tls.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#if IPP_VERSION_X100 >= 700
|
||||
|
||||
#define IPP_HISTOGRAM_PARALLEL 1
|
||||
|
||||
namespace cv { namespace ipp { unsigned long long getIppTopFeatures(); } }
|
||||
|
||||
using namespace cv;
|
||||
|
||||
namespace {
|
||||
|
||||
typedef IppStatus(CV_STDCALL * IppiHistogram_C1)(const void* pSrc, int srcStep,
|
||||
IppiSize roiSize, Ipp32u* pHist, const IppiHistogramSpec* pSpec, Ipp8u* pBuffer);
|
||||
|
||||
static IppiHistogram_C1 getIppiHistogramFunction_C1(int type)
|
||||
{
|
||||
IppiHistogram_C1 ippFunction =
|
||||
(type == CV_8UC1) ? (IppiHistogram_C1)ippiHistogram_8u_C1R :
|
||||
(type == CV_16UC1) ? (IppiHistogram_C1)ippiHistogram_16u_C1R :
|
||||
(type == CV_32FC1) ? (IppiHistogram_C1)ippiHistogram_32f_C1R :
|
||||
NULL;
|
||||
|
||||
return ippFunction;
|
||||
}
|
||||
|
||||
class ipp_calcHistParallelTLS
|
||||
{
|
||||
public:
|
||||
ipp_calcHistParallelTLS() {}
|
||||
|
||||
IppAutoBuffer<IppiHistogramSpec> spec;
|
||||
IppAutoBuffer<Ipp8u> buffer;
|
||||
IppAutoBuffer<Ipp32u> thist;
|
||||
};
|
||||
|
||||
class ipp_calcHistParallel: public ParallelLoopBody
|
||||
{
|
||||
public:
|
||||
ipp_calcHistParallel(const Mat &src, Mat &hist, Ipp32s histSize, const float *ranges, bool uniform, bool &ok):
|
||||
ParallelLoopBody(), m_src(src), m_hist(hist), m_ok(ok)
|
||||
{
|
||||
ok = true;
|
||||
|
||||
m_uniform = uniform;
|
||||
m_ranges = ranges;
|
||||
m_histSize = histSize;
|
||||
m_type = ippiGetDataType(src.type());
|
||||
m_levelsNum = histSize+1;
|
||||
ippiHistogram_C1 = getIppiHistogramFunction_C1(src.type());
|
||||
m_fullRoi = ippiSize(src.size());
|
||||
m_bufferSize = 0;
|
||||
m_specSize = 0;
|
||||
if(!ippiHistogram_C1)
|
||||
{
|
||||
ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(ippiHistogramGetBufferSize(m_type, m_fullRoi, &m_levelsNum, 1, 1, &m_specSize, &m_bufferSize) < 0)
|
||||
{
|
||||
ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
hist.setTo(0);
|
||||
}
|
||||
|
||||
virtual void operator() (const Range & range) const CV_OVERRIDE
|
||||
{
|
||||
if(!m_ok)
|
||||
return;
|
||||
|
||||
ipp_calcHistParallelTLS *pTls = m_tls.get();
|
||||
|
||||
IppiSize roi = {m_src.cols, range.end - range.start };
|
||||
bool mtLoop = false;
|
||||
if(m_fullRoi.height != roi.height)
|
||||
mtLoop = true;
|
||||
|
||||
if(!pTls->spec)
|
||||
{
|
||||
pTls->spec.allocate(m_specSize);
|
||||
if(!pTls->spec.get())
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
pTls->buffer.allocate(m_bufferSize);
|
||||
if(!pTls->buffer.get() && m_bufferSize)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(m_uniform)
|
||||
{
|
||||
if(ippiHistogramUniformInit(m_type, (Ipp32f*)&m_ranges[0], (Ipp32f*)&m_ranges[1], (Ipp32s*)&m_levelsNum, 1, pTls->spec) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
if(ippiHistogramInit(m_type, (const Ipp32f**)&m_ranges, (Ipp32s*)&m_levelsNum, 1, pTls->spec) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
pTls->thist.allocate(m_histSize*sizeof(Ipp32u));
|
||||
}
|
||||
|
||||
if(CV_INSTRUMENT_FUN_IPP(ippiHistogram_C1, m_src.ptr(range.start), (int)m_src.step, roi, pTls->thist, pTls->spec, pTls->buffer) < 0)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(mtLoop)
|
||||
{
|
||||
for(int i = 0; i < m_histSize; i++)
|
||||
CV_XADD((int*)(m_hist.ptr(i)), *(int*)((Ipp32u*)pTls->thist + i));
|
||||
}
|
||||
else
|
||||
ippiCopy_32s_C1R((Ipp32s*)pTls->thist.get(), sizeof(Ipp32u), (Ipp32s*)m_hist.ptr(), (int)m_hist.step, ippiSize(1, m_histSize));
|
||||
}
|
||||
|
||||
private:
|
||||
const Mat &m_src;
|
||||
Mat &m_hist;
|
||||
Ipp32s m_histSize;
|
||||
const float *m_ranges;
|
||||
bool m_uniform;
|
||||
|
||||
IppiHistogram_C1 ippiHistogram_C1;
|
||||
IppiSize m_fullRoi;
|
||||
IppDataType m_type;
|
||||
Ipp32s m_levelsNum;
|
||||
int m_bufferSize;
|
||||
int m_specSize;
|
||||
|
||||
mutable Mutex m_syncMutex;
|
||||
TLSData<ipp_calcHistParallelTLS> m_tls;
|
||||
|
||||
volatile bool &m_ok;
|
||||
const ipp_calcHistParallel & operator = (const ipp_calcHistParallel & );
|
||||
};
|
||||
|
||||
static bool ipp_calchist(const Mat &image, Mat &hist, int histSize, const float** ranges, bool uniform, bool accumulate)
|
||||
{
|
||||
#if IPP_VERSION_X100 < 201801
|
||||
// No SSE42 optimization for uniform 32f
|
||||
if(uniform && image.depth() == CV_32F && cv::ipp::getIppTopFeatures() == ippCPUID_SSE42)
|
||||
return false;
|
||||
#endif
|
||||
|
||||
// IPP_DISABLE_HISTOGRAM - https://github.com/opencv/opencv/issues/11544
|
||||
// and https://github.com/opencv/opencv/issues/21595
|
||||
if ((uniform && (ranges[0][1] - ranges[0][0]) != histSize) || abs(ranges[0][0]) != cvFloor(ranges[0][0]))
|
||||
return false;
|
||||
|
||||
Mat ihist = hist;
|
||||
if(accumulate)
|
||||
ihist.create(1, &histSize, CV_32S);
|
||||
|
||||
bool ok = true;
|
||||
int threads = ippiSuggestThreadsNum(image.cols, image.rows, image.elemSize(), (1+((double)ihist.total()/image.total()))*2);
|
||||
Range range(0, image.rows);
|
||||
ipp_calcHistParallel invoker(image, ihist, histSize, ranges[0], uniform, ok);
|
||||
if(!ok)
|
||||
return false;
|
||||
|
||||
if(IPP_HISTOGRAM_PARALLEL && threads > 1)
|
||||
parallel_for_(range, invoker, threads*2);
|
||||
else
|
||||
invoker(range);
|
||||
|
||||
if(ok)
|
||||
{
|
||||
if(accumulate)
|
||||
{
|
||||
IppiSize histRoi = ippiSize(1, histSize);
|
||||
IppAutoBuffer<Ipp32f> fhist(histSize*sizeof(Ipp32f));
|
||||
CV_INSTRUMENT_FUN_IPP(ippiConvert_32s32f_C1R, (Ipp32s*)ihist.ptr(), (int)ihist.step, (Ipp32f*)fhist, sizeof(Ipp32f), histRoi);
|
||||
CV_INSTRUMENT_FUN_IPP(ippiAdd_32f_C1IR, (Ipp32f*)fhist, sizeof(Ipp32f), (Ipp32f*)hist.ptr(), (int)hist.step, histRoi);
|
||||
}
|
||||
else
|
||||
CV_INSTRUMENT_FUN_IPP(ippiConvert_32s32f_C1R, (Ipp32s*)ihist.ptr(), (int)ihist.step, (Ipp32f*)hist.ptr(), (int)hist.step, ippiSize(1, histSize));
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
int ipp_hal_calcHist(const uchar* src_data, size_t src_step, int src_type, int src_width, int src_height,
|
||||
float* hist_data, int hist_size, const float** ranges, bool uniform, bool accumulate)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
Mat image(src_height, src_width, src_type, (void*)src_data, src_step);
|
||||
Mat hist(1, &hist_size, CV_32F, (void*)hist_data);
|
||||
|
||||
if(ipp_calchist(image, hist, hist_size, ranges, uniform, accumulate))
|
||||
return CV_HAL_ERROR_OK;
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,144 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, BigVision LLC, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#if IPP_VERSION_X100 >= 700
|
||||
|
||||
using namespace cv;
|
||||
|
||||
#if IPP_VERSION_X100 >= 201700
|
||||
#define IPP_HAL_MALLOC(SIZE) ippMalloc_L(SIZE)
|
||||
#else
|
||||
#define IPP_HAL_MALLOC(SIZE) ippMalloc((int)(SIZE))
|
||||
#endif
|
||||
|
||||
typedef IppStatus (CV_STDCALL * ippimatchTemplate)(const void*, int, IppiSize, const void*, int, IppiSize, Ipp32f* , int , IppEnum , Ipp8u*);
|
||||
|
||||
static bool ipp_crossCorr(const Mat& src, const Mat& tpl, Mat& dst, bool normed)
|
||||
{
|
||||
IppStatus status;
|
||||
|
||||
IppiSize srcRoiSize = {src.cols,src.rows};
|
||||
IppiSize tplRoiSize = {tpl.cols,tpl.rows};
|
||||
|
||||
int bufSize=0;
|
||||
|
||||
int depth = src.depth();
|
||||
|
||||
ippimatchTemplate ippiCrossCorrNorm =
|
||||
depth==CV_8U ? (ippimatchTemplate)ippiCrossCorrNorm_8u32f_C1R:
|
||||
depth==CV_32F? (ippimatchTemplate)ippiCrossCorrNorm_32f_C1R: 0;
|
||||
|
||||
if (ippiCrossCorrNorm==0)
|
||||
return false;
|
||||
|
||||
IppEnum funCfg = (IppEnum)(+ippAlgAuto | ippiROIValid);
|
||||
if(normed)
|
||||
funCfg |= ippiNorm;
|
||||
else
|
||||
funCfg |= ippiNormNone;
|
||||
|
||||
status = ippiCrossCorrNormGetBufferSize(srcRoiSize, tplRoiSize, funCfg, &bufSize);
|
||||
if ( status < 0 )
|
||||
return false;
|
||||
|
||||
Ipp8u* buffer = bufSize > 0 ? (Ipp8u*)IPP_HAL_MALLOC(bufSize) : 0;
|
||||
if (bufSize > 0 && buffer == 0)
|
||||
return false;
|
||||
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiCrossCorrNorm, src.ptr(), (int)src.step, srcRoiSize, tpl.ptr(), (int)tpl.step, tplRoiSize, dst.ptr<Ipp32f>(), (int)dst.step, funCfg, buffer);
|
||||
|
||||
if (buffer)
|
||||
ippFree(buffer);
|
||||
return status >= 0;
|
||||
}
|
||||
|
||||
static bool ipp_sqrDistance(const Mat& src, const Mat& tpl, Mat& dst)
|
||||
{
|
||||
IppStatus status;
|
||||
|
||||
IppiSize srcRoiSize = {src.cols,src.rows};
|
||||
IppiSize tplRoiSize = {tpl.cols,tpl.rows};
|
||||
|
||||
int bufSize=0;
|
||||
|
||||
int depth = src.depth();
|
||||
|
||||
ippimatchTemplate ippiSqrDistanceNorm =
|
||||
depth==CV_8U ? (ippimatchTemplate)ippiSqrDistanceNorm_8u32f_C1R:
|
||||
depth==CV_32F? (ippimatchTemplate)ippiSqrDistanceNorm_32f_C1R: 0;
|
||||
|
||||
if (ippiSqrDistanceNorm==0)
|
||||
return false;
|
||||
|
||||
IppEnum funCfg = (IppEnum)(+ippAlgAuto | ippiROIValid | ippiNormNone);
|
||||
status = ippiSqrDistanceNormGetBufferSize(srcRoiSize, tplRoiSize, funCfg, &bufSize);
|
||||
if ( status < 0 )
|
||||
return false;
|
||||
|
||||
Ipp8u* buffer = bufSize > 0 ? (Ipp8u*)IPP_HAL_MALLOC(bufSize) : 0;
|
||||
if (bufSize > 0 && buffer == 0)
|
||||
return false;
|
||||
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiSqrDistanceNorm, src.ptr(), (int)src.step, srcRoiSize, tpl.ptr(), (int)tpl.step, tplRoiSize, dst.ptr<Ipp32f>(), (int)dst.step, funCfg, buffer);
|
||||
|
||||
if (buffer)
|
||||
ippFree(buffer);
|
||||
dst = cv::max(dst, 0); // handle edge case from rounding in variance computation which can result in negative values
|
||||
return status >= 0;
|
||||
}
|
||||
|
||||
int ipp_hal_matchTemplate(const uchar* src_data, size_t src_step, int src_width, int src_height,
|
||||
const uchar* templ_data, size_t templ_step, int templ_width, int templ_height,
|
||||
float* result_data, size_t result_step, int depth, int cn, int method)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
if(cn != 1)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
if(depth != CV_8U && depth != CV_32F)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
Mat img(src_height, src_width, CV_MAKETYPE(depth, cn), (void*)src_data, src_step);
|
||||
Mat templ(templ_height, templ_width, CV_MAKETYPE(depth, cn), (void*)templ_data, templ_step);
|
||||
Mat result(src_height - templ_height + 1, src_width - templ_width + 1, CV_32FC1, (void*)result_data, result_step);
|
||||
|
||||
// These functions are not efficient if template size is comparable with image size
|
||||
if(templ.size().area()*4 > img.size().area())
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// CV_8U SQDIFF/SQDIFF_NORMED suffer from float32 catastrophic cancellation
|
||||
// in IPP's internal accumulators; fall through to the double-precision path instead.
|
||||
if(depth == CV_8U && (method == cv::TM_SQDIFF || method == cv::TM_SQDIFF_NORMED))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
if(method == cv::TM_SQDIFF)
|
||||
{
|
||||
if(ipp_sqrDistance(img, templ, result))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(method == cv::TM_CCORR_NORMED)
|
||||
{
|
||||
if(ipp_crossCorr(img, templ, result, true))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
else if(method == cv::TM_SQDIFF_NORMED || method == cv::TM_CCORR ||
|
||||
method == cv::TM_CCOEFF || method == cv::TM_CCOEFF_NORMED)
|
||||
{
|
||||
// Raw cross-correlation; caller finishes SQDIFF_NORMED/CCOEFF/CCOEFF_NORMED via
|
||||
// common_matchTemplate (TM_CCORR is already final).
|
||||
if(ipp_crossCorr(img, templ, result, false))
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
#endif // IPP_VERSION_X100 >= 700
|
||||
@@ -0,0 +1,80 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#include "ipp_hal_core.hpp"
|
||||
|
||||
#include <opencv2/core/core.hpp>
|
||||
#include <opencv2/core/base.hpp>
|
||||
|
||||
int ipp_hal_invSqrt32f(const float* src, float* dst, int len)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
IppStatus status = CV_INSTRUMENT_FUN_IPP(ippsInvSqrt_32f_A21, src, dst, len);
|
||||
if (status >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_invSqrt64f(const double* src, double* dst, int len)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
IppStatus status = CV_INSTRUMENT_FUN_IPP(ippsInvSqrt_64f_A50, src, dst, len);
|
||||
if (status >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_exp32f(const float* src, float* dst, int len)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
IppStatus status = CV_INSTRUMENT_FUN_IPP(ippsExp_32f_A21, src, dst, len);
|
||||
if (status >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_exp64f(const double* src, double* dst, int len)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
IppStatus status = CV_INSTRUMENT_FUN_IPP(ippsExp_64f_A50, src, dst, len);
|
||||
if (status >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_log32f(const float* src, float* dst, int len)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
IppStatus status = CV_INSTRUMENT_FUN_IPP(ippsLn_32f_A21, src, dst, len);
|
||||
if (status >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
int ipp_hal_log64f(const double* src, double* dst, int len)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
IppStatus status = CV_INSTRUMENT_FUN_IPP(ippsLn_64f_A50, src, dst, len);
|
||||
if (status >= 0)
|
||||
{
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
@@ -23,6 +23,20 @@ static inline IppiSize ippiSize(const cv::Size & _size)
|
||||
return size;
|
||||
}
|
||||
|
||||
#if IPP_VERSION_X100 >= 201700
|
||||
static inline IppiSizeL ippiSizeL(size_t width, size_t height)
|
||||
{
|
||||
IppiSizeL size = { (IppSizeL)width, (IppSizeL)height };
|
||||
return size;
|
||||
}
|
||||
|
||||
static inline IppiSizeL ippiSizeL(const cv::Size & _size)
|
||||
{
|
||||
IppiSizeL size = { _size.width, _size.height };
|
||||
return size;
|
||||
}
|
||||
#endif
|
||||
|
||||
static inline IppDataType ippiGetDataType(int depth)
|
||||
{
|
||||
depth = CV_MAT_DEPTH(depth);
|
||||
@@ -106,4 +120,30 @@ static inline int ippiSuggestRowThreadsNum(const ::ipp::IwiImage &image, size_t
|
||||
}
|
||||
#endif
|
||||
|
||||
#if IPP_VERSION_X100 >= 201700
|
||||
#define CV_IPP_MALLOC(SIZE) ippMalloc_L(SIZE)
|
||||
#else
|
||||
#define CV_IPP_MALLOC(SIZE) ippMalloc((int)SIZE)
|
||||
#endif
|
||||
|
||||
template<typename T>
|
||||
class IppAutoBuffer
|
||||
{
|
||||
public:
|
||||
IppAutoBuffer() { m_size = 0; m_pBuffer = NULL; }
|
||||
explicit IppAutoBuffer(size_t size) { m_size = 0; m_pBuffer = NULL; allocate(size); }
|
||||
~IppAutoBuffer() { deallocate(); }
|
||||
T* allocate(size_t size) { if(m_size < size) { deallocate(); m_pBuffer = (T*)CV_IPP_MALLOC(size); m_size = size; } return m_pBuffer; }
|
||||
void deallocate() { if(m_pBuffer) { ippFree(m_pBuffer); m_pBuffer = NULL; } m_size = 0; }
|
||||
inline T* get() { return (T*)m_pBuffer;}
|
||||
inline operator T* () { return (T*)m_pBuffer;}
|
||||
inline operator const T* () const { return (const T*)m_pBuffer;}
|
||||
private:
|
||||
IppAutoBuffer(IppAutoBuffer &) {}
|
||||
IppAutoBuffer& operator =(const IppAutoBuffer &) {return *this;}
|
||||
|
||||
size_t m_size;
|
||||
T* m_pBuffer;
|
||||
};
|
||||
|
||||
#endif //__PRECOMP_IPP_HPP__
|
||||
|
||||
@@ -0,0 +1,196 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#ifdef HAVE_IPP_IW
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#include "iw++/iw.hpp"
|
||||
|
||||
#include <cfloat>
|
||||
|
||||
#define IPP_RESIZE_PARALLEL 1
|
||||
|
||||
class ipp_resizeParallel: public cv::ParallelLoopBody
|
||||
{
|
||||
public:
|
||||
ipp_resizeParallel(::ipp::IwiImage &src, ::ipp::IwiImage &dst, bool &ok):
|
||||
m_src(src), m_dst(dst), m_ok(ok) {}
|
||||
~ipp_resizeParallel()
|
||||
{
|
||||
}
|
||||
|
||||
void Init(IppiInterpolationType inter)
|
||||
{
|
||||
iwiResize.InitAlloc(m_src.m_size, m_dst.m_size, m_src.m_dataType, m_src.m_channels, inter, ::ipp::IwiResizeParams(0, 0, 0.75, 4), ippBorderRepl);
|
||||
|
||||
m_ok = true;
|
||||
}
|
||||
|
||||
virtual void operator() (const cv::Range& range) const CV_OVERRIDE
|
||||
{
|
||||
if(!m_ok)
|
||||
return;
|
||||
|
||||
try
|
||||
{
|
||||
::ipp::IwiTile tile = ::ipp::IwiRoi(0, range.start, m_dst.m_size.width, range.end - range.start);
|
||||
CV_INSTRUMENT_FUN_IPP(iwiResize, m_src, m_dst, ippBorderRepl, tile);
|
||||
}
|
||||
catch(const ::ipp::IwException &)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
private:
|
||||
::ipp::IwiImage &m_src;
|
||||
::ipp::IwiImage &m_dst;
|
||||
|
||||
mutable ::ipp::IwiResize iwiResize;
|
||||
|
||||
volatile bool &m_ok;
|
||||
const ipp_resizeParallel& operator= (const ipp_resizeParallel&);
|
||||
};
|
||||
|
||||
class ipp_resizeAffineParallel: public cv::ParallelLoopBody
|
||||
{
|
||||
public:
|
||||
ipp_resizeAffineParallel(::ipp::IwiImage &src, ::ipp::IwiImage &dst, bool &ok):
|
||||
m_src(src), m_dst(dst), m_ok(ok) {}
|
||||
~ipp_resizeAffineParallel()
|
||||
{
|
||||
}
|
||||
|
||||
void Init(IppiInterpolationType inter, double scaleX, double scaleY)
|
||||
{
|
||||
double shift = (inter == ippNearest)?-1e-10:-0.5;
|
||||
double coeffs[2][3] = {
|
||||
{scaleX, 0, shift+0.5*scaleX},
|
||||
{0, scaleY, shift+0.5*scaleY}
|
||||
};
|
||||
|
||||
iwiWarpAffine.InitAlloc(m_src.m_size, m_dst.m_size, m_src.m_dataType, m_src.m_channels, coeffs, iwTransForward, inter, ::ipp::IwiWarpAffineParams(0, 0, 0.75), ippBorderRepl);
|
||||
|
||||
m_ok = true;
|
||||
}
|
||||
|
||||
virtual void operator() (const cv::Range& range) const CV_OVERRIDE
|
||||
{
|
||||
if(!m_ok)
|
||||
return;
|
||||
|
||||
try
|
||||
{
|
||||
::ipp::IwiTile tile = ::ipp::IwiRoi(0, range.start, m_dst.m_size.width, range.end - range.start);
|
||||
CV_INSTRUMENT_FUN_IPP(iwiWarpAffine, m_src, m_dst, tile);
|
||||
}
|
||||
catch(const ::ipp::IwException &)
|
||||
{
|
||||
m_ok = false;
|
||||
return;
|
||||
}
|
||||
}
|
||||
private:
|
||||
::ipp::IwiImage &m_src;
|
||||
::ipp::IwiImage &m_dst;
|
||||
|
||||
mutable ::ipp::IwiWarpAffine iwiWarpAffine;
|
||||
|
||||
volatile bool &m_ok;
|
||||
const ipp_resizeAffineParallel& operator= (const ipp_resizeAffineParallel&);
|
||||
};
|
||||
|
||||
int ipp_hal_resize(int src_type, const uchar *src_data, size_t src_step, int src_width, int src_height,
|
||||
uchar *dst_data, size_t dst_step, int dst_width, int dst_height,
|
||||
double inv_scale_x, double inv_scale_y, int interpolation)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
int depth = CV_MAT_DEPTH(src_type), channels = CV_MAT_CN(src_type);
|
||||
|
||||
IppDataType ippDataType = ippiGetDataType(depth);
|
||||
IppiInterpolationType ippInter = ippiGetInterpolation(interpolation);
|
||||
if((int)ippInter < 0)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
// Resize which doesn't match OpenCV exactly
|
||||
if (!cv::ipp::useIPP_NotExact())
|
||||
{
|
||||
if (ippInter == ippNearest || ippInter == ippSuper || (ippDataType == ipp8u && ippInter == ippLinear))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
if(ippInter != ippLinear && ippDataType == ipp64f)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
#if IPP_VERSION_X100 < 201801
|
||||
// Degradations on int^2 linear downscale
|
||||
if (ippDataType != ipp64f && ippInter == ippLinear && inv_scale_x < 1 && inv_scale_y < 1) // if downscale
|
||||
{
|
||||
int scale_x = (int)(1 / inv_scale_x);
|
||||
int scale_y = (int)(1 / inv_scale_y);
|
||||
if (1 / inv_scale_x - scale_x < DBL_EPSILON && 1 / inv_scale_y - scale_y < DBL_EPSILON) // if integer
|
||||
{
|
||||
if (!(scale_x&(scale_x - 1)) && !(scale_y&(scale_y - 1))) // if power of 2
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
bool affine = false;
|
||||
const double IPP_RESIZE_EPS = (depth == CV_64F)?0:1e-10;
|
||||
double ex = fabs((double)dst_width / src_width - inv_scale_x) / inv_scale_x;
|
||||
double ey = fabs((double)dst_height / src_height - inv_scale_y) / inv_scale_y;
|
||||
|
||||
// Use affine transform resize to allow sub-pixel accuracy
|
||||
if(ex > IPP_RESIZE_EPS || ey > IPP_RESIZE_EPS)
|
||||
affine = true;
|
||||
|
||||
// Affine doesn't support Lanczos and Super interpolations
|
||||
if(affine && (ippInter == ippLanczos || ippInter == ippSuper))
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
try
|
||||
{
|
||||
::ipp::IwiImage iwSrc(::ipp::IwiSize(src_width, src_height), ippDataType, channels, 0, (void*)src_data, src_step);
|
||||
::ipp::IwiImage iwDst(::ipp::IwiSize(dst_width, dst_height), ippDataType, channels, 0, (void*)dst_data, dst_step);
|
||||
|
||||
bool ok;
|
||||
int threads = ippiSuggestThreadsNum(iwDst, 1+((double)(src_width*src_height)/(dst_width*dst_height)));
|
||||
cv::Range range(0, dst_height);
|
||||
ipp_resizeParallel invokerGeneral(iwSrc, iwDst, ok);
|
||||
ipp_resizeAffineParallel invokerAffine(iwSrc, iwDst, ok);
|
||||
cv::ParallelLoopBody *pInvoker = NULL;
|
||||
if(affine)
|
||||
{
|
||||
pInvoker = &invokerAffine;
|
||||
invokerAffine.Init(ippInter, inv_scale_x, inv_scale_y);
|
||||
}
|
||||
else
|
||||
{
|
||||
pInvoker = &invokerGeneral;
|
||||
invokerGeneral.Init(ippInter);
|
||||
}
|
||||
|
||||
if(IPP_RESIZE_PARALLEL && threads > 1)
|
||||
cv::parallel_for_(range, *pInvoker, threads*4);
|
||||
else
|
||||
pInvoker->operator()(range);
|
||||
|
||||
if(!ok)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
catch(const ::ipp::IwException &)
|
||||
{
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
#endif // HAVE_IPP_IW
|
||||
@@ -0,0 +1,128 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, BigVision LLC, all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "ipp_hal_imgproc.hpp"
|
||||
|
||||
#include <opencv2/core.hpp>
|
||||
#include "precomp_ipp.hpp"
|
||||
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
|
||||
#if IPP_VERSION_X100 >= 700
|
||||
|
||||
int ipp_hal_threshold(const uchar* src_data, size_t src_step, uchar* dst_data, size_t dst_step,
|
||||
int width, int height, int depth, int cn, double thresh, double maxValue,
|
||||
int thresholdType)
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
CV_UNUSED(maxValue);
|
||||
|
||||
// IPP only implements these three types and depths; everything else falls back to OpenCV.
|
||||
if (thresholdType != cv::THRESH_TRUNC && thresholdType != cv::THRESH_TOZERO &&
|
||||
thresholdType != cv::THRESH_TOZERO_INV)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
if (depth != CV_8U && depth != CV_16S && depth != CV_32F)
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
|
||||
int roi_width = width * cn; // threshold is elementwise, so channels fold into the width
|
||||
int roi_height = height;
|
||||
int elemSize1 = (depth == CV_8U) ? 1 : (depth == CV_16S) ? 2 : 4;
|
||||
|
||||
if (src_step == (size_t)roi_width * elemSize1 && dst_step == (size_t)roi_width * elemSize1)
|
||||
{
|
||||
roi_width *= roi_height;
|
||||
roi_height = 1;
|
||||
src_step = dst_step = (size_t)roi_width * elemSize1;
|
||||
}
|
||||
|
||||
IppiSize sz = { roi_width, roi_height };
|
||||
const bool inplace = (src_data == dst_data);
|
||||
IppStatus status = ippStsErr;
|
||||
|
||||
CV_SUPPRESS_DEPRECATED_START
|
||||
switch (depth)
|
||||
{
|
||||
case CV_8U:
|
||||
{
|
||||
Ipp8u t = (Ipp8u)thresh;
|
||||
switch (thresholdType)
|
||||
{
|
||||
case cv::THRESH_TRUNC:
|
||||
if (inplace)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GT_8u_C1IR, dst_data, (int)dst_step, sz, t);
|
||||
if (!inplace || status < 0)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GT_8u_C1R, src_data, (int)src_step, dst_data, (int)dst_step, sz, t);
|
||||
break;
|
||||
case cv::THRESH_TOZERO:
|
||||
if (inplace)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_LTVal_8u_C1IR, dst_data, (int)dst_step, sz, t + 1, 0);
|
||||
if (!inplace || status < 0)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_LTVal_8u_C1R, src_data, (int)src_step, dst_data, (int)dst_step, sz, t + 1, 0);
|
||||
break;
|
||||
case cv::THRESH_TOZERO_INV:
|
||||
if (inplace)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GTVal_8u_C1IR, dst_data, (int)dst_step, sz, t, 0);
|
||||
if (!inplace || status < 0)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GTVal_8u_C1R, src_data, (int)src_step, dst_data, (int)dst_step, sz, t, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case CV_16S:
|
||||
{
|
||||
const Ipp16s* src = (const Ipp16s*)src_data;
|
||||
Ipp16s* dst = (Ipp16s*)dst_data;
|
||||
Ipp16s t = (Ipp16s)thresh;
|
||||
switch (thresholdType)
|
||||
{
|
||||
case cv::THRESH_TRUNC:
|
||||
if (inplace)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GT_16s_C1IR, dst, (int)dst_step, sz, t);
|
||||
if (!inplace || status < 0)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GT_16s_C1R, src, (int)src_step, dst, (int)dst_step, sz, t);
|
||||
break;
|
||||
case cv::THRESH_TOZERO:
|
||||
if (inplace)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_LTVal_16s_C1IR, dst, (int)dst_step, sz, t + 1, 0);
|
||||
if (!inplace || status < 0)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_LTVal_16s_C1R, src, (int)src_step, dst, (int)dst_step, sz, t + 1, 0);
|
||||
break;
|
||||
case cv::THRESH_TOZERO_INV:
|
||||
if (inplace)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GTVal_16s_C1IR, dst, (int)dst_step, sz, t, 0);
|
||||
if (!inplace || status < 0)
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GTVal_16s_C1R, src, (int)src_step, dst, (int)dst_step, sz, t, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case CV_32F:
|
||||
{
|
||||
const Ipp32f* src = (const Ipp32f*)src_data;
|
||||
Ipp32f* dst = (Ipp32f*)dst_data;
|
||||
Ipp32f t = (Ipp32f)thresh;
|
||||
switch (thresholdType)
|
||||
{
|
||||
case cv::THRESH_TRUNC:
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GT_32f_C1R, src, (int)src_step, dst, (int)dst_step, sz, t);
|
||||
break;
|
||||
case cv::THRESH_TOZERO:
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_LTVal_32f_C1R, src, (int)src_step, dst, (int)dst_step, sz, nextafterf(t, std::numeric_limits<float>::infinity()), 0);
|
||||
break;
|
||||
case cv::THRESH_TOZERO_INV:
|
||||
status = CV_INSTRUMENT_FUN_IPP(ippiThreshold_GTVal_32f_C1R, src, (int)src_step, dst, (int)dst_step, sz, t, 0);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
CV_SUPPRESS_DEPRECATED_END
|
||||
|
||||
return status >= 0 ? CV_HAL_ERROR_OK : CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
#endif // IPP_VERSION_X100 >= 700
|
||||
@@ -81,6 +81,8 @@ static bool IPPSet(const double value[4], void *dataPointer, int step, IppiSize
|
||||
int ipp_hal_warpAffine(int src_type, const uchar *src_data, size_t src_step, int src_width, int src_height, uchar *dst_data, size_t dst_step,
|
||||
int dst_width, int dst_height, const double M[6], int interpolation, int borderType, const double borderValue[4])
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
//CV_INSTRUMENT_REGION_IPP();
|
||||
|
||||
IppiInterpolationType ippInter = ippiGetInterpolation(interpolation);
|
||||
@@ -188,6 +190,8 @@ int ipp_hal_warpAffine(int src_type, const uchar *src_data, size_t src_step, int
|
||||
int ipp_hal_warpPerspective(int src_type, const uchar *src_data, size_t src_step, int src_width, int src_height, uchar * dst_data, size_t dst_step,
|
||||
int dst_width, int dst_height, const double M[9], int interpolation, int borderType, const double borderValue[4])
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
//CV_INSTRUMENT_REGION_IPP();
|
||||
|
||||
IppiInterpolationType ippInter = ippiGetInterpolation(interpolation);
|
||||
@@ -231,7 +235,7 @@ int ipp_hal_warpPerspective(int src_type, const uchar *src_data, size_t src_step
|
||||
{{0, 0}, {0, 0}, {0, 0}, {0, 1}}, //16U
|
||||
{{1, 1}, {0, 0}, {1, 0}, {1, 1}}, //16S
|
||||
{{1, 1}, {0, 0}, {1, 0}, {1, 1}}, //32S
|
||||
{{0, 0}, {0, 0}, {0, 0}, {1, 0}}, //32F
|
||||
{{0, 0}, {0, 0}, {0, 0}, {0, 0}}, //32F
|
||||
{{0, 0}, {0, 0}, {0, 0}, {0, 0}}}; //64F
|
||||
#endif
|
||||
|
||||
@@ -384,6 +388,8 @@ int ipp_hal_remap32f(int src_type, const uchar *src_data, size_t src_step, int s
|
||||
float *mapx, size_t mapx_step, float *mapy, size_t mapy_step,
|
||||
int interpolation, int border_type, const double border_value[4])
|
||||
{
|
||||
CV_HAL_CHECK_USE_IPP();
|
||||
|
||||
if (!((interpolation == cv::INTER_LINEAR || interpolation == cv::INTER_CUBIC || interpolation == cv::INTER_NEAREST) &&
|
||||
(border_type == cv::BORDER_CONSTANT || border_type == cv::BORDER_TRANSPARENT)))
|
||||
{
|
||||
|
||||
@@ -141,7 +141,7 @@ int filter(cvhalFilter2D *context,
|
||||
int cal_y = offset_y - ctx->anchor_y; // negative if top border exceeded
|
||||
|
||||
// calculate source border
|
||||
ctx->padding.resize(cal_width * cal_height * cnes);
|
||||
ctx->padding.resize((size_t)cal_width * cal_height * cnes);
|
||||
uchar* pad_data = &ctx->padding[0];
|
||||
int pad_step = cal_width * cnes;
|
||||
|
||||
|
||||
@@ -19,6 +19,14 @@ inline int meanStdDev_32FC1(const uchar* src_data, size_t src_step, int width, i
|
||||
|
||||
int meanStdDev(const uchar* src_data, size_t src_step, int width, int height, int src_type,
|
||||
double* mean_val, double* stddev_val, uchar* mask, size_t mask_step) {
|
||||
|
||||
// NOTE: The OpenCV imlementation with universal intrinscs is more efficient
|
||||
// See https://github.com/opencv/opencv/pull/29339#issuecomment-4778104352
|
||||
if (!mask && !stddev_val)
|
||||
{
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
switch (src_type)
|
||||
{
|
||||
case CV_8UC1:
|
||||
|
||||
@@ -56,32 +56,35 @@ inline int fast_16(const uchar* src_data, size_t src_step,
|
||||
size_t vl;
|
||||
for (; j < width - 3; j += vl, ptr += vl)
|
||||
{
|
||||
vl = __riscv_vsetvl_e16m1(width - 3 - j);
|
||||
/* u8mf2 pre-screen: same VL as i16m1, defer vzext until needed */
|
||||
vl = __riscv_vsetvl_e8mf2(width - 3 - j);
|
||||
vuint8mf2_t vcen_8 = __riscv_vle8_v_u8mf2(ptr, vl);
|
||||
vuint8mf2_t vlo_8 = __riscv_vssubu_vx_u8mf2(vcen_8, (uint8_t)threshold, vl);
|
||||
vuint8mf2_t vhi_8 = __riscv_vsaddu_vx_u8mf2(vcen_8, (uint8_t)threshold, vl);
|
||||
vuint8mf2_t vk0_8 = __riscv_vle8_v_u8mf2(ptr + pixel[0], vl);
|
||||
vuint8mf2_t vk4_8 = __riscv_vle8_v_u8mf2(ptr + pixel[4], vl);
|
||||
vuint8mf2_t vk8_8 = __riscv_vle8_v_u8mf2(ptr + pixel[8], vl);
|
||||
vuint8mf2_t vk12_8 = __riscv_vle8_v_u8mf2(ptr + pixel[12], vl);
|
||||
|
||||
/* Load center pixel and widen to int16 */
|
||||
vint16m1_t vcen = __riscv_vreinterpret_v_u16m1_i16m1(
|
||||
__riscv_vzext_vf2(__riscv_vle8_v_u8mf2(ptr, vl), vl));
|
||||
vint16m1_t vlo = __riscv_vsub_vx_i16m1(vcen, threshold, vl);
|
||||
vint16m1_t vhi = __riscv_vadd_vx_i16m1(vcen, threshold, vl);
|
||||
|
||||
/* 4-direction quick reject */
|
||||
vint16m1_t vk0 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(__riscv_vle8_v_u8mf2(ptr + pixel[0], vl), vl));
|
||||
vint16m1_t vk4 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(__riscv_vle8_v_u8mf2(ptr + pixel[4], vl), vl));
|
||||
vint16m1_t vk8 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(__riscv_vle8_v_u8mf2(ptr + pixel[8], vl), vl));
|
||||
vint16m1_t vk12 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(__riscv_vle8_v_u8mf2(ptr + pixel[12], vl), vl));
|
||||
|
||||
vbool16_t bright = __riscv_vmand_mm_b16(__riscv_vmsgt_vv_i16m1_b16(vk0, vhi, vl), __riscv_vmsgt_vv_i16m1_b16(vk4, vhi, vl), vl);
|
||||
vbool16_t dark = __riscv_vmand_mm_b16(__riscv_vmsgt_vv_i16m1_b16(vlo, vk0, vl), __riscv_vmsgt_vv_i16m1_b16(vlo, vk4, vl), vl);
|
||||
bright = __riscv_vmor_mm_b16(bright, __riscv_vmand_mm_b16(__riscv_vmsgt_vv_i16m1_b16(vk4, vhi, vl), __riscv_vmsgt_vv_i16m1_b16(vk8, vhi, vl), vl), vl);
|
||||
dark = __riscv_vmor_mm_b16(dark, __riscv_vmand_mm_b16(__riscv_vmsgt_vv_i16m1_b16(vlo, vk4, vl), __riscv_vmsgt_vv_i16m1_b16(vlo, vk8, vl), vl), vl);
|
||||
bright = __riscv_vmor_mm_b16(bright, __riscv_vmand_mm_b16(__riscv_vmsgt_vv_i16m1_b16(vk8, vhi, vl), __riscv_vmsgt_vv_i16m1_b16(vk12, vhi, vl), vl), vl);
|
||||
dark = __riscv_vmor_mm_b16(dark, __riscv_vmand_mm_b16(__riscv_vmsgt_vv_i16m1_b16(vlo, vk8, vl), __riscv_vmsgt_vv_i16m1_b16(vlo, vk12, vl), vl), vl);
|
||||
bright = __riscv_vmor_mm_b16(bright, __riscv_vmand_mm_b16(__riscv_vmsgt_vv_i16m1_b16(vk12, vhi, vl), __riscv_vmsgt_vv_i16m1_b16(vk0, vhi, vl), vl), vl);
|
||||
dark = __riscv_vmor_mm_b16(dark, __riscv_vmand_mm_b16(__riscv_vmsgt_vv_i16m1_b16(vlo, vk12, vl), __riscv_vmsgt_vv_i16m1_b16(vlo, vk0, vl), vl), vl);
|
||||
vbool16_t bright = __riscv_vmand_mm_b16(__riscv_vmsgtu_vv_u8mf2_b16(vk0_8, vhi_8, vl), __riscv_vmsgtu_vv_u8mf2_b16(vk4_8, vhi_8, vl), vl);
|
||||
vbool16_t dark = __riscv_vmand_mm_b16(__riscv_vmsgtu_vv_u8mf2_b16(vlo_8, vk0_8, vl), __riscv_vmsgtu_vv_u8mf2_b16(vlo_8, vk4_8, vl), vl);
|
||||
bright = __riscv_vmor_mm_b16(bright, __riscv_vmand_mm_b16(__riscv_vmsgtu_vv_u8mf2_b16(vk4_8, vhi_8, vl), __riscv_vmsgtu_vv_u8mf2_b16(vk8_8, vhi_8, vl), vl), vl);
|
||||
dark = __riscv_vmor_mm_b16(dark, __riscv_vmand_mm_b16(__riscv_vmsgtu_vv_u8mf2_b16(vlo_8, vk4_8, vl), __riscv_vmsgtu_vv_u8mf2_b16(vlo_8, vk8_8, vl), vl), vl);
|
||||
bright = __riscv_vmor_mm_b16(bright, __riscv_vmand_mm_b16(__riscv_vmsgtu_vv_u8mf2_b16(vk8_8, vhi_8, vl), __riscv_vmsgtu_vv_u8mf2_b16(vk12_8, vhi_8, vl), vl), vl);
|
||||
dark = __riscv_vmor_mm_b16(dark, __riscv_vmand_mm_b16(__riscv_vmsgtu_vv_u8mf2_b16(vlo_8, vk8_8, vl), __riscv_vmsgtu_vv_u8mf2_b16(vlo_8, vk12_8, vl), vl), vl);
|
||||
bright = __riscv_vmor_mm_b16(bright, __riscv_vmand_mm_b16(__riscv_vmsgtu_vv_u8mf2_b16(vk12_8, vhi_8, vl), __riscv_vmsgtu_vv_u8mf2_b16(vk0_8, vhi_8, vl), vl), vl);
|
||||
dark = __riscv_vmor_mm_b16(dark, __riscv_vmand_mm_b16(__riscv_vmsgtu_vv_u8mf2_b16(vlo_8, vk12_8, vl), __riscv_vmsgtu_vv_u8mf2_b16(vlo_8, vk0_8, vl), vl), vl);
|
||||
|
||||
if (__riscv_vfirst_m_b16(__riscv_vmor_mm_b16(bright, dark, vl), vl) < 0)
|
||||
continue;
|
||||
|
||||
/* Widen pre-loaded u8mf2 to i16m1 for score computation */
|
||||
vint16m1_t vcen = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(vcen_8, vl));
|
||||
vint16m1_t vk0 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(vk0_8, vl));
|
||||
vint16m1_t vk4 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(vk4_8, vl));
|
||||
vint16m1_t vk8 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(vk8_8, vl));
|
||||
vint16m1_t vk12 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(vk12_8, vl));
|
||||
|
||||
/* Load remaining 12 neighbors */
|
||||
vint16m1_t vk1 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(__riscv_vle8_v_u8mf2(ptr + pixel[1], vl), vl));
|
||||
vint16m1_t vk2 = __riscv_vreinterpret_v_u16m1_i16m1(__riscv_vzext_vf2(__riscv_vle8_v_u8mf2(ptr + pixel[2], vl), vl));
|
||||
|
||||
@@ -102,6 +102,8 @@ template<> struct rvv<RVV_U8M1>
|
||||
{
|
||||
static inline vfloat32m4_t vcvt0(vuint8m1_t a, size_t b) { return __riscv_vfcvt_f(__riscv_vzext_vf4(a, b), b); }
|
||||
static inline vuint8m1_t vcvt1(vfloat32m4_t a, size_t b) { return __riscv_vnclipu(__riscv_vfncvt_xu(a, b), 0, __RISCV_VXRM_RNU, b); }
|
||||
static inline vint32m4_t vwiden(vuint8m1_t a, size_t b) { return __riscv_vreinterpret_v_u32m4_i32m4(__riscv_vzext_vf4(a, b)); }
|
||||
static inline vuint8m1_t vnarrow(vint32m4_t a, size_t b) { return __riscv_vncvt_x(__riscv_vncvt_x(__riscv_vreinterpret_v_i32m4_u32m4(a), b), b); }
|
||||
static inline vuint8m1_t vloxei(const uchar* a, vuint16m2_t b, size_t c) { return __riscv_vloxei16_v_u8m1(a, b, c); }
|
||||
static inline void vloxseg2ei(const uchar* a, vuint16m2_t b, size_t c, vuint8m1_t& x, vuint8m1_t& y) { auto src = __riscv_vloxseg2ei16_v_u8m1x2(a, b, c); x = __riscv_vget_v_u8m1x2_u8m1(src, 0); y = __riscv_vget_v_u8m1x2_u8m1(src, 1); }
|
||||
static inline void vloxseg3ei(const uchar* a, vuint16m2_t b, size_t c, vuint8m1_t& x, vuint8m1_t& y, vuint8m1_t& z) { auto src = __riscv_vloxseg3ei16_v_u8m1x3(a, b, c); x = __riscv_vget_v_u8m1x3_u8m1(src, 0); y = __riscv_vget_v_u8m1x3_u8m1(src, 1); z = __riscv_vget_v_u8m1x3_u8m1(src, 2); }
|
||||
@@ -148,6 +150,8 @@ template<> struct rvv<RVV_U8MF2>
|
||||
{
|
||||
static inline vfloat32m2_t vcvt0(vuint8mf2_t a, size_t b) { return __riscv_vfcvt_f(__riscv_vzext_vf4(a, b), b); }
|
||||
static inline vuint8mf2_t vcvt1(vfloat32m2_t a, size_t b) { return __riscv_vnclipu(__riscv_vfncvt_xu(a, b), 0, __RISCV_VXRM_RNU, b); }
|
||||
static inline vint32m2_t vwiden(vuint8mf2_t a, size_t b) { return __riscv_vreinterpret_v_u32m2_i32m2(__riscv_vzext_vf4(a, b)); }
|
||||
static inline vuint8mf2_t vnarrow(vint32m2_t a, size_t b) { return __riscv_vncvt_x(__riscv_vncvt_x(__riscv_vreinterpret_v_i32m2_u32m2(a), b), b); }
|
||||
static inline vuint8mf2_t vloxei(const uchar* a, vuint16m1_t b, size_t c) { return __riscv_vloxei16_v_u8mf2(a, b, c); }
|
||||
static inline void vloxseg2ei(const uchar* a, vuint16m1_t b, size_t c, vuint8mf2_t& x, vuint8mf2_t& y) { auto src = __riscv_vloxseg2ei16_v_u8mf2x2(a, b, c); x = __riscv_vget_v_u8mf2x2_u8mf2(src, 0); y = __riscv_vget_v_u8mf2x2_u8mf2(src, 1); }
|
||||
static inline void vloxseg3ei(const uchar* a, vuint16m1_t b, size_t c, vuint8mf2_t& x, vuint8mf2_t& y, vuint8mf2_t& z) { auto src = __riscv_vloxseg3ei16_v_u8mf2x3(a, b, c); x = __riscv_vget_v_u8mf2x3_u8mf2(src, 0); y = __riscv_vget_v_u8mf2x3_u8mf2(src, 1); z = __riscv_vget_v_u8mf2x3_u8mf2(src, 2); }
|
||||
@@ -315,6 +319,150 @@ static inline int resizeLinear(int start, int end, const uchar *src_data, size_t
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
static const int INTER_RESIZE_COEF_BITS = 11;
|
||||
static const int INTER_RESIZE_COEF_SCALE = 1 << INTER_RESIZE_COEF_BITS;
|
||||
|
||||
// Bit-exact fixed-point vertical+horizontal linear combine for one channel, matching the reference
|
||||
// HResizeLinear<uchar,int,short,INTER_RESIZE_COEF_SCALE> + VResizeLinear<uchar,int,short,
|
||||
// FixedPtCast<int,uchar,INTER_RESIZE_COEF_BITS*2>> in modules/imgproc/src/resize.cpp.
|
||||
// Sxy are the four widened (int32) source corners; a0/a1 the horizontal weights (per column),
|
||||
// b0/b1 the vertical weights (per row). The staged >>4 / >>16 / >>2 shifts (total 22) reproduce the
|
||||
// reference exactly and keep the intermediates within int32.
|
||||
template<typename VI>
|
||||
static inline VI resizeLinearU8Combine(VI S00, VI S01, VI S10, VI S11, VI a0, VI a1, int b0, int b1, size_t vl)
|
||||
{
|
||||
VI h0 = __riscv_vmacc(__riscv_vmul(S00, a0, vl), S01, a1, vl); // S[y0][x0]*a0 + S[y0][x1]*a1
|
||||
VI h1 = __riscv_vmacc(__riscv_vmul(S10, a0, vl), S11, a1, vl); // S[y1][x0]*a0 + S[y1][x1]*a1
|
||||
h0 = __riscv_vsra(__riscv_vmul(__riscv_vsra(h0, 4, vl), b0, vl), 16, vl); // (b0*(h0>>4))>>16
|
||||
h1 = __riscv_vsra(__riscv_vmul(__riscv_vsra(h1, 4, vl), b1, vl), 16, vl); // (b1*(h1>>4))>>16
|
||||
return __riscv_vsra(__riscv_vadd(__riscv_vadd(h0, h1, vl), 2, vl), 2, vl); // (t0 + t1 + 2) >> 2
|
||||
}
|
||||
|
||||
// the algorithm is copied from imgproc/src/resize.cpp, HResizeLinear + VResizeLinear for uchar.
|
||||
// The generic resizeLinear<> above computes in float, which is bit-exact only for 16U/32F; 8U
|
||||
// INTER_LINEAR must use this fixed-point path to match the reference (float drift breaks tight
|
||||
// consumers such as DNN blobFromImage preprocessing, issue #28870).
|
||||
template<typename helper, int cn>
|
||||
static inline int resizeLinearU8(int start, int end, const uchar *src_data, size_t src_step, int src_height, uchar *dst_data, size_t dst_step, int dst_width, double scale_y, const ushort* x_ofs0, const ushort* x_ofs1, const int* ialpha0, const int* ialpha1)
|
||||
{
|
||||
for (int i = start; i < end; i++)
|
||||
{
|
||||
float my = (i + 0.5) * scale_y - 0.5;
|
||||
int y_ofs = static_cast<int>(std::floor(my));
|
||||
my -= y_ofs;
|
||||
|
||||
int y_ofs0 = std::min(std::max(y_ofs , 0), src_height - 1);
|
||||
int y_ofs1 = std::min(std::max(y_ofs + 1, 0), src_height - 1);
|
||||
int b0 = saturate_cast<short>((1.f - my) * INTER_RESIZE_COEF_SCALE);
|
||||
int b1 = saturate_cast<short>(my * INTER_RESIZE_COEF_SCALE);
|
||||
|
||||
const uchar* S0 = src_data + y_ofs0 * src_step;
|
||||
const uchar* S1 = src_data + y_ofs1 * src_step;
|
||||
|
||||
int vl;
|
||||
switch (cn)
|
||||
{
|
||||
case 1:
|
||||
for (int j = 0; j < dst_width; j += vl)
|
||||
{
|
||||
vl = helper::setvl(dst_width - j);
|
||||
auto ptr0 = RVV_SameLen<ushort, helper>::vload(x_ofs0 + j, vl);
|
||||
auto ptr1 = RVV_SameLen<ushort, helper>::vload(x_ofs1 + j, vl);
|
||||
auto a0 = RVV_SameLen<int, helper>::vload(ialpha0 + j, vl);
|
||||
auto a1 = RVV_SameLen<int, helper>::vload(ialpha1 + j, vl);
|
||||
|
||||
auto v00 = rvv<helper>::vwiden(rvv<helper>::vloxei(S0, ptr0, vl), vl);
|
||||
auto v01 = rvv<helper>::vwiden(rvv<helper>::vloxei(S0, ptr1, vl), vl);
|
||||
auto v02 = rvv<helper>::vwiden(rvv<helper>::vloxei(S1, ptr0, vl), vl);
|
||||
auto v03 = rvv<helper>::vwiden(rvv<helper>::vloxei(S1, ptr1, vl), vl);
|
||||
|
||||
auto d0 = resizeLinearU8Combine(v00, v01, v02, v03, a0, a1, b0, b1, vl);
|
||||
helper::vstore(reinterpret_cast<typename helper::ElemType*>(dst_data + i * dst_step) + j, rvv<helper>::vnarrow(d0, vl), vl);
|
||||
}
|
||||
break;
|
||||
case 2:
|
||||
for (int j = 0; j < dst_width; j += vl)
|
||||
{
|
||||
vl = helper::setvl(dst_width - j);
|
||||
auto ptr0 = RVV_SameLen<ushort, helper>::vload(x_ofs0 + j, vl);
|
||||
auto ptr1 = RVV_SameLen<ushort, helper>::vload(x_ofs1 + j, vl);
|
||||
auto a0 = RVV_SameLen<int, helper>::vload(ialpha0 + j, vl);
|
||||
auto a1 = RVV_SameLen<int, helper>::vload(ialpha1 + j, vl);
|
||||
|
||||
typename helper::VecType s0, s1;
|
||||
rvv<helper>::vloxseg2ei(S0, ptr0, vl, s0, s1);
|
||||
auto v00 = rvv<helper>::vwiden(s0, vl), v10 = rvv<helper>::vwiden(s1, vl);
|
||||
rvv<helper>::vloxseg2ei(S0, ptr1, vl, s0, s1);
|
||||
auto v01 = rvv<helper>::vwiden(s0, vl), v11 = rvv<helper>::vwiden(s1, vl);
|
||||
rvv<helper>::vloxseg2ei(S1, ptr0, vl, s0, s1);
|
||||
auto v02 = rvv<helper>::vwiden(s0, vl), v12 = rvv<helper>::vwiden(s1, vl);
|
||||
rvv<helper>::vloxseg2ei(S1, ptr1, vl, s0, s1);
|
||||
auto v03 = rvv<helper>::vwiden(s0, vl), v13 = rvv<helper>::vwiden(s1, vl);
|
||||
|
||||
auto d0 = resizeLinearU8Combine(v00, v01, v02, v03, a0, a1, b0, b1, vl);
|
||||
auto d1 = resizeLinearU8Combine(v10, v11, v12, v13, a0, a1, b0, b1, vl);
|
||||
rvv<helper>::vsseg2e(reinterpret_cast<typename helper::ElemType*>(dst_data + i * dst_step) + j * 2, vl, rvv<helper>::vnarrow(d0, vl), rvv<helper>::vnarrow(d1, vl));
|
||||
}
|
||||
break;
|
||||
case 3:
|
||||
for (int j = 0; j < dst_width; j += vl)
|
||||
{
|
||||
vl = helper::setvl(dst_width - j);
|
||||
auto ptr0 = RVV_SameLen<ushort, helper>::vload(x_ofs0 + j, vl);
|
||||
auto ptr1 = RVV_SameLen<ushort, helper>::vload(x_ofs1 + j, vl);
|
||||
auto a0 = RVV_SameLen<int, helper>::vload(ialpha0 + j, vl);
|
||||
auto a1 = RVV_SameLen<int, helper>::vload(ialpha1 + j, vl);
|
||||
|
||||
typename helper::VecType s0, s1, s2;
|
||||
rvv<helper>::vloxseg3ei(S0, ptr0, vl, s0, s1, s2);
|
||||
auto v00 = rvv<helper>::vwiden(s0, vl), v10 = rvv<helper>::vwiden(s1, vl), v20 = rvv<helper>::vwiden(s2, vl);
|
||||
rvv<helper>::vloxseg3ei(S0, ptr1, vl, s0, s1, s2);
|
||||
auto v01 = rvv<helper>::vwiden(s0, vl), v11 = rvv<helper>::vwiden(s1, vl), v21 = rvv<helper>::vwiden(s2, vl);
|
||||
rvv<helper>::vloxseg3ei(S1, ptr0, vl, s0, s1, s2);
|
||||
auto v02 = rvv<helper>::vwiden(s0, vl), v12 = rvv<helper>::vwiden(s1, vl), v22 = rvv<helper>::vwiden(s2, vl);
|
||||
rvv<helper>::vloxseg3ei(S1, ptr1, vl, s0, s1, s2);
|
||||
auto v03 = rvv<helper>::vwiden(s0, vl), v13 = rvv<helper>::vwiden(s1, vl), v23 = rvv<helper>::vwiden(s2, vl);
|
||||
|
||||
auto d0 = resizeLinearU8Combine(v00, v01, v02, v03, a0, a1, b0, b1, vl);
|
||||
auto d1 = resizeLinearU8Combine(v10, v11, v12, v13, a0, a1, b0, b1, vl);
|
||||
auto d2 = resizeLinearU8Combine(v20, v21, v22, v23, a0, a1, b0, b1, vl);
|
||||
rvv<helper>::vsseg3e(reinterpret_cast<typename helper::ElemType*>(dst_data + i * dst_step) + j * 3, vl, rvv<helper>::vnarrow(d0, vl), rvv<helper>::vnarrow(d1, vl), rvv<helper>::vnarrow(d2, vl));
|
||||
}
|
||||
break;
|
||||
case 4:
|
||||
for (int j = 0; j < dst_width; j += vl)
|
||||
{
|
||||
vl = helper::setvl(dst_width - j);
|
||||
auto ptr0 = RVV_SameLen<ushort, helper>::vload(x_ofs0 + j, vl);
|
||||
auto ptr1 = RVV_SameLen<ushort, helper>::vload(x_ofs1 + j, vl);
|
||||
auto a0 = RVV_SameLen<int, helper>::vload(ialpha0 + j, vl);
|
||||
auto a1 = RVV_SameLen<int, helper>::vload(ialpha1 + j, vl);
|
||||
|
||||
typename helper::VecType s0, s1, s2, s3;
|
||||
rvv<helper>::vloxseg4ei(S0, ptr0, vl, s0, s1, s2, s3);
|
||||
auto v00 = rvv<helper>::vwiden(s0, vl), v10 = rvv<helper>::vwiden(s1, vl), v20 = rvv<helper>::vwiden(s2, vl), v30 = rvv<helper>::vwiden(s3, vl);
|
||||
rvv<helper>::vloxseg4ei(S0, ptr1, vl, s0, s1, s2, s3);
|
||||
auto v01 = rvv<helper>::vwiden(s0, vl), v11 = rvv<helper>::vwiden(s1, vl), v21 = rvv<helper>::vwiden(s2, vl), v31 = rvv<helper>::vwiden(s3, vl);
|
||||
rvv<helper>::vloxseg4ei(S1, ptr0, vl, s0, s1, s2, s3);
|
||||
auto v02 = rvv<helper>::vwiden(s0, vl), v12 = rvv<helper>::vwiden(s1, vl), v22 = rvv<helper>::vwiden(s2, vl), v32 = rvv<helper>::vwiden(s3, vl);
|
||||
rvv<helper>::vloxseg4ei(S1, ptr1, vl, s0, s1, s2, s3);
|
||||
auto v03 = rvv<helper>::vwiden(s0, vl), v13 = rvv<helper>::vwiden(s1, vl), v23 = rvv<helper>::vwiden(s2, vl), v33 = rvv<helper>::vwiden(s3, vl);
|
||||
|
||||
auto d0 = resizeLinearU8Combine(v00, v01, v02, v03, a0, a1, b0, b1, vl);
|
||||
auto d1 = resizeLinearU8Combine(v10, v11, v12, v13, a0, a1, b0, b1, vl);
|
||||
auto d2 = resizeLinearU8Combine(v20, v21, v22, v23, a0, a1, b0, b1, vl);
|
||||
auto d3 = resizeLinearU8Combine(v30, v31, v32, v33, a0, a1, b0, b1, vl);
|
||||
rvv<helper>::vsseg4e(reinterpret_cast<typename helper::ElemType*>(dst_data + i * dst_step) + j * 4, vl, rvv<helper>::vnarrow(d0, vl), rvv<helper>::vnarrow(d1, vl), rvv<helper>::vnarrow(d2, vl), rvv<helper>::vnarrow(d3, vl));
|
||||
}
|
||||
break;
|
||||
default:
|
||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||
}
|
||||
}
|
||||
|
||||
return CV_HAL_ERROR_OK;
|
||||
}
|
||||
|
||||
template<int cn>
|
||||
static inline int resizeLinearExact(int start, int end, const uchar *src_data, size_t src_step, int src_height, uchar *dst_data, size_t dst_step, int dst_width, double scale_y, const ushort* x_ofs0, const ushort* x_ofs1, const ushort* x_val)
|
||||
{
|
||||
@@ -801,6 +949,37 @@ static inline int resizeLinear(int src_type, const uchar *src_data, size_t src_s
|
||||
return invoke(dst_height, {resizeLinearExact<4>}, src_data, src_step, src_height, dst_data, dst_step, dst_width, scale_y, x_ofs0.data(), x_ofs1.data(), x_val.data());
|
||||
}
|
||||
}
|
||||
else if (CV_MAT_DEPTH(src_type) == CV_8U)
|
||||
{
|
||||
// 8U INTER_LINEAR must be bit-exact fixed-point to match the reference implementation;
|
||||
// computing it in float drifts and breaks tight consumers (e.g. DNN preprocessing, #28870).
|
||||
std::vector<int> ialpha0(dst_width), ialpha1(dst_width);
|
||||
for (int i = 0; i < dst_width; i++)
|
||||
{
|
||||
float fx = (float)((i + 0.5) * scale_x - 0.5);
|
||||
int x_ofs = static_cast<int>(std::floor(fx));
|
||||
fx -= x_ofs;
|
||||
if (x_ofs < 0) { fx = 0; x_ofs = 0; }
|
||||
if (x_ofs >= src_width - 1) { fx = 0; x_ofs = src_width - 1; }
|
||||
|
||||
x_ofs0[i] = static_cast<ushort>(x_ofs) * cn;
|
||||
x_ofs1[i] = static_cast<ushort>(std::min(x_ofs + 1, src_width - 1)) * cn;
|
||||
ialpha0[i] = saturate_cast<short>((1.f - fx) * INTER_RESIZE_COEF_SCALE);
|
||||
ialpha1[i] = saturate_cast<short>(fx * INTER_RESIZE_COEF_SCALE);
|
||||
}
|
||||
|
||||
switch (src_type)
|
||||
{
|
||||
case CV_8UC1:
|
||||
return invoke(dst_height, {resizeLinearU8<RVV_U8M1, 1>}, src_data, src_step, src_height, dst_data, dst_step, dst_width, scale_y, x_ofs0.data(), x_ofs1.data(), ialpha0.data(), ialpha1.data());
|
||||
case CV_8UC2:
|
||||
return invoke(dst_height, {resizeLinearU8<RVV_U8M1, 2>}, src_data, src_step, src_height, dst_data, dst_step, dst_width, scale_y, x_ofs0.data(), x_ofs1.data(), ialpha0.data(), ialpha1.data());
|
||||
case CV_8UC3:
|
||||
return invoke(dst_height, {resizeLinearU8<RVV_U8MF2, 3>}, src_data, src_step, src_height, dst_data, dst_step, dst_width, scale_y, x_ofs0.data(), x_ofs1.data(), ialpha0.data(), ialpha1.data());
|
||||
case CV_8UC4:
|
||||
return invoke(dst_height, {resizeLinearU8<RVV_U8MF2, 4>}, src_data, src_step, src_height, dst_data, dst_step, dst_width, scale_y, x_ofs0.data(), x_ofs1.data(), ialpha0.data(), ialpha1.data());
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
std::vector<float> x_val(dst_width);
|
||||
|
||||
@@ -881,7 +881,8 @@ CV_EXPORTS_W Vec3d RQDecomp3x3( InputArray src, OutputArray mtxR, OutputArray mt
|
||||
@param projMatrix 3x4 input projection matrix P.
|
||||
@param cameraMatrix Output 3x3 camera intrinsic matrix \f$\cameramatrix{A}\f$.
|
||||
@param rotMatrix Output 3x3 external rotation matrix R.
|
||||
@param transVect Output 4x1 translation vector T.
|
||||
@param transVect Output 4x1 vector representing the camera position in homogeneous coordinates.
|
||||
To obtain the translation vector, use t = -rotMatrix * transVect[:3].
|
||||
@param rotMatrixX Optional 3x3 rotation matrix around x-axis.
|
||||
@param rotMatrixY Optional 3x3 rotation matrix around y-axis.
|
||||
@param rotMatrixZ Optional 3x3 rotation matrix around z-axis.
|
||||
|
||||
@@ -39,4 +39,35 @@ PERF_TEST_P(String_Size, asymm_circles_grid, testing::Values(
|
||||
SANITY_CHECK(ptvec, 2);
|
||||
}
|
||||
|
||||
// Perf test using synthetic keypoints (no image I/O). Exercises the RNG and
|
||||
// findLongestPath code paths directly with a pre-detected point set.
|
||||
typedef perf::TestBaseWithParam<cv::Size> CirclesGrid_RNG_Size;
|
||||
|
||||
PERF_TEST_P(CirclesGrid_RNG_Size, detect_keypoints_symmetric,
|
||||
testing::Values(cv::Size(6, 5), cv::Size(8, 6), cv::Size(10, 8), cv::Size(15, 12), cv::Size(20, 15)))
|
||||
{
|
||||
const cv::Size patternSize = GetParam();
|
||||
const float spacing = 30.f;
|
||||
|
||||
std::vector<cv::Point2f> pts;
|
||||
pts.reserve(patternSize.area());
|
||||
for (int r = 0; r < patternSize.height; r++)
|
||||
for (int c = 0; c < patternSize.width; c++)
|
||||
pts.push_back(cv::Point2f(c * spacing, r * spacing));
|
||||
|
||||
// Shuffle so the detector works from an unordered set, same as real use.
|
||||
cv::RNG& rng = cv::theRNG();
|
||||
for (int k = (int)pts.size() - 1; k > 0; k--)
|
||||
std::swap(pts[k], pts[rng.uniform(0, k + 1)]);
|
||||
|
||||
std::vector<cv::Point2f> centers;
|
||||
centers.resize(patternSize.area());
|
||||
declare.in(cv::Mat(pts)).out(centers);
|
||||
|
||||
TEST_CYCLE() ASSERT_TRUE(findCirclesGrid(cv::Mat(pts), patternSize, centers,
|
||||
CALIB_CB_SYMMETRIC_GRID, cv::Ptr<cv::FeatureDetector>()));
|
||||
|
||||
SANITY_CHECK_NOTHING();
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
#include "precomp.hpp"
|
||||
#include "opencv2/flann.hpp"
|
||||
#include "chessboard.hpp"
|
||||
#include "math.h"
|
||||
#include <math.h>
|
||||
|
||||
//#define CV_DETECTORS_CHESSBOARD_DEBUG
|
||||
#ifdef CV_DETECTORS_CHESSBOARD_DEBUG
|
||||
|
||||
@@ -43,6 +43,7 @@
|
||||
#include "precomp.hpp"
|
||||
#include "circlesgrid.hpp"
|
||||
#include <limits>
|
||||
#include <queue>
|
||||
|
||||
// Requires CMake flag: DEBUG_opencv_calib3d=ON
|
||||
//#define DEBUG_CIRCLES
|
||||
@@ -569,8 +570,6 @@ CirclesGridFinder::Segment::Segment(cv::Point2f _s, cv::Point2f _e) :
|
||||
{
|
||||
}
|
||||
|
||||
void computeShortestPath(Mat &predecessorMatrix, int v1, int v2, std::vector<int> &path);
|
||||
void computePredecessorMatrix(const Mat &dm, int verticesCount, Mat &predecessorMatrix);
|
||||
|
||||
CirclesGridFinderParameters::CirclesGridFinderParameters()
|
||||
{
|
||||
@@ -1204,79 +1203,91 @@ void CirclesGridFinder::computeRNG(Graph &rng, std::vector<cv::Point2f> &vectors
|
||||
rng = Graph(keypoints.size());
|
||||
vectors.clear();
|
||||
|
||||
//TODO: use more fast algorithm instead of naive N^3
|
||||
for (size_t i = 0; i < keypoints.size(); i++)
|
||||
{
|
||||
for (size_t j = 0; j < keypoints.size(); j++)
|
||||
{
|
||||
if (i == j)
|
||||
continue;
|
||||
|
||||
Point2f vec = keypoints[i] - keypoints[j];
|
||||
double dist = norm(vec);
|
||||
|
||||
bool isNeighbors = true;
|
||||
for (size_t k = 0; k < keypoints.size(); k++)
|
||||
{
|
||||
if (k == i || k == j)
|
||||
continue;
|
||||
|
||||
double dist1 = norm(keypoints[i] - keypoints[k]);
|
||||
double dist2 = norm(keypoints[j] - keypoints[k]);
|
||||
if (dist1 < dist && dist2 < dist)
|
||||
{
|
||||
isNeighbors = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (isNeighbors)
|
||||
{
|
||||
rng.addEdge(i, j);
|
||||
vectors.push_back(keypoints[i] - keypoints[j]);
|
||||
if (drawImage != 0)
|
||||
{
|
||||
line(*drawImage, keypoints[i], keypoints[j], Scalar(255, 0, 0), 2);
|
||||
circle(*drawImage, keypoints[i], 3, Scalar(0, 0, 255), -1);
|
||||
circle(*drawImage, keypoints[j], 3, Scalar(0, 0, 255), -1);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void computePredecessorMatrix(const Mat &dm, int verticesCount, Mat &predecessorMatrix)
|
||||
{
|
||||
CV_Assert( dm.type() == CV_32SC1 );
|
||||
predecessorMatrix.create(verticesCount, verticesCount, CV_32SC1);
|
||||
predecessorMatrix = -1;
|
||||
for (int i = 0; i < predecessorMatrix.rows; i++)
|
||||
{
|
||||
for (int j = 0; j < predecessorMatrix.cols; j++)
|
||||
{
|
||||
int dist = dm.at<int> (i, j);
|
||||
for (int k = 0; k < verticesCount; k++)
|
||||
{
|
||||
if (dm.at<int> (i, k) == dist - 1 && dm.at<int> (k, j) == 1)
|
||||
{
|
||||
predecessorMatrix.at<int> (i, j) = k;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void computeShortestPath(Mat &predecessorMatrix, size_t v1, size_t v2, std::vector<size_t> &path)
|
||||
{
|
||||
if (predecessorMatrix.at<int> ((int)v1, (int)v2) < 0)
|
||||
{
|
||||
path.push_back(v1);
|
||||
const size_t n = keypoints.size();
|
||||
if (n < 2)
|
||||
return;
|
||||
|
||||
// RNG is a subgraph of the Delaunay triangulation, so we only need to test
|
||||
// Delaunay edges as candidates. This brings the complexity from O(N^3) down
|
||||
// to O(N^2) in the worst case, and much better in practice for regular grids
|
||||
// where Delaunay edges (~3N) are almost all RNG edges anyway.
|
||||
|
||||
float minX = keypoints[0].x, minY = keypoints[0].y;
|
||||
float maxX = minX, maxY = minY;
|
||||
for (size_t i = 1; i < n; i++)
|
||||
{
|
||||
minX = std::min(minX, keypoints[i].x);
|
||||
minY = std::min(minY, keypoints[i].y);
|
||||
maxX = std::max(maxX, keypoints[i].x);
|
||||
maxY = std::max(maxY, keypoints[i].y);
|
||||
}
|
||||
|
||||
computeShortestPath(predecessorMatrix, v1, predecessorMatrix.at<int> ((int)v1, (int)v2), path);
|
||||
path.push_back(v2);
|
||||
// Subdiv2D requires a rect that strictly contains all points.
|
||||
const float margin = 1.f;
|
||||
Rect2f rect(minX - margin, minY - margin,
|
||||
(maxX - minX) + 2*margin,
|
||||
(maxY - minY) + 2*margin);
|
||||
Subdiv2D subdiv(rect);
|
||||
subdiv.insert(std::vector<Point2f>(keypoints.begin(), keypoints.end()));
|
||||
|
||||
// Map coordinates back to keypoint indices. Subdiv2D stores and returns the
|
||||
// exact float values we inserted, so direct comparison is safe here.
|
||||
std::map<std::pair<float, float>, size_t> ptToIdx;
|
||||
for (size_t i = 0; i < n; i++)
|
||||
ptToIdx[{keypoints[i].x, keypoints[i].y}] = i;
|
||||
|
||||
std::vector<Vec4f> edgeList;
|
||||
subdiv.getEdgeList(edgeList);
|
||||
|
||||
for (const Vec4f& e : edgeList)
|
||||
{
|
||||
auto it1 = ptToIdx.find({e[0], e[1]});
|
||||
auto it2 = ptToIdx.find({e[2], e[3]});
|
||||
// Edges involving the virtual bounding-rect vertices won't be in ptToIdx.
|
||||
if (it1 == ptToIdx.end() || it2 == ptToIdx.end())
|
||||
continue;
|
||||
|
||||
size_t i = it1->second;
|
||||
size_t j = it2->second;
|
||||
if (i == j)
|
||||
continue;
|
||||
if (i > j)
|
||||
std::swap(i, j);
|
||||
|
||||
Point2f vec = keypoints[i] - keypoints[j];
|
||||
double distSq = (double)vec.x*vec.x + (double)vec.y*vec.y;
|
||||
|
||||
bool isRNG = true;
|
||||
for (size_t k = 0; k < n; k++)
|
||||
{
|
||||
if (k == i || k == j)
|
||||
continue;
|
||||
Point2f d1 = keypoints[i] - keypoints[k];
|
||||
Point2f d2 = keypoints[j] - keypoints[k];
|
||||
double d1Sq = (double)d1.x*d1.x + (double)d1.y*d1.y;
|
||||
double d2Sq = (double)d2.x*d2.x + (double)d2.y*d2.y;
|
||||
if (d1Sq < distSq && d2Sq < distSq)
|
||||
{
|
||||
isRNG = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (isRNG)
|
||||
{
|
||||
rng.addEdge(i, j);
|
||||
// Push both directions; findBasis needs the full set to cluster into
|
||||
// the 4 groups (two grid axes and their negatives) via k-means.
|
||||
vectors.push_back(keypoints[i] - keypoints[j]);
|
||||
vectors.push_back(keypoints[j] - keypoints[i]);
|
||||
if (drawImage != 0)
|
||||
{
|
||||
line(*drawImage, keypoints[i], keypoints[j], Scalar(255, 0, 0), 2);
|
||||
circle(*drawImage, keypoints[i], 3, Scalar(0, 0, 255), -1);
|
||||
circle(*drawImage, keypoints[j], 3, Scalar(0, 0, 255), -1);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
size_t CirclesGridFinder::findLongestPath(std::vector<Graph> &basisGraphs, Path &bestPath)
|
||||
@@ -1285,45 +1296,93 @@ size_t CirclesGridFinder::findLongestPath(std::vector<Graph> &basisGraphs, Path
|
||||
std::vector<int> confidences;
|
||||
|
||||
size_t bestGraphIdx = 0;
|
||||
const int infinity = -1;
|
||||
for (size_t graphIdx = 0; graphIdx < basisGraphs.size(); graphIdx++)
|
||||
{
|
||||
const Graph &g = basisGraphs[graphIdx];
|
||||
Mat distanceMatrix;
|
||||
g.floydWarshall(distanceMatrix, infinity);
|
||||
Mat predecessorMatrix;
|
||||
computePredecessorMatrix(distanceMatrix, (int)g.getVerticesCount(), predecessorMatrix);
|
||||
const int n = (int)g.getVerticesCount();
|
||||
|
||||
double maxVal;
|
||||
Point maxLoc;
|
||||
minMaxLoc(distanceMatrix, 0, &maxVal, 0, &maxLoc);
|
||||
// BFS from every vertex to find the diameter (longest shortest path).
|
||||
// basisGraphs are sparse -- each vertex connects only to grid neighbors in
|
||||
// one direction -- so this is O(N^2) vs Floyd-Warshall's O(N^3).
|
||||
std::vector<int> dist(n);
|
||||
std::queue<int> q;
|
||||
|
||||
if (maxVal > longestPaths[0].length)
|
||||
int maxDist = 0;
|
||||
int srcBest = 0, dstBest = 0;
|
||||
|
||||
for (int src = 0; src < n; src++)
|
||||
{
|
||||
std::fill(dist.begin(), dist.end(), -1);
|
||||
dist[src] = 0;
|
||||
q.push(src);
|
||||
while (!q.empty())
|
||||
{
|
||||
int v = q.front(); q.pop();
|
||||
for (size_t nb : g.getNeighbors((size_t)v))
|
||||
{
|
||||
int u = (int)nb;
|
||||
if (dist[u] < 0)
|
||||
{
|
||||
dist[u] = dist[v] + 1;
|
||||
q.push(u);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int dst = 0; dst < n; dst++)
|
||||
{
|
||||
if (dist[dst] > maxDist)
|
||||
{
|
||||
maxDist = dist[dst];
|
||||
srcBest = src;
|
||||
dstBest = dst;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (maxDist > longestPaths[0].length)
|
||||
{
|
||||
longestPaths.clear();
|
||||
confidences.clear();
|
||||
bestGraphIdx = graphIdx;
|
||||
}
|
||||
if (longestPaths.empty() || (maxVal == longestPaths[0].length && graphIdx == bestGraphIdx))
|
||||
if (longestPaths.empty() || (maxDist == longestPaths[0].length && graphIdx == bestGraphIdx))
|
||||
{
|
||||
Path path = Path(maxLoc.x, maxLoc.y, cvRound(maxVal));
|
||||
CV_Assert(maxLoc.x >= 0 && maxLoc.y >= 0)
|
||||
;
|
||||
size_t id1 = static_cast<size_t> (maxLoc.x);
|
||||
size_t id2 = static_cast<size_t> (maxLoc.y);
|
||||
computeShortestPath(predecessorMatrix, id1, id2, path.vertices);
|
||||
Path path = Path(srcBest, dstBest, maxDist);
|
||||
|
||||
// BFS again from srcBest to reconstruct the path to dstBest
|
||||
std::vector<int> pred(n, -1);
|
||||
std::fill(dist.begin(), dist.end(), -1);
|
||||
dist[srcBest] = 0;
|
||||
q.push(srcBest);
|
||||
while (!q.empty())
|
||||
{
|
||||
int v = q.front(); q.pop();
|
||||
for (size_t nb : g.getNeighbors((size_t)v))
|
||||
{
|
||||
int u = (int)nb;
|
||||
if (dist[u] < 0)
|
||||
{
|
||||
dist[u] = dist[v] + 1;
|
||||
pred[u] = v;
|
||||
q.push(u);
|
||||
}
|
||||
}
|
||||
}
|
||||
std::vector<size_t> pathVertices;
|
||||
for (int cur = dstBest; cur != srcBest; cur = pred[cur])
|
||||
pathVertices.push_back((size_t)cur);
|
||||
pathVertices.push_back((size_t)srcBest);
|
||||
std::reverse(pathVertices.begin(), pathVertices.end());
|
||||
path.vertices = pathVertices;
|
||||
|
||||
longestPaths.push_back(path);
|
||||
|
||||
int conf = 0;
|
||||
for (int v2 = 0; v2 < (int)path.vertices.size(); v2++)
|
||||
{
|
||||
conf += (int)basisGraphs[1 - (int)graphIdx].getDegree(v2);
|
||||
}
|
||||
conf += (int)basisGraphs[1 - (int)graphIdx].getDegree(path.vertices[v2]);
|
||||
confidences.push_back(conf);
|
||||
}
|
||||
}
|
||||
//if( bestGraphIdx != 0 )
|
||||
//CV_Error( 0, "" );
|
||||
|
||||
int maxConf = -1;
|
||||
int bestPathIdx = -1;
|
||||
@@ -1336,7 +1395,6 @@ size_t CirclesGridFinder::findLongestPath(std::vector<Graph> &basisGraphs, Path
|
||||
}
|
||||
}
|
||||
|
||||
//int bestPathIdx = rand() % longestPaths.size();
|
||||
bestPath = longestPaths.at(bestPathIdx);
|
||||
bool needReverse = (bestGraphIdx == 0 && keypoints[bestPath.lastVertex].x < keypoints[bestPath.firstVertex].x)
|
||||
|| (bestGraphIdx == 1 && keypoints[bestPath.lastVertex].y < keypoints[bestPath.firstVertex].y);
|
||||
|
||||
@@ -517,7 +517,7 @@ void filterHomographyDecompByVisibleRefpoints(InputArrayOfArrays _rotations,
|
||||
CV_Assert(pointsMask.empty() || pointsMask.checkVector(1, CV_8U) == npoints);
|
||||
const uchar* pointsMaskPtr = pointsMask.data;
|
||||
|
||||
std::vector<uchar> solutionMask(nsolutions, (uchar)1);
|
||||
AutoBuffer<uchar> solutionMask(nsolutions, (uchar)1);
|
||||
std::vector<Mat> normals(nsolutions);
|
||||
std::vector<Mat> rotnorm(nsolutions);
|
||||
Mat R;
|
||||
@@ -559,7 +559,8 @@ void filterHomographyDecompByVisibleRefpoints(InputArrayOfArrays _rotations,
|
||||
if( solutionMask[i] )
|
||||
possibleSolutions.push_back(i);
|
||||
|
||||
Mat(possibleSolutions).copyTo(_possibleSolutions);
|
||||
constexpr int cvType = traits::Type<decltype(possibleSolutions)::value_type>::value;
|
||||
Mat(static_cast<int>(possibleSolutions.size()), 1, cvType, possibleSolutions.data()).copyTo(_possibleSolutions);
|
||||
}
|
||||
|
||||
} //namespace cv
|
||||
|
||||
@@ -849,6 +849,97 @@ TEST(Calib3d_RotatedCirclesPatternDetector, issue_24964)
|
||||
EXPECT_LE(error, precise_success_error_level);
|
||||
}
|
||||
|
||||
// Generate a perfect W x H symmetric circle grid at the given spacing.
|
||||
// Points are returned in shuffled order so the detector can't rely on input ordering.
|
||||
static std::vector<Point2f> makeSyntheticSymmetricGrid(int cols, int rows, float spacing)
|
||||
{
|
||||
std::vector<Point2f> pts;
|
||||
pts.reserve(cols * rows);
|
||||
for (int r = 0; r < rows; r++)
|
||||
for (int c = 0; c < cols; c++)
|
||||
pts.push_back(Point2f(c * spacing, r * spacing));
|
||||
|
||||
cv::RNG& rng = cv::theRNG();
|
||||
for (int k = (int)pts.size() - 1; k > 0; k--)
|
||||
std::swap(pts[k], pts[rng.uniform(0, k + 1)]);
|
||||
|
||||
return pts;
|
||||
}
|
||||
|
||||
// Generate an asymmetric circle grid. Even rows start at x=0, odd rows are offset by spacing/2.
|
||||
static std::vector<Point2f> makeSyntheticAsymmetricGrid(int cols, int rows, float spacing)
|
||||
{
|
||||
std::vector<Point2f> pts;
|
||||
pts.reserve(cols * rows);
|
||||
for (int r = 0; r < rows; r++)
|
||||
for (int c = 0; c < cols; c++)
|
||||
pts.push_back(Point2f(c * spacing + (r % 2) * spacing * 0.5f, r * spacing * 0.5f));
|
||||
|
||||
cv::RNG& rng = cv::theRNG();
|
||||
for (int k = (int)pts.size() - 1; k > 0; k--)
|
||||
std::swap(pts[k], pts[rng.uniform(0, k + 1)]);
|
||||
|
||||
return pts;
|
||||
}
|
||||
|
||||
typedef testing::TestWithParam<Size> Calib3d_CirclesGrid_RNG_Symmetric;
|
||||
|
||||
TEST_P(Calib3d_CirclesGrid_RNG_Symmetric, synthetic)
|
||||
{
|
||||
// Verify that findCirclesGrid correctly detects synthetic perfect symmetric grids of
|
||||
// various sizes. This exercises the computeRNG path (Delaunay-based) end-to-end.
|
||||
const float spacing = 30.f;
|
||||
const Size gridSize = GetParam();
|
||||
|
||||
std::vector<Point2f> pts = makeSyntheticSymmetricGrid(gridSize.width, gridSize.height, spacing);
|
||||
|
||||
std::vector<Point2f> centers;
|
||||
bool found = findCirclesGrid(Mat(pts), gridSize, centers,
|
||||
CALIB_CB_SYMMETRIC_GRID, Ptr<FeatureDetector>());
|
||||
|
||||
ASSERT_TRUE(found) << "Symmetric grid " << gridSize.width << "x" << gridSize.height << " not detected";
|
||||
ASSERT_EQ((int)centers.size(), gridSize.area());
|
||||
for (const Point2f& c : centers)
|
||||
{
|
||||
bool matched = false;
|
||||
for (const Point2f& p : pts)
|
||||
if (cv::norm(c - p) < 1.f) { matched = true; break; }
|
||||
EXPECT_TRUE(matched) << "Detected center " << c << " does not match any input point "
|
||||
<< "for grid " << gridSize.width << "x" << gridSize.height;
|
||||
}
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/**/, Calib3d_CirclesGrid_RNG_Symmetric,
|
||||
testing::Values(Size(4, 4), Size(6, 5), Size(8, 6), Size(10, 8)));
|
||||
|
||||
typedef testing::TestWithParam<Size> Calib3d_CirclesGrid_RNG_Asymmetric;
|
||||
|
||||
TEST_P(Calib3d_CirclesGrid_RNG_Asymmetric, synthetic)
|
||||
{
|
||||
const float spacing = 30.f;
|
||||
const Size gridSize = GetParam();
|
||||
|
||||
std::vector<Point2f> pts = makeSyntheticAsymmetricGrid(gridSize.width, gridSize.height, spacing);
|
||||
|
||||
std::vector<Point2f> centers;
|
||||
bool found = findCirclesGrid(Mat(pts), gridSize, centers,
|
||||
CALIB_CB_ASYMMETRIC_GRID, Ptr<FeatureDetector>());
|
||||
|
||||
ASSERT_TRUE(found) << "Asymmetric grid " << gridSize.width << "x" << gridSize.height << " not detected";
|
||||
ASSERT_EQ((int)centers.size(), gridSize.area());
|
||||
for (const Point2f& c : centers)
|
||||
{
|
||||
bool matched = false;
|
||||
for (const Point2f& p : pts)
|
||||
if (cv::norm(c - p) < 1.f) { matched = true; break; }
|
||||
EXPECT_TRUE(matched) << "Detected center " << c << " does not match any input point "
|
||||
<< "for grid " << gridSize.width << "x" << gridSize.height;
|
||||
}
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/**/, Calib3d_CirclesGrid_RNG_Asymmetric,
|
||||
testing::Values(Size(4, 6), Size(5, 8)));
|
||||
|
||||
TEST(Calib3d_CornerOrdering, issue_26830) {
|
||||
const cv::String dataDir = string(TS::ptr()->get_data_path()) + "cv/cameracalibration/";
|
||||
const cv::Mat image = cv::imread(dataDir + "checkerboard_marker_white.png");
|
||||
|
||||
@@ -1,19 +1,21 @@
|
||||
set(the_description "The Core Functionality")
|
||||
|
||||
ocv_add_dispatched_file(mathfuncs_core SSE2 AVX AVX2 LASX)
|
||||
ocv_add_dispatched_file(stat SSE4_2 AVX2 LASX)
|
||||
ocv_add_dispatched_file(stat SSE4_2 AVX2 AVX512_SKX AVX512_ICL LASX)
|
||||
ocv_add_dispatched_file(arithm SSE2 SSE4_1 AVX2 VSX3 LASX)
|
||||
ocv_add_dispatched_file(convert SSE2 AVX2 VSX3 LASX)
|
||||
ocv_add_dispatched_file(convert_scale SSE2 AVX2 LASX)
|
||||
ocv_add_dispatched_file(count_non_zero SSE2 AVX2 LASX)
|
||||
ocv_add_dispatched_file(count_non_zero SSE2 AVX2 AVX512_SKX AVX512_ICL LASX)
|
||||
ocv_add_dispatched_file(has_non_zero SSE2 AVX2 LASX )
|
||||
ocv_add_dispatched_file(matmul SSE2 SSE4_1 AVX2 AVX512_SKX NEON_DOTPROD LASX)
|
||||
ocv_add_dispatched_file(mean SSE2 AVX2 LASX)
|
||||
ocv_add_dispatched_file(mean SSE2 AVX2 AVX512_SKX AVX512_ICL LASX)
|
||||
ocv_add_dispatched_file(merge SSE2 AVX2 LASX)
|
||||
ocv_add_dispatched_file(minmax SSE2 SSE4_1 AVX2 AVX512_SKX AVX512_ICL LASX)
|
||||
ocv_add_dispatched_file(split SSE2 AVX2 LASX)
|
||||
ocv_add_dispatched_file(sum SSE2 AVX2 LASX)
|
||||
ocv_add_dispatched_file(sum SSE2 AVX2 AVX512_SKX AVX512_ICL LASX)
|
||||
ocv_add_dispatched_file(reduce SSE2 SSSE3 AVX2 NEON_DOTPROD)
|
||||
ocv_add_dispatched_file(norm SSE2 SSE4_1 AVX AVX2 NEON_DOTPROD LASX)
|
||||
ocv_add_dispatched_file(norm SSE2 SSE4_1 AVX AVX2 AVX512_SKX AVX512_ICL NEON_DOTPROD LASX)
|
||||
ocv_add_dispatched_file(lut AVX512_ICL)
|
||||
|
||||
# dispatching for accuracy tests
|
||||
ocv_add_dispatched_file_force_all(test_intrin128 TEST SSE2 SSE3 SSSE3 SSE4_1 SSE4_2 AVX FP16 AVX2 AVX512_SKX)
|
||||
|
||||
@@ -1598,6 +1598,20 @@ inline v_uint8x32 v256_lut(const uchar* tab, const int* idx) { return v_reinterp
|
||||
inline v_uint8x32 v256_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v256_lut_pairs((const schar *)tab, idx)); }
|
||||
inline v_uint8x32 v256_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v256_lut_quads((const schar *)tab, idx)); }
|
||||
|
||||
inline v_uint8x32 v256_lut(const uchar* tab, const v_uint8x32& idx)
|
||||
{
|
||||
uchar CV_DECL_ALIGNED(32) indices[32], result[32];
|
||||
_mm256_store_si256((__m256i*)indices, idx.val);
|
||||
for (int i = 0; i < 32; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x32(_mm256_load_si256((const __m256i*)result));
|
||||
}
|
||||
inline v_int8x32 v256_lut(const schar* tab, const v_uint8x32& idx)
|
||||
{ return v_reinterpret_as_s8(v256_lut((const uchar *)tab, idx)); }
|
||||
|
||||
|
||||
inline v_uint8x32 v_lut(const uchar* tab, const v_uint8x32& idx) { return v256_lut(tab, idx); }
|
||||
inline v_int8x32 v_lut(const schar* tab, const v_uint8x32& idx) { return v256_lut(tab, idx); }
|
||||
|
||||
inline v_int16x16 v256_lut(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x16(_mm256_setr_epi16(tab[idx[0]], tab[idx[1]], tab[idx[ 2]], tab[idx[ 3]], tab[idx[ 4]], tab[idx[ 5]], tab[idx[ 6]], tab[idx[ 7]],
|
||||
|
||||
@@ -1634,6 +1634,122 @@ inline v_uint8x64 v512_lut(const uchar* tab, const int* idx) { return v_reinterp
|
||||
inline v_uint8x64 v512_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v512_lut_pairs((const schar *)tab, idx)); }
|
||||
inline v_uint8x64 v512_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v512_lut_quads((const schar *)tab, idx)); }
|
||||
|
||||
// Byte-indexed LUT: takes v_uint8x64 byte indices instead of int* indices
|
||||
// Table must have at least 256 entries (byte indices cover 0-255)
|
||||
// VBMI path uses vpermb (register-to-register), fallback uses gather
|
||||
inline v_uint8x64 v512_lut(const uchar* tab, const v_uint8x64& idx)
|
||||
{
|
||||
#if CV_AVX_512VBMI
|
||||
const __m512i lut0 = _mm512_loadu_si512(tab);
|
||||
const __m512i lut1 = _mm512_loadu_si512(tab + 64);
|
||||
const __m512i lut2 = _mm512_loadu_si512(tab + 128);
|
||||
const __m512i lut3 = _mm512_loadu_si512(tab + 192);
|
||||
|
||||
__m512i idx6 = _mm512_and_si512(idx.val, _mm512_set1_epi8(0x3F));
|
||||
|
||||
__m512i r0 = _mm512_permutexvar_epi8(idx6, lut0);
|
||||
__m512i r1 = _mm512_permutexvar_epi8(idx6, lut1);
|
||||
__m512i r2 = _mm512_permutexvar_epi8(idx6, lut2);
|
||||
__m512i r3 = _mm512_permutexvar_epi8(idx6, lut3);
|
||||
|
||||
__mmask64 k6 = _mm512_test_epi8_mask(idx.val, _mm512_set1_epi8(0x40));
|
||||
__mmask64 k7 = _mm512_test_epi8_mask(idx.val, _mm512_set1_epi8((char)0x80));
|
||||
|
||||
__m512i low = _mm512_mask_blend_epi8(k6, r0, r1);
|
||||
__m512i high = _mm512_mask_blend_epi8(k6, r2, r3);
|
||||
return v_uint8x64(_mm512_mask_blend_epi8(k7, low, high));
|
||||
#else
|
||||
uchar CV_DECL_ALIGNED(64) indices[64], result[64];
|
||||
_mm512_store_si512((__m512i*)indices, idx.val);
|
||||
for (int i = 0; i < 64; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x64(_mm512_load_si512((const __m512i*)result));
|
||||
#endif
|
||||
}
|
||||
inline v_int8x64 v512_lut(const schar* tab, const v_uint8x64& idx)
|
||||
{ return v_reinterpret_as_s8(v512_lut((const uchar *)tab, idx)); }
|
||||
|
||||
// Universal v_lut overloads for vector byte indices (aliases to v512_lut)
|
||||
inline v_uint8x64 v_lut(const uchar* tab, const v_uint8x64& idx)
|
||||
{ return v512_lut(tab, idx); }
|
||||
|
||||
inline v_int8x64 v_lut(const schar* tab, const v_uint8x64& idx)
|
||||
{ return v512_lut(tab, idx); }
|
||||
|
||||
// Byte-indexed pair LUT: 32 byte indices (lower half of idx) → 32 pairs = 64 bytes
|
||||
inline v_uint8x64 v512_lut_pairs(const uchar* tab, const v_uint8x64& idx)
|
||||
{
|
||||
#if CV_AVX_512VBMI
|
||||
const __m512i lut0 = _mm512_loadu_si512(tab);
|
||||
const __m512i lut1 = _mm512_loadu_si512(tab + 64);
|
||||
const __m512i lut2 = _mm512_loadu_si512(tab + 128);
|
||||
const __m512i lut3 = _mm512_loadu_si512(tab + 192);
|
||||
|
||||
// Expand 32 byte indices to interleaved pairs: (i0, i0+1, i1, i1+1, ...)
|
||||
__m512i idx16 = _mm512_cvtepu8_epi16(_mm512_castsi512_si256(idx.val));
|
||||
__m512i idx_dup = _mm512_or_si512(idx16, _mm512_slli_epi16(idx16, 8));
|
||||
__m512i idx_pairs = _mm512_add_epi8(idx_dup, _mm512_set1_epi16(0x0100));
|
||||
|
||||
__m512i idx6 = _mm512_and_si512(idx_pairs, _mm512_set1_epi8(0x3F));
|
||||
|
||||
__m512i r0 = _mm512_permutexvar_epi8(idx6, lut0);
|
||||
__m512i r1 = _mm512_permutexvar_epi8(idx6, lut1);
|
||||
__m512i r2 = _mm512_permutexvar_epi8(idx6, lut2);
|
||||
__m512i r3 = _mm512_permutexvar_epi8(idx6, lut3);
|
||||
|
||||
__mmask64 k6 = _mm512_test_epi8_mask(idx_pairs, _mm512_set1_epi8(0x40));
|
||||
__mmask64 k7 = _mm512_test_epi8_mask(idx_pairs, _mm512_set1_epi8((char)0x80));
|
||||
|
||||
__m512i low = _mm512_mask_blend_epi8(k6, r0, r1);
|
||||
__m512i high = _mm512_mask_blend_epi8(k6, r2, r3);
|
||||
return v_uint8x64(_mm512_mask_blend_epi8(k7, low, high));
|
||||
#else
|
||||
uchar CV_DECL_ALIGNED(64) indices[64], result[64];
|
||||
_mm512_store_si512((__m512i*)indices, idx.val);
|
||||
for (int i = 0; i < 64; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x64(_mm512_load_si512((const __m512i*)result));
|
||||
#endif
|
||||
}
|
||||
inline v_int8x64 v512_lut_pairs(const schar* tab, const v_uint8x64& idx)
|
||||
{ return v_reinterpret_as_s8(v512_lut_pairs((const uchar *)tab, idx)); }
|
||||
|
||||
// Byte-indexed quad LUT: 16 byte indices (lower 16 of idx) → 16 quads = 64 bytes
|
||||
inline v_uint8x64 v512_lut_quads(const uchar* tab, const v_uint8x64& idx)
|
||||
{
|
||||
#if CV_AVX_512VBMI
|
||||
const __m512i lut0 = _mm512_loadu_si512(tab);
|
||||
const __m512i lut1 = _mm512_loadu_si512(tab + 64);
|
||||
const __m512i lut2 = _mm512_loadu_si512(tab + 128);
|
||||
const __m512i lut3 = _mm512_loadu_si512(tab + 192);
|
||||
|
||||
// Expand 16 byte indices to quad offsets: (i0, i0+1, i0+2, i0+3, i1, ...)
|
||||
__m512i idx32 = _mm512_cvtepu8_epi32(_mm512_castsi512_si128(idx.val));
|
||||
__m512i t1 = _mm512_or_si512(idx32, _mm512_slli_epi32(idx32, 8));
|
||||
__m512i idx_bcast = _mm512_or_si512(t1, _mm512_slli_epi32(t1, 16));
|
||||
__m512i idx_quads = _mm512_add_epi8(idx_bcast, _mm512_set1_epi32(0x03020100));
|
||||
|
||||
__m512i idx6 = _mm512_and_si512(idx_quads, _mm512_set1_epi8(0x3F));
|
||||
|
||||
__m512i r0 = _mm512_permutexvar_epi8(idx6, lut0);
|
||||
__m512i r1 = _mm512_permutexvar_epi8(idx6, lut1);
|
||||
__m512i r2 = _mm512_permutexvar_epi8(idx6, lut2);
|
||||
__m512i r3 = _mm512_permutexvar_epi8(idx6, lut3);
|
||||
|
||||
__mmask64 k6 = _mm512_test_epi8_mask(idx_quads, _mm512_set1_epi8(0x40));
|
||||
__mmask64 k7 = _mm512_test_epi8_mask(idx_quads, _mm512_set1_epi8((char)0x80));
|
||||
|
||||
__m512i low = _mm512_mask_blend_epi8(k6, r0, r1);
|
||||
__m512i high = _mm512_mask_blend_epi8(k6, r2, r3);
|
||||
return v_uint8x64(_mm512_mask_blend_epi8(k7, low, high));
|
||||
#else
|
||||
uchar CV_DECL_ALIGNED(64) indices[64], result[64];
|
||||
_mm512_store_si512((__m512i*)indices, idx.val);
|
||||
for (int i = 0; i < 64; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x64(_mm512_load_si512((const __m512i*)result));
|
||||
#endif
|
||||
}
|
||||
inline v_int8x64 v512_lut_quads(const schar* tab, const v_uint8x64& idx)
|
||||
{ return v_reinterpret_as_s8(v512_lut_quads((const uchar *)tab, idx)); }
|
||||
|
||||
inline v_int16x32 v512_lut(const short* tab, const int* idx)
|
||||
{
|
||||
__m256i p0 = _mm512_cvtepi32_epi16(_mm512_i32gather_epi32(_mm512_loadu_si512((const __m512i*)idx ), (const int *)tab, 2));
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
// Copyright (C) 2009, Willow Garage Inc., all rights reserved.
|
||||
// Copyright (C) 2013, OpenCV Foundation, all rights reserved.
|
||||
// Copyright (C) 2015, Itseez Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without modification,
|
||||
@@ -2714,6 +2715,16 @@ template<typename _Tp> inline v_reg<_Tp, simd128_width / sizeof(_Tp)> v_lut_quad
|
||||
return c;
|
||||
}
|
||||
|
||||
template<int n> inline v_reg<uchar, n> v_lut(const uchar* tab, const v_reg<uchar, n>& idx)
|
||||
{
|
||||
v_reg<uchar, n> c;
|
||||
for( int i = 0; i < n; i++ )
|
||||
c.s[i] = tab[idx.s[i]];
|
||||
return c;
|
||||
}
|
||||
template<int n> inline v_reg<schar, n> v_lut(const schar* tab, const v_reg<uchar, n>& idx)
|
||||
{ return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx)); }
|
||||
|
||||
template<int n> inline v_reg<int, n> v_lut(const int* tab, const v_reg<int, n>& idx)
|
||||
{
|
||||
v_reg<int, n> c;
|
||||
|
||||
@@ -1673,6 +1673,21 @@ inline v_uint8x32 v256_lut(const uchar* tab, const int* idx) { return v_reinterp
|
||||
inline v_uint8x32 v256_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v256_lut_pairs((const schar *)tab, idx)); }
|
||||
inline v_uint8x32 v256_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v256_lut_quads((const schar *)tab, idx)); }
|
||||
|
||||
// Byte-indexed LUT: 32 byte indices in a vector -> 32 looked-up bytes
|
||||
inline v_uint8x32 v256_lut(const uchar* tab, const v_uint8x32& idx)
|
||||
{
|
||||
uchar CV_DECL_ALIGNED(32) indices[32], result[32];
|
||||
__lasx_xvst(idx.val, indices, 0);
|
||||
for (int i = 0; i < 32; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x32(__lasx_xvld(result, 0));
|
||||
}
|
||||
inline v_int8x32 v256_lut(const schar* tab, const v_uint8x32& idx)
|
||||
{ return v_reinterpret_as_s8(v256_lut((const uchar*)tab, idx)); }
|
||||
|
||||
// Universal v_lut overloads for vector byte indices (aliases to v256_lut)
|
||||
inline v_uint8x32 v_lut(const uchar* tab, const v_uint8x32& idx) { return v256_lut(tab, idx); }
|
||||
inline v_int8x32 v_lut(const schar* tab, const v_uint8x32& idx) { return v256_lut(tab, idx); }
|
||||
|
||||
inline v_int16x16 v256_lut(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x16(_v256_setr_h(tab[idx[ 0]], tab[idx[ 1]], tab[idx[ 2]], tab[idx[ 3]], tab[idx[ 4]],
|
||||
|
||||
@@ -1443,6 +1443,16 @@ inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx)
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx)
|
||||
{ return v_reinterpret_as_u8(v_lut_quads((const schar*)tab, idx)); }
|
||||
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const v_uint8x16& idx)
|
||||
{
|
||||
uchar CV_DECL_ALIGNED(16) indices[16], result[16];
|
||||
__lsx_vst(idx.val, indices, 0);
|
||||
for (int i = 0; i < 16; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x16(__lsx_vld(result, 0));
|
||||
}
|
||||
inline v_int8x16 v_lut(const schar* tab, const v_uint8x16& idx)
|
||||
{ return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x8(_v128_setr_h(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]],
|
||||
|
||||
@@ -1566,6 +1566,15 @@ inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((schar*)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((schar*)tab, idx)); }
|
||||
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const v_uint8x16& idx)
|
||||
{
|
||||
uchar CV_DECL_ALIGNED(16) indices[16], result[16];
|
||||
msa_st1q_u8(indices, idx.val);
|
||||
for (int i = 0; i < 16; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x16(msa_ld1q_u8(result));
|
||||
}
|
||||
inline v_int8x16 v_lut(const schar* tab, const v_uint8x16& idx)
|
||||
{ return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
// Copyright (C) 2009, Willow Garage Inc., all rights reserved.
|
||||
// Copyright (C) 2013, OpenCV Foundation, all rights reserved.
|
||||
// Copyright (C) 2015, Itseez Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without modification,
|
||||
@@ -2370,6 +2371,48 @@ inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((schar*)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((schar*)tab, idx)); }
|
||||
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const v_uint8x16& idx)
|
||||
{
|
||||
#if CV_NEON_AARCH64
|
||||
uint8x16_t index = idx.val;
|
||||
uint8x16_t idx6 = vandq_u8(index, vdupq_n_u8(0x3F));
|
||||
|
||||
uint8x16x4_t t0, t1, t2, t3;
|
||||
t0.val[0] = vld1q_u8(tab); t0.val[1] = vld1q_u8(tab + 16);
|
||||
t0.val[2] = vld1q_u8(tab + 32); t0.val[3] = vld1q_u8(tab + 48);
|
||||
uint8x16_t r0 = vqtbl4q_u8(t0, idx6);
|
||||
|
||||
t1.val[0] = vld1q_u8(tab + 64); t1.val[1] = vld1q_u8(tab + 80);
|
||||
t1.val[2] = vld1q_u8(tab + 96); t1.val[3] = vld1q_u8(tab + 112);
|
||||
uint8x16_t r1 = vqtbl4q_u8(t1, idx6);
|
||||
|
||||
t2.val[0] = vld1q_u8(tab + 128); t2.val[1] = vld1q_u8(tab + 144);
|
||||
t2.val[2] = vld1q_u8(tab + 160); t2.val[3] = vld1q_u8(tab + 176);
|
||||
uint8x16_t r2 = vqtbl4q_u8(t2, idx6);
|
||||
|
||||
t3.val[0] = vld1q_u8(tab + 192); t3.val[1] = vld1q_u8(tab + 208);
|
||||
t3.val[2] = vld1q_u8(tab + 224); t3.val[3] = vld1q_u8(tab + 240);
|
||||
uint8x16_t r3 = vqtbl4q_u8(t3, idx6);
|
||||
|
||||
uint8x16_t bit6 = vtstq_u8(index, vdupq_n_u8(0x40));
|
||||
uint8x16_t low = vbslq_u8(bit6, r1, r0);
|
||||
uint8x16_t high = vbslq_u8(bit6, r3, r2);
|
||||
|
||||
uint8x16_t bit7 = vtstq_u8(index, vdupq_n_u8(0x80));
|
||||
return v_uint8x16(vbslq_u8(bit7, high, low));
|
||||
#else
|
||||
uchar CV_DECL_ALIGNED(16) indices[16], result[16];
|
||||
vst1q_u8(indices, idx.val);
|
||||
for (int i = 0; i < 16; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x16(vld1q_u8(result));
|
||||
#endif
|
||||
}
|
||||
|
||||
inline v_int8x16 v_lut(const schar* tab, const v_uint8x16& idx)
|
||||
{
|
||||
return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx));
|
||||
}
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
short CV_DECL_ALIGNED(32) elems[8] =
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
// Copyright (C) 2015, PingTouGe Semiconductor Co., Ltd., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#ifndef OPENCV_HAL_INTRIN_RISCVV_HPP
|
||||
#define OPENCV_HAL_INTRIN_RISCVV_HPP
|
||||
@@ -1625,6 +1626,16 @@ inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((schar*)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((schar*)tab, idx)); }
|
||||
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const v_uint8x16& idx)
|
||||
{
|
||||
uchar CV_DECL_ALIGNED(16) indices[16], result[16];
|
||||
vse8_v_u8m1(indices, idx.val, 16);
|
||||
for (int i = 0; i < 16; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x16(vle8_v_u8m1(result, 16));
|
||||
}
|
||||
inline v_int8x16 v_lut(const schar* tab, const v_uint8x16& idx)
|
||||
{ return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
#if 0
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
|
||||
// The original implementation is contributed by HAN Liutong.
|
||||
// Copyright (C) 2022, Institute of Software, Chinese Academy of Sciences.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#ifndef OPENCV_HAL_INTRIN_RVV_SCALABLE_HPP
|
||||
#define OPENCV_HAL_INTRIN_RVV_SCALABLE_HPP
|
||||
@@ -534,6 +535,13 @@ inline void v_lut_deinterleave(const double* tab, const v_int32& vidx, v_float64
|
||||
inline v_uint8 v_lut(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut((schar*)tab, idx)); }
|
||||
inline v_uint8 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((schar*)tab, idx)); }
|
||||
inline v_uint8 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((schar*)tab, idx)); }
|
||||
|
||||
// Byte-indexed LUT: vector byte indices -> looked-up bytes (uses RVV indexed load)
|
||||
inline v_uint8 v_lut(const uchar* tab, const v_uint8& idx)
|
||||
{ return __riscv_vluxei8_v_u8m2(tab, idx, VTraits<v_uint8>::vlanes()); }
|
||||
inline v_int8 v_lut(const schar* tab, const v_uint8& idx)
|
||||
{ return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx)); }
|
||||
|
||||
inline v_uint16 v_lut(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut((short*)tab, idx)); }
|
||||
inline v_uint16 v_lut_pairs(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_pairs((short*)tab, idx)); }
|
||||
inline v_uint16 v_lut_quads(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_quads((short*)tab, idx)); }
|
||||
@@ -931,7 +939,7 @@ inline scalartype v_reduce_sum(const _Tpvec& a) \
|
||||
}
|
||||
OPENCV_HAL_IMPL_RVV_REDUCE_SUM_FP(v_float32, v_float32, vfloat32m1_t, float, f32, VTraits<v_float32>::vlanes())
|
||||
#if CV_SIMD_SCALABLE_64F
|
||||
OPENCV_HAL_IMPL_RVV_REDUCE_SUM_FP(v_float64, v_float64, vfloat64m1_t, float, f64, VTraits<v_float64>::vlanes())
|
||||
OPENCV_HAL_IMPL_RVV_REDUCE_SUM_FP(v_float64, v_float64, vfloat64m1_t, double, f64, VTraits<v_float64>::vlanes())
|
||||
#endif
|
||||
|
||||
#define OPENCV_HAL_IMPL_RVV_REDUCE(_Tpvec, _nTpvec, func, scalartype, suffix, vl, red) \
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
// Copyright (C) 2009, Willow Garage Inc., all rights reserved.
|
||||
// Copyright (C) 2013, OpenCV Foundation, all rights reserved.
|
||||
// Copyright (C) 2015, Itseez Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without modification,
|
||||
@@ -3060,71 +3061,55 @@ inline v_float64x2 v_cvt_f64(const v_int64x2& v)
|
||||
|
||||
inline v_int8x16 v_lut(const schar* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int8x16(_mm_setr_epi8(tab[idx[0]], tab[idx[1]], tab[idx[ 2]], tab[idx[ 3]], tab[idx[ 4]], tab[idx[ 5]], tab[idx[ 6]], tab[idx[ 7]],
|
||||
tab[idx[8]], tab[idx[9]], tab[idx[10]], tab[idx[11]], tab[idx[12]], tab[idx[13]], tab[idx[14]], tab[idx[15]]));
|
||||
#else
|
||||
return v_int8x16(_mm_setr_epi64(
|
||||
_mm_setr_pi8(tab[idx[0]], tab[idx[1]], tab[idx[ 2]], tab[idx[ 3]], tab[idx[ 4]], tab[idx[ 5]], tab[idx[ 6]], tab[idx[ 7]]),
|
||||
_mm_setr_pi8(tab[idx[8]], tab[idx[9]], tab[idx[10]], tab[idx[11]], tab[idx[12]], tab[idx[13]], tab[idx[14]], tab[idx[15]])
|
||||
));
|
||||
#endif
|
||||
return v_int8x16(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]],
|
||||
tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]],
|
||||
tab[idx[8]], tab[idx[9]], tab[idx[10]], tab[idx[11]],
|
||||
tab[idx[12]], tab[idx[13]], tab[idx[14]], tab[idx[15]]);
|
||||
}
|
||||
inline v_int8x16 v_lut_pairs(const schar* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int8x16(_mm_setr_epi16(*(const short*)(tab + idx[0]), *(const short*)(tab + idx[1]), *(const short*)(tab + idx[2]), *(const short*)(tab + idx[3]),
|
||||
*(const short*)(tab + idx[4]), *(const short*)(tab + idx[5]), *(const short*)(tab + idx[6]), *(const short*)(tab + idx[7])));
|
||||
#else
|
||||
return v_int8x16(_mm_setr_epi64(
|
||||
_mm_setr_pi16(*(const short*)(tab + idx[0]), *(const short*)(tab + idx[1]), *(const short*)(tab + idx[2]), *(const short*)(tab + idx[3])),
|
||||
_mm_setr_pi16(*(const short*)(tab + idx[4]), *(const short*)(tab + idx[5]), *(const short*)(tab + idx[6]), *(const short*)(tab + idx[7]))
|
||||
));
|
||||
#endif
|
||||
return v_int8x16(tab[idx[0]], tab[idx[0] + 1], tab[idx[1]], tab[idx[1] + 1],
|
||||
tab[idx[2]], tab[idx[2] + 1], tab[idx[3]], tab[idx[3] + 1],
|
||||
tab[idx[4]], tab[idx[4] + 1], tab[idx[5]], tab[idx[5] + 1],
|
||||
tab[idx[6]], tab[idx[6] + 1], tab[idx[7]], tab[idx[7] + 1]);
|
||||
}
|
||||
inline v_int8x16 v_lut_quads(const schar* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int8x16(_mm_setr_epi32(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1]),
|
||||
*(const int*)(tab + idx[2]), *(const int*)(tab + idx[3])));
|
||||
#else
|
||||
return v_int8x16(_mm_setr_epi64(
|
||||
_mm_setr_pi32(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1])),
|
||||
_mm_setr_pi32(*(const int*)(tab + idx[2]), *(const int*)(tab + idx[3]))
|
||||
));
|
||||
#endif
|
||||
return v_int8x16(
|
||||
tab[idx[0]], tab[idx[0] + 1], tab[idx[0] + 2], tab[idx[0] + 3],
|
||||
tab[idx[1]], tab[idx[1] + 1], tab[idx[1] + 2], tab[idx[1] + 3],
|
||||
tab[idx[2]], tab[idx[2] + 1], tab[idx[2] + 2], tab[idx[2] + 3],
|
||||
tab[idx[3]], tab[idx[3] + 1], tab[idx[3] + 2], tab[idx[3] + 3]);
|
||||
}
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut((const schar *)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((const schar *)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((const schar *)tab, idx)); }
|
||||
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const v_uint8x16& idx)
|
||||
{
|
||||
uchar CV_DECL_ALIGNED(16) indices[16], result[16];
|
||||
_mm_store_si128((__m128i*)indices, idx.val);
|
||||
for (int i = 0; i < 16; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x16(_mm_load_si128((const __m128i*)result));
|
||||
}
|
||||
inline v_int8x16 v_lut(const schar* tab, const v_uint8x16& idx)
|
||||
{ return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int16x8(_mm_setr_epi16(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]],
|
||||
tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]]));
|
||||
#else
|
||||
return v_int16x8(_mm_setr_epi64(
|
||||
_mm_setr_pi16(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]),
|
||||
_mm_setr_pi16(tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]])
|
||||
));
|
||||
#endif
|
||||
return v_int16x8(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]],
|
||||
tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]]);
|
||||
}
|
||||
inline v_int16x8 v_lut_pairs(const short* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int16x8(_mm_setr_epi32(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1]),
|
||||
*(const int*)(tab + idx[2]), *(const int*)(tab + idx[3])));
|
||||
#else
|
||||
return v_int16x8(_mm_setr_epi64(
|
||||
_mm_setr_pi32(*(const int*)(tab + idx[0]), *(const int*)(tab + idx[1])),
|
||||
_mm_setr_pi32(*(const int*)(tab + idx[2]), *(const int*)(tab + idx[3]))
|
||||
));
|
||||
#endif
|
||||
return v_int16x8(tab[idx[0]], tab[idx[0] + 1], tab[idx[1]], tab[idx[1] + 1],
|
||||
tab[idx[2]], tab[idx[2] + 1], tab[idx[3]], tab[idx[3] + 1]);
|
||||
}
|
||||
inline v_int16x8 v_lut_quads(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x8(_mm_set_epi64x(*(const int64_t*)(tab + idx[1]), *(const int64_t*)(tab + idx[0])));
|
||||
return v_int16x8(tab[idx[0]], tab[idx[0] + 1], tab[idx[0] + 2],
|
||||
tab[idx[0] + 3], tab[idx[1]], tab[idx[1] + 1],
|
||||
tab[idx[1] + 2], tab[idx[1] + 3]);
|
||||
}
|
||||
inline v_uint16x8 v_lut(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut((const short *)tab, idx)); }
|
||||
inline v_uint16x8 v_lut_pairs(const ushort* tab, const int* idx) { return v_reinterpret_as_u16(v_lut_pairs((const short *)tab, idx)); }
|
||||
@@ -3132,15 +3117,7 @@ inline v_uint16x8 v_lut_quads(const ushort* tab, const int* idx) { return v_rein
|
||||
|
||||
inline v_int32x4 v_lut(const int* tab, const int* idx)
|
||||
{
|
||||
#if defined(_MSC_VER)
|
||||
return v_int32x4(_mm_setr_epi32(tab[idx[0]], tab[idx[1]],
|
||||
tab[idx[2]], tab[idx[3]]));
|
||||
#else
|
||||
return v_int32x4(_mm_setr_epi64(
|
||||
_mm_setr_pi32(tab[idx[0]], tab[idx[1]]),
|
||||
_mm_setr_pi32(tab[idx[2]], tab[idx[3]])
|
||||
));
|
||||
#endif
|
||||
return v_int32x4(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]]);
|
||||
}
|
||||
inline v_int32x4 v_lut_pairs(const int* tab, const int* idx)
|
||||
{
|
||||
|
||||
@@ -1175,6 +1175,16 @@ inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((const schar*)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((const schar*)tab, idx)); }
|
||||
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const v_uint8x16& idx)
|
||||
{
|
||||
uchar CV_DECL_ALIGNED(16) indices[16], result[16];
|
||||
vsx_st(idx.val, 0, indices);
|
||||
for (int i = 0; i < 16; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x16(vsx_ld(0, result));
|
||||
}
|
||||
inline v_int8x16 v_lut(const schar* tab, const v_uint8x16& idx)
|
||||
{ return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x8(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]], tab[idx[4]], tab[idx[5]], tab[idx[6]], tab[idx[7]]);
|
||||
|
||||
@@ -2584,6 +2584,16 @@ inline v_uint8x16 v_lut(const uchar* tab, const int* idx) { return v_reinterpret
|
||||
inline v_uint8x16 v_lut_pairs(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_pairs((const schar *)tab, idx)); }
|
||||
inline v_uint8x16 v_lut_quads(const uchar* tab, const int* idx) { return v_reinterpret_as_u8(v_lut_quads((const schar *)tab, idx)); }
|
||||
|
||||
inline v_uint8x16 v_lut(const uchar* tab, const v_uint8x16& idx)
|
||||
{
|
||||
uchar CV_DECL_ALIGNED(16) indices[16], result[16];
|
||||
wasm_v128_store(indices, idx.val);
|
||||
for (int i = 0; i < 16; i++) result[i] = tab[indices[i]];
|
||||
return v_uint8x16(wasm_v128_load(result));
|
||||
}
|
||||
inline v_int8x16 v_lut(const schar* tab, const v_uint8x16& idx)
|
||||
{ return v_reinterpret_as_s8(v_lut((const uchar*)tab, idx)); }
|
||||
|
||||
inline v_int16x8 v_lut(const short* tab, const int* idx)
|
||||
{
|
||||
return v_int16x8(tab[idx[0]], tab[idx[1]], tab[idx[2]], tab[idx[3]],
|
||||
|
||||
@@ -555,6 +555,7 @@ public:
|
||||
Mat_<uchar> m2({2, 3}, {1, 2, 3, 4, 5, 6}); // 2x3 Mat
|
||||
|
||||
Mat_<double> R({2, 2}, {a, -b, b, a}); // from example
|
||||
\endcode
|
||||
*/
|
||||
template<typename _Tp> class MatCommaInitializer_
|
||||
{
|
||||
|
||||
@@ -211,7 +211,6 @@ T* allocSingletonNew() { return new(allocSingletonNewBuffer(sizeof(T))) T(); }
|
||||
// Temporary disabled named IPP region. Performance
|
||||
#define IPP_DISABLE_PERF_COPYMAKE 1 // performance variations
|
||||
#define IPP_DISABLE_PERF_TRUE_DIST_MT 1 // cv::distanceTransform OpenCV MT performance is better
|
||||
#define IPP_DISABLE_PERF_CANNY_MT 1 // cv::Canny OpenCV MT performance is better
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#include "ippversion.h"
|
||||
|
||||
@@ -6,9 +6,9 @@
|
||||
#define OPENCV_VERSION_HPP
|
||||
|
||||
#define CV_VERSION_MAJOR 4
|
||||
#define CV_VERSION_MINOR 14
|
||||
#define CV_VERSION_MINOR 15
|
||||
#define CV_VERSION_REVISION 0
|
||||
#define CV_VERSION_STATUS "-pre"
|
||||
#define CV_VERSION_STATUS "-dev"
|
||||
|
||||
#define CVAUX_STR_EXP(__A) #__A
|
||||
#define CVAUX_STR(__A) CVAUX_STR_EXP(__A)
|
||||
|
||||
@@ -1202,6 +1202,9 @@ typedef TestBaseWithParam<ReduceMinMaxParams> ReduceMinMaxFixture;
|
||||
OCL_PERF_TEST_P(ReduceMinMaxFixture, Reduce,
|
||||
::testing::Combine(OCL_TEST_SIZES,
|
||||
OCL_PERF_ENUM(std::make_pair<MatType, MatType>(CV_8UC1, CV_8UC1),
|
||||
std::make_pair<MatType, MatType>(CV_8UC4, CV_8UC4),
|
||||
std::make_pair<MatType, MatType>(CV_32FC1, CV_32FC1),
|
||||
std::make_pair<MatType, MatType>(CV_32FC3, CV_32FC3),
|
||||
std::make_pair<MatType, MatType>(CV_32FC4, CV_32FC4)),
|
||||
OCL_PERF_ENUM(0, 1),
|
||||
ReduceMinMaxOp::all()))
|
||||
|
||||
@@ -4,7 +4,7 @@ namespace opencv_test
|
||||
{
|
||||
using namespace perf;
|
||||
|
||||
#define TYPICAL_MAT_TYPES_ADWEIGHTED CV_8UC1, CV_8UC4, CV_8SC1, CV_16UC1, CV_16SC1, CV_32SC1
|
||||
#define TYPICAL_MAT_TYPES_ADWEIGHTED CV_8UC1, CV_8UC4, CV_8SC1, CV_16UC1, CV_16SC1, CV_32SC1, CV_32FC1
|
||||
#define TYPICAL_MATS_ADWEIGHTED testing::Combine(testing::Values(szVGA, sz720p, sz1080p), testing::Values(TYPICAL_MAT_TYPES_ADWEIGHTED))
|
||||
|
||||
PERF_TEST_P(Size_MatType, addWeighted, TYPICAL_MATS_ADWEIGHTED)
|
||||
@@ -31,7 +31,10 @@ PERF_TEST_P(Size_MatType, addWeighted, TYPICAL_MATS_ADWEIGHTED)
|
||||
|
||||
TEST_CYCLE() cv::addWeighted( src1, alpha, src2, beta, gamma, dst, dst.type() );
|
||||
|
||||
SANITY_CHECK(dst, depth == CV_32S ? 4 : 1);
|
||||
if (depth == CV_32F)
|
||||
SANITY_CHECK(dst, 1e-4, ERROR_RELATIVE);
|
||||
else
|
||||
SANITY_CHECK(dst, depth == CV_32S ? 4 : 1);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
@@ -436,7 +436,22 @@ PERF_TEST_P_(BinaryOpTest, transpose2d)
|
||||
SANITY_CHECK_NOTHING();
|
||||
}
|
||||
|
||||
PERF_TEST_P_(BinaryOpTest, transposeND)
|
||||
static cv::Mat makeTransposeNDOutput(const cv::Mat& src, const std::vector<int>& order)
|
||||
{
|
||||
std::vector<int> new_sz(order.size());
|
||||
for (size_t i = 0; i < order.size(); ++i)
|
||||
new_sz[i] = src.size[order[i]];
|
||||
return Mat(static_cast<int>(new_sz.size()), new_sz.data(), src.type());
|
||||
}
|
||||
|
||||
static cv::Mat makeTransposeNDInput3D(Size sz, int type)
|
||||
{
|
||||
const int channels = CV_MAT_CN(type);
|
||||
int dims[] = { sz.height, sz.width, channels };
|
||||
return Mat(3, dims, CV_MAKETYPE(CV_MAT_DEPTH(type), 1));
|
||||
}
|
||||
|
||||
PERF_TEST_P_(BinaryOpTest, transposeND_identity_order)
|
||||
{
|
||||
Size sz = get<0>(GetParam());
|
||||
int type = get<1>(GetParam());
|
||||
@@ -444,14 +459,64 @@ PERF_TEST_P_(BinaryOpTest, transposeND)
|
||||
|
||||
std::vector<int> order(a.dims);
|
||||
std::iota(order.begin(), order.end(), 0);
|
||||
std::reverse(order.begin(), order.end());
|
||||
|
||||
std::vector<int> new_sz(a.dims);
|
||||
std::copy(a.size.p, a.size.p + a.dims, new_sz.begin());
|
||||
std::reverse(new_sz.begin(), new_sz.end());
|
||||
cv::Mat b = Mat(new_sz, type);
|
||||
cv::Mat b = makeTransposeNDOutput(a, order);
|
||||
|
||||
declare.in(a,WARMUP_RNG).out(b);
|
||||
declare.in(a, WARMUP_RNG).out(b);
|
||||
|
||||
TEST_CYCLE() cv::transposeND(a, order, b);
|
||||
|
||||
SANITY_CHECK_NOTHING();
|
||||
}
|
||||
|
||||
PERF_TEST_P_(BinaryOpTest, transposeND_2d_swap_order)
|
||||
{
|
||||
Size sz = get<0>(GetParam());
|
||||
int type = get<1>(GetParam());
|
||||
cv::Mat a = Mat(sz, type).reshape(1);
|
||||
|
||||
std::vector<int> order{1, 0};
|
||||
|
||||
cv::Mat b = makeTransposeNDOutput(a, order);
|
||||
|
||||
declare.in(a, WARMUP_RNG).out(b);
|
||||
|
||||
TEST_CYCLE() cv::transposeND(a, order, b);
|
||||
|
||||
SANITY_CHECK_NOTHING();
|
||||
}
|
||||
|
||||
|
||||
PERF_TEST_P_(BinaryOpTest, transposeND_generic_keep_tail_order)
|
||||
{
|
||||
Size sz = get<0>(GetParam());
|
||||
int type = get<1>(GetParam());
|
||||
cv::Mat a = makeTransposeNDInput3D(sz, type);
|
||||
|
||||
std::vector<int> order{1, 0, 2};
|
||||
|
||||
cv::Mat b = makeTransposeNDOutput(a, order);
|
||||
|
||||
randu(a, 0, 255);
|
||||
declare.in(a).out(b);
|
||||
|
||||
TEST_CYCLE() cv::transposeND(a, order, b);
|
||||
|
||||
SANITY_CHECK_NOTHING();
|
||||
}
|
||||
|
||||
PERF_TEST_P_(BinaryOpTest, transposeND_generic_move_tail_order)
|
||||
{
|
||||
Size sz = get<0>(GetParam());
|
||||
int type = get<1>(GetParam());
|
||||
cv::Mat a = makeTransposeNDInput3D(sz, type);
|
||||
|
||||
std::vector<int> order{2, 0, 1};
|
||||
|
||||
cv::Mat b = makeTransposeNDOutput(a, order);
|
||||
|
||||
randu(a, 0, 255);
|
||||
declare.in(a).out(b);
|
||||
|
||||
TEST_CYCLE() cv::transposeND(a, order, b);
|
||||
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
#include "perf_precomp.hpp"
|
||||
|
||||
namespace opencv_test
|
||||
{
|
||||
using namespace perf;
|
||||
|
||||
CV_FLAGS(GemmFlag, 0, GEMM_1_T, GEMM_2_T, GEMM_3_T)
|
||||
|
||||
typedef tuple<tuple<int, int, int>, MatType, GemmFlag> GemmTestParams_t;
|
||||
class GemmTest : public perf::TestBaseWithParam<GemmTestParams_t>
|
||||
{
|
||||
public:
|
||||
void runGemmTest(const GemmTestParams_t& params)
|
||||
{
|
||||
int M = get<0>(get<0>(params));
|
||||
int N = get<1>(get<0>(params));
|
||||
int K = get<2>(get<0>(params));
|
||||
int type = get<1>(params);
|
||||
int flags = get<2>(params);
|
||||
|
||||
Size aSize = (flags & GEMM_1_T) ? Size(M, K) : Size(K, M);
|
||||
Size bSize = (flags & GEMM_2_T) ? Size(K, N) : Size(N, K);
|
||||
|
||||
Mat src1(aSize, type), src2(bSize, type), src3(M, N, type), dst(M, N, type);
|
||||
declare.in(src1, src2, src3, WARMUP_RNG).out(dst);
|
||||
|
||||
TEST_CYCLE() cv::gemm(src1, src2, 1.0, src3, 1.0, dst, flags);
|
||||
|
||||
if (dst.total() * dst.channels() < 26)
|
||||
SANITY_CHECK_NOTHING();
|
||||
else
|
||||
SANITY_CHECK(dst, (CV_MAT_DEPTH(type) == CV_32F) ? 1e-4 : 1e-6, ERROR_RELATIVE);
|
||||
};
|
||||
};
|
||||
|
||||
// Sparse coverage: exercise tiny/small/rectangular shapes and m=1/n=1 edge cases.
|
||||
// Large square sizes (640+) are covered by opencl/perf_gemm.cpp on the CPU perf tool.
|
||||
PERF_TEST_P(GemmTest, gemmTiny,
|
||||
testing::Combine(
|
||||
testing::Values(
|
||||
make_tuple(2, 2, 2),
|
||||
make_tuple(3, 3, 3)
|
||||
),
|
||||
testing::Values(CV_32FC1, CV_64FC1),
|
||||
testing::Values(0, (int)GEMM_1_T, (int)GEMM_2_T,
|
||||
(int)(GEMM_1_T | GEMM_2_T))
|
||||
))
|
||||
{
|
||||
GemmTest::runGemmTest(GetParam());
|
||||
}
|
||||
|
||||
PERF_TEST_P(GemmTest, gemmSmall,
|
||||
testing::Combine(
|
||||
testing::Values(
|
||||
make_tuple(8, 8, 8),
|
||||
make_tuple(32, 32, 32),
|
||||
make_tuple(64, 64, 64),
|
||||
make_tuple(8, 16, 32)
|
||||
),
|
||||
testing::Values(CV_32FC1),
|
||||
testing::Values(0, (int)GEMM_2_T)
|
||||
))
|
||||
{
|
||||
GemmTest::runGemmTest(GetParam());
|
||||
}
|
||||
|
||||
PERF_TEST_P(GemmTest, gemmSquare,
|
||||
testing::Combine(
|
||||
testing::Values(
|
||||
make_tuple(256, 256, 256),
|
||||
make_tuple(512, 512, 512)
|
||||
),
|
||||
testing::Values(CV_32FC1),
|
||||
testing::Values(0)
|
||||
))
|
||||
{
|
||||
GemmTest::runGemmTest(GetParam());
|
||||
}
|
||||
|
||||
PERF_TEST_P(GemmTest, gemmRect,
|
||||
testing::Combine(
|
||||
testing::Values(
|
||||
make_tuple(1024, 64, 256),
|
||||
make_tuple(256, 1024, 512)
|
||||
),
|
||||
testing::Values(CV_32FC1),
|
||||
testing::Values(0)
|
||||
))
|
||||
{
|
||||
GemmTest::runGemmTest(GetParam());
|
||||
}
|
||||
|
||||
PERF_TEST_P(GemmTest, gemmM1,
|
||||
testing::Combine(
|
||||
testing::Values(
|
||||
make_tuple(1, 64, 2500)
|
||||
),
|
||||
testing::Values(CV_32FC1),
|
||||
testing::Values(0, (int)GEMM_1_T)
|
||||
))
|
||||
{
|
||||
GemmTest::runGemmTest(GetParam());
|
||||
}
|
||||
|
||||
PERF_TEST_P(GemmTest, gemmN1,
|
||||
testing::Combine(
|
||||
testing::Values(
|
||||
make_tuple(256, 1, 256)
|
||||
),
|
||||
testing::Values(CV_32FC1),
|
||||
testing::Values(0)
|
||||
))
|
||||
{
|
||||
GemmTest::runGemmTest(GetParam());
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -1566,7 +1566,7 @@ void mul_loop_d<double, v_float64>(const double* src1, size_t step1, const doubl
|
||||
DEFINE_SIMD_FUN(fun, _T1, v_float64, _OP)
|
||||
|
||||
DEFINE_SIMD_SAT(mul, mul_loop)
|
||||
DEFINE_SIMD_F32(mul, mul_loop_d)
|
||||
DEFINE_SIMD_F32(mul, mul_loop)
|
||||
DEFINE_SIMD_S32(mul, mul_loop_d)
|
||||
DEFINE_SIMD_F64(mul, mul_loop_d)
|
||||
|
||||
@@ -1834,7 +1834,7 @@ void add_weighted_loop_d<double, v_float64>(const double* src1, size_t step1, co
|
||||
|
||||
DEFINE_SIMD_SAT(addWeighted, add_weighted_loop)
|
||||
DEFINE_SIMD_S32(addWeighted, add_weighted_loop_d)
|
||||
DEFINE_SIMD_F32(addWeighted, add_weighted_loop_d)
|
||||
DEFINE_SIMD_F32(addWeighted, add_weighted_loop)
|
||||
DEFINE_SIMD_F64(addWeighted, add_weighted_loop_d)
|
||||
|
||||
//=======================================
|
||||
|
||||
@@ -2660,7 +2660,7 @@ cvReshapeMatND( const CvArr* arr,
|
||||
mat = &stub;
|
||||
}
|
||||
|
||||
if( CV_IS_MAT_CONT( mat->type ))
|
||||
if( !CV_IS_MAT_CONT( mat->type ))
|
||||
CV_Error( cv::Error::StsBadArg, "Non-continuous nD arrays are not supported" );
|
||||
|
||||
size1 = mat->dim[0].size;
|
||||
|
||||
@@ -29,16 +29,138 @@ static int countNonZero_(const T* src, int len )
|
||||
return nz;
|
||||
}
|
||||
|
||||
static int countNonZero8u( const void* src_ptr, int len )
|
||||
{
|
||||
const uchar* src = static_cast<const uchar*>(src_ptr);
|
||||
int i=0, nz = 0;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
int len0 = len & -VTraits<v_uint8>::vlanes();
|
||||
|
||||
#if defined(CV_CPU_COMPILE_AVX512_SKX) || defined(CV_CPU_COMPILE_AVX512_ICL)
|
||||
#define CV_COUNTNONZERO_AVX512 1
|
||||
#else
|
||||
#define CV_COUNTNONZERO_AVX512 0
|
||||
#endif
|
||||
|
||||
#if CV_COUNTNONZERO_AVX512
|
||||
|
||||
static inline int cnz_popcount32(unsigned x)
|
||||
{
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
return __builtin_popcount(x);
|
||||
#else
|
||||
int c = 0;
|
||||
for (; x; x &= x - 1)
|
||||
++c;
|
||||
return c;
|
||||
#endif
|
||||
}
|
||||
|
||||
static inline int cnz_popcount64(uint64 x)
|
||||
{
|
||||
#if defined(__GNUC__) || defined(__clang__)
|
||||
return (int)__builtin_popcountll((unsigned long long)x);
|
||||
#else
|
||||
int c = 0;
|
||||
for (; x; x &= x - 1)
|
||||
++c;
|
||||
return c;
|
||||
#endif
|
||||
}
|
||||
|
||||
static inline int cnz_lane_pop(int m) { return cnz_popcount32((unsigned)m); }
|
||||
static inline int cnz_lane_pop(int64 m) { return cnz_popcount64((uint64)m); }
|
||||
|
||||
template<typename VT> static inline int cnz_pop_ne(const VT& v, const VT& z);
|
||||
|
||||
template<> inline int cnz_pop_ne<v_uint8>(const v_uint8& v, const v_uint8& z)
|
||||
{ return cnz_lane_pop(v_signmask(v_reinterpret_as_s8(v_ne(v, z)))); }
|
||||
|
||||
template<> inline int cnz_pop_ne<v_uint16>(const v_uint16& v, const v_uint16& z)
|
||||
{ return cnz_lane_pop(v_signmask(v_reinterpret_as_s16(v_ne(v, z)))); }
|
||||
|
||||
template<> inline int cnz_pop_ne<v_int32>(const v_int32& v, const v_int32& z)
|
||||
{ return cnz_lane_pop(v_signmask(v_ne(v, z))); }
|
||||
|
||||
template<> inline int cnz_pop_ne<v_float32>(const v_float32& v, const v_float32& z)
|
||||
{ return cnz_lane_pop(v_signmask(v_ne(v, z))); }
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
template<> inline int cnz_pop_ne<v_float64>(const v_float64& v, const v_float64& z)
|
||||
{ return cnz_lane_pop(v_signmask(v_ne(v, z))); }
|
||||
#endif
|
||||
|
||||
#endif // CV_COUNTNONZERO_AVX512
|
||||
|
||||
struct cnz_pack16u
|
||||
{
|
||||
static inline v_int8 eq(const ushort* src, int k, v_uint16 z)
|
||||
{
|
||||
const int n = VTraits<v_uint16>::vlanes();
|
||||
return v_pack(v_reinterpret_as_s16(v_eq(vx_load(src + k), z)),
|
||||
v_reinterpret_as_s16(v_eq(vx_load(src + k + n), z)));
|
||||
}
|
||||
};
|
||||
|
||||
struct cnz_pack32s
|
||||
{
|
||||
static inline v_int8 eq(const int* src, int k, v_int32 z)
|
||||
{
|
||||
const int n = VTraits<v_int32>::vlanes();
|
||||
return v_pack(v_pack(v_eq(vx_load(src + k), z), v_eq(vx_load(src + k + n), z)),
|
||||
v_pack(v_eq(vx_load(src + k + 2 * n), z), v_eq(vx_load(src + k + 3 * n), z)));
|
||||
}
|
||||
};
|
||||
|
||||
struct cnz_pack32f
|
||||
{
|
||||
static inline v_int8 eq(const float* src, int k, v_float32 z)
|
||||
{
|
||||
const int n = VTraits<v_float32>::vlanes();
|
||||
return v_pack(v_pack(v_reinterpret_as_s32(v_eq(vx_load(src + k), z)),
|
||||
v_reinterpret_as_s32(v_eq(vx_load(src + k + n), z))),
|
||||
v_pack(v_reinterpret_as_s32(v_eq(vx_load(src + k + 2 * n), z)),
|
||||
v_reinterpret_as_s32(v_eq(vx_load(src + k + 3 * n), z))));
|
||||
}
|
||||
};
|
||||
|
||||
template<typename PackOp, typename ST, typename ZVT>
|
||||
static int countNonZeroVT_batched(const ST* src, int len, const ZVT& z)
|
||||
{
|
||||
int i = 0, nz = 0;
|
||||
const int vl8 = VTraits<v_int8>::vlanes();
|
||||
const int len0 = len & -vl8;
|
||||
v_int8 v_one = vx_setall_s8(1);
|
||||
v_int32 v_sum32 = vx_setzero_s32();
|
||||
|
||||
while (i < len0)
|
||||
{
|
||||
v_int16 v_sum16 = vx_setzero_s16();
|
||||
int j = i;
|
||||
while (j < std::min(len0, i + 32766 * VTraits<v_int16>::vlanes()))
|
||||
{
|
||||
v_int8 v_sum8 = vx_setzero_s8();
|
||||
int k = j;
|
||||
for (; k < std::min(len0, j + 127 * vl8); k += vl8)
|
||||
v_sum8 = v_add(v_sum8, v_and(v_one, PackOp::eq(src, k, z)));
|
||||
v_int16 part1, part2;
|
||||
v_expand(v_sum8, part1, part2);
|
||||
v_sum16 = v_add(v_sum16, v_add(part1, part2));
|
||||
j = k;
|
||||
}
|
||||
v_int32 part1, part2;
|
||||
v_expand(v_sum16, part1, part2);
|
||||
v_sum32 = v_add(v_sum32, v_add(part1, part2));
|
||||
i = j;
|
||||
}
|
||||
nz = i - v_reduce_sum(v_sum32);
|
||||
v_cleanup();
|
||||
return nz + countNonZero_(src + i, len - i);
|
||||
}
|
||||
|
||||
static int countNonZeroVT_u8(const uchar* src, int len)
|
||||
{
|
||||
int i = 0, nz = 0;
|
||||
const int len0 = len & -VTraits<v_uint8>::vlanes();
|
||||
v_uint8 v_zero = vx_setzero_u8();
|
||||
v_uint8 v_one = vx_setall_u8(1);
|
||||
|
||||
v_uint32 v_sum32 = vx_setzero_u32();
|
||||
|
||||
while (i < len0)
|
||||
{
|
||||
v_uint16 v_sum16 = vx_setzero_u16();
|
||||
@@ -61,143 +183,145 @@ static int countNonZero8u( const void* src_ptr, int len )
|
||||
}
|
||||
nz = i - v_reduce_sum(v_sum32);
|
||||
v_cleanup();
|
||||
#endif
|
||||
for( ; i < len; i++ )
|
||||
for (; i < len; i++)
|
||||
nz += src[i] != 0;
|
||||
return nz;
|
||||
}
|
||||
|
||||
static int countNonZero16u( const void* src_ptr, int len )
|
||||
{
|
||||
const ushort* src = static_cast<const ushort*>(src_ptr);
|
||||
int i = 0, nz = 0;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
int len0 = len & -VTraits<v_int8>::vlanes();
|
||||
v_uint16 v_zero = vx_setzero_u16();
|
||||
v_int8 v_one = vx_setall_s8(1);
|
||||
|
||||
v_int32 v_sum32 = vx_setzero_s32();
|
||||
while (i < len0)
|
||||
{
|
||||
v_int16 v_sum16 = vx_setzero_s16();
|
||||
int j = i;
|
||||
while (j < std::min(len0, i + 32766 * VTraits<v_int16>::vlanes()))
|
||||
{
|
||||
v_int8 v_sum8 = vx_setzero_s8();
|
||||
int k = j;
|
||||
for (; k < std::min(len0, j + 127 * VTraits<v_int8>::vlanes()); k += VTraits<v_int8>::vlanes())
|
||||
v_sum8 = v_add(v_sum8, v_and(v_one, v_pack(v_reinterpret_as_s16(v_eq(vx_load(src + k), v_zero)), v_reinterpret_as_s16(v_eq(vx_load(src + k + VTraits<v_uint16>::vlanes()), v_zero)))));
|
||||
v_int16 part1, part2;
|
||||
v_expand(v_sum8, part1, part2);
|
||||
v_sum16 = v_add(v_sum16, v_add(part1, part2));
|
||||
j = k;
|
||||
}
|
||||
v_int32 part1, part2;
|
||||
v_expand(v_sum16, part1, part2);
|
||||
v_sum32 = v_add(v_sum32, v_add(part1, part2));
|
||||
i = j;
|
||||
}
|
||||
nz = i - v_reduce_sum(v_sum32);
|
||||
v_cleanup();
|
||||
#endif
|
||||
return nz + countNonZero_(src + i, len - i);
|
||||
}
|
||||
|
||||
static int countNonZero32s( const void* src_ptr, int len )
|
||||
{
|
||||
const int* src = static_cast<const int*>(src_ptr);
|
||||
int i = 0, nz = 0;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
int len0 = len & -VTraits<v_int8>::vlanes();
|
||||
v_int32 v_zero = vx_setzero_s32();
|
||||
v_int8 v_one = vx_setall_s8(1);
|
||||
|
||||
v_int32 v_sum32 = vx_setzero_s32();
|
||||
while (i < len0)
|
||||
{
|
||||
v_int16 v_sum16 = vx_setzero_s16();
|
||||
int j = i;
|
||||
while (j < std::min(len0, i + 32766 * VTraits<v_int16>::vlanes()))
|
||||
{
|
||||
v_int8 v_sum8 = vx_setzero_s8();
|
||||
int k = j;
|
||||
for (; k < std::min(len0, j + 127 * VTraits<v_int8>::vlanes()); k += VTraits<v_int8>::vlanes())
|
||||
v_sum8 = v_add(v_sum8, v_and(v_one, v_pack(v_pack(v_eq(vx_load(src + k), v_zero), v_eq(vx_load(src + k + VTraits<v_int32>::vlanes()), v_zero)), v_pack(v_eq(vx_load(src + k + 2 * VTraits<v_int32>::vlanes()), v_zero), v_eq(vx_load(src + k + 3 * VTraits<v_int32>::vlanes()), v_zero)))));
|
||||
v_int16 part1, part2;
|
||||
v_expand(v_sum8, part1, part2);
|
||||
v_sum16 = v_add(v_sum16, v_add(part1, part2));
|
||||
j = k;
|
||||
}
|
||||
v_int32 part1, part2;
|
||||
v_expand(v_sum16, part1, part2);
|
||||
v_sum32 = v_add(v_sum32, v_add(part1, part2));
|
||||
i = j;
|
||||
}
|
||||
nz = i - v_reduce_sum(v_sum32);
|
||||
v_cleanup();
|
||||
#endif
|
||||
return nz + countNonZero_(src + i, len - i);
|
||||
}
|
||||
|
||||
static int countNonZero32f( const void* src_ptr, int len )
|
||||
{
|
||||
const float* src = static_cast<const float*>(src_ptr);
|
||||
int i = 0, nz = 0;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
int len0 = len & -VTraits<v_int8>::vlanes();
|
||||
v_float32 v_zero = vx_setzero_f32();
|
||||
v_int8 v_one = vx_setall_s8(1);
|
||||
|
||||
v_int32 v_sum32 = vx_setzero_s32();
|
||||
while (i < len0)
|
||||
{
|
||||
v_int16 v_sum16 = vx_setzero_s16();
|
||||
int j = i;
|
||||
while (j < std::min(len0, i + 32766 * VTraits<v_int16>::vlanes()))
|
||||
{
|
||||
v_int8 v_sum8 = vx_setzero_s8();
|
||||
int k = j;
|
||||
for (; k < std::min(len0, j + 127 * VTraits<v_int8>::vlanes()); k += VTraits<v_int8>::vlanes())
|
||||
v_sum8 = v_add(v_sum8, v_and(v_one, v_pack(v_pack(v_reinterpret_as_s32(v_eq(vx_load(src + k), v_zero)), v_reinterpret_as_s32(v_eq(vx_load(src + k + VTraits<v_float32>::vlanes()), v_zero))), v_pack(v_reinterpret_as_s32(v_eq(vx_load(src + k + 2 * VTraits<v_float32>::vlanes()), v_zero)), v_reinterpret_as_s32(v_eq(vx_load(src + k + 3 * VTraits<v_float32>::vlanes()), v_zero))))));
|
||||
v_int16 part1, part2;
|
||||
v_expand(v_sum8, part1, part2);
|
||||
v_sum16 = v_add(v_sum16, v_add(part1, part2));
|
||||
j = k;
|
||||
}
|
||||
v_int32 part1, part2;
|
||||
v_expand(v_sum16, part1, part2);
|
||||
v_sum32 = v_add(v_sum32, v_add(part1, part2));
|
||||
i = j;
|
||||
}
|
||||
nz = i - v_reduce_sum(v_sum32);
|
||||
v_cleanup();
|
||||
#endif
|
||||
return nz + countNonZero_(src + i, len - i);
|
||||
}
|
||||
|
||||
static int countNonZero64f( const void* src_ptr, int len )
|
||||
{
|
||||
const double* src = static_cast<const double*>(src_ptr);
|
||||
int nz = 0, i = 0;
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
static int countNonZeroVT_f64(const double* src, int len)
|
||||
{
|
||||
int nz = 0, i = 0;
|
||||
v_int64 sum1 = vx_setzero_s64();
|
||||
v_int64 sum2 = vx_setzero_s64();
|
||||
v_float64 zero = vx_setzero_f64();
|
||||
int step = VTraits<v_float64>::vlanes() * 2;
|
||||
int len0 = len & -step;
|
||||
const int step = VTraits<v_float64>::vlanes() * 2;
|
||||
const int len0 = len & -step;
|
||||
|
||||
for(i = 0; i < len0; i += step )
|
||||
{
|
||||
for (i = 0; i < len0; i += step)
|
||||
{
|
||||
sum1 = v_add(sum1, v_reinterpret_as_s64(v_eq(vx_load(&src[i]), zero)));
|
||||
sum2 = v_add(sum2, v_reinterpret_as_s64(v_eq(vx_load(&src[i + step / 2]), zero)));
|
||||
}
|
||||
}
|
||||
|
||||
// N.B the value is incremented by -1 (0xF...F) for each value
|
||||
nz = i + (int)v_reduce_sum(v_add(sum1, sum2));
|
||||
v_cleanup();
|
||||
#endif
|
||||
return nz + countNonZero_(src + i, len - i);
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename VT> struct cnz_vt_traits;
|
||||
|
||||
template<>
|
||||
struct cnz_vt_traits<v_uint8>
|
||||
{
|
||||
typedef uchar ST;
|
||||
static inline v_uint8 zero() { return vx_setzero_u8(); }
|
||||
static int legacy(const uchar* src, int len, const v_uint8&) { return countNonZeroVT_u8(src, len); }
|
||||
};
|
||||
|
||||
template<>
|
||||
struct cnz_vt_traits<v_uint16>
|
||||
{
|
||||
typedef ushort ST;
|
||||
static inline v_uint16 zero() { return vx_setzero_u16(); }
|
||||
static int legacy(const ushort* src, int len, const v_uint16& z)
|
||||
{ return countNonZeroVT_batched<cnz_pack16u, ushort, v_uint16>(src, len, z); }
|
||||
};
|
||||
|
||||
template<>
|
||||
struct cnz_vt_traits<v_int32>
|
||||
{
|
||||
typedef int ST;
|
||||
static inline v_int32 zero() { return vx_setzero_s32(); }
|
||||
static int legacy(const int* src, int len, const v_int32& z)
|
||||
{ return countNonZeroVT_batched<cnz_pack32s, int, v_int32>(src, len, z); }
|
||||
};
|
||||
|
||||
template<>
|
||||
struct cnz_vt_traits<v_float32>
|
||||
{
|
||||
typedef float ST;
|
||||
static inline v_float32 zero() { return vx_setzero_f32(); }
|
||||
static int legacy(const float* src, int len, const v_float32& z)
|
||||
{ return countNonZeroVT_batched<cnz_pack32f, float, v_float32>(src, len, z); }
|
||||
};
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
template<>
|
||||
struct cnz_vt_traits<v_float64>
|
||||
{
|
||||
typedef double ST;
|
||||
static inline v_float64 zero() { return vx_setzero_f64(); }
|
||||
static int legacy(const double* src, int len, const v_float64&)
|
||||
{ return countNonZeroVT_f64(src, len); }
|
||||
};
|
||||
#endif
|
||||
|
||||
template<typename VT, typename ST>
|
||||
static int countNonZeroVT(const ST* src, int len, const VT& z)
|
||||
{
|
||||
#if CV_COUNTNONZERO_AVX512
|
||||
const int nlanes = VTraits<VT>::vlanes();
|
||||
const int step = nlanes * 2;
|
||||
const int len0 = len & -step;
|
||||
int i = 0, nz = 0;
|
||||
|
||||
for (; i < len0; i += step)
|
||||
{
|
||||
nz += cnz_pop_ne<VT>(vx_load(src + i), z);
|
||||
nz += cnz_pop_ne<VT>(vx_load(src + i + nlanes), z);
|
||||
}
|
||||
for (; i <= len - nlanes; i += nlanes)
|
||||
nz += cnz_pop_ne<VT>(vx_load(src + i), z);
|
||||
|
||||
v_cleanup();
|
||||
return nz + countNonZero_(src + i, len - i);
|
||||
#else
|
||||
return cnz_vt_traits<VT>::legacy(src, len, z);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<typename VT>
|
||||
static int countNonZeroSimd(const void* src_ptr, int len)
|
||||
{
|
||||
typedef cnz_vt_traits<VT> traits;
|
||||
const typename traits::ST* src = static_cast<const typename traits::ST*>(src_ptr);
|
||||
return countNonZeroVT<VT>(src, len, traits::zero());
|
||||
}
|
||||
|
||||
#endif // CV_SIMD
|
||||
|
||||
template<typename VT, typename ST>
|
||||
static int countNonZeroDepth(const void* src_ptr, int len)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
return countNonZeroSimd<VT>(src_ptr, len);
|
||||
#else
|
||||
return countNonZero_(static_cast<const ST*>(src_ptr), len);
|
||||
#endif
|
||||
}
|
||||
|
||||
static int countNonZero8u( const void* src_ptr, int len )
|
||||
{ return countNonZeroDepth<v_uint8, uchar>(src_ptr, len); }
|
||||
|
||||
static int countNonZero16u( const void* src_ptr, int len )
|
||||
{ return countNonZeroDepth<v_uint16, ushort>(src_ptr, len); }
|
||||
|
||||
static int countNonZero32s( const void* src_ptr, int len )
|
||||
{ return countNonZeroDepth<v_int32, int>(src_ptr, len); }
|
||||
|
||||
static int countNonZero32f( const void* src_ptr, int len )
|
||||
{ return countNonZeroDepth<v_float32, float>(src_ptr, len); }
|
||||
|
||||
static int countNonZero64f( const void* src_ptr, int len )
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE) && (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
return countNonZeroSimd<v_float64>(src_ptr, len);
|
||||
#else
|
||||
return countNonZero_(static_cast<const double*>(src_ptr), len);
|
||||
#endif
|
||||
}
|
||||
|
||||
CountNonZeroFunc getCountNonZeroTab(int depth)
|
||||
{
|
||||
|
||||
@@ -47,7 +47,6 @@
|
||||
|
||||
#ifdef HAVE_LAPACK
|
||||
|
||||
#include <complex.h>
|
||||
#include "opencv_lapack.h"
|
||||
|
||||
#include <cmath>
|
||||
@@ -710,4 +709,4 @@ int lapack_gemm64fc(const double *src1, size_t src1_step, const double *src2, si
|
||||
#pragma clang diagnostic pop
|
||||
#endif
|
||||
|
||||
#endif //HAVE_LAPACK
|
||||
#endif //HAVE_LAPACK
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
|
||||
#include "precomp.hpp"
|
||||
@@ -8,6 +9,11 @@
|
||||
#include "convert.hpp"
|
||||
#include <sys/types.h>
|
||||
|
||||
namespace cv {
|
||||
void LUT8u_dispatch( const uchar* src, const uchar* lut, uchar* dst, int len, int cn, int lutcn );
|
||||
void LUT16u_dispatch( const uchar* src, const ushort* lut, ushort* dst, int len, int cn, int lutcn );
|
||||
} // namespace cv
|
||||
|
||||
/****************************************************************************************\
|
||||
* LUT Transform *
|
||||
\****************************************************************************************/
|
||||
@@ -40,9 +46,9 @@ static LUTFunc getLUTFunc(const int srcDepth, const int dstDepth)
|
||||
{
|
||||
switch(dstDepth)
|
||||
{
|
||||
case CV_8U: ret = (LUTFunc)LUT_<uint8_t, uint8_t>; break;
|
||||
case CV_8U: ret = (LUTFunc)LUT8u_dispatch; break;
|
||||
case CV_8S: ret = (LUTFunc)LUT_<uint8_t, int8_t>; break;
|
||||
case CV_16U: ret = (LUTFunc)LUT_<uint8_t, uint16_t>; break;
|
||||
case CV_16U: ret = (LUTFunc)LUT16u_dispatch; break;
|
||||
case CV_16S: ret = (LUTFunc)LUT_<uint8_t, int16_t>; break;
|
||||
case CV_32S: ret = (LUTFunc)LUT_<uint8_t, int32_t>; break;
|
||||
case CV_32F: ret = (LUTFunc)LUT_<uint8_t, int32_t>; break; // float
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#include "precomp.hpp"
|
||||
|
||||
#include "lut.simd.hpp"
|
||||
#include "lut.simd_declarations.hpp" // defines CV_CPU_DISPATCH_MODES_ALL=AVX512_ICL,...,BASELINE based on CMakeLists.txt content
|
||||
|
||||
namespace cv {
|
||||
|
||||
namespace {
|
||||
|
||||
typedef void (*LUT8uFunc)( const uchar*, const uchar*, uchar*, int, int, int );
|
||||
typedef void (*LUT16uFunc)( const uchar*, const ushort*, ushort*, int, int, int );
|
||||
|
||||
static LUT8uFunc resolveLUT8uFunc()
|
||||
{
|
||||
#if CV_TRY_AVX512_ICL
|
||||
if (cv::checkHardwareSupport(CV_CPU_AVX512_ICL))
|
||||
return opt_AVX512_ICL::LUT8u_;
|
||||
#endif
|
||||
return cpu_baseline::LUT8u_;
|
||||
}
|
||||
|
||||
static LUT16uFunc resolveLUT16uFunc()
|
||||
{
|
||||
#if CV_TRY_AVX512_ICL
|
||||
if (cv::checkHardwareSupport(CV_CPU_AVX512_ICL))
|
||||
return opt_AVX512_ICL::LUT16u_;
|
||||
#endif
|
||||
return cpu_baseline::LUT16u_;
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
void LUT8u_dispatch( const uchar* src, const uchar* lut, uchar* dst, int len, int cn, int lutcn );
|
||||
|
||||
void LUT8u_dispatch( const uchar* src, const uchar* lut, uchar* dst, int len, int cn, int lutcn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
static LUT8uFunc fn = resolveLUT8uFunc();
|
||||
fn(src, lut, dst, len, cn, lutcn);
|
||||
}
|
||||
|
||||
void LUT16u_dispatch( const uchar* src, const ushort* lut, ushort* dst, int len, int cn, int lutcn );
|
||||
|
||||
void LUT16u_dispatch( const uchar* src, const ushort* lut, ushort* dst, int len, int cn, int lutcn )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
static LUT16uFunc fn = resolveLUT16uFunc();
|
||||
fn(src, lut, dst, len, cn, lutcn);
|
||||
}
|
||||
|
||||
} // namespace cv
|
||||
@@ -0,0 +1,146 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#include "precomp.hpp"
|
||||
|
||||
// VBMI vpermb is the only x86 SIMD path faster than scalar for byte LUT.
|
||||
// x86 gather (AVX2/SSE) is slower than scalar due to high gather latency.
|
||||
// Non-x86 (NEON, etc.) benefits from v_lut implementation.
|
||||
#if CV_AVX_512VBMI
|
||||
#define CV_LUT8U_SIMD 1
|
||||
#elif (defined(CV_CPU_COMPILE_SSE) || defined(CV_CPU_COMPILE_SSE2) || defined(CV_CPU_COMPILE_SSE3) || \
|
||||
defined(CV_CPU_COMPILE_SSSE3) || defined(CV_CPU_COMPILE_SSE4_1) || defined(CV_CPU_COMPILE_SSE4_2) || \
|
||||
defined(CV_CPU_COMPILE_AVX) || defined(CV_CPU_COMPILE_AVX2) || defined(CV_CPU_COMPILE_AVX_512F) || \
|
||||
defined(CV_CPU_COMPILE_AVX512_COMMON) || defined(CV_CPU_COMPILE_AVX512_SKX) || \
|
||||
defined(CV_CPU_COMPILE_AVX512_CNL) || defined(CV_CPU_COMPILE_AVX512_CLX))
|
||||
#define CV_LUT8U_SIMD 0
|
||||
#elif CV_SIMD || CV_SIMD_SCALABLE
|
||||
#define CV_LUT8U_SIMD 1
|
||||
#else
|
||||
#define CV_LUT8U_SIMD 0
|
||||
#endif
|
||||
|
||||
namespace cv {
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
void LUT8u_( const uchar* src, const uchar* lut, uchar* dst, int len, int cn, int lutcn );
|
||||
void LUT16u_( const uchar* src, const ushort* lut, ushort* dst, int len, int cn, int lutcn );
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
void LUT8u_( const uchar* src, const uchar* lut, uchar* dst, int len, int cn, int lutcn )
|
||||
{
|
||||
const int total = len * cn;
|
||||
|
||||
if( lutcn == 1 )
|
||||
{
|
||||
int i = 0;
|
||||
#if CV_LUT8U_SIMD
|
||||
const int nlanes = VTraits<v_uint8>::vlanes();
|
||||
for( ; i <= total - nlanes; i += nlanes )
|
||||
{
|
||||
v_uint8 idx = vx_load(src + i);
|
||||
v_uint8 result = v_lut(lut, idx);
|
||||
v_store(dst + i, result);
|
||||
}
|
||||
#endif
|
||||
for( ; i < total; i++ )
|
||||
dst[i] = lut[src[i]];
|
||||
}
|
||||
#if CV_LUT8U_SIMD
|
||||
else if( cn == 3 )
|
||||
{
|
||||
// Deinterleave the 3-channel LUT into per-channel tables
|
||||
uchar CV_DECL_ALIGNED(64) lut0[256], lut1[256], lut2[256];
|
||||
const int nlanes = VTraits<v_uint8>::vlanes();
|
||||
{
|
||||
unsigned j = 0;
|
||||
for( ; j + (unsigned)nlanes <= 256; j += (unsigned)nlanes )
|
||||
{
|
||||
v_uint8 a, b, c;
|
||||
v_load_deinterleave(lut + j * 3, a, b, c);
|
||||
v_store(lut0 + j, a);
|
||||
v_store(lut1 + j, b);
|
||||
v_store(lut2 + j, c);
|
||||
}
|
||||
for( ; j < 256; j++ )
|
||||
{
|
||||
const unsigned idx = j * 3;
|
||||
lut0[j] = lut[idx];
|
||||
lut1[j] = lut[idx + 1];
|
||||
lut2[j] = lut[idx + 2];
|
||||
}
|
||||
}
|
||||
int i = 0;
|
||||
for( ; i <= total - nlanes*3; i += nlanes*3 )
|
||||
{
|
||||
v_uint8 r, g, b;
|
||||
v_load_deinterleave(src + i, r, g, b);
|
||||
r = v_lut(lut0, r);
|
||||
g = v_lut(lut1, g);
|
||||
b = v_lut(lut2, b);
|
||||
v_store_interleave(dst + i, r, g, b);
|
||||
}
|
||||
for( ; i < total; i += 3 )
|
||||
{
|
||||
dst[i] = lut0[src[i]];
|
||||
dst[i+1] = lut1[src[i+1]];
|
||||
dst[i+2] = lut2[src[i+2]];
|
||||
}
|
||||
}
|
||||
#endif
|
||||
else
|
||||
{
|
||||
for( int i = 0; i < total; i += cn )
|
||||
for( int k = 0; k < cn; k++ )
|
||||
dst[i+k] = lut[src[i+k]*cn + k];
|
||||
}
|
||||
}
|
||||
|
||||
void LUT16u_( const uchar* src, const ushort* lut, ushort* dst, int len, int cn, int lutcn )
|
||||
{
|
||||
const int total = len * cn;
|
||||
|
||||
if( lutcn == 1 )
|
||||
{
|
||||
int i = 0;
|
||||
#if CV_LUT8U_SIMD
|
||||
// Split ushort LUT into low-byte and high-byte tables,
|
||||
// then use two byte v_lut (vpermb on VBMI) + interleave.
|
||||
uchar CV_DECL_ALIGNED(64) lut_lo[256], lut_hi[256];
|
||||
for( int j = 0; j < 256; j++ )
|
||||
{
|
||||
lut_lo[j] = (uchar)(lut[j]);
|
||||
lut_hi[j] = (uchar)(lut[j] >> 8);
|
||||
}
|
||||
const int nlanes8 = VTraits<v_uint8>::vlanes();
|
||||
const int nlanes16 = VTraits<v_uint16>::vlanes();
|
||||
for( ; i <= total - nlanes8; i += nlanes8 )
|
||||
{
|
||||
v_uint8 idx = vx_load(src + i);
|
||||
v_uint8 lo = v_lut(lut_lo, idx);
|
||||
v_uint8 hi = v_lut(lut_hi, idx);
|
||||
// Zip low and high bytes: [l0,h0,l1,h1,...] → ushort values on little-endian
|
||||
v_uint8 res0, res1;
|
||||
v_zip(lo, hi, res0, res1);
|
||||
v_store(dst + i, v_reinterpret_as_u16(res0));
|
||||
v_store(dst + i + nlanes16, v_reinterpret_as_u16(res1));
|
||||
}
|
||||
#endif
|
||||
for( ; i < total; i++ )
|
||||
dst[i] = lut[src[i]];
|
||||
}
|
||||
else
|
||||
{
|
||||
for( int i = 0; i < total; i += cn )
|
||||
for( int k = 0; k < cn; k++ )
|
||||
dst[i+k] = lut[src[i+k]*cn + k];
|
||||
}
|
||||
}
|
||||
|
||||
#endif // CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_END
|
||||
} // namespace cv
|
||||
@@ -107,7 +107,6 @@ void invSqrt32f(const float* src, float* dst, int len)
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
CALL_HAL(invSqrt32f, cv_hal_invSqrt32f, src, dst, len);
|
||||
CV_IPP_RUN_FAST(CV_INSTRUMENT_FUN_IPP(ippsInvSqrt_32f_A21, src, dst, len) >= 0);
|
||||
|
||||
CV_CPU_DISPATCH(invSqrt32f, (src, dst, len),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
@@ -119,7 +118,6 @@ void invSqrt64f(const double* src, double* dst, int len)
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
CALL_HAL(invSqrt64f, cv_hal_invSqrt64f, src, dst, len);
|
||||
CV_IPP_RUN_FAST(CV_INSTRUMENT_FUN_IPP(ippsInvSqrt_64f_A50, src, dst, len) >= 0);
|
||||
|
||||
CV_CPU_DISPATCH(invSqrt64f, (src, dst, len),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
@@ -152,7 +150,6 @@ void exp32f(const float *src, float *dst, int n)
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
CALL_HAL(exp32f, cv_hal_exp32f, src, dst, n);
|
||||
CV_IPP_RUN_FAST(CV_INSTRUMENT_FUN_IPP(ippsExp_32f_A21, src, dst, n) >= 0);
|
||||
|
||||
CV_CPU_DISPATCH(exp32f, (src, dst, n),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
@@ -163,7 +160,6 @@ void exp64f(const double *src, double *dst, int n)
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
CALL_HAL(exp64f, cv_hal_exp64f, src, dst, n);
|
||||
CV_IPP_RUN_FAST(CV_INSTRUMENT_FUN_IPP(ippsExp_64f_A50, src, dst, n) >= 0);
|
||||
|
||||
CV_CPU_DISPATCH(exp64f, (src, dst, n),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
@@ -174,7 +170,6 @@ void log32f(const float *src, float *dst, int n)
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
CALL_HAL(log32f, cv_hal_log32f, src, dst, n);
|
||||
CV_IPP_RUN_FAST(CV_INSTRUMENT_FUN_IPP(ippsLn_32f_A21, src, dst, n) >= 0);
|
||||
|
||||
CV_CPU_DISPATCH(log32f, (src, dst, n),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
@@ -185,7 +180,6 @@ void log64f(const double *src, double *dst, int n)
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
CALL_HAL(log64f, cv_hal_log64f, src, dst, n);
|
||||
CV_IPP_RUN_FAST(CV_INSTRUMENT_FUN_IPP(ippsLn_64f_A50, src, dst, n) >= 0);
|
||||
|
||||
CV_CPU_DISPATCH(log64f, (src, dst, n),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
|
||||
@@ -595,7 +595,7 @@ void sqrt64f(const double* src, double* dst, int len)
|
||||
// Workaround for ICE in MSVS 2015 update 3 (issue #7795)
|
||||
// CV_AVX is not used here, because generated code is faster in non-AVX mode.
|
||||
// (tested with disabled IPP on i5-6300U)
|
||||
#if (defined _MSC_VER && _MSC_VER >= 1900) || defined(__EMSCRIPTEN__)
|
||||
#if (defined _MSC_VER && _MSC_VER >= 1900 && (defined(_M_IX86) || defined(_M_X64))) || defined(__EMSCRIPTEN__)
|
||||
void exp32f(const float *src, float *dst, int n)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
// Copyright (C) 2000-2008, Intel Corporation, all rights reserved.
|
||||
// Copyright (C) 2009-2011, Willow Garage Inc., all rights reserved.
|
||||
// Copyright (C) 2014-2015, Itseez Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
//
|
||||
// Redistribution and use in source and binary forms, with or without modification,
|
||||
@@ -175,6 +176,337 @@ GEMM_TransposeBlock( const uchar* src, size_t src_step,
|
||||
}
|
||||
}
|
||||
|
||||
#if CV_SIMD
|
||||
template<typename ST> struct VecTraits;
|
||||
|
||||
template<> struct VecTraits<float> {
|
||||
typedef v_float32 VT;
|
||||
static inline VT zero() { return vx_setzero_f32(); }
|
||||
static inline VT load(const float* p) { return vx_load(p); }
|
||||
static inline VT setall(float v) { return vx_setall_f32(v); }
|
||||
};
|
||||
#endif
|
||||
|
||||
#if CV_SIMD_64F
|
||||
template<> struct VecTraits<double> {
|
||||
typedef v_float64 VT;
|
||||
static inline VT zero() { return vx_setzero_f64(); }
|
||||
static inline VT load(const double* p) { return vx_load(p); }
|
||||
static inline VT setall(double v) { return vx_setall_f64(v); }
|
||||
};
|
||||
#endif
|
||||
|
||||
#if CV_SIMD
|
||||
template<typename ST>
|
||||
static inline ST simdDotProduct(const ST* a, const ST* b, int n, int& k)
|
||||
{
|
||||
typedef typename VecTraits<ST>::VT VecT;
|
||||
const int vlanes = VTraits<VecT>::vlanes();
|
||||
const int stride4 = vlanes * 4;
|
||||
VecT vs0 = VecTraits<ST>::zero();
|
||||
VecT vs1 = VecTraits<ST>::zero();
|
||||
VecT vs2 = VecTraits<ST>::zero();
|
||||
VecT vs3 = VecTraits<ST>::zero();
|
||||
for( ; k <= n - stride4; k += stride4 )
|
||||
{
|
||||
vs0 = v_muladd(VecTraits<ST>::load(a + k), VecTraits<ST>::load(b + k), vs0);
|
||||
vs1 = v_muladd(VecTraits<ST>::load(a + k + vlanes), VecTraits<ST>::load(b + k + vlanes), vs1);
|
||||
vs2 = v_muladd(VecTraits<ST>::load(a + k + vlanes * 2), VecTraits<ST>::load(b + k + vlanes * 2), vs2);
|
||||
vs3 = v_muladd(VecTraits<ST>::load(a + k + vlanes * 3), VecTraits<ST>::load(b + k + vlanes * 3), vs3);
|
||||
}
|
||||
vs0 = v_add(v_add(vs0, vs1), v_add(vs2, vs3));
|
||||
for( ; k <= n - vlanes; k += vlanes )
|
||||
vs0 = v_muladd(VecTraits<ST>::load(a + k), VecTraits<ST>::load(b + k), vs0);
|
||||
return v_reduce_sum(vs0);
|
||||
}
|
||||
#endif
|
||||
|
||||
#if CV_SIMD_64F
|
||||
// Float input, double accumulation dot product (for 32FC1 where WT=double)
|
||||
static inline double simdDotProduct_f32f64(const float* a, const float* b, int n, int& k)
|
||||
{
|
||||
const int vlanes_f32 = VTraits<v_float32>::vlanes();
|
||||
const int stride2 = vlanes_f32 * 2;
|
||||
v_float64 vs0 = vx_setzero_f64();
|
||||
v_float64 vs1 = vx_setzero_f64();
|
||||
v_float64 vs2 = vx_setzero_f64();
|
||||
v_float64 vs3 = vx_setzero_f64();
|
||||
for( ; k <= n - stride2; k += stride2 )
|
||||
{
|
||||
v_float32 va0 = vx_load(a + k);
|
||||
v_float32 vb0 = vx_load(b + k);
|
||||
vs0 = v_muladd(v_cvt_f64(va0), v_cvt_f64(vb0), vs0);
|
||||
vs1 = v_muladd(v_cvt_f64_high(va0), v_cvt_f64_high(vb0), vs1);
|
||||
|
||||
v_float32 va1 = vx_load(a + k + vlanes_f32);
|
||||
v_float32 vb1 = vx_load(b + k + vlanes_f32);
|
||||
vs2 = v_muladd(v_cvt_f64(va1), v_cvt_f64(vb1), vs2);
|
||||
vs3 = v_muladd(v_cvt_f64_high(va1), v_cvt_f64_high(vb1), vs3);
|
||||
}
|
||||
vs0 = v_add(v_add(vs0, vs1), v_add(vs2, vs3));
|
||||
for( ; k <= n - vlanes_f32; k += vlanes_f32 )
|
||||
{
|
||||
v_float32 va = vx_load(a + k);
|
||||
v_float32 vb = vx_load(b + k);
|
||||
vs0 = v_muladd(v_cvt_f64(va), v_cvt_f64(vb), vs0);
|
||||
vs0 = v_muladd(v_cvt_f64_high(va), v_cvt_f64_high(vb), vs0);
|
||||
}
|
||||
return v_reduce_sum(vs0);
|
||||
}
|
||||
|
||||
// Float input, double accumulation GEMM k-outer/j-inner (for 32FC1 where WT=double)
|
||||
static inline void simdGEMM_kj_f32f64(
|
||||
const float* a_data, const float* _b_data, size_t b_step,
|
||||
const float* _c_data, size_t c_step1,
|
||||
float* d_data,
|
||||
int n, int m,
|
||||
double alpha, double beta,
|
||||
int& j)
|
||||
{
|
||||
const int vlanes_f32 = VTraits<v_float32>::vlanes();
|
||||
const int vlanes_f64 = VTraits<v_float64>::vlanes();
|
||||
|
||||
if( m >= 2 * vlanes_f32 && n >= 64 )
|
||||
{
|
||||
// k-outer / j-inner with double accumulation
|
||||
const int max_m = 1600 / (int)sizeof(float);
|
||||
cv::AutoBuffer<double> _s_buf(max_m);
|
||||
double* s_buf = _s_buf.data();
|
||||
|
||||
{
|
||||
v_float64 vz = vx_setzero_f64();
|
||||
for( j = 0; j <= m - vlanes_f64; j += vlanes_f64 )
|
||||
v_store(s_buf + j, vz);
|
||||
for( ; j < m; j++ )
|
||||
s_buf[j] = 0.0;
|
||||
}
|
||||
|
||||
{
|
||||
const float* b_ptr = _b_data;
|
||||
for( int k = 0; k < n; k++, b_ptr += b_step )
|
||||
{
|
||||
v_float64 va = vx_setall_f64((double)a_data[k]);
|
||||
j = 0;
|
||||
for( ; j <= m - vlanes_f32; j += vlanes_f32 )
|
||||
{
|
||||
v_float32 vb = vx_load(b_ptr + j);
|
||||
v_float64 vb_lo = v_cvt_f64(vb);
|
||||
v_float64 vb_hi = v_cvt_f64_high(vb);
|
||||
v_float64 vd_lo = vx_load(s_buf + j);
|
||||
v_float64 vd_hi = vx_load(s_buf + j + vlanes_f64);
|
||||
vd_lo = v_muladd(va, vb_lo, vd_lo);
|
||||
vd_hi = v_muladd(va, vb_hi, vd_hi);
|
||||
v_store(s_buf + j, vd_lo);
|
||||
v_store(s_buf + j + vlanes_f64, vd_hi);
|
||||
}
|
||||
double a_val = (double)a_data[k];
|
||||
for( ; j < m; j++ )
|
||||
s_buf[j] += a_val * (double)b_ptr[j];
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
v_float64 valpha = vx_setall_f64(alpha);
|
||||
const float* c_data = _c_data;
|
||||
j = 0;
|
||||
if( !_c_data )
|
||||
{
|
||||
for( ; j <= m - vlanes_f32; j += vlanes_f32 )
|
||||
{
|
||||
v_float64 vs_lo = v_mul(vx_load(s_buf + j), valpha);
|
||||
v_float64 vs_hi = v_mul(vx_load(s_buf + j + vlanes_f64), valpha);
|
||||
v_store(d_data + j, v_cvt_f32(vs_lo, vs_hi));
|
||||
}
|
||||
for( ; j < m; j++ )
|
||||
d_data[j] = (float)(s_buf[j] * alpha);
|
||||
}
|
||||
else if( c_step1 == 1 )
|
||||
{
|
||||
v_float64 vbeta = vx_setall_f64(beta);
|
||||
for( ; j <= m - vlanes_f32; j += vlanes_f32 )
|
||||
{
|
||||
v_float64 vs_lo = v_mul(vx_load(s_buf + j), valpha);
|
||||
v_float64 vs_hi = v_mul(vx_load(s_buf + j + vlanes_f64), valpha);
|
||||
v_float32 vc = vx_load(c_data + j);
|
||||
vs_lo = v_muladd(v_cvt_f64(vc), vbeta, vs_lo);
|
||||
vs_hi = v_muladd(v_cvt_f64_high(vc), vbeta, vs_hi);
|
||||
v_store(d_data + j, v_cvt_f32(vs_lo, vs_hi));
|
||||
}
|
||||
for( ; j < m; j++ )
|
||||
d_data[j] = (float)(s_buf[j] * alpha + (double)c_data[j] * beta);
|
||||
}
|
||||
else
|
||||
{
|
||||
for( j = 0; j < m; j++, c_data += c_step1 )
|
||||
d_data[j] = (float)(s_buf[j] * alpha + (double)c_data[0] * beta);
|
||||
}
|
||||
}
|
||||
j = m;
|
||||
}
|
||||
else if( vlanes_f32 * (int)sizeof(float) >= 32 )
|
||||
{
|
||||
// j-outer with SIMD: double accumulation per vlanes_f64 group
|
||||
const float* c_data = _c_data;
|
||||
for( ; j <= m - vlanes_f32; j += vlanes_f32, c_data += vlanes_f32 * c_step1 )
|
||||
{
|
||||
const float* b = _b_data + j;
|
||||
v_float64 vs_lo = vx_setzero_f64();
|
||||
v_float64 vs_hi = vx_setzero_f64();
|
||||
for( int k = 0; k < n; k++, b += b_step )
|
||||
{
|
||||
v_float64 va = vx_setall_f64((double)a_data[k]);
|
||||
v_float32 vb = vx_load(b);
|
||||
vs_lo = v_muladd(va, v_cvt_f64(vb), vs_lo);
|
||||
vs_hi = v_muladd(va, v_cvt_f64_high(vb), vs_hi);
|
||||
}
|
||||
v_float64 valpha = vx_setall_f64(alpha);
|
||||
vs_lo = v_mul(vs_lo, valpha);
|
||||
vs_hi = v_mul(vs_hi, valpha);
|
||||
if( !_c_data )
|
||||
{
|
||||
v_store(d_data + j, v_cvt_f32(vs_lo, vs_hi));
|
||||
}
|
||||
else if( c_step1 == 1 )
|
||||
{
|
||||
v_float64 vbeta = vx_setall_f64(beta);
|
||||
v_float32 vc = vx_load(c_data);
|
||||
vs_lo = v_muladd(v_cvt_f64(vc), vbeta, vs_lo);
|
||||
vs_hi = v_muladd(v_cvt_f64_high(vc), vbeta, vs_hi);
|
||||
v_store(d_data + j, v_cvt_f32(vs_lo, vs_hi));
|
||||
}
|
||||
else
|
||||
{
|
||||
CV_DECL_ALIGNED(CV_SIMD_WIDTH) float buf[VTraits<v_float32>::max_nlanes];
|
||||
v_store_aligned(buf, v_cvt_f32(vs_lo, vs_hi));
|
||||
for( int jj = 0; jj < vlanes_f32; jj++ )
|
||||
d_data[j + jj] = buf[jj] + c_data[jj * c_step1] * (float)beta;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
#if CV_SIMD
|
||||
template<typename ST>
|
||||
static inline void simdGEMM_kj(
|
||||
const ST* a_data, const ST* _b_data, size_t b_step,
|
||||
const ST* _c_data, size_t c_step1,
|
||||
ST* d_data,
|
||||
int n, int m,
|
||||
double alpha, double beta,
|
||||
int& j)
|
||||
{
|
||||
typedef typename VecTraits<ST>::VT VecT;
|
||||
const int vlanes = VTraits<VecT>::vlanes();
|
||||
|
||||
if( m >= 2 * vlanes && n >= 64 )
|
||||
{
|
||||
// k-outer / j-inner SIMD: sequential B-row access, good for large n
|
||||
const int max_m = 1600 / (int)sizeof(ST);
|
||||
CV_DECL_ALIGNED(CV_SIMD_WIDTH) ST s_buf[max_m];
|
||||
|
||||
{
|
||||
VecT vz = VecTraits<ST>::zero();
|
||||
for( j = 0; j <= m - vlanes; j += vlanes )
|
||||
v_store_aligned(s_buf + j, vz);
|
||||
for( ; j < m; j++ )
|
||||
s_buf[j] = (ST)0;
|
||||
}
|
||||
|
||||
{
|
||||
const ST* b_ptr = _b_data;
|
||||
for( int k = 0; k < n; k++, b_ptr += b_step )
|
||||
{
|
||||
VecT va = VecTraits<ST>::setall(a_data[k]);
|
||||
j = 0;
|
||||
for( ; j <= m - vlanes; j += vlanes )
|
||||
{
|
||||
VecT vd = vx_load_aligned(s_buf + j);
|
||||
VecT vb = VecTraits<ST>::load(b_ptr + j);
|
||||
vd = v_muladd(va, vb, vd);
|
||||
v_store_aligned(s_buf + j, vd);
|
||||
}
|
||||
ST a_val = a_data[k];
|
||||
for( ; j < m; j++ )
|
||||
s_buf[j] += a_val * b_ptr[j];
|
||||
}
|
||||
}
|
||||
|
||||
{
|
||||
VecT valpha = VecTraits<ST>::setall((ST)alpha);
|
||||
const ST* c_data = _c_data;
|
||||
j = 0;
|
||||
if( !_c_data )
|
||||
{
|
||||
for( ; j <= m - vlanes; j += vlanes )
|
||||
{
|
||||
VecT vs = v_mul(vx_load_aligned(s_buf + j), valpha);
|
||||
v_store(d_data + j, vs);
|
||||
}
|
||||
for( ; j < m; j++ )
|
||||
d_data[j] = s_buf[j] * (ST)alpha;
|
||||
}
|
||||
else if( c_step1 == 1 )
|
||||
{
|
||||
VecT vbeta = VecTraits<ST>::setall((ST)beta);
|
||||
for( ; j <= m - vlanes; j += vlanes )
|
||||
{
|
||||
VecT vs = v_mul(vx_load_aligned(s_buf + j), valpha);
|
||||
VecT vc = VecTraits<ST>::load(c_data + j);
|
||||
vs = v_muladd(vc, vbeta, vs);
|
||||
v_store(d_data + j, vs);
|
||||
}
|
||||
for( ; j < m; j++ )
|
||||
d_data[j] = s_buf[j] * (ST)alpha + c_data[j] * (ST)beta;
|
||||
}
|
||||
else
|
||||
{
|
||||
for( j = 0; j < m; j++, c_data += c_step1 )
|
||||
d_data[j] = s_buf[j] * (ST)alpha + c_data[0] * (ST)beta;
|
||||
}
|
||||
}
|
||||
j = m; // all columns handled
|
||||
}
|
||||
else if( vlanes * (int)sizeof(ST) >= 32 )
|
||||
{
|
||||
// j-outer with SIMD: process vlanes columns per group
|
||||
// (skip on narrow SIMD where 4-column scalar is faster)
|
||||
VecT valpha = VecTraits<ST>::setall((ST)alpha);
|
||||
const ST* c_data = _c_data;
|
||||
for( ; j <= m - vlanes; j += vlanes, c_data += vlanes * c_step1 )
|
||||
{
|
||||
const ST* b = _b_data + j;
|
||||
VecT vs = VecTraits<ST>::zero();
|
||||
for( int k = 0; k < n; k++, b += b_step )
|
||||
{
|
||||
VecT va = VecTraits<ST>::setall(a_data[k]);
|
||||
VecT vb = VecTraits<ST>::load(b);
|
||||
vs = v_muladd(va, vb, vs);
|
||||
}
|
||||
vs = v_mul(vs, valpha);
|
||||
if( !_c_data )
|
||||
{
|
||||
v_store(d_data + j, vs);
|
||||
}
|
||||
else if( c_step1 == 1 )
|
||||
{
|
||||
VecT vc = VecTraits<ST>::load(c_data);
|
||||
vs = v_muladd(vc, VecTraits<ST>::setall((ST)beta), vs);
|
||||
v_store(d_data + j, vs);
|
||||
}
|
||||
else
|
||||
{
|
||||
CV_DECL_ALIGNED(CV_SIMD_WIDTH) ST buf[VTraits<VecT>::max_nlanes];
|
||||
v_store_aligned(buf, vs);
|
||||
for( int jj = 0; jj < vlanes; jj++ )
|
||||
d_data[j + jj] = buf[jj] + c_data[jj * c_step1] * (ST)beta;
|
||||
}
|
||||
}
|
||||
// Remaining columns handled by 4-column scalar in caller
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
template<typename T, typename WT> static void
|
||||
GEMMSingleMul( const T* a_data, size_t a_step,
|
||||
@@ -286,20 +618,38 @@ GEMMSingleMul( const T* a_data, size_t a_step,
|
||||
for( j = 0; j < d_size.width; j++, b_data += b_step,
|
||||
c_data += c_step1 )
|
||||
{
|
||||
WT s0(0), s1(0), s2(0), s3(0);
|
||||
WT s0(0);
|
||||
k = 0;
|
||||
#if CV_ENABLE_UNROLLED
|
||||
for( ; k <= n - 4; k += 4 )
|
||||
#if CV_SIMD
|
||||
if( sizeof(WT) == sizeof(double) )
|
||||
{
|
||||
s0 += WT(a_data[k])*WT(b_data[k]);
|
||||
s1 += WT(a_data[k+1])*WT(b_data[k+1]);
|
||||
s2 += WT(a_data[k+2])*WT(b_data[k+2]);
|
||||
s3 += WT(a_data[k+3])*WT(b_data[k+3]);
|
||||
#if CV_SIMD_64F
|
||||
if( sizeof(T) == sizeof(double) )
|
||||
s0 = (WT)simdDotProduct((const double*)a_data, (const double*)b_data, n, k);
|
||||
else if( sizeof(T) == sizeof(float) )
|
||||
s0 = (WT)simdDotProduct_f32f64((const float*)a_data, (const float*)b_data, n, k);
|
||||
else
|
||||
#endif
|
||||
s0 = (WT)simdDotProduct((const float*)a_data, (const float*)b_data, n, k);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
WT s1(0), s2(0), s3(0);
|
||||
#if CV_ENABLE_UNROLLED
|
||||
for( ; k <= n - 4; k += 4 )
|
||||
{
|
||||
s0 += WT(a_data[k])*WT(b_data[k]);
|
||||
s1 += WT(a_data[k+1])*WT(b_data[k+1]);
|
||||
s2 += WT(a_data[k+2])*WT(b_data[k+2]);
|
||||
s3 += WT(a_data[k+3])*WT(b_data[k+3]);
|
||||
}
|
||||
#endif
|
||||
s0 += s1 + s2 + s3;
|
||||
}
|
||||
#endif
|
||||
for( ; k < n; k++ )
|
||||
s0 += WT(a_data[k])*WT(b_data[k]);
|
||||
s0 = (s0+s1+s2+s3)*alpha;
|
||||
s0 = s0*alpha;
|
||||
|
||||
if( !c_data )
|
||||
d_data[j] = T(s0);
|
||||
@@ -323,7 +673,29 @@ GEMMSingleMul( const T* a_data, size_t a_step,
|
||||
a_data = a_buf;
|
||||
}
|
||||
|
||||
for( j = 0; j <= m - 4; j += 4, c_data += 4*c_step1 )
|
||||
j = 0;
|
||||
#if CV_SIMD
|
||||
if( sizeof(WT) == sizeof(double) )
|
||||
{
|
||||
#if CV_SIMD_64F
|
||||
if( sizeof(T) == sizeof(double) )
|
||||
simdGEMM_kj((const double*)a_data, (const double*)_b_data, b_step,
|
||||
(const double*)_c_data, c_step1,
|
||||
(double*)d_data, n, m, alpha, beta, j);
|
||||
else if( sizeof(T) == sizeof(float) )
|
||||
simdGEMM_kj_f32f64((const float*)a_data, (const float*)_b_data, b_step,
|
||||
(const float*)_c_data, c_step1,
|
||||
(float*)d_data, n, m, alpha, beta, j);
|
||||
else
|
||||
#endif
|
||||
simdGEMM_kj((const float*)a_data, (const float*)_b_data, b_step,
|
||||
(const float*)_c_data, c_step1,
|
||||
(float*)d_data, n, m, alpha, beta, j);
|
||||
}
|
||||
#endif
|
||||
c_data = _c_data + j * c_step1;
|
||||
// 4-column j-outer with register accumulators for remaining columns
|
||||
for( ; j <= m - 4; j += 4, c_data += 4*c_step1 )
|
||||
{
|
||||
const T* b = _b_data + j;
|
||||
WT s0(0), s1(0), s2(0), s3(0);
|
||||
@@ -394,6 +766,41 @@ GEMMSingleMul( const T* a_data, size_t a_step,
|
||||
{
|
||||
WT al(a_data[k]);
|
||||
j=0;
|
||||
#if CV_SIMD_64F
|
||||
if( sizeof(WT) == sizeof(double) )
|
||||
{
|
||||
if( sizeof(T) == sizeof(double) )
|
||||
{
|
||||
const int vlanes = VTraits<v_float64>::vlanes();
|
||||
v_float64 val = vx_setall_f64(*((const double*)&al));
|
||||
for( ; j <= m - vlanes; j += vlanes )
|
||||
{
|
||||
v_float64 vd = vx_load((const double*)(d_buf + j));
|
||||
v_float64 vb = vx_load((const double*)(b_data + j));
|
||||
vd = v_muladd(vb, val, vd);
|
||||
v_store((double*)(d_buf + j), vd);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const int vlanes_f32 = VTraits<v_float32>::vlanes();
|
||||
const int vlanes_f64 = VTraits<v_float64>::vlanes();
|
||||
v_float64 val = vx_setall_f64(*((const double*)&al));
|
||||
for( ; j <= m - vlanes_f32; j += vlanes_f32 )
|
||||
{
|
||||
v_float32 vb = vx_load((const float*)(b_data + j));
|
||||
v_float64 vb_lo = v_cvt_f64(vb);
|
||||
v_float64 vb_hi = v_cvt_f64_high(vb);
|
||||
v_float64 vd_lo = vx_load((const double*)(d_buf + j));
|
||||
v_float64 vd_hi = vx_load((const double*)(d_buf + j + vlanes_f64));
|
||||
vd_lo = v_muladd(vb_lo, val, vd_lo);
|
||||
vd_hi = v_muladd(vb_hi, val, vd_hi);
|
||||
v_store((double*)(d_buf + j), vd_lo);
|
||||
v_store((double*)(d_buf + j + vlanes_f64), vd_hi);
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
for( ; j < m; j++ )
|
||||
d_buf[j] += WT(b_data[j])*al;
|
||||
}
|
||||
@@ -410,7 +817,96 @@ GEMMSingleMul( const T* a_data, size_t a_step,
|
||||
}
|
||||
}
|
||||
}
|
||||
#if CV_SIMD_64F
|
||||
template<typename WT>
|
||||
static inline void setDoubleDataZero(WT* d_data, int m, int vlanes)
|
||||
{
|
||||
v_float64 vz = vx_setzero_f64();
|
||||
int j = 0;
|
||||
for( ; j <= m - vlanes; j += vlanes )
|
||||
v_store((double*)(d_data + j), vz);
|
||||
for( ; j < m; j++ )
|
||||
d_data[j] = WT(0);
|
||||
}
|
||||
|
||||
template<typename ST>
|
||||
static inline void scalarTail_kj(const ST* a_k, const ST* b_j,
|
||||
double* d_j, int j, int m)
|
||||
{
|
||||
double al = (double)(*a_k);
|
||||
for( ; j < m; j++ )
|
||||
d_j[j] += al * (double)(b_j[j]);
|
||||
}
|
||||
|
||||
template<typename ST>
|
||||
static inline void simdMulAdd(const ST* b, double* d, int m, double a_val);
|
||||
|
||||
template<>
|
||||
inline void simdMulAdd<float>(
|
||||
const float* b, double* d, int m, double a_val)
|
||||
{
|
||||
const int vlanes_f32 = VTraits<v_float32>::vlanes();
|
||||
const int vlanes_f64 = VTraits<v_float64>::vlanes();
|
||||
v_float64 va = vx_setall_f64(a_val);
|
||||
int j = 0;
|
||||
for( ; j <= m - vlanes_f32; j += vlanes_f32 )
|
||||
{
|
||||
v_float32 vb = vx_load(b + j);
|
||||
v_float64 vb_lo = v_cvt_f64(vb);
|
||||
v_float64 vb_hi = v_cvt_f64_high(vb);
|
||||
v_float64 vd_lo = vx_load(d + j);
|
||||
v_float64 vd_hi = vx_load(d + j + vlanes_f64);
|
||||
vd_lo = v_muladd(va, vb_lo, vd_lo);
|
||||
vd_hi = v_muladd(va, vb_hi, vd_hi);
|
||||
v_store(d + j, vd_lo);
|
||||
v_store(d + j + vlanes_f64, vd_hi);
|
||||
}
|
||||
for( ; j < m; j++ )
|
||||
d[j] += a_val * (double)b[j];
|
||||
}
|
||||
|
||||
template<>
|
||||
inline void simdMulAdd<double>(
|
||||
const double* b, double* d, int m, double a_val)
|
||||
{
|
||||
const int vlanes = VTraits<v_float64>::vlanes();
|
||||
const int stride2 = vlanes * 2;
|
||||
v_float64 va = vx_setall_f64(a_val);
|
||||
int j = 0;
|
||||
for( ; j <= m - stride2; j += stride2 )
|
||||
{
|
||||
v_float64 vd0 = vx_load(d + j);
|
||||
v_float64 vb0 = vx_load(b + j);
|
||||
v_float64 vd1 = vx_load(d + j + vlanes);
|
||||
v_float64 vb1 = vx_load(b + j + vlanes);
|
||||
vd0 = v_muladd(va, vb0, vd0);
|
||||
vd1 = v_muladd(va, vb1, vd1);
|
||||
v_store(d + j, vd0);
|
||||
v_store(d + j + vlanes, vd1);
|
||||
}
|
||||
for( ; j <= m - vlanes; j += vlanes )
|
||||
{
|
||||
v_float64 vd = vx_load(d + j);
|
||||
v_float64 vb = vx_load(b + j);
|
||||
vd = v_muladd(va, vb, vd);
|
||||
v_store(d + j, vd);
|
||||
}
|
||||
for( ; j < m; j++ )
|
||||
d[j] += a_val * b[j];
|
||||
}
|
||||
|
||||
template<typename ST>
|
||||
static inline void simdBlockMul_kj(
|
||||
const ST* a_data, const ST* b_data, size_t b_step,
|
||||
double* d_data, int n, int m, bool do_acc)
|
||||
{
|
||||
const int vlanes = VTraits<v_float64>::vlanes();
|
||||
if( !do_acc )
|
||||
setDoubleDataZero(d_data, m, vlanes);
|
||||
for( int k = 0; k < n; k++, b_data += b_step )
|
||||
simdMulAdd(b_data, d_data, m, (double)a_data[k]);
|
||||
}
|
||||
#endif // CV_SIMD_64F
|
||||
|
||||
template<typename T, typename WT> static void
|
||||
GEMMBlockMul( const T* a_data, size_t a_step,
|
||||
@@ -456,17 +952,35 @@ GEMMBlockMul( const T* a_data, size_t a_step,
|
||||
|
||||
for( j = 0; j < d_size.width; j++, b_data += b_step )
|
||||
{
|
||||
WT s0 = do_acc ? d_data[j]:WT(0), s1(0);
|
||||
for( k = 0; k <= n - 2; k += 2 )
|
||||
WT s0 = do_acc ? d_data[j] : WT(0);
|
||||
k = 0;
|
||||
#if CV_SIMD
|
||||
if( sizeof(WT) == sizeof(double) )
|
||||
{
|
||||
s0 += WT(a_data[k])*WT(b_data[k]);
|
||||
s1 += WT(a_data[k+1])*WT(b_data[k+1]);
|
||||
#if CV_SIMD_64F
|
||||
if( sizeof(T) == sizeof(double) )
|
||||
s0 += (WT)simdDotProduct((const double*)a_data, (const double*)b_data, n, k);
|
||||
else if( sizeof(T) == sizeof(float) )
|
||||
s0 += (WT)simdDotProduct_f32f64((const float*)a_data, (const float*)b_data, n, k);
|
||||
else
|
||||
#endif
|
||||
s0 += (WT)simdDotProduct((const float*)a_data, (const float*)b_data, n, k);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
WT s1(0);
|
||||
for( ; k <= n - 2; k += 2 )
|
||||
{
|
||||
s0 += WT(a_data[k])*WT(b_data[k]);
|
||||
s1 += WT(a_data[k+1])*WT(b_data[k+1]);
|
||||
}
|
||||
s0 += s1;
|
||||
}
|
||||
|
||||
for( ; k < n; k++ )
|
||||
s0 += WT(a_data[k])*WT(b_data[k]);
|
||||
|
||||
d_data[j] = s0 + s1;
|
||||
d_data[j] = s0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -482,40 +996,55 @@ GEMMBlockMul( const T* a_data, size_t a_step,
|
||||
a_buf[k] = a_data[a_step1*k];
|
||||
a_data = a_buf;
|
||||
}
|
||||
|
||||
for( j = 0; j <= m - 4; j += 4 )
|
||||
#if CV_SIMD_64F
|
||||
if( sizeof(WT) == sizeof(double) )
|
||||
{
|
||||
WT s0, s1, s2, s3;
|
||||
const T* b = b_data + j;
|
||||
|
||||
if( do_acc )
|
||||
{
|
||||
s0 = d_data[j]; s1 = d_data[j+1];
|
||||
s2 = d_data[j+2]; s3 = d_data[j+3];
|
||||
}
|
||||
if( sizeof(T) == sizeof(double) )
|
||||
simdBlockMul_kj(
|
||||
(const double*)a_data, (const double*)b_data, b_step,
|
||||
(double*)d_data, n, m, do_acc != 0);
|
||||
else
|
||||
s0 = s1 = s2 = s3 = WT(0);
|
||||
|
||||
for( k = 0; k < n; k++, b += b_step )
|
||||
simdBlockMul_kj(
|
||||
(const float*)a_data, (const float*)b_data, b_step,
|
||||
(double*)d_data, n, m, do_acc != 0);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
{
|
||||
for( j = 0; j <= m - 4; j += 4 )
|
||||
{
|
||||
WT a(a_data[k]);
|
||||
s0 += a * WT(b[0]); s1 += a * WT(b[1]);
|
||||
s2 += a * WT(b[2]); s3 += a * WT(b[3]);
|
||||
WT s0, s1, s2, s3;
|
||||
const T* b = b_data + j;
|
||||
|
||||
if( do_acc )
|
||||
{
|
||||
s0 = d_data[j]; s1 = d_data[j+1];
|
||||
s2 = d_data[j+2]; s3 = d_data[j+3];
|
||||
}
|
||||
else
|
||||
s0 = s1 = s2 = s3 = WT(0);
|
||||
|
||||
for( k = 0; k < n; k++, b += b_step )
|
||||
{
|
||||
WT a(a_data[k]);
|
||||
s0 += a * WT(b[0]); s1 += a * WT(b[1]);
|
||||
s2 += a * WT(b[2]); s3 += a * WT(b[3]);
|
||||
}
|
||||
|
||||
d_data[j] = s0; d_data[j+1] = s1;
|
||||
d_data[j+2] = s2; d_data[j+3] = s3;
|
||||
}
|
||||
|
||||
d_data[j] = s0; d_data[j+1] = s1;
|
||||
d_data[j+2] = s2; d_data[j+3] = s3;
|
||||
}
|
||||
for( ; j < m; j++ )
|
||||
{
|
||||
const T* b = b_data + j;
|
||||
WT s0 = do_acc ? d_data[j] : WT(0);
|
||||
|
||||
for( ; j < m; j++ )
|
||||
{
|
||||
const T* b = b_data + j;
|
||||
WT s0 = do_acc ? d_data[j] : WT(0);
|
||||
for( k = 0; k < n; k++, b += b_step )
|
||||
s0 += WT(a_data[k]) * WT(b[0]);
|
||||
|
||||
for( k = 0; k < n; k++, b += b_step )
|
||||
s0 += WT(a_data[k]) * WT(b[0]);
|
||||
|
||||
d_data[j] = s0;
|
||||
d_data[j] = s0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -680,7 +1209,7 @@ static void GEMMBlockMul_32fc( const Complexf* a_data, size_t a_step,
|
||||
Complexd* d_data, size_t d_step,
|
||||
Size a_size, Size d_size, int flags )
|
||||
{
|
||||
GEMMBlockMul(a_data, a_step, b_data, b_step, d_data, d_step, a_size, d_size, flags);
|
||||
GEMMBlockMul<Complexf,Complexd>(a_data, a_step, b_data, b_step, d_data, d_step, a_size, d_size, flags);
|
||||
}
|
||||
|
||||
|
||||
@@ -689,7 +1218,7 @@ static void GEMMBlockMul_64fc( const Complexd* a_data, size_t a_step,
|
||||
Complexd* d_data, size_t d_step,
|
||||
Size a_size, Size d_size, int flags )
|
||||
{
|
||||
GEMMBlockMul(a_data, a_step, b_data, b_step, d_data, d_step, a_size, d_size, flags);
|
||||
GEMMBlockMul<Complexd,Complexd>(a_data, a_step, b_data, b_step, d_data, d_step, a_size, d_size, flags);
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -1738,8 +1738,16 @@ MatExpr Mat::mul(InputArray m, double scale) const
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
Mat b = m.getMat();
|
||||
// Unless the argument is a refcounted Mat/UMat, the header returned by getMat() may be
|
||||
// a non-owning view of caller memory (e.g. a scalar bound to _InputArray(const double&),
|
||||
// a Matx or a Vec) that the returned MatExpr can outlive, so snapshot it.
|
||||
// See https://github.com/opencv/opencv/issues/23577
|
||||
if( !m.isMat() && !m.isUMat() )
|
||||
b = b.clone();
|
||||
|
||||
MatExpr e;
|
||||
MatOp_Bin::makeExpr(e, '*', *this, m.getMat(), scale);
|
||||
MatOp_Bin::makeExpr(e, '*', *this, b, scale);
|
||||
return e;
|
||||
}
|
||||
|
||||
|
||||
@@ -6,11 +6,8 @@
|
||||
#include "opencv2/core/mat.hpp"
|
||||
#include "opencl_kernels_core.hpp"
|
||||
|
||||
#undef HAVE_IPP
|
||||
#undef CV_IPP_RUN_FAST
|
||||
#define CV_IPP_RUN_FAST(f, ...)
|
||||
#undef CV_IPP_RUN
|
||||
#define CV_IPP_RUN(c, f, ...)
|
||||
#define IPP_DISABLE_REDUCE 1
|
||||
#define IPP_DISABLE_SORT 0
|
||||
|
||||
/*************************************************************************************************\
|
||||
Matrix Operations
|
||||
@@ -341,9 +338,13 @@ cv::Mat cv::Mat::cross(InputArray _m) const
|
||||
namespace cv
|
||||
{
|
||||
|
||||
typedef void (*ReduceSumFunc)(const Mat& src, Mat& dst);
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
|
||||
template <typename T, typename WT, typename Op>
|
||||
struct ReduceR_SIMD
|
||||
@@ -354,7 +355,6 @@ struct ReduceR_SIMD
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<typename T, typename ST, typename WT, class Op, class OpInit>
|
||||
class ReduceR_Invoker : public ParallelLoopBody
|
||||
{
|
||||
@@ -474,8 +474,6 @@ reduceC_( const Mat& srcmat, Mat& dstmat)
|
||||
parallel_for_(Range(0, srcmat.size().height), body);
|
||||
}
|
||||
|
||||
typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
|
||||
}
|
||||
|
||||
#define reduceSumR8u32s reduceR_<uchar, int, OpAdd<int>, OpNop<int> >
|
||||
@@ -512,7 +510,7 @@ typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
#define reduceMinR32f reduceR_<float, float, OpMin<float> >
|
||||
#define reduceMinR64f reduceR_<double,double,OpMin<double> >
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#if defined(HAVE_IPP) && !IPP_DISABLE_REDUCE
|
||||
static inline bool ipp_reduceSumC_8u16u16s32f_64f(const cv::Mat& srcmat, cv::Mat& dstmat)
|
||||
{
|
||||
int sstep = (int)srcmat.step, stype = srcmat.type(),
|
||||
@@ -601,7 +599,7 @@ static inline void reduceSumC_8u16u16s32f_64f(const cv::Mat& srcmat, cv::Mat& ds
|
||||
#define reduceSum2C32f32f reduceC_<float, float, OpAddSqr<float>, OpSqr<float> >
|
||||
#define reduceSum2C64f64f reduceC_<double,double,OpAddSqr<double>,OpSqr<double> >
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#if defined(HAVE_IPP) && !IPP_DISABLE_REDUCE
|
||||
#define reduceSumC8u64f reduceSumC_8u16u16s32f_64f
|
||||
#define reduceSumC16u64f reduceSumC_8u16u16s32f_64f
|
||||
#define reduceSumC16s64f reduceSumC_8u16u16s32f_64f
|
||||
@@ -611,14 +609,14 @@ static inline void reduceSumC_8u16u16s32f_64f(const cv::Mat& srcmat, cv::Mat& ds
|
||||
#define reduceSumC16u64f reduceC_<ushort,double,OpAdd<double> >
|
||||
#define reduceSumC16s64f reduceC_<short, double,OpAdd<double> >
|
||||
#define reduceSumC32f64f reduceC_<float, double,OpAdd<double> >
|
||||
#endif
|
||||
|
||||
#define reduceSum2C8u64f reduceC_<uchar, double,OpAddSqr<int>, OpSqr<int> >
|
||||
#define reduceSum2C16u64f reduceC_<ushort,double,OpAddSqr<double>,OpSqr<double> >
|
||||
#define reduceSum2C16s64f reduceC_<short, double,OpAddSqr<double>,OpSqr<double> >
|
||||
#define reduceSum2C32f64f reduceC_<float, double,OpAddSqr<double>,OpSqr<double> >
|
||||
#endif
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#if defined(HAVE_IPP) && !IPP_DISABLE_REDUCE
|
||||
#define REDUCE_OP(favor, optype, type1, type2) \
|
||||
static inline bool ipp_reduce##optype##C##favor(const cv::Mat& srcmat, cv::Mat& dstmat) \
|
||||
{ \
|
||||
@@ -643,7 +641,7 @@ static inline void reduce##optype##C##favor(const cv::Mat& srcmat, cv::Mat& dstm
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#if defined(HAVE_IPP) && !IPP_DISABLE_REDUCE
|
||||
REDUCE_OP(8u, Max, uchar, uchar)
|
||||
REDUCE_OP(16u, Max, ushort, ushort)
|
||||
REDUCE_OP(16s, Max, short, short)
|
||||
@@ -656,7 +654,7 @@ REDUCE_OP(32f, Max, float, float)
|
||||
#endif
|
||||
#define reduceMaxC64f reduceC_<double,double,OpMax<double> >
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#if defined(HAVE_IPP) && !IPP_DISABLE_REDUCE
|
||||
REDUCE_OP(8u, Min, uchar, uchar)
|
||||
REDUCE_OP(16u, Min, ushort, ushort)
|
||||
REDUCE_OP(16s, Min, short, short)
|
||||
@@ -821,9 +819,9 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
{
|
||||
if( op == REDUCE_SUM )
|
||||
{
|
||||
ReduceSumFunc simd_func = getReduceRSumFunc(sdepth, ddepth);
|
||||
ReduceFunc simd_func = getReduceRSumFunc(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = (ReduceFunc)simd_func;
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32S)
|
||||
func = reduceSumR8u32s;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32F)
|
||||
@@ -899,9 +897,11 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
{
|
||||
if(op == REDUCE_SUM)
|
||||
{
|
||||
ReduceSumFunc simd_func = getReduceCSumFunc(sdepth, ddepth);
|
||||
ReduceFunc simd_func = op0 == REDUCE_AVG
|
||||
? getReduceCAvgFunc(sdepth, ddepth)
|
||||
: getReduceCSumFunc(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = (ReduceFunc)simd_func;
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32S)
|
||||
func = reduceSumC8u32s;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32F)
|
||||
@@ -925,7 +925,10 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
}
|
||||
else if(op == REDUCE_MAX)
|
||||
{
|
||||
if(sdepth == CV_8U && ddepth == CV_8U)
|
||||
ReduceFunc simd_func = getReduceCMaxFunc(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_8U)
|
||||
func = reduceMaxC8u;
|
||||
else if(sdepth == CV_16U && ddepth == CV_16U)
|
||||
func = reduceMaxC16u;
|
||||
@@ -938,7 +941,10 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
}
|
||||
else if(op == REDUCE_MIN)
|
||||
{
|
||||
if(sdepth == CV_8U && ddepth == CV_8U)
|
||||
ReduceFunc simd_func = getReduceCMinFunc(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_8U)
|
||||
func = reduceMinC8u;
|
||||
else if(sdepth == CV_16U && ddepth == CV_16U)
|
||||
func = reduceMinC16u;
|
||||
@@ -951,7 +957,10 @@ void cv::reduce(InputArray _src, OutputArray _dst, int dim, int op, int dtype)
|
||||
}
|
||||
else if(op == REDUCE_SUM2)
|
||||
{
|
||||
if(sdepth == CV_8U && ddepth == CV_32S)
|
||||
ReduceFunc simd_func = getReduceCSum2Func(sdepth, ddepth);
|
||||
if(simd_func)
|
||||
func = simd_func;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32S)
|
||||
func = reduceSum2C8u32s;
|
||||
else if(sdepth == CV_8U && ddepth == CV_32F)
|
||||
func = reduceSum2C8u32f;
|
||||
@@ -992,7 +1001,6 @@ namespace cv
|
||||
|
||||
template<typename T> static void sort_( const Mat& src, Mat& dst, int flags )
|
||||
{
|
||||
AutoBuffer<T> buf;
|
||||
int n, len;
|
||||
bool sortRows = (flags & 1) == SORT_EVERY_ROW;
|
||||
bool inplace = src.data == dst.data;
|
||||
@@ -1001,45 +1009,48 @@ template<typename T> static void sort_( const Mat& src, Mat& dst, int flags )
|
||||
if( sortRows )
|
||||
n = src.rows, len = src.cols;
|
||||
else
|
||||
{
|
||||
n = src.cols, len = src.rows;
|
||||
buf.allocate(len);
|
||||
}
|
||||
T* bptr = buf.data();
|
||||
|
||||
for( int i = 0; i < n; i++ )
|
||||
parallel_for_(Range(0, n), [&](const Range& range)
|
||||
{
|
||||
T* ptr = bptr;
|
||||
if( sortRows )
|
||||
{
|
||||
T* dptr = dst.ptr<T>(i);
|
||||
if( !inplace )
|
||||
{
|
||||
const T* sptr = src.ptr<T>(i);
|
||||
memcpy(dptr, sptr, sizeof(T) * len);
|
||||
}
|
||||
ptr = dptr;
|
||||
}
|
||||
else
|
||||
{
|
||||
for( int j = 0; j < len; j++ )
|
||||
ptr[j] = src.ptr<T>(j)[i];
|
||||
}
|
||||
|
||||
std::sort( ptr, ptr + len );
|
||||
if( sortDescending )
|
||||
{
|
||||
for( int j = 0; j < len/2; j++ )
|
||||
std::swap(ptr[j], ptr[len-1-j]);
|
||||
}
|
||||
|
||||
AutoBuffer<T> buf;
|
||||
if( !sortRows )
|
||||
for( int j = 0; j < len; j++ )
|
||||
dst.ptr<T>(j)[i] = ptr[j];
|
||||
}
|
||||
buf.allocate(len);
|
||||
T* bptr = buf.data();
|
||||
|
||||
for( int i = range.start; i < range.end; i++ )
|
||||
{
|
||||
T* ptr = bptr;
|
||||
if( sortRows )
|
||||
{
|
||||
T* dptr = dst.ptr<T>(i);
|
||||
if( !inplace )
|
||||
{
|
||||
const T* sptr = src.ptr<T>(i);
|
||||
memcpy(dptr, sptr, sizeof(T) * len);
|
||||
}
|
||||
ptr = dptr;
|
||||
}
|
||||
else
|
||||
{
|
||||
for( int j = 0; j < len; j++ )
|
||||
ptr[j] = src.ptr<T>(j)[i];
|
||||
}
|
||||
|
||||
std::sort( ptr, ptr + len );
|
||||
if( sortDescending )
|
||||
{
|
||||
for( int j = 0; j < len/2; j++ )
|
||||
std::swap(ptr[j], ptr[len-1-j]);
|
||||
}
|
||||
|
||||
if( !sortRows )
|
||||
for( int j = 0; j < len; j++ )
|
||||
dst.ptr<T>(j)[i] = ptr[j];
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#if defined(HAVE_IPP) && !IPP_DISABLE_SORT
|
||||
typedef IppStatus (CV_STDCALL *IppSortFunc)(void *pSrcDst, int len, Ipp8u *pBuffer);
|
||||
|
||||
static IppSortFunc getSortFunc(int depth, bool sortDescending)
|
||||
@@ -1076,54 +1087,64 @@ static bool ipp_sort(const Mat& src, Mat& dst, int flags)
|
||||
if(!ippsSortRadix_I)
|
||||
return false;
|
||||
|
||||
if(sortRows)
|
||||
{
|
||||
AutoBuffer<Ipp8u> buffer;
|
||||
int bufferSize;
|
||||
if(ippsSortRadixGetBufferSize(src.cols, type, &bufferSize) < 0)
|
||||
return false;
|
||||
|
||||
buffer.allocate(bufferSize);
|
||||
|
||||
if(!inplace)
|
||||
src.copyTo(dst);
|
||||
|
||||
for(int i = 0; i < dst.rows; i++)
|
||||
{
|
||||
if(CV_INSTRUMENT_FUN_IPP(ippsSortRadix_I, (void*)dst.ptr(i), dst.cols, buffer.data()) < 0)
|
||||
return false;
|
||||
}
|
||||
}
|
||||
int n, len;
|
||||
if( sortRows )
|
||||
n = src.rows, len = src.cols;
|
||||
else
|
||||
n = src.cols, len = src.rows;
|
||||
|
||||
int bufferSize;
|
||||
if(ippsSortRadixGetBufferSize(len, type, &bufferSize) < 0)
|
||||
return false;
|
||||
|
||||
// Each row/column sorts independently, so parallelize over n like sort_ does.
|
||||
// IPP scratch (buffer) and per-column temporaries are thread-local; a shared flag
|
||||
// records IPP failures since the lambda cannot return from ipp_sort.
|
||||
volatile bool ok = true;
|
||||
parallel_for_(Range(0, n), [&](const Range& range)
|
||||
{
|
||||
AutoBuffer<Ipp8u> buffer;
|
||||
int bufferSize;
|
||||
if(ippsSortRadixGetBufferSize(src.rows, type, &bufferSize) < 0)
|
||||
return false;
|
||||
|
||||
buffer.allocate(bufferSize);
|
||||
|
||||
Mat row(1, src.rows, src.type());
|
||||
Mat srcSub;
|
||||
Mat dstSub;
|
||||
AutoBuffer<Ipp8u> buffer(bufferSize);
|
||||
Mat row;
|
||||
Rect subRect(0,0,1,src.rows);
|
||||
if( !sortRows )
|
||||
row.create(1, src.rows, src.type());
|
||||
|
||||
for(int i = 0; i < src.cols; i++)
|
||||
for( int i = range.start; i < range.end; i++ )
|
||||
{
|
||||
subRect.x = i;
|
||||
srcSub = Mat(src, subRect);
|
||||
dstSub = Mat(dst, subRect);
|
||||
srcSub.copyTo(row);
|
||||
if( !ok )
|
||||
break;
|
||||
void* ptr;
|
||||
if( sortRows )
|
||||
{
|
||||
uchar* dptr = dst.ptr(i);
|
||||
if( !inplace )
|
||||
memcpy(dptr, src.ptr(i), src.cols * src.elemSize());
|
||||
ptr = dptr;
|
||||
}
|
||||
else
|
||||
{
|
||||
subRect.x = i;
|
||||
Mat(src, subRect).copyTo(row);
|
||||
ptr = row.ptr();
|
||||
}
|
||||
|
||||
if(CV_INSTRUMENT_FUN_IPP(ippsSortRadix_I, (void*)row.ptr(), dst.rows, buffer.data()) < 0)
|
||||
return false;
|
||||
if( CV_INSTRUMENT_FUN_IPP(ippsSortRadix_I, ptr, len, buffer.data()) < 0 )
|
||||
{
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
|
||||
row = row.reshape(1, dstSub.rows);
|
||||
row.copyTo(dstSub);
|
||||
if( !sortRows )
|
||||
{
|
||||
Mat dstSub(dst, subRect);
|
||||
row = row.reshape(1, dstSub.rows);
|
||||
row.copyTo(dstSub);
|
||||
row = row.reshape(1, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
return true;
|
||||
return ok;
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1187,7 +1208,7 @@ template<typename T> static void sortIdx_( const Mat& src, Mat& dst, int flags )
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef HAVE_IPP
|
||||
#if defined(HAVE_IPP) && !IPP_DISABLE_SORT
|
||||
typedef IppStatus (CV_STDCALL *IppSortIndexFunc)(const void* pSrc, Ipp32s srcStrideBytes, Ipp32s *pDstIndx, int len, Ipp8u *pBuffer);
|
||||
|
||||
static IppSortIndexFunc getSortIndexFunc(int depth, bool sortDescending)
|
||||
@@ -1278,7 +1299,9 @@ void cv::sort( InputArray _src, OutputArray _dst, int flags )
|
||||
CV_Assert( src.dims <= 2 && src.channels() == 1 );
|
||||
_dst.create( src.size(), src.type() );
|
||||
Mat dst = _dst.getMat();
|
||||
#if !IPP_DISABLE_SORT
|
||||
CV_IPP_RUN_FAST(ipp_sort(src, dst, flags));
|
||||
#endif
|
||||
|
||||
static SortFunc tab[CV_DEPTH_MAX] =
|
||||
{
|
||||
@@ -1303,7 +1326,9 @@ void cv::sortIdx( InputArray _src, OutputArray _dst, int flags )
|
||||
_dst.create( src.size(), CV_32S );
|
||||
dst = _dst.getMat();
|
||||
|
||||
#if !IPP_DISABLE_SORT
|
||||
CV_IPP_RUN_FAST(ipp_sortIdx(src, dst, flags));
|
||||
#endif
|
||||
|
||||
static SortFunc tab[CV_DEPTH_MAX] =
|
||||
{
|
||||
|
||||
@@ -551,6 +551,33 @@ void transposeND(InputArray src_, const std::vector<int>& order, OutputArray dst
|
||||
CV_CheckEQ(inp.channels(), 1, "Input array should be single-channel");
|
||||
CV_CheckEQ(order.size(), static_cast<size_t>(inp.dims), "Number of dimensions shouldn't change");
|
||||
|
||||
bool isIdentityOrder = true;
|
||||
for (size_t i = 0; i < order.size(); ++i)
|
||||
{
|
||||
if (order[i] != static_cast<int>(i))
|
||||
{
|
||||
isIdentityOrder = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (isIdentityOrder)
|
||||
{
|
||||
dst_.create(inp.dims, inp.size.p, inp.type());
|
||||
Mat out = dst_.getMat();
|
||||
CV_Assert(out.isContinuous());
|
||||
|
||||
if (inp.data != out.data)
|
||||
inp.copyTo(out);
|
||||
return;
|
||||
}
|
||||
|
||||
const bool is2DSwap = inp.dims == 2 && order.size() == 2 && order[0] == 1 && order[1] == 0;
|
||||
if (is2DSwap)
|
||||
{
|
||||
transpose(inp, dst_);
|
||||
return;
|
||||
}
|
||||
|
||||
auto order_ = order;
|
||||
std::sort(order_.begin(), order_.end());
|
||||
for (size_t i = 0; i < order_.size(); ++i)
|
||||
@@ -558,7 +585,7 @@ void transposeND(InputArray src_, const std::vector<int>& order, OutputArray dst
|
||||
CV_CheckEQ(static_cast<size_t>(order_[i]), i, "New order should be a valid permutation of the old one");
|
||||
}
|
||||
|
||||
std::vector<int> newShape(order.size());
|
||||
AutoBuffer<int> newShape(order.size());
|
||||
for (size_t i = 0; i < order.size(); ++i)
|
||||
{
|
||||
newShape[i] = inp.size[order[i]];
|
||||
@@ -582,7 +609,7 @@ void transposeND(InputArray src_, const std::vector<int>& order, OutputArray dst
|
||||
size_t continuous_size = continuous_idx == 0 ? out.total() : out.step1(continuous_idx - 1);
|
||||
size_t outer_size = out.total() / continuous_size;
|
||||
|
||||
std::vector<size_t> steps(order.size());
|
||||
AutoBuffer<size_t> steps(order.size());
|
||||
for (int i = 0; i < static_cast<int>(steps.size()); ++i)
|
||||
{
|
||||
steps[i] = inp.step1(order[i]);
|
||||
@@ -1229,7 +1256,7 @@ void broadcast(InputArray _src, InputArray _shape, OutputArray _dst) {
|
||||
// impl
|
||||
_dst.create(dims_shape, shape.ptr<int>(), src.type());
|
||||
Mat dst = _dst.getMat();
|
||||
std::vector<int> is_same_shape(dims_shape, 0);
|
||||
AutoBuffer<int> is_same_shape(dims_shape, 0);
|
||||
for (int i = 0; i < static_cast<int>(shape_src.size()); ++i) {
|
||||
if (shape_src[i] == ptr_shape[i]) {
|
||||
is_same_shape[i] = 1;
|
||||
@@ -1328,7 +1355,7 @@ void broadcast(InputArray _src, InputArray _shape, OutputArray _dst) {
|
||||
std::memcpy(p_dst + dst_offset, p_src + src_offset, dst.elemSize());
|
||||
}
|
||||
// broadcast copy (dst inplace)
|
||||
std::vector<int> cumulative_shape(dims_shape, 1);
|
||||
AutoBuffer<int> cumulative_shape(dims_shape, 1);
|
||||
int total = static_cast<int>(dst.total());
|
||||
for (int i = dims_shape - 1; i >= 0; --i) {
|
||||
cumulative_shape[i] = static_cast<int>(total / ptr_shape[i]);
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#include "precomp.hpp"
|
||||
#include "stat.hpp"
|
||||
@@ -31,7 +32,35 @@ struct SumSqr_SIMD<uchar, int, int>
|
||||
{
|
||||
int operator () (const uchar * src0, const uchar * mask, int * sum, int * sqsum, int len, int cn) const
|
||||
{
|
||||
if (mask || (cn != 1 && cn != 2 && cn != 4))
|
||||
if (mask)
|
||||
return 0;
|
||||
// cn==3: deinterleave so each lane belongs to one channel, then reduce per
|
||||
// channel. sum/sqsum stay exact in s32 within the dispatcher's 1<<15 block.
|
||||
if (cn == 3)
|
||||
{
|
||||
const int vl8 = VTraits<v_uint8>::vlanes();
|
||||
v_int32 vs0 = vx_setzero_s32(), vs1 = vx_setzero_s32(), vs2 = vx_setzero_s32();
|
||||
v_int32 vq0 = vx_setzero_s32(), vq1 = vx_setzero_s32(), vq2 = vx_setzero_s32();
|
||||
auto acc = [](const v_uint8& ch, v_int32& vs, v_int32& vq)
|
||||
{
|
||||
v_uint16 lo, hi; v_expand(ch, lo, hi);
|
||||
v_uint32 s0, s1, s2, s3; v_expand(lo, s0, s1); v_expand(hi, s2, s3);
|
||||
vs = v_add(vs, v_reinterpret_as_s32(v_add(v_add(s0, s1), v_add(s2, s3))));
|
||||
v_int16 d0 = v_reinterpret_as_s16(lo), d1 = v_reinterpret_as_s16(hi);
|
||||
vq = v_add(vq, v_add(v_dotprod(d0, d0), v_dotprod(d1, d1)));
|
||||
};
|
||||
int e = 0;
|
||||
for (; e <= len - vl8; e += vl8)
|
||||
{
|
||||
v_uint8 a, b, c; v_load_deinterleave(src0 + e * 3, a, b, c);
|
||||
acc(a, vs0, vq0); acc(b, vs1, vq1); acc(c, vs2, vq2);
|
||||
}
|
||||
sum[0] += v_reduce_sum(vs0); sum[1] += v_reduce_sum(vs1); sum[2] += v_reduce_sum(vs2);
|
||||
sqsum[0] += v_reduce_sum(vq0); sqsum[1] += v_reduce_sum(vq1); sqsum[2] += v_reduce_sum(vq2);
|
||||
v_cleanup();
|
||||
return e;
|
||||
}
|
||||
if (cn != 1 && cn != 2 && cn != 4)
|
||||
return 0;
|
||||
len *= cn;
|
||||
|
||||
@@ -98,7 +127,32 @@ struct SumSqr_SIMD<schar, int, int>
|
||||
{
|
||||
int operator () (const schar * src0, const uchar * mask, int * sum, int * sqsum, int len, int cn) const
|
||||
{
|
||||
if (mask || (cn != 1 && cn != 2 && cn != 4))
|
||||
if (mask)
|
||||
return 0;
|
||||
if (cn == 3)
|
||||
{
|
||||
const int vl8 = VTraits<v_int8>::vlanes();
|
||||
v_int32 vs0 = vx_setzero_s32(), vs1 = vx_setzero_s32(), vs2 = vx_setzero_s32();
|
||||
v_int32 vq0 = vx_setzero_s32(), vq1 = vx_setzero_s32(), vq2 = vx_setzero_s32();
|
||||
auto acc = [](const v_int8& ch, v_int32& vs, v_int32& vq)
|
||||
{
|
||||
v_int16 lo, hi; v_expand(ch, lo, hi);
|
||||
v_int32 s0, s1, s2, s3; v_expand(lo, s0, s1); v_expand(hi, s2, s3);
|
||||
vs = v_add(vs, v_add(v_add(s0, s1), v_add(s2, s3)));
|
||||
vq = v_add(vq, v_add(v_dotprod(lo, lo), v_dotprod(hi, hi)));
|
||||
};
|
||||
int e = 0;
|
||||
for (; e <= len - vl8; e += vl8)
|
||||
{
|
||||
v_int8 a, b, c; v_load_deinterleave(src0 + e * 3, a, b, c);
|
||||
acc(a, vs0, vq0); acc(b, vs1, vq1); acc(c, vs2, vq2);
|
||||
}
|
||||
sum[0] += v_reduce_sum(vs0); sum[1] += v_reduce_sum(vs1); sum[2] += v_reduce_sum(vs2);
|
||||
sqsum[0] += v_reduce_sum(vq0); sqsum[1] += v_reduce_sum(vq1); sqsum[2] += v_reduce_sum(vq2);
|
||||
v_cleanup();
|
||||
return e;
|
||||
}
|
||||
if (cn != 1 && cn != 2 && cn != 4)
|
||||
return 0;
|
||||
len *= cn;
|
||||
|
||||
@@ -160,6 +214,212 @@ struct SumSqr_SIMD<schar, int, int>
|
||||
}
|
||||
};
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
// Expand a 16-bit vector into two s32 halves. The unsigned overload reinterprets
|
||||
// after a widening expand (values 0..65535 stay non-negative in s32, so the
|
||||
// subsequent v_cvt_f64 is exact); the signed overload expands directly. This is
|
||||
// the only real difference between the ushort and short reductions below.
|
||||
static inline void sumSqrExpandS32(const v_uint16& v, v_int32& lo, v_int32& hi)
|
||||
{
|
||||
v_uint32 a, b; v_expand(v, a, b);
|
||||
lo = v_reinterpret_as_s32(a); hi = v_reinterpret_as_s32(b);
|
||||
}
|
||||
static inline void sumSqrExpandS32(const v_int16& v, v_int32& lo, v_int32& hi)
|
||||
{
|
||||
v_expand(v, lo, hi);
|
||||
}
|
||||
|
||||
// Shared 16-bit -> {s32 sum, f64 sqsum} reduction for ushort/short. The register
|
||||
// type is deduced from the pointer, so the body is identical for both depths.
|
||||
template <typename T>
|
||||
static inline int sumSqr16ToF64(const T* src0, int* sum, double* sqsum, int len, int cn)
|
||||
{
|
||||
typedef decltype(vx_load(src0)) VT; // v_uint16 or v_int16
|
||||
|
||||
if (cn == 3)
|
||||
{
|
||||
const int vl = VTraits<VT>::vlanes();
|
||||
v_int32 vs0 = vx_setzero_s32(), vs1 = vx_setzero_s32(), vs2 = vx_setzero_s32();
|
||||
v_float64 vq0 = vx_setzero_f64(), vq1 = vx_setzero_f64(), vq2 = vx_setzero_f64();
|
||||
auto acc = [](const VT& ch, v_int32& vs, v_float64& vq)
|
||||
{
|
||||
v_int32 lo, hi; sumSqrExpandS32(ch, lo, hi);
|
||||
vs = v_add(vs, v_add(lo, hi));
|
||||
v_float64 f0 = v_cvt_f64(lo), f1 = v_cvt_f64_high(lo);
|
||||
v_float64 f2 = v_cvt_f64(hi), f3 = v_cvt_f64_high(hi);
|
||||
vq = v_fma(f0, f0, v_fma(f1, f1, v_fma(f2, f2, v_fma(f3, f3, vq))));
|
||||
};
|
||||
int e = 0;
|
||||
for (; e <= len - vl; e += vl)
|
||||
{
|
||||
VT a, b, c; v_load_deinterleave(src0 + e * 3, a, b, c);
|
||||
acc(a, vs0, vq0); acc(b, vs1, vq1); acc(c, vs2, vq2);
|
||||
}
|
||||
sum[0] += v_reduce_sum(vs0); sum[1] += v_reduce_sum(vs1); sum[2] += v_reduce_sum(vs2);
|
||||
sqsum[0] += v_reduce_sum(vq0); sqsum[1] += v_reduce_sum(vq1); sqsum[2] += v_reduce_sum(vq2);
|
||||
v_cleanup();
|
||||
return e;
|
||||
}
|
||||
if (cn != 1 && cn != 2 && cn != 4)
|
||||
return 0;
|
||||
len *= cn;
|
||||
|
||||
const int vl16 = VTraits<VT>::vlanes();
|
||||
const int vl32 = VTraits<v_int32>::vlanes();
|
||||
const int vl64 = VTraits<v_float64>::vlanes();
|
||||
|
||||
// The lane->channel scatter below (i % cn) and the returned pixel count
|
||||
// (x / cn) require the lane counts to be a multiple of cn. This always holds
|
||||
// for fixed-width SIMD (vlanes is a power of two >= cn for cn in {1,2,4}),
|
||||
// but scalable backends (e.g. RVV) may report a non-multiple; fall back to
|
||||
// scalar there. vl16 == 2*vl32, so checking vl32 covers both scatter loops.
|
||||
if (vl32 % cn != 0)
|
||||
return 0;
|
||||
|
||||
v_int32 v_sum = vx_setzero_s32();
|
||||
v_float64 v_sq0 = vx_setzero_f64(), v_sq1 = vx_setzero_f64();
|
||||
v_float64 v_sq2 = vx_setzero_f64(), v_sq3 = vx_setzero_f64();
|
||||
|
||||
int x = 0;
|
||||
for (; x <= len - vl16; x += vl16)
|
||||
{
|
||||
VT v_src = vx_load(src0 + x);
|
||||
v_int32 lo, hi; sumSqrExpandS32(v_src, lo, hi);
|
||||
v_sum = v_add(v_sum, v_add(lo, hi));
|
||||
|
||||
v_float64 f0 = v_cvt_f64(lo);
|
||||
v_float64 f1 = v_cvt_f64_high(lo);
|
||||
v_float64 f2 = v_cvt_f64(hi);
|
||||
v_float64 f3 = v_cvt_f64_high(hi);
|
||||
v_sq0 = v_fma(f0, f0, v_sq0);
|
||||
v_sq1 = v_fma(f1, f1, v_sq1);
|
||||
v_sq2 = v_fma(f2, f2, v_sq2);
|
||||
v_sq3 = v_fma(f3, f3, v_sq3);
|
||||
}
|
||||
|
||||
int CV_DECL_ALIGNED(CV_SIMD_WIDTH) sbuf[VTraits<v_int32>::max_nlanes];
|
||||
double CV_DECL_ALIGNED(CV_SIMD_WIDTH) qbuf[VTraits<v_float64>::max_nlanes * 4];
|
||||
v_store_aligned(sbuf, v_sum);
|
||||
v_store_aligned(qbuf + vl64 * 0, v_sq0);
|
||||
v_store_aligned(qbuf + vl64 * 1, v_sq1);
|
||||
v_store_aligned(qbuf + vl64 * 2, v_sq2);
|
||||
v_store_aligned(qbuf + vl64 * 3, v_sq3);
|
||||
|
||||
// sbuf lane j (j<vl32) holds samples j and j+vl32 -> channel j%cn (vl32%cn==0).
|
||||
// qbuf index i (i<vl16) holds sample lane i -> channel i%cn (vl16%cn==0).
|
||||
for (int i = 0; i < vl32; ++i)
|
||||
sum[i % cn] += sbuf[i];
|
||||
for (int i = 0; i < vl16; ++i)
|
||||
sqsum[i % cn] += qbuf[i];
|
||||
|
||||
v_cleanup();
|
||||
return x / cn;
|
||||
}
|
||||
|
||||
// Shared 32-bit -> f64 reduction for int/float. Both sum and sqsum accumulate in
|
||||
// f64 to match the scalar reference's double accumulation; values are widened to
|
||||
// f64 before squaring so the square is computed in double (same as (double)v*v).
|
||||
// The register type is deduced from the pointer, so int and float share one body.
|
||||
template <typename T>
|
||||
static inline int sumSqr32ToF64(const T* src0, double* sum, double* sqsum, int len, int cn)
|
||||
{
|
||||
typedef decltype(vx_load(src0)) VT; // v_int32 or v_float32
|
||||
|
||||
if (cn == 3)
|
||||
{
|
||||
const int vl = VTraits<VT>::vlanes();
|
||||
v_float64 vs0 = vx_setzero_f64(), vs1 = vx_setzero_f64(), vs2 = vx_setzero_f64();
|
||||
v_float64 vq0 = vx_setzero_f64(), vq1 = vx_setzero_f64(), vq2 = vx_setzero_f64();
|
||||
auto acc = [](const VT& ch, v_float64& vs, v_float64& vq)
|
||||
{
|
||||
v_float64 f0 = v_cvt_f64(ch), f1 = v_cvt_f64_high(ch);
|
||||
vs = v_add(vs, v_add(f0, f1));
|
||||
vq = v_fma(f0, f0, v_fma(f1, f1, vq));
|
||||
};
|
||||
int e = 0;
|
||||
for (; e <= len - vl; e += vl)
|
||||
{
|
||||
VT a, b, c; v_load_deinterleave(src0 + e * 3, a, b, c);
|
||||
acc(a, vs0, vq0); acc(b, vs1, vq1); acc(c, vs2, vq2);
|
||||
}
|
||||
sum[0] += v_reduce_sum(vs0); sum[1] += v_reduce_sum(vs1); sum[2] += v_reduce_sum(vs2);
|
||||
sqsum[0] += v_reduce_sum(vq0); sqsum[1] += v_reduce_sum(vq1); sqsum[2] += v_reduce_sum(vq2);
|
||||
v_cleanup();
|
||||
return e;
|
||||
}
|
||||
if (cn != 1 && cn != 2 && cn != 4)
|
||||
return 0;
|
||||
len *= cn;
|
||||
|
||||
const int vl32 = VTraits<VT>::vlanes();
|
||||
const int vl64 = VTraits<v_float64>::vlanes();
|
||||
|
||||
// See note in sumSqr16ToF64: guard scalable backends whose lane count may
|
||||
// not divide cn (the i % cn scatter / x / cn rely on it).
|
||||
if (vl32 % cn != 0)
|
||||
return 0;
|
||||
|
||||
v_float64 vs0 = vx_setzero_f64(), vs1 = vx_setzero_f64();
|
||||
v_float64 vq0 = vx_setzero_f64(), vq1 = vx_setzero_f64();
|
||||
|
||||
int x = 0;
|
||||
for (; x <= len - vl32; x += vl32)
|
||||
{
|
||||
VT v_src = vx_load(src0 + x);
|
||||
v_float64 f0 = v_cvt_f64(v_src);
|
||||
v_float64 f1 = v_cvt_f64_high(v_src);
|
||||
vs0 = v_add(vs0, f0);
|
||||
vs1 = v_add(vs1, f1);
|
||||
vq0 = v_fma(f0, f0, vq0);
|
||||
vq1 = v_fma(f1, f1, vq1);
|
||||
}
|
||||
|
||||
double CV_DECL_ALIGNED(CV_SIMD_WIDTH) sbuf[VTraits<v_float64>::max_nlanes * 2];
|
||||
double CV_DECL_ALIGNED(CV_SIMD_WIDTH) qbuf[VTraits<v_float64>::max_nlanes * 2];
|
||||
v_store_aligned(sbuf, vs0); v_store_aligned(sbuf + vl64, vs1);
|
||||
v_store_aligned(qbuf, vq0); v_store_aligned(qbuf + vl64, vq1);
|
||||
|
||||
for (int i = 0; i < vl32; ++i)
|
||||
{
|
||||
sum[i % cn] += sbuf[i];
|
||||
sqsum[i % cn] += qbuf[i];
|
||||
}
|
||||
|
||||
v_cleanup();
|
||||
return x / cn;
|
||||
}
|
||||
|
||||
template <>
|
||||
struct SumSqr_SIMD<ushort, int, double>
|
||||
{
|
||||
int operator () (const ushort* src0, const uchar* mask, int* sum, double* sqsum, int len, int cn) const
|
||||
{ return mask ? 0 : sumSqr16ToF64(src0, sum, sqsum, len, cn); }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct SumSqr_SIMD<short, int, double>
|
||||
{
|
||||
int operator () (const short* src0, const uchar* mask, int* sum, double* sqsum, int len, int cn) const
|
||||
{ return mask ? 0 : sumSqr16ToF64(src0, sum, sqsum, len, cn); }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct SumSqr_SIMD<int, double, double>
|
||||
{
|
||||
int operator () (const int* src0, const uchar* mask, double* sum, double* sqsum, int len, int cn) const
|
||||
{ return mask ? 0 : sumSqr32ToF64(src0, sum, sqsum, len, cn); }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct SumSqr_SIMD<float, double, double>
|
||||
{
|
||||
int operator () (const float* src0, const uchar* mask, double* sum, double* sqsum, int len, int cn) const
|
||||
{ return mask ? 0 : sumSqr32ToF64(src0, sum, sqsum, len, cn); }
|
||||
};
|
||||
|
||||
#endif // CV_SIMD_64F
|
||||
|
||||
#endif
|
||||
|
||||
template<typename T, typename ST, typename SQT>
|
||||
@@ -253,6 +513,21 @@ static int sumsqr_(const T* src0, const uchar* mask, ST* sum, SQT* sqsum, int le
|
||||
sum[0] = s0;
|
||||
sqsum[0] = sq0;
|
||||
}
|
||||
else if( cn == 2 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1];
|
||||
SQT sq0 = sqsum[0], sq1 = sqsum[1];
|
||||
for( i = 0; i < len; i++, src += 2 )
|
||||
if( mask[i] )
|
||||
{
|
||||
T v0 = src[0], v1 = src[1];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
nzm++;
|
||||
}
|
||||
sum[0] = s0; sum[1] = s1;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1;
|
||||
}
|
||||
else if( cn == 3 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1], s2 = sum[2];
|
||||
@@ -269,6 +544,23 @@ static int sumsqr_(const T* src0, const uchar* mask, ST* sum, SQT* sqsum, int le
|
||||
sum[0] = s0; sum[1] = s1; sum[2] = s2;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1; sqsum[2] = sq2;
|
||||
}
|
||||
else if( cn == 4 )
|
||||
{
|
||||
ST s0 = sum[0], s1 = sum[1], s2 = sum[2], s3 = sum[3];
|
||||
SQT sq0 = sqsum[0], sq1 = sqsum[1], sq2 = sqsum[2], sq3 = sqsum[3];
|
||||
for( i = 0; i < len; i++, src += 4 )
|
||||
if( mask[i] )
|
||||
{
|
||||
T v0 = src[0], v1 = src[1], v2 = src[2], v3 = src[3];
|
||||
s0 += v0; sq0 += (SQT)v0*v0;
|
||||
s1 += v1; sq1 += (SQT)v1*v1;
|
||||
s2 += v2; sq2 += (SQT)v2*v2;
|
||||
s3 += v3; sq3 += (SQT)v3*v3;
|
||||
nzm++;
|
||||
}
|
||||
sum[0] = s0; sum[1] = s1; sum[2] = s2; sum[3] = s3;
|
||||
sqsum[0] = sq0; sqsum[1] = sq1; sqsum[2] = sq2; sqsum[3] = sq3;
|
||||
}
|
||||
else
|
||||
{
|
||||
for( i = 0; i < len; i++, src += cn )
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,540 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#include "precomp.hpp"
|
||||
#include "opencl_kernels_core.hpp"
|
||||
#include "stat.hpp"
|
||||
#include "opencv2/core/detail/dispatch_helper.impl.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
#include "minmax.simd.hpp"
|
||||
#include "minmax.simd_declarations.hpp" // defines CV_CPU_DISPATCH_MODES_ALL=AVX2,...,BASELINE based on CMakeLists.txt content
|
||||
|
||||
/****************************************************************************************\
|
||||
* minMaxLoc *
|
||||
\****************************************************************************************/
|
||||
|
||||
namespace cv
|
||||
{
|
||||
|
||||
static MinMaxIdxFunc getMinmaxTab(int depth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getMinmaxTab, (depth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
// The function expects 1-based indexing for ofs
|
||||
// Zero is treated as invalid offset (not found)
|
||||
static void ofs2idx(const Mat& a, size_t ofs, int* idx)
|
||||
{
|
||||
int i, d = a.dims;
|
||||
if( ofs > 0 )
|
||||
{
|
||||
ofs--;
|
||||
for( i = d-1; i >= 0; i-- )
|
||||
{
|
||||
int sz = a.size[i];
|
||||
idx[i] = (int)(ofs % sz);
|
||||
ofs /= sz;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for( i = d-1; i >= 0; i-- )
|
||||
idx[i] = -1;
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
|
||||
#define MINMAX_STRUCT_ALIGNMENT 8 // sizeof double
|
||||
|
||||
template <typename T>
|
||||
void getMinMaxRes(const Mat & db, double * minVal, double * maxVal,
|
||||
int* minLoc, int* maxLoc,
|
||||
int groupnum, int cols, double * maxVal2)
|
||||
{
|
||||
uint index_max = std::numeric_limits<uint>::max();
|
||||
T minval = std::numeric_limits<T>::max();
|
||||
T maxval = std::numeric_limits<T>::min() > 0 ? -std::numeric_limits<T>::max() : std::numeric_limits<T>::min(), maxval2 = maxval;
|
||||
uint minloc = index_max, maxloc = index_max;
|
||||
|
||||
size_t index = 0;
|
||||
const T * minptr = NULL, * maxptr = NULL, * maxptr2 = NULL;
|
||||
const uint * minlocptr = NULL, * maxlocptr = NULL;
|
||||
if (minVal || minLoc)
|
||||
{
|
||||
minptr = db.ptr<T>();
|
||||
index += sizeof(T) * groupnum;
|
||||
index = alignSize(index, MINMAX_STRUCT_ALIGNMENT);
|
||||
}
|
||||
if (maxVal || maxLoc)
|
||||
{
|
||||
maxptr = (const T *)(db.ptr() + index);
|
||||
index += sizeof(T) * groupnum;
|
||||
index = alignSize(index, MINMAX_STRUCT_ALIGNMENT);
|
||||
}
|
||||
if (minLoc)
|
||||
{
|
||||
minlocptr = (const uint *)(db.ptr() + index);
|
||||
index += sizeof(uint) * groupnum;
|
||||
index = alignSize(index, MINMAX_STRUCT_ALIGNMENT);
|
||||
}
|
||||
if (maxLoc)
|
||||
{
|
||||
maxlocptr = (const uint *)(db.ptr() + index);
|
||||
index += sizeof(uint) * groupnum;
|
||||
index = alignSize(index, MINMAX_STRUCT_ALIGNMENT);
|
||||
}
|
||||
if (maxVal2)
|
||||
maxptr2 = (const T *)(db.ptr() + index);
|
||||
|
||||
for (int i = 0; i < groupnum; i++)
|
||||
{
|
||||
if (minptr && minptr[i] <= minval)
|
||||
{
|
||||
if (minptr[i] == minval)
|
||||
{
|
||||
if (minlocptr)
|
||||
minloc = std::min(minlocptr[i], minloc);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (minlocptr)
|
||||
minloc = minlocptr[i];
|
||||
minval = minptr[i];
|
||||
}
|
||||
}
|
||||
if (maxptr && maxptr[i] >= maxval)
|
||||
{
|
||||
if (maxptr[i] == maxval)
|
||||
{
|
||||
if (maxlocptr)
|
||||
maxloc = std::min(maxlocptr[i], maxloc);
|
||||
}
|
||||
else
|
||||
{
|
||||
if (maxlocptr)
|
||||
maxloc = maxlocptr[i];
|
||||
maxval = maxptr[i];
|
||||
}
|
||||
}
|
||||
if (maxptr2 && maxptr2[i] > maxval2)
|
||||
maxval2 = maxptr2[i];
|
||||
}
|
||||
bool zero_mask = (minLoc && minloc == index_max) ||
|
||||
(maxLoc && maxloc == index_max);
|
||||
|
||||
if (minVal)
|
||||
*minVal = zero_mask ? 0 : (double)minval;
|
||||
if (maxVal)
|
||||
*maxVal = zero_mask ? 0 : (double)maxval;
|
||||
if (maxVal2)
|
||||
*maxVal2 = zero_mask ? 0 : (double)maxval2;
|
||||
|
||||
if (minLoc)
|
||||
{
|
||||
minLoc[0] = zero_mask ? -1 : minloc / cols;
|
||||
minLoc[1] = zero_mask ? -1 : minloc % cols;
|
||||
}
|
||||
if (maxLoc)
|
||||
{
|
||||
maxLoc[0] = zero_mask ? -1 : maxloc / cols;
|
||||
maxLoc[1] = zero_mask ? -1 : maxloc % cols;
|
||||
}
|
||||
}
|
||||
|
||||
typedef void (*getMinMaxResFunc)(const Mat & db, double * minVal, double * maxVal,
|
||||
int * minLoc, int *maxLoc, int gropunum, int cols, double * maxVal2);
|
||||
|
||||
bool ocl_minMaxIdx( InputArray _src, double* minVal, double* maxVal, int* minLoc, int* maxLoc, InputArray _mask,
|
||||
int ddepth, bool absValues, InputArray _src2, double * maxVal2)
|
||||
{
|
||||
const ocl::Device & dev = ocl::Device::getDefault();
|
||||
|
||||
#ifdef __ANDROID__
|
||||
if (dev.isNVidia())
|
||||
return false;
|
||||
#endif
|
||||
|
||||
if (dev.deviceVersionMajor() == 1 && dev.deviceVersionMinor() < 2)
|
||||
{
|
||||
// 'static' storage class specifier used by "minmaxloc" is available from OpenCL 1.2+ only
|
||||
return false;
|
||||
}
|
||||
|
||||
bool doubleSupport = dev.doubleFPConfig() > 0, haveMask = !_mask.empty(),
|
||||
haveSrc2 = _src2.kind() != _InputArray::NONE;
|
||||
int type = _src.type(), depth = CV_MAT_DEPTH(type), cn = CV_MAT_CN(type),
|
||||
kercn = haveMask ? cn : std::min(4, ocl::predictOptimalVectorWidth(_src, _src2));
|
||||
|
||||
if (depth >= CV_16F)
|
||||
return false;
|
||||
|
||||
// disabled following modes since it occasionally fails on AMD devices (e.g. A10-6800K, sep. 2014)
|
||||
if ((haveMask || type == CV_32FC1) && dev.isAMD())
|
||||
return false;
|
||||
|
||||
CV_Assert( (cn == 1 && (!haveMask || _mask.type() == CV_8U)) ||
|
||||
(cn >= 1 && !minLoc && !maxLoc) );
|
||||
|
||||
if (ddepth < 0)
|
||||
ddepth = depth;
|
||||
|
||||
CV_Assert(!haveSrc2 || _src2.type() == type);
|
||||
|
||||
if (depth == CV_32S)
|
||||
return false;
|
||||
|
||||
if ((depth == CV_64F || ddepth == CV_64F) && !doubleSupport)
|
||||
return false;
|
||||
|
||||
int groupnum = dev.maxComputeUnits();
|
||||
size_t wgs = dev.maxWorkGroupSize();
|
||||
|
||||
int wgs2_aligned = 1;
|
||||
while (wgs2_aligned < (int)wgs)
|
||||
wgs2_aligned <<= 1;
|
||||
wgs2_aligned >>= 1;
|
||||
|
||||
bool needMinVal = minVal || minLoc, needMinLoc = minLoc != NULL,
|
||||
needMaxVal = maxVal || maxLoc, needMaxLoc = maxLoc != NULL;
|
||||
|
||||
// in case of mask we must know whether mask is filled with zeros or not
|
||||
// so let's calculate min or max location, if it's undefined, so mask is zeros
|
||||
if (!(needMaxLoc || needMinLoc) && haveMask)
|
||||
{
|
||||
if (needMinVal)
|
||||
needMinLoc = true;
|
||||
else
|
||||
needMaxLoc = true;
|
||||
}
|
||||
|
||||
char cvt[2][50];
|
||||
String opts = format("-D DEPTH_%d -D srcT1=%s%s -D WGS=%d -D srcT=%s"
|
||||
" -D WGS2_ALIGNED=%d%s%s%s -D kercn=%d%s%s%s%s"
|
||||
" -D dstT1=%s -D dstT=%s -D convertToDT=%s%s%s%s%s -D wdepth=%d -D convertFromU=%s"
|
||||
" -D MINMAX_STRUCT_ALIGNMENT=%d",
|
||||
depth, ocl::typeToStr(depth), haveMask ? " -D HAVE_MASK" : "", (int)wgs,
|
||||
ocl::typeToStr(CV_MAKE_TYPE(depth, kercn)), wgs2_aligned,
|
||||
doubleSupport ? " -D DOUBLE_SUPPORT" : "",
|
||||
_src.isContinuous() ? " -D HAVE_SRC_CONT" : "",
|
||||
_mask.isContinuous() ? " -D HAVE_MASK_CONT" : "", kercn,
|
||||
needMinVal ? " -D NEED_MINVAL" : "", needMaxVal ? " -D NEED_MAXVAL" : "",
|
||||
needMinLoc ? " -D NEED_MINLOC" : "", needMaxLoc ? " -D NEED_MAXLOC" : "",
|
||||
ocl::typeToStr(ddepth), ocl::typeToStr(CV_MAKE_TYPE(ddepth, kercn)),
|
||||
ocl::convertTypeStr(depth, ddepth, kercn, cvt[0], sizeof(cvt[0])),
|
||||
absValues ? " -D OP_ABS" : "",
|
||||
haveSrc2 ? " -D HAVE_SRC2" : "", maxVal2 ? " -D OP_CALC2" : "",
|
||||
haveSrc2 && _src2.isContinuous() ? " -D HAVE_SRC2_CONT" : "", ddepth,
|
||||
depth <= CV_32S && ddepth == CV_32S ? ocl::convertTypeStr(CV_8U, ddepth, kercn, cvt[1], sizeof(cvt[1])) : "noconvert",
|
||||
MINMAX_STRUCT_ALIGNMENT);
|
||||
|
||||
ocl::Kernel k("minmaxloc", ocl::core::minmaxloc_oclsrc, opts);
|
||||
if (k.empty())
|
||||
return false;
|
||||
|
||||
int esz = CV_ELEM_SIZE(ddepth), esz32s = CV_ELEM_SIZE1(CV_32S),
|
||||
dbsize = groupnum * ((needMinVal ? esz : 0) + (needMaxVal ? esz : 0) +
|
||||
(needMinLoc ? esz32s : 0) + (needMaxLoc ? esz32s : 0) +
|
||||
(maxVal2 ? esz : 0))
|
||||
+ 5 * MINMAX_STRUCT_ALIGNMENT;
|
||||
UMat src = _src.getUMat(), src2 = _src2.getUMat(), db(1, dbsize, CV_8UC1), mask = _mask.getUMat();
|
||||
|
||||
if (cn > 1 && !haveMask)
|
||||
{
|
||||
src = src.reshape(1);
|
||||
src2 = src2.reshape(1);
|
||||
}
|
||||
|
||||
if (haveSrc2)
|
||||
{
|
||||
if (!haveMask)
|
||||
k.args(ocl::KernelArg::ReadOnlyNoSize(src), src.cols, (int)src.total(),
|
||||
groupnum, ocl::KernelArg::PtrWriteOnly(db), ocl::KernelArg::ReadOnlyNoSize(src2));
|
||||
else
|
||||
k.args(ocl::KernelArg::ReadOnlyNoSize(src), src.cols, (int)src.total(),
|
||||
groupnum, ocl::KernelArg::PtrWriteOnly(db), ocl::KernelArg::ReadOnlyNoSize(mask),
|
||||
ocl::KernelArg::ReadOnlyNoSize(src2));
|
||||
}
|
||||
else
|
||||
{
|
||||
if (!haveMask)
|
||||
k.args(ocl::KernelArg::ReadOnlyNoSize(src), src.cols, (int)src.total(),
|
||||
groupnum, ocl::KernelArg::PtrWriteOnly(db));
|
||||
else
|
||||
k.args(ocl::KernelArg::ReadOnlyNoSize(src), src.cols, (int)src.total(),
|
||||
groupnum, ocl::KernelArg::PtrWriteOnly(db), ocl::KernelArg::ReadOnlyNoSize(mask));
|
||||
}
|
||||
|
||||
size_t globalsize = groupnum * wgs;
|
||||
if (!k.run(1, &globalsize, &wgs, true))
|
||||
return false;
|
||||
|
||||
static const getMinMaxResFunc functab[7] =
|
||||
{
|
||||
getMinMaxRes<uchar>,
|
||||
getMinMaxRes<char>,
|
||||
getMinMaxRes<ushort>,
|
||||
getMinMaxRes<short>,
|
||||
getMinMaxRes<int>,
|
||||
getMinMaxRes<float>,
|
||||
getMinMaxRes<double>
|
||||
};
|
||||
|
||||
CV_Assert(ddepth <= CV_64F);
|
||||
getMinMaxResFunc func = functab[ddepth];
|
||||
|
||||
int locTemp[2];
|
||||
func(db.getMat(ACCESS_READ), minVal, maxVal,
|
||||
needMinLoc ? minLoc ? minLoc : locTemp : minLoc,
|
||||
needMaxLoc ? maxLoc ? maxLoc : locTemp : maxLoc,
|
||||
groupnum, src.cols, maxVal2);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
}
|
||||
|
||||
void cv::minMaxIdx(InputArray _src, double* minVal,
|
||||
double* maxVal, int* minIdx, int* maxIdx,
|
||||
InputArray _mask)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
int type = _src.type(), depth = CV_MAT_DEPTH(type), cn = CV_MAT_CN(type);
|
||||
CV_Assert( (cn == 1 && (_mask.empty() || _mask.type() == CV_8U)) ||
|
||||
(cn > 1 && _mask.empty() && !minIdx && !maxIdx) );
|
||||
|
||||
CV_OCL_RUN(OCL_PERFORMANCE_CHECK(_src.isUMat()) && _src.dims() <= 2 && (_mask.empty() || _src.size() == _mask.size()),
|
||||
ocl_minMaxIdx(_src, minVal, maxVal, minIdx, maxIdx, _mask))
|
||||
|
||||
Mat src = _src.getMat(), mask = _mask.getMat();
|
||||
|
||||
if (src.dims <= 2)
|
||||
{
|
||||
if ((size_t)src.step == (size_t)mask.step || mask.empty())
|
||||
{
|
||||
CALL_HAL(minMaxIdx, cv_hal_minMaxIdx, src.data, src.step, src.cols*cn, src.rows,
|
||||
src.depth(), minVal, maxVal, minIdx, maxIdx, mask.data);
|
||||
}
|
||||
|
||||
CALL_HAL(minMaxIdxMaskStep, cv_hal_minMaxIdxMaskStep, src.data, src.step, src.cols*cn, src.rows,
|
||||
src.depth(), minVal, maxVal, minIdx, maxIdx, mask.data, mask.step);
|
||||
}
|
||||
else if (src.isContinuous() && (mask.isContinuous() || mask.empty()))
|
||||
{
|
||||
int res = cv_hal_minMaxIdx(src.data, 0, (int)src.total()*cn, 1, src.depth(),
|
||||
minVal, maxVal, minIdx, maxIdx, mask.data);
|
||||
|
||||
if (res == CV_HAL_ERROR_OK)
|
||||
{
|
||||
// minIdx[0] and minIdx[0] are always 0 for "flatten" version
|
||||
if (minIdx)
|
||||
ofs2idx(src, minIdx[1]+1, minIdx);
|
||||
if (maxIdx)
|
||||
ofs2idx(src, maxIdx[1]+1, maxIdx);
|
||||
return;
|
||||
}
|
||||
else if (res != CV_HAL_ERROR_NOT_IMPLEMENTED)
|
||||
{
|
||||
CV_Error_(cv::Error::StsInternal,
|
||||
("HAL implementation minMaxIdx ==> " CVAUX_STR(cv_hal_minMaxIdx) " returned %d (0x%08x)", res, res));
|
||||
}
|
||||
}
|
||||
|
||||
MinMaxIdxFunc func = getMinmaxTab(depth);
|
||||
CV_Assert( func != 0 );
|
||||
|
||||
const Mat* arrays[] = {&src, &mask, 0};
|
||||
uchar* ptrs[2] = {};
|
||||
NAryMatIterator it(arrays, ptrs);
|
||||
|
||||
size_t minidx = 0, maxidx = 0;
|
||||
int iminval = INT_MAX, imaxval = INT_MIN;
|
||||
float fminval = std::numeric_limits<float>::infinity(), fmaxval = -fminval;
|
||||
double dminval = std::numeric_limits<double>::infinity(), dmaxval = -dminval;
|
||||
size_t startidx = 1;
|
||||
void *minval = &iminval, *maxval = &imaxval;
|
||||
int planeSize = (int)it.size*cn;
|
||||
|
||||
if( depth == CV_32F )
|
||||
minval = &fminval, maxval = &fmaxval;
|
||||
else if( depth == CV_64F )
|
||||
minval = &dminval, maxval = &dmaxval;
|
||||
|
||||
for( size_t i = 0; i < it.nplanes; i++, ++it, startidx += planeSize )
|
||||
func( ptrs[0], ptrs[1], minval, maxval, &minidx, &maxidx, planeSize, startidx );
|
||||
|
||||
if (!src.empty() && mask.empty())
|
||||
{
|
||||
if( minidx == 0 )
|
||||
minidx = 1;
|
||||
if( maxidx == 0 )
|
||||
maxidx = 1;
|
||||
}
|
||||
|
||||
if( minidx == 0 )
|
||||
dminval = dmaxval = 0;
|
||||
else if( depth == CV_32F )
|
||||
dminval = fminval, dmaxval = fmaxval;
|
||||
else if( depth <= CV_32S )
|
||||
dminval = iminval, dmaxval = imaxval;
|
||||
|
||||
if( minVal )
|
||||
*minVal = dminval;
|
||||
if( maxVal )
|
||||
*maxVal = dmaxval;
|
||||
|
||||
if( minIdx )
|
||||
ofs2idx(src, minidx, minIdx);
|
||||
if( maxIdx )
|
||||
ofs2idx(src, maxidx, maxIdx);
|
||||
}
|
||||
|
||||
void cv::minMaxLoc( InputArray _img, double* minVal, double* maxVal,
|
||||
Point* minLoc, Point* maxLoc, InputArray mask )
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
int dims = _img.dims();
|
||||
CV_CheckLE(dims, 2, "");
|
||||
|
||||
minMaxIdx(_img, minVal, maxVal, (int*)minLoc, (int*)maxLoc, mask);
|
||||
if( minLoc )
|
||||
{
|
||||
if (dims == 2)
|
||||
std::swap(minLoc->x, minLoc->y);
|
||||
else
|
||||
minLoc->y = 0;
|
||||
}
|
||||
if( maxLoc )
|
||||
{
|
||||
if (dims == 2)
|
||||
std::swap(maxLoc->x, maxLoc->y);
|
||||
else
|
||||
maxLoc->y = 0;
|
||||
}
|
||||
}
|
||||
|
||||
enum class ReduceMode
|
||||
{
|
||||
FIRST_MIN = 0, //!< get index of first min occurrence
|
||||
LAST_MIN = 1, //!< get index of last min occurrence
|
||||
FIRST_MAX = 2, //!< get index of first max occurrence
|
||||
LAST_MAX = 3, //!< get index of last max occurrence
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct reduceMinMaxImpl
|
||||
{
|
||||
void operator()(const cv::Mat& src, cv::Mat& dst, ReduceMode mode, const int axis) const
|
||||
{
|
||||
switch(mode)
|
||||
{
|
||||
case ReduceMode::FIRST_MIN:
|
||||
reduceMinMaxApply<std::less>(src, dst, axis);
|
||||
break;
|
||||
case ReduceMode::LAST_MIN:
|
||||
reduceMinMaxApply<std::less_equal>(src, dst, axis);
|
||||
break;
|
||||
case ReduceMode::FIRST_MAX:
|
||||
reduceMinMaxApply<std::greater>(src, dst, axis);
|
||||
break;
|
||||
case ReduceMode::LAST_MAX:
|
||||
reduceMinMaxApply<std::greater_equal>(src, dst, axis);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
template <template<class> class Cmp>
|
||||
static void reduceMinMaxApply(const cv::Mat& src, cv::Mat& dst, const int axis)
|
||||
{
|
||||
Cmp<T> cmp;
|
||||
|
||||
const auto *src_ptr = src.ptr<T>();
|
||||
auto *dst_ptr = dst.ptr<int32_t>();
|
||||
|
||||
const size_t outer_size = src.total(0, axis);
|
||||
const auto mid_size = static_cast<size_t>(src.size[axis]);
|
||||
|
||||
const size_t outer_step = src.total(axis);
|
||||
const size_t dst_step = dst.total(axis);
|
||||
|
||||
const size_t mid_step = src.total(axis + 1);
|
||||
|
||||
for (size_t outer = 0; outer < outer_size; ++outer)
|
||||
{
|
||||
const size_t outer_offset = outer * outer_step;
|
||||
const size_t dst_offset = outer * dst_step;
|
||||
for (size_t mid = 0; mid != mid_size; ++mid)
|
||||
{
|
||||
const size_t src_offset = outer_offset + mid * mid_step;
|
||||
for (size_t inner = 0; inner < mid_step; inner++)
|
||||
{
|
||||
int32_t& index = dst_ptr[dst_offset + inner];
|
||||
|
||||
const size_t prev = outer_offset + index * mid_step + inner;
|
||||
const size_t curr = src_offset + inner;
|
||||
|
||||
if (cmp(src_ptr[curr], src_ptr[prev]))
|
||||
{
|
||||
index = static_cast<int32_t>(mid);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
static void reduceMinMax(cv::InputArray src, cv::OutputArray dst, ReduceMode mode, int axis)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
|
||||
cv::Mat srcMat = src.getMat();
|
||||
axis = (axis + srcMat.dims) % srcMat.dims;
|
||||
CV_Assert(srcMat.channels() == 1 && axis >= 0 && axis < srcMat.dims);
|
||||
|
||||
cv::AutoBuffer<int> sizes(srcMat.dims);
|
||||
std::copy(srcMat.size.p, srcMat.size.p + srcMat.dims, sizes.begin());
|
||||
sizes[axis] = 1;
|
||||
|
||||
dst.create(srcMat.dims, sizes.data(), CV_32SC1); // indices
|
||||
cv::Mat dstMat = dst.getMat();
|
||||
dstMat.setTo(cv::Scalar::all(0));
|
||||
|
||||
if (!srcMat.isContinuous())
|
||||
{
|
||||
srcMat = srcMat.clone();
|
||||
}
|
||||
|
||||
bool needs_copy = !dstMat.isContinuous();
|
||||
if (needs_copy)
|
||||
{
|
||||
dstMat = dstMat.clone();
|
||||
}
|
||||
|
||||
cv::detail::depthDispatch<reduceMinMaxImpl>(srcMat.depth(), srcMat, dstMat, mode, axis);
|
||||
|
||||
if (needs_copy)
|
||||
{
|
||||
dstMat.copyTo(dst);
|
||||
}
|
||||
}
|
||||
|
||||
void cv::reduceArgMin(InputArray src, OutputArray dst, int axis, bool lastIndex)
|
||||
{
|
||||
reduceMinMax(src, dst, lastIndex ? ReduceMode::LAST_MIN : ReduceMode::FIRST_MIN, axis);
|
||||
}
|
||||
|
||||
void cv::reduceArgMax(InputArray src, OutputArray dst, int axis, bool lastIndex)
|
||||
{
|
||||
reduceMinMax(src, dst, lastIndex ? ReduceMode::LAST_MAX : ReduceMode::FIRST_MAX, axis);
|
||||
}
|
||||
@@ -0,0 +1,471 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
#include "precomp.hpp"
|
||||
#include "stat.hpp"
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
namespace cv {
|
||||
|
||||
typedef void (*MinMaxIdxFunc)(const void*, const uchar*, void*, void*, size_t*, size_t*, int, size_t);
|
||||
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
MinMaxIdxFunc getMinmaxTab(int depth);
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
template<typename T, typename WT> static void
|
||||
minMaxIdx_( const T* src, const uchar* mask, WT* _minVal, WT* _maxVal,
|
||||
size_t* _minIdx, size_t* _maxIdx, int len, size_t startIdx )
|
||||
{
|
||||
WT minVal = *_minVal, maxVal = *_maxVal;
|
||||
size_t minIdx = *_minIdx, maxIdx = *_maxIdx;
|
||||
|
||||
if( !mask )
|
||||
{
|
||||
for( int i = 0; i < len; i++ )
|
||||
{
|
||||
T val = src[i];
|
||||
if( val < minVal )
|
||||
{
|
||||
minVal = val;
|
||||
minIdx = startIdx + i;
|
||||
}
|
||||
if( val > maxVal )
|
||||
{
|
||||
maxVal = val;
|
||||
maxIdx = startIdx + i;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for( int i = 0; i < len; i++ )
|
||||
{
|
||||
T val = src[i];
|
||||
if( mask[i] && val < minVal )
|
||||
{
|
||||
minVal = val;
|
||||
minIdx = startIdx + i;
|
||||
}
|
||||
if( mask[i] && val > maxVal )
|
||||
{
|
||||
maxVal = val;
|
||||
maxIdx = startIdx + i;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
*_minIdx = minIdx;
|
||||
*_maxIdx = maxIdx;
|
||||
*_minVal = minVal;
|
||||
*_maxVal = maxVal;
|
||||
}
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
template<typename T, typename WT> CV_ALWAYS_INLINE void
|
||||
minMaxIdx_init( const T* src, const uchar* mask, WT* minval, WT* maxval,
|
||||
size_t* minidx, size_t* maxidx, WT &minVal, WT &maxVal,
|
||||
size_t &minIdx, size_t &maxIdx, const WT minInit, const WT maxInit,
|
||||
const int nlanes, int len, size_t startidx, int &j, int &len0 )
|
||||
{
|
||||
len0 = len & -nlanes;
|
||||
j = 0;
|
||||
|
||||
minVal = *minval, maxVal = *maxval;
|
||||
minIdx = *minidx, maxIdx = *maxidx;
|
||||
|
||||
// To handle start values out of range
|
||||
if ( minVal < minInit || maxVal < minInit || minVal > maxInit || maxVal > maxInit )
|
||||
{
|
||||
uchar done = 0x00;
|
||||
|
||||
for ( ; (j < len) && (done != 0x03); j++ )
|
||||
{
|
||||
if ( !mask || mask[j] ) {
|
||||
T val = src[j];
|
||||
if ( val < minVal )
|
||||
{
|
||||
minVal = val;
|
||||
minIdx = startidx + j;
|
||||
done |= 0x01;
|
||||
}
|
||||
if ( val > maxVal )
|
||||
{
|
||||
maxVal = val;
|
||||
maxIdx = startidx + j;
|
||||
done |= 0x02;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
len0 = j + ((len - j) & -nlanes);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename T, typename WT> CV_ALWAYS_INLINE void
|
||||
minMaxIdx_finish( const T* src, const uchar* mask, WT* minval, WT* maxval,
|
||||
size_t* minidx, size_t* maxidx, WT minVal, WT maxVal,
|
||||
size_t minIdx, size_t maxIdx, int len, size_t startidx,
|
||||
int j )
|
||||
{
|
||||
for ( ; j < len ; j++ )
|
||||
{
|
||||
if ( !mask || mask[j] )
|
||||
{
|
||||
T val = src[j];
|
||||
if ( val < minVal )
|
||||
{
|
||||
minVal = val;
|
||||
minIdx = startidx + j;
|
||||
}
|
||||
if ( val > maxVal )
|
||||
{
|
||||
maxVal = val;
|
||||
maxIdx = startidx + j;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
*minidx = minIdx;
|
||||
*maxidx = maxIdx;
|
||||
*minval = minVal;
|
||||
*maxval = maxVal;
|
||||
}
|
||||
|
||||
//============================================================================
|
||||
// Templated SIMD core (universal intrinsics) shared by every depth. The
|
||||
// per-depth wrappers below only select the value/index vector + result types.
|
||||
//============================================================================
|
||||
|
||||
// Fast single-instruction broadcast into any universal vector (value or index).
|
||||
template<typename V> static inline V mm_set(typename VTraits<V>::lane_type v);
|
||||
template<> inline v_uint8 mm_set<v_uint8 >(uchar v) { return vx_setall_u8(v); }
|
||||
template<> inline v_int8 mm_set<v_int8 >(schar v) { return vx_setall_s8(v); }
|
||||
template<> inline v_uint16 mm_set<v_uint16 >(ushort v) { return vx_setall_u16(v); }
|
||||
template<> inline v_int16 mm_set<v_int16 >(short v) { return vx_setall_s16(v); }
|
||||
template<> inline v_uint32 mm_set<v_uint32 >(uint v) { return vx_setall_u32(v); }
|
||||
template<> inline v_int32 mm_set<v_int32 >(int v) { return vx_setall_s32(v); }
|
||||
template<> inline v_float32 mm_set<v_float32>(float v) { return vx_setall_f32(v); }
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
template<> inline v_uint64 mm_set<v_uint64 >(uint64_t v) { return vx_setall_u64(v); }
|
||||
template<> inline v_float64 mm_set<v_float64>(double v) { return vx_setall_f64(v); }
|
||||
#endif
|
||||
|
||||
// Reinterpret a value-domain comparison mask to the matching unsigned index vector.
|
||||
static inline v_uint8 mm_idxmask(const v_uint8& m) { return m; }
|
||||
static inline v_uint8 mm_idxmask(const v_int8& m) { return v_reinterpret_as_u8(m); }
|
||||
static inline v_uint16 mm_idxmask(const v_uint16& m) { return m; }
|
||||
static inline v_uint16 mm_idxmask(const v_int16& m) { return v_reinterpret_as_u16(m); }
|
||||
static inline v_uint32 mm_idxmask(const v_int32& m) { return v_reinterpret_as_u32(m); }
|
||||
static inline v_uint32 mm_idxmask(const v_float32& m) { return v_reinterpret_as_u32(m); }
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
static inline v_uint64 mm_idxmask(const v_float64& m) { return v_reinterpret_as_u64(m); }
|
||||
#endif
|
||||
|
||||
// Blend for index vectors (universal API has no v_select for u64).
|
||||
template<typename IT> static inline IT mm_sel(const IT& mask, const IT& a, const IT& b)
|
||||
{ return v_xor(b, v_and(v_xor(a, b), mask)); }
|
||||
|
||||
// Value reduce-min / reduce-max (f64 has no universal v_reduce_*).
|
||||
static inline int mm_vmin(const v_uint8& v){ return v_reduce_min(v); }
|
||||
static inline int mm_vmin(const v_int8& v){ return v_reduce_min(v); }
|
||||
static inline int mm_vmin(const v_uint16& v){ return v_reduce_min(v); }
|
||||
static inline int mm_vmin(const v_int16& v){ return v_reduce_min(v); }
|
||||
static inline int mm_vmin(const v_int32& v){ return v_reduce_min(v); }
|
||||
static inline float mm_vmin(const v_float32&v){ return v_reduce_min(v); }
|
||||
static inline int mm_vmax(const v_uint8& v){ return v_reduce_max(v); }
|
||||
static inline int mm_vmax(const v_int8& v){ return v_reduce_max(v); }
|
||||
static inline int mm_vmax(const v_uint16& v){ return v_reduce_max(v); }
|
||||
static inline int mm_vmax(const v_int16& v){ return v_reduce_max(v); }
|
||||
static inline int mm_vmax(const v_int32& v){ return v_reduce_max(v); }
|
||||
static inline float mm_vmax(const v_float32&v){ return v_reduce_max(v); }
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
static inline double mm_vmin(const v_float64& v)
|
||||
{ double b[VTraits<v_float64>::max_nlanes]; v_store(b, v); double r=b[0]; const int n=VTraits<v_float64>::vlanes(); for(int i=1;i<n;i++) if(b[i]<r) r=b[i]; return r; }
|
||||
static inline double mm_vmax(const v_float64& v)
|
||||
{ double b[VTraits<v_float64>::max_nlanes]; v_store(b, v); double r=b[0]; const int n=VTraits<v_float64>::vlanes(); for(int i=1;i<n;i++) if(b[i]>r) r=b[i]; return r; }
|
||||
#endif
|
||||
|
||||
// Index reduce-min (u64 has no universal v_reduce_min).
|
||||
static inline unsigned mm_imin(const v_uint8& v){ return v_reduce_min(v); }
|
||||
static inline unsigned mm_imin(const v_uint16& v){ return v_reduce_min(v); }
|
||||
static inline unsigned mm_imin(const v_uint32& v){ return v_reduce_min(v); }
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
static inline uint64_t mm_imin(const v_uint64& v)
|
||||
{ uint64_t b[VTraits<v_uint64>::max_nlanes]; v_store(b, v); uint64_t r=b[0]; const int n=VTraits<v_uint64>::vlanes(); for(int i=1;i<n;i++) if(b[i]<r) r=b[i]; return r; }
|
||||
#endif
|
||||
|
||||
// Active-lane mask in the *value* vector domain, from a uchar* mask.
|
||||
template<typename VT> static inline VT mm_active(const uchar* mask, int k, int nlanes);
|
||||
template<> inline v_uint8 mm_active<v_uint8 >(const uchar* mask, int k, int)
|
||||
{ return v_ne(vx_load(mask + k), vx_setzero_u8()); }
|
||||
template<> inline v_int8 mm_active<v_int8 >(const uchar* mask, int k, int)
|
||||
{ return v_reinterpret_as_s8(v_ne(vx_load(mask + k), vx_setzero_u8())); }
|
||||
template<> inline v_uint16 mm_active<v_uint16>(const uchar* mask, int k, int)
|
||||
{ return v_ne(vx_load_expand(mask + k), vx_setzero_u16()); }
|
||||
template<> inline v_int16 mm_active<v_int16 >(const uchar* mask, int k, int)
|
||||
{ return v_reinterpret_as_s16(v_ne(vx_load_expand(mask + k), vx_setzero_u16())); }
|
||||
template<> inline v_int32 mm_active<v_int32 >(const uchar* mask, int k, int nlanes)
|
||||
{ uint32_t b[VTraits<v_uint32>::max_nlanes]; for(int t=0;t<nlanes;t++) b[t]=mask[k+t]?~0u:0u; return v_reinterpret_as_s32(vx_load(b)); }
|
||||
template<> inline v_float32 mm_active<v_float32>(const uchar* mask, int k, int nlanes)
|
||||
{ uint32_t b[VTraits<v_uint32>::max_nlanes]; for(int t=0;t<nlanes;t++) b[t]=mask[k+t]?~0u:0u; return v_reinterpret_as_f32(vx_load(b)); }
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
template<> inline v_float64 mm_active<v_float64>(const uchar* mask, int k, int nlanes)
|
||||
{ uint64_t b[VTraits<v_uint64>::max_nlanes]; for(int t=0;t<nlanes;t++) b[t]=mask[k+t]?~0ull:0ull; return v_reinterpret_as_f64(vx_load(b)); }
|
||||
#endif
|
||||
|
||||
// Fold one accumulator stream's (value,index) extremum into the running result.
|
||||
// Scalar tie-break (smaller index wins) lets several independent streams combine
|
||||
// correctly without any vector index comparison (works for u64 indices too).
|
||||
template<typename VT, typename IT, typename WT>
|
||||
static inline void mm_fold_min(const VT& valMin, const IT& idxMin, const IT& none,
|
||||
size_t delta, WT& minVal, size_t& minIdx)
|
||||
{
|
||||
if ( v_check_any(v_ne(idxMin, none)) )
|
||||
{
|
||||
WT cv = (WT)mm_vmin(valMin);
|
||||
IT sel = mm_sel(mm_idxmask(v_eq(mm_set<VT>((typename VTraits<VT>::lane_type)cv), valMin)), idxMin, none);
|
||||
size_t ci = (size_t)mm_imin(sel) + delta;
|
||||
if ( cv < minVal || (cv == minVal && ci < minIdx) ) { minVal = cv; minIdx = ci; }
|
||||
}
|
||||
}
|
||||
template<typename VT, typename IT, typename WT>
|
||||
static inline void mm_fold_max(const VT& valMax, const IT& idxMax, const IT& none,
|
||||
size_t delta, WT& maxVal, size_t& maxIdx)
|
||||
{
|
||||
if ( v_check_any(v_ne(idxMax, none)) )
|
||||
{
|
||||
WT cv = (WT)mm_vmax(valMax);
|
||||
IT sel = mm_sel(mm_idxmask(v_eq(mm_set<VT>((typename VTraits<VT>::lane_type)cv), valMax)), idxMax, none);
|
||||
size_t ci = (size_t)mm_imin(sel) + delta;
|
||||
if ( cv > maxVal || (cv == maxVal && ci < maxIdx) ) { maxVal = cv; maxIdx = ci; }
|
||||
}
|
||||
}
|
||||
|
||||
// IST = unsigned index lane type (uchar / ushort / uint / uint64_t).
|
||||
template<typename T, typename VT, typename IT, typename IST, typename WT>
|
||||
static void minMaxIdx_simd_(const T* src, const uchar* mask, WT* minval, WT* maxval,
|
||||
size_t* minidx, size_t* maxidx, int len, size_t startidx,
|
||||
WT minInit, WT maxInit)
|
||||
{
|
||||
const int nlanes = VTraits<VT>::vlanes();
|
||||
if ( len >= nlanes )
|
||||
{
|
||||
int j, len0;
|
||||
WT minVal, maxVal;
|
||||
size_t minIdx, maxIdx;
|
||||
|
||||
minMaxIdx_init( src, mask, minval, maxval, minidx, maxidx, minVal, maxVal, minIdx, maxIdx,
|
||||
minInit, maxInit, nlanes, len, startidx, j, len0 );
|
||||
|
||||
if ( j <= len0 - nlanes )
|
||||
{
|
||||
IST idxbuf[VTraits<IT>::max_nlanes];
|
||||
for ( int t = 0; t < nlanes; t++ ) idxbuf[t] = (IST)t;
|
||||
const IT idxStart = vx_load(idxbuf);
|
||||
const IT inc = mm_set<IT>((IST)nlanes);
|
||||
const IT none = mm_set<IT>((IST)~(IST)0);
|
||||
// Reduce before the per-lane index could reach the 'none' sentinel.
|
||||
// For >= 32-bit indices (len <= INT_MAX) one block covers everything.
|
||||
const int64_t idxcap = (sizeof(IST) <= 2) ? (int64_t)std::numeric_limits<IST>::max() : (int64_t)INT_MAX;
|
||||
const int blockStep = (int)((idxcap / nlanes) * nlanes);
|
||||
|
||||
const IT inc2 = mm_set<IT>((IST)(nlanes * 2));
|
||||
|
||||
do
|
||||
{
|
||||
// Two independent accumulator streams break the serial
|
||||
// min/max + index-select dependency chain so the CPU can
|
||||
// overlap iterations (latency-bound loop). Stream 0 covers the
|
||||
// even vector slots, stream 1 the odd ones; results are merged
|
||||
// by mm_fold_*. (2 streams is the sweet spot: 4 spills the 16
|
||||
// YMM regs on AVX2 and gives no gain on AVX-512.) The masked
|
||||
// path keeps a single stream (stream 0).
|
||||
VT vMin0 = mm_set<VT>((T)minVal), vMin1 = vMin0;
|
||||
VT vMax0 = mm_set<VT>((T)maxVal), vMax1 = vMax0;
|
||||
IT idx0 = idxStart, idx1 = v_add(idxStart, inc);
|
||||
IT iMin0 = none, iMin1 = none, iMax0 = none, iMax1 = none;
|
||||
|
||||
int k = j;
|
||||
size_t delta = startidx + j;
|
||||
// 64-bit math: blockStep can be ~INT_MAX (32/64-bit indices), so
|
||||
// j + blockStep would overflow int once j advances past the start.
|
||||
const int limit = (int)std::min<int64_t>((int64_t)len0, (int64_t)j + (int64_t)blockStep);
|
||||
|
||||
if ( !mask )
|
||||
{
|
||||
// Dual stream only for >= 16-bit indices: 8-bit forces tiny
|
||||
// reduce-blocks (u8 index < 256), where extra per-block folds
|
||||
// outweigh the dependency-break benefit.
|
||||
for ( ; (sizeof(IST) > 1) && k <= limit - 2 * nlanes; k += 2 * nlanes )
|
||||
{
|
||||
VT d0 = vx_load(src + k), d1 = vx_load(src + k + nlanes);
|
||||
iMin0 = mm_sel(mm_idxmask(v_lt(d0, vMin0)), idx0, iMin0);
|
||||
iMin1 = mm_sel(mm_idxmask(v_lt(d1, vMin1)), idx1, iMin1);
|
||||
iMax0 = mm_sel(mm_idxmask(v_gt(d0, vMax0)), idx0, iMax0);
|
||||
iMax1 = mm_sel(mm_idxmask(v_gt(d1, vMax1)), idx1, iMax1);
|
||||
vMin0 = v_min(d0, vMin0); vMin1 = v_min(d1, vMin1);
|
||||
vMax0 = v_max(d0, vMax0); vMax1 = v_max(d1, vMax1);
|
||||
idx0 = v_add(idx0, inc2); idx1 = v_add(idx1, inc2);
|
||||
}
|
||||
for ( ; k < limit; k += nlanes ) // odd trailing vector
|
||||
{
|
||||
VT d0 = vx_load(src + k);
|
||||
iMin0 = mm_sel(mm_idxmask(v_lt(d0, vMin0)), idx0, iMin0);
|
||||
iMax0 = mm_sel(mm_idxmask(v_gt(d0, vMax0)), idx0, iMax0);
|
||||
vMin0 = v_min(d0, vMin0); vMax0 = v_max(d0, vMax0);
|
||||
idx0 = v_add(idx0, inc);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for ( ; k < limit; k += nlanes )
|
||||
{
|
||||
VT data = vx_load(src + k);
|
||||
VT active = mm_active<VT>(mask, k, nlanes);
|
||||
VT cmpMin = v_and(v_lt(data, vMin0), active);
|
||||
VT cmpMax = v_and(v_gt(data, vMax0), active);
|
||||
iMin0 = mm_sel(mm_idxmask(cmpMin), idx0, iMin0);
|
||||
iMax0 = mm_sel(mm_idxmask(cmpMax), idx0, iMax0);
|
||||
vMin0 = v_select(cmpMin, data, vMin0);
|
||||
vMax0 = v_select(cmpMax, data, vMax0);
|
||||
idx0 = v_add(idx0, inc);
|
||||
}
|
||||
}
|
||||
|
||||
j = k;
|
||||
|
||||
mm_fold_min(vMin0, iMin0, none, delta, minVal, minIdx);
|
||||
mm_fold_min(vMin1, iMin1, none, delta, minVal, minIdx);
|
||||
mm_fold_max(vMax0, iMax0, none, delta, maxVal, maxIdx);
|
||||
mm_fold_max(vMax1, iMax1, none, delta, maxVal, maxIdx);
|
||||
}
|
||||
while ( j < len0 );
|
||||
}
|
||||
|
||||
minMaxIdx_finish( src, mask, minval, maxval, minidx, maxidx, minVal, maxVal,
|
||||
minIdx, maxIdx, len, startidx, j );
|
||||
vx_cleanup();
|
||||
}
|
||||
else
|
||||
{
|
||||
minMaxIdx_(src, mask, minval, maxval, minidx, maxidx, len, startidx);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
static void minMaxIdx_8u(const void* src_, const uchar* mask, void* minval_, void* maxval_,
|
||||
size_t* minidx, size_t* maxidx, int len, size_t startidx )
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
minMaxIdx_simd_<uchar, v_uint8, v_uint8, uchar, int>(
|
||||
static_cast<const uchar*>(src_), mask, static_cast<int*>(minval_), static_cast<int*>(maxval_),
|
||||
minidx, maxidx, len, startidx, (int)0, (int)UCHAR_MAX);
|
||||
#else
|
||||
minMaxIdx_(static_cast<const uchar*>(src_), mask, static_cast<int*>(minval_),
|
||||
static_cast<int*>(maxval_), minidx, maxidx, len, startidx);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void minMaxIdx_8s(const void* src_, const uchar* mask, void* minval_, void* maxval_,
|
||||
size_t* minidx, size_t* maxidx, int len, size_t startidx )
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
minMaxIdx_simd_<schar, v_int8, v_uint8, uchar, int>(
|
||||
static_cast<const schar*>(src_), mask, static_cast<int*>(minval_), static_cast<int*>(maxval_),
|
||||
minidx, maxidx, len, startidx, (int)SCHAR_MIN, (int)SCHAR_MAX);
|
||||
#else
|
||||
minMaxIdx_(static_cast<const schar*>(src_), mask, static_cast<int*>(minval_),
|
||||
static_cast<int*>(maxval_), minidx, maxidx, len, startidx);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void minMaxIdx_16u(const void* src_, const uchar* mask, void* minval_, void* maxval_,
|
||||
size_t* minidx, size_t* maxidx, int len, size_t startidx )
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
minMaxIdx_simd_<ushort, v_uint16, v_uint16, ushort, int>(
|
||||
static_cast<const ushort*>(src_), mask, static_cast<int*>(minval_), static_cast<int*>(maxval_),
|
||||
minidx, maxidx, len, startidx, (int)0, (int)USHRT_MAX);
|
||||
#else
|
||||
minMaxIdx_(static_cast<const ushort*>(src_), mask, static_cast<int*>(minval_),
|
||||
static_cast<int*>(maxval_), minidx, maxidx, len, startidx);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void minMaxIdx_16s(const void* src_, const uchar* mask, void* minval_, void* maxval_,
|
||||
size_t* minidx, size_t* maxidx, int len, size_t startidx )
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
minMaxIdx_simd_<short, v_int16, v_uint16, ushort, int>(
|
||||
static_cast<const short*>(src_), mask, static_cast<int*>(minval_), static_cast<int*>(maxval_),
|
||||
minidx, maxidx, len, startidx, (int)SHRT_MIN, (int)SHRT_MAX);
|
||||
#else
|
||||
minMaxIdx_(static_cast<const short*>(src_), mask, static_cast<int*>(minval_),
|
||||
static_cast<int*>(maxval_), minidx, maxidx, len, startidx);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void minMaxIdx_32s(const void* src_, const uchar* mask, void* minval_, void* maxval_,
|
||||
size_t* minidx, size_t* maxidx, int len, size_t startidx )
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
minMaxIdx_simd_<int, v_int32, v_uint32, uint, int>(
|
||||
static_cast<const int*>(src_), mask, static_cast<int*>(minval_), static_cast<int*>(maxval_),
|
||||
minidx, maxidx, len, startidx, INT_MIN, INT_MAX);
|
||||
#else
|
||||
minMaxIdx_(static_cast<const int*>(src_), mask, static_cast<int*>(minval_),
|
||||
static_cast<int*>(maxval_), minidx, maxidx, len, startidx);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void minMaxIdx_32f(const void* src_, const uchar* mask, void* minval_, void* maxval_,
|
||||
size_t* minidx, size_t* maxidx, int len, size_t startidx )
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
minMaxIdx_simd_<float, v_float32, v_uint32, uint, float>(
|
||||
static_cast<const float*>(src_), mask, static_cast<float*>(minval_), static_cast<float*>(maxval_),
|
||||
minidx, maxidx, len, startidx, FLT_MIN, FLT_MAX);
|
||||
#else
|
||||
minMaxIdx_(static_cast<const float*>(src_), mask, static_cast<float*>(minval_),
|
||||
static_cast<float*>(maxval_), minidx, maxidx, len, startidx);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void minMaxIdx_64f(const void* src_, const uchar* mask, void* minval_, void* maxval_,
|
||||
size_t* minidx, size_t* maxidx, int len, size_t startidx )
|
||||
{
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
minMaxIdx_simd_<double, v_float64, v_uint64, uint64_t, double>(
|
||||
static_cast<const double*>(src_), mask, static_cast<double*>(minval_), static_cast<double*>(maxval_),
|
||||
minidx, maxidx, len, startidx, DBL_MIN, DBL_MAX);
|
||||
#else
|
||||
minMaxIdx_(static_cast<const double*>(src_), mask, static_cast<double*>(minval_),
|
||||
static_cast<double*>(maxval_), minidx, maxidx, len, startidx);
|
||||
#endif
|
||||
}
|
||||
|
||||
MinMaxIdxFunc getMinmaxTab(int depth)
|
||||
{
|
||||
static MinMaxIdxFunc minmaxTab[CV_DEPTH_MAX] =
|
||||
{
|
||||
GET_OPTIMIZED(minMaxIdx_8u), GET_OPTIMIZED(minMaxIdx_8s),
|
||||
GET_OPTIMIZED(minMaxIdx_16u), GET_OPTIMIZED(minMaxIdx_16s),
|
||||
GET_OPTIMIZED(minMaxIdx_32s),
|
||||
GET_OPTIMIZED(minMaxIdx_32f), GET_OPTIMIZED(minMaxIdx_64f),
|
||||
0
|
||||
};
|
||||
|
||||
return minmaxTab[depth];
|
||||
}
|
||||
|
||||
#endif // CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_END
|
||||
} // namespace cv
|
||||
@@ -1,6 +1,8 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
//
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
|
||||
|
||||
#include "precomp.hpp"
|
||||
@@ -155,20 +157,27 @@ float normL2Sqr_(const float* a, const float* b, int n)
|
||||
{
|
||||
int j = 0; float d = 0.f;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
const int vl = VTraits<v_float32>::vlanes();
|
||||
v_float32 v_d0 = vx_setzero_f32(), v_d1 = vx_setzero_f32();
|
||||
v_float32 v_d2 = vx_setzero_f32(), v_d3 = vx_setzero_f32();
|
||||
for (; j <= n - 4 * VTraits<v_float32>::vlanes(); j += 4 * VTraits<v_float32>::vlanes())
|
||||
for (; j <= n - 4 * vl; j += 4 * vl)
|
||||
{
|
||||
v_float32 t0 = v_sub(vx_load(a + j), vx_load(b + j));
|
||||
v_float32 t1 = v_sub(vx_load(a + j + VTraits<v_float32>::vlanes()), vx_load(b + j + VTraits<v_float32>::vlanes()));
|
||||
v_float32 t1 = v_sub(vx_load(a + j + vl), vx_load(b + j + vl));
|
||||
v_d0 = v_muladd(t0, t0, v_d0);
|
||||
v_float32 t2 = v_sub(vx_load(a + j + 2 * VTraits<v_float32>::vlanes()), vx_load(b + j + 2 * VTraits<v_float32>::vlanes()));
|
||||
v_float32 t2 = v_sub(vx_load(a + j + 2 * vl), vx_load(b + j + 2 * vl));
|
||||
v_d1 = v_muladd(t1, t1, v_d1);
|
||||
v_float32 t3 = v_sub(vx_load(a + j + 3 * VTraits<v_float32>::vlanes()), vx_load(b + j + 3 * VTraits<v_float32>::vlanes()));
|
||||
v_float32 t3 = v_sub(vx_load(a + j + 3 * vl), vx_load(b + j + 3 * vl));
|
||||
v_d2 = v_muladd(t2, t2, v_d2);
|
||||
v_d3 = v_muladd(t3, t3, v_d3);
|
||||
}
|
||||
d = v_reduce_sum(v_add(v_add(v_add(v_d0, v_d1), v_d2), v_d3));
|
||||
v_d0 = v_add(v_add(v_d0, v_d1), v_add(v_d2, v_d3));
|
||||
for (; j <= n - vl; j += vl)
|
||||
{
|
||||
v_float32 t0 = v_sub(vx_load(a + j), vx_load(b + j));
|
||||
v_d0 = v_muladd(t0, t0, v_d0);
|
||||
}
|
||||
d = v_reduce_sum(v_d0);
|
||||
#endif
|
||||
for( ; j < n; j++ )
|
||||
{
|
||||
@@ -183,16 +192,20 @@ float normL1_(const float* a, const float* b, int n)
|
||||
{
|
||||
int j = 0; float d = 0.f;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
const int vl = VTraits<v_float32>::vlanes();
|
||||
v_float32 v_d0 = vx_setzero_f32(), v_d1 = vx_setzero_f32();
|
||||
v_float32 v_d2 = vx_setzero_f32(), v_d3 = vx_setzero_f32();
|
||||
for (; j <= n - 4 * VTraits<v_float32>::vlanes(); j += 4 * VTraits<v_float32>::vlanes())
|
||||
for (; j <= n - 4 * vl; j += 4 * vl)
|
||||
{
|
||||
v_d0 = v_add(v_d0, v_absdiff(vx_load(a + j), vx_load(b + j)));
|
||||
v_d1 = v_add(v_d1, v_absdiff(vx_load(a + j + VTraits<v_float32>::vlanes()), vx_load(b + j + VTraits<v_float32>::vlanes())));
|
||||
v_d2 = v_add(v_d2, v_absdiff(vx_load(a + j + 2 * VTraits<v_float32>::vlanes()), vx_load(b + j + 2 * VTraits<v_float32>::vlanes())));
|
||||
v_d3 = v_add(v_d3, v_absdiff(vx_load(a + j + 3 * VTraits<v_float32>::vlanes()), vx_load(b + j + 3 * VTraits<v_float32>::vlanes())));
|
||||
v_d1 = v_add(v_d1, v_absdiff(vx_load(a + j + vl), vx_load(b + j + vl)));
|
||||
v_d2 = v_add(v_d2, v_absdiff(vx_load(a + j + 2 * vl), vx_load(b + j + 2 * vl)));
|
||||
v_d3 = v_add(v_d3, v_absdiff(vx_load(a + j + 3 * vl), vx_load(b + j + 3 * vl)));
|
||||
}
|
||||
d = v_reduce_sum(v_add(v_add(v_add(v_d0, v_d1), v_d2), v_d3));
|
||||
v_d0 = v_add(v_add(v_d0, v_d1), v_add(v_d2, v_d3));
|
||||
for (; j <= n - vl; j += vl)
|
||||
v_d0 = v_add(v_d0, v_absdiff(vx_load(a + j), vx_load(b + j)));
|
||||
d = v_reduce_sum(v_d0);
|
||||
#endif
|
||||
for( ; j < n; j++ )
|
||||
d += std::abs(a[j] - b[j]);
|
||||
@@ -203,11 +216,14 @@ int normL1_(const uchar* a, const uchar* b, int n)
|
||||
{
|
||||
int j = 0, d = 0;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
for (; j <= n - 4 * VTraits<v_uint8>::vlanes(); j += 4 * VTraits<v_uint8>::vlanes())
|
||||
const int vl = VTraits<v_uint8>::vlanes();
|
||||
for (; j <= n - 4 * vl; j += 4 * vl)
|
||||
d += v_reduce_sad(vx_load(a + j), vx_load(b + j)) +
|
||||
v_reduce_sad(vx_load(a + j + VTraits<v_uint8>::vlanes()), vx_load(b + j + VTraits<v_uint8>::vlanes())) +
|
||||
v_reduce_sad(vx_load(a + j + 2 * VTraits<v_uint8>::vlanes()), vx_load(b + j + 2 * VTraits<v_uint8>::vlanes())) +
|
||||
v_reduce_sad(vx_load(a + j + 3 * VTraits<v_uint8>::vlanes()), vx_load(b + j + 3 * VTraits<v_uint8>::vlanes()));
|
||||
v_reduce_sad(vx_load(a + j + vl), vx_load(b + j + vl)) +
|
||||
v_reduce_sad(vx_load(a + j + 2 * vl), vx_load(b + j + 2 * vl)) +
|
||||
v_reduce_sad(vx_load(a + j + 3 * vl), vx_load(b + j + 3 * vl));
|
||||
for (; j <= n - vl; j += vl)
|
||||
d += v_reduce_sad(vx_load(a + j), vx_load(b + j));
|
||||
#endif
|
||||
for( ; j < n; j++ )
|
||||
d += std::abs(a[j] - b[j]);
|
||||
|
||||
+481
-48
@@ -3,6 +3,7 @@
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
//
|
||||
// Copyright (C) 2025, SpaceMIT Inc., all rights reserved.
|
||||
// Copyright (C) 2026, Advanced Micro Devices, Inc., all rights reserved.
|
||||
// Third party copyrights are property of their respective owners.
|
||||
|
||||
#include "precomp.hpp"
|
||||
@@ -1462,21 +1463,39 @@ struct MaskedNormInf_SIMD<uchar, int> {
|
||||
result = std::max(result, (int)src[i]);
|
||||
}
|
||||
}
|
||||
else if (cn == 4 && len >= VTraits<v_uint8>::vlanes()) {
|
||||
const int vstep = VTraits<v_uint8>::vlanes();
|
||||
v_uint8 acc = vx_setzero_u8();
|
||||
int i = 0;
|
||||
for (;;) {
|
||||
if (i > len - vstep)
|
||||
i = len - vstep;
|
||||
v_uint8 c0, c1, c2, c3;
|
||||
v_load_deinterleave(src + i * 4, c0, c1, c2, c3);
|
||||
v_uint8 pmax = v_max(v_max(c0, c1), v_max(c2, c3));
|
||||
v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8());
|
||||
acc = v_max(acc, v_and(pmax, m));
|
||||
if (i >= len - vstep)
|
||||
break;
|
||||
i += vstep;
|
||||
}
|
||||
result = (int)v_reduce_max(acc);
|
||||
}
|
||||
else if (cn == 4) {
|
||||
// len < one vector: pure scalar
|
||||
for (int i = 0; i < len; i++) {
|
||||
if (mask[i]) {
|
||||
const uchar* elem = src + i * 4;
|
||||
result = std::max(result, (int)std::max(std::max(elem[0], elem[1]),
|
||||
std::max(elem[2], elem[3])));
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (int i = 0; i < len; i++) {
|
||||
if (mask[i]) {
|
||||
const uchar* elem = src + i * cn;
|
||||
int k = 0;
|
||||
const int vstep = VTraits<v_uint8>::vlanes();
|
||||
v_uint8 acc = vx_setzero_u8();
|
||||
|
||||
for (; k <= cn - vstep; k += vstep) {
|
||||
acc = v_max(acc, vx_load(elem + k));
|
||||
}
|
||||
|
||||
result = std::max(result, (int)v_reduce_max(acc));
|
||||
|
||||
for (; k < cn; k++)
|
||||
for (int k = 0; k < cn; k++)
|
||||
result = std::max(result, (int)elem[k]);
|
||||
}
|
||||
}
|
||||
@@ -1508,22 +1527,30 @@ struct MaskedNormL1_SIMD<uchar, int> {
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (int i = 0; i < len; i++) {
|
||||
if (mask[i]) {
|
||||
const uchar* elem = src + i * cn;
|
||||
int k = 0;
|
||||
const int vstep = VTraits<v_uint8>::vlanes() / 4;
|
||||
v_uint32 acc = vx_setzero_u32();
|
||||
|
||||
for (; k <= cn - vstep; k += vstep) {
|
||||
v_uint32 s = vx_load_expand_q(elem + k);
|
||||
acc = v_add(acc, s);
|
||||
const int vstep = VTraits<v_uint8>::vlanes() / 4;
|
||||
if (cn >= vstep) {
|
||||
for (int i = 0; i < len; i++) {
|
||||
if (mask[i]) {
|
||||
const uchar* elem = src + i * cn;
|
||||
int k = 0;
|
||||
v_uint32 acc = vx_setzero_u32();
|
||||
for (; k <= cn - vstep; k += vstep) {
|
||||
v_uint32 s = vx_load_expand_q(elem + k);
|
||||
acc = v_add(acc, s);
|
||||
}
|
||||
result += (int)v_reduce_sum(acc);
|
||||
for (; k < cn; k++)
|
||||
result += elem[k];
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
for (int i = 0; i < len; i++) {
|
||||
if (mask[i]) {
|
||||
const uchar* elem = src + i * cn;
|
||||
for (int k = 0; k < cn; k++)
|
||||
result += elem[k];
|
||||
}
|
||||
|
||||
result += (int)v_reduce_sum(acc);
|
||||
|
||||
for (; k < cn; k++)
|
||||
result += elem[k];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1738,6 +1765,428 @@ normL2_(const T* src, const uchar* mask, ST* _result, int len, int cn) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
inline v_uint8 v_diffmask(const v_uint8& d, const uchar* m) { return v_and(d, v_gt(vx_load(m), vx_setzero_u8())); }
|
||||
inline v_uint16 v_diffmask(const v_uint16& d, const uchar* m) { return v_and(d, v_gt(vx_load_expand(m), vx_setzero_u16())); }
|
||||
inline v_uint32 v_diffmask(const v_uint32& d, const uchar* m) { return v_and(d, v_gt(vx_load_expand_q(m), vx_setzero_u32())); }
|
||||
inline v_float32 v_diffmask(const v_float32& d, const uchar* m) {
|
||||
v_uint32 cm = v_gt(vx_load_expand_q(m), vx_setzero_u32());
|
||||
return v_reinterpret_as_f32(v_and(v_reinterpret_as_u32(d), cm));
|
||||
}
|
||||
#endif
|
||||
|
||||
template <typename T, typename ST>
|
||||
struct MaskedNormDiffInf_SIMD {
|
||||
inline ST operator()(const T* s1, const T* s2, const uchar* mask, int len, int cn) const {
|
||||
ST result = 0;
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
const int vstep = VTraits<decltype(vx_load(s1))>::vlanes();
|
||||
if (cn == 1 && len >= vstep) {
|
||||
auto acc = v_diffmask(v_absdiff(vx_load(s1), vx_load(s2)), mask);
|
||||
int i = vstep;
|
||||
for (; i <= len - vstep; i += vstep)
|
||||
acc = v_max(acc, v_diffmask(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), mask + i));
|
||||
result = (ST)v_reduce_max(acc);
|
||||
for (; i < len; i++) if (mask[i]) result = std::max(result, (ST)std::abs((double)s1[i] - (double)s2[i]));
|
||||
return result;
|
||||
}
|
||||
#endif
|
||||
for (int i = 0; i < len; i++) if (mask[i]) {
|
||||
const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn;
|
||||
for (int k = 0; k < cn; k++) result = std::max(result, (ST)std::abs((double)e1[k] - (double)e2[k]));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
};
|
||||
template <typename T, typename ST>
|
||||
struct MaskedNormDiffL1_SIMD {
|
||||
inline ST operator()(const T* s1, const T* s2, const uchar* mask, int len, int cn) const {
|
||||
ST s = 0;
|
||||
for (int i = 0; i < len; i++) if (mask[i]) {
|
||||
const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn;
|
||||
for (int k = 0; k < cn; k++) s += std::abs(e1[k] - e2[k]);
|
||||
}
|
||||
return s;
|
||||
}
|
||||
};
|
||||
template <typename T, typename ST>
|
||||
struct MaskedNormDiffL2_SIMD {
|
||||
inline ST operator()(const T* s1, const T* s2, const uchar* mask, int len, int cn) const {
|
||||
ST s = 0;
|
||||
for (int i = 0; i < len; i++) if (mask[i]) {
|
||||
const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn;
|
||||
for (int k = 0; k < cn; k++) { ST v = (ST)e1[k] - (ST)e2[k]; s += v*v; }
|
||||
}
|
||||
return s;
|
||||
}
|
||||
};
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
// Shared scalar cn>1 fallback for masked norm-diff (RT = accumulator type).
|
||||
// Used by every SIMD specialization's multi-channel branch.
|
||||
template<typename T, typename RT>
|
||||
static inline RT maskedNormDiffL1Tail(const T* s1, const T* s2, const uchar* mask, int len, int cn, RT result) {
|
||||
for (int i = 0; i < len; i++) if (mask[i]) {
|
||||
const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn;
|
||||
for (int k = 0; k < cn; k++) result += (RT)std::abs((RT)e1[k] - (RT)e2[k]);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
template<typename T, typename RT>
|
||||
static inline RT maskedNormDiffL2Tail(const T* s1, const T* s2, const uchar* mask, int len, int cn, RT result) {
|
||||
for (int i = 0; i < len; i++) if (mask[i]) {
|
||||
const T* e1 = s1 + i*cn; const T* e2 = s2 + i*cn;
|
||||
for (int k = 0; k < cn; k++) { RT v = (RT)e1[k] - (RT)e2[k]; result += v*v; }
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// Shared 8-bit cn==1 masked kernels (uchar/schar): v_absdiff() yields v_uint8
|
||||
// for both, so a single typename-T body serves both depths.
|
||||
template<typename T>
|
||||
static inline int maskedNormDiffL1_8(const T* s1, const T* s2, const uchar* mask, int len) {
|
||||
int i = 0; const int vstep = VTraits<v_uint8>::vlanes();
|
||||
const v_uint8 one = vx_setall_u8(1);
|
||||
v_uint32 acc = vx_setzero_u32();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8());
|
||||
v_uint8 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m);
|
||||
acc = v_dotprod_expand_fast(ad, one, acc); // sum of masked |diff|
|
||||
}
|
||||
int result = (int)v_reduce_sum(acc);
|
||||
for (; i < len; i++) if (mask[i]) result += std::abs((int)s1[i] - (int)s2[i]);
|
||||
return result;
|
||||
}
|
||||
template<typename T>
|
||||
static inline int maskedNormDiffL2_8(const T* s1, const T* s2, const uchar* mask, int len) {
|
||||
int i = 0; const int vstep = VTraits<v_uint8>::vlanes();
|
||||
v_uint32 acc = vx_setzero_u32();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8());
|
||||
v_uint8 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m);
|
||||
acc = v_dotprod_expand_fast(ad, ad, acc); // sum of masked diff^2
|
||||
}
|
||||
int result = (int)v_reduce_sum(acc);
|
||||
for (; i < len; i++) if (mask[i]) { int v = (int)s1[i] - (int)s2[i]; result += v*v; }
|
||||
return result;
|
||||
}
|
||||
|
||||
template<>
|
||||
struct MaskedNormDiffInf_SIMD<uchar, int> {
|
||||
int operator()(const uchar* s1, const uchar* s2, const uchar* mask, int len, int cn) const {
|
||||
int result = 0;
|
||||
const int vstep = VTraits<v_uint8>::vlanes();
|
||||
if (cn == 1) {
|
||||
int i = 0;
|
||||
v_uint8 acc = vx_setzero_u8();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint8 ad = v_absdiff(vx_load(s1 + i), vx_load(s2 + i));
|
||||
v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8());
|
||||
acc = v_max(acc, v_and(ad, m));
|
||||
}
|
||||
result = (int)v_reduce_max(acc);
|
||||
for (; i < len; i++) if (mask[i]) result = std::max(result, std::abs((int)s1[i] - (int)s2[i]));
|
||||
}
|
||||
else if (cn == 4 && len >= vstep) {
|
||||
v_uint8 acc = vx_setzero_u8();
|
||||
int i = 0;
|
||||
for (;;) {
|
||||
if (i > len - vstep) i = len - vstep; // back-step (max is idempotent)
|
||||
v_uint8 a0,a1,a2,a3,b0,b1,b2,b3;
|
||||
v_load_deinterleave(s1 + i*4, a0,a1,a2,a3);
|
||||
v_load_deinterleave(s2 + i*4, b0,b1,b2,b3);
|
||||
v_uint8 ad = v_max(v_max(v_absdiff(a0,b0), v_absdiff(a1,b1)),
|
||||
v_max(v_absdiff(a2,b2), v_absdiff(a3,b3)));
|
||||
v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8());
|
||||
acc = v_max(acc, v_and(ad, m));
|
||||
if (i >= len - vstep) break;
|
||||
i += vstep;
|
||||
}
|
||||
result = (int)v_reduce_max(acc);
|
||||
}
|
||||
else {
|
||||
for (int i = 0; i < len; i++) if (mask[i]) {
|
||||
const uchar* e1 = s1 + i*cn; const uchar* e2 = s2 + i*cn;
|
||||
for (int k = 0; k < cn; k++) result = std::max(result, std::abs((int)e1[k] - (int)e2[k]));
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct MaskedNormDiffInf_SIMD<int, int> {
|
||||
int operator()(const int* s1, const int* s2, const uchar* mask, int len, int cn) const {
|
||||
int result = 0;
|
||||
const int vstep = VTraits<v_int32>::vlanes();
|
||||
if (cn == 1 && len >= vstep) {
|
||||
// Use wrapping int subtraction (v_abs(v_sub)) to match the non-masked
|
||||
// NormDiffInf_SIMD<int,int> kernel. v_absdiff would compute the true
|
||||
// unsigned |a-b| which can exceed INT_MAX and overflow on the cast to int.
|
||||
v_uint32 acc = vx_setzero_u32();
|
||||
int i = 0;
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint32 ad = v_abs(v_sub(vx_load(s1 + i), vx_load(s2 + i)));
|
||||
v_uint32 m = v_gt(vx_load_expand_q(mask + i), vx_setzero_u32());
|
||||
acc = v_max(acc, v_and(ad, m));
|
||||
}
|
||||
result = (int)v_reduce_max(acc);
|
||||
for (; i < len; i++) if (mask[i]) result = std::max(result, (int)std::abs(s1[i] - s2[i]));
|
||||
return result;
|
||||
}
|
||||
for (int i = 0; i < len; i++) if (mask[i]) {
|
||||
const int* e1 = s1 + i*cn; const int* e2 = s2 + i*cn;
|
||||
for (int k = 0; k < cn; k++) result = std::max(result, (int)std::abs(e1[k] - e2[k]));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct MaskedNormDiffL1_SIMD<uchar, int> {
|
||||
int operator()(const uchar* s1, const uchar* s2, const uchar* mask, int len, int cn) const {
|
||||
if (cn == 1) return maskedNormDiffL1_8(s1, s2, mask, len);
|
||||
if (cn == 4) {
|
||||
int result = 0;
|
||||
int i = 0;
|
||||
const int vstep = VTraits<v_uint8>::vlanes();
|
||||
const v_uint8 one = vx_setall_u8(1);
|
||||
v_uint32 acc = vx_setzero_u32();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint8 a0,a1,a2,a3,b0,b1,b2,b3;
|
||||
v_load_deinterleave(s1 + i*4, a0,a1,a2,a3);
|
||||
v_load_deinterleave(s2 + i*4, b0,b1,b2,b3);
|
||||
v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8());
|
||||
acc = v_dotprod_expand_fast(v_and(v_absdiff(a0,b0), m), one, acc);
|
||||
acc = v_dotprod_expand_fast(v_and(v_absdiff(a1,b1), m), one, acc);
|
||||
acc = v_dotprod_expand_fast(v_and(v_absdiff(a2,b2), m), one, acc);
|
||||
acc = v_dotprod_expand_fast(v_and(v_absdiff(a3,b3), m), one, acc);
|
||||
}
|
||||
result = (int)v_reduce_sum(acc);
|
||||
for (; i < len; i++) if (mask[i]) {
|
||||
const uchar* e1 = s1 + i*4; const uchar* e2 = s2 + i*4;
|
||||
for (int k = 0; k < 4; k++) result += std::abs((int)e1[k] - (int)e2[k]);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
return maskedNormDiffL1Tail<uchar, int>(s1, s2, mask, len, cn, 0);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct MaskedNormDiffL2_SIMD<uchar, int> {
|
||||
int operator()(const uchar* s1, const uchar* s2, const uchar* mask, int len, int cn) const {
|
||||
if (cn == 1) return maskedNormDiffL2_8(s1, s2, mask, len);
|
||||
if (cn == 4) {
|
||||
int result = 0;
|
||||
int i = 0;
|
||||
const int vstep = VTraits<v_uint8>::vlanes();
|
||||
v_uint32 acc = vx_setzero_u32();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint8 a0,a1,a2,a3,b0,b1,b2,b3;
|
||||
v_load_deinterleave(s1 + i*4, a0,a1,a2,a3);
|
||||
v_load_deinterleave(s2 + i*4, b0,b1,b2,b3);
|
||||
v_uint8 m = v_gt(vx_load(mask + i), vx_setzero_u8());
|
||||
v_uint8 d0 = v_and(v_absdiff(a0,b0), m), d1 = v_and(v_absdiff(a1,b1), m);
|
||||
v_uint8 d2 = v_and(v_absdiff(a2,b2), m), d3 = v_and(v_absdiff(a3,b3), m);
|
||||
acc = v_dotprod_expand_fast(d0, d0, acc);
|
||||
acc = v_dotprod_expand_fast(d1, d1, acc);
|
||||
acc = v_dotprod_expand_fast(d2, d2, acc);
|
||||
acc = v_dotprod_expand_fast(d3, d3, acc);
|
||||
}
|
||||
result = (int)v_reduce_sum(acc);
|
||||
for (; i < len; i++) if (mask[i]) {
|
||||
const uchar* e1 = s1 + i*4; const uchar* e2 = s2 + i*4;
|
||||
for (int k = 0; k < 4; k++) { int v = (int)e1[k] - (int)e2[k]; result += v*v; }
|
||||
}
|
||||
return result;
|
||||
}
|
||||
return maskedNormDiffL2Tail<uchar, int>(s1, s2, mask, len, cn, 0);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct MaskedNormDiffInf_SIMD<double, double> {
|
||||
inline double operator()(const double* s1, const double* s2, const uchar* mask, int len, int cn) const {
|
||||
double result = 0.0;
|
||||
for (int i = 0; i < len; i++) if (mask[i]) {
|
||||
const double* e1 = s1 + i*cn; const double* e2 = s2 + i*cn;
|
||||
for (int k = 0; k < cn; k++) result = std::max(result, std::abs(e1[k] - e2[k]));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
};
|
||||
|
||||
#if CV_SIMD_64F || CV_SIMD_SCALABLE_64F
|
||||
template<>
|
||||
struct MaskedNormDiffL1_SIMD<int, double> {
|
||||
double operator()(const int* s1, const int* s2, const uchar* mask, int len, int cn) const {
|
||||
double result = 0.0;
|
||||
if (cn == 1) {
|
||||
int i = 0;
|
||||
const int vstep = VTraits<v_int32>::vlanes();
|
||||
v_float64 acc = vx_setzero_f64();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint32 m = v_gt(vx_load_expand_q(mask + i), vx_setzero_u32());
|
||||
v_uint32 ad = v_and(v_abs(v_sub(vx_load(s1 + i), vx_load(s2 + i))), m);
|
||||
v_int32 adi = v_reinterpret_as_s32(ad);
|
||||
acc = v_add(acc, v_cvt_f64(adi));
|
||||
acc = v_add(acc, v_cvt_f64_high(adi));
|
||||
}
|
||||
result = v_reduce_sum(acc);
|
||||
// Use wrapping int subtraction to match the SIMD body (v_abs(v_sub))
|
||||
// and the non-masked NormDiffL1_SIMD<int,double> kernel.
|
||||
for (; i < len; i++) if (mask[i]) result += std::abs(s1[i] - s2[i]);
|
||||
}
|
||||
else {
|
||||
result = maskedNormDiffL1Tail<int, double>(s1, s2, mask, len, cn, 0.0);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
};
|
||||
|
||||
// Masked L1/L2 SIMD for the remaining depths (cn==1). Without these the masked
|
||||
// norm-diff for 8s/16u/16s/32s/32f fell back to the scalar base template, which
|
||||
// the compiler autovectorizes poorly once AVX-512 dispatch is enabled.
|
||||
template<> struct MaskedNormDiffL1_SIMD<schar, int> {
|
||||
int operator()(const schar* s1, const schar* s2, const uchar* mask, int len, int cn) const {
|
||||
if (cn == 1) return maskedNormDiffL1_8(s1, s2, mask, len);
|
||||
return maskedNormDiffL1Tail<schar, int>(s1, s2, mask, len, cn, 0);
|
||||
}
|
||||
};
|
||||
|
||||
// Shared 16-bit masked L1 kernel (ushort/short): v_absdiff() yields v_uint16
|
||||
// for both, so a single typename-T body serves both depths.
|
||||
template<typename T>
|
||||
static inline int maskedNormDiffL1_16(const T* s1, const T* s2, const uchar* mask, int len, int cn) {
|
||||
int result = 0;
|
||||
if (cn == 1) {
|
||||
int i = 0; const int vstep = VTraits<v_uint16>::vlanes();
|
||||
v_uint32 acc = vx_setzero_u32();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint16 m = v_gt(vx_load_expand(mask + i), vx_setzero_u16());
|
||||
v_uint16 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m);
|
||||
v_uint32 lo, hi; v_expand(ad, lo, hi);
|
||||
acc = v_add(acc, v_add(lo, hi));
|
||||
}
|
||||
result = (int)v_reduce_sum(acc);
|
||||
for (; i < len; i++) if (mask[i]) result += std::abs((int)s1[i] - (int)s2[i]);
|
||||
} else {
|
||||
result = maskedNormDiffL1Tail<T, int>(s1, s2, mask, len, cn, 0);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
template<> struct MaskedNormDiffL1_SIMD<ushort, int> {
|
||||
int operator()(const ushort* s1, const ushort* s2, const uchar* mask, int len, int cn) const
|
||||
{ return maskedNormDiffL1_16(s1, s2, mask, len, cn); } };
|
||||
template<> struct MaskedNormDiffL1_SIMD<short, int> {
|
||||
int operator()(const short* s1, const short* s2, const uchar* mask, int len, int cn) const
|
||||
{ return maskedNormDiffL1_16(s1, s2, mask, len, cn); } };
|
||||
|
||||
template<>
|
||||
struct MaskedNormDiffL1_SIMD<float, double> {
|
||||
double operator()(const float* s1, const float* s2, const uchar* mask, int len, int cn) const {
|
||||
double result = 0.0;
|
||||
if (cn == 1) {
|
||||
int i = 0; const int vstep = VTraits<v_float32>::vlanes();
|
||||
v_float64 acc0 = vx_setzero_f64(), acc1 = vx_setzero_f64();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_float32 m = v_reinterpret_as_f32(v_gt(vx_load_expand_q(mask + i), vx_setzero_u32()));
|
||||
v_float32 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m);
|
||||
acc0 = v_add(acc0, v_cvt_f64(ad));
|
||||
acc1 = v_add(acc1, v_cvt_f64_high(ad));
|
||||
}
|
||||
result = v_reduce_sum(v_add(acc0, acc1));
|
||||
for (; i < len; i++) if (mask[i]) result += std::abs((double)s1[i] - (double)s2[i]);
|
||||
} else {
|
||||
result = maskedNormDiffL1Tail<float, double>(s1, s2, mask, len, cn, 0.0);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
};
|
||||
|
||||
template<> struct MaskedNormDiffL2_SIMD<schar, int> {
|
||||
int operator()(const schar* s1, const schar* s2, const uchar* mask, int len, int cn) const {
|
||||
if (cn == 1) return maskedNormDiffL2_8(s1, s2, mask, len);
|
||||
return maskedNormDiffL2Tail<schar, int>(s1, s2, mask, len, cn, 0);
|
||||
}
|
||||
};
|
||||
|
||||
// Shared 16-bit masked L2 kernel (ushort/short): v_absdiff() yields v_uint16
|
||||
// for both, so a single typename-T body serves both depths.
|
||||
template<typename T>
|
||||
static inline double maskedNormDiffL2_16(const T* s1, const T* s2, const uchar* mask, int len, int cn) {
|
||||
double result = 0.0;
|
||||
if (cn == 1) {
|
||||
int i = 0; const int vstep = VTraits<v_uint16>::vlanes();
|
||||
v_float64 acc = vx_setzero_f64();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint16 m = v_gt(vx_load_expand(mask + i), vx_setzero_u16());
|
||||
v_uint16 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m);
|
||||
v_uint64 u = v_dotprod_expand_fast(ad, ad);
|
||||
acc = v_add(acc, v_cvt_f64(v_reinterpret_as_s64(u)));
|
||||
}
|
||||
result = v_reduce_sum(acc);
|
||||
for (; i < len; i++) if (mask[i]) { double v = (double)s1[i] - (double)s2[i]; result += v*v; }
|
||||
} else {
|
||||
result = maskedNormDiffL2Tail<T, double>(s1, s2, mask, len, cn, 0.0);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
template<> struct MaskedNormDiffL2_SIMD<ushort, double> {
|
||||
double operator()(const ushort* s1, const ushort* s2, const uchar* mask, int len, int cn) const
|
||||
{ return maskedNormDiffL2_16(s1, s2, mask, len, cn); } };
|
||||
template<> struct MaskedNormDiffL2_SIMD<short, double> {
|
||||
double operator()(const short* s1, const short* s2, const uchar* mask, int len, int cn) const
|
||||
{ return maskedNormDiffL2_16(s1, s2, mask, len, cn); } };
|
||||
|
||||
template<>
|
||||
struct MaskedNormDiffL2_SIMD<int, double> {
|
||||
double operator()(const int* s1, const int* s2, const uchar* mask, int len, int cn) const {
|
||||
double result = 0.0;
|
||||
if (cn == 1) {
|
||||
int i = 0; const int vstep = VTraits<v_int32>::vlanes();
|
||||
v_float64 r0 = vx_setzero_f64(), r1 = vx_setzero_f64();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_uint32 m = v_gt(vx_load_expand_q(mask + i), vx_setzero_u32());
|
||||
v_uint32 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m);
|
||||
v_uint64 e0, e1; v_expand(ad, e0, e1);
|
||||
v_float64 f0 = v_cvt_f64(v_reinterpret_as_s64(e0)), f1 = v_cvt_f64(v_reinterpret_as_s64(e1));
|
||||
r0 = v_fma(f0, f0, r0); r1 = v_fma(f1, f1, r1);
|
||||
}
|
||||
result = v_reduce_sum(v_add(r0, r1));
|
||||
for (; i < len; i++) if (mask[i]) { double v = (double)s1[i] - (double)s2[i]; result += v*v; }
|
||||
} else {
|
||||
result = maskedNormDiffL2Tail<int, double>(s1, s2, mask, len, cn, 0.0);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct MaskedNormDiffL2_SIMD<float, double> {
|
||||
double operator()(const float* s1, const float* s2, const uchar* mask, int len, int cn) const {
|
||||
double result = 0.0;
|
||||
if (cn == 1) {
|
||||
int i = 0; const int vstep = VTraits<v_float32>::vlanes();
|
||||
v_float64 r0 = vx_setzero_f64(), r1 = vx_setzero_f64();
|
||||
for (; i <= len - vstep; i += vstep) {
|
||||
v_float32 m = v_reinterpret_as_f32(v_gt(vx_load_expand_q(mask + i), vx_setzero_u32()));
|
||||
v_float32 ad = v_and(v_absdiff(vx_load(s1 + i), vx_load(s2 + i)), m);
|
||||
v_float64 f0 = v_cvt_f64(ad), f1 = v_cvt_f64_high(ad);
|
||||
r0 = v_fma(f0, f0, r0); r1 = v_fma(f1, f1, r1);
|
||||
}
|
||||
result = v_reduce_sum(v_add(r0, r1));
|
||||
for (; i < len; i++) if (mask[i]) { double v = (double)s1[i] - (double)s2[i]; result += v*v; }
|
||||
} else {
|
||||
result = maskedNormDiffL2Tail<float, double>(s1, s2, mask, len, cn, 0.0);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
};
|
||||
#endif
|
||||
#endif
|
||||
|
||||
template<typename T, typename ST> int
|
||||
normDiffInf_(const T* src1, const T* src2, const uchar* mask, ST* _result, int len, int cn) {
|
||||
ST result = *_result;
|
||||
@@ -1745,13 +2194,8 @@ normDiffInf_(const T* src1, const T* src2, const uchar* mask, ST* _result, int l
|
||||
NormDiffInf_SIMD<T, ST> op;
|
||||
result = std::max(result, op(src1, src2, len*cn));
|
||||
} else {
|
||||
for( int i = 0; i < len; i++, src1 += cn, src2 += cn ) {
|
||||
if( mask[i] ) {
|
||||
for( int k = 0; k < cn; k++ ) {
|
||||
result = std::max(result, (ST)std::abs(src1[k] - src2[k]));
|
||||
}
|
||||
}
|
||||
}
|
||||
MaskedNormDiffInf_SIMD<T, ST> op;
|
||||
result = std::max(result, op(src1, src2, mask, len, cn));
|
||||
}
|
||||
*_result = result;
|
||||
return 0;
|
||||
@@ -1765,13 +2209,8 @@ normDiffL1_(const T* src1, const T* src2, const uchar* mask, ST* _result, int le
|
||||
result += op(src1, src2, len*cn);
|
||||
}
|
||||
else {
|
||||
for( int i = 0; i < len; i++, src1 += cn, src2 += cn ) {
|
||||
if( mask[i] ) {
|
||||
for( int k = 0; k < cn; k++ ) {
|
||||
result += std::abs(src1[k] - src2[k]);
|
||||
}
|
||||
}
|
||||
}
|
||||
MaskedNormDiffL1_SIMD<T, ST> op;
|
||||
result += op(src1, src2, mask, len, cn);
|
||||
}
|
||||
*_result = result;
|
||||
return 0;
|
||||
@@ -1784,14 +2223,8 @@ normDiffL2_(const T* src1, const T* src2, const uchar* mask, ST* _result, int le
|
||||
NormDiffL2_SIMD<T, ST> op;
|
||||
result += op(src1, src2, len*cn);
|
||||
} else {
|
||||
for( int i = 0; i < len; i++, src1 += cn, src2 += cn ) {
|
||||
if( mask[i] ) {
|
||||
for( int k = 0; k < cn; k++ ) {
|
||||
ST v = (ST)src1[k] - (ST)src2[k];
|
||||
result += v*v;
|
||||
}
|
||||
}
|
||||
}
|
||||
MaskedNormDiffL2_SIMD<T, ST> op;
|
||||
result += op(src1, src2, mask, len, cn);
|
||||
}
|
||||
*_result = result;
|
||||
return 0;
|
||||
|
||||
@@ -1682,27 +1682,12 @@ Context& initializeContextFromGL()
|
||||
|
||||
if(extensionSize > 0)
|
||||
{
|
||||
char* extensions = nullptr;
|
||||
|
||||
try {
|
||||
extensions = new char[extensionSize];
|
||||
|
||||
status = clGetDeviceInfo(devices[j], CL_DEVICE_EXTENSIONS, extensionSize, extensions, &extensionSize);
|
||||
if (status != CL_SUCCESS)
|
||||
continue;
|
||||
} catch(...) {
|
||||
CV_Error(cv::Error::OpenCLInitError, "OpenCL: Exception thrown during device extensions gathering");
|
||||
}
|
||||
|
||||
std::string devString;
|
||||
|
||||
if(extensions != nullptr) {
|
||||
devString = extensions;
|
||||
delete[] extensions;
|
||||
}
|
||||
else {
|
||||
CV_Error(cv::Error::OpenCLInitError, "OpenCL: Unexpected error during device extensions gathering");
|
||||
}
|
||||
std::string devString(extensionSize, '\0');
|
||||
status = clGetDeviceInfo(devices[j], CL_DEVICE_EXTENSIONS, devString.size(), &devString[0], &extensionSize);
|
||||
if (status != CL_SUCCESS)
|
||||
continue;
|
||||
if (extensionSize > 0 && devString.size() >= extensionSize)
|
||||
devString.resize(extensionSize - 1);
|
||||
|
||||
size_t oldPos = 0;
|
||||
size_t spacePos = devString.find(' ', oldPos); // extensions string is space delimited
|
||||
|
||||
@@ -266,7 +266,7 @@ int decodeFormat( const char* dt, int* fmt_pairs, int max_len )
|
||||
int calcElemSize( const char* dt, int initial_size )
|
||||
{
|
||||
int size = 0;
|
||||
int fmt_pairs[CV_FS_MAX_FMT_PAIRS], i, fmt_pair_count;
|
||||
int fmt_pairs[CV_FS_MAX_FMT_PAIRS*2], i, fmt_pair_count;
|
||||
int comp_size;
|
||||
|
||||
fmt_pair_count = decodeFormat( dt, fmt_pairs, CV_FS_MAX_FMT_PAIRS );
|
||||
@@ -316,7 +316,7 @@ int calcStructSize( const char* dt, int initial_size )
|
||||
int decodeSimpleFormat( const char* dt )
|
||||
{
|
||||
int elem_type = -1;
|
||||
int fmt_pairs[CV_FS_MAX_FMT_PAIRS], fmt_pair_count;
|
||||
int fmt_pairs[CV_FS_MAX_FMT_PAIRS*2], fmt_pair_count;
|
||||
|
||||
fmt_pair_count = decodeFormat( dt, fmt_pairs, CV_FS_MAX_FMT_PAIRS );
|
||||
if( fmt_pair_count != 1 || fmt_pairs[0] >= CV_CN_MAX)
|
||||
@@ -2376,7 +2376,7 @@ FileNode::operator float() const
|
||||
|
||||
if( type == INT )
|
||||
{
|
||||
return (float)readInt(p);
|
||||
return (float)readLong(p);
|
||||
}
|
||||
else if( type == REAL )
|
||||
{
|
||||
@@ -2397,7 +2397,7 @@ FileNode::operator double() const
|
||||
|
||||
if( type == INT )
|
||||
{
|
||||
return (double)readInt(p);
|
||||
return (double)readLong(p);
|
||||
}
|
||||
else if( type == REAL )
|
||||
{
|
||||
|
||||
@@ -65,15 +65,15 @@ public:
|
||||
/*
|
||||
* a convertor must provide :
|
||||
* - `operator >> (uchar * & dst)` for writing current binary data to `dst` and moving to next data.
|
||||
* - `operator bool` for checking if current loaction is valid and not the end.
|
||||
* - `operator bool` for checking if current location is valid and not the end.
|
||||
*/
|
||||
template<typename _to_binary_convertor_t> inline
|
||||
Base64ContextEmitter & write(_to_binary_convertor_t & convertor)
|
||||
{
|
||||
static const size_t BUFFER_MAX_LEN = 1024U;
|
||||
constexpr size_t BUFFER_MAX_LEN = 1024U;
|
||||
|
||||
std::vector<uchar> buffer(BUFFER_MAX_LEN);
|
||||
uchar * beg = buffer.data();
|
||||
uchar buffer[BUFFER_MAX_LEN];
|
||||
uchar * beg = buffer;
|
||||
uchar * end = beg;
|
||||
|
||||
while (convertor) {
|
||||
|
||||
@@ -75,7 +75,7 @@ void write( FileStorage& fs, const String& name, const SparseMat& m )
|
||||
fs << "data" << "[:";
|
||||
|
||||
size_t i = 0, n = m.nzcount();
|
||||
std::vector<const SparseMat::Node*> elems(n);
|
||||
AutoBuffer<const SparseMat::Node*> elems(n);
|
||||
SparseMatConstIterator it = m.begin(), it_end = m.end();
|
||||
|
||||
for( ; it != it_end; ++it )
|
||||
@@ -142,6 +142,7 @@ void read(const FileNode& node, Mat& m, const Mat& default_mat)
|
||||
CV_Assert( !sizes_node.empty() );
|
||||
|
||||
dims = (int)sizes_node.size();
|
||||
CV_Assert( dims > 0 && dims <= CV_MAX_DIM );
|
||||
sizes_node.readRaw("i", sizes, dims*sizeof(sizes[0]));
|
||||
|
||||
m.create(dims, sizes, elem_type);
|
||||
@@ -180,6 +181,7 @@ void read( const FileNode& node, SparseMat& m, const SparseMat& default_mat )
|
||||
CV_Assert( !sizes_node.empty() );
|
||||
|
||||
int dims = (int)sizes_node.size();
|
||||
CV_Assert( dims > 0 && dims <= CV_MAX_DIM );
|
||||
sizes_node.readRaw("i", sizes, dims*sizeof(sizes[0]));
|
||||
|
||||
m.create(dims, sizes, elem_type);
|
||||
|
||||
@@ -9,18 +9,50 @@
|
||||
|
||||
namespace cv {
|
||||
|
||||
typedef void (*ReduceSumFunc)(const Mat& src, Mat& dst);
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCSumFunc, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth)
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCAvgFunc, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCMaxFunc, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCMinFunc, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceCSum2Func, (sdepth, ddepth),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
CV_CPU_DISPATCH(getReduceRSumFunc, (sdepth, ddepth),
|
||||
|
||||
@@ -7,12 +7,20 @@
|
||||
namespace cv {
|
||||
CV_CPU_OPTIMIZATION_NAMESPACE_BEGIN
|
||||
|
||||
typedef void (*ReduceSumFunc)(const Mat& src, Mat& dst);
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
typedef void (*ReduceFunc)( const Mat& src, Mat& dst );
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth);
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth);
|
||||
|
||||
#ifndef CV_CPU_OPTIMIZATION_DECLARATIONS_ONLY
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
#include "reduce_c.simd.hpp"
|
||||
#endif
|
||||
|
||||
// =====================================================================
|
||||
// Col reduce SUM (dim=1): sum each row into cn output values
|
||||
// =====================================================================
|
||||
@@ -1089,7 +1097,7 @@ static void reduceRowSum_64f64f(const Mat& srcmat, Mat& dstmat)
|
||||
// Dispatchers
|
||||
// =====================================================================
|
||||
|
||||
ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
ReduceFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_32S) return reduceColSum_8u32s;
|
||||
@@ -1108,7 +1116,68 @@ ReduceSumFunc getReduceCSumFunc(int sdepth, int ddepth)
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ReduceSumFunc getReduceRSumFunc(int sdepth, int ddepth)
|
||||
ReduceFunc getReduceCAvgFunc(int sdepth, int ddepth)
|
||||
{
|
||||
return getReduceCSumFunc(sdepth, ddepth);
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCMaxFunc(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_8U) return reduceColMax_8u;
|
||||
if (sdepth == CV_16U && ddepth == CV_16U) return reduceColMax_16u;
|
||||
if (sdepth == CV_16S && ddepth == CV_16S) return reduceColMax_16s;
|
||||
if (sdepth == CV_32F && ddepth == CV_32F) return reduceColMax_32f;
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
if (sdepth == CV_64F && ddepth == CV_64F) return reduceColMax_64f;
|
||||
#endif
|
||||
#else
|
||||
CV_UNUSED(sdepth);
|
||||
CV_UNUSED(ddepth);
|
||||
#endif
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCMinFunc(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_8U) return reduceColMin_8u;
|
||||
if (sdepth == CV_16U && ddepth == CV_16U) return reduceColMin_16u;
|
||||
if (sdepth == CV_16S && ddepth == CV_16S) return reduceColMin_16s;
|
||||
if (sdepth == CV_32F && ddepth == CV_32F) return reduceColMin_32f;
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
if (sdepth == CV_64F && ddepth == CV_64F) return reduceColMin_64f;
|
||||
#endif
|
||||
#else
|
||||
CV_UNUSED(sdepth);
|
||||
CV_UNUSED(ddepth);
|
||||
#endif
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ReduceFunc getReduceCSum2Func(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_32S) return reduceColSum2_8u32s;
|
||||
if (sdepth == CV_8U && ddepth == CV_32F) return reduceColSum2_8u32f;
|
||||
if (sdepth == CV_8U && ddepth == CV_64F) return reduceColSum2_8u64f;
|
||||
if (sdepth == CV_16U && ddepth == CV_32F) return reduceColSum2_16u32f;
|
||||
if (sdepth == CV_16S && ddepth == CV_32F) return reduceColSum2_16s32f;
|
||||
if (sdepth == CV_32F && ddepth == CV_32F) return reduceColSum2_32f32f;
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
if (sdepth == CV_16U && ddepth == CV_64F) return reduceColSum2_16u64f;
|
||||
if (sdepth == CV_16S && ddepth == CV_64F) return reduceColSum2_16s64f;
|
||||
if (sdepth == CV_32F && ddepth == CV_64F) return reduceColSum2_32f64f;
|
||||
if (sdepth == CV_64F && ddepth == CV_64F) return reduceColSum2_64f64f;
|
||||
#endif
|
||||
#else
|
||||
CV_UNUSED(sdepth);
|
||||
CV_UNUSED(ddepth);
|
||||
#endif
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
ReduceFunc getReduceRSumFunc(int sdepth, int ddepth)
|
||||
{
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
if (sdepth == CV_8U && ddepth == CV_32S) return reduceRowSum_8u32s;
|
||||
|
||||
@@ -0,0 +1,539 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#include "reduce_c_generic.hpp"
|
||||
|
||||
#if CV_RVV
|
||||
#include "reduce_c_rvv.hpp"
|
||||
#endif
|
||||
|
||||
#if CV_NEON
|
||||
#include "reduce_c_neon.hpp"
|
||||
#endif
|
||||
|
||||
#if CV_AVX2
|
||||
#include "reduce_c_avx2.hpp"
|
||||
#endif
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax8uC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax8uC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax8uC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_8uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax8uC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax8uC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax8uC3<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_8uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax8uC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax8uC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax8uC4<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_8uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cn = srcmat.channels();
|
||||
if (cn == 1)
|
||||
reduceColMinMax_8uC1<isMax>(srcmat, dstmat);
|
||||
else if (cn == 3)
|
||||
reduceColMinMax_8uC3<isMax>(srcmat, dstmat);
|
||||
else if (cn == 4)
|
||||
reduceColMinMax_8uC4<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_8uFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_8u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_8u<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_8u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_8u<false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16uC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16uC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16uC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16uC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16uC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16uC4<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16uC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16uC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16uC3<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16uFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() == 1)
|
||||
reduceColMinMax_16uC1<isMax>(srcmat, dstmat);
|
||||
else if (srcmat.channels() == 3)
|
||||
reduceColMinMax_16uC3<isMax>(srcmat, dstmat);
|
||||
else if (srcmat.channels() == 4)
|
||||
reduceColMinMax_16uC4<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_16uFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_16u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_16u<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_16u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_16u<false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16sC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16sC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16sC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16sC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16sFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16sC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16sC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16sC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16sC4<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16sFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16sC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax16sC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax16sC3<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax16sC3<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_16sFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16s(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() == 1)
|
||||
reduceColMinMax_16sC1<isMax>(srcmat, dstmat);
|
||||
else if (srcmat.channels() == 3)
|
||||
reduceColMinMax_16sC3<isMax>(srcmat, dstmat);
|
||||
else if (srcmat.channels() == 4)
|
||||
reduceColMinMax_16sC4<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_16sFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_16s(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_16s<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_16s(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_16s<false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax32fC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax32fC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax32fC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_32fFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32fC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax32fC3<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_32fFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::minMax32fC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::minMax32fC4<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax32fC4<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_32fFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cn = srcmat.channels();
|
||||
if (cn == 1)
|
||||
reduceColMinMax_32fC1<isMax>(srcmat, dstmat);
|
||||
else if (cn == 3)
|
||||
reduceColMinMax_32fC3<isMax>(srcmat, dstmat);
|
||||
else if (cn == 4)
|
||||
reduceColMinMax_32fC4<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_32fFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_32f<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_32f<false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::minMax64fC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::minMax64fC1<isMax>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::minMax64fC1<isMax>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColMinMax_64fFallback<isMax>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() == 1)
|
||||
reduceColMinMax_64fC1<isMax>(srcmat, dstmat);
|
||||
else
|
||||
reduceColMinMax_64fFallback<isMax>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMax_64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_64f<true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColMin_64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColMinMax_64f<false>(srcmat, dstmat);
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_8uC1<DT>(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_8uC1<DT>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_8uC1<DT>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_8uFallback<DT>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_8uC3<DT>(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_8uC3<DT>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_8uC3<DT>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_8uFallback<DT>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_8uC4<DT>(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_8uC4<DT>(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_8uC4<DT>(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_8uFallback<DT>(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8u(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cn = srcmat.channels();
|
||||
if (cn == 1)
|
||||
reduceColSum2_8uC1<DT>(srcmat, dstmat);
|
||||
else if (cn == 3)
|
||||
reduceColSum2_8uC3<DT>(srcmat, dstmat);
|
||||
else if (cn == 4)
|
||||
reduceColSum2_8uC4<DT>(srcmat, dstmat);
|
||||
else
|
||||
reduceColSum2_8uFallback<DT>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_8u32s(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColSum2_8u<int>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_8u32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColSum2_8u<float>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_8u64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColSum2_8u<double>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_16u32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_16u32fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_16u32fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_16u32fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_16u32fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_16u32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_16s32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_16s32fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_16s32fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_16s32fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_16s32fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_16s32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_32fC1(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::sum2_32fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_32fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_32f32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32fC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_32fC3(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_32f32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
#if CV_RVV
|
||||
reduce_c_rvv::sum2_32fC4(srcmat, dstmat);
|
||||
#elif CV_NEON
|
||||
reduce_c_neon::sum2_32fC4(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_32fC4(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_32f32fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cn = srcmat.channels();
|
||||
if (cn == 1)
|
||||
reduceColSum2_32f32fC1(srcmat, dstmat);
|
||||
else if (cn == 3)
|
||||
reduceColSum2_32f32fC3(srcmat, dstmat);
|
||||
else if (cn == 4)
|
||||
reduceColSum2_32f32fC4(srcmat, dstmat);
|
||||
else
|
||||
reduceColSum2_32f32fFallback(srcmat, dstmat);
|
||||
}
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
static void reduceColSum2_16u64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_16u64fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::sum2_16u64fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_16u64fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_16u64fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_16u64fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_16s64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_16s64fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::sum2_16s64fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_16s64fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_16s64fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_16s64fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_32f64fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::sum2_32f64fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_32f64fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_32f64fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_32f64fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
|
||||
static void reduceColSum2_64f64f(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (srcmat.channels() != 1)
|
||||
{
|
||||
reduceColSum2_64f64fFallback(srcmat, dstmat);
|
||||
return;
|
||||
}
|
||||
#if CV_RVV && CV_SIMD_SCALABLE_64F
|
||||
reduce_c_rvv::sum2_64f64fC1(srcmat, dstmat);
|
||||
#elif CV_NEON && (defined(__aarch64__) || defined(_M_ARM64))
|
||||
reduce_c_neon::sum2_64f64fC1(srcmat, dstmat);
|
||||
#elif CV_AVX2
|
||||
reduce_c_avx2::sum2_64f64fC1(srcmat, dstmat);
|
||||
#else
|
||||
reduceColSum2_64f64fFallback(srcmat, dstmat);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,989 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
namespace reduce_c_avx2
|
||||
{
|
||||
|
||||
// Optimized ReduceC support in this backend:
|
||||
//
|
||||
// | Input -> output type/channel | SUM | AVG | MIN | MAX | SUM2 |
|
||||
// |------------------------------------|:---:|:---:|:---:|:---:|:----:|
|
||||
// | 8UC1/C3/C4 -> 8UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 8UC1/C3/C4 -> 32SC1/C3/C4 | x | x | - | - | x |
|
||||
// | 8UC1/C3/C4 -> 32FC1/C3/C4 | x | x | - | - | x |
|
||||
// | 8UC1/C3/C4 -> 64FC1/C3/C4 | - | - | - | - | x |
|
||||
// | 16UC1/C3/C4 -> 16UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16UC1 -> 32FC1 | x | x | - | - | x |
|
||||
// | 16UC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16UC1 -> 64FC1 | - | - | - | - | x |
|
||||
// | 16UC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 16SC1/C3/C4 -> 16SC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16SC1 -> 32FC1 | x | x | - | - | x |
|
||||
// | 16SC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16SC1 -> 64FC1 | - | - | - | - | x |
|
||||
// | 16SC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 32FC1 -> 32FC1 | x | x | x | x | x |
|
||||
// | 32FC3 -> 32FC3 | x | x | - | - | - |
|
||||
// | 32FC4 -> 32FC4 | x | x | x | x | x |
|
||||
// | 32FC1 -> 64FC1 | x | x | - | - | x |
|
||||
// | 32FC3/C4 -> 64FC3/C4 | x | x | - | - | - |
|
||||
// | 64FC1 -> 64FC1 | x | x | x | x | x |
|
||||
// | 64FC3/C4 -> 64FC3/C4 | x | x | - | - | - |
|
||||
//
|
||||
// 'x' in SUM/AVG denotes the existing shared universal-intrinsics kernel; 'x'
|
||||
// in MIN/MAX/SUM2 denotes a native AVX2 kernel. For legal MIN/MAX/SUM2
|
||||
// combinations marked '-', and for other channel counts, dispatch uses the
|
||||
// shared generic fallback.
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
uchar result = isMax ? 0 : UCHAR_MAX;
|
||||
int x = 0;
|
||||
__m256i acc = _mm256_set1_epi8((char)result);
|
||||
for (; x <= cols - 32; x += 32)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
acc = isMax ? _mm256_max_epu8(acc, v) : _mm256_min_epu8(acc, v);
|
||||
}
|
||||
uchar lanes[32];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc);
|
||||
for (int i = 0; i < 32; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dst[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort result = isMax ? 0 : USHRT_MAX;
|
||||
int x = 0;
|
||||
__m256i acc = _mm256_set1_epi16((short)result);
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
acc = isMax ? _mm256_max_epu16(acc, v) : _mm256_min_epu16(acc, v);
|
||||
}
|
||||
ushort lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc);
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<ushort>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short result = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
int x = 0;
|
||||
__m256i acc = _mm256_set1_epi16(result);
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
acc = isMax ? _mm256_max_epi16(acc, v) : _mm256_min_epi16(acc, v);
|
||||
}
|
||||
short lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc);
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<short>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
const __m256i accInit = _mm256_set1_epi16((short)initial);
|
||||
const __m256i validMask = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(-1, -1, -1, -1, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0));
|
||||
const __m256i masks[4] = {
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(0, 1, 8, 9, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(2, 3, 10, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(4, 5, 12, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(6, 7, 14, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1))
|
||||
};
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort* dst = dstmat.ptr<ushort>(y);
|
||||
__m256i acc[4] = {accInit, accInit, accInit, accInit};
|
||||
int x = 0;
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 4));
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
__m256i vc = _mm256_shuffle_epi8(v, masks[c]);
|
||||
vc = _mm256_blendv_epi8(accInit, vc, validMask);
|
||||
acc[c] = isMax ? _mm256_max_epu16(acc[c], vc)
|
||||
: _mm256_min_epu16(acc[c], vc);
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
ushort lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc[c]);
|
||||
ushort result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 4 + c])
|
||||
: std::min(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
const __m256i accInit = _mm256_set1_epi16(initial);
|
||||
const __m256i validMask = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(-1, -1, -1, -1, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0));
|
||||
const __m256i masks[4] = {
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(0, 1, 8, 9, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(2, 3, 10, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(4, 5, 12, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)),
|
||||
_mm256_broadcastsi128_si256(_mm_setr_epi8(6, 7, 14, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1))
|
||||
};
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short* dst = dstmat.ptr<short>(y);
|
||||
__m256i acc[4] = {accInit, accInit, accInit, accInit};
|
||||
int x = 0;
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 4));
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
__m256i vc = _mm256_shuffle_epi8(v, masks[c]);
|
||||
vc = _mm256_blendv_epi8(accInit, vc, validMask);
|
||||
acc[c] = isMax ? _mm256_max_epi16(acc[c], vc)
|
||||
: _mm256_min_epi16(acc[c], vc);
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
short lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc[c]);
|
||||
short result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 4 + c])
|
||||
: std::min(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename T, bool isMax, bool isSigned>
|
||||
static void minMax16C3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const T initial = isMax ? std::numeric_limits<T>::lowest() : std::numeric_limits<T>::max();
|
||||
const __m256i accInit = _mm256_set1_epi16((short)initial);
|
||||
const __m256i masks[3] = {
|
||||
_mm256_setr_epi8(0, 1, 6, 7, 12, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
2, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1),
|
||||
_mm256_setr_epi8(2, 3, 8, 9, 14, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
4, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1),
|
||||
_mm256_setr_epi8(4, 5, 10, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 1, 6, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1)
|
||||
};
|
||||
const __m256i validMasks[3] = {
|
||||
_mm256_setr_epi8(-1, -1, -1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
-1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0),
|
||||
_mm256_setr_epi8(-1, -1, -1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
-1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0),
|
||||
_mm256_setr_epi8(-1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
-1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0)
|
||||
};
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const T* src = srcmat.ptr<T>(y);
|
||||
T* dst = dstmat.ptr<T>(y);
|
||||
__m256i acc[3] = {accInit, accInit, accInit};
|
||||
int x = 0;
|
||||
for (; x <= cols - 6; x += 4)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 3));
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
__m256i vc = _mm256_shuffle_epi8(v, masks[c]);
|
||||
vc = _mm256_blendv_epi8(accInit, vc, validMasks[c]);
|
||||
if (isSigned)
|
||||
acc[c] = isMax ? _mm256_max_epi16(acc[c], vc) : _mm256_min_epi16(acc[c], vc);
|
||||
else
|
||||
acc[c] = isMax ? _mm256_max_epu16(acc[c], vc) : _mm256_min_epu16(acc[c], vc);
|
||||
}
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
T lanes[16];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc[c]);
|
||||
T result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 3 + c])
|
||||
: std::min(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
minMax16C3<ushort, isMax, false>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
minMax16C3<short, isMax, true>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
const __m256i accInit = _mm256_set1_epi8((char)initial);
|
||||
const __m256i mask0 = _mm256_setr_epi8(
|
||||
0, 3, 6, 9, 12, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
2, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i mask1 = _mm256_setr_epi8(
|
||||
1, 4, 7, 10, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 3, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i mask2 = _mm256_setr_epi8(
|
||||
2, 5, 8, 11, 14, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
1, 4, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i invalid0 = _mm256_setr_epi8(
|
||||
0, 0, 0, 0, 0, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i invalid12 = _mm256_setr_epi8(
|
||||
0, 0, 0, 0, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 0, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
__m256i acc0 = accInit, acc1 = accInit, acc2 = accInit;
|
||||
int x = 0;
|
||||
for (; x <= cols - 11; x += 8)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 3));
|
||||
__m256i v0 = _mm256_shuffle_epi8(v, mask0);
|
||||
__m256i v1 = _mm256_shuffle_epi8(v, mask1);
|
||||
__m256i v2 = _mm256_shuffle_epi8(v, mask2);
|
||||
if (!isMax)
|
||||
{
|
||||
v0 = _mm256_or_si256(v0, invalid0);
|
||||
v1 = _mm256_or_si256(v1, invalid12);
|
||||
v2 = _mm256_or_si256(v2, invalid12);
|
||||
}
|
||||
acc0 = isMax ? _mm256_max_epu8(acc0, v0) : _mm256_min_epu8(acc0, v0);
|
||||
acc1 = isMax ? _mm256_max_epu8(acc1, v1) : _mm256_min_epu8(acc1, v1);
|
||||
acc2 = isMax ? _mm256_max_epu8(acc2, v2) : _mm256_min_epu8(acc2, v2);
|
||||
}
|
||||
|
||||
__m256i accs[3] = {acc0, acc1, acc2};
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
uchar lanes[32];
|
||||
_mm256_storeu_si256((__m256i*)lanes, accs[c]);
|
||||
uchar result = initial;
|
||||
for (int i = 0; i < 32; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
const __m256i accInit = _mm256_set1_epi8((char)initial);
|
||||
const __m256i maskInvalid = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(0, 0, 0, 0, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask0 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(0, 4, 8, 12, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask1 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(1, 5, 9, 13, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask2 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(2, 6, 10, 14, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask3 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(3, 7, 11, 15, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
__m256i acc0 = accInit, acc1 = accInit, acc2 = accInit, acc3 = accInit;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 4));
|
||||
__m256i v0 = _mm256_shuffle_epi8(v, mask0);
|
||||
__m256i v1 = _mm256_shuffle_epi8(v, mask1);
|
||||
__m256i v2 = _mm256_shuffle_epi8(v, mask2);
|
||||
__m256i v3 = _mm256_shuffle_epi8(v, mask3);
|
||||
if (!isMax)
|
||||
{
|
||||
v0 = _mm256_or_si256(v0, maskInvalid);
|
||||
v1 = _mm256_or_si256(v1, maskInvalid);
|
||||
v2 = _mm256_or_si256(v2, maskInvalid);
|
||||
v3 = _mm256_or_si256(v3, maskInvalid);
|
||||
}
|
||||
acc0 = isMax ? _mm256_max_epu8(acc0, v0) : _mm256_min_epu8(acc0, v0);
|
||||
acc1 = isMax ? _mm256_max_epu8(acc1, v1) : _mm256_min_epu8(acc1, v1);
|
||||
acc2 = isMax ? _mm256_max_epu8(acc2, v2) : _mm256_min_epu8(acc2, v2);
|
||||
acc3 = isMax ? _mm256_max_epu8(acc3, v3) : _mm256_min_epu8(acc3, v3);
|
||||
}
|
||||
|
||||
__m256i accs[4] = {acc0, acc1, acc2, acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
uchar lanes[32];
|
||||
_mm256_storeu_si256((__m256i*)lanes, accs[c]);
|
||||
uchar result = initial;
|
||||
for (int i = 0; i < 32; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = src[0];
|
||||
int x = 0;
|
||||
__m256 acc = _mm256_set1_ps(result);
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x);
|
||||
acc = isMax ? _mm256_max_ps(acc, v) : _mm256_min_ps(acc, v);
|
||||
}
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const float initial = isMax ? std::numeric_limits<float>::lowest() : std::numeric_limits<float>::max();
|
||||
const __m256 accInit = _mm256_set1_ps(initial);
|
||||
const __m256 validMask = _mm256_castsi256_ps(_mm256_setr_epi32(-1, -1, 0, 0, 0, 0, 0, 0));
|
||||
const __m256i idx0 = _mm256_setr_epi32(0, 4, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx1 = _mm256_setr_epi32(1, 5, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx2 = _mm256_setr_epi32(2, 6, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx3 = _mm256_setr_epi32(3, 7, 0, 0, 0, 0, 0, 0);
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
__m256 acc0 = accInit, acc1 = accInit, acc2 = accInit, acc3 = accInit;
|
||||
int x = 0;
|
||||
for (; x <= cols - 2; x += 2)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x * 4);
|
||||
__m256 v0 = _mm256_blendv_ps(accInit, _mm256_permutevar8x32_ps(v, idx0), validMask);
|
||||
__m256 v1 = _mm256_blendv_ps(accInit, _mm256_permutevar8x32_ps(v, idx1), validMask);
|
||||
__m256 v2 = _mm256_blendv_ps(accInit, _mm256_permutevar8x32_ps(v, idx2), validMask);
|
||||
__m256 v3 = _mm256_blendv_ps(accInit, _mm256_permutevar8x32_ps(v, idx3), validMask);
|
||||
acc0 = isMax ? _mm256_max_ps(acc0, v0) : _mm256_min_ps(acc0, v0);
|
||||
acc1 = isMax ? _mm256_max_ps(acc1, v1) : _mm256_min_ps(acc1, v1);
|
||||
acc2 = isMax ? _mm256_max_ps(acc2, v2) : _mm256_min_ps(acc2, v2);
|
||||
acc3 = isMax ? _mm256_max_ps(acc3, v3) : _mm256_min_ps(acc3, v3);
|
||||
}
|
||||
|
||||
__m256 accs[4] = {acc0, acc1, acc2, acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, accs[c]);
|
||||
float result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const double initial = isMax ? std::numeric_limits<double>::lowest()
|
||||
: std::numeric_limits<double>::max();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
double result = initial;
|
||||
int x = 0;
|
||||
__m256d acc0 = _mm256_set1_pd(initial);
|
||||
__m256d acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256d v0 = _mm256_loadu_pd(src + x);
|
||||
__m256d v1 = _mm256_loadu_pd(src + x + 4);
|
||||
__m256d v2 = _mm256_loadu_pd(src + x + 8);
|
||||
__m256d v3 = _mm256_loadu_pd(src + x + 12);
|
||||
acc0 = isMax ? _mm256_max_pd(acc0, v0) : _mm256_min_pd(acc0, v0);
|
||||
acc1 = isMax ? _mm256_max_pd(acc1, v1) : _mm256_min_pd(acc1, v1);
|
||||
acc2 = isMax ? _mm256_max_pd(acc2, v2) : _mm256_min_pd(acc2, v2);
|
||||
acc3 = isMax ? _mm256_max_pd(acc3, v3) : _mm256_min_pd(acc3, v3);
|
||||
}
|
||||
acc0 = isMax ? _mm256_max_pd(acc0, acc1) : _mm256_min_pd(acc0, acc1);
|
||||
acc2 = isMax ? _mm256_max_pd(acc2, acc3) : _mm256_min_pd(acc2, acc3);
|
||||
acc0 = isMax ? _mm256_max_pd(acc0, acc2) : _mm256_min_pd(acc0, acc2);
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t result = 0;
|
||||
int x = 0;
|
||||
__m256i acc = _mm256_setzero_si256();
|
||||
for (; x <= cols - 32; x += 32)
|
||||
{
|
||||
__m256i bytes = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m128i lo = _mm256_castsi256_si128(bytes);
|
||||
__m128i hi = _mm256_extracti128_si256(bytes, 1);
|
||||
__m256i lo16 = _mm256_cvtepu8_epi16(lo);
|
||||
__m256i hi16 = _mm256_cvtepu8_epi16(hi);
|
||||
acc = _mm256_add_epi32(acc, _mm256_madd_epi16(lo16, lo16));
|
||||
acc = _mm256_add_epi32(acc, _mm256_madd_epi16(hi16, hi16));
|
||||
}
|
||||
uint32_t lanes[8];
|
||||
_mm256_storeu_si256((__m256i*)lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
result += (uint32_t)src[x] * src[x];
|
||||
dst[0] = (DT)(int32_t)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
__m256 acc0 = _mm256_setzero_ps(), acc1 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m256 v0 = _mm256_cvtepi32_ps(_mm256_cvtepu16_epi32(_mm256_castsi256_si128(v)));
|
||||
__m256 v1 = _mm256_cvtepi32_ps(_mm256_cvtepu16_epi32(_mm256_extracti128_si256(v, 1)));
|
||||
acc0 = _mm256_add_ps(acc0, _mm256_mul_ps(v0, v0));
|
||||
acc1 = _mm256_add_ps(acc1, _mm256_mul_ps(v1, v1));
|
||||
}
|
||||
acc0 = _mm256_add_ps(acc0, acc1);
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, acc0);
|
||||
float result = 0;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
float value = (float)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
__m256 acc0 = _mm256_setzero_ps(), acc1 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m256 v0 = _mm256_cvtepi32_ps(_mm256_cvtepi16_epi32(_mm256_castsi256_si128(v)));
|
||||
__m256 v1 = _mm256_cvtepi32_ps(_mm256_cvtepi16_epi32(_mm256_extracti128_si256(v, 1)));
|
||||
acc0 = _mm256_add_ps(acc0, _mm256_mul_ps(v0, v0));
|
||||
acc1 = _mm256_add_ps(acc1, _mm256_mul_ps(v1, v1));
|
||||
}
|
||||
acc0 = _mm256_add_ps(acc0, acc1);
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, acc0);
|
||||
float result = 0;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
float value = (float)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
__m256d acc0 = _mm256_setzero_pd(), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m256i v0 = _mm256_cvtepu16_epi32(_mm256_castsi256_si128(v));
|
||||
__m256i v1 = _mm256_cvtepu16_epi32(_mm256_extracti128_si256(v, 1));
|
||||
__m256d d0 = _mm256_cvtepi32_pd(_mm256_castsi256_si128(v0));
|
||||
__m256d d1 = _mm256_cvtepi32_pd(_mm256_extracti128_si256(v0, 1));
|
||||
__m256d d2 = _mm256_cvtepi32_pd(_mm256_castsi256_si128(v1));
|
||||
__m256d d3 = _mm256_cvtepi32_pd(_mm256_extracti128_si256(v1, 1));
|
||||
acc0 = _mm256_add_pd(acc0, _mm256_mul_pd(d0, d0));
|
||||
acc1 = _mm256_add_pd(acc1, _mm256_mul_pd(d1, d1));
|
||||
acc2 = _mm256_add_pd(acc2, _mm256_mul_pd(d2, d2));
|
||||
acc3 = _mm256_add_pd(acc3, _mm256_mul_pd(d3, d3));
|
||||
}
|
||||
acc0 = _mm256_add_pd(_mm256_add_pd(acc0, acc1), _mm256_add_pd(acc2, acc3));
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1] + lanes[2] + lanes[3];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
double value = (double)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
__m256d acc0 = _mm256_setzero_pd(), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x));
|
||||
__m256i v0 = _mm256_cvtepi16_epi32(_mm256_castsi256_si128(v));
|
||||
__m256i v1 = _mm256_cvtepi16_epi32(_mm256_extracti128_si256(v, 1));
|
||||
__m256d d0 = _mm256_cvtepi32_pd(_mm256_castsi256_si128(v0));
|
||||
__m256d d1 = _mm256_cvtepi32_pd(_mm256_extracti128_si256(v0, 1));
|
||||
__m256d d2 = _mm256_cvtepi32_pd(_mm256_castsi256_si128(v1));
|
||||
__m256d d3 = _mm256_cvtepi32_pd(_mm256_extracti128_si256(v1, 1));
|
||||
acc0 = _mm256_add_pd(acc0, _mm256_mul_pd(d0, d0));
|
||||
acc1 = _mm256_add_pd(acc1, _mm256_mul_pd(d1, d1));
|
||||
acc2 = _mm256_add_pd(acc2, _mm256_mul_pd(d2, d2));
|
||||
acc3 = _mm256_add_pd(acc3, _mm256_mul_pd(d3, d3));
|
||||
}
|
||||
acc0 = _mm256_add_pd(_mm256_add_pd(acc0, acc1), _mm256_add_pd(acc2, acc3));
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1] + lanes[2] + lanes[3];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
double value = (double)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
__m256d acc0 = _mm256_setzero_pd(), acc1 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x);
|
||||
__m256d v0 = _mm256_cvtps_pd(_mm256_castps256_ps128(v));
|
||||
__m256d v1 = _mm256_cvtps_pd(_mm256_extractf128_ps(v, 1));
|
||||
acc0 = _mm256_add_pd(acc0, _mm256_mul_pd(v0, v0));
|
||||
acc1 = _mm256_add_pd(acc1, _mm256_mul_pd(v1, v1));
|
||||
}
|
||||
acc0 = _mm256_add_pd(acc0, acc1);
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1] + lanes[2] + lanes[3];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
double value = (double)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_64f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
__m256d acc0 = _mm256_setzero_pd(), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
__m256d v0 = _mm256_loadu_pd(src + x);
|
||||
__m256d v1 = _mm256_loadu_pd(src + x + 4);
|
||||
__m256d v2 = _mm256_loadu_pd(src + x + 8);
|
||||
__m256d v3 = _mm256_loadu_pd(src + x + 12);
|
||||
acc0 = _mm256_add_pd(acc0, _mm256_mul_pd(v0, v0));
|
||||
acc1 = _mm256_add_pd(acc1, _mm256_mul_pd(v1, v1));
|
||||
acc2 = _mm256_add_pd(acc2, _mm256_mul_pd(v2, v2));
|
||||
acc3 = _mm256_add_pd(acc3, _mm256_mul_pd(v3, v3));
|
||||
}
|
||||
acc0 = _mm256_add_pd(_mm256_add_pd(acc0, acc1), _mm256_add_pd(acc2, acc3));
|
||||
double lanes[4];
|
||||
_mm256_storeu_pd(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1] + lanes[2] + lanes[3];
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const __m256i mask0 = _mm256_setr_epi8(
|
||||
0, 3, 6, 9, 12, 15, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
2, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i mask1 = _mm256_setr_epi8(
|
||||
1, 4, 7, 10, 13, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
0, 3, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
const __m256i mask2 = _mm256_setr_epi8(
|
||||
2, 5, 8, 11, 14, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1,
|
||||
1, 4, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
__m256i acc0 = _mm256_setzero_si256();
|
||||
__m256i acc1 = _mm256_setzero_si256();
|
||||
__m256i acc2 = _mm256_setzero_si256();
|
||||
int x = 0;
|
||||
for (; x <= cols - 11; x += 8)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 3));
|
||||
__m256i channels[3] = {
|
||||
_mm256_shuffle_epi8(v, mask0),
|
||||
_mm256_shuffle_epi8(v, mask1),
|
||||
_mm256_shuffle_epi8(v, mask2)
|
||||
};
|
||||
__m256i* accs[3] = {&acc0, &acc1, &acc2};
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
__m128i lo = _mm256_castsi256_si128(channels[c]);
|
||||
__m128i hi = _mm256_extracti128_si256(channels[c], 1);
|
||||
__m256i lo16 = _mm256_cvtepu8_epi16(lo);
|
||||
__m256i hi16 = _mm256_cvtepu8_epi16(hi);
|
||||
*accs[c] = _mm256_add_epi32(*accs[c], _mm256_madd_epi16(lo16, lo16));
|
||||
*accs[c] = _mm256_add_epi32(*accs[c], _mm256_madd_epi16(hi16, hi16));
|
||||
}
|
||||
}
|
||||
|
||||
__m256i accs[3] = {acc0, acc1, acc2};
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
uint32_t lanes[8];
|
||||
uint32_t result = 0;
|
||||
_mm256_storeu_si256((__m256i*)lanes, accs[c]);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (int i = x; i < cols; i++)
|
||||
{
|
||||
uint32_t value = src[i * 3 + c];
|
||||
result += value * value;
|
||||
}
|
||||
dst[c] = (DT)(int32_t)result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const __m256i mask0 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(0, 4, 8, 12, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask1 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(1, 5, 9, 13, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask2 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(2, 6, 10, 14, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
const __m256i mask3 = _mm256_broadcastsi128_si256(
|
||||
_mm_setr_epi8(3, 7, 11, 15, -1, -1, -1, -1,
|
||||
-1, -1, -1, -1, -1, -1, -1, -1));
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
__m256i acc0 = _mm256_setzero_si256();
|
||||
__m256i acc1 = _mm256_setzero_si256();
|
||||
__m256i acc2 = _mm256_setzero_si256();
|
||||
__m256i acc3 = _mm256_setzero_si256();
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256i v = _mm256_loadu_si256((const __m256i*)(src + x * 4));
|
||||
__m256i channels[4] = {
|
||||
_mm256_shuffle_epi8(v, mask0),
|
||||
_mm256_shuffle_epi8(v, mask1),
|
||||
_mm256_shuffle_epi8(v, mask2),
|
||||
_mm256_shuffle_epi8(v, mask3)
|
||||
};
|
||||
__m256i* accs[4] = {&acc0, &acc1, &acc2, &acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
__m128i lo = _mm256_castsi256_si128(channels[c]);
|
||||
__m128i hi = _mm256_extracti128_si256(channels[c], 1);
|
||||
__m256i lo16 = _mm256_cvtepu8_epi16(lo);
|
||||
__m256i hi16 = _mm256_cvtepu8_epi16(hi);
|
||||
*accs[c] = _mm256_add_epi32(*accs[c], _mm256_madd_epi16(lo16, lo16));
|
||||
*accs[c] = _mm256_add_epi32(*accs[c], _mm256_madd_epi16(hi16, hi16));
|
||||
}
|
||||
}
|
||||
|
||||
__m256i accs[4] = {acc0, acc1, acc2, acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
uint32_t lanes[8];
|
||||
uint32_t result = 0;
|
||||
_mm256_storeu_si256((__m256i*)lanes, accs[c]);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (int i = x; i < cols; i++)
|
||||
{
|
||||
uint32_t value = src[i * 4 + c];
|
||||
result += value * value;
|
||||
}
|
||||
dst[c] = (DT)(int32_t)result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = 0;
|
||||
int x = 0;
|
||||
__m256 acc = _mm256_setzero_ps();
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x);
|
||||
acc = _mm256_add_ps(acc, _mm256_mul_ps(v, v));
|
||||
}
|
||||
float lanes[8];
|
||||
_mm256_storeu_ps(lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const __m256 zero = _mm256_setzero_ps();
|
||||
const __m256 validMask = _mm256_castsi256_ps(_mm256_setr_epi32(-1, -1, 0, 0, 0, 0, 0, 0));
|
||||
const __m256i idx0 = _mm256_setr_epi32(0, 4, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx1 = _mm256_setr_epi32(1, 5, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx2 = _mm256_setr_epi32(2, 6, 0, 0, 0, 0, 0, 0);
|
||||
const __m256i idx3 = _mm256_setr_epi32(3, 7, 0, 0, 0, 0, 0, 0);
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
__m256 acc0 = zero, acc1 = zero, acc2 = zero, acc3 = zero;
|
||||
int x = 0;
|
||||
for (; x <= cols - 2; x += 2)
|
||||
{
|
||||
__m256 v = _mm256_loadu_ps(src + x * 4);
|
||||
__m256 v0 = _mm256_blendv_ps(zero, _mm256_permutevar8x32_ps(v, idx0), validMask);
|
||||
__m256 v1 = _mm256_blendv_ps(zero, _mm256_permutevar8x32_ps(v, idx1), validMask);
|
||||
__m256 v2 = _mm256_blendv_ps(zero, _mm256_permutevar8x32_ps(v, idx2), validMask);
|
||||
__m256 v3 = _mm256_blendv_ps(zero, _mm256_permutevar8x32_ps(v, idx3), validMask);
|
||||
acc0 = _mm256_add_ps(acc0, _mm256_mul_ps(v0, v0));
|
||||
acc1 = _mm256_add_ps(acc1, _mm256_mul_ps(v1, v1));
|
||||
acc2 = _mm256_add_ps(acc2, _mm256_mul_ps(v2, v2));
|
||||
acc3 = _mm256_add_ps(acc3, _mm256_mul_ps(v3, v3));
|
||||
}
|
||||
|
||||
__m256 accs[4] = {acc0, acc1, acc2, acc3};
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
float lanes[8];
|
||||
float result = 0;
|
||||
_mm256_storeu_ps(lanes, accs[c]);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result += lanes[i];
|
||||
for (int i = x; i < cols; i++)
|
||||
result += src[i * 4 + c] * src[i * 4 + c];
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
} // namespace reduce_c_avx2
|
||||
@@ -0,0 +1,701 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
template<typename T, typename VT>
|
||||
static inline VT vx_load_strided(const T *ptr, size_t step)
|
||||
{
|
||||
constexpr int nlanes = VTraits<VT>::max_nlanes;
|
||||
T buf[nlanes];
|
||||
for (int i = 0; i < VTraits<VT>::vlanes(); i++)
|
||||
{
|
||||
buf[i] = *ptr;
|
||||
ptr += step;
|
||||
}
|
||||
return vx_load(buf);
|
||||
}
|
||||
|
||||
#if CV_RVV
|
||||
template<>
|
||||
inline v_uint8 vx_load_strided(const uchar *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse8_v_u8m2(ptr, step * sizeof(uchar), __riscv_vsetvlmax_e8m2());
|
||||
}
|
||||
template<>
|
||||
inline v_uint16 vx_load_strided(const ushort *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse16_v_u16m2(ptr, step * sizeof(ushort), __riscv_vsetvlmax_e16m2());
|
||||
}
|
||||
template<>
|
||||
inline v_int16 vx_load_strided(const short *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse16_v_i16m2(ptr, step * sizeof(short), __riscv_vsetvlmax_e16m2());
|
||||
}
|
||||
template<>
|
||||
inline v_int32 vx_load_strided(const int *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse32_v_i32m2(ptr, step * sizeof(int), __riscv_vsetvlmax_e32m2());
|
||||
}
|
||||
template<>
|
||||
inline v_float32 vx_load_strided(const float *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse32_v_f32m2(ptr, step * sizeof(float), __riscv_vsetvlmax_e32m2());
|
||||
}
|
||||
template<>
|
||||
inline v_float64 vx_load_strided(const double *ptr, size_t step)
|
||||
{
|
||||
return __riscv_vlse64_v_f64m2(ptr, step * sizeof(double), __riscv_vsetvlmax_e64m2());
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
template<typename stype, typename itype>
|
||||
struct ReduceOpAddSqr
|
||||
{
|
||||
using v_stype = stype;
|
||||
using v_itype = itype;
|
||||
static const int vlanes;
|
||||
static inline stype load(const stype *ptr, size_t step) { (void)step; return *ptr; }
|
||||
static inline itype init() { return (itype)0; }
|
||||
static inline itype reduce(const itype &val) { return val; }
|
||||
inline itype operator()(const itype &a, const stype &b) const { return a + (itype)b * (itype)b; }
|
||||
};
|
||||
template<typename stype, typename itype>
|
||||
const int ReduceOpAddSqr<stype, itype>::vlanes = 1;
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
template<typename stype, typename itype>
|
||||
struct ReduceVecOpAddSqr;
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<uchar, int>
|
||||
{
|
||||
using stype = uchar;
|
||||
using itype = int;
|
||||
using v_stype = v_uint8;
|
||||
using v_itype = v_int32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_s32(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_uint16 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
|
||||
v_uint32 s00, s01;
|
||||
v_expand(b0, s00, s01);
|
||||
s00 = v_add(s00, s01);
|
||||
v_uint32 s10, s11;
|
||||
v_expand(b1, s10, s11);
|
||||
s10 = v_add(s10, s11);
|
||||
return v_add(a, v_reinterpret_as_s32(v_add(s00, s10)));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<uchar, int>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<ushort, float>
|
||||
{
|
||||
using stype = ushort;
|
||||
using itype = float;
|
||||
using v_stype = v_uint16;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f32(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_uint32 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
v_int32 sb = v_reinterpret_as_s32(v_add(b0, b1));
|
||||
return v_add(a, v_cvt_f32(sb));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<ushort, float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<short, float>
|
||||
{
|
||||
using stype = short;
|
||||
using itype = float;
|
||||
using v_stype = v_int16;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f32(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_int32 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
v_int32 sb = v_add(b0, b1);
|
||||
return v_add(a, v_cvt_f32(sb));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<short, float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<float, float>
|
||||
{
|
||||
using stype = float;
|
||||
using itype = float;
|
||||
using v_stype = v_float32;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f32(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_add(a, v_mul(b, b)); }
|
||||
};
|
||||
const int ReduceVecOpAddSqr<float, float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<short, double>
|
||||
{
|
||||
using stype = short;
|
||||
using itype = double;
|
||||
using v_stype = v_int16;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f64(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_int32 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
v_int32 sb = v_add(b0, b1);
|
||||
return v_add(a, v_add(v_cvt_f64(sb), v_cvt_f64_high(sb)));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<short, double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<ushort, double>
|
||||
{
|
||||
using stype = ushort;
|
||||
using itype = double;
|
||||
using v_stype = v_uint16;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f64(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_uint32 b0, b1;
|
||||
v_mul_expand(b, b, b0, b1);
|
||||
v_int32 sb = v_reinterpret_as_s32(v_add(b0, b1));
|
||||
return v_add(a, v_add(v_cvt_f64(sb), v_cvt_f64_high(sb)));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<ushort, double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<float, double>
|
||||
{
|
||||
using stype = float;
|
||||
using itype = double;
|
||||
using v_stype = v_float32;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f64(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const
|
||||
{
|
||||
v_itype b0 = v_cvt_f64(b), b1 = v_cvt_f64_high(b);
|
||||
return v_add(a, v_add(v_mul(b0, b0), v_mul(b1, b1)));
|
||||
}
|
||||
};
|
||||
const int ReduceVecOpAddSqr<float, double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpAddSqr<double, double>
|
||||
{
|
||||
using stype = double;
|
||||
using itype = double;
|
||||
using v_stype = v_float64;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setzero_f64(); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_sum(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_add(a, v_mul(b, b)); }
|
||||
};
|
||||
const int ReduceVecOpAddSqr<double, double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
template<typename stype>
|
||||
struct ReduceOpMax
|
||||
{
|
||||
using v_stype = stype;
|
||||
using v_itype = stype;
|
||||
static const int vlanes;
|
||||
static inline stype load(const stype *ptr, size_t step) { (void)step; return *ptr; }
|
||||
static inline stype init() { return std::numeric_limits<stype>::lowest(); }
|
||||
static inline stype reduce(const stype &val) { return val; }
|
||||
inline stype operator()(const stype &a, const stype &b) const { return std::max(a, b); }
|
||||
};
|
||||
template<typename stype>
|
||||
const int ReduceOpMax<stype>::vlanes = 1;
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
template<typename stype>
|
||||
struct ReduceVecOpMax;
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<uchar>
|
||||
{
|
||||
using stype = uchar;
|
||||
using itype = uchar;
|
||||
using v_stype = v_uint8;
|
||||
using v_itype = v_uint8;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_u8(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_max(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<uchar>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<ushort>
|
||||
{
|
||||
using stype = ushort;
|
||||
using itype = ushort;
|
||||
using v_stype = v_uint16;
|
||||
using v_itype = v_uint16;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_u16(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_max(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<ushort>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<short>
|
||||
{
|
||||
using stype = short;
|
||||
using itype = short;
|
||||
using v_stype = v_int16;
|
||||
using v_itype = v_int16;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_s16(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_max(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<short>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<float>
|
||||
{
|
||||
using stype = float;
|
||||
using itype = float;
|
||||
using v_stype = v_float32;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_f32(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_max(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMax<double>
|
||||
{
|
||||
using stype = double;
|
||||
using itype = double;
|
||||
using v_stype = v_float64;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_f64(std::numeric_limits<itype>::lowest()); }
|
||||
static inline itype reduce(const v_itype &val)
|
||||
{
|
||||
constexpr int nlanes = VTraits<v_itype>::max_nlanes;
|
||||
itype buf[nlanes];
|
||||
vx_store(buf, val);
|
||||
itype m = buf[0];
|
||||
for (int i = 1; i < VTraits<v_itype>::vlanes(); i++)
|
||||
{
|
||||
if (m < buf[i]) m = buf[i];
|
||||
}
|
||||
return m;
|
||||
}
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_max(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMax<double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
template<typename stype>
|
||||
struct ReduceOpMin
|
||||
{
|
||||
using v_stype = stype;
|
||||
using v_itype = stype;
|
||||
static const int vlanes;
|
||||
static inline stype load(const stype *ptr, size_t step) { (void)step; return *ptr; }
|
||||
static inline stype init() { return std::numeric_limits<stype>::max(); }
|
||||
static inline stype reduce(const stype &val) { return (stype)val; }
|
||||
inline stype operator()(const stype &a, const stype &b) const { return std::min(a, b); }
|
||||
};
|
||||
template<typename stype>
|
||||
const int ReduceOpMin<stype>::vlanes = 1;
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
template<typename stype>
|
||||
struct ReduceVecOpMin;
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<uchar>
|
||||
{
|
||||
using stype = uchar;
|
||||
using itype = uchar;
|
||||
using v_stype = v_uint8;
|
||||
using v_itype = v_uint8;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_u8(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_min(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<uchar>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<ushort>
|
||||
{
|
||||
using stype = ushort;
|
||||
using itype = ushort;
|
||||
using v_stype = v_uint16;
|
||||
using v_itype = v_uint16;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_u16(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_min(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<ushort>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<short>
|
||||
{
|
||||
using stype = short;
|
||||
using itype = short;
|
||||
using v_stype = v_int16;
|
||||
using v_itype = v_int16;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_s16(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_min(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<short>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<float>
|
||||
{
|
||||
using stype = float;
|
||||
using itype = float;
|
||||
using v_stype = v_float32;
|
||||
using v_itype = v_float32;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_f32(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val) { return v_reduce_min(val); }
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<float>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
template<>
|
||||
struct ReduceVecOpMin<double>
|
||||
{
|
||||
using stype = double;
|
||||
using itype = double;
|
||||
using v_stype = v_float64;
|
||||
using v_itype = v_float64;
|
||||
static const int vlanes;
|
||||
static inline v_stype load(const stype *ptr, size_t step) { return vx_load_strided<stype, v_stype>(ptr, step); }
|
||||
static inline v_itype init() { return vx_setall_f64(std::numeric_limits<itype>::max()); }
|
||||
static inline itype reduce(const v_itype &val)
|
||||
{
|
||||
constexpr int nlanes = VTraits<v_itype>::max_nlanes;
|
||||
itype buf[nlanes];
|
||||
vx_store(buf, val);
|
||||
itype m = buf[0];
|
||||
for (int i = 1; i < VTraits<v_itype>::vlanes(); i++)
|
||||
{
|
||||
if (m > buf[i]) m = buf[i];
|
||||
}
|
||||
return m;
|
||||
}
|
||||
inline v_itype operator()(const v_itype &a, const v_stype &b) const { return v_min(a, b); }
|
||||
};
|
||||
const int ReduceVecOpMin<double>::vlanes = VTraits<v_stype>::vlanes();
|
||||
|
||||
#endif
|
||||
|
||||
|
||||
using ReduceOpAddSqr_8U32S = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceOpAddSqr_8U32F = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceOpAddSqr_8U64F = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceOpAddSqr_16U32F = ReduceOpAddSqr<ushort, float>;
|
||||
using ReduceOpAddSqr_16U64F = ReduceOpAddSqr<ushort, double>;
|
||||
using ReduceOpAddSqr_16S32F = ReduceOpAddSqr<short, float>;
|
||||
using ReduceOpAddSqr_16S64F = ReduceOpAddSqr<short, double>;
|
||||
using ReduceOpAddSqr_32F32F = ReduceOpAddSqr<float, float>;
|
||||
using ReduceOpAddSqr_32F64F = ReduceOpAddSqr<float, double>;
|
||||
using ReduceOpAddSqr_64F64F = ReduceOpAddSqr<double, double>;
|
||||
|
||||
using ReduceOpMax_8U = ReduceOpMax<uchar>;
|
||||
using ReduceOpMax_16U = ReduceOpMax<ushort>;
|
||||
using ReduceOpMax_16S = ReduceOpMax<short>;
|
||||
using ReduceOpMax_32F = ReduceOpMax<float>;
|
||||
using ReduceOpMax_64F = ReduceOpMax<double>;
|
||||
|
||||
using ReduceOpMin_8U = ReduceOpMin<uchar>;
|
||||
using ReduceOpMin_16U = ReduceOpMin<ushort>;
|
||||
using ReduceOpMin_16S = ReduceOpMin<short>;
|
||||
using ReduceOpMin_32F = ReduceOpMin<float>;
|
||||
using ReduceOpMin_64F = ReduceOpMin<double>;
|
||||
|
||||
#if (CV_SIMD || CV_SIMD_SCALABLE)
|
||||
|
||||
using ReduceVecOpAddSqr_8U32S = ReduceVecOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_8U32F = ReduceVecOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_8U64F = ReduceVecOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_16U32F = ReduceVecOpAddSqr<ushort, float>;
|
||||
using ReduceVecOpAddSqr_16S32F = ReduceVecOpAddSqr<short, float>;
|
||||
using ReduceVecOpAddSqr_32F32F = ReduceVecOpAddSqr<float, float>;
|
||||
using ReduceVecOpMax_8U = ReduceVecOpMax<uchar>;
|
||||
using ReduceVecOpMax_16U = ReduceVecOpMax<ushort>;
|
||||
using ReduceVecOpMax_16S = ReduceVecOpMax<short>;
|
||||
using ReduceVecOpMax_32F = ReduceVecOpMax<float>;
|
||||
using ReduceVecOpMin_8U = ReduceVecOpMin<uchar>;
|
||||
using ReduceVecOpMin_16U = ReduceVecOpMin<ushort>;
|
||||
using ReduceVecOpMin_16S = ReduceVecOpMin<short>;
|
||||
using ReduceVecOpMin_32F = ReduceVecOpMin<float>;
|
||||
|
||||
#else
|
||||
|
||||
using ReduceVecOpAddSqr_8U32S = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_8U32F = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_8U64F = ReduceOpAddSqr<uchar, int>;
|
||||
using ReduceVecOpAddSqr_16U32F = ReduceOpAddSqr<ushort, float>;
|
||||
using ReduceVecOpAddSqr_16S32F = ReduceOpAddSqr<short, float>;
|
||||
using ReduceVecOpAddSqr_32F32F = ReduceOpAddSqr<float, float>;
|
||||
using ReduceVecOpMax_8U = ReduceOpMax<uchar>;
|
||||
using ReduceVecOpMax_16U = ReduceOpMax<ushort>;
|
||||
using ReduceVecOpMax_16S = ReduceOpMax<short>;
|
||||
using ReduceVecOpMax_32F = ReduceOpMax<float>;
|
||||
using ReduceVecOpMin_8U = ReduceOpMin<uchar>;
|
||||
using ReduceVecOpMin_16U = ReduceOpMin<ushort>;
|
||||
using ReduceVecOpMin_16S = ReduceOpMin<short>;
|
||||
using ReduceVecOpMin_32F = ReduceOpMin<float>;
|
||||
|
||||
#endif
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
|
||||
using ReduceVecOpAddSqr_16U64F = ReduceVecOpAddSqr<ushort, double>;
|
||||
using ReduceVecOpAddSqr_16S64F = ReduceVecOpAddSqr<short, double>;
|
||||
using ReduceVecOpAddSqr_32F64F = ReduceVecOpAddSqr<float, double>;
|
||||
using ReduceVecOpAddSqr_64F64F = ReduceVecOpAddSqr<double, double>;
|
||||
using ReduceVecOpMax_64F = ReduceVecOpMax<double>;
|
||||
using ReduceVecOpMin_64F = ReduceVecOpMin<double>;
|
||||
|
||||
#else
|
||||
|
||||
using ReduceVecOpAddSqr_16U64F = ReduceOpAddSqr<ushort, double>;
|
||||
using ReduceVecOpAddSqr_16S64F = ReduceOpAddSqr<short, double>;
|
||||
using ReduceVecOpAddSqr_32F64F = ReduceOpAddSqr<float, double>;
|
||||
using ReduceVecOpAddSqr_64F64F = ReduceOpAddSqr<double, double>;
|
||||
using ReduceVecOpMax_64F = ReduceOpMax<double>;
|
||||
using ReduceVecOpMin_64F = ReduceOpMin<double>;
|
||||
|
||||
#endif
|
||||
|
||||
template<typename T, typename ST, class Op, class VecOp>
|
||||
class ReduceC_Invoker : public ParallelLoopBody
|
||||
{
|
||||
using WT = typename Op::v_itype;
|
||||
using VT = typename VecOp::v_itype;
|
||||
public:
|
||||
ReduceC_Invoker(const Mat& aSrcmat, Mat& aDstmat, Op& aOp, VecOp& aVop)
|
||||
:srcmat(aSrcmat),dstmat(aDstmat),op(aOp),vop(aVop)
|
||||
{
|
||||
}
|
||||
void operator()(const Range& range) const CV_OVERRIDE
|
||||
{
|
||||
int channels = srcmat.channels();
|
||||
int width = srcmat.cols;
|
||||
|
||||
const int nlanes = VecOp::vlanes;
|
||||
|
||||
for (int h = range.start; h < range.end; h++)
|
||||
{
|
||||
const T *srcrow = srcmat.ptr<T>(h);
|
||||
ST *dst = dstmat.ptr<ST>(h);
|
||||
for (int cn = 0; cn < channels; cn++)
|
||||
{
|
||||
const T *src = srcrow + cn;
|
||||
VT vbuf = vop.init();
|
||||
int w = 0;
|
||||
for (; w <= width - nlanes; w += nlanes)
|
||||
{
|
||||
vbuf = vop(vbuf, vop.load(src+w*channels, channels));
|
||||
}
|
||||
WT wbuf = vop.reduce(vbuf);
|
||||
for (; w < width; w++)
|
||||
{
|
||||
wbuf = op(wbuf, op.load(src+w*channels, channels));
|
||||
}
|
||||
dst[cn] = (ST)op.reduce(wbuf);
|
||||
}
|
||||
}
|
||||
}
|
||||
private:
|
||||
const Mat& srcmat;
|
||||
Mat& dstmat;
|
||||
Op& op;
|
||||
VecOp& vop;
|
||||
};
|
||||
|
||||
template<typename T, typename ST, class Op, class VecOp> static void
|
||||
reduceColGeneric(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
Op op;
|
||||
VecOp vop;
|
||||
|
||||
ReduceC_Invoker<T, ST, Op, VecOp> body(srcmat, dstmat, op, vop);
|
||||
parallel_for_(Range(0, srcmat.size().height), body);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static inline uchar reduceScalarMinMax(uchar a, uchar b)
|
||||
{
|
||||
return isMax ? std::max(a, b) : std::min(a, b);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_8uFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<uchar, uchar, ReduceOpMax_8U, ReduceVecOpMax_8U>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<uchar, uchar, ReduceOpMin_8U, ReduceVecOpMin_8U>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16uFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<ushort, ushort, ReduceOpMax_16U, ReduceVecOpMax_16U>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<ushort, ushort, ReduceOpMin_16U, ReduceVecOpMin_16U>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_16sFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<short, short, ReduceOpMax_16S, ReduceVecOpMax_16S>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<short, short, ReduceOpMin_16S, ReduceVecOpMin_16S>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static inline float reduceScalarMinMax(float a, float b)
|
||||
{
|
||||
return isMax ? std::max(a, b) : std::min(a, b);
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_32fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<float, float, ReduceOpMax_32F, ReduceVecOpMax_32F>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<float, float, ReduceOpMin_32F, ReduceVecOpMin_32F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
template<bool isMax>
|
||||
static void reduceColMinMax_64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (isMax)
|
||||
reduceColGeneric<double, double, ReduceOpMax_64F, ReduceVecOpMax_64F>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<double, double, ReduceOpMin_64F, ReduceVecOpMin_64F>(srcmat, dstmat);
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename DT>
|
||||
static void reduceColSum2_8uFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
if (std::is_same<DT, int>::value)
|
||||
reduceColGeneric<uchar, int, ReduceOpAddSqr_8U32S, ReduceVecOpAddSqr_8U32S>(srcmat, dstmat);
|
||||
else if (std::is_same<DT, float>::value)
|
||||
reduceColGeneric<uchar, float, ReduceOpAddSqr_8U32F, ReduceVecOpAddSqr_8U32F>(srcmat, dstmat);
|
||||
else
|
||||
reduceColGeneric<uchar, double, ReduceOpAddSqr_8U64F, ReduceVecOpAddSqr_8U64F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_16u32fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<ushort, float, ReduceOpAddSqr_16U32F, ReduceVecOpAddSqr_16U32F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_16s32fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<short, float, ReduceOpAddSqr_16S32F, ReduceVecOpAddSqr_16S32F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f32fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<float, float, ReduceOpAddSqr_32F32F, ReduceVecOpAddSqr_32F32F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
#if (CV_SIMD_64F || CV_SIMD_SCALABLE_64F)
|
||||
static void reduceColSum2_16u64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<ushort, double, ReduceOpAddSqr_16U64F, ReduceVecOpAddSqr_16U64F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_16s64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<short, double, ReduceOpAddSqr_16S64F, ReduceVecOpAddSqr_16S64F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_32f64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<float, double, ReduceOpAddSqr_32F64F, ReduceVecOpAddSqr_32F64F>(srcmat, dstmat);
|
||||
}
|
||||
|
||||
static void reduceColSum2_64f64fFallback(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
reduceColGeneric<double, double, ReduceOpAddSqr_64F64F, ReduceVecOpAddSqr_64F64F>(srcmat, dstmat);
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,827 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
namespace reduce_c_neon
|
||||
{
|
||||
|
||||
// Optimized ReduceC support in this backend:
|
||||
//
|
||||
// | Input -> output type/channel | SUM | AVG | MIN | MAX | SUM2 |
|
||||
// |------------------------------------|:---:|:---:|:---:|:---:|:----:|
|
||||
// | 8UC1/C3/C4 -> 8UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 8UC1/C3/C4 -> 32SC1/C3/C4 | x | x | - | - | x* |
|
||||
// | 8UC1/C3/C4 -> 32FC1/C3/C4 | x | x | - | - | x* |
|
||||
// | 8UC1/C3/C4 -> 64FC1/C3/C4 | - | - | - | - | x* |
|
||||
// | 16UC1/C3/C4 -> 16UC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16UC1 -> 32FC1 | x | x | - | - | x* |
|
||||
// | 16UC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16UC1 -> 64FC1 | - | - | - | - | x* |
|
||||
// | 16UC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 16SC1/C3/C4 -> 16SC1/C3/C4 | - | - | x | x | - |
|
||||
// | 16SC1 -> 32FC1 | x | x | - | - | x* |
|
||||
// | 16SC3/C4 -> 32FC3/C4 | x | x | - | - | - |
|
||||
// | 16SC1 -> 64FC1 | - | - | - | - | x* |
|
||||
// | 16SC3/C4 -> 64FC3/C4 | - | - | - | - | - |
|
||||
// | 32FC1 -> 32FC1 | x | x | x | x | x |
|
||||
// | 32FC3 -> 32FC3 | x | x | - | - | - |
|
||||
// | 32FC4 -> 32FC4 | x | x | x | x | x |
|
||||
// | 32FC1 -> 64FC1 | x* | x* | - | - | x* |
|
||||
// | 32FC3/C4 -> 64FC3/C4 | x* | x* | - | - | - |
|
||||
// | 64FC1 -> 64FC1 | x* | x* | x* | x* | x* |
|
||||
// | 64FC3/C4 -> 64FC3/C4 | x* | x* | - | - | - |
|
||||
//
|
||||
// 'x' in SUM/AVG denotes the existing shared universal-intrinsics kernel; 'x'
|
||||
// in MIN/MAX/SUM2 denotes a native NEON kernel. '*' requires AArch64. For legal
|
||||
// MIN/MAX/SUM2 combinations marked '-', and for other channel counts, dispatch
|
||||
// uses the shared generic fallback.
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
uchar result = isMax ? 0 : UCHAR_MAX;
|
||||
int x = 0;
|
||||
uint8x16_t acc = vdupq_n_u8(result);
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16_t v = vld1q_u8(src + x);
|
||||
acc = isMax ? vmaxq_u8(acc, v) : vminq_u8(acc, v);
|
||||
}
|
||||
uchar lanes[16];
|
||||
vst1q_u8(lanes, acc);
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dst[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort result = isMax ? 0 : USHRT_MAX;
|
||||
int x = 0;
|
||||
uint16x8_t acc = vdupq_n_u16(result);
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
uint16x8_t v = vld1q_u16(src + x);
|
||||
acc = isMax ? vmaxq_u16(acc, v) : vminq_u16(acc, v);
|
||||
}
|
||||
ushort lanes[8];
|
||||
vst1q_u16(lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<ushort>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short result = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
int x = 0;
|
||||
int16x8_t acc = vdupq_n_s16(result);
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
int16x8_t v = vld1q_s16(src + x);
|
||||
acc = isMax ? vmaxq_s16(acc, v) : vminq_s16(acc, v);
|
||||
}
|
||||
short lanes[8];
|
||||
vst1q_s16(lanes, acc);
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<short>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort* dst = dstmat.ptr<ushort>(y);
|
||||
uint16x8x4_t acc = {{
|
||||
vdupq_n_u16(initial), vdupq_n_u16(initial),
|
||||
vdupq_n_u16(initial), vdupq_n_u16(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
uint16x8x4_t v = vld4q_u16(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = isMax ? vmaxq_u16(acc.val[c], v.val[c])
|
||||
: vminq_u16(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
ushort lanes[8];
|
||||
vst1q_u16(lanes, acc.val[c]);
|
||||
ushort result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 4 + c])
|
||||
: std::min(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short* dst = dstmat.ptr<short>(y);
|
||||
int16x8x4_t acc = {{
|
||||
vdupq_n_s16(initial), vdupq_n_s16(initial),
|
||||
vdupq_n_s16(initial), vdupq_n_s16(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
int16x8x4_t v = vld4q_s16(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = isMax ? vmaxq_s16(acc.val[c], v.val[c])
|
||||
: vminq_s16(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
short lanes[8];
|
||||
vst1q_s16(lanes, acc.val[c]);
|
||||
short result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 4 + c])
|
||||
: std::min(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const ushort initial = isMax ? 0 : USHRT_MAX;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
ushort* dst = dstmat.ptr<ushort>(y);
|
||||
uint16x8x3_t acc = {{vdupq_n_u16(initial), vdupq_n_u16(initial), vdupq_n_u16(initial)}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
uint16x8x3_t v = vld3q_u16(src + x * 3);
|
||||
for (int c = 0; c < 3; c++)
|
||||
acc.val[c] = isMax ? vmaxq_u16(acc.val[c], v.val[c])
|
||||
: vminq_u16(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
ushort lanes[8];
|
||||
vst1q_u16(lanes, acc.val[c]);
|
||||
ushort result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 3 + c])
|
||||
: std::min(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax16sC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const short initial = isMax ? SHRT_MIN : SHRT_MAX;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
short* dst = dstmat.ptr<short>(y);
|
||||
int16x8x3_t acc = {{vdupq_n_s16(initial), vdupq_n_s16(initial), vdupq_n_s16(initial)}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
int16x8x3_t v = vld3q_s16(src + x * 3);
|
||||
for (int c = 0; c < 3; c++)
|
||||
acc.val[c] = isMax ? vmaxq_s16(acc.val[c], v.val[c])
|
||||
: vminq_s16(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
short lanes[8];
|
||||
vst1q_s16(lanes, acc.val[c]);
|
||||
short result = initial;
|
||||
for (int i = 0; i < 8; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = isMax ? std::max(result, src[i * 3 + c])
|
||||
: std::min(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
uint8x16x3_t acc = {{
|
||||
vdupq_n_u8(initial), vdupq_n_u8(initial), vdupq_n_u8(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16x3_t v = vld3q_u8(src + x * 3);
|
||||
for (int c = 0; c < 3; c++)
|
||||
acc.val[c] = isMax ? vmaxq_u8(acc.val[c], v.val[c])
|
||||
: vminq_u8(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
uchar lanes[16];
|
||||
vst1q_u8(lanes, acc.val[c]);
|
||||
uchar result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 3 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
uchar* dst = dstmat.ptr<uchar>(y);
|
||||
const uchar initial = isMax ? 0 : UCHAR_MAX;
|
||||
uint8x16x4_t acc = {{
|
||||
vdupq_n_u8(initial), vdupq_n_u8(initial),
|
||||
vdupq_n_u8(initial), vdupq_n_u8(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16x4_t v = vld4q_u8(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = isMax ? vmaxq_u8(acc.val[c], v.val[c])
|
||||
: vminq_u8(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
uchar lanes[16];
|
||||
vst1q_u8(lanes, acc.val[c]);
|
||||
uchar result = initial;
|
||||
for (int i = 0; i < 16; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = src[0];
|
||||
int x = 0;
|
||||
float32x4_t acc = vdupq_n_f32(result);
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
float32x4_t v = vld1q_f32(src + x);
|
||||
acc = isMax ? vmaxq_f32(acc, v) : vminq_f32(acc, v);
|
||||
}
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[x]);
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<bool isMax>
|
||||
static void minMax32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const float initial = isMax ? std::numeric_limits<float>::lowest() : std::numeric_limits<float>::max();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
float32x4x4_t acc = {{
|
||||
vdupq_n_f32(initial), vdupq_n_f32(initial),
|
||||
vdupq_n_f32(initial), vdupq_n_f32(initial)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
float32x4x4_t v = vld4q_f32(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = isMax ? vmaxq_f32(acc.val[c], v.val[c])
|
||||
: vminq_f32(acc.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc.val[c]);
|
||||
float result = initial;
|
||||
for (int i = 0; i < 4; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, lanes[i]);
|
||||
for (int i = x; i < cols; i++)
|
||||
result = reduceScalarMinMax<isMax>(result, src[i * 4 + c]);
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
template<bool isMax>
|
||||
static void minMax64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
const double initial = isMax ? std::numeric_limits<double>::lowest()
|
||||
: std::numeric_limits<double>::max();
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
double result = initial;
|
||||
int x = 0;
|
||||
float64x2_t acc0 = vdupq_n_f64(initial);
|
||||
float64x2_t acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
float64x2_t v0 = vld1q_f64(src + x);
|
||||
float64x2_t v1 = vld1q_f64(src + x + 2);
|
||||
float64x2_t v2 = vld1q_f64(src + x + 4);
|
||||
float64x2_t v3 = vld1q_f64(src + x + 6);
|
||||
acc0 = isMax ? vmaxq_f64(acc0, v0) : vminq_f64(acc0, v0);
|
||||
acc1 = isMax ? vmaxq_f64(acc1, v1) : vminq_f64(acc1, v1);
|
||||
acc2 = isMax ? vmaxq_f64(acc2, v2) : vminq_f64(acc2, v2);
|
||||
acc3 = isMax ? vmaxq_f64(acc3, v3) : vminq_f64(acc3, v3);
|
||||
}
|
||||
acc0 = isMax ? vmaxq_f64(acc0, acc1) : vminq_f64(acc0, acc1);
|
||||
acc2 = isMax ? vmaxq_f64(acc2, acc3) : vminq_f64(acc2, acc3);
|
||||
acc0 = isMax ? vmaxq_f64(acc0, acc2) : vminq_f64(acc0, acc2);
|
||||
double lanes[2];
|
||||
vst1q_f64(lanes, acc0);
|
||||
for (int i = 0; i < 2; i++)
|
||||
result = isMax ? std::max(result, lanes[i]) : std::min(result, lanes[i]);
|
||||
for (; x < cols; x++)
|
||||
result = isMax ? std::max(result, src[x]) : std::min(result, src[x]);
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(__aarch64__) || defined(_M_ARM64)
|
||||
static inline uint32_t reduceSum2_8u_NEON(uint8x16_t v)
|
||||
{
|
||||
uint16x8_t lo = vmull_u8(vget_low_u8(v), vget_low_u8(v));
|
||||
uint16x8_t hi = vmull_u8(vget_high_u8(v), vget_high_u8(v));
|
||||
return vaddvq_u32(vpaddlq_u16(lo)) + vaddvq_u32(vpaddlq_u16(hi));
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t result = 0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
result += reduceSum2_8u_NEON(vld1q_u8(src + x));
|
||||
for (; x < cols; x++)
|
||||
result += (uint32_t)src[x] * src[x];
|
||||
dst[0] = (DT)(int32_t)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
float result = 0;
|
||||
float32x4_t acc0 = vdupq_n_f32(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint16x8_t v01 = vld1q_u16(src + x);
|
||||
uint16x8_t v23 = vld1q_u16(src + x + 8);
|
||||
float32x4_t v0 = vcvtq_f32_u32(vmovl_u16(vget_low_u16(v01)));
|
||||
float32x4_t v1 = vcvtq_f32_u32(vmovl_u16(vget_high_u16(v01)));
|
||||
float32x4_t v2 = vcvtq_f32_u32(vmovl_u16(vget_low_u16(v23)));
|
||||
float32x4_t v3 = vcvtq_f32_u32(vmovl_u16(vget_high_u16(v23)));
|
||||
acc0 = vmlaq_f32(acc0, v0, v0);
|
||||
acc1 = vmlaq_f32(acc1, v1, v1);
|
||||
acc2 = vmlaq_f32(acc2, v2, v2);
|
||||
acc3 = vmlaq_f32(acc3, v3, v3);
|
||||
}
|
||||
acc0 = vaddq_f32(vaddq_f32(acc0, acc1), vaddq_f32(acc2, acc3));
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc0);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
float value = (float)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
float result = 0;
|
||||
float32x4_t acc0 = vdupq_n_f32(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
int16x8_t v01 = vld1q_s16(src + x);
|
||||
int16x8_t v23 = vld1q_s16(src + x + 8);
|
||||
float32x4_t v0 = vcvtq_f32_s32(vmovl_s16(vget_low_s16(v01)));
|
||||
float32x4_t v1 = vcvtq_f32_s32(vmovl_s16(vget_high_s16(v01)));
|
||||
float32x4_t v2 = vcvtq_f32_s32(vmovl_s16(vget_low_s16(v23)));
|
||||
float32x4_t v3 = vcvtq_f32_s32(vmovl_s16(vget_high_s16(v23)));
|
||||
acc0 = vmlaq_f32(acc0, v0, v0);
|
||||
acc1 = vmlaq_f32(acc1, v1, v1);
|
||||
acc2 = vmlaq_f32(acc2, v2, v2);
|
||||
acc3 = vmlaq_f32(acc3, v3, v3);
|
||||
}
|
||||
acc0 = vaddq_f32(vaddq_f32(acc0, acc1), vaddq_f32(acc2, acc3));
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc0);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
float value = (float)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16u64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const ushort* src = srcmat.ptr<ushort>(y);
|
||||
uint64x2_t acc0 = vdupq_n_u64(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
uint16x8_t v = vld1q_u16(src + x);
|
||||
uint32x4_t sq0 = vmull_u16(vget_low_u16(v), vget_low_u16(v));
|
||||
uint32x4_t sq1 = vmull_u16(vget_high_u16(v), vget_high_u16(v));
|
||||
acc0 = vaddq_u64(acc0, vmovl_u32(vget_low_u32(sq0)));
|
||||
acc1 = vaddq_u64(acc1, vmovl_high_u32(sq0));
|
||||
acc2 = vaddq_u64(acc2, vmovl_u32(vget_low_u32(sq1)));
|
||||
acc3 = vaddq_u64(acc3, vmovl_high_u32(sq1));
|
||||
}
|
||||
acc0 = vaddq_u64(vaddq_u64(acc0, acc1), vaddq_u64(acc2, acc3));
|
||||
uint64_t lanes[2];
|
||||
vst1q_u64(lanes, acc0);
|
||||
uint64_t result = lanes[0] + lanes[1];
|
||||
for (; x < cols; x++)
|
||||
result += (uint64_t)src[x] * src[x];
|
||||
dstmat.ptr<double>(y)[0] = (double)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_16s64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const short* src = srcmat.ptr<short>(y);
|
||||
uint64x2_t acc0 = vdupq_n_u64(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
int16x8_t v = vld1q_s16(src + x);
|
||||
int32x4_t sq0 = vmull_s16(vget_low_s16(v), vget_low_s16(v));
|
||||
int32x4_t sq1 = vmull_s16(vget_high_s16(v), vget_high_s16(v));
|
||||
uint32x4_t usq0 = vreinterpretq_u32_s32(sq0);
|
||||
uint32x4_t usq1 = vreinterpretq_u32_s32(sq1);
|
||||
acc0 = vaddq_u64(acc0, vmovl_u32(vget_low_u32(usq0)));
|
||||
acc1 = vaddq_u64(acc1, vmovl_high_u32(usq0));
|
||||
acc2 = vaddq_u64(acc2, vmovl_u32(vget_low_u32(usq1)));
|
||||
acc3 = vaddq_u64(acc3, vmovl_high_u32(usq1));
|
||||
}
|
||||
acc0 = vaddq_u64(vaddq_u64(acc0, acc1), vaddq_u64(acc2, acc3));
|
||||
uint64_t lanes[2];
|
||||
vst1q_u64(lanes, acc0);
|
||||
uint64_t result = lanes[0] + lanes[1];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
int64_t value = src[x];
|
||||
result += (uint64_t)(value * value);
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = (double)result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float64x2_t acc0 = vdupq_n_f64(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
float32x4_t v01 = vld1q_f32(src + x);
|
||||
float32x4_t v23 = vld1q_f32(src + x + 4);
|
||||
float64x2_t v0 = vcvt_f64_f32(vget_low_f32(v01));
|
||||
float64x2_t v1 = vcvt_high_f64_f32(v01);
|
||||
float64x2_t v2 = vcvt_f64_f32(vget_low_f32(v23));
|
||||
float64x2_t v3 = vcvt_high_f64_f32(v23);
|
||||
acc0 = vmlaq_f64(acc0, v0, v0);
|
||||
acc1 = vmlaq_f64(acc1, v1, v1);
|
||||
acc2 = vmlaq_f64(acc2, v2, v2);
|
||||
acc3 = vmlaq_f64(acc3, v3, v3);
|
||||
}
|
||||
acc0 = vaddq_f64(vaddq_f64(acc0, acc1), vaddq_f64(acc2, acc3));
|
||||
double lanes[2];
|
||||
vst1q_f64(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1];
|
||||
for (; x < cols; x++)
|
||||
{
|
||||
double value = (double)src[x];
|
||||
result += value * value;
|
||||
}
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_64f64fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const double* src = srcmat.ptr<double>(y);
|
||||
float64x2_t acc0 = vdupq_n_f64(0), acc1 = acc0, acc2 = acc0, acc3 = acc0;
|
||||
int x = 0;
|
||||
for (; x <= cols - 8; x += 8)
|
||||
{
|
||||
float64x2_t v0 = vld1q_f64(src + x);
|
||||
float64x2_t v1 = vld1q_f64(src + x + 2);
|
||||
float64x2_t v2 = vld1q_f64(src + x + 4);
|
||||
float64x2_t v3 = vld1q_f64(src + x + 6);
|
||||
acc0 = vmlaq_f64(acc0, v0, v0);
|
||||
acc1 = vmlaq_f64(acc1, v1, v1);
|
||||
acc2 = vmlaq_f64(acc2, v2, v2);
|
||||
acc3 = vmlaq_f64(acc3, v3, v3);
|
||||
}
|
||||
acc0 = vaddq_f64(vaddq_f64(acc0, acc1), vaddq_f64(acc2, acc3));
|
||||
double lanes[2];
|
||||
vst1q_f64(lanes, acc0);
|
||||
double result = lanes[0] + lanes[1];
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<double>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC3(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t results[3] = {0, 0, 0};
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16x3_t v = vld3q_u8(src + x * 3);
|
||||
for (int c = 0; c < 3; c++)
|
||||
results[c] += reduceSum2_8u_NEON(v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 3; c++)
|
||||
{
|
||||
for (int i = x; i < cols; i++)
|
||||
{
|
||||
uint32_t value = src[i * 3 + c];
|
||||
results[c] += value * value;
|
||||
}
|
||||
dst[c] = (DT)(int32_t)results[c];
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
template<typename DT>
|
||||
static void sum2_8uC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const uchar* src = srcmat.ptr<uchar>(y);
|
||||
DT* dst = dstmat.ptr<DT>(y);
|
||||
uint32_t results[4] = {0, 0, 0, 0};
|
||||
int x = 0;
|
||||
for (; x <= cols - 16; x += 16)
|
||||
{
|
||||
uint8x16x4_t v = vld4q_u8(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
results[c] += reduceSum2_8u_NEON(v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
for (int i = x; i < cols; i++)
|
||||
{
|
||||
uint32_t value = src[i * 4 + c];
|
||||
results[c] += value * value;
|
||||
}
|
||||
dst[c] = (DT)(int32_t)results[c];
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
static void sum2_32fC1(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float result = 0;
|
||||
int x = 0;
|
||||
float32x4_t acc = vdupq_n_f32(0);
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
float32x4_t v = vld1q_f32(src + x);
|
||||
acc = vaddq_f32(acc, vmulq_f32(v, v));
|
||||
}
|
||||
float lanes[4];
|
||||
vst1q_f32(lanes, acc);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result += lanes[i];
|
||||
for (; x < cols; x++)
|
||||
result += src[x] * src[x];
|
||||
dstmat.ptr<float>(y)[0] = result;
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
static void sum2_32fC4(const Mat& srcmat, Mat& dstmat)
|
||||
{
|
||||
const int cols = srcmat.cols;
|
||||
parallel_for_(Range(0, srcmat.rows), [&](const Range& range) {
|
||||
for (int y = range.start; y < range.end; y++)
|
||||
{
|
||||
const float* src = srcmat.ptr<float>(y);
|
||||
float* dst = dstmat.ptr<float>(y);
|
||||
float32x4x4_t acc = {{
|
||||
vdupq_n_f32(0), vdupq_n_f32(0),
|
||||
vdupq_n_f32(0), vdupq_n_f32(0)
|
||||
}};
|
||||
int x = 0;
|
||||
for (; x <= cols - 4; x += 4)
|
||||
{
|
||||
float32x4x4_t v = vld4q_f32(src + x * 4);
|
||||
for (int c = 0; c < 4; c++)
|
||||
acc.val[c] = vmlaq_f32(acc.val[c], v.val[c], v.val[c]);
|
||||
}
|
||||
for (int c = 0; c < 4; c++)
|
||||
{
|
||||
float lanes[4];
|
||||
float result = 0;
|
||||
vst1q_f32(lanes, acc.val[c]);
|
||||
for (int i = 0; i < 4; i++)
|
||||
result += lanes[i];
|
||||
for (int i = x; i < cols; i++)
|
||||
result += src[i * 4 + c] * src[i * 4 + c];
|
||||
dst[c] = result;
|
||||
}
|
||||
}
|
||||
});
|
||||
v_cleanup();
|
||||
}
|
||||
|
||||
} // namespace reduce_c_neon
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user