Compare commits
224 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| dc15187f1b | |||
| ae1a249a0a | |||
| 171fbf879f | |||
| 8a1caa0d16 | |||
| cdcf7e62f3 | |||
| 083727b0a6 | |||
| 718dd9f170 | |||
| 6c218c7b0f | |||
| e87a0baa4b | |||
| 8ee5f2ad89 | |||
| ee76cef1ff | |||
| 6da05f7086 | |||
| e24b1629a0 | |||
| d9ea9bedb2 | |||
| 1546b9bf99 | |||
| 62b5d37b6b | |||
| d78d9bf151 | |||
| a5f0fb6008 | |||
| f2bb3d0d80 | |||
| 037a72debd | |||
| 1ef4b7ae5a | |||
| b314cc4c23 | |||
| 644de8f22a | |||
| 36f61f3879 | |||
| aa11f7d8a3 | |||
| f30aafc3cc | |||
| 2f065b8b4c | |||
| aece3e732e | |||
| ebe9327a92 | |||
| 102d8f67cd | |||
| a00fe15abd | |||
| 3acf8cfd63 | |||
| d34717d8c9 | |||
| a5b8f163d7 | |||
| 6add3b9161 | |||
| 1ddd61f98a | |||
| 3503a36e5e | |||
| 2ea7269450 | |||
| a968eadbf1 | |||
| c6b63e0e28 | |||
| 7bd8ddc8fa | |||
| 14e264f646 | |||
| 969b55036f | |||
| cbf978d1f7 | |||
| 94e0ac7d9f | |||
| 08b36e4f48 | |||
| d62d880316 | |||
| 361b5e0ebf | |||
| c16e2e6234 | |||
| c08f29c803 | |||
| ecb6d03ccd | |||
| e59793cc75 | |||
| 236ad4aeda | |||
| 97bb91d5fa | |||
| 233030e417 | |||
| 6256e425f3 | |||
| 48368dc9a1 | |||
| 691c655630 | |||
| b88ad7f2d9 | |||
| b300b6b3bd | |||
| 234117800f | |||
| ac58b2f857 | |||
| 220b37144b | |||
| 74c8ccb45b | |||
| 9cfe981e1f | |||
| 2f9072efdc | |||
| 5e90802b1a | |||
| 45fee13f3d | |||
| 261ad78122 | |||
| 4fa82809df | |||
| 3e3787ecb6 | |||
| a723aaedd8 | |||
| ebb528976f | |||
| f77c2d700f | |||
| 8c44b8306b | |||
| e668cff573 | |||
| fa953e4205 | |||
| 540982cc9d | |||
| a61546680b | |||
| 8cb7eae5c9 | |||
| 18440c1faf | |||
| 9f69ca503a | |||
| 49ba744130 | |||
| 4b24ddd70d | |||
| 1612db5f91 | |||
| 7dfe68cac6 | |||
| 83807811cd | |||
| 23e71d1aa2 | |||
| b542a1804c | |||
| 2dff2f36bf | |||
| 6b674709b8 | |||
| f56445d7ca | |||
| 03bee14372 | |||
| 50ff40d684 | |||
| c31164bf1e | |||
| 3835ab394e | |||
| 20b23da8e2 | |||
| f6795d75a6 | |||
| fa11b98800 | |||
| 64c67a93d3 | |||
| 154380ccf5 | |||
| 1f2c83845d | |||
| 2349a09736 | |||
| 1df533c914 | |||
| 44bf748479 | |||
| f9fbd29c14 | |||
| 31dc3e9256 | |||
| 698b2bf729 | |||
| 1d42560018 | |||
| efcf307b4c | |||
| 6876f3b91d | |||
| 7495a4722f | |||
| 7ce56b3a47 | |||
| 5376863c0c | |||
| a86e036594 | |||
| 01324b02e7 | |||
| 792722865f | |||
| 9aa401a7d0 | |||
| a160e4fb6b | |||
| c7422e4d90 | |||
| 485d2b593c | |||
| 1bea537731 | |||
| 75e88d613f | |||
| 379b83e946 | |||
| dda1bf1887 | |||
| f7e524cbe6 | |||
| e503ac508e | |||
| f5ba3f51ce | |||
| d392b11dfb | |||
| cd01ee9a54 | |||
| 8c1af09989 | |||
| f53ff0d01c | |||
| 4f48dab023 | |||
| a7ffcaab28 | |||
| f617f18e46 | |||
| 4372d75b26 | |||
| 885fb703cf | |||
| cd00d8f3f0 | |||
| b3755e617c | |||
| fc0f9da7a7 | |||
| 277961fa25 | |||
| 14a9103fc0 | |||
| 6d1f7c2b1b | |||
| 2c1f3487a4 | |||
| acc6189da0 | |||
| ac177b849c | |||
| 29aeebf5bc | |||
| cc769ff19d | |||
| 1834eed809 | |||
| b34234ac14 | |||
| 7ec9f52509 | |||
| 68f527267b | |||
| 7f22b34c78 | |||
| ad63d24dba | |||
| 3b5813c035 | |||
| 339b963e6b | |||
| 00890aecdf | |||
| 2171cae8ff | |||
| 3f55152ca0 | |||
| f3cebb3e1b | |||
| 2b227f00f2 | |||
| 98de57c6c4 | |||
| e231be86b7 | |||
| 7ec221e734 | |||
| fe9ff64d64 | |||
| 3f65c12d0c | |||
| 336627a776 | |||
| 6226ea0085 | |||
| 1067cd0649 | |||
| 1537ecd931 | |||
| 5b5c42d2c7 | |||
| 161890dad4 | |||
| 35846fe735 | |||
| 96ce65f021 | |||
| 8457e471fd | |||
| 246de2b7f5 | |||
| e8c26963e9 | |||
| 793e7c0d9f | |||
| 3b337a12c9 | |||
| 2c32cb743c | |||
| 883b995fd6 | |||
| a28533933f | |||
| d695208727 | |||
| 613ff61de7 | |||
| 65b02cc8f2 | |||
| 1c8ee3f957 | |||
| 922108060d | |||
| ce74285c5e | |||
| c262eea84a | |||
| f2ca7e664a | |||
| cf8f65d806 | |||
| bc221bdb90 | |||
| 6bed5c181b | |||
| f162c08cda | |||
| 0d1d452b79 | |||
| 3c3e131c38 | |||
| a218cf3c61 | |||
| 866468cc3e | |||
| f66fc199a2 | |||
| 11ac26bfb4 | |||
| 2241bfb0df | |||
| d92af2aa85 | |||
| e4d573a080 | |||
| f86c8656a3 | |||
| 6697f0ea86 | |||
| e935f06f16 | |||
| 1e18004bdd | |||
| 19646ad049 | |||
| 7e943808b6 | |||
| 88dbee3589 | |||
| 5bfa43f7d5 | |||
| 7ed37b3fa5 | |||
| f0271e54d9 | |||
| 0ac2f0e04c | |||
| ef1690ef45 | |||
| 0fa06b1db0 | |||
| abceef74e0 | |||
| 5444a6b11c | |||
| 9411cd6c07 | |||
| a35d4f9029 | |||
| 284d26da05 | |||
| 17c430da88 | |||
| 96a501c08b | |||
| 05fbd1e5bc |
@@ -40,4 +40,5 @@ if(WITH_NEON)
|
||||
target_compile_definitions(carotene_objs PRIVATE "-DWITH_NEON")
|
||||
endif()
|
||||
|
||||
add_library(carotene STATIC EXCLUDE_FROM_ALL "$<TARGET_OBJECTS:carotene_objs>")
|
||||
# we add dummy file to fix XCode build
|
||||
add_library(carotene STATIC EXCLUDE_FROM_ALL "$<TARGET_OBJECTS:carotene_objs>" "${CAROTENE_SOURCE_DIR}/dummy.cpp")
|
||||
|
||||
@@ -80,7 +80,8 @@ set_property(DIRECTORY APPEND PROPERTY COMPILE_DEFINITIONS ${carotene_defs})
|
||||
# set_source_files_properties(impl.cpp $<TARGET_OBJECTS:carotene_objs> COMPILE_FLAGS "--param ipcp-unit-growth=100000 --param inline-unit-growth=100000 --param large-stack-frame-growth=5000")
|
||||
endif()
|
||||
|
||||
add_library(tegra_hal STATIC $<TARGET_OBJECTS:carotene_objs>)
|
||||
# we add dummy file to fix XCode build
|
||||
add_library(tegra_hal STATIC $<TARGET_OBJECTS:carotene_objs> "dummy.cpp")
|
||||
set_target_properties(tegra_hal PROPERTIES ARCHIVE_OUTPUT_DIRECTORY ${3P_LIBRARY_OUTPUT_PATH})
|
||||
set(OPENCV_SRC_DIR "${CMAKE_SOURCE_DIR}")
|
||||
if(NOT BUILD_SHARED_LIBS)
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
// This file is needed for compilation on some platforms e.g. with XCode generator
|
||||
// Related issue: https://gitlab.kitware.com/cmake/cmake/-/issues/17457
|
||||
@@ -0,0 +1,2 @@
|
||||
// This file is needed for compilation on some platforms e.g. with XCode generator
|
||||
// Related issue: https://gitlab.kitware.com/cmake/cmake/-/issues/17457
|
||||
@@ -1,8 +1,8 @@
|
||||
# Binaries branch name: ffmpeg/3.4_20200608
|
||||
# Binaries were created for OpenCV: 458f1d5ebe31e22789d9d781d0ca2ca936758fde
|
||||
ocv_update(FFMPEG_BINARIES_COMMIT "57064cd66d98994503b34aade3c8d8ff25007b46")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "6fff20f5617bd1b7362058790db52caa")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "15df55131471191b575668a424dff385")
|
||||
# Binaries branch name: ffmpeg/3.4_20200907
|
||||
# Binaries were created for OpenCV: 03bee14372f5537daa56c62e771ec16181ca1f98
|
||||
ocv_update(FFMPEG_BINARIES_COMMIT "2a96257b743695a47f8012aab1ffb995a1dee8b4")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "5e68a3ff82f43ac6524e50e448a34c9c")
|
||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "205db629d893e7d4865fd1459807ff47")
|
||||
ocv_update(FFMPEG_FILE_HASH_CMAKE "3b90f67f4b429e77d3da36698cef700c")
|
||||
|
||||
function(download_win_ffmpeg script_var)
|
||||
|
||||
@@ -463,6 +463,7 @@ OCV_OPTION(BUILD_JAVA "Enable Java support"
|
||||
# OpenCV installation options
|
||||
# ===================================================
|
||||
OCV_OPTION(INSTALL_CREATE_DISTRIB "Change install rules to build the distribution package" OFF )
|
||||
OCV_OPTION(INSTALL_BIN_EXAMPLES "Install prebuilt examples" WIN32 IF BUILD_EXAMPLES)
|
||||
OCV_OPTION(INSTALL_C_EXAMPLES "Install C examples" OFF )
|
||||
OCV_OPTION(INSTALL_PYTHON_EXAMPLES "Install Python examples" OFF )
|
||||
OCV_OPTION(INSTALL_ANDROID_EXAMPLES "Install Android examples" OFF IF ANDROID )
|
||||
|
||||
@@ -78,14 +78,19 @@ if(CUDA_FOUND)
|
||||
|
||||
message(STATUS "CUDA detected: " ${CUDA_VERSION})
|
||||
|
||||
set(_generations "Fermi" "Kepler" "Maxwell" "Pascal" "Volta" "Turing" "Ampere")
|
||||
OCV_OPTION(CUDA_ENABLE_DEPRECATED_GENERATION "Enable deprecated generations in the list" OFF)
|
||||
set(_generations "Maxwell" "Pascal" "Volta" "Turing" "Ampere")
|
||||
if(CUDA_ENABLE_DEPRECATED_GENERATION)
|
||||
set(_generations "Fermi" "${_generations}")
|
||||
set(_generations "Kepler" "${_generations}")
|
||||
endif()
|
||||
set(_arch_fermi "2.0")
|
||||
set(_arch_kepler "3.0;3.5;3.7")
|
||||
set(_arch_maxwell "5.0;5.2")
|
||||
set(_arch_pascal "6.0;6.1")
|
||||
set(_arch_volta "7.0")
|
||||
set(_arch_turing "7.5")
|
||||
set(_arch_ampere "8.0")
|
||||
set(_arch_ampere "8.0;8.6")
|
||||
if(NOT CMAKE_CROSSCOMPILING)
|
||||
list(APPEND _generations "Auto")
|
||||
endif()
|
||||
@@ -193,16 +198,12 @@ if(CUDA_FOUND)
|
||||
|
||||
if(${status} EQUAL 0)
|
||||
# cache detected values
|
||||
set(OPENCV_CACHE_CUDA_ACTIVE_CC ${${result_list}} CACHE INTERNAL "")
|
||||
set(OPENCV_CACHE_CUDA_ACTIVE_CC ${${output}} CACHE INTERNAL "")
|
||||
set(OPENCV_CACHE_CUDA_ACTIVE_CC_check "${__cache_key_check}" CACHE INTERNAL "")
|
||||
endif()
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
macro(ocv_wipeout_deprecated _arch_bin_list)
|
||||
string(REPLACE "2.1" "2.1(2.0)" ${_arch_bin_list} "${${_arch_bin_list}}")
|
||||
endmacro()
|
||||
|
||||
set(__cuda_arch_ptx "")
|
||||
if(CUDA_GENERATION STREQUAL "Fermi")
|
||||
set(__cuda_arch_bin ${_arch_fermi})
|
||||
@@ -265,7 +266,6 @@ if(CUDA_FOUND)
|
||||
)
|
||||
endif()
|
||||
endif()
|
||||
ocv_wipeout_deprecated(__cuda_arch_bin)
|
||||
|
||||
set(CUDA_ARCH_BIN ${__cuda_arch_bin} CACHE STRING "Specify 'real' GPU architectures to build binaries for, BIN(PTX) format is supported")
|
||||
set(CUDA_ARCH_PTX ${__cuda_arch_ptx} CACHE STRING "Specify 'virtual' PTX architectures to build PTX intermediate code for")
|
||||
@@ -273,10 +273,14 @@ if(CUDA_FOUND)
|
||||
string(REGEX REPLACE "\\." "" ARCH_BIN_NO_POINTS "${CUDA_ARCH_BIN}")
|
||||
string(REGEX REPLACE "\\." "" ARCH_PTX_NO_POINTS "${CUDA_ARCH_PTX}")
|
||||
|
||||
# Check if user specified 1.0 compute capability: we don't support it
|
||||
if(" ${CUDA_ARCH_BIN} ${CUDA_ARCH_PTX}" MATCHES " 1.0")
|
||||
message(SEND_ERROR "CUDA: 1.0 compute capability is not supported - exclude it from ARCH/PTX list are re-run CMake")
|
||||
endif()
|
||||
# Check if user specified 1.0/2.1 compute capability: we don't support it
|
||||
macro(ocv_wipeout_deprecated_cc target_cc)
|
||||
if(" ${CUDA_ARCH_BIN} ${CUDA_ARCH_PTX}" MATCHES " ${target_cc}")
|
||||
message(SEND_ERROR "CUDA: ${target_cc} compute capability is not supported - exclude it from ARCH/PTX list and re-run CMake")
|
||||
endif()
|
||||
endmacro()
|
||||
ocv_wipeout_deprecated_cc("1.0")
|
||||
ocv_wipeout_deprecated_cc("2.1")
|
||||
|
||||
# NVCC flags to be set
|
||||
set(NVCC_FLAGS_EXTRA "")
|
||||
|
||||
@@ -252,6 +252,7 @@ if(NOT DEFINED IPPROOT)
|
||||
else()
|
||||
ocv_install_3rdparty_licenses(ippicv "${ICV_PACKAGE_ROOT}/EULA.txt")
|
||||
endif()
|
||||
ocv_install_3rdparty_licenses(ippicv "${ICV_PACKAGE_ROOT}/third-party-programs.txt")
|
||||
endif()
|
||||
|
||||
file(TO_CMAKE_PATH "${IPPROOT}" __IPPROOT)
|
||||
|
||||
@@ -1337,8 +1337,8 @@ function(ocv_add_samples)
|
||||
endif()
|
||||
add_dependencies(${parent_target} ${the_target})
|
||||
|
||||
if(WIN32)
|
||||
install(TARGETS ${the_target} RUNTIME DESTINATION "samples/${module_id}" COMPONENT samples)
|
||||
if(INSTALL_BIN_EXAMPLES)
|
||||
install(TARGETS ${the_target} RUNTIME DESTINATION "${OPENCV_SAMPLES_BIN_INSTALL_PATH}/${module_id}" COMPONENT samples)
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
|
Before Width: | Height: | Size: 4.4 KiB After Width: | Height: | Size: 5.2 KiB |
@@ -9,6 +9,9 @@ MathJax.Hub.Config(
|
||||
forkfour: ["\\left\\{ \\begin{array}{l l} #1 & \\mbox{#2}\\\\ #3 & \\mbox{#4}\\\\ #5 & \\mbox{#6}\\\\ #7 & \\mbox{#8}\\\\ \\end{array} \\right.", 8],
|
||||
vecthree: ["\\begin{bmatrix} #1\\\\ #2\\\\ #3 \\end{bmatrix}", 3],
|
||||
vecthreethree: ["\\begin{bmatrix} #1 & #2 & #3\\\\ #4 & #5 & #6\\\\ #7 & #8 & #9 \\end{bmatrix}", 9],
|
||||
cameramatrix: ["#1 = \\begin{bmatrix} f_x & 0 & c_x\\\\ 0 & f_y & c_y\\\\ 0 & 0 & 1 \\end{bmatrix}", 1],
|
||||
distcoeffs: ["(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \\tau_x, \\tau_y]]]]) \\text{ of 4, 5, 8, 12 or 14 elements}"],
|
||||
distcoeffsfisheye: ["(k_1, k_2, k_3, k_4)"],
|
||||
hdotsfor: ["\\dots", 1],
|
||||
mathbbm: ["\\mathbb{#1}", 1],
|
||||
bordermatrix: ["\\matrix{#1}", 1]
|
||||
|
||||
@@ -51,3 +51,20 @@
|
||||
#7 & #8 & #9
|
||||
\end{bmatrix}
|
||||
}
|
||||
|
||||
\newcommand{\cameramatrix}[1]{
|
||||
#1 =
|
||||
\begin{bmatrix}
|
||||
f_x & 0 & c_x\\
|
||||
0 & f_y & c_y\\
|
||||
0 & 0 & 1
|
||||
\end{bmatrix}
|
||||
}
|
||||
|
||||
\newcommand{\distcoeffs}[]{
|
||||
(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]]) \text{ of 4, 5, 8, 12 or 14 elements}
|
||||
}
|
||||
|
||||
\newcommand{\distcoeffsfisheye}[]{
|
||||
(k_1, k_2, k_3, k_4)
|
||||
}
|
||||
|
||||
|
Before Width: | Height: | Size: 1.4 KiB After Width: | Height: | Size: 2.1 KiB |
|
Before Width: | Height: | Size: 7.9 KiB After Width: | Height: | Size: 9.5 KiB |
@@ -0,0 +1,9 @@
|
||||
OpenCV logo has been originally designed and contributed to OpenCV by Adi Shavit in 2006. The graphical part consists of three stylized letters O, C, V, colored in the primary R, G, B color components, used by humans and computers to perceive the world. It is shaped in a way to mimic the famous [Kanizsa's triangle](https://en.wikipedia.org/wiki/Illusory_contours) to emphasize that the prior knowledge and internal processing are at least as important as the actually acquired "raw" data.
|
||||
|
||||
The restyled version of the logo has been designed and contributed by [xperience.ai](https://xperience.ai/) in July 2020 for the [20th anniversary](https://opencv.org/anniversary/) of OpenCV.
|
||||
|
||||
The logo uses [Exo 2](https://fonts.google.com/specimen/Exo+2#about) font by Natanael Gama distributed under OFL license.
|
||||
|
||||
Higher-resolution version of the logo, as well as SVG version of it, can be obtained at OpenCV [Media Kit](https://opencv.org/resources/media-kit/).
|
||||
|
||||

|
||||
|
Before Width: | Height: | Size: 17 KiB After Width: | Height: | Size: 36 KiB |
|
Before Width: | Height: | Size: 24 KiB After Width: | Height: | Size: 42 KiB |
@@ -584,6 +584,16 @@
|
||||
pages = {1033--1040},
|
||||
publisher = {IEEE}
|
||||
}
|
||||
@article{YM11,
|
||||
author = {Yu, Guoshen and Morel, Jean-Michel},
|
||||
title = {ASIFT: An Algorithm for Fully Affine Invariant Comparison},
|
||||
year = {2011},
|
||||
pages = {11--38},
|
||||
journal = {Image Processing On Line},
|
||||
volume = {1},
|
||||
doi = {10.5201/ipol.2011.my-asift},
|
||||
url = {http://www.ipol.im/pub/algo/my_affine_sift/}
|
||||
}
|
||||
@inproceedings{LCS11,
|
||||
author = {Leutenegger, Stefan and Chli, Margarita and Siegwart, Roland Yves},
|
||||
title = {BRISK: Binary robust invariant scalable keypoints},
|
||||
@@ -1215,3 +1225,23 @@
|
||||
year = {1996},
|
||||
publisher = {Elsevier}
|
||||
}
|
||||
@Article{Wu2009,
|
||||
author={Wu, Kesheng
|
||||
and Otoo, Ekow
|
||||
and Suzuki, Kenji},
|
||||
title={Optimizing two-pass connected-component labeling algorithms},
|
||||
journal={Pattern Analysis and Applications},
|
||||
year={2009},
|
||||
month={Jun},
|
||||
day={01},
|
||||
volume={12},
|
||||
number={2},
|
||||
pages={117-135},
|
||||
}
|
||||
@inproceedings{forstner1987fast,
|
||||
title={A fast operator for detection and precise location of distincs points, corners and center of circular features},
|
||||
author={FORSTNER, W},
|
||||
booktitle={Proc. of the Intercommission Conference on Fast Processing of Photogrammetric Data, Interlaken, Switzerland, 1987},
|
||||
pages={281--305},
|
||||
year={1987}
|
||||
}
|
||||
|
||||
|
Before Width: | Height: | Size: 4.7 KiB After Width: | Height: | Size: 4.2 KiB |
@@ -36,18 +36,27 @@ class PatternMaker:
|
||||
def make_circles_pattern(self):
|
||||
spacing = self.square_size
|
||||
r = spacing / self.radius_rate
|
||||
for x in range(1, self.cols + 1):
|
||||
for y in range(1, self.rows + 1):
|
||||
dot = SVG("circle", cx=x * spacing, cy=y * spacing, r=r, fill="black", stroke="none")
|
||||
pattern_width = ((self.cols - 1.0) * spacing) + (2.0 * r)
|
||||
pattern_height = ((self.rows - 1.0) * spacing) + (2.0 * r)
|
||||
x_spacing = (self.width - pattern_width) / 2.0
|
||||
y_spacing = (self.height - pattern_height) / 2.0
|
||||
for x in range(0, self.cols):
|
||||
for y in range(0, self.rows):
|
||||
dot = SVG("circle", cx=(x * spacing) + x_spacing + r,
|
||||
cy=(y * spacing) + y_spacing + r, r=r, fill="black", stroke="none")
|
||||
self.g.append(dot)
|
||||
|
||||
def make_acircles_pattern(self):
|
||||
spacing = self.square_size
|
||||
r = spacing / self.radius_rate
|
||||
for i in range(0, self.rows):
|
||||
for j in range(0, self.cols):
|
||||
dot = SVG("circle", cx=((j * 2 + i % 2) * spacing) + spacing, cy=self.height - (i * spacing + spacing),
|
||||
r=r, fill="black", stroke="none")
|
||||
pattern_width = ((self.cols-1.0) * 2 * spacing) + spacing + (2.0 * r)
|
||||
pattern_height = ((self.rows-1.0) * spacing) + (2.0 * r)
|
||||
x_spacing = (self.width - pattern_width) / 2.0
|
||||
y_spacing = (self.height - pattern_height) / 2.0
|
||||
for x in range(0, self.cols):
|
||||
for y in range(0, self.rows):
|
||||
dot = SVG("circle", cx=(2 * x * spacing) + (y % 2)*spacing + x_spacing + r,
|
||||
cy=(y * spacing) + y_spacing + r, r=r, fill="black", stroke="none")
|
||||
self.g.append(dot)
|
||||
|
||||
def make_checkerboard_pattern(self):
|
||||
@@ -84,9 +93,9 @@ def main():
|
||||
parser.add_argument("-R", "--radius_rate", help="circles_radius = square_size/radius_rate", default="5.0",
|
||||
action="store", dest="radius_rate", type=float)
|
||||
parser.add_argument("-w", "--page_width", help="page width in units", default="216", action="store",
|
||||
dest="page_width", type=int)
|
||||
dest="page_width", type=float)
|
||||
parser.add_argument("-h", "--page_height", help="page height in units", default="279", action="store",
|
||||
dest="page_width", type=int)
|
||||
dest="page_width", type=float)
|
||||
parser.add_argument("-a", "--page_size", help="page size, supersedes -h -w arguments", default="A4", action="store",
|
||||
dest="page_size", choices=["A0", "A1", "A2", "A3", "A4", "A5"])
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -78,7 +78,7 @@ pixelpoints = np.transpose(np.nonzero(mask))
|
||||
Here, two methods, one using Numpy functions, next one using OpenCV function (last commented line)
|
||||
are given to do the same. Results are also same, but with a slight difference. Numpy gives
|
||||
coordinates in **(row, column)** format, while OpenCV gives coordinates in **(x,y)** format. So
|
||||
basically the answers will be interchanged. Note that, **row = x** and **column = y**.
|
||||
basically the answers will be interchanged. Note that, **row = y** and **column = x**.
|
||||
|
||||
7. Maximum Value, Minimum Value and their locations
|
||||
---------------------------------------------------
|
||||
|
||||
|
Before Width: | Height: | Size: 4.4 KiB After Width: | Height: | Size: 5.2 KiB |
@@ -6,12 +6,11 @@ body, table, div, p, dl {
|
||||
}
|
||||
|
||||
code {
|
||||
font: 12px Consolas, "Liberation Mono", Courier, monospace;
|
||||
font-size: 85%;
|
||||
font-family: "SFMono-Regular",Consolas,"Liberation Mono",Menlo,Courier,monospace;
|
||||
white-space: pre-wrap;
|
||||
padding: 1px 5px;
|
||||
padding: 0;
|
||||
background-color: #ddd;
|
||||
background-color: rgb(223, 229, 241);
|
||||
vertical-align: baseline;
|
||||
}
|
||||
|
||||
@@ -20,6 +19,16 @@ body {
|
||||
margin: 0 auto;
|
||||
}
|
||||
|
||||
div.fragment {
|
||||
padding: 3px;
|
||||
padding-bottom: 0px;
|
||||
}
|
||||
|
||||
div.line {
|
||||
padding-bottom: 3px;
|
||||
font-family: "SFMono-Regular",Consolas,"Liberation Mono",Menlo,Courier,monospace;
|
||||
}
|
||||
|
||||
div.contents {
|
||||
width: 980px;
|
||||
margin: 0 auto;
|
||||
@@ -35,3 +44,11 @@ span.arrow {
|
||||
div.image img{
|
||||
max-width: 900px;
|
||||
}
|
||||
|
||||
#projectlogo
|
||||
{
|
||||
text-align: center;
|
||||
vertical-align: middle;
|
||||
border-collapse: separate;
|
||||
padding-left: 0.5em;
|
||||
}
|
||||
|
||||
@@ -15,7 +15,7 @@ Tutorial was written for the following versions of corresponding software:
|
||||
|
||||
- Download and install Android Studio from https://developer.android.com/studio.
|
||||
|
||||
- Get the latest pre-built OpenCV for Android release from https://github.com/opencv/opencv/releases and unpack it (for example, `opencv-3.4.11-android-sdk.zip`).
|
||||
- Get the latest pre-built OpenCV for Android release from https://github.com/opencv/opencv/releases and unpack it (for example, `opencv-3.X.Y-android-sdk.zip`).
|
||||
|
||||
- Download MobileNet object detection model from https://github.com/chuanqi305/MobileNet-SSD. We need a configuration file `MobileNetSSD_deploy.prototxt` and weights `MobileNetSSD_deploy.caffemodel`.
|
||||
|
||||
|
||||
@@ -37,7 +37,7 @@ J_{12} & J_{22}
|
||||
where \f$J_{11} = M[Z_{x}^{2}]\f$, \f$J_{22} = M[Z_{y}^{2}]\f$, \f$J_{12} = M[Z_{x}Z_{y}]\f$ - components of the tensor, \f$M[]\f$ is a symbol of mathematical expectation (we can consider this operation as averaging in a window w), \f$Z_{x}\f$ and \f$Z_{y}\f$ are partial derivatives of an image \f$Z\f$ with respect to \f$x\f$ and \f$y\f$.
|
||||
|
||||
The eigenvalues of the tensor can be found in the below formula:
|
||||
\f[\lambda_{1,2} = J_{11} + J_{22} \pm \sqrt{(J_{11} - J_{22})^{2} + 4J_{12}^{2}}\f]
|
||||
\f[\lambda_{1,2} = \frac{1}{2} \left [ J_{11} + J_{22} \pm \sqrt{(J_{11} - J_{22})^{2} + 4J_{12}^{2}} \right ] \f]
|
||||
where \f$\lambda_1\f$ - largest eigenvalue, \f$\lambda_2\f$ - smallest eigenvalue.
|
||||
|
||||
### How to estimate orientation and coherency of an anisotropic image by gradient structure tensor?
|
||||
|
||||
@@ -39,14 +39,14 @@ Open your Doxyfile using your favorite text editor and search for the key
|
||||
`TAGFILES`. Change it as follows:
|
||||
|
||||
@code
|
||||
TAGFILES = ./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/3.4.11
|
||||
TAGFILES = ./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/3.4.12
|
||||
@endcode
|
||||
|
||||
If you had other definitions already, you can append the line using a `\`:
|
||||
|
||||
@code
|
||||
TAGFILES = ./docs/doxygen-tags/libstdc++.tag=https://gcc.gnu.org/onlinedocs/libstdc++/latest-doxygen \
|
||||
./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/3.4.11
|
||||
./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/3.4.12
|
||||
@endcode
|
||||
|
||||
Doxygen can now use the information from the tag file to link to the OpenCV
|
||||
|
||||
@@ -64,17 +64,17 @@ The distortion-free projective transformation given by a pinhole camera model i
|
||||
\f[s \; p = A \begin{bmatrix} R|t \end{bmatrix} P_w,\f]
|
||||
|
||||
where \f$P_w\f$ is a 3D point expressed with respect to the world coordinate system,
|
||||
\f$p\f$ is a 2D pixel in the image plane, \f$A\f$ is the intrinsic camera matrix,
|
||||
\f$p\f$ is a 2D pixel in the image plane, \f$A\f$ is the camera intrinsic matrix,
|
||||
\f$R\f$ and \f$t\f$ are the rotation and translation that describe the change of coordinates from
|
||||
world to camera coordinate systems (or camera frame) and \f$s\f$ is the projective transformation's
|
||||
arbitrary scaling and not part of the camera model.
|
||||
|
||||
The intrinsic camera matrix \f$A\f$ (notation used as in @cite Zhang2000 and also generally notated
|
||||
The camera intrinsic matrix \f$A\f$ (notation used as in @cite Zhang2000 and also generally notated
|
||||
as \f$K\f$) projects 3D points given in the camera coordinate system to 2D pixel coordinates, i.e.
|
||||
|
||||
\f[p = A P_c.\f]
|
||||
|
||||
The camera matrix \f$A\f$ is composed of the focal lengths \f$f_x\f$ and \f$f_y\f$, which are
|
||||
The camera intrinsic matrix \f$A\f$ is composed of the focal lengths \f$f_x\f$ and \f$f_y\f$, which are
|
||||
expressed in pixel units, and the principal point \f$(c_x, c_y)\f$, that is usually close to the
|
||||
image center:
|
||||
|
||||
@@ -382,9 +382,9 @@ R & t \\
|
||||
\end{bmatrix} P_{h_0}.\f]
|
||||
|
||||
@note
|
||||
- Many functions in this module take a camera matrix as an input parameter. Although all
|
||||
- Many functions in this module take a camera intrinsic matrix as an input parameter. Although all
|
||||
functions assume the same structure of this parameter, they may name it differently. The
|
||||
parameter's description, however, will be clear in that a camera matrix with the structure
|
||||
parameter's description, however, will be clear in that a camera intrinsic matrix with the structure
|
||||
shown above is required.
|
||||
- A calibration sample for 3 cameras in a horizontal position can be found at
|
||||
opencv_source_code/samples/cpp/3calibration.cpp
|
||||
@@ -450,8 +450,10 @@ enum SolvePnPMethod {
|
||||
SOLVEPNP_ITERATIVE = 0,
|
||||
SOLVEPNP_EPNP = 1, //!< EPnP: Efficient Perspective-n-Point Camera Pose Estimation @cite lepetit2009epnp
|
||||
SOLVEPNP_P3P = 2, //!< Complete Solution Classification for the Perspective-Three-Point Problem @cite gao2003complete
|
||||
SOLVEPNP_DLS = 3, //!< A Direct Least-Squares (DLS) Method for PnP @cite hesch2011direct
|
||||
SOLVEPNP_UPNP = 4, //!< Exhaustive Linearization for Robust Camera Pose and Focal Length Estimation @cite penate2013exhaustive
|
||||
SOLVEPNP_DLS = 3, //!< **Broken implementation. Using this flag will fallback to EPnP.** \n
|
||||
//!< A Direct Least-Squares (DLS) Method for PnP @cite hesch2011direct
|
||||
SOLVEPNP_UPNP = 4, //!< **Broken implementation. Using this flag will fallback to EPnP.** \n
|
||||
//!< Exhaustive Linearization for Robust Camera Pose and Focal Length Estimation @cite penate2013exhaustive
|
||||
SOLVEPNP_AP3P = 5, //!< An Efficient Algebraic Solution to the Perspective-Three-Point Problem @cite Ke17
|
||||
SOLVEPNP_IPPE = 6, //!< Infinitesimal Plane-Based Pose Estimation @cite Collins14 \n
|
||||
//!< Object points must be coplanar.
|
||||
@@ -648,10 +650,10 @@ CV_EXPORTS_W Vec3d RQDecomp3x3( InputArray src, OutputArray mtxR, OutputArray mt
|
||||
OutputArray Qy = noArray(),
|
||||
OutputArray Qz = noArray());
|
||||
|
||||
/** @brief Decomposes a projection matrix into a rotation matrix and a camera matrix.
|
||||
/** @brief Decomposes a projection matrix into a rotation matrix and a camera intrinsic matrix.
|
||||
|
||||
@param projMatrix 3x4 input projection matrix P.
|
||||
@param cameraMatrix Output 3x3 camera matrix K.
|
||||
@param cameraMatrix Output 3x3 camera intrinsic matrix \f$\cameramatrix{A}\f$.
|
||||
@param rotMatrix Output 3x3 external rotation matrix R.
|
||||
@param transVect Output 4x1 translation vector T.
|
||||
@param rotMatrixX Optional 3x3 rotation matrix around x-axis.
|
||||
@@ -736,10 +738,9 @@ CV_EXPORTS_W void composeRT( InputArray rvec1, InputArray tvec1,
|
||||
@param rvec The rotation vector (@ref Rodrigues) that, together with tvec, performs a change of
|
||||
basis from world to camera coordinate system, see @ref calibrateCamera for details.
|
||||
@param tvec The translation vector, see parameter description above.
|
||||
@param cameraMatrix Camera matrix \f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{_1}\f$ .
|
||||
@param cameraMatrix Camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is empty, the zero distortion coefficients are assumed.
|
||||
\f$\distcoeffs\f$ . If the vector is empty, the zero distortion coefficients are assumed.
|
||||
@param imagePoints Output array of image points, 1xN/Nx1 2-channel, or
|
||||
vector\<Point2f\> .
|
||||
@param jacobian Optional output 2Nx(10+\<numDistCoeffs\>) jacobian matrix of derivatives of image
|
||||
@@ -793,10 +794,9 @@ Number of input points must be 4. Object points must be defined in the following
|
||||
1xN/Nx1 3-channel, where N is the number of points. vector\<Point3d\> can be also passed here.
|
||||
@param imagePoints Array of corresponding image points, Nx2 1-channel or 1xN/Nx1 2-channel,
|
||||
where N is the number of points. vector\<Point2d\> can be also passed here.
|
||||
@param cameraMatrix Input camera matrix \f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Input camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
\f$\distcoeffs\f$. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
assumed.
|
||||
@param rvec Output rotation vector (see @ref Rodrigues ) that, together with tvec, brings points from
|
||||
the model coordinate system to the camera coordinate system.
|
||||
@@ -808,7 +808,7 @@ vectors, respectively, and further optimizes them.
|
||||
- **SOLVEPNP_ITERATIVE** Iterative method is based on a Levenberg-Marquardt optimization. In
|
||||
this case the function finds such a pose that minimizes reprojection error, that is the sum
|
||||
of squared distances between the observed projections imagePoints and the projected (using
|
||||
projectPoints ) objectPoints .
|
||||
@ref projectPoints ) objectPoints .
|
||||
- **SOLVEPNP_P3P** Method is based on the paper of X.S. Gao, X.-R. Hou, J. Tang, H.-F. Chang
|
||||
"Complete Solution Classification for the Perspective-Three-Point Problem" (@cite gao2003complete).
|
||||
In this case the function requires exactly four object and image points.
|
||||
@@ -817,9 +817,11 @@ In this case the function requires exactly four object and image points.
|
||||
In this case the function requires exactly four object and image points.
|
||||
- **SOLVEPNP_EPNP** Method has been introduced by F. Moreno-Noguer, V. Lepetit and P. Fua in the
|
||||
paper "EPnP: Efficient Perspective-n-Point Camera Pose Estimation" (@cite lepetit2009epnp).
|
||||
- **SOLVEPNP_DLS** Method is based on the paper of J. Hesch and S. Roumeliotis.
|
||||
- **SOLVEPNP_DLS** **Broken implementation. Using this flag will fallback to EPnP.** \n
|
||||
Method is based on the paper of J. Hesch and S. Roumeliotis.
|
||||
"A Direct Least-Squares (DLS) Method for PnP" (@cite hesch2011direct).
|
||||
- **SOLVEPNP_UPNP** Method is based on the paper of A. Penate-Sanchez, J. Andrade-Cetto,
|
||||
- **SOLVEPNP_UPNP** **Broken implementation. Using this flag will fallback to EPnP.** \n
|
||||
Method is based on the paper of A. Penate-Sanchez, J. Andrade-Cetto,
|
||||
F. Moreno-Noguer. "Exhaustive Linearization for Robust Camera Pose and Focal Length
|
||||
Estimation" (@cite penate2013exhaustive). In this case the function also estimates the parameters \f$f_x\f$ and \f$f_y\f$
|
||||
assuming that both have the same value. Then the cameraMatrix is updated with the estimated
|
||||
@@ -835,7 +837,7 @@ It requires 4 coplanar object points defined in the following order:
|
||||
- point 3: [-squareLength / 2, -squareLength / 2, 0]
|
||||
|
||||
The function estimates the object pose given a set of object points, their corresponding image
|
||||
projections, as well as the camera matrix and the distortion coefficients, see the figure below
|
||||
projections, as well as the camera intrinsic matrix and the distortion coefficients, see the figure below
|
||||
(more precisely, the X-axis of the camera frame is pointing to the right, the Y-axis downward
|
||||
and the Z-axis forward).
|
||||
|
||||
@@ -968,10 +970,9 @@ CV_EXPORTS_W bool solvePnP( InputArray objectPoints, InputArray imagePoints,
|
||||
1xN/Nx1 3-channel, where N is the number of points. vector\<Point3d\> can be also passed here.
|
||||
@param imagePoints Array of corresponding image points, Nx2 1-channel or 1xN/Nx1 2-channel,
|
||||
where N is the number of points. vector\<Point2d\> can be also passed here.
|
||||
@param cameraMatrix Input camera matrix \f$A = \vecthreethree{fx}{0}{cx}{0}{fy}{cy}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Input camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
\f$\distcoeffs\f$. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
assumed.
|
||||
@param rvec Output rotation vector (see @ref Rodrigues ) that, together with tvec, brings points from
|
||||
the model coordinate system to the camera coordinate system.
|
||||
@@ -988,7 +989,7 @@ an inlier.
|
||||
@param flags Method for solving a PnP problem (see @ref solvePnP ).
|
||||
|
||||
The function estimates an object pose given a set of object points, their corresponding image
|
||||
projections, as well as the camera matrix and the distortion coefficients. This function finds such
|
||||
projections, as well as the camera intrinsic matrix and the distortion coefficients. This function finds such
|
||||
a pose that minimizes reprojection error, that is, the sum of squared distances between the observed
|
||||
projections imagePoints and the projected (using @ref projectPoints ) objectPoints. The use of RANSAC
|
||||
makes the function resistant to outliers.
|
||||
@@ -1017,10 +1018,9 @@ CV_EXPORTS_W bool solvePnPRansac( InputArray objectPoints, InputArray imagePoint
|
||||
1x3/3x1 3-channel. vector\<Point3f\> can be also passed here.
|
||||
@param imagePoints Array of corresponding image points, 3x2 1-channel or 1x3/3x1 2-channel.
|
||||
vector\<Point2f\> can be also passed here.
|
||||
@param cameraMatrix Input camera matrix \f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Input camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
\f$\distcoeffs\f$. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
assumed.
|
||||
@param rvecs Output rotation vectors (see @ref Rodrigues ) that, together with tvecs, brings points from
|
||||
the model coordinate system to the camera coordinate system. A P3P problem has up to 4 solutions.
|
||||
@@ -1032,7 +1032,7 @@ the model coordinate system to the camera coordinate system. A P3P problem has u
|
||||
"An Efficient Algebraic Solution to the Perspective-Three-Point Problem" (@cite Ke17).
|
||||
|
||||
The function estimates the object pose given 3 object points, their corresponding image
|
||||
projections, as well as the camera matrix and the distortion coefficients.
|
||||
projections, as well as the camera intrinsic matrix and the distortion coefficients.
|
||||
|
||||
@note
|
||||
The solutions are sorted by reprojection errors (lowest to highest).
|
||||
@@ -1049,10 +1049,9 @@ to the camera coordinate frame) from a 3D-2D point correspondences and starting
|
||||
where N is the number of points. vector\<Point3d\> can also be passed here.
|
||||
@param imagePoints Array of corresponding image points, Nx2 1-channel or 1xN/Nx1 2-channel,
|
||||
where N is the number of points. vector\<Point2d\> can also be passed here.
|
||||
@param cameraMatrix Input camera matrix \f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Input camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
\f$\distcoeffs\f$. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
assumed.
|
||||
@param rvec Input/Output rotation vector (see @ref Rodrigues ) that, together with tvec, brings points from
|
||||
the model coordinate system to the camera coordinate system. Input values are used as an initial solution.
|
||||
@@ -1061,7 +1060,7 @@ the model coordinate system to the camera coordinate system. Input values are us
|
||||
|
||||
The function refines the object pose given at least 3 object points, their corresponding image
|
||||
projections, an initial solution for the rotation and translation vector,
|
||||
as well as the camera matrix and the distortion coefficients.
|
||||
as well as the camera intrinsic matrix and the distortion coefficients.
|
||||
The function minimizes the projection error with respect to the rotation and the translation vectors, according
|
||||
to a Levenberg-Marquardt iterative minimization @cite Madsen04 @cite Eade13 process.
|
||||
*/
|
||||
@@ -1077,10 +1076,9 @@ to the camera coordinate frame) from a 3D-2D point correspondences and starting
|
||||
where N is the number of points. vector\<Point3d\> can also be passed here.
|
||||
@param imagePoints Array of corresponding image points, Nx2 1-channel or 1xN/Nx1 2-channel,
|
||||
where N is the number of points. vector\<Point2d\> can also be passed here.
|
||||
@param cameraMatrix Input camera matrix \f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Input camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
\f$\distcoeffs\f$. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
assumed.
|
||||
@param rvec Input/Output rotation vector (see @ref Rodrigues ) that, together with tvec, brings points from
|
||||
the model coordinate system to the camera coordinate system. Input values are used as an initial solution.
|
||||
@@ -1091,7 +1089,7 @@ gain in the Damped Gauss-Newton formulation.
|
||||
|
||||
The function refines the object pose given at least 3 object points, their corresponding image
|
||||
projections, an initial solution for the rotation and translation vector,
|
||||
as well as the camera matrix and the distortion coefficients.
|
||||
as well as the camera intrinsic matrix and the distortion coefficients.
|
||||
The function minimizes the projection error with respect to the rotation and the translation vectors, using a
|
||||
virtual visual servoing (VVS) @cite Chaumette06 @cite Marchand16 scheme.
|
||||
*/
|
||||
@@ -1119,10 +1117,9 @@ Only 1 solution is returned.
|
||||
1xN/Nx1 3-channel, where N is the number of points. vector\<Point3d\> can be also passed here.
|
||||
@param imagePoints Array of corresponding image points, Nx2 1-channel or 1xN/Nx1 2-channel,
|
||||
where N is the number of points. vector\<Point2d\> can be also passed here.
|
||||
@param cameraMatrix Input camera matrix \f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Input camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
\f$\distcoeffs\f$. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
assumed.
|
||||
@param rvecs Vector of output rotation vectors (see @ref Rodrigues ) that, together with tvecs, brings points from
|
||||
the model coordinate system to the camera coordinate system.
|
||||
@@ -1143,9 +1140,11 @@ In this case the function requires exactly four object and image points.
|
||||
In this case the function requires exactly four object and image points.
|
||||
- **SOLVEPNP_EPNP** Method has been introduced by F.Moreno-Noguer, V.Lepetit and P.Fua in the
|
||||
paper "EPnP: Efficient Perspective-n-Point Camera Pose Estimation" (@cite lepetit2009epnp).
|
||||
- **SOLVEPNP_DLS** Method is based on the paper of Joel A. Hesch and Stergios I. Roumeliotis.
|
||||
- **SOLVEPNP_DLS** **Broken implementation. Using this flag will fallback to EPnP.** \n
|
||||
Method is based on the paper of Joel A. Hesch and Stergios I. Roumeliotis.
|
||||
"A Direct Least-Squares (DLS) Method for PnP" (@cite hesch2011direct).
|
||||
- **SOLVEPNP_UPNP** Method is based on the paper of A.Penate-Sanchez, J.Andrade-Cetto,
|
||||
- **SOLVEPNP_UPNP** **Broken implementation. Using this flag will fallback to EPnP.** \n
|
||||
Method is based on the paper of A.Penate-Sanchez, J.Andrade-Cetto,
|
||||
F.Moreno-Noguer. "Exhaustive Linearization for Robust Camera Pose and Focal Length
|
||||
Estimation" (@cite penate2013exhaustive). In this case the function also estimates the parameters \f$f_x\f$ and \f$f_y\f$
|
||||
assuming that both have the same value. Then the cameraMatrix is updated with the estimated
|
||||
@@ -1168,7 +1167,7 @@ and useExtrinsicGuess is set to true.
|
||||
and the 3D object points projected with the estimated pose.
|
||||
|
||||
The function estimates the object pose given a set of object points, their corresponding image
|
||||
projections, as well as the camera matrix and the distortion coefficients, see the figure below
|
||||
projections, as well as the camera intrinsic matrix and the distortion coefficients, see the figure below
|
||||
(more precisely, the X-axis of the camera frame is pointing to the right, the Y-axis downward
|
||||
and the Z-axis forward).
|
||||
|
||||
@@ -1297,7 +1296,7 @@ CV_EXPORTS_W int solvePnPGeneric( InputArray objectPoints, InputArray imagePoint
|
||||
InputArray rvec = noArray(), InputArray tvec = noArray(),
|
||||
OutputArray reprojectionError = noArray() );
|
||||
|
||||
/** @brief Finds an initial camera matrix from 3D-2D point correspondences.
|
||||
/** @brief Finds an initial camera intrinsic matrix from 3D-2D point correspondences.
|
||||
|
||||
@param objectPoints Vector of vectors of the calibration pattern points in the calibration pattern
|
||||
coordinate space. In the old interface all the per-view vectors are concatenated. See
|
||||
@@ -1308,7 +1307,7 @@ old interface all the per-view vectors are concatenated.
|
||||
@param aspectRatio If it is zero or negative, both \f$f_x\f$ and \f$f_y\f$ are estimated independently.
|
||||
Otherwise, \f$f_x = f_y * \texttt{aspectRatio}\f$ .
|
||||
|
||||
The function estimates and returns an initial camera matrix for the camera calibration process.
|
||||
The function estimates and returns an initial camera intrinsic matrix for the camera calibration process.
|
||||
Currently, the function only supports planar calibration patterns, which are patterns where each
|
||||
object point has z-coordinate =0.
|
||||
*/
|
||||
@@ -1390,10 +1389,9 @@ CV_EXPORTS_W void drawChessboardCorners( InputOutputArray image, Size patternSiz
|
||||
|
||||
@param image Input/output image. It must have 1 or 3 channels. The number of channels is not altered.
|
||||
@param cameraMatrix Input 3x3 floating-point matrix of camera intrinsic parameters.
|
||||
\f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$
|
||||
\f$\cameramatrix{A}\f$
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is empty, the zero distortion coefficients are assumed.
|
||||
\f$\distcoeffs\f$. If the vector is empty, the zero distortion coefficients are assumed.
|
||||
@param rvec Rotation vector (see @ref Rodrigues ) that, together with tvec, brings points from
|
||||
the model coordinate system to the camera coordinate system.
|
||||
@param tvec Translation vector.
|
||||
@@ -1503,14 +1501,13 @@ pattern points (e.g. std::vector<std::vector<cv::Vec2f>>). imagePoints.size() an
|
||||
objectPoints.size(), and imagePoints[i].size() and objectPoints[i].size() for each i, must be equal,
|
||||
respectively. In the old interface all the vectors of object points from different views are
|
||||
concatenated together.
|
||||
@param imageSize Size of the image used only to initialize the intrinsic camera matrix.
|
||||
@param cameraMatrix Input/output 3x3 floating-point camera matrix
|
||||
\f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ . If CV\_CALIB\_USE\_INTRINSIC\_GUESS
|
||||
@param imageSize Size of the image used only to initialize the camera intrinsic matrix.
|
||||
@param cameraMatrix Input/output 3x3 floating-point camera intrinsic matrix
|
||||
\f$\cameramatrix{A}\f$ . If CV\_CALIB\_USE\_INTRINSIC\_GUESS
|
||||
and/or CALIB_FIX_ASPECT_RATIO are specified, some or all of fx, fy, cx, cy must be
|
||||
initialized before calling the function.
|
||||
@param distCoeffs Input/output vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements.
|
||||
\f$\distcoeffs\f$.
|
||||
@param rvecs Output vector of rotation vectors (@ref Rodrigues ) estimated for each pattern view
|
||||
(e.g. std::vector<cv::Mat>>). That is, each i-th rotation vector together with the corresponding
|
||||
i-th translation vector (see the next output parameter description) brings the calibration pattern
|
||||
@@ -1628,9 +1625,9 @@ CV_EXPORTS_W double calibrateCamera( InputArrayOfArrays objectPoints,
|
||||
int flags = 0, TermCriteria criteria = TermCriteria(
|
||||
TermCriteria::COUNT + TermCriteria::EPS, 30, DBL_EPSILON) );
|
||||
|
||||
/** @brief Computes useful camera characteristics from the camera matrix.
|
||||
/** @brief Computes useful camera characteristics from the camera intrinsic matrix.
|
||||
|
||||
@param cameraMatrix Input camera matrix that can be estimated by calibrateCamera or
|
||||
@param cameraMatrix Input camera intrinsic matrix that can be estimated by calibrateCamera or
|
||||
stereoCalibrate .
|
||||
@param imageSize Input image size in pixels.
|
||||
@param apertureWidth Physical width in mm of the sensor.
|
||||
@@ -1666,15 +1663,15 @@ be equal for each i.
|
||||
observed by the first camera. The same structure as in @ref calibrateCamera.
|
||||
@param imagePoints2 Vector of vectors of the projections of the calibration pattern points,
|
||||
observed by the second camera. The same structure as in @ref calibrateCamera.
|
||||
@param cameraMatrix1 Input/output camera matrix for the first camera, the same as in
|
||||
@param cameraMatrix1 Input/output camera intrinsic matrix for the first camera, the same as in
|
||||
@ref calibrateCamera. Furthermore, for the stereo case, additional flags may be used, see below.
|
||||
@param distCoeffs1 Input/output vector of distortion coefficients, the same as in
|
||||
@ref calibrateCamera.
|
||||
@param cameraMatrix2 Input/output second camera matrix for the second camera. See description for
|
||||
@param cameraMatrix2 Input/output second camera intrinsic matrix for the second camera. See description for
|
||||
cameraMatrix1.
|
||||
@param distCoeffs2 Input/output lens distortion coefficients for the second camera. See
|
||||
description for distCoeffs1.
|
||||
@param imageSize Size of the image used only to initialize the intrinsic camera matrices.
|
||||
@param imageSize Size of the image used only to initialize the camera intrinsic matrices.
|
||||
@param R Output rotation matrix. Together with the translation vector T, this matrix brings
|
||||
points given in the first camera's coordinate system to points in the second camera's
|
||||
coordinate system. In more technical terms, the tuple of R and T performs a change of basis
|
||||
@@ -1795,9 +1792,9 @@ CV_EXPORTS_W double stereoCalibrate( InputArrayOfArrays objectPoints,
|
||||
|
||||
/** @brief Computes rectification transforms for each head of a calibrated stereo camera.
|
||||
|
||||
@param cameraMatrix1 First camera matrix.
|
||||
@param cameraMatrix1 First camera intrinsic matrix.
|
||||
@param distCoeffs1 First camera distortion parameters.
|
||||
@param cameraMatrix2 Second camera matrix.
|
||||
@param cameraMatrix2 Second camera intrinsic matrix.
|
||||
@param distCoeffs2 Second camera distortion parameters.
|
||||
@param imageSize Size of the image used for stereo calibration.
|
||||
@param R Rotation matrix from the coordinate system of the first camera to the second camera,
|
||||
@@ -1953,12 +1950,11 @@ CV_EXPORTS_W float rectify3Collinear( InputArray cameraMatrix1, InputArray distC
|
||||
OutputArray Q, double alpha, Size newImgSize,
|
||||
CV_OUT Rect* roi1, CV_OUT Rect* roi2, int flags );
|
||||
|
||||
/** @brief Returns the new camera matrix based on the free scaling parameter.
|
||||
/** @brief Returns the new camera intrinsic matrix based on the free scaling parameter.
|
||||
|
||||
@param cameraMatrix Input camera matrix.
|
||||
@param cameraMatrix Input camera intrinsic matrix.
|
||||
@param distCoeffs Input vector of distortion coefficients
|
||||
\f$(k_1, k_2, p_1, p_2[, k_3[, k_4, k_5, k_6 [, s_1, s_2, s_3, s_4[, \tau_x, \tau_y]]]])\f$ of
|
||||
4, 5, 8, 12 or 14 elements. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
\f$\distcoeffs\f$. If the vector is NULL/empty, the zero distortion coefficients are
|
||||
assumed.
|
||||
@param imageSize Original image size.
|
||||
@param alpha Free scaling parameter between 0 (when all the pixels in the undistorted image are
|
||||
@@ -1967,17 +1963,17 @@ stereoRectify for details.
|
||||
@param newImgSize Image size after rectification. By default, it is set to imageSize .
|
||||
@param validPixROI Optional output rectangle that outlines all-good-pixels region in the
|
||||
undistorted image. See roi1, roi2 description in stereoRectify .
|
||||
@param centerPrincipalPoint Optional flag that indicates whether in the new camera matrix the
|
||||
@param centerPrincipalPoint Optional flag that indicates whether in the new camera intrinsic matrix the
|
||||
principal point should be at the image center or not. By default, the principal point is chosen to
|
||||
best fit a subset of the source image (determined by alpha) to the corrected image.
|
||||
@return new_camera_matrix Output new camera matrix.
|
||||
@return new_camera_matrix Output new camera intrinsic matrix.
|
||||
|
||||
The function computes and returns the optimal new camera matrix based on the free scaling parameter.
|
||||
The function computes and returns the optimal new camera intrinsic matrix based on the free scaling parameter.
|
||||
By varying this parameter, you may retrieve only sensible pixels alpha=0 , keep all the original
|
||||
image pixels if there is valuable information in the corners alpha=1 , or get something in between.
|
||||
When alpha\>0 , the undistorted result is likely to have some black pixels corresponding to
|
||||
"virtual" pixels outside of the captured distorted image. The original camera matrix, distortion
|
||||
coefficients, the computed new camera matrix, and newImageSize should be passed to
|
||||
"virtual" pixels outside of the captured distorted image. The original camera intrinsic matrix, distortion
|
||||
coefficients, the computed new camera intrinsic matrix, and newImageSize should be passed to
|
||||
initUndistortRectifyMap to produce the maps for remap .
|
||||
*/
|
||||
CV_EXPORTS_W Mat getOptimalNewCameraMatrix( InputArray cameraMatrix, InputArray distCoeffs,
|
||||
@@ -1989,23 +1985,23 @@ CV_EXPORTS_W Mat getOptimalNewCameraMatrix( InputArray cameraMatrix, InputArray
|
||||
|
||||
@param[in] R_gripper2base Rotation part extracted from the homogeneous matrix that transforms a point
|
||||
expressed in the gripper frame to the robot base frame (\f$_{}^{b}\textrm{T}_g\f$).
|
||||
This is a vector (`vector<Mat>`) that contains the rotation matrices for all the transformations
|
||||
from gripper frame to robot base frame.
|
||||
This is a vector (`vector<Mat>`) that contains the rotation, `(3x3)` rotation matrices or `(3x1)` rotation vectors,
|
||||
for all the transformations from gripper frame to robot base frame.
|
||||
@param[in] t_gripper2base Translation part extracted from the homogeneous matrix that transforms a point
|
||||
expressed in the gripper frame to the robot base frame (\f$_{}^{b}\textrm{T}_g\f$).
|
||||
This is a vector (`vector<Mat>`) that contains the translation vectors for all the transformations
|
||||
This is a vector (`vector<Mat>`) that contains the `(3x1)` translation vectors for all the transformations
|
||||
from gripper frame to robot base frame.
|
||||
@param[in] R_target2cam Rotation part extracted from the homogeneous matrix that transforms a point
|
||||
expressed in the target frame to the camera frame (\f$_{}^{c}\textrm{T}_t\f$).
|
||||
This is a vector (`vector<Mat>`) that contains the rotation matrices for all the transformations
|
||||
from calibration target frame to camera frame.
|
||||
This is a vector (`vector<Mat>`) that contains the rotation, `(3x3)` rotation matrices or `(3x1)` rotation vectors,
|
||||
for all the transformations from calibration target frame to camera frame.
|
||||
@param[in] t_target2cam Rotation part extracted from the homogeneous matrix that transforms a point
|
||||
expressed in the target frame to the camera frame (\f$_{}^{c}\textrm{T}_t\f$).
|
||||
This is a vector (`vector<Mat>`) that contains the translation vectors for all the transformations
|
||||
This is a vector (`vector<Mat>`) that contains the `(3x1)` translation vectors for all the transformations
|
||||
from calibration target frame to camera frame.
|
||||
@param[out] R_cam2gripper Estimated rotation part extracted from the homogeneous matrix that transforms a point
|
||||
@param[out] R_cam2gripper Estimated `(3x3)` rotation part extracted from the homogeneous matrix that transforms a point
|
||||
expressed in the camera frame to the gripper frame (\f$_{}^{g}\textrm{T}_c\f$).
|
||||
@param[out] t_cam2gripper Estimated translation part extracted from the homogeneous matrix that transforms a point
|
||||
@param[out] t_cam2gripper Estimated `(3x1)` translation part extracted from the homogeneous matrix that transforms a point
|
||||
expressed in the camera frame to the gripper frame (\f$_{}^{g}\textrm{T}_c\f$).
|
||||
@param[in] method One of the implemented Hand-Eye calibration method, see cv::HandEyeCalibrationMethod
|
||||
|
||||
@@ -2167,7 +2163,7 @@ final fundamental matrix. It can be set to something like 1-3, depending on the
|
||||
point localization, image resolution, and the image noise.
|
||||
@param confidence Parameter used for the RANSAC and LMedS methods only. It specifies a desirable level
|
||||
of confidence (probability) that the estimated matrix is correct.
|
||||
@param mask
|
||||
@param[out] mask optional output mask
|
||||
@param maxIters The maximum number of robust method iterations.
|
||||
|
||||
The epipolar geometry is described by the following equation:
|
||||
@@ -2222,11 +2218,11 @@ CV_EXPORTS Mat findFundamentalMat( InputArray points1, InputArray points2,
|
||||
@param points1 Array of N (N \>= 5) 2D points from the first image. The point coordinates should
|
||||
be floating-point (single or double precision).
|
||||
@param points2 Array of the second image points of the same size and format as points1 .
|
||||
@param cameraMatrix Camera matrix \f$K = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
Note that this function assumes that points1 and points2 are feature points from cameras with the
|
||||
same camera matrix. If this assumption does not hold for your use case, use
|
||||
same camera intrinsic matrix. If this assumption does not hold for your use case, use
|
||||
`undistortPoints()` with `P = cv::NoArray()` for both cameras to transform image points
|
||||
to normalized image coordinates, which are valid for the identity camera matrix. When
|
||||
to normalized image coordinates, which are valid for the identity camera intrinsic matrix. When
|
||||
passing these coordinates, pass the identity matrix for this parameter.
|
||||
@param method Method for computing an essential matrix.
|
||||
- **RANSAC** for the RANSAC algorithm.
|
||||
@@ -2273,10 +2269,10 @@ confidence (probability) that the estimated matrix is correct.
|
||||
@param mask Output array of N elements, every element of which is set to 0 for outliers and to 1
|
||||
for the other points. The array is computed only in the RANSAC and LMedS methods.
|
||||
|
||||
This function differs from the one above that it computes camera matrix from focal length and
|
||||
This function differs from the one above that it computes camera intrinsic matrix from focal length and
|
||||
principal point:
|
||||
|
||||
\f[K =
|
||||
\f[A =
|
||||
\begin{bmatrix}
|
||||
f & 0 & x_{pp} \\
|
||||
0 & f & y_{pp} \\
|
||||
@@ -2316,9 +2312,9 @@ inliers that pass the check.
|
||||
@param points1 Array of N 2D points from the first image. The point coordinates should be
|
||||
floating-point (single or double precision).
|
||||
@param points2 Array of the second image points of the same size and format as points1 .
|
||||
@param cameraMatrix Camera matrix \f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
Note that this function assumes that points1 and points2 are feature points from cameras with the
|
||||
same camera matrix.
|
||||
same camera intrinsic matrix.
|
||||
@param R Output rotation matrix. Together with the translation vector, this matrix makes up a tuple
|
||||
that performs a change of basis from the first camera's coordinate system to the second camera's
|
||||
coordinate system. Note that, in general, t can not be used for this tuple, see the parameter
|
||||
@@ -2381,7 +2377,7 @@ are feature points from cameras with same focal length and principal point.
|
||||
inliers in points1 and points2 for then given essential matrix E. Only these inliers will be used to
|
||||
recover pose. In the output mask only inliers which pass the cheirality check.
|
||||
|
||||
This function differs from the one above that it computes camera matrix from focal length and
|
||||
This function differs from the one above that it computes camera intrinsic matrix from focal length and
|
||||
principal point:
|
||||
|
||||
\f[A =
|
||||
@@ -2401,9 +2397,9 @@ CV_EXPORTS_W int recoverPose( InputArray E, InputArray points1, InputArray point
|
||||
@param points1 Array of N 2D points from the first image. The point coordinates should be
|
||||
floating-point (single or double precision).
|
||||
@param points2 Array of the second image points of the same size and format as points1.
|
||||
@param cameraMatrix Camera matrix \f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ .
|
||||
@param cameraMatrix Camera intrinsic matrix \f$\cameramatrix{A}\f$ .
|
||||
Note that this function assumes that points1 and points2 are feature points from cameras with the
|
||||
same camera matrix.
|
||||
same camera intrinsic matrix.
|
||||
@param R Output rotation matrix. Together with the translation vector, this matrix makes up a tuple
|
||||
that performs a change of basis from the first camera's coordinate system to the second camera's
|
||||
coordinate system. Note that, in general, t can not be used for this tuple, see the parameter
|
||||
@@ -2762,7 +2758,7 @@ Check @ref tutorial_homography "the corresponding tutorial" for more details.
|
||||
/** @brief Decompose a homography matrix to rotation(s), translation(s) and plane normal(s).
|
||||
|
||||
@param H The input homography matrix between two images.
|
||||
@param K The input intrinsic camera calibration matrix.
|
||||
@param K The input camera intrinsic matrix.
|
||||
@param rotations Array of rotation matrices.
|
||||
@param translations Array of translation matrices.
|
||||
@param normals Array of plane normal matrices.
|
||||
@@ -3020,8 +3016,8 @@ namespace fisheye
|
||||
@param imagePoints Output array of image points, 2xN/Nx2 1-channel or 1xN/Nx1 2-channel, or
|
||||
vector\<Point2f\>.
|
||||
@param affine
|
||||
@param K Camera matrix \f$K = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{_1}\f$.
|
||||
@param D Input vector of distortion coefficients \f$(k_1, k_2, k_3, k_4)\f$.
|
||||
@param K Camera intrinsic matrix \f$cameramatrix{K}\f$.
|
||||
@param D Input vector of distortion coefficients \f$\distcoeffsfisheye\f$.
|
||||
@param alpha The skew coefficient.
|
||||
@param jacobian Optional output 2Nx15 jacobian matrix of derivatives of image points with respect
|
||||
to components of the focal lengths, coordinates of the principal point, distortion coefficients,
|
||||
@@ -3044,12 +3040,12 @@ namespace fisheye
|
||||
|
||||
@param undistorted Array of object points, 1xN/Nx1 2-channel (or vector\<Point2f\> ), where N is
|
||||
the number of points in the view.
|
||||
@param K Camera matrix \f$K = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{_1}\f$.
|
||||
@param D Input vector of distortion coefficients \f$(k_1, k_2, k_3, k_4)\f$.
|
||||
@param K Camera intrinsic matrix \f$cameramatrix{K}\f$.
|
||||
@param D Input vector of distortion coefficients \f$\distcoeffsfisheye\f$.
|
||||
@param alpha The skew coefficient.
|
||||
@param distorted Output array of image points, 1xN/Nx1 2-channel, or vector\<Point2f\> .
|
||||
|
||||
Note that the function assumes the camera matrix of the undistorted points to be identity.
|
||||
Note that the function assumes the camera intrinsic matrix of the undistorted points to be identity.
|
||||
This means if you want to transform back points undistorted with undistortPoints() you have to
|
||||
multiply them with \f$P^{-1}\f$.
|
||||
*/
|
||||
@@ -3059,11 +3055,11 @@ namespace fisheye
|
||||
|
||||
@param distorted Array of object points, 1xN/Nx1 2-channel (or vector\<Point2f\> ), where N is the
|
||||
number of points in the view.
|
||||
@param K Camera matrix \f$K = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{_1}\f$.
|
||||
@param D Input vector of distortion coefficients \f$(k_1, k_2, k_3, k_4)\f$.
|
||||
@param K Camera intrinsic matrix \f$cameramatrix{K}\f$.
|
||||
@param D Input vector of distortion coefficients \f$\distcoeffsfisheye\f$.
|
||||
@param R Rectification transformation in the object space: 3x3 1-channel, or vector: 3x1/1x3
|
||||
1-channel or 1x1 3-channel
|
||||
@param P New camera matrix (3x3) or new projection matrix (3x4)
|
||||
@param P New camera intrinsic matrix (3x3) or new projection matrix (3x4)
|
||||
@param undistorted Output array of image points, 1xN/Nx1 2-channel, or vector\<Point2f\> .
|
||||
*/
|
||||
CV_EXPORTS_W void undistortPoints(InputArray distorted, OutputArray undistorted,
|
||||
@@ -3072,11 +3068,11 @@ namespace fisheye
|
||||
/** @brief Computes undistortion and rectification maps for image transform by cv::remap(). If D is empty zero
|
||||
distortion is used, if R or P is empty identity matrixes are used.
|
||||
|
||||
@param K Camera matrix \f$K = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{_1}\f$.
|
||||
@param D Input vector of distortion coefficients \f$(k_1, k_2, k_3, k_4)\f$.
|
||||
@param K Camera intrinsic matrix \f$cameramatrix{K}\f$.
|
||||
@param D Input vector of distortion coefficients \f$\distcoeffsfisheye\f$.
|
||||
@param R Rectification transformation in the object space: 3x3 1-channel, or vector: 3x1/1x3
|
||||
1-channel or 1x1 3-channel
|
||||
@param P New camera matrix (3x3) or new projection matrix (3x4)
|
||||
@param P New camera intrinsic matrix (3x3) or new projection matrix (3x4)
|
||||
@param size Undistorted image size.
|
||||
@param m1type Type of the first output map that can be CV_32FC1 or CV_16SC2 . See convertMaps()
|
||||
for details.
|
||||
@@ -3090,9 +3086,9 @@ namespace fisheye
|
||||
|
||||
@param distorted image with fisheye lens distortion.
|
||||
@param undistorted Output image with compensated fisheye lens distortion.
|
||||
@param K Camera matrix \f$K = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{_1}\f$.
|
||||
@param D Input vector of distortion coefficients \f$(k_1, k_2, k_3, k_4)\f$.
|
||||
@param Knew Camera matrix of the distorted image. By default, it is the identity matrix but you
|
||||
@param K Camera intrinsic matrix \f$cameramatrix{K}\f$.
|
||||
@param D Input vector of distortion coefficients \f$\distcoeffsfisheye\f$.
|
||||
@param Knew Camera intrinsic matrix of the distorted image. By default, it is the identity matrix but you
|
||||
may additionally scale and shift the result by using a different matrix.
|
||||
@param new_size the new size
|
||||
|
||||
@@ -3117,14 +3113,14 @@ namespace fisheye
|
||||
CV_EXPORTS_W void undistortImage(InputArray distorted, OutputArray undistorted,
|
||||
InputArray K, InputArray D, InputArray Knew = cv::noArray(), const Size& new_size = Size());
|
||||
|
||||
/** @brief Estimates new camera matrix for undistortion or rectification.
|
||||
/** @brief Estimates new camera intrinsic matrix for undistortion or rectification.
|
||||
|
||||
@param K Camera matrix \f$K = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{_1}\f$.
|
||||
@param K Camera intrinsic matrix \f$cameramatrix{K}\f$.
|
||||
@param image_size Size of the image
|
||||
@param D Input vector of distortion coefficients \f$(k_1, k_2, k_3, k_4)\f$.
|
||||
@param D Input vector of distortion coefficients \f$\distcoeffsfisheye\f$.
|
||||
@param R Rectification transformation in the object space: 3x3 1-channel, or vector: 3x1/1x3
|
||||
1-channel or 1x1 3-channel
|
||||
@param P New camera matrix (3x3) or new projection matrix (3x4)
|
||||
@param P New camera intrinsic matrix (3x3) or new projection matrix (3x4)
|
||||
@param balance Sets the new focal length in range between the min focal length and the max focal
|
||||
length. Balance is in range of [0, 1].
|
||||
@param new_size the new size
|
||||
@@ -3140,12 +3136,12 @@ namespace fisheye
|
||||
@param imagePoints vector of vectors of the projections of calibration pattern points.
|
||||
imagePoints.size() and objectPoints.size() and imagePoints[i].size() must be equal to
|
||||
objectPoints[i].size() for each i.
|
||||
@param image_size Size of the image used only to initialize the intrinsic camera matrix.
|
||||
@param K Output 3x3 floating-point camera matrix
|
||||
\f$A = \vecthreethree{f_x}{0}{c_x}{0}{f_y}{c_y}{0}{0}{1}\f$ . If
|
||||
@param image_size Size of the image used only to initialize the camera intrinsic matrix.
|
||||
@param K Output 3x3 floating-point camera intrinsic matrix
|
||||
\f$\cameramatrix{A}\f$ . If
|
||||
fisheye::CALIB_USE_INTRINSIC_GUESS/ is specified, some or all of fx, fy, cx, cy must be
|
||||
initialized before calling the function.
|
||||
@param D Output vector of distortion coefficients \f$(k_1, k_2, k_3, k_4)\f$.
|
||||
@param D Output vector of distortion coefficients \f$\distcoeffsfisheye\f$.
|
||||
@param rvecs Output vector of rotation vectors (see Rodrigues ) estimated for each pattern view.
|
||||
That is, each k-th rotation vector together with the corresponding k-th translation vector (see
|
||||
the next output parameter description) brings the calibration pattern from the model coordinate
|
||||
@@ -3172,9 +3168,9 @@ optimization. It stays at the center or at a different location specified when C
|
||||
|
||||
/** @brief Stereo rectification for fisheye camera model
|
||||
|
||||
@param K1 First camera matrix.
|
||||
@param K1 First camera intrinsic matrix.
|
||||
@param D1 First camera distortion parameters.
|
||||
@param K2 Second camera matrix.
|
||||
@param K2 Second camera intrinsic matrix.
|
||||
@param D2 Second camera distortion parameters.
|
||||
@param imageSize Size of the image used for stereo calibration.
|
||||
@param R Rotation matrix between the coordinate systems of the first and the second
|
||||
@@ -3211,15 +3207,15 @@ optimization. It stays at the center or at a different location specified when C
|
||||
observed by the first camera.
|
||||
@param imagePoints2 Vector of vectors of the projections of the calibration pattern points,
|
||||
observed by the second camera.
|
||||
@param K1 Input/output first camera matrix:
|
||||
@param K1 Input/output first camera intrinsic matrix:
|
||||
\f$\vecthreethree{f_x^{(j)}}{0}{c_x^{(j)}}{0}{f_y^{(j)}}{c_y^{(j)}}{0}{0}{1}\f$ , \f$j = 0,\, 1\f$ . If
|
||||
any of fisheye::CALIB_USE_INTRINSIC_GUESS , fisheye::CALIB_FIX_INTRINSIC are specified,
|
||||
some or all of the matrix components must be initialized.
|
||||
@param D1 Input/output vector of distortion coefficients \f$(k_1, k_2, k_3, k_4)\f$ of 4 elements.
|
||||
@param K2 Input/output second camera matrix. The parameter is similar to K1 .
|
||||
@param D1 Input/output vector of distortion coefficients \f$\distcoeffsfisheye\f$ of 4 elements.
|
||||
@param K2 Input/output second camera intrinsic matrix. The parameter is similar to K1 .
|
||||
@param D2 Input/output lens distortion coefficients for the second camera. The parameter is
|
||||
similar to D1 .
|
||||
@param imageSize Size of the image used only to initialize intrinsic camera matrix.
|
||||
@param imageSize Size of the image used only to initialize camera intrinsic matrix.
|
||||
@param R Output rotation matrix between the 1st and the 2nd camera coordinate systems.
|
||||
@param T Output translation vector between the coordinate systems of the cameras.
|
||||
@param flags Different flags that may be zero or a combination of the following values:
|
||||
|
||||
@@ -791,6 +791,7 @@ int ChessBoardDetector::orderFoundConnectedQuads(std::vector<ChessBoardQuad*>& q
|
||||
|
||||
for (int i = 0; i < 4; i++)
|
||||
{
|
||||
CV_DbgAssert(q);
|
||||
ChessBoardQuad *neighbor = q->neighbors[i];
|
||||
switch(i) // adjust col, row for this quad
|
||||
{ // start at top left, go clockwise
|
||||
@@ -1271,6 +1272,7 @@ int ChessBoardDetector::cleanFoundConnectedQuads(std::vector<ChessBoardQuad*>& q
|
||||
for (int i = 0; i < quad_count; ++i)
|
||||
{
|
||||
ChessBoardQuad *q = quad_group[i];
|
||||
CV_DbgAssert(q);
|
||||
for (int j = 0; j < 4; ++j)
|
||||
{
|
||||
if (q->neighbors[j] == q0)
|
||||
@@ -1328,6 +1330,7 @@ void ChessBoardDetector::findConnectedQuads(std::vector<ChessBoardQuad*>& out_gr
|
||||
stack.pop();
|
||||
for (int k = 0; k < 4; k++ )
|
||||
{
|
||||
CV_DbgAssert(q);
|
||||
ChessBoardQuad *neighbor = q->neighbors[k];
|
||||
if (neighbor && neighbor->count > 0 && neighbor->group_idx < 0 )
|
||||
{
|
||||
@@ -1716,6 +1719,7 @@ void ChessBoardDetector::findQuadNeighbors()
|
||||
int k = 0;
|
||||
for (; k < 4; k++ )
|
||||
{
|
||||
CV_DbgAssert(q);
|
||||
if (!q->neighbors[k])
|
||||
{
|
||||
if (normL2Sqr<float>(closest_corner.pt - q->corners[k]->pt) < min_dist)
|
||||
@@ -2090,6 +2094,7 @@ void drawChessboardCorners( InputOutputArray image, Size patternSize,
|
||||
return;
|
||||
Mat corners = _corners.getMat();
|
||||
const Point2f* corners_data = corners.ptr<Point2f>(0);
|
||||
CV_DbgAssert(corners_data);
|
||||
int nelems = corners.checkVector(2, CV_32F, true);
|
||||
CV_Assert(nelems >= 0);
|
||||
|
||||
|
||||
@@ -978,9 +978,9 @@ CV_IMPL void cvFindExtrinsicCameraParams2( const CvMat* objectPoints,
|
||||
|
||||
int i, count;
|
||||
double a[9], ar[9]={1,0,0,0,1,0,0,0,1}, R[9];
|
||||
double MM[9], U[9], V[9], W[3];
|
||||
double MM[9] = { 0 }, U[9] = { 0 }, V[9] = { 0 }, W[3] = { 0 };
|
||||
cv::Scalar Mc;
|
||||
double param[6];
|
||||
double param[6] = { 0 };
|
||||
CvMat matA = cvMat( 3, 3, CV_64F, a );
|
||||
CvMat _Ar = cvMat( 3, 3, CV_64F, ar );
|
||||
CvMat matR = cvMat( 3, 3, CV_64F, R );
|
||||
@@ -1199,8 +1199,9 @@ CV_IMPL void cvInitIntrinsicParams2D( const CvMat* objectPoints,
|
||||
CvMat matH = cvMat( 3, 3, CV_64F, H );
|
||||
CvMat _f = cvMat( 2, 1, CV_64F, f );
|
||||
|
||||
assert( CV_MAT_TYPE(npoints->type) == CV_32SC1 &&
|
||||
CV_IS_MAT_CONT(npoints->type) );
|
||||
CV_Assert(npoints);
|
||||
CV_Assert(CV_MAT_TYPE(npoints->type) == CV_32SC1);
|
||||
CV_Assert(CV_IS_MAT_CONT(npoints->type));
|
||||
nimages = npoints->rows + npoints->cols - 1;
|
||||
|
||||
if( (CV_MAT_TYPE(objectPoints->type) != CV_32FC3 &&
|
||||
@@ -1221,6 +1222,9 @@ CV_IMPL void cvInitIntrinsicParams2D( const CvMat* objectPoints,
|
||||
// extract vanishing points in order to obtain initial value for the focal length
|
||||
for( i = 0, pos = 0; i < nimages; i++, pos += ni )
|
||||
{
|
||||
CV_DbgAssert(npoints->data.i);
|
||||
CV_DbgAssert(matA && matA->data.db);
|
||||
CV_DbgAssert(_b && _b->data.db);
|
||||
double* Ap = matA->data.db + i*4;
|
||||
double* bp = _b->data.db + i*2;
|
||||
ni = npoints->data.i[i];
|
||||
@@ -1231,6 +1235,7 @@ CV_IMPL void cvInitIntrinsicParams2D( const CvMat* objectPoints,
|
||||
cvGetCols( imagePoints, &_m, pos, pos + ni );
|
||||
|
||||
cvFindHomography( &matM, &_m, &matH );
|
||||
CV_DbgAssert(_allH && _allH->data.db);
|
||||
memcpy( _allH->data.db + i*9, H, sizeof(H) );
|
||||
|
||||
H[0] -= H[6]*a[2]; H[1] -= H[7]*a[2]; H[2] -= H[8]*a[2];
|
||||
@@ -3828,6 +3833,7 @@ static void adjust3rdMatrix(InputArrayOfArrays _imgpt1_0,
|
||||
|
||||
double y1_ = 0, y2_ = 0, y1y1_ = 0, y1y2_ = 0;
|
||||
size_t n = imgpt1.size();
|
||||
CV_DbgAssert(n > 0);
|
||||
|
||||
for( size_t i = 0; i < n; i++ )
|
||||
{
|
||||
|
||||
@@ -29,6 +29,7 @@ static Mat homogeneousInverse(const Mat& T)
|
||||
// q = sin(theta/2) * v
|
||||
// theta - rotation angle
|
||||
// v - unit rotation axis, |v| = 1
|
||||
// Reference: http://www.euclideanspace.com/maths/geometry/rotations/conversions/matrixToQuaternion/
|
||||
static Mat rot2quatMinimal(const Mat& R)
|
||||
{
|
||||
CV_Assert(R.type() == CV_64FC1 && R.rows >= 3 && R.cols >= 3);
|
||||
@@ -44,7 +45,7 @@ static Mat rot2quatMinimal(const Mat& R)
|
||||
qx = (m21 - m12) / S;
|
||||
qy = (m02 - m20) / S;
|
||||
qz = (m10 - m01) / S;
|
||||
} else if ((m00 > m11)&(m00 > m22)) {
|
||||
} else if (m00 > m11 && m00 > m22) {
|
||||
double S = sqrt(1.0 + m00 - m11 - m22) * 2; // S=4*qx
|
||||
qx = 0.25 * S;
|
||||
qy = (m01 + m10) / S;
|
||||
@@ -98,6 +99,7 @@ static Mat quatMinimal2rot(const Mat& q)
|
||||
//
|
||||
// q - 4x1 unit quaternion <qw, qx, qy, qz>
|
||||
// R - 3x3 rotation matrix
|
||||
// Reference: http://www.euclideanspace.com/maths/geometry/rotations/conversions/matrixToQuaternion/
|
||||
static Mat rot2quat(const Mat& R)
|
||||
{
|
||||
CV_Assert(R.type() == CV_64FC1 && R.rows >= 3 && R.cols >= 3);
|
||||
@@ -114,7 +116,7 @@ static Mat rot2quat(const Mat& R)
|
||||
qx = (m21 - m12) / S;
|
||||
qy = (m02 - m20) / S;
|
||||
qz = (m10 - m01) / S;
|
||||
} else if ((m00 > m11)&(m00 > m22)) {
|
||||
} else if (m00 > m11 && m00 > m22) {
|
||||
double S = sqrt(1.0 + m00 - m11 - m22) * 2; // S=4*qx
|
||||
qw = (m21 - m12) / S;
|
||||
qx = 0.25 * S;
|
||||
@@ -572,7 +574,11 @@ static void calibrateHandEyeAndreff(const std::vector<Mat>& Hg, const std::vecto
|
||||
R = R.reshape(1, 2, newSize);
|
||||
//Eq 15
|
||||
double det = determinant(R);
|
||||
R = pow(sign_double(det) / abs(det), 1.0/3.0) * R;
|
||||
if (std::fabs(det) < FLT_EPSILON)
|
||||
{
|
||||
CV_Error(Error::StsNoConv, "calibrateHandEye() with CALIB_HAND_EYE_ANDREFF method: determinant(R) is null");
|
||||
}
|
||||
R = cubeRoot(static_cast<float>(sign_double(det) / abs(det))) * R;
|
||||
|
||||
Mat w, u, vt;
|
||||
SVDecomp(R, w, u, vt);
|
||||
@@ -712,7 +718,10 @@ void calibrateHandEye(InputArrayOfArrays R_gripper2base, InputArrayOfArrays t_gr
|
||||
{
|
||||
Mat m = Mat::eye(4, 4, CV_64FC1);
|
||||
Mat R = m(Rect(0, 0, 3, 3));
|
||||
R_gripper2base_[i].convertTo(R, CV_64F);
|
||||
if(R_gripper2base_[i].size() == Size(3, 3))
|
||||
R_gripper2base_[i].convertTo(R, CV_64F);
|
||||
else
|
||||
Rodrigues(R_gripper2base_[i], R);
|
||||
|
||||
Mat t = m(Rect(3, 0, 1, 3));
|
||||
t_gripper2base_[i].convertTo(t, CV_64F);
|
||||
@@ -727,7 +736,10 @@ void calibrateHandEye(InputArrayOfArrays R_gripper2base, InputArrayOfArrays t_gr
|
||||
{
|
||||
Mat m = Mat::eye(4, 4, CV_64FC1);
|
||||
Mat R = m(Rect(0, 0, 3, 3));
|
||||
R_target2cam_[i].convertTo(R, CV_64F);
|
||||
if(R_target2cam_[i].size() == Size(3, 3))
|
||||
R_target2cam_[i].convertTo(R, CV_64F);
|
||||
else
|
||||
Rodrigues(R_target2cam_[i], R);
|
||||
|
||||
Mat t = m(Rect(3, 0, 1, 3));
|
||||
t_target2cam_[i].convertTo(t, CV_64F);
|
||||
|
||||
@@ -374,6 +374,9 @@ cv::Mat cv::findHomography( InputArray _points1, InputArray _points2,
|
||||
return Mat();
|
||||
convertPointsFromHomogeneous(p, p);
|
||||
}
|
||||
// Need at least 4 point correspondences to calculate Homography
|
||||
if( npoints < 4 )
|
||||
CV_Error(Error::StsVecLengthErr , "The input arrays should have at least 4 corresponding point sets to calculate Homography");
|
||||
p.reshape(2, npoints).convertTo(m, CV_32F);
|
||||
}
|
||||
|
||||
|
||||
@@ -77,18 +77,18 @@ void PoseSolver::solveGeneric(InputArray _objectPoints, InputArray _normalizedIn
|
||||
OutputArray _Ma, OutputArray _Mb)
|
||||
{
|
||||
//argument checking:
|
||||
size_t n = static_cast<size_t>(_objectPoints.rows() * _objectPoints.cols()); //number of points
|
||||
size_t n = static_cast<size_t>(_normalizedInputPoints.rows()) * static_cast<size_t>(_normalizedInputPoints.cols()); //number of points
|
||||
int objType = _objectPoints.type();
|
||||
int type_input = _normalizedInputPoints.type();
|
||||
|
||||
CV_CheckType(objType, objType == CV_32FC3 || objType == CV_64FC3,
|
||||
"Type of _objectPoints must be CV_32FC3 or CV_64FC3" );
|
||||
CV_CheckType(type_input, type_input == CV_32FC2 || type_input == CV_64FC2,
|
||||
"Type of _normalizedInputPoints must be CV_32FC3 or CV_64FC3" );
|
||||
"Type of _normalizedInputPoints must be CV_32FC2 or CV_64FC2" );
|
||||
CV_Assert(_objectPoints.rows() == 1 || _objectPoints.cols() == 1);
|
||||
CV_Assert(_objectPoints.rows() >= 4 || _objectPoints.cols() >= 4);
|
||||
CV_Assert(_normalizedInputPoints.rows() == 1 || _normalizedInputPoints.cols() == 1);
|
||||
CV_Assert(static_cast<size_t>(_objectPoints.rows() * _objectPoints.cols()) == n);
|
||||
CV_Assert(static_cast<size_t>(_objectPoints.rows()) * static_cast<size_t>(_objectPoints.cols()) == n);
|
||||
|
||||
Mat normalizedInputPoints;
|
||||
if (type_input == CV_32FC2)
|
||||
@@ -101,7 +101,7 @@ void PoseSolver::solveGeneric(InputArray _objectPoints, InputArray _normalizedIn
|
||||
}
|
||||
|
||||
Mat objectInputPoints;
|
||||
if (type_input == CV_32FC3)
|
||||
if (objType == CV_32FC3)
|
||||
{
|
||||
_objectPoints.getMat().convertTo(objectInputPoints, CV_64F);
|
||||
}
|
||||
|
||||
@@ -48,6 +48,7 @@
|
||||
#include "ap3p.h"
|
||||
#include "ippe.hpp"
|
||||
#include "opencv2/calib3d/calib3d_c.h"
|
||||
#include <opencv2/core/utils/logger.hpp>
|
||||
|
||||
namespace cv
|
||||
{
|
||||
@@ -780,6 +781,15 @@ int solvePnPGeneric( InputArray _opoints, InputArray _ipoints,
|
||||
vector<Mat> vec_rvecs, vec_tvecs;
|
||||
if (flags == SOLVEPNP_EPNP || flags == SOLVEPNP_DLS || flags == SOLVEPNP_UPNP)
|
||||
{
|
||||
if (flags == SOLVEPNP_DLS)
|
||||
{
|
||||
CV_LOG_DEBUG(NULL, "Broken implementation for SOLVEPNP_DLS. Fallback to EPnP.");
|
||||
}
|
||||
else if (flags == SOLVEPNP_UPNP)
|
||||
{
|
||||
CV_LOG_DEBUG(NULL, "Broken implementation for SOLVEPNP_UPNP. Fallback to EPnP.");
|
||||
}
|
||||
|
||||
Mat undistortedPoints;
|
||||
undistortPoints(ipoints, undistortedPoints, cameraMatrix, distCoeffs);
|
||||
epnp PnP(cameraMatrix, opoints, undistortedPoints);
|
||||
|
||||
@@ -7,6 +7,38 @@
|
||||
|
||||
namespace opencv_test { namespace {
|
||||
|
||||
static std::string getMethodName(HandEyeCalibrationMethod method)
|
||||
{
|
||||
std::string method_name = "";
|
||||
switch (method)
|
||||
{
|
||||
case CALIB_HAND_EYE_TSAI:
|
||||
method_name = "Tsai";
|
||||
break;
|
||||
|
||||
case CALIB_HAND_EYE_PARK:
|
||||
method_name = "Park";
|
||||
break;
|
||||
|
||||
case CALIB_HAND_EYE_HORAUD:
|
||||
method_name = "Horaud";
|
||||
break;
|
||||
|
||||
case CALIB_HAND_EYE_ANDREFF:
|
||||
method_name = "Andreff";
|
||||
break;
|
||||
|
||||
case CALIB_HAND_EYE_DANIILIDIS:
|
||||
method_name = "Daniilidis";
|
||||
break;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return method_name;
|
||||
}
|
||||
|
||||
class CV_CalibrateHandEyeTest : public cvtest::BaseTest
|
||||
{
|
||||
public:
|
||||
@@ -48,7 +80,6 @@ protected:
|
||||
std::vector<Mat> &R_target2cam, std::vector<Mat> &t_target2cam,
|
||||
bool noise, Mat& R_cam2gripper, Mat& t_cam2gripper);
|
||||
Mat homogeneousInverse(const Mat& T);
|
||||
std::string getMethodName(HandEyeCalibrationMethod method);
|
||||
double sign_double(double val);
|
||||
|
||||
double eps_rvec[5];
|
||||
@@ -317,7 +348,10 @@ void CV_CalibrateHandEyeTest::simulateData(RNG& rng, int nPoses,
|
||||
t_gripper2base_noise.at<double>(2,0) += rng.gaussian(0.001);
|
||||
}
|
||||
|
||||
R_target2cam.push_back(T_target2cam(Rect(0, 0, 3, 3)));
|
||||
// test rvec represenation
|
||||
Mat rvec_target2cam;
|
||||
cv::Rodrigues(T_target2cam(Rect(0, 0, 3, 3)), rvec_target2cam);
|
||||
R_target2cam.push_back(rvec_target2cam);
|
||||
t_target2cam.push_back(T_target2cam(Rect(3, 0, 1, 3)));
|
||||
}
|
||||
}
|
||||
@@ -337,38 +371,6 @@ Mat CV_CalibrateHandEyeTest::homogeneousInverse(const Mat& T)
|
||||
return Tinv;
|
||||
}
|
||||
|
||||
std::string CV_CalibrateHandEyeTest::getMethodName(HandEyeCalibrationMethod method)
|
||||
{
|
||||
std::string method_name = "";
|
||||
switch (method)
|
||||
{
|
||||
case CALIB_HAND_EYE_TSAI:
|
||||
method_name = "Tsai";
|
||||
break;
|
||||
|
||||
case CALIB_HAND_EYE_PARK:
|
||||
method_name = "Park";
|
||||
break;
|
||||
|
||||
case CALIB_HAND_EYE_HORAUD:
|
||||
method_name = "Horaud";
|
||||
break;
|
||||
|
||||
case CALIB_HAND_EYE_ANDREFF:
|
||||
method_name = "Andreff";
|
||||
break;
|
||||
|
||||
case CALIB_HAND_EYE_DANIILIDIS:
|
||||
method_name = "Daniilidis";
|
||||
break;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
return method_name;
|
||||
}
|
||||
|
||||
double CV_CalibrateHandEyeTest::sign_double(double val)
|
||||
{
|
||||
return (0 < val) - (val < 0);
|
||||
@@ -378,4 +380,86 @@ double CV_CalibrateHandEyeTest::sign_double(double val)
|
||||
|
||||
TEST(Calib3d_CalibrateHandEye, regression) { CV_CalibrateHandEyeTest test; test.safe_run(); }
|
||||
|
||||
TEST(Calib3d_CalibrateHandEye, regression_17986)
|
||||
{
|
||||
const std::string camera_poses_filename = findDataFile("cv/hand_eye_calibration/cali.txt");
|
||||
const std::string end_effector_poses = findDataFile("cv/hand_eye_calibration/robot_cali.txt");
|
||||
|
||||
std::vector<Mat> R_target2cam;
|
||||
std::vector<Mat> t_target2cam;
|
||||
// Parse camera poses
|
||||
{
|
||||
std::ifstream file(camera_poses_filename.c_str());
|
||||
ASSERT_TRUE(file.is_open());
|
||||
|
||||
int ndata = 0;
|
||||
file >> ndata;
|
||||
R_target2cam.reserve(ndata);
|
||||
t_target2cam.reserve(ndata);
|
||||
|
||||
std::string image_name;
|
||||
Matx33d cameraMatrix;
|
||||
Matx33d R;
|
||||
Matx31d t;
|
||||
Matx16d distCoeffs;
|
||||
Matx13d distCoeffs2;
|
||||
while (file >> image_name >>
|
||||
cameraMatrix(0,0) >> cameraMatrix(0,1) >> cameraMatrix(0,2) >>
|
||||
cameraMatrix(1,0) >> cameraMatrix(1,1) >> cameraMatrix(1,2) >>
|
||||
cameraMatrix(2,0) >> cameraMatrix(2,1) >> cameraMatrix(2,2) >>
|
||||
R(0,0) >> R(0,1) >> R(0,2) >>
|
||||
R(1,0) >> R(1,1) >> R(1,2) >>
|
||||
R(2,0) >> R(2,1) >> R(2,2) >>
|
||||
t(0) >> t(1) >> t(2) >>
|
||||
distCoeffs(0) >> distCoeffs(1) >> distCoeffs(2) >> distCoeffs(3) >> distCoeffs(4) >>
|
||||
distCoeffs2(0) >> distCoeffs2(1) >> distCoeffs2(2)) {
|
||||
R_target2cam.push_back(Mat(R));
|
||||
t_target2cam.push_back(Mat(t));
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<Mat> R_gripper2base;
|
||||
std::vector<Mat> t_gripper2base;
|
||||
// Parse end-effector poses
|
||||
{
|
||||
std::ifstream file(end_effector_poses.c_str());
|
||||
ASSERT_TRUE(file.is_open());
|
||||
|
||||
int ndata = 0;
|
||||
file >> ndata;
|
||||
R_gripper2base.reserve(ndata);
|
||||
t_gripper2base.reserve(ndata);
|
||||
|
||||
Matx33d R;
|
||||
Matx31d t;
|
||||
Matx14d last_row;
|
||||
while (file >>
|
||||
R(0,0) >> R(0,1) >> R(0,2) >> t(0) >>
|
||||
R(1,0) >> R(1,1) >> R(1,2) >> t(1) >>
|
||||
R(2,0) >> R(2,1) >> R(2,2) >> t(2) >>
|
||||
last_row(0) >> last_row(1) >> last_row(2) >> last_row(3)) {
|
||||
R_gripper2base.push_back(Mat(R));
|
||||
t_gripper2base.push_back(Mat(t));
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<HandEyeCalibrationMethod> methods;
|
||||
methods.push_back(CALIB_HAND_EYE_TSAI);
|
||||
methods.push_back(CALIB_HAND_EYE_PARK);
|
||||
methods.push_back(CALIB_HAND_EYE_HORAUD);
|
||||
methods.push_back(CALIB_HAND_EYE_ANDREFF);
|
||||
methods.push_back(CALIB_HAND_EYE_DANIILIDIS);
|
||||
|
||||
for (size_t idx = 0; idx < methods.size(); idx++) {
|
||||
SCOPED_TRACE(cv::format("method=%s", getMethodName(methods[idx]).c_str()));
|
||||
|
||||
Matx33d R_cam2gripper_est;
|
||||
Matx31d t_cam2gripper_est;
|
||||
calibrateHandEye(R_gripper2base, t_gripper2base, R_target2cam, t_target2cam, R_cam2gripper_est, t_cam2gripper_est, methods[idx]);
|
||||
|
||||
EXPECT_TRUE(checkRange(R_cam2gripper_est));
|
||||
EXPECT_TRUE(checkRange(t_cam2gripper_est));
|
||||
}
|
||||
}
|
||||
|
||||
}} // namespace
|
||||
|
||||
@@ -63,6 +63,7 @@ namespace opencv_test { namespace {
|
||||
#define MESSAGE_RANSAC_DIFF "Reprojection error for current pair of points more than required."
|
||||
|
||||
#define MAX_COUNT_OF_POINTS 303
|
||||
#define MIN_COUNT_OF_POINTS 4
|
||||
#define COUNT_NORM_TYPES 3
|
||||
#define METHODS_COUNT 4
|
||||
|
||||
@@ -249,7 +250,7 @@ void CV_HomographyTest::print_information_8(int _method, int j, int N, int k, in
|
||||
|
||||
void CV_HomographyTest::run(int)
|
||||
{
|
||||
for (int N = 4; N <= MAX_COUNT_OF_POINTS; ++N)
|
||||
for (int N = MIN_COUNT_OF_POINTS; N <= MAX_COUNT_OF_POINTS; ++N)
|
||||
{
|
||||
RNG& rng = ts->get_rng();
|
||||
|
||||
@@ -711,4 +712,27 @@ TEST(Calib3d_Homography, fromImages)
|
||||
ASSERT_GE(ninliers1, 80);
|
||||
}
|
||||
|
||||
TEST(Calib3d_Homography, minPoints)
|
||||
{
|
||||
float pt1data[] =
|
||||
{
|
||||
2.80073029e+002f, 2.39591217e+002f, 2.21912201e+002f, 2.59783997e+002f
|
||||
};
|
||||
|
||||
float pt2data[] =
|
||||
{
|
||||
1.84072723e+002f, 1.43591202e+002f, 1.25912483e+002f, 1.63783859e+002f
|
||||
};
|
||||
|
||||
int npoints = (int)(sizeof(pt1data)/sizeof(pt1data[0])/2);
|
||||
printf("npoints = %d\n", npoints); // npoints = 2
|
||||
|
||||
Mat p1(1, npoints, CV_32FC2, pt1data);
|
||||
Mat p2(1, npoints, CV_32FC2, pt2data);
|
||||
Mat mask;
|
||||
|
||||
// findHomography should raise an error since npoints < MIN_COUNT_OF_POINTS
|
||||
EXPECT_THROW(findHomography(p1, p2, RANSAC, 0.01, mask), cv::Exception);
|
||||
}
|
||||
|
||||
}} // namespace
|
||||
|
||||
@@ -1615,7 +1615,9 @@ elements.
|
||||
CV_EXPORTS_W bool checkRange(InputArray a, bool quiet = true, CV_OUT Point* pos = 0,
|
||||
double minVal = -DBL_MAX, double maxVal = DBL_MAX);
|
||||
|
||||
/** @brief converts NaN's to the given number
|
||||
/** @brief converts NaNs to the given number
|
||||
@param a input/output matrix (CV_32F type).
|
||||
@param val value to convert the NaNs
|
||||
*/
|
||||
CV_EXPORTS_W void patchNaNs(InputOutputArray a, double val = 0);
|
||||
|
||||
|
||||
@@ -63,7 +63,7 @@ struct CheckContext {
|
||||
#define CV__CHECK_LOCATION_VARNAME(id) CVAUX_CONCAT(CVAUX_CONCAT(__cv_check_, id), __LINE__)
|
||||
#define CV__DEFINE_CHECK_CONTEXT(id, message, testOp, p1_str, p2_str) \
|
||||
static const cv::detail::CheckContext CV__CHECK_LOCATION_VARNAME(id) = \
|
||||
{ CV__CHECK_FUNCTION, CV__CHECK_FILENAME, __LINE__, testOp, message, p1_str, p2_str }
|
||||
{ CV__CHECK_FUNCTION, CV__CHECK_FILENAME, __LINE__, testOp, "" message, "" p1_str, "" p2_str }
|
||||
|
||||
CV_EXPORTS void CV_NORETURN check_failed_auto(const int v1, const int v2, const CheckContext& ctx);
|
||||
CV_EXPORTS void CV_NORETURN check_failed_auto(const size_t v1, const size_t v2, const CheckContext& ctx);
|
||||
|
||||
@@ -58,11 +58,13 @@
|
||||
#pragma warning( disable: 4244 ) //conversion from '__int64' to 'int', possible loss of data
|
||||
#endif
|
||||
|
||||
#if !defined(OPENCV_DISABLE_EIGEN_TENSOR_SUPPORT)
|
||||
#if EIGEN_WORLD_VERSION == 3 && EIGEN_MAJOR_VERSION >= 3 \
|
||||
&& defined(CV_CXX11) && defined(CV_CXX_STD_ARRAY)
|
||||
#include <unsupported/Eigen/CXX11/Tensor>
|
||||
#define OPENCV_EIGEN_TENSOR_SUPPORT
|
||||
#endif // EIGEN_WORLD_VERSION == 3 && EIGEN_MAJOR_VERSION >= 3
|
||||
#define OPENCV_EIGEN_TENSOR_SUPPORT 1
|
||||
#endif // EIGEN_WORLD_VERSION == 3 && EIGEN_MAJOR_VERSION >= 3
|
||||
#endif // !defined(OPENCV_DISABLE_EIGEN_TENSOR_SUPPORT)
|
||||
|
||||
namespace cv
|
||||
{
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
#define CV_VERSION_MAJOR 3
|
||||
#define CV_VERSION_MINOR 4
|
||||
#define CV_VERSION_REVISION 11
|
||||
#define CV_VERSION_REVISION 12
|
||||
#define CV_VERSION_STATUS ""
|
||||
|
||||
#define CVAUX_STR_EXP(__A) #__A
|
||||
|
||||
@@ -414,6 +414,29 @@ void Mat::copyTo( OutputArray _dst, InputArray _mask ) const
|
||||
copymask(ptrs[0], 0, ptrs[2], 0, ptrs[1], 0, sz, &esz);
|
||||
}
|
||||
|
||||
|
||||
static bool can_apply_memset(const Mat &mat, const Scalar &s, int &fill_value)
|
||||
{
|
||||
// check if depth is 1 byte.
|
||||
switch (mat.depth())
|
||||
{
|
||||
case CV_8U: fill_value = saturate_cast<uchar>( s.val[0] ); break;
|
||||
case CV_8S: fill_value = saturate_cast<schar>( s.val[0] ); break;
|
||||
default: return false;
|
||||
}
|
||||
|
||||
// check if all element is same.
|
||||
const int64* is = (const int64*)&s.val[0];
|
||||
switch (mat.channels())
|
||||
{
|
||||
case 1: return true;
|
||||
case 2: return (is[0] == is[1]);
|
||||
case 3: return (is[0] == is[1] && is[1] == is[2]);
|
||||
case 4: return (is[0] == is[1] && is[1] == is[2] && is[2] == is[3]);
|
||||
default: return false;
|
||||
}
|
||||
}
|
||||
|
||||
Mat& Mat::operator = (const Scalar& s)
|
||||
{
|
||||
CV_INSTRUMENT_REGION();
|
||||
@@ -434,6 +457,14 @@ Mat& Mat::operator = (const Scalar& s)
|
||||
}
|
||||
else
|
||||
{
|
||||
int fill_value = 0;
|
||||
if ( can_apply_memset(*this, s, fill_value) )
|
||||
{
|
||||
for (size_t i = 0; i < it.nplanes; i++, ++it)
|
||||
memset(dptr, fill_value, elsize);
|
||||
return *this;
|
||||
}
|
||||
|
||||
if( it.nplanes > 0 )
|
||||
{
|
||||
double scalar[12];
|
||||
|
||||
@@ -561,7 +561,7 @@ void cv::cuda::GpuMat::convertTo(OutputArray _dst, int rtype, Stream& stream) co
|
||||
{convertToNoScale<double, uchar>, convertToNoScale<double, schar>, convertToNoScale<double, ushort>, convertToNoScale<double, short>, convertToNoScale<double, int>, convertToNoScale<double, float>, 0}
|
||||
};
|
||||
|
||||
funcs[sdepth][ddepth](reshape(1), dst.reshape(1), stream);
|
||||
funcs[sdepth][ddepth](src.reshape(1), dst.reshape(1), stream);
|
||||
}
|
||||
|
||||
void cv::cuda::GpuMat::convertTo(OutputArray _dst, int rtype, double alpha, double beta, Stream& stream) const
|
||||
@@ -591,7 +591,7 @@ void cv::cuda::GpuMat::convertTo(OutputArray _dst, int rtype, double alpha, doub
|
||||
{convertToScale<double, uchar>, convertToScale<double, schar>, convertToScale<double, ushort>, convertToScale<double, short>, convertToScale<double, int>, convertToScale<double, float>, convertToScale<double, double>}
|
||||
};
|
||||
|
||||
funcs[sdepth][ddepth](reshape(1), dst.reshape(1), alpha, beta, stream);
|
||||
funcs[sdepth][ddepth](src.reshape(1), dst.reshape(1), alpha, beta, stream);
|
||||
}
|
||||
|
||||
void cv::cuda::convertFp16(InputArray _src, OutputArray _dst, Stream& stream)
|
||||
|
||||
@@ -237,12 +237,19 @@ void setSize( Mat& m, int _dims, const int* _sz, const size_t* _steps, bool auto
|
||||
|
||||
if( _steps )
|
||||
{
|
||||
if (_steps[i] % esz1 != 0)
|
||||
if (i < _dims-1)
|
||||
{
|
||||
CV_Error(Error::BadStep, "Step must be a multiple of esz1");
|
||||
}
|
||||
if (_steps[i] % esz1 != 0)
|
||||
{
|
||||
CV_Error_(Error::BadStep, ("Step %zu for dimension %d must be a multiple of esz1 %zu", _steps[i], i, esz1));
|
||||
}
|
||||
|
||||
m.step.p[i] = i < _dims-1 ? _steps[i] : esz;
|
||||
m.step.p[i] = _steps[i];
|
||||
}
|
||||
else
|
||||
{
|
||||
m.step.p[i] = esz;
|
||||
}
|
||||
}
|
||||
else if( autoSteps )
|
||||
{
|
||||
|
||||
@@ -1247,6 +1247,7 @@ void _OutputArray::create(int d, const int* sizes, int mtype, int i,
|
||||
{
|
||||
CV_Assert( i < 0 );
|
||||
Mat& m = *(Mat*)obj;
|
||||
CV_Assert(!(m.empty() && fixedType() && fixedSize()) && "Can't reallocate empty Mat with locked layout (probably due to misused 'const' modifier)");
|
||||
if (allowTransposed && !m.empty() &&
|
||||
d == 2 && m.dims == 2 &&
|
||||
m.type() == mtype && m.rows == sizes[1] && m.cols == sizes[0] &&
|
||||
@@ -1260,13 +1261,13 @@ void _OutputArray::create(int d, const int* sizes, int mtype, int i,
|
||||
if(CV_MAT_CN(mtype) == m.channels() && ((1 << CV_MAT_TYPE(flags)) & fixedDepthMask) != 0 )
|
||||
mtype = m.type();
|
||||
else
|
||||
CV_CheckTypeEQ(m.type(), CV_MAT_TYPE(mtype), "");
|
||||
CV_CheckTypeEQ(m.type(), CV_MAT_TYPE(mtype), "Can't reallocate Mat with locked type (probably due to misused 'const' modifier)");
|
||||
}
|
||||
if(fixedSize())
|
||||
{
|
||||
CV_CheckEQ(m.dims, d, "");
|
||||
CV_CheckEQ(m.dims, d, "Can't reallocate Mat with locked size (probably due to misused 'const' modifier)");
|
||||
for(int j = 0; j < d; ++j)
|
||||
CV_CheckEQ(m.size[j], sizes[j], "");
|
||||
CV_CheckEQ(m.size[j], sizes[j], "Can't reallocate Mat with locked size (probably due to misused 'const' modifier)");
|
||||
}
|
||||
m.create(d, sizes, mtype);
|
||||
return;
|
||||
@@ -1276,6 +1277,7 @@ void _OutputArray::create(int d, const int* sizes, int mtype, int i,
|
||||
{
|
||||
CV_Assert( i < 0 );
|
||||
UMat& m = *(UMat*)obj;
|
||||
CV_Assert(!(m.empty() && fixedType() && fixedSize()) && "Can't reallocate empty UMat with locked layout (probably due to misused 'const' modifier)");
|
||||
if (allowTransposed && !m.empty() &&
|
||||
d == 2 && m.dims == 2 &&
|
||||
m.type() == mtype && m.rows == sizes[1] && m.cols == sizes[0] &&
|
||||
@@ -1289,13 +1291,13 @@ void _OutputArray::create(int d, const int* sizes, int mtype, int i,
|
||||
if(CV_MAT_CN(mtype) == m.channels() && ((1 << CV_MAT_TYPE(flags)) & fixedDepthMask) != 0 )
|
||||
mtype = m.type();
|
||||
else
|
||||
CV_CheckTypeEQ(m.type(), CV_MAT_TYPE(mtype), "");
|
||||
CV_CheckTypeEQ(m.type(), CV_MAT_TYPE(mtype), "Can't reallocate UMat with locked type (probably due to misused 'const' modifier)");
|
||||
}
|
||||
if(fixedSize())
|
||||
{
|
||||
CV_CheckEQ(m.dims, d, "");
|
||||
CV_CheckEQ(m.dims, d, "Can't reallocate UMat with locked size (probably due to misused 'const' modifier)");
|
||||
for(int j = 0; j < d; ++j)
|
||||
CV_CheckEQ(m.size[j], sizes[j], "");
|
||||
CV_CheckEQ(m.size[j], sizes[j], "Can't reallocate UMat with locked size (probably due to misused 'const' modifier)");
|
||||
}
|
||||
m.create(d, sizes, mtype);
|
||||
return;
|
||||
|
||||
@@ -40,6 +40,11 @@
|
||||
//M*/
|
||||
|
||||
#include "precomp.hpp"
|
||||
|
||||
#ifndef HAVE_OPENCL
|
||||
#include "ocl_disabled.impl.hpp"
|
||||
#else // HAVE_OPENCL
|
||||
|
||||
#include <list>
|
||||
#include <map>
|
||||
#include <deque>
|
||||
@@ -106,23 +111,7 @@
|
||||
#include "opencv2/core/opencl/runtime/opencl_clamdblas.hpp"
|
||||
#include "opencv2/core/opencl/runtime/opencl_clamdfft.hpp"
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
#include "opencv2/core/opencl/runtime/opencl_core.hpp"
|
||||
#else
|
||||
#if defined(_MSC_VER)
|
||||
#pragma warning(push)
|
||||
#pragma warning(disable : 4100)
|
||||
#pragma warning(disable : 4702)
|
||||
#elif defined(__clang__)
|
||||
#pragma clang diagnostic push
|
||||
#pragma clang diagnostic ignored "-Wunused-parameter"
|
||||
#elif defined(__GNUC__)
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-parameter"
|
||||
#endif
|
||||
// TODO FIXIT: This file can't be build without OPENCL
|
||||
#include "ocl_deprecated.hpp"
|
||||
#endif // HAVE_OPENCL
|
||||
|
||||
#ifdef HAVE_OPENCL_SVM
|
||||
#include "opencv2/core/opencl/runtime/opencl_svm_20.hpp"
|
||||
@@ -147,31 +136,6 @@ cv::utils::AllocatorStatisticsInterface& getOpenCLAllocatorStatistics()
|
||||
return opencl_allocator_stats;
|
||||
}
|
||||
|
||||
#ifndef HAVE_OPENCL
|
||||
#define CV_OPENCL_NO_SUPPORT() CV_Error(cv::Error::OpenCLApiCallError, "OpenCV build without OpenCL support")
|
||||
namespace {
|
||||
struct DummyImpl
|
||||
{
|
||||
DummyImpl() { CV_OPENCL_NO_SUPPORT(); }
|
||||
~DummyImpl() { /* do not throw in desctructors */ }
|
||||
IMPLEMENT_REFCOUNTABLE();
|
||||
};
|
||||
} // namespace
|
||||
|
||||
// TODO Replace to empty body (without HAVE_OPENCL)
|
||||
#define CV_OCL_TRACE_CHECK_RESULT(status, message) /* nothing */
|
||||
#define CV_OCL_API_ERROR_MSG(check_result, msg) cv::String()
|
||||
#define CV_OCL_CHECK_RESULT(check_result, msg) (void)check_result
|
||||
#define CV_OCL_CHECK_(expr, check_result) expr; (void)check_result
|
||||
#define CV_OCL_CHECK(expr) do { cl_int __cl_result = (expr); CV_OCL_CHECK_RESULT(__cl_result, #expr); } while (0)
|
||||
#define CV_OCL_DBG_CHECK_RESULT(check_result, msg) (void)check_result
|
||||
#define CV_OCL_DBG_CHECK_(expr, check_result) expr; (void)check_result
|
||||
#define CV_OCL_DBG_CHECK(expr) do { cl_int __cl_result = (expr); CV_OCL_CHECK_RESULT(__cl_result, #expr); } while (0)
|
||||
|
||||
static const bool CV_OPENCL_DISABLE_BUFFER_RECT_OPERATIONS = false;
|
||||
|
||||
#else // HAVE_OPENCL
|
||||
|
||||
#ifndef _DEBUG
|
||||
static bool isRaiseError()
|
||||
{
|
||||
@@ -270,7 +234,6 @@ static const String getBuildExtraOptions()
|
||||
static const bool CV_OPENCL_ENABLE_MEM_USE_HOST_PTR = utils::getConfigurationParameterBool("OPENCV_OPENCL_ENABLE_MEM_USE_HOST_PTR", true);
|
||||
static const size_t CV_OPENCL_ALIGNMENT_MEM_USE_HOST_PTR = utils::getConfigurationParameterSizeT("OPENCV_OPENCL_ALIGNMENT_MEM_USE_HOST_PTR", 4);
|
||||
|
||||
#endif // HAVE_OPENCL
|
||||
|
||||
struct UMat2D
|
||||
{
|
||||
@@ -331,7 +294,7 @@ static uint64 crc64( const uchar* data, size_t size, uint64 crc0=0 )
|
||||
return ~crc;
|
||||
}
|
||||
|
||||
#if defined HAVE_OPENCL && OPENCV_HAVE_FILESYSTEM_SUPPORT
|
||||
#if OPENCV_HAVE_FILESYSTEM_SUPPORT
|
||||
struct OpenCLBinaryCacheConfigurator
|
||||
{
|
||||
cv::String cache_path_;
|
||||
@@ -872,7 +835,6 @@ static bool g_isOpenCVActivated = false;
|
||||
bool haveOpenCL()
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
#ifdef HAVE_OPENCL
|
||||
static bool g_isOpenCLInitialized = false;
|
||||
static bool g_isOpenCLAvailable = false;
|
||||
|
||||
@@ -902,9 +864,6 @@ bool haveOpenCL()
|
||||
g_isOpenCLInitialized = true;
|
||||
}
|
||||
return g_isOpenCLAvailable;
|
||||
#else
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
bool useOpenCL()
|
||||
@@ -924,14 +883,12 @@ bool useOpenCL()
|
||||
return data.useOpenCL > 0;
|
||||
}
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
bool isOpenCLActivated()
|
||||
{
|
||||
if (!g_isOpenCVActivated)
|
||||
return false; // prevent unnecessary OpenCL activation via useOpenCL()->haveOpenCL() calls
|
||||
return useOpenCL();
|
||||
}
|
||||
#endif
|
||||
|
||||
void setUseOpenCL(bool flag)
|
||||
{
|
||||
@@ -1958,7 +1915,6 @@ static unsigned int getSVMCapabilitiesMask()
|
||||
} // namespace
|
||||
#endif
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
static size_t getProgramCountLimit()
|
||||
{
|
||||
static bool initialized = false;
|
||||
@@ -1970,7 +1926,6 @@ static size_t getProgramCountLimit()
|
||||
}
|
||||
return count;
|
||||
}
|
||||
#endif
|
||||
|
||||
struct Context::Impl
|
||||
{
|
||||
@@ -2800,7 +2755,7 @@ KernelArg KernelArg::Constant(const Mat& m)
|
||||
struct Kernel::Impl
|
||||
{
|
||||
Impl(const char* kname, const Program& prog) :
|
||||
refcount(1), handle(NULL), isInProgress(false), nu(0)
|
||||
refcount(1), handle(NULL), isInProgress(false), isAsyncRun(false), nu(0)
|
||||
{
|
||||
cl_program ph = (cl_program)prog.ptr();
|
||||
cl_int retval = 0;
|
||||
@@ -2877,6 +2832,7 @@ struct Kernel::Impl
|
||||
enum { MAX_ARRS = 16 };
|
||||
UMatData* u[MAX_ARRS];
|
||||
bool isInProgress;
|
||||
bool isAsyncRun; // true if kernel was scheduled in async mode
|
||||
int nu;
|
||||
std::list<Image2D> images;
|
||||
bool haveTempDstUMats;
|
||||
@@ -3156,13 +3112,45 @@ bool Kernel::run(int dims, size_t _globalsize[], size_t _localsize[],
|
||||
}
|
||||
|
||||
|
||||
static bool isRaiseErrorOnReuseAsyncKernel()
|
||||
{
|
||||
static bool initialized = false;
|
||||
static bool value = false;
|
||||
if (!initialized)
|
||||
{
|
||||
value = cv::utils::getConfigurationParameterBool("OPENCV_OPENCL_RAISE_ERROR_REUSE_ASYNC_KERNEL", false);
|
||||
initialized = true;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
bool Kernel::Impl::run(int dims, size_t globalsize[], size_t localsize[],
|
||||
bool sync, int64* timeNS, const Queue& q)
|
||||
{
|
||||
CV_INSTRUMENT_REGION_OPENCL_RUN(name.c_str());
|
||||
|
||||
if (!handle || isInProgress)
|
||||
if (!handle)
|
||||
{
|
||||
CV_LOG_ERROR(NULL, "OpenCL kernel has zero handle: " << name);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (isAsyncRun)
|
||||
{
|
||||
CV_LOG_ERROR(NULL, "OpenCL kernel can't be reused in async mode: " << name);
|
||||
if (isRaiseErrorOnReuseAsyncKernel())
|
||||
CV_Assert(0);
|
||||
return false; // OpenCV 5.0: raise error
|
||||
}
|
||||
isAsyncRun = !sync;
|
||||
|
||||
if (isInProgress)
|
||||
{
|
||||
CV_LOG_ERROR(NULL, "Previous OpenCL kernel launch is not finished: " << name);
|
||||
if (isRaiseErrorOnReuseAsyncKernel())
|
||||
CV_Assert(0);
|
||||
return false; // OpenCV 5.0: raise error
|
||||
}
|
||||
|
||||
cl_command_queue qq = getQueue(q);
|
||||
if (haveTempDstUMats)
|
||||
@@ -3553,8 +3541,6 @@ internal::ProgramEntry::operator ProgramSource&() const
|
||||
|
||||
/////////////////////////////////////////// Program /////////////////////////////////////////////
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
|
||||
static
|
||||
cv::String joinBuildOptions(const cv::String& a, const cv::String& b)
|
||||
{
|
||||
@@ -3968,10 +3954,6 @@ struct Program::Impl
|
||||
String sourceName_;
|
||||
};
|
||||
|
||||
#else // HAVE_OPENCL
|
||||
struct Program::Impl : public DummyImpl {};
|
||||
#endif // HAVE_OPENCL
|
||||
|
||||
|
||||
Program::Program() { p = 0; }
|
||||
|
||||
@@ -4014,7 +3996,6 @@ bool Program::create(const ProgramSource& src,
|
||||
p->release();
|
||||
p = NULL;
|
||||
}
|
||||
#ifdef HAVE_OPENCL
|
||||
p = new Impl(src, buildflags, errmsg);
|
||||
if(!p->handle)
|
||||
{
|
||||
@@ -4022,18 +4003,11 @@ bool Program::create(const ProgramSource& src,
|
||||
p = 0;
|
||||
}
|
||||
return p != 0;
|
||||
#else
|
||||
CV_OPENCL_NO_SUPPORT();
|
||||
#endif
|
||||
}
|
||||
|
||||
void* Program::ptr() const
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
return p ? p->handle : 0;
|
||||
#else
|
||||
CV_OPENCL_NO_SUPPORT();
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifndef OPENCV_REMOVE_DEPRECATED_API
|
||||
@@ -4056,44 +4030,30 @@ bool Program::write(String& bin) const
|
||||
|
||||
String Program::getPrefix() const
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
if(!p)
|
||||
return String();
|
||||
Context::Impl* ctx_ = Context::getDefault().getImpl();
|
||||
CV_Assert(ctx_);
|
||||
return cv::format("opencl=%s\nbuildflags=%s", ctx_->getPrefixString().c_str(), p->buildflags.c_str());
|
||||
#else
|
||||
CV_OPENCL_NO_SUPPORT();
|
||||
#endif
|
||||
}
|
||||
|
||||
String Program::getPrefix(const String& buildflags)
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
Context::Impl* ctx_ = Context::getDefault().getImpl();
|
||||
CV_Assert(ctx_);
|
||||
return cv::format("opencl=%s\nbuildflags=%s", ctx_->getPrefixString().c_str(), buildflags.c_str());
|
||||
#else
|
||||
CV_OPENCL_NO_SUPPORT();
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
#endif // OPENCV_REMOVE_DEPRECATED_API
|
||||
|
||||
void Program::getBinary(std::vector<char>& binary) const
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
CV_Assert(p && "Empty program");
|
||||
p->getProgramBinary(binary);
|
||||
#else
|
||||
binary.clear();
|
||||
CV_OPENCL_NO_SUPPORT();
|
||||
#endif
|
||||
}
|
||||
|
||||
Program Context::Impl::getProg(const ProgramSource& src,
|
||||
const String& buildflags, String& errmsg)
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
size_t limit = getProgramCountLimit();
|
||||
const ProgramSource::Impl* src_ = src.getImpl();
|
||||
CV_Assert(src_);
|
||||
@@ -4145,9 +4105,6 @@ Program Context::Impl::getProg(const ProgramSource& src,
|
||||
cacheList.push_front(key);
|
||||
}
|
||||
return prog;
|
||||
#else
|
||||
CV_OPENCL_NO_SUPPORT();
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -4707,9 +4664,6 @@ public:
|
||||
|
||||
bool allocate(UMatData* u, int accessFlags, UMatUsageFlags usageFlags) const CV_OVERRIDE
|
||||
{
|
||||
#ifndef HAVE_OPENCL
|
||||
return false;
|
||||
#else
|
||||
if(!u)
|
||||
return false;
|
||||
|
||||
@@ -4828,7 +4782,6 @@ public:
|
||||
u->markHostCopyObsolete(true);
|
||||
opencl_allocator_stats.onAllocate(u->size);
|
||||
return true;
|
||||
#endif // HAVE_OPENCL
|
||||
}
|
||||
|
||||
/*void sync(UMatData* u) const
|
||||
@@ -6458,6 +6411,9 @@ struct Image2D::Impl
|
||||
CV_Error(Error::OpenCLApiCallError, "OpenCL runtime not found!");
|
||||
|
||||
cl_context context = (cl_context)Context::getDefault().ptr();
|
||||
if (!context)
|
||||
return false;
|
||||
|
||||
// Figure out how many formats are supported by this context.
|
||||
cl_uint numFormats = 0;
|
||||
cl_int err = clGetSupportedImageFormats(context, CL_MEM_READ_WRITE,
|
||||
@@ -6696,27 +6652,19 @@ struct Timer::Impl
|
||||
|
||||
void start()
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
CV_OCL_DBG_CHECK(clFinish((cl_command_queue)queue.ptr()));
|
||||
timer.start();
|
||||
#endif
|
||||
}
|
||||
|
||||
void stop()
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
CV_OCL_DBG_CHECK(clFinish((cl_command_queue)queue.ptr()));
|
||||
timer.stop();
|
||||
#endif
|
||||
}
|
||||
|
||||
uint64 durationNS() const
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
return (uint64)(timer.getTimeSec() * 1e9);
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
TickMeter timer;
|
||||
@@ -6743,13 +6691,6 @@ uint64 Timer::durationNS() const
|
||||
return p->durationNS();
|
||||
}
|
||||
|
||||
#ifndef HAVE_OPENCL
|
||||
#if defined(_MSC_VER)
|
||||
#pragma warning(pop)
|
||||
#elif defined(__clang__)
|
||||
#pragma clang diagnostic pop
|
||||
#elif defined(__GNUC__)
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
#endif
|
||||
}} // namespace
|
||||
|
||||
#endif // HAVE_OPENCL
|
||||
|
||||
@@ -0,0 +1,366 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
|
||||
#include "opencv2/core/ocl_genbase.hpp"
|
||||
|
||||
#if defined(_MSC_VER)
|
||||
#pragma warning(push)
|
||||
#pragma warning(disable : 4100)
|
||||
#pragma warning(disable : 4702)
|
||||
#elif defined(__clang__)
|
||||
#pragma clang diagnostic push
|
||||
#pragma clang diagnostic ignored "-Wunused-parameter"
|
||||
#elif defined(__GNUC__)
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wunused-parameter"
|
||||
#endif
|
||||
|
||||
namespace cv { namespace ocl {
|
||||
|
||||
static
|
||||
CV_NORETURN void throw_no_ocl()
|
||||
{
|
||||
CV_Error(Error::OpenCLApiCallError, "OpenCV build without OpenCL support");
|
||||
}
|
||||
#define OCL_NOT_AVAILABLE() throw_no_ocl();
|
||||
|
||||
CV_EXPORTS_W bool haveOpenCL() { return false; }
|
||||
CV_EXPORTS_W bool useOpenCL() { return false; }
|
||||
CV_EXPORTS_W bool haveAmdBlas() { return false; }
|
||||
CV_EXPORTS_W bool haveAmdFft() { return false; }
|
||||
CV_EXPORTS_W void setUseOpenCL(bool flag) { /* nothing */ }
|
||||
CV_EXPORTS_W void finish() { /* nothing */ }
|
||||
|
||||
CV_EXPORTS bool haveSVM() { return false; }
|
||||
|
||||
Device::Device() : p(NULL) { }
|
||||
Device::Device(void* d) : p(NULL) { OCL_NOT_AVAILABLE(); }
|
||||
Device::Device(const Device& d) : p(NULL) { }
|
||||
Device& Device::operator=(const Device& d) { return *this; }
|
||||
Device::~Device() { }
|
||||
|
||||
void Device::set(void* d) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
String Device::name() const { OCL_NOT_AVAILABLE(); }
|
||||
String Device::extensions() const { OCL_NOT_AVAILABLE(); }
|
||||
bool Device::isExtensionSupported(const String& extensionName) const { OCL_NOT_AVAILABLE(); }
|
||||
String Device::version() const { OCL_NOT_AVAILABLE(); }
|
||||
String Device::vendorName() const { OCL_NOT_AVAILABLE(); }
|
||||
String Device::OpenCL_C_Version() const { OCL_NOT_AVAILABLE(); }
|
||||
String Device::OpenCLVersion() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::deviceVersionMajor() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::deviceVersionMinor() const { OCL_NOT_AVAILABLE(); }
|
||||
String Device::driverVersion() const { OCL_NOT_AVAILABLE(); }
|
||||
void* Device::ptr() const { /*OCL_NOT_AVAILABLE();*/ return NULL; }
|
||||
|
||||
int Device::type() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::addressBits() const { OCL_NOT_AVAILABLE(); }
|
||||
bool Device::available() const { OCL_NOT_AVAILABLE(); }
|
||||
bool Device::compilerAvailable() const { OCL_NOT_AVAILABLE(); }
|
||||
bool Device::linkerAvailable() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::doubleFPConfig() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::singleFPConfig() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::halfFPConfig() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
bool Device::endianLittle() const { OCL_NOT_AVAILABLE(); }
|
||||
bool Device::errorCorrectionSupport() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::executionCapabilities() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Device::globalMemCacheSize() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::globalMemCacheType() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::globalMemCacheLineSize() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Device::globalMemSize() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Device::localMemSize() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::localMemType() const { return NO_LOCAL_MEM; }
|
||||
bool Device::hostUnifiedMemory() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
bool Device::imageSupport() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
bool Device::imageFromBufferSupport() const { OCL_NOT_AVAILABLE(); }
|
||||
uint Device::imagePitchAlignment() const { OCL_NOT_AVAILABLE(); }
|
||||
uint Device::imageBaseAddressAlignment() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
bool Device::intelSubgroupsSupport() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Device::image2DMaxWidth() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Device::image2DMaxHeight() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Device::image3DMaxWidth() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Device::image3DMaxHeight() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Device::image3DMaxDepth() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Device::imageMaxBufferSize() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Device::imageMaxArraySize() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::vendorID() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::maxClockFrequency() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::maxComputeUnits() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::maxConstantArgs() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Device::maxConstantBufferSize() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Device::maxMemAllocSize() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Device::maxParameterSize() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::maxReadImageArgs() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::maxWriteImageArgs() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::maxSamplers() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Device::maxWorkGroupSize() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::maxWorkItemDims() const { OCL_NOT_AVAILABLE(); }
|
||||
void Device::maxWorkItemSizes(size_t*) const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::memBaseAddrAlign() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::nativeVectorWidthChar() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::nativeVectorWidthShort() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::nativeVectorWidthInt() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::nativeVectorWidthLong() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::nativeVectorWidthFloat() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::nativeVectorWidthDouble() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::nativeVectorWidthHalf() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Device::preferredVectorWidthChar() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::preferredVectorWidthShort() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::preferredVectorWidthInt() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::preferredVectorWidthLong() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::preferredVectorWidthFloat() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::preferredVectorWidthDouble() const { OCL_NOT_AVAILABLE(); }
|
||||
int Device::preferredVectorWidthHalf() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Device::printfBufferSize() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Device::profilingTimerResolution() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
/* static */
|
||||
const Device& Device::getDefault()
|
||||
{
|
||||
static Device dummy;
|
||||
return dummy;
|
||||
}
|
||||
|
||||
|
||||
Context::Context() : p(NULL) { }
|
||||
Context::Context(int dtype) : p(NULL) { }
|
||||
Context::~Context() { }
|
||||
Context::Context(const Context& c) : p(NULL) { }
|
||||
Context& Context::operator=(const Context& c) { return *this; }
|
||||
|
||||
bool Context::create() { return false; }
|
||||
bool Context::create(int dtype) { return false; }
|
||||
size_t Context::ndevices() const { return 0; }
|
||||
const Device& Context::device(size_t idx) const { OCL_NOT_AVAILABLE(); }
|
||||
Program Context::getProg(const ProgramSource& prog, const String& buildopt, String& errmsg) { OCL_NOT_AVAILABLE(); }
|
||||
void Context::unloadProg(Program& prog) { }
|
||||
|
||||
/* static */
|
||||
Context& Context::getDefault(bool initialize)
|
||||
{
|
||||
static Context dummy;
|
||||
return dummy;
|
||||
}
|
||||
void* Context::ptr() const { return NULL; }
|
||||
|
||||
bool Context::useSVM() const { return false; }
|
||||
void Context::setUseSVM(bool enabled) { }
|
||||
|
||||
Platform::Platform() : p(NULL) { }
|
||||
Platform::~Platform() { }
|
||||
Platform::Platform(const Platform&) : p(NULL) { }
|
||||
Platform& Platform::operator=(const Platform&) { return *this; }
|
||||
|
||||
void* Platform::ptr() const { return NULL; }
|
||||
|
||||
/* static */
|
||||
Platform& Platform::getDefault()
|
||||
{
|
||||
static Platform dummy;
|
||||
return dummy;
|
||||
}
|
||||
|
||||
void attachContext(const String& platformName, void* platformID, void* context, void* deviceID) { OCL_NOT_AVAILABLE(); }
|
||||
void convertFromBuffer(void* cl_mem_buffer, size_t step, int rows, int cols, int type, UMat& dst) { OCL_NOT_AVAILABLE(); }
|
||||
void convertFromImage(void* cl_mem_image, UMat& dst) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
void initializeContextFromHandle(Context& ctx, void* platform, void* context, void* device) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
Queue::Queue() : p(NULL) { }
|
||||
Queue::Queue(const Context& c, const Device& d) : p(NULL) { OCL_NOT_AVAILABLE(); }
|
||||
Queue::~Queue() { }
|
||||
Queue::Queue(const Queue& q) {}
|
||||
Queue& Queue::operator=(const Queue& q) { return *this; }
|
||||
|
||||
bool Queue::create(const Context& c, const Device& d) { OCL_NOT_AVAILABLE(); }
|
||||
void Queue::finish() {}
|
||||
void* Queue::ptr() const { return NULL; }
|
||||
/* static */
|
||||
Queue& Queue::getDefault()
|
||||
{
|
||||
static Queue dummy;
|
||||
return dummy;
|
||||
}
|
||||
|
||||
/// @brief Returns OpenCL command queue with enable profiling mode support
|
||||
const Queue& Queue::getProfilingQueue() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
|
||||
KernelArg::KernelArg()
|
||||
: flags(0), m(0), obj(0), sz(0), wscale(1), iwscale(1)
|
||||
{
|
||||
}
|
||||
|
||||
KernelArg::KernelArg(int _flags, UMat* _m, int _wscale, int _iwscale, const void* _obj, size_t _sz)
|
||||
: flags(_flags), m(_m), obj(_obj), sz(_sz), wscale(_wscale), iwscale(_iwscale)
|
||||
{
|
||||
OCL_NOT_AVAILABLE();
|
||||
}
|
||||
|
||||
KernelArg KernelArg::Constant(const Mat& m)
|
||||
{
|
||||
OCL_NOT_AVAILABLE();
|
||||
}
|
||||
|
||||
|
||||
Kernel::Kernel() : p(NULL) { }
|
||||
Kernel::Kernel(const char* kname, const Program& prog) : p(NULL) { OCL_NOT_AVAILABLE(); }
|
||||
Kernel::Kernel(const char* kname, const ProgramSource& prog, const String& buildopts, String* errmsg) : p(NULL) { OCL_NOT_AVAILABLE(); }
|
||||
Kernel::~Kernel() { }
|
||||
Kernel::Kernel(const Kernel& k) : p(NULL) { }
|
||||
Kernel& Kernel::operator=(const Kernel& k) { return *this; }
|
||||
|
||||
bool Kernel::empty() const { return true; }
|
||||
bool Kernel::create(const char* kname, const Program& prog) { OCL_NOT_AVAILABLE(); }
|
||||
bool Kernel::create(const char* kname, const ProgramSource& prog, const String& buildopts, String* errmsg) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int Kernel::set(int i, const void* value, size_t sz) { OCL_NOT_AVAILABLE(); }
|
||||
int Kernel::set(int i, const Image2D& image2D) { OCL_NOT_AVAILABLE(); }
|
||||
int Kernel::set(int i, const UMat& m) { OCL_NOT_AVAILABLE(); }
|
||||
int Kernel::set(int i, const KernelArg& arg) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
bool Kernel::run(int dims, size_t globalsize[], size_t localsize[], bool sync, const Queue& q) { OCL_NOT_AVAILABLE(); }
|
||||
bool Kernel::runTask(bool sync, const Queue& q) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int64 Kernel::runProfiling(int dims, size_t globalsize[], size_t localsize[], const Queue& q) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
size_t Kernel::workGroupSize() const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Kernel::preferedWorkGroupSizeMultiple() const { OCL_NOT_AVAILABLE(); }
|
||||
bool Kernel::compileWorkGroupSize(size_t wsz[]) const { OCL_NOT_AVAILABLE(); }
|
||||
size_t Kernel::localMemSize() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
void* Kernel::ptr() const { return NULL; }
|
||||
|
||||
|
||||
Program::Program() : p(NULL) { }
|
||||
Program::Program(const ProgramSource& src, const String& buildflags, String& errmsg) : p(NULL) { OCL_NOT_AVAILABLE(); }
|
||||
Program::Program(const Program& prog) : p(NULL) { }
|
||||
Program& Program::operator=(const Program& prog) { return *this; }
|
||||
Program::~Program() { }
|
||||
|
||||
bool Program::create(const ProgramSource& src, const String& buildflags, String& errmsg) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
void* Program::ptr() const { return NULL; }
|
||||
|
||||
void Program::getBinary(std::vector<char>& binary) const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
bool Program::read(const String& buf, const String& buildflags) { OCL_NOT_AVAILABLE(); }
|
||||
bool Program::write(String& buf) const { OCL_NOT_AVAILABLE(); }
|
||||
const ProgramSource& Program::source() const { OCL_NOT_AVAILABLE(); }
|
||||
String Program::getPrefix() const { OCL_NOT_AVAILABLE(); }
|
||||
/* static */ String Program::getPrefix(const String& buildflags) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
|
||||
ProgramSource::ProgramSource() : p(NULL) { }
|
||||
ProgramSource::ProgramSource(const String& module, const String& name, const String& codeStr, const String& codeHash) : p(NULL) { }
|
||||
ProgramSource::ProgramSource(const String& prog) : p(NULL) { }
|
||||
ProgramSource::ProgramSource(const char* prog) : p(NULL) { }
|
||||
ProgramSource::~ProgramSource() { }
|
||||
ProgramSource::ProgramSource(const ProgramSource& prog) : p(NULL) { }
|
||||
ProgramSource& ProgramSource::operator=(const ProgramSource& prog) { return *this; }
|
||||
|
||||
const String& ProgramSource::source() const { OCL_NOT_AVAILABLE(); }
|
||||
ProgramSource::hash_t ProgramSource::hash() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
/* static */ ProgramSource ProgramSource::fromBinary(const String& module, const String& name, const unsigned char* binary, const size_t size, const cv::String& buildOptions) { OCL_NOT_AVAILABLE(); }
|
||||
/* static */ ProgramSource ProgramSource::fromSPIR(const String& module, const String& name, const unsigned char* binary, const size_t size, const cv::String& buildOptions) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
|
||||
PlatformInfo::PlatformInfo() : p(NULL) { }
|
||||
PlatformInfo::PlatformInfo(void* id) : p(NULL) { OCL_NOT_AVAILABLE(); }
|
||||
PlatformInfo::~PlatformInfo() { }
|
||||
|
||||
PlatformInfo::PlatformInfo(const PlatformInfo& i) : p(NULL) { }
|
||||
PlatformInfo& PlatformInfo::operator=(const PlatformInfo& i) { return *this; }
|
||||
|
||||
String PlatformInfo::name() const { OCL_NOT_AVAILABLE(); }
|
||||
String PlatformInfo::vendor() const { OCL_NOT_AVAILABLE(); }
|
||||
String PlatformInfo::version() const { OCL_NOT_AVAILABLE(); }
|
||||
int PlatformInfo::deviceNumber() const { OCL_NOT_AVAILABLE(); }
|
||||
void PlatformInfo::getDevice(Device& device, int d) const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
const char* convertTypeStr(int sdepth, int ddepth, int cn, char* buf) { OCL_NOT_AVAILABLE(); }
|
||||
const char* typeToStr(int t) { OCL_NOT_AVAILABLE(); }
|
||||
const char* memopTypeToStr(int t) { OCL_NOT_AVAILABLE(); }
|
||||
const char* vecopTypeToStr(int t) { OCL_NOT_AVAILABLE(); }
|
||||
const char* getOpenCLErrorString(int errorCode) { OCL_NOT_AVAILABLE(); }
|
||||
String kernelToStr(InputArray _kernel, int ddepth, const char* name) { OCL_NOT_AVAILABLE(); }
|
||||
void getPlatfomsInfo(std::vector<PlatformInfo>& platform_info) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
|
||||
int predictOptimalVectorWidth(InputArray src1, InputArray src2, InputArray src3,
|
||||
InputArray src4, InputArray src5, InputArray src6,
|
||||
InputArray src7, InputArray src8, InputArray src9,
|
||||
OclVectorStrategy strat)
|
||||
{ OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int checkOptimalVectorWidth(const int *vectorWidths,
|
||||
InputArray src1, InputArray src2, InputArray src3,
|
||||
InputArray src4, InputArray src5, InputArray src6,
|
||||
InputArray src7, InputArray src8, InputArray src9,
|
||||
OclVectorStrategy strat)
|
||||
{ OCL_NOT_AVAILABLE(); }
|
||||
|
||||
int predictOptimalVectorWidthMax(InputArray src1, InputArray src2, InputArray src3,
|
||||
InputArray src4, InputArray src5, InputArray src6,
|
||||
InputArray src7, InputArray src8, InputArray src9)
|
||||
{ OCL_NOT_AVAILABLE(); }
|
||||
|
||||
void buildOptionsAddMatrixDescription(String& buildOptions, const String& name, InputArray _m) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
|
||||
Image2D::Image2D() : p(NULL) { }
|
||||
Image2D::Image2D(const UMat &src, bool norm, bool alias) { OCL_NOT_AVAILABLE(); }
|
||||
Image2D::Image2D(const Image2D & i) : p(NULL) { OCL_NOT_AVAILABLE(); }
|
||||
Image2D::~Image2D() { }
|
||||
Image2D& Image2D::operator=(const Image2D & i) { return *this; }
|
||||
|
||||
/* static */ bool Image2D::canCreateAlias(const UMat &u) { OCL_NOT_AVAILABLE(); }
|
||||
/* static */ bool Image2D::isFormatSupported(int depth, int cn, bool norm) { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
void* Image2D::ptr() const { return NULL; }
|
||||
|
||||
|
||||
Timer::Timer(const Queue& q) : p(NULL) {}
|
||||
Timer::~Timer() {}
|
||||
void Timer::start() { OCL_NOT_AVAILABLE(); }
|
||||
void Timer::stop() { OCL_NOT_AVAILABLE();}
|
||||
|
||||
uint64 Timer::durationNS() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
MatAllocator* getOpenCLAllocator() { return NULL; }
|
||||
|
||||
internal::ProgramEntry::operator ProgramSource&() const { OCL_NOT_AVAILABLE(); }
|
||||
|
||||
}}
|
||||
|
||||
#if defined(_MSC_VER)
|
||||
#pragma warning(pop)
|
||||
#elif defined(__clang__)
|
||||
#pragma clang diagnostic pop
|
||||
#elif defined(__GNUC__)
|
||||
#pragma GCC diagnostic pop
|
||||
#endif
|
||||
@@ -54,7 +54,8 @@
|
||||
#endif
|
||||
|
||||
#if defined __linux__ || defined __APPLE__ || defined __GLIBC__ \
|
||||
|| defined __HAIKU__ || defined __EMSCRIPTEN__ || defined __FreeBSD__
|
||||
|| defined __HAIKU__ || defined __EMSCRIPTEN__ || defined __FreeBSD__ \
|
||||
|| defined __OpenBSD__
|
||||
#include <unistd.h>
|
||||
#include <stdio.h>
|
||||
#include <sys/types.h>
|
||||
|
||||
@@ -82,7 +82,12 @@ static bool getParameterTraceEnable()
|
||||
static int param_maxRegionDepthOpenCV = (int)utils::getConfigurationParameterSizeT("OPENCV_TRACE_DEPTH_OPENCV", 1);
|
||||
static int param_maxRegionChildrenOpenCV = (int)utils::getConfigurationParameterSizeT("OPENCV_TRACE_MAX_CHILDREN_OPENCV", 1000);
|
||||
static int param_maxRegionChildren = (int)utils::getConfigurationParameterSizeT("OPENCV_TRACE_MAX_CHILDREN", 10000);
|
||||
static cv::String param_traceLocation = utils::getConfigurationParameterString("OPENCV_TRACE_LOCATION", "OpenCVTrace");
|
||||
|
||||
static const cv::String& getParameterTraceLocation()
|
||||
{
|
||||
static cv::String param_traceLocation = utils::getConfigurationParameterString("OPENCV_TRACE_LOCATION", "OpenCVTrace");
|
||||
return param_traceLocation;
|
||||
}
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
static bool param_synchronizeOpenCL = utils::getConfigurationParameterBool("OPENCV_TRACE_SYNC_OPENCL", false);
|
||||
@@ -813,7 +818,7 @@ TraceStorage* TraceManagerThreadLocal::getStorage() const
|
||||
TraceStorage* global = getTraceManager().trace_storage.get();
|
||||
if (global)
|
||||
{
|
||||
const std::string filepath = cv::format("%s-%03d.txt", param_traceLocation.c_str(), threadID).c_str();
|
||||
const std::string filepath = cv::format("%s-%03d.txt", getParameterTraceLocation().c_str(), threadID).c_str();
|
||||
TraceMessage msg;
|
||||
const char* pos = strrchr(filepath.c_str(), '/'); // extract filename
|
||||
#ifdef _WIN32
|
||||
@@ -848,7 +853,7 @@ TraceManager::TraceManager()
|
||||
activated = getParameterTraceEnable();
|
||||
|
||||
if (activated)
|
||||
trace_storage.reset(new SyncTraceStorage(std::string(param_traceLocation) + ".txt"));
|
||||
trace_storage.reset(new SyncTraceStorage(std::string(getParameterTraceLocation()) + ".txt"));
|
||||
|
||||
#ifdef OPENCV_WITH_ITT
|
||||
if (isITTEnabled())
|
||||
|
||||
@@ -80,6 +80,7 @@ UMatData::~UMatData()
|
||||
CV_Assert(mapcount == 0);
|
||||
data = origdata = 0;
|
||||
size = 0;
|
||||
bool isAsyncCleanup = !!(flags & UMatData::ASYNC_CLEANUP);
|
||||
flags = 0;
|
||||
handle = 0;
|
||||
userdata = 0;
|
||||
@@ -106,7 +107,7 @@ UMatData::~UMatData()
|
||||
showWarn = true;
|
||||
if (zero_Ref && zero_URef) // oops, we need to free resources
|
||||
{
|
||||
showWarn = true;
|
||||
showWarn = !isAsyncCleanup;
|
||||
// simulate UMat::deallocate
|
||||
u->currAllocator->deallocate(u);
|
||||
}
|
||||
|
||||
@@ -2185,4 +2185,32 @@ TEST(Mat, empty_iterator_16855)
|
||||
EXPECT_TRUE(m.begin<uchar>() == m.end<uchar>());
|
||||
}
|
||||
|
||||
|
||||
TEST(Mat, regression_18473)
|
||||
{
|
||||
std::vector<int> sizes(3);
|
||||
sizes[0] = 20;
|
||||
sizes[1] = 50;
|
||||
sizes[2] = 100;
|
||||
#if 1 // with the fix
|
||||
std::vector<size_t> steps(2);
|
||||
steps[0] = 50*100*2;
|
||||
steps[1] = 100*2;
|
||||
#else // without the fix
|
||||
std::vector<size_t> steps(3);
|
||||
steps[0] = 50*100*2;
|
||||
steps[1] = 100*2;
|
||||
steps[2] = 2;
|
||||
#endif
|
||||
std::vector<short> data(20*50*100, 0); // 1Mb
|
||||
data[data.size() - 1] = 5;
|
||||
|
||||
// param steps Array of ndims-1 steps
|
||||
Mat m(sizes, CV_16SC1, (void*)data.data(), (const size_t*)steps.data());
|
||||
|
||||
ASSERT_FALSE(m.empty());
|
||||
EXPECT_EQ((int)5, (int)m.at<short>(19, 49, 99));
|
||||
}
|
||||
|
||||
|
||||
}} // namespace
|
||||
|
||||
@@ -1154,6 +1154,30 @@ TEST(UMat, map_unmap_counting)
|
||||
}
|
||||
|
||||
|
||||
static void process_with_async_cleanup(Mat& frame)
|
||||
{
|
||||
UMat blurResult;
|
||||
{
|
||||
UMat umat_buffer = frame.getUMat(ACCESS_READ);
|
||||
cv::blur(umat_buffer, blurResult, Size(3, 3)); // UMat doesn't support inplace, this call is not synchronized
|
||||
}
|
||||
Mat result;
|
||||
blurResult.copyTo(result);
|
||||
swap(result, frame);
|
||||
// umat_buffer cleanup is done asynchronously, silence warning about original 'frame' cleanup here (through 'result')
|
||||
// - release input 'frame' (as 'result')
|
||||
// - release 'umat_buffer' asynchronously and silence warning about "parent" buffer (in debug builds)
|
||||
}
|
||||
TEST(UMat, async_cleanup_without_call_chain_warning)
|
||||
{
|
||||
Mat frame(Size(640, 480), CV_8UC1, Scalar::all(128));
|
||||
for (int i = 0; i < 10; i++)
|
||||
{
|
||||
process_with_async_cleanup(frame);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
///////////// oclCleanupCallback threadsafe check (#5062) /////////////////////
|
||||
|
||||
// Case 1: reuse of old src Mat in OCL pipe. Hard to catch!
|
||||
|
||||
@@ -102,6 +102,34 @@ namespace
|
||||
cudaSafeCall( cudaDeviceSynchronize() );
|
||||
}
|
||||
};
|
||||
|
||||
template <int DEPTH> struct NppMirrorIFunc
|
||||
{
|
||||
typedef typename NppTypeTraits<DEPTH>::npp_t npp_t;
|
||||
|
||||
typedef NppStatus (*func_t)(npp_t* pSrcDst, int nSrcDstStep, NppiSize oROI, NppiAxis flip);
|
||||
};
|
||||
|
||||
template <int DEPTH, typename NppMirrorIFunc<DEPTH>::func_t func> struct NppMirrorI
|
||||
{
|
||||
typedef typename NppMirrorIFunc<DEPTH>::npp_t npp_t;
|
||||
|
||||
static void call(GpuMat& srcDst, int flipCode, cudaStream_t stream)
|
||||
{
|
||||
NppStreamHandler h(stream);
|
||||
|
||||
NppiSize sz;
|
||||
sz.width = srcDst.cols;
|
||||
sz.height = srcDst.rows;
|
||||
|
||||
nppSafeCall( func(srcDst.ptr<npp_t>(), static_cast<int>(srcDst.step),
|
||||
sz,
|
||||
(flipCode == 0 ? NPP_HORIZONTAL_AXIS : (flipCode > 0 ? NPP_VERTICAL_AXIS : NPP_BOTH_AXIS))) );
|
||||
|
||||
if (stream == 0)
|
||||
cudaSafeCall( cudaDeviceSynchronize() );
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
void cv::cuda::flip(InputArray _src, OutputArray _dst, int flipCode, Stream& stream)
|
||||
@@ -117,6 +145,17 @@ void cv::cuda::flip(InputArray _src, OutputArray _dst, int flipCode, Stream& str
|
||||
{NppMirror<CV_32F, nppiMirror_32f_C1R>::call, 0, NppMirror<CV_32F, nppiMirror_32f_C3R>::call, NppMirror<CV_32F, nppiMirror_32f_C4R>::call}
|
||||
};
|
||||
|
||||
typedef void (*ifunc_t)(GpuMat& srcDst, int flipCode, cudaStream_t stream);
|
||||
static const ifunc_t ifuncs[6][4] =
|
||||
{
|
||||
{NppMirrorI<CV_8U, nppiMirror_8u_C1IR>::call, 0, NppMirrorI<CV_8U, nppiMirror_8u_C3IR>::call, NppMirrorI<CV_8U, nppiMirror_8u_C4IR>::call},
|
||||
{0,0,0,0},
|
||||
{NppMirrorI<CV_16U, nppiMirror_16u_C1IR>::call, 0, NppMirrorI<CV_16U, nppiMirror_16u_C3IR>::call, NppMirrorI<CV_16U, nppiMirror_16u_C4IR>::call},
|
||||
{0,0,0,0},
|
||||
{NppMirrorI<CV_32S, nppiMirror_32s_C1IR>::call, 0, NppMirrorI<CV_32S, nppiMirror_32s_C3IR>::call, NppMirrorI<CV_32S, nppiMirror_32s_C4IR>::call},
|
||||
{NppMirrorI<CV_32F, nppiMirror_32f_C1IR>::call, 0, NppMirrorI<CV_32F, nppiMirror_32f_C3IR>::call, NppMirrorI<CV_32F, nppiMirror_32f_C4IR>::call}
|
||||
};
|
||||
|
||||
GpuMat src = getInputMat(_src, stream);
|
||||
|
||||
CV_Assert(src.depth() == CV_8U || src.depth() == CV_16U || src.depth() == CV_32S || src.depth() == CV_32F);
|
||||
@@ -124,8 +163,15 @@ void cv::cuda::flip(InputArray _src, OutputArray _dst, int flipCode, Stream& str
|
||||
|
||||
_dst.create(src.size(), src.type());
|
||||
GpuMat dst = getOutputMat(_dst, src.size(), src.type(), stream);
|
||||
bool isInplace = (src.data == dst.data) || (src.refcount == dst.refcount);
|
||||
bool isSizeOdd = (src.cols & 1) == 1 || (src.rows & 1) == 1;
|
||||
if (isInplace && isSizeOdd)
|
||||
CV_Error(Error::BadROISize, "In-place version of flip only accepts even width/height");
|
||||
|
||||
funcs[src.depth()][src.channels() - 1](src, dst, flipCode, StreamAccessor::getStream(stream));
|
||||
if (isInplace == false)
|
||||
funcs[src.depth()][src.channels() - 1](src, dst, flipCode, StreamAccessor::getStream(stream));
|
||||
else // in-place
|
||||
ifuncs[src.depth()][src.channels() - 1](src, flipCode, StreamAccessor::getStream(stream));
|
||||
|
||||
syncOutput(dst, _dst, stream);
|
||||
}
|
||||
|
||||
@@ -279,6 +279,24 @@ CUDA_TEST_P(Flip, Accuracy)
|
||||
EXPECT_MAT_NEAR(dst_gold, dst, 0.0);
|
||||
}
|
||||
|
||||
CUDA_TEST_P(Flip, AccuracyInplace)
|
||||
{
|
||||
cv::Mat src = randomMat(size, type);
|
||||
bool isSizeOdd = ((size.width & 1) == 1) || ((size.height & 1) == 1);
|
||||
cv::cuda::GpuMat srcDst = loadMat(src, useRoi);
|
||||
if(isSizeOdd)
|
||||
{
|
||||
EXPECT_THROW(cv::cuda::flip(srcDst, srcDst, flip_code), cv::Exception);
|
||||
return;
|
||||
}
|
||||
cv::cuda::flip(srcDst, srcDst, flip_code);
|
||||
|
||||
cv::Mat dst_gold;
|
||||
cv::flip(src, dst_gold, flip_code);
|
||||
|
||||
EXPECT_MAT_NEAR(dst_gold, srcDst, 0.0);
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(CUDA_Arithm, Flip, testing::Combine(
|
||||
ALL_DEVICES,
|
||||
DIFFERENT_SIZES,
|
||||
|
||||
@@ -2778,7 +2778,7 @@ CUDA_TEST_P(PolarToCart, Accuracy)
|
||||
{
|
||||
cv::Mat magnitude = randomMat(size, type);
|
||||
cv::Mat angle = randomMat(size, type);
|
||||
const double tol = (type == CV_32FC1 ? 1.6e-4 : 1e-4) * (angleInDegrees ? 1.0 : 19.0);
|
||||
const double tol = (type == CV_32FC1 ? 1.6e-4 : 1e-4) * (angleInDegrees ? 1.0 : 19.47);
|
||||
|
||||
cv::cuda::GpuMat x = createMat(size, type, useRoi);
|
||||
cv::cuda::GpuMat y = createMat(size, type, useRoi);
|
||||
|
||||
@@ -320,6 +320,65 @@ CUDA_TEST_P(GpuMat_ConvertTo, WithScaling)
|
||||
}
|
||||
}
|
||||
|
||||
CUDA_TEST_P(GpuMat_ConvertTo, InplaceWithOutScaling)
|
||||
{
|
||||
cv::Mat src = randomMat(size, depth1);
|
||||
|
||||
if ((depth1 == CV_64F || depth2 == CV_64F) && !supportFeature(devInfo, cv::cuda::NATIVE_DOUBLE))
|
||||
{
|
||||
try
|
||||
{
|
||||
cv::cuda::GpuMat d_srcDst = loadMat(src);
|
||||
d_srcDst.convertTo(d_srcDst, depth2);
|
||||
}
|
||||
catch (const cv::Exception& e)
|
||||
{
|
||||
ASSERT_EQ(cv::Error::StsUnsupportedFormat, e.code);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
cv::cuda::GpuMat d_srcDst = loadMat(src, useRoi);
|
||||
d_srcDst.convertTo(d_srcDst, depth2);
|
||||
|
||||
cv::Mat dst_gold;
|
||||
src.convertTo(dst_gold, depth2);
|
||||
|
||||
EXPECT_MAT_NEAR(dst_gold, d_srcDst, depth2 < CV_32F ? 1.0 : 1e-4);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
CUDA_TEST_P(GpuMat_ConvertTo, InplaceWithScaling)
|
||||
{
|
||||
cv::Mat src = randomMat(size, depth1);
|
||||
double a = randomDouble(0.0, 1.0);
|
||||
double b = randomDouble(-10.0, 10.0);
|
||||
|
||||
if ((depth1 == CV_64F || depth2 == CV_64F) && !supportFeature(devInfo, cv::cuda::NATIVE_DOUBLE))
|
||||
{
|
||||
try
|
||||
{
|
||||
cv::cuda::GpuMat d_srcDst = loadMat(src);
|
||||
d_srcDst.convertTo(d_srcDst, depth2, a, b);
|
||||
}
|
||||
catch (const cv::Exception& e)
|
||||
{
|
||||
ASSERT_EQ(cv::Error::StsUnsupportedFormat, e.code);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
cv::cuda::GpuMat d_srcDst = loadMat(src, useRoi);
|
||||
d_srcDst.convertTo(d_srcDst, depth2, a, b);
|
||||
|
||||
cv::Mat dst_gold;
|
||||
src.convertTo(dst_gold, depth2, a, b);
|
||||
|
||||
EXPECT_MAT_NEAR(dst_gold, d_srcDst, depth2 < CV_32F ? 1.0 : 1e-4);
|
||||
}
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(CUDA, GpuMat_ConvertTo, testing::Combine(
|
||||
ALL_DEVICES,
|
||||
DIFFERENT_SIZES,
|
||||
|
||||
@@ -257,18 +257,15 @@ namespace hist
|
||||
|
||||
namespace hist
|
||||
{
|
||||
__constant__ int c_lut[256];
|
||||
|
||||
struct EqualizeHist : unary_function<uchar, uchar>
|
||||
{
|
||||
float scale;
|
||||
const uchar* lut;
|
||||
|
||||
__host__ EqualizeHist(float _scale) : scale(_scale) {}
|
||||
__host__ EqualizeHist(const uchar* _lut) : lut(_lut) {}
|
||||
|
||||
__device__ __forceinline__ uchar operator ()(uchar val) const
|
||||
{
|
||||
const int lut = c_lut[val];
|
||||
return __float2int_rn(scale * lut);
|
||||
return lut[val];
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -283,16 +280,137 @@ namespace cv { namespace cuda { namespace device
|
||||
|
||||
namespace hist
|
||||
{
|
||||
void equalizeHist(PtrStepSzb src, PtrStepSzb dst, const int* lut, cudaStream_t stream)
|
||||
void equalizeHist(PtrStepSzb src, PtrStepSzb dst, const uchar* lut, cudaStream_t stream)
|
||||
{
|
||||
device::transform(src, dst, EqualizeHist(lut), WithOutMask(), stream);
|
||||
}
|
||||
|
||||
__global__ void buildLutKernel(int* hist, unsigned char* lut, int size)
|
||||
{
|
||||
__shared__ int warp_smem[8];
|
||||
__shared__ int hist_smem[8][33];
|
||||
|
||||
#define HIST_SMEM_NO_BANK_CONFLICT(idx) hist_smem[(idx) >> 5][(idx) & 31]
|
||||
|
||||
const int tId = threadIdx.x;
|
||||
const int warpId = threadIdx.x / 32;
|
||||
const int laneId = threadIdx.x % 32;
|
||||
|
||||
// Step1 - Find minimum non-zero value in hist and make it zero
|
||||
HIST_SMEM_NO_BANK_CONFLICT(tId) = hist[tId];
|
||||
int nonZeroIdx = HIST_SMEM_NO_BANK_CONFLICT(tId) > 0 ? tId : 256;
|
||||
|
||||
__syncthreads();
|
||||
|
||||
for (int delta = 16; delta > 0; delta /= 2)
|
||||
{
|
||||
#if __CUDACC_VER_MAJOR__ >= 9
|
||||
int shflVal = __shfl_down_sync(0xFFFFFFFF, nonZeroIdx, delta);
|
||||
#else
|
||||
int shflVal = __shfl_down(nonZeroIdx, delta);
|
||||
#endif
|
||||
if (laneId < delta)
|
||||
nonZeroIdx = min(nonZeroIdx, shflVal);
|
||||
}
|
||||
|
||||
if (laneId == 0)
|
||||
warp_smem[warpId] = nonZeroIdx;
|
||||
|
||||
__syncthreads();
|
||||
|
||||
if (tId < 8)
|
||||
{
|
||||
int warpVal = warp_smem[tId];
|
||||
for (int delta = 4; delta > 0; delta /= 2)
|
||||
{
|
||||
#if __CUDACC_VER_MAJOR__ >= 9
|
||||
int shflVal = __shfl_down_sync(0x000000FF, warpVal, delta);
|
||||
#else
|
||||
int shflVal = __shfl_down(warpVal, delta);
|
||||
#endif
|
||||
if (tId < delta)
|
||||
warpVal = min(warpVal, shflVal);
|
||||
}
|
||||
if (tId == 0)
|
||||
{
|
||||
warp_smem[0] = warpVal; // warpVal - minimum index
|
||||
}
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
const int minNonZeroIdx = warp_smem[0];
|
||||
const int minNonZeroVal = HIST_SMEM_NO_BANK_CONFLICT(minNonZeroIdx);
|
||||
if (minNonZeroVal == size)
|
||||
{
|
||||
// This is a special case: the whole image has the same color
|
||||
|
||||
lut[tId] = 0;
|
||||
if (tId == minNonZeroIdx)
|
||||
lut[tId] = minNonZeroIdx;
|
||||
return;
|
||||
}
|
||||
|
||||
if (tId == 0)
|
||||
HIST_SMEM_NO_BANK_CONFLICT(minNonZeroIdx) = 0;
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Step2 - Inclusive sum
|
||||
// Algorithm from GPU Gems 3 (A Work-Efficient Parallel Scan)
|
||||
// https://developer.nvidia.com/gpugems/gpugems3/part-vi-gpu-computing/chapter-39-parallel-prefix-sum-scan-cuda
|
||||
|
||||
// Step2 Phase1 - The Up-Sweep Phase
|
||||
for (int delta = 1; delta < 256; delta *= 2)
|
||||
{
|
||||
if (tId < 128 / delta)
|
||||
{
|
||||
int idx = 255 - 2 * tId * delta;
|
||||
HIST_SMEM_NO_BANK_CONFLICT(idx) += HIST_SMEM_NO_BANK_CONFLICT(idx - delta);
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
// Step2 Phase2 - The Down-Sweep Phase
|
||||
if (tId == 0)
|
||||
HIST_SMEM_NO_BANK_CONFLICT(255) = 0;
|
||||
|
||||
for (int delta = 128; delta >= 1; delta /= 2)
|
||||
{
|
||||
if (tId < 128 / delta)
|
||||
{
|
||||
int rootIdx = 255 - tId * delta * 2;
|
||||
int leftIdx = rootIdx - delta;
|
||||
int tmp = HIST_SMEM_NO_BANK_CONFLICT(leftIdx);
|
||||
HIST_SMEM_NO_BANK_CONFLICT(leftIdx) = HIST_SMEM_NO_BANK_CONFLICT(rootIdx);
|
||||
HIST_SMEM_NO_BANK_CONFLICT(rootIdx) += tmp;
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
// Step2 Phase3 - Convert exclusive sum to inclusive sum
|
||||
int tmp = HIST_SMEM_NO_BANK_CONFLICT(tId);
|
||||
__syncthreads();
|
||||
if (tId >= 1)
|
||||
HIST_SMEM_NO_BANK_CONFLICT(tId - 1) = tmp;
|
||||
if (tId == 255)
|
||||
HIST_SMEM_NO_BANK_CONFLICT(tId) = tmp + hist[tId];
|
||||
__syncthreads();
|
||||
|
||||
// Step3 - Scale values to build lut
|
||||
|
||||
lut[tId] = saturate_cast<unsigned char>(HIST_SMEM_NO_BANK_CONFLICT(tId) * (255.0f / (size - minNonZeroVal)));
|
||||
|
||||
#undef HIST_SMEM_NO_BANK_CONFLICT
|
||||
}
|
||||
|
||||
void buildLut(PtrStepSzi hist, PtrStepSzb lut, int size, cudaStream_t stream)
|
||||
{
|
||||
buildLutKernel<<<1, 256, 0, stream>>>(hist.data, lut.data, size);
|
||||
cudaSafeCall( cudaGetLastError() );
|
||||
|
||||
if (stream == 0)
|
||||
cudaSafeCall( cudaMemcpyToSymbol(c_lut, lut, 256 * sizeof(int), 0, cudaMemcpyDeviceToDevice) );
|
||||
else
|
||||
cudaSafeCall( cudaMemcpyToSymbolAsync(c_lut, lut, 256 * sizeof(int), 0, cudaMemcpyDeviceToDevice, stream) );
|
||||
|
||||
const float scale = 255.0f / (src.cols * src.rows);
|
||||
|
||||
device::transform(src, dst, EqualizeHist(scale), WithOutMask(), stream);
|
||||
cudaSafeCall( cudaDeviceSynchronize() );
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -102,7 +102,8 @@ void cv::cuda::calcHist(InputArray _src, InputArray _mask, OutputArray _hist, St
|
||||
|
||||
namespace hist
|
||||
{
|
||||
void equalizeHist(PtrStepSzb src, PtrStepSzb dst, const int* lut, cudaStream_t stream);
|
||||
void equalizeHist(PtrStepSzb src, PtrStepSzb dst, const uchar* lut, cudaStream_t stream);
|
||||
void buildLut(PtrStepSzi hist, PtrStepSzb lut, int size, cudaStream_t stream);
|
||||
}
|
||||
|
||||
void cv::cuda::equalizeHist(InputArray _src, OutputArray _dst, Stream& _stream)
|
||||
@@ -114,26 +115,21 @@ void cv::cuda::equalizeHist(InputArray _src, OutputArray _dst, Stream& _stream)
|
||||
_dst.create(src.size(), src.type());
|
||||
GpuMat dst = _dst.getGpuMat();
|
||||
|
||||
int intBufSize;
|
||||
nppSafeCall( nppsIntegralGetBufferSize_32s(256, &intBufSize) );
|
||||
|
||||
size_t bufSize = intBufSize + 2 * 256 * sizeof(int);
|
||||
size_t bufSize = 256 * sizeof(int) + 256 * sizeof(uchar);
|
||||
|
||||
BufferPool pool(_stream);
|
||||
GpuMat buf = pool.getBuffer(1, static_cast<int>(bufSize), CV_8UC1);
|
||||
|
||||
GpuMat hist(1, 256, CV_32SC1, buf.data);
|
||||
GpuMat lut(1, 256, CV_32SC1, buf.data + 256 * sizeof(int));
|
||||
GpuMat intBuf(1, intBufSize, CV_8UC1, buf.data + 2 * 256 * sizeof(int));
|
||||
GpuMat lut(1, 256, CV_8UC1, buf.data + 256 * sizeof(int));
|
||||
|
||||
cuda::calcHist(src, hist, _stream);
|
||||
|
||||
cudaStream_t stream = StreamAccessor::getStream(_stream);
|
||||
NppStreamHandler h(stream);
|
||||
|
||||
nppSafeCall( nppsIntegral_32s(hist.ptr<Npp32s>(), lut.ptr<Npp32s>(), 256, intBuf.ptr<Npp8u>()) );
|
||||
hist::buildLut(hist, lut, src.rows * src.cols, stream);
|
||||
|
||||
hist::equalizeHist(src, dst, lut.ptr<int>(), stream);
|
||||
hist::equalizeHist(src, dst, lut.data, stream);
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -208,7 +208,7 @@ CUDA_TEST_P(EqualizeHist, Async)
|
||||
cv::Mat dst_gold;
|
||||
cv::equalizeHist(src, dst_gold);
|
||||
|
||||
EXPECT_MAT_NEAR(dst_gold, dst, 3.0);
|
||||
EXPECT_MAT_NEAR(dst_gold, dst, 0.0);
|
||||
}
|
||||
|
||||
CUDA_TEST_P(EqualizeHist, Accuracy)
|
||||
@@ -221,13 +221,91 @@ CUDA_TEST_P(EqualizeHist, Accuracy)
|
||||
cv::Mat dst_gold;
|
||||
cv::equalizeHist(src, dst_gold);
|
||||
|
||||
EXPECT_MAT_NEAR(dst_gold, dst, 3.0);
|
||||
EXPECT_MAT_NEAR(dst_gold, dst, 0.0);
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(CUDA_ImgProc, EqualizeHist, testing::Combine(
|
||||
ALL_DEVICES,
|
||||
DIFFERENT_SIZES));
|
||||
|
||||
TEST(EqualizeHistIssue, Issue18035)
|
||||
{
|
||||
std::vector<std::string> imgPaths;
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/3MP.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/5MP.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/airplane.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/baboon.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/box.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/box_in_scene.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/fruits.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/fruits_ecc.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/graffiti.png");
|
||||
imgPaths.push_back(std::string(cvtest::TS::ptr()->get_data_path()) + "../cv/shared/lena.png");
|
||||
|
||||
for (size_t i = 0; i < imgPaths.size(); ++i)
|
||||
{
|
||||
std::string imgPath = imgPaths[i];
|
||||
cv::Mat src = cv::imread(imgPath, cv::IMREAD_GRAYSCALE);
|
||||
src = src / 30;
|
||||
|
||||
cv::cuda::GpuMat d_src, dst;
|
||||
d_src.upload(src);
|
||||
cv::cuda::equalizeHist(d_src, dst);
|
||||
|
||||
cv::Mat dst_gold;
|
||||
cv::equalizeHist(src, dst_gold);
|
||||
|
||||
EXPECT_MAT_NEAR(dst_gold, dst, 0.0);
|
||||
}
|
||||
}
|
||||
|
||||
PARAM_TEST_CASE(EqualizeHistExtreme, cv::cuda::DeviceInfo, cv::Size, int)
|
||||
{
|
||||
cv::cuda::DeviceInfo devInfo;
|
||||
cv::Size size;
|
||||
int val;
|
||||
|
||||
virtual void SetUp()
|
||||
{
|
||||
devInfo = GET_PARAM(0);
|
||||
size = GET_PARAM(1);
|
||||
val = GET_PARAM(2);
|
||||
|
||||
cv::cuda::setDevice(devInfo.deviceID());
|
||||
}
|
||||
};
|
||||
|
||||
CUDA_TEST_P(EqualizeHistExtreme, Case1)
|
||||
{
|
||||
cv::Mat src(size, CV_8UC1, val);
|
||||
|
||||
cv::cuda::GpuMat dst;
|
||||
cv::cuda::equalizeHist(loadMat(src), dst);
|
||||
|
||||
cv::Mat dst_gold;
|
||||
cv::equalizeHist(src, dst_gold);
|
||||
|
||||
EXPECT_MAT_NEAR(dst_gold, dst, 0.0);
|
||||
}
|
||||
|
||||
CUDA_TEST_P(EqualizeHistExtreme, Case2)
|
||||
{
|
||||
cv::Mat src = randomMat(size, CV_8UC1, val);
|
||||
|
||||
cv::cuda::GpuMat dst;
|
||||
cv::cuda::equalizeHist(loadMat(src), dst);
|
||||
|
||||
cv::Mat dst_gold;
|
||||
cv::equalizeHist(src, dst_gold);
|
||||
|
||||
EXPECT_MAT_NEAR(dst_gold, dst, 0.0);
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(CUDA_ImgProc, EqualizeHistExtreme, testing::Combine(
|
||||
ALL_DEVICES,
|
||||
DIFFERENT_SIZES,
|
||||
testing::Range(0, 256)));
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
// CLAHE
|
||||
|
||||
|
||||
@@ -116,15 +116,15 @@ namespace tvl1flow
|
||||
texture<float, cudaTextureType2D, cudaReadModeElementType> tex_I1y(false, cudaFilterModePoint, cudaAddressModeClamp);
|
||||
struct SrcTexRef : SrcTex
|
||||
{
|
||||
__device__ __forceinline__ float I1(float x, float y) const override
|
||||
__device__ __forceinline__ float I1(float x, float y) const CV_OVERRIDE
|
||||
{
|
||||
return tex2D(tex_I1, x, y);
|
||||
}
|
||||
__device__ __forceinline__ float I1x(float x, float y) const override
|
||||
__device__ __forceinline__ float I1x(float x, float y) const CV_OVERRIDE
|
||||
{
|
||||
return tex2D(tex_I1x, x, y);
|
||||
}
|
||||
__device__ __forceinline__ float I1y(float x, float y) const override
|
||||
__device__ __forceinline__ float I1y(float x, float y) const CV_OVERRIDE
|
||||
{
|
||||
return tex2D(tex_I1y, x, y);
|
||||
}
|
||||
@@ -135,15 +135,15 @@ namespace tvl1flow
|
||||
__host__ SrcTexObj(cudaTextureObject_t tex_obj_I1_, cudaTextureObject_t tex_obj_I1x_, cudaTextureObject_t tex_obj_I1y_)
|
||||
: tex_obj_I1(tex_obj_I1_), tex_obj_I1x(tex_obj_I1x_), tex_obj_I1y(tex_obj_I1y_) {}
|
||||
|
||||
__device__ __forceinline__ float I1(float x, float y) const override
|
||||
__device__ __forceinline__ float I1(float x, float y) const CV_OVERRIDE
|
||||
{
|
||||
return tex2D<float>(tex_obj_I1, x, y);
|
||||
}
|
||||
__device__ __forceinline__ float I1x(float x, float y) const override
|
||||
__device__ __forceinline__ float I1x(float x, float y) const CV_OVERRIDE
|
||||
{
|
||||
return tex2D<float>(tex_obj_I1x, x, y);
|
||||
}
|
||||
__device__ __forceinline__ float I1y(float x, float y) const override
|
||||
__device__ __forceinline__ float I1y(float x, float y) const CV_OVERRIDE
|
||||
{
|
||||
return tex2D<float>(tex_obj_I1y, x, y);
|
||||
}
|
||||
|
||||
@@ -598,6 +598,14 @@ CV__DNN_EXPERIMENTAL_NS_BEGIN
|
||||
static Ptr<RegionLayer> create(const LayerParams& params);
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Detection output layer.
|
||||
*
|
||||
* The layer size is: @f$ (1 \times 1 \times N \times 7) @f$
|
||||
* where N is [keep_top_k] parameter multiplied by batch size. Each row is:
|
||||
* [image_id, label, confidence, xmin, ymin, xmax, ymax]
|
||||
* where image_id is the index of image input in the batch.
|
||||
*/
|
||||
class CV_EXPORTS DetectionOutputLayer : public Layer
|
||||
{
|
||||
public:
|
||||
|
||||
@@ -47,9 +47,9 @@
|
||||
#include "opencv2/core/async.hpp"
|
||||
|
||||
#if !defined CV_DOXYGEN && !defined CV_STATIC_ANALYSIS && !defined CV_DNN_DONT_ADD_EXPERIMENTAL_NS
|
||||
#define CV__DNN_EXPERIMENTAL_NS_BEGIN namespace experimental_dnn_34_v18 {
|
||||
#define CV__DNN_EXPERIMENTAL_NS_BEGIN namespace experimental_dnn_34_v19 {
|
||||
#define CV__DNN_EXPERIMENTAL_NS_END }
|
||||
namespace cv { namespace dnn { namespace experimental_dnn_34_v18 { } using namespace experimental_dnn_34_v18; }}
|
||||
namespace cv { namespace dnn { namespace experimental_dnn_34_v19 { } using namespace experimental_dnn_34_v19; }}
|
||||
#else
|
||||
#define CV__DNN_EXPERIMENTAL_NS_BEGIN
|
||||
#define CV__DNN_EXPERIMENTAL_NS_END
|
||||
|
||||
@@ -111,6 +111,10 @@ PERF_TEST_P_(DNNTestNetwork, ENet)
|
||||
if ((backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && target != DNN_TARGET_CPU) ||
|
||||
(backend == DNN_BACKEND_OPENCV && target == DNN_TARGET_OPENCL_FP16))
|
||||
throw SkipTestException("");
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_GE(2021010000)
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
throw SkipTestException("");
|
||||
#endif
|
||||
processNet("dnn/Enet-model-best.net", "", "enet.yml",
|
||||
Mat(cv::Size(512, 256), CV_32FC3));
|
||||
}
|
||||
@@ -202,6 +206,10 @@ PERF_TEST_P_(DNNTestNetwork, YOLOv3)
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && target == DNN_TARGET_OPENCL_FP16)
|
||||
throw SkipTestException("Test is disabled in OpenVINO 2020.4");
|
||||
#endif
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2021010000) // nGraph compilation failure
|
||||
if (target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("");
|
||||
#endif
|
||||
|
||||
Mat sample = imread(findDataFile("dnn/dog416.png"));
|
||||
cvtColor(sample, sample, COLOR_BGR2RGB);
|
||||
@@ -214,7 +222,7 @@ PERF_TEST_P_(DNNTestNetwork, YOLOv4)
|
||||
{
|
||||
if (backend == DNN_BACKEND_HALIDE)
|
||||
throw SkipTestException("");
|
||||
if (target == DNN_TARGET_MYRIAD)
|
||||
if (target == DNN_TARGET_MYRIAD) // not enough resources
|
||||
throw SkipTestException("");
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2020040000) // nGraph compilation failure
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && target == DNN_TARGET_OPENCL)
|
||||
@@ -233,6 +241,10 @@ PERF_TEST_P_(DNNTestNetwork, YOLOv4_tiny)
|
||||
{
|
||||
if (backend == DNN_BACKEND_HALIDE)
|
||||
throw SkipTestException("");
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2021010000) // nGraph compilation failure
|
||||
if (target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("");
|
||||
#endif
|
||||
Mat sample = imread(findDataFile("dnn/dog416.png"));
|
||||
cvtColor(sample, sample, COLOR_BGR2RGB);
|
||||
Mat inp;
|
||||
@@ -263,6 +275,10 @@ PERF_TEST_P_(DNNTestNetwork, Inception_v2_Faster_RCNN)
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2019020000)
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
throw SkipTestException("Test is disabled in OpenVINO 2019R2");
|
||||
#endif
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2021010000)
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && target == DNN_TARGET_MYRIAD)
|
||||
throw SkipTestException("Test is disabled in OpenVINO 2021.1 / MYRIAD");
|
||||
#endif
|
||||
if (backend == DNN_BACKEND_HALIDE ||
|
||||
(backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && target != DNN_TARGET_CPU) ||
|
||||
|
||||
@@ -221,6 +221,10 @@ namespace cv {
|
||||
{
|
||||
cv::dnn::LayerParams activation_param;
|
||||
if (type == "relu")
|
||||
{
|
||||
activation_param.type = "ReLU";
|
||||
}
|
||||
else if (type == "leaky")
|
||||
{
|
||||
activation_param.set<float>("negative_slope", 0.1f);
|
||||
activation_param.type = "ReLU";
|
||||
@@ -862,24 +866,8 @@ namespace cv {
|
||||
}
|
||||
|
||||
std::string activation = getParam<std::string>(layer_params, "activation", "linear");
|
||||
if (activation == "leaky")
|
||||
{
|
||||
setParams.setActivation("relu");
|
||||
}
|
||||
else if (activation == "swish")
|
||||
{
|
||||
setParams.setActivation("swish");
|
||||
}
|
||||
else if (activation == "mish")
|
||||
{
|
||||
setParams.setActivation("mish");
|
||||
}
|
||||
else if (activation == "logistic")
|
||||
{
|
||||
setParams.setActivation("logistic");
|
||||
}
|
||||
else if (activation != "linear")
|
||||
CV_Error(cv::Error::StsParseError, "Unsupported activation: " + activation);
|
||||
if (activation != "linear")
|
||||
setParams.setActivation(activation);
|
||||
|
||||
net->out_channels_vec[layers_counter] = tensor_shape[0];
|
||||
}
|
||||
@@ -996,8 +984,8 @@ namespace cv {
|
||||
}
|
||||
|
||||
std::string activation = getParam<std::string>(layer_params, "activation", "linear");
|
||||
if(activation == "leaky" || activation == "swish" || activation == "mish" || activation == "logistic")
|
||||
++cv_layers_counter; // For ReLU, Swish, Mish, Sigmoid
|
||||
if (activation != "linear")
|
||||
++cv_layers_counter; // For ReLU, Swish, Mish, Sigmoid, etc
|
||||
|
||||
if(!darknet_layers_counter)
|
||||
tensor_shape.resize(1);
|
||||
|
||||
@@ -2113,7 +2113,9 @@ struct Net::Impl : public detail::NetImplBase
|
||||
|
||||
auto ieInpNode = inputNodes[i].dynamicCast<InfEngineNgraphNode>();
|
||||
CV_Assert(oid < ieInpNode->node->get_output_size());
|
||||
#if INF_ENGINE_VER_MAJOR_GT(INF_ENGINE_RELEASE_2020_3)
|
||||
#if INF_ENGINE_VER_MAJOR_GT(INF_ENGINE_RELEASE_2020_4)
|
||||
inputNodes[i] = Ptr<BackendNode>(new InfEngineNgraphNode(ieInpNode->node));
|
||||
#elif INF_ENGINE_VER_MAJOR_GT(INF_ENGINE_RELEASE_2020_3)
|
||||
inputNodes[i] = Ptr<BackendNode>(new InfEngineNgraphNode(ieInpNode->node->get_output_as_single_output_node(oid)));
|
||||
#else
|
||||
inputNodes[i] = Ptr<BackendNode>(new InfEngineNgraphNode(ieInpNode->node->get_output_as_single_output_node(oid, false)));
|
||||
@@ -2411,14 +2413,42 @@ struct Net::Impl : public detail::NetImplBase
|
||||
}
|
||||
|
||||
// fuse convolution layer followed by eltwise + relu
|
||||
if ( IS_DNN_OPENCL_TARGET(preferableTarget) && ld.layerInstance->type == "Convolution" )
|
||||
while (nextData && IS_DNN_OPENCL_TARGET(preferableTarget) && ld.layerInstance->type == "Convolution") // semantic of 'if'
|
||||
{
|
||||
Ptr<EltwiseLayer> nextEltwiseLayer;
|
||||
if( nextData )
|
||||
nextEltwiseLayer = nextData->layerInstance.dynamicCast<EltwiseLayer>();
|
||||
Ptr<EltwiseLayer> nextEltwiseLayer = nextData->layerInstance.dynamicCast<EltwiseLayer>();
|
||||
if (nextEltwiseLayer.empty())
|
||||
break;
|
||||
|
||||
if (pinsToKeep.count(lpNext) != 0)
|
||||
break;
|
||||
if (nextData->inputBlobsId.size() != 2)
|
||||
break;
|
||||
|
||||
if (!nextData->params.has("operation") || nextData->params.get<String>("operation").toLowerCase() == "sum")
|
||||
{
|
||||
if (nextData->params.has("coeff"))
|
||||
{
|
||||
DictValue paramCoeff = nextData->params.get("coeff");
|
||||
int n = paramCoeff.size();
|
||||
bool isCoeffOneOne = (n == 2);
|
||||
for (int i = 0; isCoeffOneOne && i < n; i++)
|
||||
{
|
||||
float c = paramCoeff.get<float>(i);
|
||||
isCoeffOneOne &= (c == 1.0f);
|
||||
}
|
||||
if (!isCoeffOneOne)
|
||||
{
|
||||
CV_LOG_DEBUG(NULL, "DNN/OpenCL: fusion of 'Sum' without coeffs (or {1.0, 1.0}) is supported only");
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
CV_LOG_DEBUG(NULL, "DNN/OpenCL: fusion with eltwise operation is not supported: " << nextData->params.get<String>("operation"));
|
||||
break;
|
||||
}
|
||||
|
||||
if( !nextEltwiseLayer.empty() && pinsToKeep.count(lpNext) == 0 &&
|
||||
nextData && nextData->inputBlobsId.size() == 2 )
|
||||
{
|
||||
LayerData *eltwiseData = nextData;
|
||||
|
||||
@@ -2458,10 +2488,12 @@ struct Net::Impl : public detail::NetImplBase
|
||||
if( nextData )
|
||||
nextActivLayer = nextData->layerInstance.dynamicCast<ActivationLayer>();
|
||||
|
||||
if( !nextActivLayer.empty() && pinsToKeep.count(lpNext) == 0 &&
|
||||
Ptr<PowerLayer> activ_power;
|
||||
if( !nextActivLayer.empty() &&
|
||||
(!nextData->type.compare("ReLU") ||
|
||||
!nextData->type.compare("ChannelsPReLU") ||
|
||||
!nextData->type.compare("Power")) &&
|
||||
(!nextData->type.compare("Power") && (activ_power = nextActivLayer.dynamicCast<PowerLayer>()) && activ_power->scale == 1.0f)
|
||||
) &&
|
||||
currLayer->setActivation(nextActivLayer) )
|
||||
{
|
||||
CV_Assert_N(biasLayerData->outputBlobsWrappers.size() == 1, ld.inputBlobsWrappers.size() == 1);
|
||||
@@ -2513,6 +2545,8 @@ struct Net::Impl : public detail::NetImplBase
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2694,11 +2728,11 @@ struct Net::Impl : public detail::NetImplBase
|
||||
|
||||
Ptr<Layer> layer = ld.layerInstance;
|
||||
|
||||
TickMeter tm;
|
||||
tm.start();
|
||||
|
||||
if( !ld.skip )
|
||||
{
|
||||
TickMeter tm;
|
||||
tm.start();
|
||||
|
||||
std::map<int, Ptr<BackendNode> >::iterator it = ld.backendNodes.find(preferableBackend);
|
||||
if (preferableBackend == DNN_BACKEND_OPENCV || it == ld.backendNodes.end() || it->second.empty())
|
||||
{
|
||||
@@ -2877,12 +2911,15 @@ struct Net::Impl : public detail::NetImplBase
|
||||
CV_Error(Error::StsNotImplemented, "Unknown backend identifier");
|
||||
}
|
||||
}
|
||||
|
||||
tm.stop();
|
||||
int64 t = tm.getTimeTicks();
|
||||
layersTimings[ld.id] = (t > 0) ? t : t + 1; // zero for skipped layers only
|
||||
}
|
||||
else
|
||||
tm.reset();
|
||||
|
||||
tm.stop();
|
||||
layersTimings[ld.id] = tm.getTimeTicks();
|
||||
{
|
||||
layersTimings[ld.id] = 0;
|
||||
}
|
||||
|
||||
ld.flag = 1;
|
||||
}
|
||||
@@ -3486,11 +3523,16 @@ void Net::connect(String _outPin, String _inPin)
|
||||
Mat Net::forward(const String& outputName)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
CV_Assert(!empty());
|
||||
|
||||
String layerName = outputName;
|
||||
|
||||
if (layerName.empty())
|
||||
layerName = getLayerNames().back();
|
||||
{
|
||||
std::vector<String> layerNames = getLayerNames();
|
||||
CV_Assert(!layerNames.empty());
|
||||
layerName = layerNames.back();
|
||||
}
|
||||
|
||||
std::vector<LayerPin> pins(1, impl->getPinByAlias(layerName));
|
||||
impl->setUpNet(pins);
|
||||
@@ -3502,11 +3544,17 @@ Mat Net::forward(const String& outputName)
|
||||
AsyncArray Net::forwardAsync(const String& outputName)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
CV_Assert(!empty());
|
||||
|
||||
#ifdef CV_CXX11
|
||||
String layerName = outputName;
|
||||
|
||||
if (layerName.empty())
|
||||
layerName = getLayerNames().back();
|
||||
{
|
||||
std::vector<String> layerNames = getLayerNames();
|
||||
CV_Assert(!layerNames.empty());
|
||||
layerName = layerNames.back();
|
||||
}
|
||||
|
||||
std::vector<LayerPin> pins(1, impl->getPinByAlias(layerName));
|
||||
impl->setUpNet(pins);
|
||||
@@ -3527,11 +3575,16 @@ AsyncArray Net::forwardAsync(const String& outputName)
|
||||
void Net::forward(OutputArrayOfArrays outputBlobs, const String& outputName)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
CV_Assert(!empty());
|
||||
|
||||
String layerName = outputName;
|
||||
|
||||
if (layerName.empty())
|
||||
layerName = getLayerNames().back();
|
||||
{
|
||||
std::vector<String> layerNames = getLayerNames();
|
||||
CV_Assert(!layerNames.empty());
|
||||
layerName = layerNames.back();
|
||||
}
|
||||
|
||||
std::vector<LayerPin> pins(1, impl->getPinByAlias(layerName));
|
||||
impl->setUpNet(pins);
|
||||
@@ -4118,6 +4171,8 @@ std::vector<Ptr<Layer> > Net::getLayerInputs(LayerId layerId)
|
||||
|
||||
std::vector<String> Net::getLayerNames() const
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
|
||||
std::vector<String> res;
|
||||
res.reserve(impl->layers.size());
|
||||
|
||||
|
||||
@@ -109,6 +109,12 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
#if INF_ENGINE_VER_MAJOR_GE(INF_ENGINE_RELEASE_2020_4)
|
||||
std::shared_ptr<ngraph::Node> clone_with_new_inputs(const ngraph::OutputVector& new_args) const override
|
||||
{
|
||||
return std::make_shared<NgraphCustomOp>(new_args, params);
|
||||
}
|
||||
#else
|
||||
std::shared_ptr<ngraph::Node> copy_with_new_args(const ngraph::NodeVector& new_args) const override
|
||||
{
|
||||
#if INF_ENGINE_VER_MAJOR_GE(INF_ENGINE_RELEASE_2020_3)
|
||||
@@ -117,6 +123,7 @@ public:
|
||||
return std::make_shared<NgraphCustomOp>(new_args, params);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
|
||||
bool visit_attributes(ngraph::AttributeVisitor& visitor) override
|
||||
{
|
||||
@@ -380,7 +387,11 @@ void InfEngineNgraphNet::setNodePtr(std::shared_ptr<ngraph::Node>* ptr) {
|
||||
|
||||
void InfEngineNgraphNet::release() {
|
||||
for (auto& node : components.back()) {
|
||||
#if INF_ENGINE_VER_MAJOR_GT(INF_ENGINE_RELEASE_2020_4)
|
||||
if (!(ngraph::op::is_parameter(node) || ngraph::op::is_output(node) || ngraph::op::is_constant(node)) ) {
|
||||
#else
|
||||
if (!(node->is_parameter() || node->is_output() || node->is_constant()) ) {
|
||||
#endif
|
||||
auto it = all_nodes.find(node->get_friendly_name());
|
||||
if (it != all_nodes.end()) {
|
||||
unconnectedNodes.erase(*(it->second));
|
||||
@@ -447,11 +458,19 @@ void InfEngineNgraphNet::createNet(Target targetId) {
|
||||
ngraph::ResultVector outputs;
|
||||
ngraph::ParameterVector inps;
|
||||
for (auto& node : components.back()) {
|
||||
#if INF_ENGINE_VER_MAJOR_GT(INF_ENGINE_RELEASE_2020_4)
|
||||
if (ngraph::op::is_parameter(node)) {
|
||||
#else
|
||||
if (node->is_parameter()) {
|
||||
#endif
|
||||
auto parameter = std::dynamic_pointer_cast<ngraph::op::Parameter>(node);
|
||||
inps.push_back(parameter);
|
||||
}
|
||||
#if INF_ENGINE_VER_MAJOR_GT(INF_ENGINE_RELEASE_2020_4)
|
||||
else if (ngraph::op::is_output(node)) {
|
||||
#else
|
||||
else if (node->is_output()) {
|
||||
#endif
|
||||
auto result = std::dynamic_pointer_cast<ngraph::op::Result>(node);
|
||||
outputs.push_back(result);
|
||||
}
|
||||
@@ -665,7 +684,11 @@ void InfEngineNgraphNet::initPlugin(InferenceEngine::CNNNetwork& net)
|
||||
}
|
||||
std::map<std::string, std::string> config;
|
||||
if (device_name == "MYRIAD") {
|
||||
#if INF_ENGINE_VER_MAJOR_GT(INF_ENGINE_RELEASE_2020_4)
|
||||
config.emplace("MYRIAD_DETECT_NETWORK_BATCH", CONFIG_VALUE(NO));
|
||||
#else
|
||||
config.emplace("VPU_DETECT_NETWORK_BATCH", CONFIG_VALUE(NO));
|
||||
#endif
|
||||
}
|
||||
|
||||
bool isHetero = device_name == "FPGA";
|
||||
|
||||
@@ -46,6 +46,8 @@
|
||||
#include "../op_inf_engine.hpp"
|
||||
#include "../ie_ngraph.hpp"
|
||||
|
||||
#include <opencv2/core/utils/logger.hpp>
|
||||
|
||||
#include "opencv2/core/hal/hal.hpp"
|
||||
#include "opencv2/core/hal/intrin.hpp"
|
||||
#include <iostream>
|
||||
@@ -106,18 +108,19 @@ public:
|
||||
inputs_arr.getMatVector(inputs);
|
||||
outputs_arr.getMatVector(outputs);
|
||||
|
||||
CV_Assert(inputs.size() > 0);
|
||||
CV_Assert((inputs.size() > outputs.size() && blobs.empty()) ||
|
||||
(!inputs.empty() && (blobs.size() == 1 || blobs.size() == 2)));
|
||||
MatSize weightShape = blobs.empty() ? inputs[1].size : blobs[0].size;
|
||||
|
||||
CV_Assert(blobs.size() == 1 || blobs.size() == 2);
|
||||
CV_Assert(inputs[0].dims == outputs[0].dims);
|
||||
CV_Assert(blobs[0].dims == kernel_size.size() + 2);
|
||||
CV_Assert(weightShape.dims() == kernel_size.size() + 2);
|
||||
for (int i = 0; i < kernel_size.size(); i++) {
|
||||
CV_Assert(blobs[0].size[i + 2] == kernel_size[i]);
|
||||
CV_Assert(weightShape[i + 2] == kernel_size[i]);
|
||||
}
|
||||
|
||||
const Mat &input = inputs[0];
|
||||
CV_Assert((input.dims == 4 || input.dims == 5) && (input.type() == CV_32F || input.type() == CV_16S));
|
||||
for (size_t i = 0; i < inputs.size(); i++)
|
||||
for (size_t i = 0; i < outputs.size(); i++)
|
||||
{
|
||||
CV_Assert(inputs[i].type() == input.type());
|
||||
CV_Assert((inputs[i].dims == 4 || inputs[i].dims == 5) && inputs[i].size[1] == input.size[1]);
|
||||
@@ -245,6 +248,7 @@ public:
|
||||
|
||||
MatShape computeColRowShape(const MatShape &inpShape, const MatShape &outShape) const CV_OVERRIDE
|
||||
{
|
||||
CV_Assert(!blobs.empty());
|
||||
int dims = inpShape.size();
|
||||
int inpD = dims == 5 ? inpShape[2] : 1;
|
||||
int inpH = inpShape[dims - 2];
|
||||
@@ -262,12 +266,14 @@ public:
|
||||
{
|
||||
if (kernel_size.size() == 3)
|
||||
return preferableTarget == DNN_TARGET_CPU;
|
||||
if ((backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 || preferableTarget != DNN_TARGET_MYRIAD) && blobs.empty())
|
||||
return false;
|
||||
return (preferableTarget != DNN_TARGET_MYRIAD || dilation.width == dilation.height);
|
||||
}
|
||||
else
|
||||
#endif
|
||||
return (kernel_size.size() == 3 && preferableTarget == DNN_TARGET_CPU && backendId == DNN_BACKEND_OPENCV) ||
|
||||
(kernel_size.size() == 2 && (backendId == DNN_BACKEND_OPENCV || backendId == DNN_BACKEND_HALIDE));
|
||||
(kernel_size.size() == 2 && (backendId == DNN_BACKEND_OPENCV || (backendId == DNN_BACKEND_HALIDE && !blobs.empty())));
|
||||
}
|
||||
|
||||
bool getMemoryShapes(const std::vector<MatShape> &inputs,
|
||||
@@ -275,16 +281,16 @@ public:
|
||||
std::vector<MatShape> &outputs,
|
||||
std::vector<MatShape> &internals) const CV_OVERRIDE
|
||||
{
|
||||
CV_Assert(blobs.size() != 0);
|
||||
CV_Assert(!hasBias() || blobs[1].total() == (size_t)blobs[0].size[0]);
|
||||
CV_Assert(inputs.size() == (size_t)1);
|
||||
CV_Assert(!blobs.empty() || inputs.size() > 1);
|
||||
const int* weightShape = blobs.empty() ? &inputs[1][0] : blobs[0].size.p;
|
||||
CV_Assert(!hasBias() || blobs[1].total() == (size_t)weightShape[0]);
|
||||
|
||||
internals.clear();
|
||||
|
||||
CV_Assert(inputs.size() != 0);
|
||||
std::vector<int> inpShape(inputs[0].begin() + 2, inputs[0].end());
|
||||
|
||||
int outCn = blobs[0].size[0];
|
||||
int outCn = weightShape[0];
|
||||
std::vector<int> outShape;
|
||||
outShape.push_back(inputs[0][0]);
|
||||
outShape.push_back(outCn);
|
||||
@@ -300,10 +306,10 @@ public:
|
||||
getConvPoolOutParams(inpShape, kernel_size, strides, padMode, dilations, outShape);
|
||||
}
|
||||
|
||||
int ngroups = inpCn / blobs[0].size[1];
|
||||
if (ngroups == 0 || ngroups * blobs[0].size[1] != inpCn)
|
||||
int ngroups = inpCn / weightShape[1];
|
||||
if (ngroups == 0 || ngroups * weightShape[1] != inpCn)
|
||||
CV_Error(Error::StsError, format("Number of input channels should "
|
||||
"be multiple of %d but got %d", blobs[0].size[1], inpCn));
|
||||
"be multiple of %d but got %d", weightShape[1], inpCn));
|
||||
CV_Assert(ngroups > 0 && inpCn % ngroups == 0 && outCn % ngroups == 0);
|
||||
|
||||
outputs.resize(1, outShape);
|
||||
@@ -315,15 +321,15 @@ public:
|
||||
{
|
||||
BaseConvolutionLayerImpl::finalize(inputs_arr, outputs_arr);
|
||||
|
||||
CV_Assert(!blobs.empty());
|
||||
const int outCn = blobs[0].size[0];
|
||||
std::vector<Mat> inputs;
|
||||
inputs_arr.getMatVector(inputs);
|
||||
// prepare weightsMat where each row is aligned and has enough zero padding on the right to
|
||||
// use vectorized (i.e. with intrinsics) loops without tail processing
|
||||
Mat wm = blobs[0].reshape(1, outCn);
|
||||
Mat wm = blobs.empty() ? inputs[1].reshape(1, numOutput) : blobs[0].reshape(1, numOutput);
|
||||
if( wm.step1() % VEC_ALIGN != 0 )
|
||||
{
|
||||
int newcols = (int)alignSize(wm.step1(), VEC_ALIGN);
|
||||
Mat wm_buffer = Mat(outCn, newcols, wm.type());
|
||||
Mat wm_buffer = Mat(numOutput, newcols, wm.type());
|
||||
Mat wm_padding = wm_buffer.colRange(wm.cols, newcols);
|
||||
wm_padding.setTo(Scalar::all(0.));
|
||||
Mat wm_aligned = wm_buffer.colRange(0, wm.cols);
|
||||
@@ -331,18 +337,18 @@ public:
|
||||
wm = wm_aligned;
|
||||
}
|
||||
weightsMat = wm;
|
||||
weightsMultipliers.assign(outCn, 1.0);
|
||||
weightsMultipliers.assign(numOutput, 1.0);
|
||||
|
||||
Mat biasMat = hasBias() ? blobs[1].reshape(1, outCn) : Mat();
|
||||
biasvec.resize(outCn+2);
|
||||
Mat biasMat = hasBias() ? blobs[1].reshape(1, numOutput) : Mat();
|
||||
biasvec.resize(numOutput+2);
|
||||
if( biasMat.empty() )
|
||||
{
|
||||
for(int i = 0; i < outCn; i++ )
|
||||
for(int i = 0; i < numOutput; i++ )
|
||||
biasvec[i] = 0.f;
|
||||
}
|
||||
else
|
||||
{
|
||||
for(int i = 0; i < outCn; i++ )
|
||||
for(int i = 0; i < numOutput; i++ )
|
||||
biasvec[i] = biasMat.at<float>(i);
|
||||
}
|
||||
#ifdef HAVE_OPENCL
|
||||
@@ -352,7 +358,7 @@ public:
|
||||
|
||||
bool setActivation(const Ptr<ActivationLayer>& layer) CV_OVERRIDE
|
||||
{
|
||||
if (!activ.empty() && !layer.empty())
|
||||
if ((!activ.empty() && !layer.empty()) || blobs.empty())
|
||||
return false;
|
||||
|
||||
activ = layer;
|
||||
@@ -367,6 +373,14 @@ public:
|
||||
Ptr<PowerLayer> activ_power = activ.dynamicCast<PowerLayer>();
|
||||
if (!activ_power.empty())
|
||||
{
|
||||
if (activ_power->scale != 1.0f) // not supported well by implementation, #17964
|
||||
{
|
||||
// FIXIT no way to check number of blobs (like, eltwise input)
|
||||
CV_LOG_DEBUG(NULL, "DNN/OpenCL: can't configure Power activation (scale != 1.0f)");
|
||||
activ.release();
|
||||
newActiv = false;
|
||||
return false;
|
||||
}
|
||||
if (activ_power->scale != 1.f || activ_power->shift != 0.f)
|
||||
{
|
||||
const int outCh = blobs[0].size[0];
|
||||
@@ -537,37 +551,50 @@ public:
|
||||
virtual Ptr<BackendNode> initNgraph(const std::vector<Ptr<BackendWrapper> > &inputs,
|
||||
const std::vector<Ptr<BackendNode> >& nodes) CV_OVERRIDE
|
||||
{
|
||||
CV_Assert_N(inputs.size() == 1, nodes.size() == 1);
|
||||
CV_Assert_N(inputs.size() >= 1, nodes.size() >= 1);
|
||||
auto& ieInpNode = nodes[0].dynamicCast<InfEngineNgraphNode>()->node;
|
||||
std::vector<size_t> dims = ieInpNode->get_shape();
|
||||
CV_Assert(dims.size() == 4 || dims.size() == 5);
|
||||
std::shared_ptr<ngraph::Node> ieWeights = nodes.size() > 1 ? nodes[1].dynamicCast<InfEngineNgraphNode>()->node : nullptr;
|
||||
if (nodes.size() > 1)
|
||||
CV_Assert(ieWeights); // dynamic_cast should not fail
|
||||
const int inpCn = dims[1];
|
||||
const int outCn = blobs[0].size[0];
|
||||
const int inpGroupCn = blobs[0].size[1];
|
||||
const int inpGroupCn = nodes.size() > 1 ? ieWeights->get_shape()[1] : blobs[0].size[1];
|
||||
const int group = inpCn / inpGroupCn;
|
||||
|
||||
std::vector<size_t> kernel_shape = getShape<size_t>(blobs[0]);
|
||||
std::vector<size_t> kernel_shape;
|
||||
if (group != 1)
|
||||
{
|
||||
kernel_shape[0] /= group;
|
||||
kernel_shape.insert(kernel_shape.begin(), group);
|
||||
kernel_shape.push_back(group);
|
||||
}
|
||||
kernel_shape.push_back(numOutput / group);
|
||||
kernel_shape.push_back(inpCn / group);
|
||||
std::copy(kernel_size.begin(), kernel_size.end(), back_inserter(kernel_shape));
|
||||
|
||||
auto ieWeights = std::make_shared<ngraph::op::Constant>(ngraph::element::f32, kernel_shape, blobs[0].data);
|
||||
if (fusedWeights)
|
||||
if (nodes.size() == 1)
|
||||
{
|
||||
if (weightsMat.isContinuous())
|
||||
ieWeights = std::make_shared<ngraph::op::Constant>(ngraph::element::f32, kernel_shape, blobs[0].data);
|
||||
if (fusedWeights)
|
||||
{
|
||||
ieWeights = std::make_shared<ngraph::op::Constant>(ngraph::element::f32, kernel_shape, weightsMat.data);
|
||||
}
|
||||
else
|
||||
{
|
||||
Mat newWeights;
|
||||
Mat cvWeights = weightsMat.colRange(0, blobs[0].total() / outCn);
|
||||
cvWeights.copyTo(newWeights);
|
||||
ieWeights = std::make_shared<ngraph::op::Constant>(ngraph::element::f32, kernel_shape, newWeights.data);
|
||||
if (weightsMat.isContinuous())
|
||||
{
|
||||
ieWeights = std::make_shared<ngraph::op::Constant>(ngraph::element::f32, kernel_shape, weightsMat.data);
|
||||
}
|
||||
else
|
||||
{
|
||||
Mat newWeights;
|
||||
Mat cvWeights = weightsMat.colRange(0, blobs[0].total() / numOutput);
|
||||
cvWeights.copyTo(newWeights);
|
||||
ieWeights = std::make_shared<ngraph::op::Constant>(ngraph::element::f32, kernel_shape, newWeights.data);
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
auto shape = std::make_shared<ngraph::op::Constant>(ngraph::element::i64,
|
||||
ngraph::Shape{kernel_shape.size()}, kernel_shape.data());
|
||||
ieWeights = std::make_shared<ngraph::op::v1::Reshape>(ieWeights, shape, true);
|
||||
}
|
||||
|
||||
ngraph::op::PadType pad_type = ngraph::op::PadType::EXPLICIT;
|
||||
if (!padMode.empty())
|
||||
@@ -592,11 +619,21 @@ public:
|
||||
pad_type);
|
||||
}
|
||||
|
||||
if (hasBias() || fusedBias)
|
||||
if (hasBias() || fusedBias || nodes.size() == 3)
|
||||
{
|
||||
std::vector<size_t> shape(conv_node->get_shape().size(), 1);
|
||||
shape[1] = outCn;
|
||||
auto bias = std::make_shared<ngraph::op::Constant>(ngraph::element::f32, ngraph::Shape(shape), biasvec.data());
|
||||
shape[1] = conv_node->get_shape()[1];
|
||||
std::shared_ptr<ngraph::Node> bias;
|
||||
if (nodes.size() == 3)
|
||||
{
|
||||
auto bias_shape = std::make_shared<ngraph::op::Constant>(ngraph::element::i64,
|
||||
ngraph::Shape{shape.size()}, shape.data());
|
||||
bias = std::make_shared<ngraph::op::v1::Reshape>(nodes[2].dynamicCast<InfEngineNgraphNode>()->node, bias_shape, true);
|
||||
}
|
||||
else
|
||||
{
|
||||
bias = std::make_shared<ngraph::op::Constant>(ngraph::element::f32, ngraph::Shape(shape), biasvec.data());
|
||||
}
|
||||
auto conv_bias = std::make_shared<ngraph::op::v1::Add>(conv_node, bias, ngraph::op::AutoBroadcastType::NUMPY);
|
||||
return Ptr<BackendNode>(new InfEngineNgraphNode(conv_bias));
|
||||
}
|
||||
@@ -623,10 +660,12 @@ public:
|
||||
bool useAVX;
|
||||
bool useAVX2;
|
||||
bool useAVX512;
|
||||
int blk_size_cn;
|
||||
|
||||
ParallelConv()
|
||||
: input_(0), weights_(0), output_(0), ngroups_(0), nstripes_(0),
|
||||
biasvec_(0), reluslope_(0), activ_(0), is1x1_(false), useAVX(false), useAVX2(false), useAVX512(false)
|
||||
, blk_size_cn(0)
|
||||
{}
|
||||
|
||||
static void run( const Mat& input, Mat& output, const Mat& weights,
|
||||
@@ -679,12 +718,17 @@ public:
|
||||
p.useAVX2 = checkHardwareSupport(CPU_AVX2) && isConv2D;
|
||||
p.useAVX512 = CV_CPU_HAS_SUPPORT_AVX512_SKX && isConv2D;
|
||||
|
||||
int ncn = std::min(inpCn, (int)BLK_SIZE_CN);
|
||||
|
||||
int kernel_d = !isConv2D? kernel_size[0] : 1;
|
||||
int kernel_h = kernel_size[kernel_size.size() - 2];
|
||||
int kernel_w = kernel_size.back();
|
||||
|
||||
int blk_size_cn0 = cvCeil(800./(kernel_w*kernel_h));
|
||||
int ncn = 16;
|
||||
while (ncn*2 < blk_size_cn0 && ncn < inpCn)
|
||||
ncn *= 2;
|
||||
ncn = std::min(ncn, inpCn);
|
||||
p.blk_size_cn = ncn;
|
||||
|
||||
int dil_d = !isConv2D? dilations[0] : 1;
|
||||
int dil_h = dilations[dilations.size() - 2];
|
||||
int dil_w = dilations.back();
|
||||
@@ -752,18 +796,26 @@ public:
|
||||
int dilation_w = dilations.back();
|
||||
|
||||
int i, j, k, d;
|
||||
size_t inpPlaneSize = input_->total(2);
|
||||
size_t outPlaneSize = output_->total(2);
|
||||
int inpPlaneSize = (int)input_->total(2);
|
||||
int outPlaneSize = (int)output_->total(2);
|
||||
bool is1x1 = is1x1_;
|
||||
|
||||
int stripesPerSample;
|
||||
size_t stripeSize;
|
||||
int stripeSize;
|
||||
Range r = r0;
|
||||
bool depthWiseConvolution = !is1x1 && isConv2D && ngroups > 1 && inpCn == 1 &&
|
||||
outCn == 1 && kernel_d == 1 && dilation_d == 1 && stride_d == 0 && pad_d == 0 &&
|
||||
width >= 16 + dilation_w*(kernel_w - 1);
|
||||
// for now only 3x3 depth-wise convolutions are supported
|
||||
depthWiseConvolution = depthWiseConvolution && kernel_w == 3 && kernel_h == 3 &&
|
||||
// computing at most 1 pixel from each side can involve padding
|
||||
max(stride_w, dilation_w) >= pad_l && max(stride_h, dilation_h) >= pad_t &&
|
||||
pad_l <= 1 && pad_t <= 1;
|
||||
|
||||
if( nstripes >= batchSize*2 )
|
||||
if( !depthWiseConvolution && nstripes >= batchSize*2 )
|
||||
{
|
||||
stripesPerSample = nstripes/batchSize;
|
||||
stripeSize = alignSize((outPlaneSize + stripesPerSample - 1)/stripesPerSample, valign);
|
||||
stripeSize = (int)alignSize((outPlaneSize + stripesPerSample - 1)/stripesPerSample, valign);
|
||||
stripeSize = std::min(stripeSize, outPlaneSize);
|
||||
}
|
||||
else
|
||||
@@ -782,20 +834,29 @@ public:
|
||||
const float* biasptr_ = &biasvec_->at(0);
|
||||
const float* reluptr_ = reluslope_->empty() ? 0 : &reluslope_->at(0);
|
||||
float* data_out0_ = output_->ptr<float>();
|
||||
size_t rowbufsz = (size_t)karea*BLK_SIZE_CN*BLK_SIZE;
|
||||
AutoBuffer<float> rowbuf0_(rowbufsz + valign);
|
||||
float* rowbuf0 = alignPtr(rowbuf0_.data(), (int)(valign*sizeof(float)));
|
||||
AutoBuffer<float> rowbuf0_;
|
||||
float* rowbuf0 = 0;
|
||||
bool use_rowbuf = !depthWiseConvolution;
|
||||
int blk_size = depthWiseConvolution ? outPlaneSize : min((int)BLK_SIZE, stripeSize);
|
||||
|
||||
// we clear the buffer once; ultimately, it lets us to avoid
|
||||
// tail processing after running the unrolled/vectorized loop.
|
||||
// the main idea is to make sure that the tail (a.k.a. padding) of each row
|
||||
// (i.e. the elements with indices between vsz=karea*ncn and vsz_a)
|
||||
// does not contain NaNs or Infs. Because the padding in the weights
|
||||
// matrix is explicitly initialized with 0's, we handle all other
|
||||
// cases nicely, i.e. we can skip expliciting re-initialization
|
||||
// of the padding - we just retain elements from the previous iteration
|
||||
// of the loop over channels (cn0).
|
||||
memset(rowbuf0, 0, rowbufsz*sizeof(rowbuf0[0]) );
|
||||
// im2row buffer is not used for depth-wise convolution
|
||||
if(use_rowbuf)
|
||||
{
|
||||
size_t rowbufsz = alignSize(karea*blk_size_cn, valign)*min((int)BLK_SIZE, blk_size);
|
||||
//printf("karea=%d, blk_size_cn=%d, rowbufsz=%d, stripeSize=%d\n", karea, blk_size_cn, (int)rowbufsz, stripeSize);
|
||||
rowbuf0_.allocate(rowbufsz + valign);
|
||||
rowbuf0 = alignPtr(rowbuf0_.data(), (int)(valign*sizeof(float)));
|
||||
// we clear the buffer once; ultimately, it lets us to avoid
|
||||
// tail processing after running the unrolled/vectorized loop.
|
||||
// the main idea is to make sure that the tail (a.k.a. padding) of each row
|
||||
// (i.e. the elements with indices between vsz=karea*ncn and vsz_a)
|
||||
// does not contain NaNs or Infs. Because the padding in the weights
|
||||
// matrix is explicitly initialized with 0's, we handle all other
|
||||
// cases nicely, i.e. we can skip expliciting re-initialization
|
||||
// of the padding - we just retain elements from the previous iteration
|
||||
// of the loop over channels (cn0).
|
||||
memset(rowbuf0, 0, rowbufsz*sizeof(rowbuf0[0]) );
|
||||
}
|
||||
|
||||
for( int stripe = r.start; stripe < r.end; stripe++ )
|
||||
{
|
||||
@@ -810,28 +871,213 @@ public:
|
||||
const float* wptr_orig = wptr_orig_ + wstep*startOutCn;
|
||||
const float* biasptr = biasptr_ + startOutCn;
|
||||
|
||||
for( int cn0 = 0; cn0 < inpCn; cn0 += BLK_SIZE_CN )
|
||||
for( int cn0 = 0; cn0 < inpCn; cn0 += blk_size_cn )
|
||||
{
|
||||
int cn1 = std::min(cn0 + BLK_SIZE_CN, inpCn);
|
||||
int cn1 = std::min(cn0 + blk_size_cn, inpCn);
|
||||
int ncn = cn1 - cn0, vsz = karea*ncn;
|
||||
int vsz_a = (int)alignSize(vsz, valign);
|
||||
const float* wptr = wptr_orig + cn0*karea;
|
||||
// we apply [Channels][P]ReLU (if any) during the final pass only.
|
||||
const float* relu = cn1 == inpCn && reluptr_ ? reluptr_ + startOutCn : 0;
|
||||
|
||||
for( int ofs0 = stripeStart; ofs0 < stripeEnd; ofs0 += BLK_SIZE )
|
||||
for( int ofs0 = stripeStart; ofs0 < stripeEnd; ofs0 += blk_size )
|
||||
{
|
||||
int ofs, ofs1 = std::min(ofs0 + BLK_SIZE, stripeEnd);
|
||||
int ofs, ofs1 = std::min(ofs0 + blk_size, stripeEnd);
|
||||
int bsz = ofs1 - ofs0;
|
||||
|
||||
int out_d = ofs0 / (outH * outW);
|
||||
int out_i = (ofs0 - out_d * outH * outW) / outW;
|
||||
int out_j = ofs0 % outW;
|
||||
|
||||
if (depthWiseConvolution)
|
||||
{
|
||||
CV_Assert(out_i == 0 && out_j == 0);
|
||||
int in_d = out_d * stride_d - pad_d;
|
||||
const float* inptr_ = data_inp0 + (cn0*depth*height + in_d*height)*width;
|
||||
float* outptr_ = data_out0 + ofs0;
|
||||
|
||||
#if CV_TRY_AVX2
|
||||
if(useAVX2)
|
||||
opt_AVX2::fastDepthwiseConv(wptr, kernel_h, kernel_w,
|
||||
stride_h, stride_w, dilation_h, dilation_w, pad_t, pad_l,
|
||||
biasptr, relu, inptr_, height, width, outptr_, out_d, outH, outW);
|
||||
else
|
||||
#endif
|
||||
#if CV_TRY_AVX
|
||||
if(useAVX)
|
||||
opt_AVX::fastDepthwiseConv(wptr, kernel_h, kernel_w,
|
||||
stride_h, stride_w, dilation_h, dilation_w, pad_t, pad_l,
|
||||
biasptr, relu, inptr_, height, width, outptr_, out_d, outH, outW);
|
||||
else
|
||||
#endif
|
||||
{
|
||||
const float w00_ = wptr[0], w01_ = wptr[1], w02_ = wptr[2],
|
||||
w10 = wptr[3], w11 = wptr[4], w12 = wptr[5],
|
||||
w20_ = wptr[6], w21_ = wptr[7], w22_ = wptr[8];
|
||||
int outW1 = min(outW, (width - dilation_w*(kernel_w - 1) + pad_l)/stride_w);
|
||||
float relu_coeff = relu ? relu[out_d] : 1.f, bias = biasptr[out_d];
|
||||
|
||||
for (int out_i = 0; out_i < outH; out_i++)
|
||||
{
|
||||
int in_i = out_i * stride_h - pad_t, out_j = 0;
|
||||
const float* imgptr0 = inptr_ + in_i*width;
|
||||
const float* imgptr1 = imgptr0 + dilation_h*width;
|
||||
const float* imgptr2 = imgptr0 + (dilation_h*2)*width;
|
||||
float out, w00 = w00_, w01 = w01_, w02 = w02_;
|
||||
float w20 = w20_, w21 = w21_, w22 = w22_;
|
||||
if (in_i < 0)
|
||||
{
|
||||
w00 = w01 = w02 = 0.f;
|
||||
imgptr0 = imgptr1;
|
||||
}
|
||||
else if (in_i + dilation_h*(kernel_h-1) >= height)
|
||||
{
|
||||
w20 = w21 = w22 = 0.f;
|
||||
imgptr2 = imgptr1;
|
||||
}
|
||||
float* outptr = outptr_ + out_i*outW;
|
||||
if (pad_l > 0)
|
||||
{
|
||||
out = imgptr0[0]*w01 + imgptr0[dilation_w]*w02 +
|
||||
imgptr1[0]*w11 + imgptr1[dilation_w]*w12 +
|
||||
imgptr2[0]*w21 + imgptr2[dilation_w]*w22 + bias;
|
||||
if (relu)
|
||||
out = out > 0.f ? out : out*relu_coeff;
|
||||
outptr[0] = out;
|
||||
out_j = 1;
|
||||
}
|
||||
|
||||
#if CV_SIMD
|
||||
// maybe with AVX or AVX512 strided depthwise convolution
|
||||
// can be accelerated with vector code, but with 4xfloat vectors
|
||||
// it's hardly the case
|
||||
if( stride_w == 1 )
|
||||
{
|
||||
const int VECSZ = v_float32::nlanes;
|
||||
const int out_delta = VECSZ/stride_w;
|
||||
v_float32 vw00 = vx_setall_f32(w00), vw01 = vx_setall_f32(w01), vw02 = vx_setall_f32(w02),
|
||||
vw10 = vx_setall_f32(w10), vw11 = vx_setall_f32(w11), vw12 = vx_setall_f32(w12),
|
||||
vw20 = vx_setall_f32(w20), vw21 = vx_setall_f32(w21), vw22 = vx_setall_f32(w22);
|
||||
v_float32 z = vx_setzero_f32(), vbias = vx_setall_f32(bias), vrc = vx_setall_f32(relu_coeff);
|
||||
for( ; out_j < outW1; out_j += out_delta )
|
||||
{
|
||||
if (out_j + out_delta > outW1)
|
||||
{
|
||||
if (out_j <= pad_l)
|
||||
break;
|
||||
out_j = outW1 - out_delta;
|
||||
}
|
||||
int in_j = out_j * stride_w - pad_l;
|
||||
v_float32 v00 = vx_load(imgptr0 + in_j),
|
||||
v01 = vx_load(imgptr0 + in_j + dilation_w),
|
||||
v02 = vx_load(imgptr0 + in_j + dilation_w*2),
|
||||
v10 = vx_load(imgptr1 + in_j),
|
||||
v11 = vx_load(imgptr1 + in_j + dilation_w),
|
||||
v12 = vx_load(imgptr1 + in_j + dilation_w*2),
|
||||
v20 = vx_load(imgptr2 + in_j),
|
||||
v21 = vx_load(imgptr2 + in_j + dilation_w),
|
||||
v22 = vx_load(imgptr2 + in_j + dilation_w*2);
|
||||
|
||||
v_float32 vout = v00*vw00 + v01*vw01 + v02*vw02 +
|
||||
v10*vw10 + v11*vw11 + v12*vw12 +
|
||||
v20*vw20 + v21*vw21 + v22*vw22 + vbias;
|
||||
if (relu)
|
||||
vout = v_select(vout > z, vout, vout*vrc);
|
||||
vx_store(outptr + out_j, vout);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
for (; out_j < outW1; out_j++)
|
||||
{
|
||||
int in_j = out_j * stride_w - pad_l;
|
||||
out = imgptr0[in_j]*w00 + imgptr0[in_j + dilation_w]*w01 + imgptr0[in_j + dilation_w*2]*w02 +
|
||||
imgptr1[in_j]*w10 + imgptr1[in_j + dilation_w]*w11 + imgptr1[in_j + dilation_w*2]*w12 +
|
||||
imgptr2[in_j]*w20 + imgptr2[in_j + dilation_w]*w21 + imgptr2[in_j + dilation_w*2]*w22 + bias;
|
||||
if (relu)
|
||||
out = out > 0.f ? out : out*relu_coeff;
|
||||
outptr[out_j] = out;
|
||||
}
|
||||
|
||||
for (; out_j < outW; out_j++ )
|
||||
{
|
||||
int in_j0 = out_j * stride_w - pad_l, in_j1 = in_j0 + dilation_w, in_j2 = in_j0 + dilation_w*2;
|
||||
float s0 = 1.f, s1 = 1.f, s2 = 1.f;
|
||||
if (in_j0 >= width)
|
||||
{
|
||||
in_j0 = 0;
|
||||
s0 = 0.f;
|
||||
}
|
||||
if (in_j1 >= width)
|
||||
{
|
||||
in_j1 = 0;
|
||||
s1 = 0.f;
|
||||
}
|
||||
if (in_j2 >= width)
|
||||
{
|
||||
in_j2 = 0;
|
||||
s2 = 0.f;
|
||||
}
|
||||
out = imgptr0[in_j0]*w00*s0 + imgptr0[in_j1]*w01*s1 + imgptr0[in_j2]*w02*s2 +
|
||||
imgptr1[in_j0]*w10*s0 + imgptr1[in_j1]*w11*s1 + imgptr1[in_j2]*w12*s2 +
|
||||
imgptr2[in_j0]*w20*s0 + imgptr2[in_j1]*w21*s1 + imgptr2[in_j2]*w22*s2 + bias;
|
||||
if (relu)
|
||||
out = out > 0.f ? out : out*relu_coeff;
|
||||
outptr[out_j] = out;
|
||||
}
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// do im2row for a part of input tensor
|
||||
float* rowbuf = rowbuf0;
|
||||
|
||||
if (isConv2D)
|
||||
{
|
||||
if( is1x1 && stride_w == 1 && stride_h == 1 )
|
||||
{
|
||||
const float* imgptr = data_inp0 + (cn0*height + out_i)*width + out_j;
|
||||
for( int j = 0; j < bsz; j++, rowbuf += vsz_a )
|
||||
{
|
||||
if( j + 4 <= bsz )
|
||||
{
|
||||
k = 0;
|
||||
#if CV_SIMD128
|
||||
for( ; k <= vsz - 4; k += 4 )
|
||||
{
|
||||
const float* inp = imgptr + j + k*inpPlaneSize;
|
||||
v_float32x4 p0 = v_load(inp), p1 = v_load(inp + inpPlaneSize);
|
||||
v_float32x4 p2 = v_load(inp + inpPlaneSize*2), p3 = v_load(inp + inpPlaneSize*3);
|
||||
v_float32x4 r0, r1, r2, r3;
|
||||
v_transpose4x4(p0, p1, p2, p3, r0, r1, r2, r3);
|
||||
v_store(rowbuf + k, r0);
|
||||
v_store(rowbuf + k + vsz_a, r1);
|
||||
v_store(rowbuf + k + vsz_a*2, r2);
|
||||
v_store(rowbuf + k + vsz_a*3, r3);
|
||||
}
|
||||
#endif
|
||||
for( ; k < vsz; k++ )
|
||||
{
|
||||
const float* inp = imgptr + j + k*inpPlaneSize;
|
||||
float v0 = inp[0], v1 = inp[1], v2 = inp[2], v3 = inp[3];
|
||||
rowbuf[k] = v0;
|
||||
rowbuf[k + vsz_a] = v1;
|
||||
rowbuf[k + vsz_a*2] = v2;
|
||||
rowbuf[k + vsz_a*3] = v3;
|
||||
}
|
||||
j += 3;
|
||||
rowbuf += vsz_a*3;
|
||||
}
|
||||
else
|
||||
{
|
||||
for( k = 0; k < vsz; k++ )
|
||||
{
|
||||
rowbuf[k] = imgptr[j + k*inpPlaneSize];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
for( ofs = ofs0; ofs < ofs1; out_j = 0, ++out_i )
|
||||
{
|
||||
int delta = std::min(ofs1 - ofs, outW - out_j);
|
||||
@@ -951,7 +1197,6 @@ public:
|
||||
|
||||
// now compute dot product of the weights
|
||||
// and im2row-transformed part of the tensor
|
||||
int bsz = ofs1 - ofs0;
|
||||
#if CV_TRY_AVX512_SKX
|
||||
/* AVX512 convolution requires an alignment of 16, and ROI is only there for larger vector sizes */
|
||||
if(useAVX512)
|
||||
@@ -1103,6 +1348,26 @@ public:
|
||||
for (int i = 0; i < inputs.size(); ++i)
|
||||
CV_Assert(inputs[i].u != outputs[0].u);
|
||||
|
||||
if (blobs.empty())
|
||||
{
|
||||
size_t n = inputs.size() - 1;
|
||||
umat_blobs.resize(n);
|
||||
for (size_t i = 0; i < n; i++)
|
||||
{
|
||||
if (use_half)
|
||||
{
|
||||
Mat matFP32;
|
||||
convertFp16(inputs[i + 1], matFP32);
|
||||
matFP32.copyTo(umat_blobs[i]);
|
||||
}
|
||||
else
|
||||
{
|
||||
inputs[i + 1].copyTo(umat_blobs[i]);
|
||||
}
|
||||
}
|
||||
inputs.resize(1);
|
||||
}
|
||||
|
||||
if (umat_blobs.empty())
|
||||
{
|
||||
size_t n = blobs.size();
|
||||
@@ -1113,7 +1378,7 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
if (convolutionOp.empty())
|
||||
if (convolutionOp.empty() || blobs.empty())
|
||||
{
|
||||
OCL4DNNConvConfig config;
|
||||
config.in_shape = shape(inputs[0]);
|
||||
@@ -1123,7 +1388,7 @@ public:
|
||||
config.stride = stride;
|
||||
config.dilation = dilation;
|
||||
config.group = inputs[0].size[1] / umat_blobs[0].size[1];
|
||||
config.bias_term = (hasBias()) ? true : false;
|
||||
config.bias_term = umat_blobs.size() == 2;
|
||||
config.use_half = use_half;
|
||||
|
||||
convolutionOp = Ptr<OCL4DNNConvSpatial<float> >(new OCL4DNNConvSpatial<float>(config));
|
||||
@@ -1250,16 +1515,37 @@ public:
|
||||
inputs_arr.getMatVector(inputs);
|
||||
outputs_arr.getMatVector(outputs);
|
||||
|
||||
int outCn = blobs.empty() ? inputs[1].size[0] : blobs[0].size[0];
|
||||
// Need to align non-const blobs
|
||||
if (blobs.empty())
|
||||
{
|
||||
Mat wm = inputs[1].reshape(1, outCn);
|
||||
if( wm.step1() % VEC_ALIGN != 0 )
|
||||
{
|
||||
wm.copyTo(weightsMat);
|
||||
if (inputs.size() > 2)
|
||||
{
|
||||
Mat biasMat = inputs[2].reshape(1, outCn);
|
||||
biasMat.col(0).copyTo(biasvec);
|
||||
biasvec.resize(outCn + 2);
|
||||
}
|
||||
else
|
||||
{
|
||||
biasvec.resize(outCn + 2, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*printf("conv %s: input (%d x %d x %d x %d), kernel (%d x %d), pad (%d x %d), stride (%d x %d), dilation (%d x %d)\n",
|
||||
name.c_str(), inputs[0].size[0], inputs[0].size[1], inputs[0].size[2], inputs[0].size[3],
|
||||
kernel.width, kernel.height, pad.width, pad.height,
|
||||
stride.width, stride.height, dilation.width, dilation.height);*/
|
||||
CV_Assert_N(inputs.size() == (size_t)1, inputs[0].size[1] % blobs[0].size[1] == 0,
|
||||
int inpGroupCn = blobs.empty() ? inputs[1].size[1] : blobs[0].size[1];
|
||||
CV_Assert_N(inputs.size() >= (size_t)1, inputs[0].size[1] % inpGroupCn == 0,
|
||||
outputs.size() == 1, inputs[0].data != outputs[0].data);
|
||||
|
||||
int ngroups = inputs[0].size[1]/blobs[0].size[1];
|
||||
int ngroups = inputs[0].size[1] / inpGroupCn;
|
||||
CV_Assert(outputs[0].size[1] % ngroups == 0);
|
||||
int outCn = blobs[0].size[0];
|
||||
|
||||
reluslope.clear();
|
||||
if( activ )
|
||||
@@ -1328,11 +1614,11 @@ public:
|
||||
virtual int64 getFLOPS(const std::vector<MatShape> &inputs,
|
||||
const std::vector<MatShape> &outputs) const CV_OVERRIDE
|
||||
{
|
||||
CV_Assert(inputs.size() == outputs.size());
|
||||
CV_Assert(inputs.size() == outputs.size() || inputs.size() == outputs.size() + blobs.size());
|
||||
|
||||
int64 flops = 0;
|
||||
int karea = std::accumulate(kernel_size.begin(), kernel_size.end(), 1, std::multiplies<size_t>());
|
||||
for (int i = 0; i < inputs.size(); i++)
|
||||
for (int i = 0; i < outputs.size(); i++)
|
||||
{
|
||||
flops += total(outputs[i])*(CV_BIG_INT(2)*karea*inputs[i][1] + 1);
|
||||
}
|
||||
|
||||
@@ -669,9 +669,14 @@ struct MishFunctor : public BaseFunctor
|
||||
{
|
||||
// Use fast approximation introduced in https://github.com/opencv/opencv/pull/17200
|
||||
float x = srcptr[i];
|
||||
float eX = exp(std::min(x, 20.f));
|
||||
float n = (eX + 2) * eX;
|
||||
dstptr[i] = (x * n) / (n + 2);
|
||||
if (x >= 8.f)
|
||||
dstptr[i] = x;
|
||||
else
|
||||
{
|
||||
float eX = exp(x);
|
||||
float n = (eX + 2) * eX;
|
||||
dstptr[i] = (x * n) / (n + 2);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -116,7 +116,6 @@ public:
|
||||
CV_CheckEQ(inputs.size(), (size_t)2, "");
|
||||
numOutput = inputs[1].back();
|
||||
cAxis = inputs[0].size() - 1;
|
||||
CV_CheckEQ(numOutput, inputs[0][cAxis - 1], "");
|
||||
int dims = inputs[0].size();
|
||||
CV_CheckEQ(inputs[1].size(), (size_t)dims, "");
|
||||
CV_CheckGE(dims, 2, "");
|
||||
@@ -565,7 +564,7 @@ public:
|
||||
}
|
||||
else
|
||||
{
|
||||
std::vector<size_t> data = {(size_t)ieInpNode->get_shape()[0], (size_t)blobs[0].size[1]};
|
||||
std::vector<int64_t> data = {(int64_t)ieInpNode->get_shape()[0], (int64_t)blobs[0].size[1]};
|
||||
auto new_shape = std::make_shared<ngraph::op::Constant>(ngraph::element::i64, ngraph::Shape{2}, data.data());
|
||||
auto inp = std::make_shared<ngraph::op::v1::Reshape>(ieInpNode, new_shape, true);
|
||||
|
||||
|
||||
@@ -50,6 +50,16 @@ void fastConv( const float* weights, size_t wstep, const float* bias,
|
||||
const float* rowbuf, float* output, const int* outShape,
|
||||
int blockSize, int vecsize, int vecsize_aligned,
|
||||
const float* relu, bool initOutput );
|
||||
void fastDepthwiseConv( const float* weights,
|
||||
int kernel_h, int kernel_w,
|
||||
int stride_h, int stride_w,
|
||||
int dilation_h, int dilation_w,
|
||||
int pad_t, int pad_l,
|
||||
const float* bias, const float* relu,
|
||||
const float* inptr,
|
||||
int height, int width,
|
||||
float* outptr,
|
||||
int out_d, int outH, int outW );
|
||||
void fastGEMM1T( const float* vec, const float* weights,
|
||||
size_t wstep, const float* bias,
|
||||
float* dst, int nvecs, int vecsize );
|
||||
@@ -64,6 +74,8 @@ void fastGEMM( const float* aptr, size_t astep, const float* bptr,
|
||||
#define _mm256_fmadd_ps(a, b, c) _mm256_add_ps(c, _mm256_mul_ps(a, b))
|
||||
#endif
|
||||
|
||||
enum { FASCONV_BASE_VECSZ = 4 };
|
||||
|
||||
void fastConv( const float* weights, size_t wstep, const float* bias,
|
||||
const float* rowbuf, float* output, const int* outShape,
|
||||
int blockSize, int vecsize, int vecsize_aligned,
|
||||
@@ -73,6 +85,11 @@ void fastConv( const float* weights, size_t wstep, const float* bias,
|
||||
size_t outPlaneSize = outShape[2]*outShape[3];
|
||||
float r0 = 1.f, r1 = 1.f, r2 = 1.f;
|
||||
__m128 vr0 = _mm_set1_ps(1.f), vr1 = vr0, vr2 = vr0, z = _mm_setzero_ps();
|
||||
int CV_DECL_ALIGNED(16) maskbuf[FASCONV_BASE_VECSZ] = {0};
|
||||
int rsz = blockSize % FASCONV_BASE_VECSZ;
|
||||
for( int i = 0; i < rsz; i++ )
|
||||
maskbuf[FASCONV_BASE_VECSZ - i - 1] = -1;
|
||||
__m128 mask = _mm_loadu_ps((const float*)maskbuf);
|
||||
|
||||
// now compute dot product of the weights
|
||||
// and im2row-transformed part of the tensor
|
||||
@@ -114,8 +131,16 @@ void fastConv( const float* weights, size_t wstep, const float* bias,
|
||||
}
|
||||
|
||||
int j = 0;
|
||||
for( ; j <= blockSize - 4; j += 4 )
|
||||
for( ; j < blockSize; j += FASCONV_BASE_VECSZ )
|
||||
{
|
||||
bool tail = false;
|
||||
if (j + FASCONV_BASE_VECSZ > blockSize)
|
||||
{
|
||||
if (j == 0)
|
||||
break;
|
||||
j = blockSize - FASCONV_BASE_VECSZ;
|
||||
tail = true;
|
||||
}
|
||||
int k = 0;
|
||||
const float* rptr = rowbuf + j*vecsize_aligned;
|
||||
|
||||
@@ -243,9 +268,16 @@ void fastConv( const float* weights, size_t wstep, const float* bias,
|
||||
__m128 m0 = _mm_cmp_ps(s0, z, _CMP_GT_OS);
|
||||
__m128 m1 = _mm_cmp_ps(s1, z, _CMP_GT_OS);
|
||||
__m128 m2 = _mm_cmp_ps(s2, z, _CMP_GT_OS);
|
||||
s0 = _mm_xor_ps(s0, _mm_andnot_ps(m0, _mm_xor_ps(_mm_mul_ps(s0, vr0), s0)));
|
||||
s1 = _mm_xor_ps(s1, _mm_andnot_ps(m1, _mm_xor_ps(_mm_mul_ps(s1, vr1), s1)));
|
||||
s2 = _mm_xor_ps(s2, _mm_andnot_ps(m2, _mm_xor_ps(_mm_mul_ps(s2, vr2), s2)));
|
||||
s0 = _mm_blendv_ps(_mm_mul_ps(s0, vr0), s0, m0);
|
||||
s1 = _mm_blendv_ps(_mm_mul_ps(s1, vr1), s1, m1);
|
||||
s2 = _mm_blendv_ps(_mm_mul_ps(s2, vr2), s2, m2);
|
||||
}
|
||||
|
||||
if( tail )
|
||||
{
|
||||
s0 = _mm_blendv_ps(_mm_loadu_ps(outptr0 + j), s0, mask);
|
||||
s1 = _mm_blendv_ps(_mm_loadu_ps(outptr1 + j), s1, mask);
|
||||
s2 = _mm_blendv_ps(_mm_loadu_ps(outptr2 + j), s2, mask);
|
||||
}
|
||||
|
||||
_mm_storeu_ps(outptr0 + j, s0);
|
||||
@@ -253,9 +285,55 @@ void fastConv( const float* weights, size_t wstep, const float* bias,
|
||||
_mm_storeu_ps(outptr2 + j, s2);
|
||||
}
|
||||
|
||||
for( ; j <= blockSize - 2; j += 2 )
|
||||
{
|
||||
const float* rptr0 = rowbuf + j*vecsize_aligned;
|
||||
const float* rptr1 = rowbuf + (j+1)*vecsize_aligned;
|
||||
float s00, s01, s10, s11, s20, s21;
|
||||
|
||||
if( initOutput )
|
||||
{
|
||||
s00 = s01 = bias0;
|
||||
s10 = s11 = bias1;
|
||||
s20 = s21 = bias2;
|
||||
}
|
||||
else
|
||||
{
|
||||
s00 = outptr0[j]; s01 = outptr0[j+1];
|
||||
s10 = outptr1[j]; s11 = outptr1[j+1];
|
||||
s20 = outptr2[j]; s21 = outptr2[j+1];
|
||||
}
|
||||
|
||||
for( int k = 0; k < vecsize; k++ )
|
||||
{
|
||||
float w0 = wptr0[k], w1 = wptr1[k], w2 = wptr2[k];
|
||||
float r = rptr0[k];
|
||||
s00 += w0*r; s10 += w1*r; s20 += w2*r;
|
||||
r = rptr1[k];
|
||||
s01 += w0*r; s11 += w1*r; s21 += w2*r;
|
||||
}
|
||||
|
||||
if( relu )
|
||||
{
|
||||
s00 = s00 > 0.f ? s00 : s00*r0;
|
||||
s01 = s01 > 0.f ? s01 : s01*r0;
|
||||
s10 = s10 > 0.f ? s10 : s10*r1;
|
||||
s11 = s11 > 0.f ? s11 : s11*r1;
|
||||
s20 = s20 > 0.f ? s20 : s20*r2;
|
||||
s21 = s21 > 0.f ? s21 : s21*r2;
|
||||
}
|
||||
|
||||
outptr0[j] = s00;
|
||||
outptr0[j+1] = s01;
|
||||
outptr1[j] = s10;
|
||||
outptr1[j+1] = s11;
|
||||
outptr2[j] = s20;
|
||||
outptr2[j+1] = s21;
|
||||
}
|
||||
|
||||
for( ; j < blockSize; j++ )
|
||||
{
|
||||
const float* rptr = rowbuf + j*vecsize_aligned;
|
||||
const float* rptr0 = rowbuf + j*vecsize_aligned;
|
||||
float s00, s10, s20;
|
||||
|
||||
if( initOutput )
|
||||
@@ -273,10 +351,9 @@ void fastConv( const float* weights, size_t wstep, const float* bias,
|
||||
|
||||
for( int k = 0; k < vecsize; k++ )
|
||||
{
|
||||
float r0 = rptr[k];
|
||||
s00 += wptr0[k]*r0;
|
||||
s10 += wptr1[k]*r0;
|
||||
s20 += wptr2[k]*r0;
|
||||
float w0 = wptr0[k], w1 = wptr1[k], w2 = wptr2[k];
|
||||
float r = rptr0[k];
|
||||
s00 += w0*r; s10 += w1*r; s20 += w2*r;
|
||||
}
|
||||
|
||||
if( relu )
|
||||
@@ -294,6 +371,185 @@ void fastConv( const float* weights, size_t wstep, const float* bias,
|
||||
_mm256_zeroupper();
|
||||
}
|
||||
|
||||
static inline void _mm256_load_deinterleave(const float* ptr, __m256& a, __m256& b)
|
||||
{
|
||||
__m256 t0 = _mm256_loadu_ps(ptr);
|
||||
__m256 t1 = _mm256_loadu_ps(ptr + 8);
|
||||
|
||||
__m256 lo = _mm256_permute2f128_ps(t0, t1, 0+2*16);
|
||||
__m256 hi = _mm256_permute2f128_ps(t0, t1, 1+3*16);
|
||||
a = _mm256_shuffle_ps(lo, hi, 0x88);
|
||||
b = _mm256_shuffle_ps(lo, hi, 0xdd);
|
||||
}
|
||||
|
||||
void fastDepthwiseConv( const float* wptr,
|
||||
int kernel_h, int kernel_w,
|
||||
int stride_h, int stride_w,
|
||||
int dilation_h, int dilation_w,
|
||||
int pad_t, int pad_l,
|
||||
const float* biasptr, const float* relu,
|
||||
const float* inptr_,
|
||||
int height, int width,
|
||||
float* outptr_,
|
||||
int out_d, int outH, int outW )
|
||||
{
|
||||
const float w00_ = wptr[0], w01_ = wptr[1], w02_ = wptr[2],
|
||||
w10 = wptr[3], w11 = wptr[4], w12 = wptr[5],
|
||||
w20_ = wptr[6], w21_ = wptr[7], w22_ = wptr[8];
|
||||
int outW1 = min(outW, (width - dilation_w*(kernel_w - 1) + pad_l)/stride_w);
|
||||
float relu_coeff = relu ? relu[out_d] : 1.f, bias = biasptr[out_d];
|
||||
|
||||
for (int out_i = 0; out_i < outH; out_i++)
|
||||
{
|
||||
int in_i = out_i * stride_h - pad_t, out_j = 0;
|
||||
const float* imgptr0 = inptr_ + in_i*width;
|
||||
const float* imgptr1 = imgptr0 + dilation_h*width;
|
||||
const float* imgptr2 = imgptr0 + (dilation_h*2)*width;
|
||||
float out, w00 = w00_, w01 = w01_, w02 = w02_;
|
||||
float w20 = w20_, w21 = w21_, w22 = w22_;
|
||||
if (in_i < 0)
|
||||
{
|
||||
w00 = w01 = w02 = 0.f;
|
||||
imgptr0 = imgptr1;
|
||||
}
|
||||
else if (in_i + dilation_h*(kernel_h-1) >= height)
|
||||
{
|
||||
w20 = w21 = w22 = 0.f;
|
||||
imgptr2 = imgptr1;
|
||||
}
|
||||
float* outptr = outptr_ + out_i*outW;
|
||||
if (pad_l > 0)
|
||||
{
|
||||
out = imgptr0[0]*w01 + imgptr0[dilation_w]*w02 +
|
||||
imgptr1[0]*w11 + imgptr1[dilation_w]*w12 +
|
||||
imgptr2[0]*w21 + imgptr2[dilation_w]*w22 + bias;
|
||||
if (relu)
|
||||
out = out > 0.f ? out : out*relu_coeff;
|
||||
outptr[0] = out;
|
||||
out_j = 1;
|
||||
}
|
||||
|
||||
if (stride_w == 1 || (stride_w == 2 && dilation_w == 1))
|
||||
{
|
||||
const int VECSZ = 8;
|
||||
__m256 vw00 = _mm256_set1_ps(w00), vw01 = _mm256_set1_ps(w01), vw02 = _mm256_set1_ps(w02),
|
||||
vw10 = _mm256_set1_ps(w10), vw11 = _mm256_set1_ps(w11), vw12 = _mm256_set1_ps(w12),
|
||||
vw20 = _mm256_set1_ps(w20), vw21 = _mm256_set1_ps(w21), vw22 = _mm256_set1_ps(w22);
|
||||
__m256 z = _mm256_setzero_ps(), vbias = _mm256_set1_ps(bias), vrc = _mm256_set1_ps(relu_coeff);
|
||||
|
||||
if( stride_w == 1 )
|
||||
for( ; out_j < outW1; out_j += VECSZ )
|
||||
{
|
||||
if (out_j + VECSZ > outW1 && out_j > pad_l)
|
||||
out_j = outW1 - VECSZ;
|
||||
int in_j = out_j * stride_w - pad_l;
|
||||
__m256 v00 = _mm256_loadu_ps(imgptr0 + in_j),
|
||||
v01 = _mm256_loadu_ps(imgptr0 + in_j + dilation_w),
|
||||
v02 = _mm256_loadu_ps(imgptr0 + in_j + dilation_w*2),
|
||||
v10 = _mm256_loadu_ps(imgptr1 + in_j),
|
||||
v11 = _mm256_loadu_ps(imgptr1 + in_j + dilation_w),
|
||||
v12 = _mm256_loadu_ps(imgptr1 + in_j + dilation_w*2),
|
||||
v20 = _mm256_loadu_ps(imgptr2 + in_j),
|
||||
v21 = _mm256_loadu_ps(imgptr2 + in_j + dilation_w),
|
||||
v22 = _mm256_loadu_ps(imgptr2 + in_j + dilation_w*2);
|
||||
|
||||
__m256 vout0 = _mm256_fmadd_ps(v00, vw00, vbias);
|
||||
__m256 vout1 = _mm256_mul_ps(v01, vw01);
|
||||
__m256 vout2 = _mm256_mul_ps(v02, vw02);
|
||||
|
||||
vout0 = _mm256_fmadd_ps(v10, vw10, vout0);
|
||||
vout1 = _mm256_fmadd_ps(v11, vw11, vout1);
|
||||
vout2 = _mm256_fmadd_ps(v12, vw12, vout2);
|
||||
|
||||
vout0 = _mm256_fmadd_ps(v20, vw20, vout0);
|
||||
vout1 = _mm256_fmadd_ps(v21, vw21, vout1);
|
||||
vout2 = _mm256_fmadd_ps(v22, vw22, vout2);
|
||||
|
||||
vout0 = _mm256_add_ps(_mm256_add_ps(vout0, vout1), vout2);
|
||||
if (relu)
|
||||
{
|
||||
__m256 m = _mm256_cmp_ps(vout0, z, _CMP_GT_OQ);
|
||||
vout0 = _mm256_blendv_ps(_mm256_mul_ps(vout0, vrc), vout0, m);
|
||||
}
|
||||
_mm256_storeu_ps(outptr + out_j, vout0);
|
||||
}
|
||||
else
|
||||
for( ; out_j < outW1; out_j += VECSZ )
|
||||
{
|
||||
if (out_j + VECSZ > outW1 && out_j > pad_l)
|
||||
out_j = outW1 - VECSZ;
|
||||
int in_j = out_j * stride_w - pad_l;
|
||||
__m256 v00, v01, v02, v10, v11, v12, v20, v21, v22, unused;
|
||||
_mm256_load_deinterleave(imgptr0 + in_j, v00, v01);
|
||||
_mm256_load_deinterleave(imgptr0 + in_j + 2, v02, unused);
|
||||
_mm256_load_deinterleave(imgptr1 + in_j, v10, v11);
|
||||
_mm256_load_deinterleave(imgptr1 + in_j + 2, v12, unused);
|
||||
_mm256_load_deinterleave(imgptr2 + in_j, v20, v21);
|
||||
_mm256_load_deinterleave(imgptr2 + in_j + 2, v22, unused);
|
||||
|
||||
__m256 vout0 = _mm256_fmadd_ps(v00, vw00, vbias);
|
||||
__m256 vout1 = _mm256_mul_ps(v01, vw01);
|
||||
__m256 vout2 = _mm256_mul_ps(v02, vw02);
|
||||
|
||||
vout0 = _mm256_fmadd_ps(v10, vw10, vout0);
|
||||
vout1 = _mm256_fmadd_ps(v11, vw11, vout1);
|
||||
vout2 = _mm256_fmadd_ps(v12, vw12, vout2);
|
||||
|
||||
vout0 = _mm256_fmadd_ps(v20, vw20, vout0);
|
||||
vout1 = _mm256_fmadd_ps(v21, vw21, vout1);
|
||||
vout2 = _mm256_fmadd_ps(v22, vw22, vout2);
|
||||
|
||||
vout0 = _mm256_add_ps(_mm256_add_ps(vout0, vout1), vout2);
|
||||
if (relu)
|
||||
{
|
||||
__m256 m = _mm256_cmp_ps(vout0, z, _CMP_GT_OQ);
|
||||
vout0 = _mm256_blendv_ps(_mm256_mul_ps(vout0, vrc), vout0, m);
|
||||
}
|
||||
_mm256_storeu_ps(outptr + out_j, vout0);
|
||||
}
|
||||
}
|
||||
|
||||
for (; out_j < outW1; out_j++)
|
||||
{
|
||||
int in_j = out_j * stride_w - pad_l;
|
||||
out = imgptr0[in_j]*w00 + imgptr0[in_j + dilation_w]*w01 + imgptr0[in_j + dilation_w*2]*w02 +
|
||||
imgptr1[in_j]*w10 + imgptr1[in_j + dilation_w]*w11 + imgptr1[in_j + dilation_w*2]*w12 +
|
||||
imgptr2[in_j]*w20 + imgptr2[in_j + dilation_w]*w21 + imgptr2[in_j + dilation_w*2]*w22 + bias;
|
||||
if (relu)
|
||||
out = out > 0.f ? out : out*relu_coeff;
|
||||
outptr[out_j] = out;
|
||||
}
|
||||
|
||||
for (; out_j < outW; out_j++ )
|
||||
{
|
||||
int in_j0 = out_j * stride_w - pad_l, in_j1 = in_j0 + dilation_w, in_j2 = in_j0 + dilation_w*2;
|
||||
float s0 = 1.f, s1 = 1.f, s2 = 1.f;
|
||||
if (in_j0 >= width)
|
||||
{
|
||||
in_j0 = 0;
|
||||
s0 = 0.f;
|
||||
}
|
||||
if (in_j1 >= width)
|
||||
{
|
||||
in_j1 = 0;
|
||||
s1 = 0.f;
|
||||
}
|
||||
if (in_j2 >= width)
|
||||
{
|
||||
in_j2 = 0;
|
||||
s2 = 0.f;
|
||||
}
|
||||
out = imgptr0[in_j0]*w00*s0 + imgptr0[in_j1]*w01*s1 + imgptr0[in_j2]*w02*s2 +
|
||||
imgptr1[in_j0]*w10*s0 + imgptr1[in_j1]*w11*s1 + imgptr1[in_j2]*w12*s2 +
|
||||
imgptr2[in_j0]*w20*s0 + imgptr2[in_j1]*w21*s1 + imgptr2[in_j2]*w22*s2 + bias;
|
||||
if (relu)
|
||||
out = out > 0.f ? out : out*relu_coeff;
|
||||
outptr[out_j] = out;
|
||||
}
|
||||
}
|
||||
_mm256_zeroupper();
|
||||
}
|
||||
|
||||
// dst = vec * weights^t + bias
|
||||
void fastGEMM1T( const float* vec, const float* weights,
|
||||
size_t wstep, const float* bias,
|
||||
|
||||
@@ -174,21 +174,9 @@ public:
|
||||
computeStrides(shape(inputs[0]), shape(outputs[0]));
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
if (uorder.empty())
|
||||
{
|
||||
std::vector<int> orderVec(_order.begin(), _order.end());;
|
||||
Mat morder(1, orderVec.size(), CV_32SC1, &orderVec[0]);
|
||||
|
||||
std::vector<int> oldStrideVec(_oldStride.begin(), _oldStride.end());
|
||||
Mat mold_stride(1, _oldStride.size(), CV_32SC1, &oldStrideVec[0]);
|
||||
|
||||
std::vector<int> newStrideVec(_newStride.begin(), _newStride.end());
|
||||
Mat mnew_stride(1, newStrideVec.size(), CV_32SC1, &newStrideVec[0]);
|
||||
|
||||
morder.copyTo(uorder);
|
||||
mold_stride.copyTo(uold_stride);
|
||||
mnew_stride.copyTo(unew_stride);
|
||||
}
|
||||
uorder.release();
|
||||
uold_stride.release();
|
||||
unew_stride.release();
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -276,6 +264,22 @@ public:
|
||||
if (!_needsPermute)
|
||||
return false;
|
||||
|
||||
if (uorder.empty())
|
||||
{
|
||||
std::vector<int> orderVec(_order.begin(), _order.end());;
|
||||
Mat morder(1, orderVec.size(), CV_32SC1, &orderVec[0]);
|
||||
|
||||
std::vector<int> oldStrideVec(_oldStride.begin(), _oldStride.end());
|
||||
Mat mold_stride(1, _oldStride.size(), CV_32SC1, &oldStrideVec[0]);
|
||||
|
||||
std::vector<int> newStrideVec(_newStride.begin(), _newStride.end());
|
||||
Mat mnew_stride(1, newStrideVec.size(), CV_32SC1, &newStrideVec[0]);
|
||||
|
||||
morder.copyTo(uorder);
|
||||
mold_stride.copyTo(uold_stride);
|
||||
mnew_stride.copyTo(unew_stride);
|
||||
}
|
||||
|
||||
bool use_half = (inps.depth() == CV_16S);
|
||||
String opts = format("-DDtype=%s", use_half ? "half" : "float");
|
||||
for (size_t i = 0; i < inputs.size(); i++)
|
||||
@@ -385,8 +389,9 @@ public:
|
||||
const std::vector<Ptr<BackendNode> >& nodes) CV_OVERRIDE
|
||||
{
|
||||
auto& ieInpNode = nodes[0].dynamicCast<InfEngineNgraphNode>()->node;
|
||||
std::vector<int64_t> order(_order.begin(), _order.end());
|
||||
auto tr_axes = std::make_shared<ngraph::op::Constant>(ngraph::element::i64,
|
||||
ngraph::Shape({_order.size()}), _order.data());
|
||||
ngraph::Shape({order.size()}), order.data());
|
||||
auto transpose = std::make_shared<ngraph::op::Transpose>(ieInpNode, tr_axes);
|
||||
return Ptr<BackendNode>(new InfEngineNgraphNode(transpose));
|
||||
}
|
||||
|
||||
@@ -98,6 +98,8 @@ public:
|
||||
type = AVE;
|
||||
else if (pool == "stochastic")
|
||||
type = STOCHASTIC;
|
||||
else if (pool == "sum")
|
||||
type = SUM;
|
||||
else
|
||||
CV_Error(Error::StsBadArg, "Unknown pooling type \"" + pool + "\"");
|
||||
|
||||
@@ -195,7 +197,7 @@ public:
|
||||
return type == MAX || type == AVE;
|
||||
}
|
||||
else
|
||||
return type != STOCHASTIC;
|
||||
return type != STOCHASTIC && type != SUM;
|
||||
}
|
||||
#endif
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
@@ -288,7 +290,7 @@ public:
|
||||
maxPooling(inputs[0], outputs[0], mask);
|
||||
break;
|
||||
}
|
||||
case AVE:
|
||||
case AVE: case SUM:
|
||||
CV_Assert_N(inputs.size() == 1, outputs.size() == 1);
|
||||
avePooling(inputs[0], outputs[0]);
|
||||
break;
|
||||
@@ -366,7 +368,7 @@ public:
|
||||
virtual Ptr<BackendNode> initNgraph(const std::vector<Ptr<BackendWrapper> >& inputs,
|
||||
const std::vector<Ptr<BackendNode> >& nodes) CV_OVERRIDE
|
||||
{
|
||||
CV_Assert_N((inputs.size() == 1 && (type == MAX || type == AVE)) || inputs.size() == 2, nodes.size() == inputs.size());
|
||||
CV_Assert_N((inputs.size() == 1 && (type == MAX || type == AVE || type == SUM)) || inputs.size() == 2, nodes.size() == inputs.size());
|
||||
auto& ieInpNode = nodes[0].dynamicCast<InfEngineNgraphNode>()->node;
|
||||
|
||||
ngraph::op::PadType pad_type = ngraph::op::PadType::EXPLICIT;
|
||||
@@ -381,6 +383,19 @@ virtual Ptr<BackendNode> initNgraph(const std::vector<Ptr<BackendWrapper> >& inp
|
||||
exclude_pad, rounding_type, pad_type);
|
||||
return Ptr<BackendNode>(new InfEngineNgraphNode(ave_pool));
|
||||
}
|
||||
else if (type == SUM) {
|
||||
ngraph::Shape inpShape = ieInpNode->get_shape();
|
||||
CV_Assert(inpShape.size() == 2 + kernel_size.size());
|
||||
std::vector<int64_t> axes;
|
||||
for (size_t i = 0; i < kernel_size.size(); i++)
|
||||
{
|
||||
if (inpShape[2 + i] == kernel_size[i])
|
||||
axes.push_back(2 + i);
|
||||
}
|
||||
auto reduction_axes = std::make_shared<ngraph::op::Constant>(ngraph::element::i64, ngraph::Shape{axes.size()}, axes);
|
||||
auto reduce_sum = std::make_shared<ngraph::op::v1::ReduceSum>(ieInpNode, reduction_axes, true);
|
||||
return Ptr<BackendNode>(new InfEngineNgraphNode(reduce_sum));
|
||||
}
|
||||
else if (type == MAX) {
|
||||
auto max_pool = std::make_shared<ngraph::op::v1::MaxPool>(ieInpNode, ngraph::Strides(strides),
|
||||
ngraph::Shape(pads_begin), ngraph::Shape(pads_end), ngraph::Shape(kernel_size),
|
||||
@@ -739,7 +754,7 @@ virtual Ptr<BackendNode> initNgraph(const std::vector<Ptr<BackendWrapper> >& inp
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (poolingType == AVE)
|
||||
else if (poolingType == AVE || poolingType == SUM)
|
||||
{
|
||||
for( ; x0 < x1; ++x0)
|
||||
{
|
||||
@@ -750,7 +765,7 @@ virtual Ptr<BackendNode> initNgraph(const std::vector<Ptr<BackendWrapper> >& inp
|
||||
xend = min(xend, inp_width);
|
||||
float inv_kernel_area = avePoolPaddedArea ? xdelta * ydelta * ddelta :
|
||||
((dend - dstart) * (yend - ystart) * (xend - xstart));
|
||||
inv_kernel_area = 1.0 / inv_kernel_area;
|
||||
inv_kernel_area = poolingType == AVE ? 1.0 / inv_kernel_area : 1.0;
|
||||
#if CV_SIMD128
|
||||
if( isPool2D && xstart > 0 && x0 + 7 < x1 && (x0 + 7) * stride_w - pad_l + kernel_w < inp_width )
|
||||
{
|
||||
@@ -1095,6 +1110,7 @@ private:
|
||||
MAX,
|
||||
AVE,
|
||||
STOCHASTIC,
|
||||
SUM,
|
||||
ROI, // RoI pooling, https://arxiv.org/pdf/1504.08083.pdf
|
||||
PSROI // Position-sensitive RoI pooling, https://arxiv.org/pdf/1605.06409.pdf
|
||||
};
|
||||
|
||||
@@ -39,7 +39,7 @@ public:
|
||||
CV_Assert(params.has("zoom_factor_x") && params.has("zoom_factor_y"));
|
||||
}
|
||||
interpolation = params.get<String>("interpolation");
|
||||
CV_Assert(interpolation == "nearest" || interpolation == "opencv_linear" || interpolation == "bilinear");
|
||||
CV_Check(interpolation, interpolation == "nearest" || interpolation == "opencv_linear" || interpolation == "bilinear", "");
|
||||
|
||||
alignCorners = params.get<bool>("align_corners", false);
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@ Implementation of Scale layer.
|
||||
#include "../op_inf_engine.hpp"
|
||||
#include "../ie_ngraph.hpp"
|
||||
|
||||
#include <opencv2/imgproc.hpp>
|
||||
#include <opencv2/dnn/shape_utils.hpp>
|
||||
|
||||
namespace cv
|
||||
@@ -324,7 +325,7 @@ public:
|
||||
std::vector<MatShape> &internals) const CV_OVERRIDE
|
||||
{
|
||||
CV_Assert_N(inputs.size() == 1, blobs.size() == 3);
|
||||
CV_Assert_N(blobs[0].total() == 1, blobs[1].total() == total(inputs[0], 1),
|
||||
CV_Assert_N(blobs[0].total() == 1,
|
||||
blobs[2].total() == inputs[0][1]);
|
||||
|
||||
outputs.assign(1, inputs[0]);
|
||||
@@ -347,15 +348,20 @@ public:
|
||||
float* outData = outputs[0].ptr<float>();
|
||||
|
||||
Mat data_mean_cpu = blobs[1].clone();
|
||||
Mat mean_resize = Mat(inputs[0].size[3], inputs[0].size[2], CV_32FC3);
|
||||
Mat mean_3d = Mat(data_mean_cpu.size[3], data_mean_cpu.size[2], CV_32FC3, data_mean_cpu.ptr<float>(0));
|
||||
resize(mean_3d, mean_resize, Size(inputs[0].size[3], inputs[0].size[2]));
|
||||
int new_size[] = {1, mean_resize.channels(), mean_resize.cols, mean_resize.rows};
|
||||
Mat data_mean_cpu_resize = mean_resize.reshape(1, *new_size);
|
||||
Mat data_mean_per_channel_cpu = blobs[2].clone();
|
||||
|
||||
const int numWeights = data_mean_cpu.total();
|
||||
const int numWeights = data_mean_cpu_resize.total();
|
||||
CV_Assert(numWeights != 0);
|
||||
|
||||
++num_iter;
|
||||
if (num_iter <= recompute_mean)
|
||||
{
|
||||
data_mean_cpu *= (num_iter - 1);
|
||||
data_mean_cpu_resize *= (num_iter - 1);
|
||||
const int batch = inputs[0].size[0];
|
||||
float alpha = 1.0 / batch;
|
||||
|
||||
@@ -364,15 +370,15 @@ public:
|
||||
Mat inpSlice(1, numWeights, CV_32F, inpData);
|
||||
inpSlice = alpha * inpSlice;
|
||||
|
||||
add(data_mean_cpu.reshape(1, 1), inpSlice, data_mean_cpu.reshape(1, 1));
|
||||
add(data_mean_cpu_resize.reshape(1, 1), inpSlice, data_mean_cpu_resize.reshape(1, 1));
|
||||
inpData += numWeights;
|
||||
}
|
||||
data_mean_cpu *= (1.0 / num_iter);
|
||||
data_mean_cpu_resize *= (1.0 / num_iter);
|
||||
|
||||
int newsize[] = {blobs[1].size[1], (int)blobs[1].total(2)};
|
||||
reduce(data_mean_cpu.reshape(1, 2, &newsize[0]), data_mean_per_channel_cpu, 1, REDUCE_SUM, CV_32F);
|
||||
int newsize[] = {inputs[0].size[1], (int)inputs[0].total(2)};
|
||||
reduce(data_mean_cpu_resize.reshape(1, 2, &newsize[0]), data_mean_per_channel_cpu, 1, REDUCE_SUM, CV_32F);
|
||||
|
||||
int area = blobs[1].total(2);
|
||||
int area = inputs[0].total(2);
|
||||
data_mean_per_channel_cpu *= (1.0 / area);
|
||||
}
|
||||
|
||||
@@ -387,7 +393,7 @@ public:
|
||||
Mat inpSlice(1, numWeights, CV_32F, inpData);
|
||||
Mat outSlice(1, numWeights, CV_32F, outData);
|
||||
|
||||
add(inpSlice, (-1) * data_mean_cpu, outSlice);
|
||||
add(inpSlice, (-1) * data_mean_cpu_resize, outSlice);
|
||||
inpData += numWeights;
|
||||
outData += numWeights;
|
||||
}
|
||||
|
||||
@@ -160,6 +160,10 @@ public:
|
||||
|
||||
void finalize(InputArrayOfArrays inputs_arr, OutputArrayOfArrays outputs_arr) CV_OVERRIDE
|
||||
{
|
||||
#ifdef HAVE_OPENCL
|
||||
ocl_exec_cache.clear();
|
||||
#endif
|
||||
|
||||
std::vector<Mat> inputs, outputs;
|
||||
inputs_arr.getMatVector(inputs);
|
||||
outputs_arr.getMatVector(outputs);
|
||||
@@ -214,26 +218,33 @@ public:
|
||||
}
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
bool forward_ocl(InputArrayOfArrays inputs_, OutputArrayOfArrays outputs_, OutputArrayOfArrays internals_)
|
||||
struct OpenCLExecInfo
|
||||
{
|
||||
std::vector<UMat> inputs;
|
||||
std::vector<UMat> outputs;
|
||||
std::string kernel_name;
|
||||
std::string build_opts;
|
||||
size_t local_size[2];
|
||||
size_t global_size[2];
|
||||
|
||||
inputs_.getUMatVector(inputs);
|
||||
outputs_.getUMatVector(outputs);
|
||||
OpenCLExecInfo()
|
||||
{
|
||||
local_size[0] = local_size[1] = 0;
|
||||
global_size[0] = global_size[1] = 0;
|
||||
}
|
||||
};
|
||||
std::vector<OpenCLExecInfo> ocl_exec_cache;
|
||||
|
||||
void ocl_prepare(const std::vector<UMat>& inputs, const std::vector<UMat>& outputs)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
|
||||
CV_Assert(outputs.size() == finalSliceRanges.size());
|
||||
ocl_exec_cache.resize(outputs.size());
|
||||
|
||||
const UMat& input = inputs[0];
|
||||
if (input.dims > 5)
|
||||
{
|
||||
CV_LOG_INFO(NULL, "DNN/OpenCL/Slice: implementation doesn't support dims=" << input.dims << ". Fallback to CPU");
|
||||
return false;
|
||||
}
|
||||
const int dims = input.dims;
|
||||
|
||||
size_t WSZ = 128;
|
||||
|
||||
const int dims = input.dims;
|
||||
const int elemSize = (int)input.elemSize();
|
||||
String opts0 = cv::format(
|
||||
"-DDIMS=%d -DELEMSIZE=%d",
|
||||
@@ -243,10 +254,11 @@ public:
|
||||
{
|
||||
opts0 += cv::format(" -DSRC_STEP_%d=%d", d, (int)input.step[dims - 1 - d]);
|
||||
}
|
||||
String kname = cv::format("slice_%d", dims);
|
||||
for (size_t i = 0; i < outputs.size(); i++)
|
||||
{
|
||||
UMat& output = outputs[i];
|
||||
OpenCLExecInfo& ocl = ocl_exec_cache[i];
|
||||
|
||||
const UMat& output = outputs[i];
|
||||
const std::vector<Range>& range = finalSliceRanges[i];
|
||||
|
||||
String opts = opts0;
|
||||
@@ -262,6 +274,8 @@ public:
|
||||
CV_CheckEQ(range[d].size(), (int)output.size[d], "");
|
||||
}
|
||||
|
||||
const size_t param_LIMIT_BLOCK_SIZE_PER_WG = WSZ * 64;
|
||||
|
||||
int block_dims = 0;
|
||||
size_t block_size = elemSize;
|
||||
for (int i = dims - 1; i >= 0; --i)
|
||||
@@ -270,12 +284,14 @@ public:
|
||||
break;
|
||||
block_size *= output.size[i];
|
||||
block_dims++;
|
||||
if (block_size >= param_LIMIT_BLOCK_SIZE_PER_WG)
|
||||
break;
|
||||
}
|
||||
|
||||
const size_t total = output.total() * elemSize;
|
||||
size_t num_blocks = total / block_size;
|
||||
|
||||
if ((num_blocks <= 8 && block_size >= WSZ * 4) || (block_size >= WSZ * 64))
|
||||
if ((num_blocks <= 8 && block_size >= WSZ * 4) || (block_size >= param_LIMIT_BLOCK_SIZE_PER_WG))
|
||||
{
|
||||
// use 1D copy mode
|
||||
opts += cv::format(" -DUSE_COPY_1D=1");
|
||||
@@ -345,23 +361,98 @@ public:
|
||||
|
||||
opts += cv::format(" -DWSZ=%d", (int)WSZ);
|
||||
|
||||
size_t local[] = { WSZ, 1 };
|
||||
size_t global[] = { WSZ, num_blocks };
|
||||
std::ostringstream kernel_suffix;
|
||||
kernel_suffix << dims << 'x' << elemSize << "_bsz" << block_size;
|
||||
kernel_suffix << "__src_";
|
||||
for (int d = 0; d < dims; d++)
|
||||
{
|
||||
kernel_suffix << input.size[dims - 1 - d] << '_';
|
||||
}
|
||||
kernel_suffix << '_';
|
||||
/*for (int d = 0; d < dims; d++)
|
||||
{
|
||||
kernel_suffix << input.step[dims - 1 - d] << '_';
|
||||
}
|
||||
kernel_suffix << '_';*/
|
||||
|
||||
ocl::Kernel kernel(kname.c_str(), ocl::dnn::slice_oclsrc, opts);
|
||||
kernel_suffix << "dst_";
|
||||
for (int d = 0; d < dims; d++)
|
||||
{
|
||||
kernel_suffix << output.size[dims - 1 - d] << '_';
|
||||
}
|
||||
/*kernel_suffix << '_';
|
||||
for (int d = 0; d < dims; d++)
|
||||
{
|
||||
kernel_suffix << output.step[dims - 1 - d] << '_';
|
||||
}*/
|
||||
kernel_suffix << "_slice_";
|
||||
for (int d = 0; d < dims; d++)
|
||||
{
|
||||
kernel_suffix << range[dims - 1 - d].start << '_';
|
||||
}
|
||||
for (int d = 0; d < dims; d++)
|
||||
{
|
||||
kernel_suffix << '_' << range[dims - 1 - d].end;
|
||||
}
|
||||
|
||||
std::string kernel_suffix_str = kernel_suffix.str();
|
||||
opts += cv::format(" -DSLICE_KERNEL_SUFFIX=%s", kernel_suffix_str.c_str());
|
||||
|
||||
ocl.kernel_name = cv::format("slice_%s", kernel_suffix_str.c_str());
|
||||
ocl.build_opts = opts;
|
||||
ocl.local_size[0] = WSZ;
|
||||
ocl.local_size[1] = 1;
|
||||
ocl.global_size[0] = WSZ;
|
||||
ocl.global_size[1] = num_blocks;
|
||||
} // for outputs.size()
|
||||
} // ocl_prepare
|
||||
|
||||
bool forward_ocl(InputArrayOfArrays inputs_, OutputArrayOfArrays outputs_, OutputArrayOfArrays internals_)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
|
||||
std::vector<UMat> inputs;
|
||||
std::vector<UMat> outputs;
|
||||
|
||||
inputs_.getUMatVector(inputs);
|
||||
outputs_.getUMatVector(outputs);
|
||||
|
||||
CV_Assert(outputs.size() == finalSliceRanges.size());
|
||||
|
||||
const UMat& input = inputs[0];
|
||||
const int dims = input.dims;
|
||||
if (dims > 5)
|
||||
{
|
||||
CV_LOG_INFO(NULL, "DNN/OpenCL/Slice: implementation doesn't support dims=" << dims << ". Fallback to CPU");
|
||||
return false;
|
||||
}
|
||||
|
||||
if (ocl_exec_cache.empty())
|
||||
{
|
||||
ocl_prepare(inputs, outputs);
|
||||
}
|
||||
CV_CheckEQ(ocl_exec_cache.size(), outputs.size(), "");
|
||||
|
||||
for (size_t i = 0; i < outputs.size(); i++)
|
||||
{
|
||||
const OpenCLExecInfo& ocl = ocl_exec_cache[i];
|
||||
|
||||
UMat& output = outputs[i];
|
||||
|
||||
ocl::Kernel kernel(ocl.kernel_name.c_str(), ocl::dnn::slice_oclsrc, ocl.build_opts);
|
||||
if (kernel.empty())
|
||||
return false;
|
||||
bool ret = kernel.args(
|
||||
ocl::KernelArg::PtrReadOnly(input),
|
||||
ocl::KernelArg::PtrWriteOnly(output)
|
||||
)
|
||||
.run(2, global, local, false);
|
||||
.run(2, (size_t*)ocl.global_size, (size_t*)ocl.local_size, false);
|
||||
if (!ret)
|
||||
return false;
|
||||
} // for outputs.size()
|
||||
|
||||
return true;
|
||||
}
|
||||
} // forward_ocl
|
||||
#endif
|
||||
|
||||
void forward(InputArrayOfArrays inputs_arr, OutputArrayOfArrays outputs_arr, OutputArrayOfArrays internals_arr) CV_OVERRIDE
|
||||
|
||||
@@ -433,7 +433,7 @@ class OCL4DNNInnerProduct
|
||||
UMat& top_data);
|
||||
private:
|
||||
OCL4DNNInnerProductConfig config_;
|
||||
int32_t axis_;
|
||||
//int32_t axis_;
|
||||
int32_t num_output_;
|
||||
int32_t M_;
|
||||
int32_t N_;
|
||||
|
||||
@@ -46,6 +46,8 @@
|
||||
#include <vector>
|
||||
#include "opencl_kernels_dnn.hpp"
|
||||
|
||||
#include "opencv2/core/utils/logger.hpp"
|
||||
|
||||
namespace cv { namespace dnn { namespace ocl4dnn {
|
||||
|
||||
enum gemm_data_type_t
|
||||
@@ -238,10 +240,6 @@ static bool ocl4dnnFastImageGEMM(const CBLAS_TRANSPOSE TransA,
|
||||
kernel_name += "_float";
|
||||
}
|
||||
|
||||
ocl::Kernel oclk_gemm_float(kernel_name.c_str(), ocl::dnn::gemm_image_oclsrc, opts);
|
||||
if (oclk_gemm_float.empty())
|
||||
return false;
|
||||
|
||||
while (C_start_y < M)
|
||||
{
|
||||
blockC_width = std::min(static_cast<int>(N) - C_start_x, blocksize);
|
||||
@@ -348,6 +346,10 @@ static bool ocl4dnnFastImageGEMM(const CBLAS_TRANSPOSE TransA,
|
||||
}
|
||||
local[1] = 1;
|
||||
|
||||
ocl::Kernel oclk_gemm_float(kernel_name.c_str(), ocl::dnn::gemm_image_oclsrc, opts);
|
||||
if (oclk_gemm_float.empty())
|
||||
return false;
|
||||
|
||||
cl_uint arg_idx = 0;
|
||||
if (is_image_a)
|
||||
oclk_gemm_float.set(arg_idx++, ocl::KernelArg::PtrReadOnly(A));
|
||||
@@ -378,7 +380,10 @@ static bool ocl4dnnFastImageGEMM(const CBLAS_TRANSPOSE TransA,
|
||||
oclk_gemm_float.set(arg_idx++, isFirstColBlock);
|
||||
|
||||
if (!oclk_gemm_float.run(2, global, local, false))
|
||||
{
|
||||
CV_LOG_WARNING(NULL, "OpenCL kernel enqueue failed: " << kernel_name);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (TransA == CblasNoTrans)
|
||||
A_start_x += blockA_width;
|
||||
|
||||
@@ -607,6 +607,7 @@ void OCL4DNNConvSpatial<Dtype>::calculateBenchmark(const UMat &bottom, UMat &ver
|
||||
{
|
||||
options_.str(""); options_.clear(); // clear contents and state flags
|
||||
createBasicKernel(1, 1, 1);
|
||||
CV_Assert(!kernelQueue.empty()); // basic kernel must be available
|
||||
kernel_index_ = kernelQueue.size() - 1;
|
||||
convolve(bottom, verifyTop, weight, bias, numImages, kernelQueue[kernel_index_]);
|
||||
CV_Assert(phash.find(kernelQueue[kernel_index_]->kernelName) != phash.end());
|
||||
@@ -1713,6 +1714,7 @@ void OCL4DNNConvSpatial<float>::useFirstAvailable(const UMat &bottom,
|
||||
tunerItems[i]->blockHeight,
|
||||
tunerItems[i]->blockDepth))
|
||||
{
|
||||
CV_Assert(!kernelQueue.empty()); // basic kernel must be available
|
||||
int kernelIdx = kernelQueue.size() - 1;
|
||||
kernelConfig* config = kernelQueue[kernelIdx].get();
|
||||
bool failed = false;
|
||||
@@ -1883,6 +1885,7 @@ void OCL4DNNConvSpatial<float>::setupConvolution(const UMat &bottom,
|
||||
CV_LOG_INFO(NULL, "fallback to basic kernel");
|
||||
options_.str(""); options_.clear(); // clear contents and state flags
|
||||
createBasicKernel(1, 1, 1);
|
||||
CV_Assert(!kernelQueue.empty()); // basic kernel must be available
|
||||
kernel_index_ = kernelQueue.size() - 1;
|
||||
}
|
||||
this->bestKernelConfig = kernelQueue[kernel_index_];
|
||||
|
||||
@@ -262,6 +262,24 @@ public:
|
||||
}
|
||||
};
|
||||
|
||||
class ExpandSubgraph : public Subgraph
|
||||
{
|
||||
public:
|
||||
ExpandSubgraph()
|
||||
{
|
||||
int input = addNodeToMatch("");
|
||||
int values = addNodeToMatch("");
|
||||
int init = addNodeToMatch("ConstantOfShape", values);
|
||||
int coeff = addNodeToMatch("Constant");
|
||||
int mul = addNodeToMatch("Mul", init, coeff);
|
||||
int shape = addNodeToMatch("Constant");
|
||||
int condition = addNodeToMatch("Equal", shape, mul);
|
||||
int where = addNodeToMatch("Where", condition, init, addNodeToMatch("Constant"));
|
||||
addNodeToMatch("Expand", input, where);
|
||||
setFusedNode("Expand", input, shape);
|
||||
}
|
||||
};
|
||||
|
||||
class MulCastSubgraph : public Subgraph
|
||||
{
|
||||
public:
|
||||
@@ -459,6 +477,7 @@ void simplifySubgraphs(opencv_onnx::GraphProto& net)
|
||||
subgraphs.push_back(makePtr<NormalizeSubgraph3>());
|
||||
subgraphs.push_back(makePtr<BatchNormalizationSubgraph1>());
|
||||
subgraphs.push_back(makePtr<BatchNormalizationSubgraph2>());
|
||||
subgraphs.push_back(makePtr<ExpandSubgraph>());
|
||||
|
||||
simplifySubgraphs(Ptr<ImportGraphWrapper>(new ONNXGraphWrapper(net)), subgraphs);
|
||||
}
|
||||
|
||||
@@ -205,7 +205,7 @@ __kernel void ConvolveBasic(
|
||||
#if APPLY_BIAS
|
||||
ACTIVATION_FUNCTION(convolved_image, offset, sum[kern] + bias[biasIndex + kern], biasIndex + kern);
|
||||
#else
|
||||
ACTIVATION_FUNCTION(convolved_image, offset, sum[kern], biasIndex + kern);
|
||||
ACTIVATION_FUNCTION(convolved_image, offset, sum[kern], kernelNum + kern);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -83,7 +83,7 @@ __kernel void TEMPLATE(lrn_full_no_scale,Dtype)(const int nthreads, __global con
|
||||
* in_off[(head - size) * step];
|
||||
}
|
||||
scale_val = k + accum_scale * alpha_over_size;
|
||||
out_off[(head - post_pad) * step] = in_off[(head - post_pad) * step] * (Dtype)native_powr((Dtype)scale_val, (Dtype)negative_beta);
|
||||
out_off[(head - post_pad) * step] = in_off[(head - post_pad) * step] * (Dtype)native_powr(scale_val, negative_beta);
|
||||
++head;
|
||||
}
|
||||
// subtract only
|
||||
@@ -93,7 +93,7 @@ __kernel void TEMPLATE(lrn_full_no_scale,Dtype)(const int nthreads, __global con
|
||||
* in_off[(head - size) * step];
|
||||
}
|
||||
scale_val = k + accum_scale * alpha_over_size;
|
||||
out_off[(head - post_pad) * step] = in_off[(head - post_pad) * step] * (Dtype)native_powr((Dtype)scale_val, (Dtype)negative_beta);
|
||||
out_off[(head - post_pad) * step] = in_off[(head - post_pad) * step] * (Dtype)native_powr(scale_val, negative_beta);
|
||||
++head;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -114,6 +114,6 @@ __kernel void clip(const int nthreads,
|
||||
for (int index = get_global_id(0); index < nthreads; index += get_global_size(0))
|
||||
{
|
||||
Dtype4 vec = vload4(index, dst);
|
||||
vstore4(clamp(vec, 0.0f, 1.0f), index, dst);
|
||||
vstore4(clamp(vec, (Dtype)0.0f, (Dtype)1.0f), index, dst);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,19 +48,85 @@ global: <WSZ, number_of_copy_blocks, 1>
|
||||
#define BLOCK_COLS_X4 (BLOCK_COLS / 4)
|
||||
#define BLOCK_COLS_X16 (BLOCK_COLS / 16)
|
||||
|
||||
#ifdef USE_COPY_1D
|
||||
|
||||
static inline
|
||||
__attribute__((always_inline))
|
||||
void copy_block_1d(
|
||||
__attribute__((reqd_work_group_size(WSZ, 1, 1)))
|
||||
__kernel void
|
||||
CONCAT(slice_, SLICE_KERNEL_SUFFIX)(
|
||||
__global const uchar* src0,
|
||||
const uint src_offset,
|
||||
__global uchar* dst0,
|
||||
const uint dst_offset
|
||||
__global uchar* dst0
|
||||
)
|
||||
{
|
||||
__global const uchar* src = src0 + src_offset;
|
||||
__global uchar* dst = dst0 + dst_offset;
|
||||
uint block_id = get_global_id(1);
|
||||
uint dst_offset0 = block_id * BLOCK_SIZE;
|
||||
uint src_offset0 = 0;
|
||||
|
||||
{ // calculate src_offset0
|
||||
|
||||
#define CALC_SRC_INDEX(dim) \
|
||||
{ \
|
||||
uint plane_sz = CONCAT(DST_STEP_, dim) / BLOCK_SIZE; \
|
||||
CONCAT(idx_, dim) = block_id / plane_sz; \
|
||||
block_id = block_id - CONCAT(idx_, dim) * plane_sz; \
|
||||
}
|
||||
#define UPDATE_SRC_OFFSET(dim) \
|
||||
src_offset0 = mad24((uint)(CONCAT(idx_, dim) + CONCAT(SRC_START_, dim)), (uint)CONCAT(SRC_STEP_, dim), (uint)src_offset0);
|
||||
/*
|
||||
if (get_global_id(0) == 0 && get_global_id(1) == 0) \
|
||||
printf("(%d, %d): @%d src_offset0=%d idx_dim=%d block_id=%d\n", \
|
||||
get_global_id(0), get_global_id(1), \
|
||||
dim, src_offset0, CONCAT(idx_, dim), block_id \
|
||||
);
|
||||
*/
|
||||
|
||||
#if DIMS > 5
|
||||
#error "invalid configuration"
|
||||
#endif
|
||||
#if DIMS > 4
|
||||
uint idx_4 = 0;
|
||||
#if BLOCK_DIMS <= 4
|
||||
CALC_SRC_INDEX(4)
|
||||
#endif
|
||||
UPDATE_SRC_OFFSET(4)
|
||||
#endif
|
||||
#if DIMS > 3
|
||||
uint idx_3 = 0;
|
||||
#if BLOCK_DIMS <= 3
|
||||
CALC_SRC_INDEX(3)
|
||||
#endif
|
||||
UPDATE_SRC_OFFSET(3)
|
||||
#endif
|
||||
#if DIMS > 2
|
||||
uint idx_2 = 0;
|
||||
#if BLOCK_DIMS <= 2
|
||||
CALC_SRC_INDEX(2)
|
||||
#endif
|
||||
UPDATE_SRC_OFFSET(2)
|
||||
#endif
|
||||
#if DIMS > 1
|
||||
uint idx_1 = 0;
|
||||
#if BLOCK_DIMS <= 1
|
||||
CALC_SRC_INDEX(1)
|
||||
#endif
|
||||
UPDATE_SRC_OFFSET(1)
|
||||
#endif
|
||||
#if DIMS > 0
|
||||
uint idx_0 = 0;
|
||||
UPDATE_SRC_OFFSET(0)
|
||||
#endif
|
||||
|
||||
/*
|
||||
if (get_global_id(0) == 0)
|
||||
printf("(%d, %d): src_offset0=%d dst_offset0=%d\n",
|
||||
get_global_id(0), get_global_id(1),
|
||||
src_offset0, dst_offset0
|
||||
);
|
||||
*/
|
||||
|
||||
} // calculate src_offset0
|
||||
|
||||
#ifdef USE_COPY_1D
|
||||
{ // copy_block_1d
|
||||
__global const uchar* src = src0 + src_offset0;
|
||||
__global uchar* dst = dst0 + dst_offset0;
|
||||
|
||||
uint processed = 0;
|
||||
|
||||
@@ -70,8 +136,9 @@ void copy_block_1d(
|
||||
uint i = get_local_id(0) * 16; // uchar16
|
||||
while (i < BLOCK_COLS_X16 * 16)
|
||||
{
|
||||
uint4 idx = (uint4)(i, i + 16 * WSZ, i + 32 * WSZ, i + 48 * WSZ);
|
||||
idx = select((uint4)i, idx, idx < (BLOCK_COLS_X16 * 16));
|
||||
uint4 idx0 = (uint4)i;
|
||||
uint4 idx = idx0 + (uint4)(0, 16 * WSZ, 32 * WSZ, 48 * WSZ);
|
||||
idx = select(idx0, idx, idx < (BLOCK_COLS_X16 * 16));
|
||||
|
||||
uchar16 a0 = vload16(0, src + idx.s0);
|
||||
uchar16 a1 = vload16(0, src + idx.s1);
|
||||
@@ -97,8 +164,9 @@ void copy_block_1d(
|
||||
uint i = get_local_id(0) * 4 + processed; // uchar4
|
||||
while (i < BLOCK_COLS_X4 * 4)
|
||||
{
|
||||
uint4 idx = (uint4)(i, i + 4 * WSZ, i + 8 * WSZ, i + 12 * WSZ);
|
||||
idx = select((uint4)i, idx, idx < (BLOCK_COLS_X4 * 4));
|
||||
uint4 idx0 = (uint4)i;
|
||||
uint4 idx = idx0 + (uint4)(0, 4 * WSZ, 8 * WSZ, 12 * WSZ);
|
||||
idx = select(idx0, idx, idx < (BLOCK_COLS_X4 * 4));
|
||||
|
||||
uchar4 a0 = vload4(0, src + idx.s0);
|
||||
uchar4 a1 = vload4(0, src + idx.s1);
|
||||
@@ -130,19 +198,11 @@ void copy_block_1d(
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
} // copy_block_1d
|
||||
|
||||
#else // USE_COPY_1D
|
||||
#else
|
||||
|
||||
static inline
|
||||
__attribute__((always_inline))
|
||||
void copy_block_2d(
|
||||
__global const uchar* src0,
|
||||
const uint src_offset0,
|
||||
__global uchar* dst0,
|
||||
const uint dst_offset0
|
||||
)
|
||||
{
|
||||
{ // copy_block_2d
|
||||
__global const uchar* src = src0 + src_offset0;
|
||||
__global uchar* dst = dst0 + dst_offset0;
|
||||
|
||||
@@ -199,85 +259,6 @@ void copy_block_2d(
|
||||
#endif // BLOCK_COLS_FILL_X4 != BLOCK_COLS
|
||||
i += WSZ * 4;
|
||||
}
|
||||
}
|
||||
|
||||
#endif // USE_COPY_1D
|
||||
|
||||
__kernel void
|
||||
CONCAT(slice_, DIMS)(
|
||||
__global const uchar* src,
|
||||
__global uchar* dst
|
||||
)
|
||||
{
|
||||
uint block_id = get_global_id(1);
|
||||
|
||||
uint dst_offset = block_id * BLOCK_SIZE;
|
||||
|
||||
uint src_offset = 0;
|
||||
|
||||
#define CALC_SRC_INDEX(dim) \
|
||||
{ \
|
||||
uint plane_sz = CONCAT(DST_STEP_, dim) / BLOCK_SIZE; \
|
||||
CONCAT(idx_, dim) = block_id / plane_sz; \
|
||||
block_id = block_id - CONCAT(idx_, dim) * plane_sz; \
|
||||
}
|
||||
#define UPDATE_SRC_OFFSET(dim) \
|
||||
src_offset = mad24((uint)(CONCAT(idx_, dim) + CONCAT(SRC_START_, dim)), (uint)CONCAT(SRC_STEP_, dim), (uint)src_offset);
|
||||
/*
|
||||
if (get_global_id(0) == 0 && get_global_id(1) == 0) \
|
||||
printf("(%d, %d): @%d src_offset=%d idx_dim=%d block_id=%d\n", \
|
||||
get_global_id(0), get_global_id(1), \
|
||||
dim, src_offset, CONCAT(idx_, dim), block_id \
|
||||
);
|
||||
*/
|
||||
|
||||
#if DIMS > 5
|
||||
#error "invalid configuration"
|
||||
#endif
|
||||
#if DIMS > 4
|
||||
uint idx_4 = 0;
|
||||
#if BLOCK_DIMS <= 4
|
||||
CALC_SRC_INDEX(4)
|
||||
#endif
|
||||
UPDATE_SRC_OFFSET(4)
|
||||
#endif
|
||||
#if DIMS > 3
|
||||
uint idx_3 = 0;
|
||||
#if BLOCK_DIMS <= 3
|
||||
CALC_SRC_INDEX(3)
|
||||
#endif
|
||||
UPDATE_SRC_OFFSET(3)
|
||||
#endif
|
||||
#if DIMS > 2
|
||||
uint idx_2 = 0;
|
||||
#if BLOCK_DIMS <= 2
|
||||
CALC_SRC_INDEX(2)
|
||||
#endif
|
||||
UPDATE_SRC_OFFSET(2)
|
||||
#endif
|
||||
#if DIMS > 1
|
||||
uint idx_1 = 0;
|
||||
#if BLOCK_DIMS <= 1
|
||||
CALC_SRC_INDEX(1)
|
||||
#endif
|
||||
UPDATE_SRC_OFFSET(1)
|
||||
#endif
|
||||
#if DIMS > 0
|
||||
uint idx_0 = 0;
|
||||
UPDATE_SRC_OFFSET(0)
|
||||
#endif
|
||||
|
||||
/*
|
||||
if (get_global_id(0) == 0)
|
||||
printf("(%d, %d): src_offset=%d dst_offset=%d\n",
|
||||
get_global_id(0), get_global_id(1),
|
||||
src_offset, dst_offset
|
||||
);
|
||||
*/
|
||||
|
||||
#ifdef USE_COPY_1D
|
||||
copy_block_1d(src, src_offset, dst, dst_offset);
|
||||
#else
|
||||
copy_block_2d(src, src_offset, dst, dst_offset);
|
||||
} // copy_block_2d
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -2067,7 +2067,7 @@ void TFImporter::populateNet(Net dstNet)
|
||||
connect(layer_id, dstNet, parsePin(layer.input(0)), id, 0);
|
||||
connect(layer_id, dstNet, parsePin(layer.input(1)), id, 1);
|
||||
}
|
||||
else if (type == "Mean")
|
||||
else if (type == "Mean" || type == "Sum")
|
||||
{
|
||||
// Computes the mean of elements across dimensions of a tensor.
|
||||
// If keepdims is false (default) reduces input_tensor along the dimensions given in axis,
|
||||
@@ -2116,7 +2116,7 @@ void TFImporter::populateNet(Net dstNet)
|
||||
LayerParams avgLp;
|
||||
std::string avgName = name + "/avg";
|
||||
CV_Assert(layer_id.find(avgName) == layer_id.end());
|
||||
avgLp.set("pool", "ave");
|
||||
avgLp.set("pool", type == "Mean" ? "ave" : "sum");
|
||||
// pooling kernel H x 1
|
||||
avgLp.set("global_pooling_h", true);
|
||||
avgLp.set("kernel_w", 1);
|
||||
@@ -2153,11 +2153,44 @@ void TFImporter::populateNet(Net dstNet)
|
||||
layer_id[name] = id;
|
||||
connect(layer_id, dstNet, Pin(avgName), id, 0);
|
||||
connect(layer_id, dstNet, Pin(layerShapeName), id, 1);
|
||||
} else if (indices.total() == 1) {
|
||||
int axis = toNCHW(indices.at<int>(0));
|
||||
if (axis == 2 || axis == 3)
|
||||
{
|
||||
layerParams.set("pool", type == "Mean" ? "ave" : "sum");
|
||||
layerParams.set(axis == 2 ? "kernel_w" : "kernel_h", 1);
|
||||
layerParams.set(axis == 2 ? "global_pooling_h" : "global_pooling_w", true);
|
||||
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
||||
layer_id[name] = id;
|
||||
connect(layer_id, dstNet, parsePin(layer.input(0)), id, 0);
|
||||
|
||||
if (!keepDims)
|
||||
{
|
||||
// To keep correct order after squeeze dims we first need to change layout from NCHW to NHWC
|
||||
LayerParams permLP;
|
||||
int order[] = {0, 2, 3, 1}; // From OpenCV's NCHW to NHWC.
|
||||
permLP.set("order", DictValue::arrayInt<int*>(order, 4));
|
||||
std::string permName = name + "/nchw";
|
||||
CV_Assert(layer_id.find(permName) == layer_id.end());
|
||||
int permId = dstNet.addLayer(permName, "Permute", permLP);
|
||||
layer_id[permName] = permId;
|
||||
connect(layer_id, dstNet, Pin(name), permId, 0);
|
||||
|
||||
LayerParams squeezeLp;
|
||||
std::string squeezeName = name + "/squeeze";
|
||||
CV_Assert(layer_id.find(squeezeName) == layer_id.end());
|
||||
squeezeLp.set("axis", indices.at<int>(0));
|
||||
squeezeLp.set("end_axis", indices.at<int>(0) + 1);
|
||||
int squeezeId = dstNet.addLayer(squeezeName, "Flatten", squeezeLp);
|
||||
layer_id[squeezeName] = squeezeId;
|
||||
connect(layer_id, dstNet, Pin(permName), squeezeId, 0);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (indices.total() != 2 || indices.at<int>(0) != 1 || indices.at<int>(1) != 2)
|
||||
CV_Error(Error::StsNotImplemented, "Unsupported mode of reduce_mean operation.");
|
||||
CV_Error(Error::StsNotImplemented, "Unsupported mode of reduce_mean or reduce_sum operation.");
|
||||
|
||||
layerParams.set("pool", "ave");
|
||||
layerParams.set("pool", type == "Mean" ? "ave" : "sum");
|
||||
layerParams.set("global_pooling", true);
|
||||
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
||||
layer_id[name] = id;
|
||||
|
||||
@@ -63,10 +63,10 @@ void normAssert(
|
||||
double l1 /*= 0.00001*/, double lInf /*= 0.0001*/)
|
||||
{
|
||||
double normL1 = cvtest::norm(ref, test, cv::NORM_L1) / ref.getMat().total();
|
||||
EXPECT_LE(normL1, l1) << comment;
|
||||
EXPECT_LE(normL1, l1) << comment << " |ref| = " << cvtest::norm(ref, cv::NORM_INF);
|
||||
|
||||
double normInf = cvtest::norm(ref, test, cv::NORM_INF);
|
||||
EXPECT_LE(normInf, lInf) << comment;
|
||||
EXPECT_LE(normInf, lInf) << comment << " |ref| = " << cvtest::norm(ref, cv::NORM_INF);
|
||||
}
|
||||
|
||||
std::vector<cv::Rect2d> matToBoxes(const cv::Mat& m)
|
||||
|
||||
@@ -625,6 +625,11 @@ TEST_P(Test_Darknet_nets, YOLOv4_tiny)
|
||||
target == DNN_TARGET_CPU ? CV_TEST_TAG_MEMORY_512MB : CV_TEST_TAG_MEMORY_1GB
|
||||
);
|
||||
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2021010000) // nGraph compilation failure
|
||||
if (target == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||
#endif
|
||||
|
||||
const double confThreshold = 0.5;
|
||||
// batchId, classId, confidence, left, top, right, bottom
|
||||
const int N0 = 2;
|
||||
@@ -753,6 +758,13 @@ TEST_P(Test_Darknet_layers, connected)
|
||||
testDarknetLayer("connected", true);
|
||||
}
|
||||
|
||||
TEST_P(Test_Darknet_layers, relu)
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && target == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD);
|
||||
testDarknetLayer("relu");
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/**/, Test_Darknet_layers, dnnBackendsAndTargets());
|
||||
|
||||
}} // namespace
|
||||
|
||||
@@ -646,6 +646,8 @@ TEST_P(Test_Caffe_layers, DataAugmentation)
|
||||
if (backend == DNN_BACKEND_OPENCV && target == DNN_TARGET_OPENCL_FP16)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_OPENCL_FP16);
|
||||
testLayerUsingCaffeModels("data_augmentation", true, false);
|
||||
testLayerUsingCaffeModels("data_augmentation_2x1", true, false);
|
||||
testLayerUsingCaffeModels("data_augmentation_8x6", true, false);
|
||||
}
|
||||
|
||||
TEST_P(Test_Caffe_layers, Resample)
|
||||
@@ -1108,6 +1110,9 @@ TEST_P(Layer_Test_Convolution_DLDT, Accuracy)
|
||||
const Backend backendId = get<0>(GetParam());
|
||||
const Target targetId = get<1>(GetParam());
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && targetId == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
|
||||
if (backendId != DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && backendId != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
throw SkipTestException("No support for async forward");
|
||||
|
||||
@@ -1118,9 +1123,8 @@ TEST_P(Layer_Test_Convolution_DLDT, Accuracy)
|
||||
else
|
||||
FAIL() << "Unknown backendId";
|
||||
|
||||
std::string suffix = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD) ? "_fp16" : "";
|
||||
Net netDefault = readNet(_tf("layer_convolution.caffemodel"), _tf("layer_convolution.prototxt"));
|
||||
Net net = readNet(_tf("layer_convolution" + suffix + ".xml"), _tf("layer_convolution" + suffix + ".bin"));
|
||||
Net net = readNet(_tf("layer_convolution.xml"), _tf("layer_convolution.bin"));
|
||||
|
||||
Mat inp = blobFromNPY(_tf("blob.npy"));
|
||||
|
||||
@@ -1140,7 +1144,10 @@ TEST_P(Layer_Test_Convolution_DLDT, Accuracy)
|
||||
|
||||
std::vector<int> outLayers = net.getUnconnectedOutLayers();
|
||||
ASSERT_EQ(net.getLayer(outLayers[0])->name, "output");
|
||||
ASSERT_EQ(net.getLayer(outLayers[0])->type, "Convolution");
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
ASSERT_EQ(net.getLayer(outLayers[0])->type, "Convolution");
|
||||
else
|
||||
ASSERT_EQ(net.getLayer(outLayers[0])->type, "Add");
|
||||
}
|
||||
|
||||
TEST_P(Layer_Test_Convolution_DLDT, setInput_uint8)
|
||||
@@ -1148,6 +1155,9 @@ TEST_P(Layer_Test_Convolution_DLDT, setInput_uint8)
|
||||
const Backend backendId = get<0>(GetParam());
|
||||
const Target targetId = get<1>(GetParam());
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && targetId == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
|
||||
if (backendId != DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && backendId != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
throw SkipTestException("No support for async forward");
|
||||
|
||||
@@ -1164,12 +1174,10 @@ TEST_P(Layer_Test_Convolution_DLDT, setInput_uint8)
|
||||
randu(inputs[0], 0, 255);
|
||||
inputs[0].convertTo(inputs[1], CV_32F);
|
||||
|
||||
std::string suffix = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD) ? "_fp16" : "";
|
||||
|
||||
Mat outs[2];
|
||||
for (int i = 0; i < 2; ++i)
|
||||
{
|
||||
Net net = readNet(_tf("layer_convolution" + suffix + ".xml"), _tf("layer_convolution" + suffix + ".bin"));
|
||||
Net net = readNet(_tf("layer_convolution.xml"), _tf("layer_convolution.bin"));
|
||||
net.setPreferableBackend(backendId);
|
||||
net.setPreferableTarget(targetId);
|
||||
net.setInput(inputs[i]);
|
||||
@@ -1185,6 +1193,9 @@ TEST_P(Layer_Test_Convolution_DLDT, multithreading)
|
||||
const Backend backendId = get<0>(GetParam());
|
||||
const Target targetId = get<1>(GetParam());
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && targetId == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
|
||||
if (backendId != DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && backendId != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
throw SkipTestException("No support for async forward");
|
||||
|
||||
@@ -1195,9 +1206,8 @@ TEST_P(Layer_Test_Convolution_DLDT, multithreading)
|
||||
else
|
||||
FAIL() << "Unknown backendId";
|
||||
|
||||
std::string suffix = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD) ? "_fp16" : "";
|
||||
std::string xmlPath = _tf("layer_convolution" + suffix + ".xml");
|
||||
std::string binPath = _tf("layer_convolution" + suffix + ".bin");
|
||||
std::string xmlPath = _tf("layer_convolution.xml");
|
||||
std::string binPath = _tf("layer_convolution.bin");
|
||||
Net firstNet = readNet(xmlPath, binPath);
|
||||
Net secondNet = readNet(xmlPath, binPath);
|
||||
Mat inp = blobFromNPY(_tf("blob.npy"));
|
||||
@@ -1256,8 +1266,7 @@ TEST_P(Test_DLDT_two_inputs_3dim, as_IR)
|
||||
int secondInpType = get<1>(GetParam());
|
||||
Target targetId = get<2>(GetParam());
|
||||
|
||||
std::string suffix = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD) ? "_fp16" : "";
|
||||
Net net = readNet(_tf("net_two_inputs" + suffix + ".xml"), _tf("net_two_inputs.bin"));
|
||||
Net net = readNet(_tf("net_two_inputs.xml"), _tf("net_two_inputs.bin"));
|
||||
std::vector<int> inpSize = get<3>(GetParam());
|
||||
Mat firstInp(3, inpSize.data(), firstInpType);
|
||||
Mat secondInp(3, inpSize.data(), secondInpType);
|
||||
@@ -2046,4 +2055,401 @@ TEST_P(Layer_Test_BatchNorm, fusion)
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/**/, Layer_Test_BatchNorm, dnnBackendsAndTargets());
|
||||
|
||||
class TestLayerFusion : public DNNTestLayer {
|
||||
public:
|
||||
static void makeDefaultTestConvolutionLayer(LayerParams& convParams, int in_channels, int num_filters, bool bias_term)
|
||||
{
|
||||
const int kernel_h = 3, kernel_w = 3;
|
||||
const int pad_h = kernel_h / 2, pad_w = kernel_w / 2;
|
||||
|
||||
convParams.set("kernel_h", kernel_h);
|
||||
convParams.set("kernel_w", kernel_w);
|
||||
convParams.set("pad_h", pad_h);
|
||||
convParams.set("pad_w", pad_w);
|
||||
convParams.set("num_output", num_filters);
|
||||
convParams.set("bias_term", bias_term);
|
||||
convParams.type = "Convolution";
|
||||
convParams.name = "convolution";
|
||||
|
||||
float conv_init_magnitude = 1.0f / in_channels / kernel_h / kernel_w;
|
||||
int weightsShape[] = {num_filters, in_channels, kernel_h, kernel_w};
|
||||
Mat weights(4, &weightsShape[0], CV_32F);
|
||||
randu(weights, -conv_init_magnitude, conv_init_magnitude);
|
||||
convParams.blobs.push_back(weights);
|
||||
if (bias_term)
|
||||
{
|
||||
Mat bias(1, num_filters, CV_32F);
|
||||
randu(bias, -1.0f, 1.0f);
|
||||
convParams.blobs.push_back(bias);
|
||||
}
|
||||
}
|
||||
|
||||
static void makeDefaultTestActivationLayer(LayerParams& activationParams, const std::string& type, int in_channels)
|
||||
{
|
||||
activationParams.type = type;
|
||||
activationParams.name = "activation";
|
||||
if (activationParams.type == "ReLU")
|
||||
activationParams.set("negative_slope", 0.1f);
|
||||
else if (activationParams.type == "Power")
|
||||
{
|
||||
activationParams.set("power", 2.0f);
|
||||
activationParams.set("scale", 0.5f);
|
||||
activationParams.set("shift", 0.3f);
|
||||
}
|
||||
else if (activationParams.type == "ReLU6")
|
||||
{
|
||||
activationParams.set("min_value", -1.0f);
|
||||
activationParams.set("max_value", 1.0f);
|
||||
}
|
||||
else if (activationParams.type == "ChannelsPReLU")
|
||||
{
|
||||
Mat scales(1, in_channels, CV_32F);
|
||||
randu(scales, -1.0f, 1.0f);
|
||||
activationParams.blobs.push_back(scales);
|
||||
}
|
||||
}
|
||||
|
||||
static void makeDefaultTestEltwiseLayer(LayerParams& eltwiseParams, const std::string& op, bool withCoefficients)
|
||||
{
|
||||
eltwiseParams.type = "Eltwise";
|
||||
eltwiseParams.name = "eltwise";
|
||||
eltwiseParams.set("operation", op);
|
||||
if (withCoefficients)
|
||||
{
|
||||
float coeff[] = {0.3f, 0.5f};
|
||||
eltwiseParams.set("coeff", DictValue::arrayReal<float*>(coeff, 2));
|
||||
}
|
||||
}
|
||||
|
||||
static void test(Mat& input, Net& net, Backend backendId, Target targetId, std::vector<int> expectedFusedLayers = std::vector<int>(), double l1 = 0.0, double lInf = 0.0)
|
||||
{
|
||||
DNNTestLayer::checkBackend(backendId, targetId);
|
||||
|
||||
net.enableFusion(false);
|
||||
net.setPreferableBackend(DNN_BACKEND_OPENCV);
|
||||
net.setPreferableTarget(DNN_TARGET_CPU);
|
||||
net.setInput(input);
|
||||
Mat outputReference = net.forward().clone();
|
||||
std::vector<double> refTimings;
|
||||
net.getPerfProfile(refTimings);
|
||||
for (int i = 0; i < refTimings.size(); i++)
|
||||
{
|
||||
CV_Assert(refTimings[i] != 0.0);
|
||||
}
|
||||
|
||||
net.enableFusion(true);
|
||||
net.setPreferableBackend(backendId);
|
||||
net.setPreferableTarget(targetId);
|
||||
net.setInput(input);
|
||||
Mat outputTest = net.forward().clone();
|
||||
std::vector<double> testTimings;
|
||||
net.getPerfProfile(testTimings);
|
||||
for (int i = 0; i < testTimings.size(); i++)
|
||||
{
|
||||
if(std::find(expectedFusedLayers.begin(), expectedFusedLayers.end(), i + 1) != expectedFusedLayers.end())
|
||||
{
|
||||
EXPECT_EQ(testTimings[i], 0.0);
|
||||
}
|
||||
else
|
||||
{
|
||||
EXPECT_NE(testTimings[i], 0.0);
|
||||
}
|
||||
}
|
||||
|
||||
// double ref_max_value, ref_min_value;
|
||||
// minMaxLoc(outputReference.reshape(1, 1), &ref_min_value, &ref_max_value);
|
||||
// std::cout << "reference range: " << ref_min_value << ' ' << ref_max_value << std::endl;
|
||||
|
||||
double default_l1, default_lInf;
|
||||
DNNTestLayer::getDefaultThresholds(backendId, targetId, &default_l1, &default_lInf);
|
||||
if (l1 == 0.0)
|
||||
l1 = default_l1;
|
||||
if (lInf == 0.0)
|
||||
lInf = default_lInf;
|
||||
normAssert(outputReference, outputTest, "", l1, lInf);
|
||||
}
|
||||
|
||||
static testing::internal::ParamGenerator<std::string> eltwiseOpList()
|
||||
{
|
||||
// TODO: automate list generation
|
||||
return Values("sum", "max", "prod", "div");
|
||||
}
|
||||
|
||||
static testing::internal::ParamGenerator<std::string> activationLayersList()
|
||||
{
|
||||
// TODO: automate list generation
|
||||
return Values("ReLU", "ReLU6", "ChannelsPReLU", "TanH", "Swish", "Mish", "Sigmoid", "ELU", "AbsVal", "BNLL", "Power");
|
||||
}
|
||||
|
||||
static testing::internal::ParamGenerator<tuple<Backend, Target> > dnnBackendsAndTargetsForFusionTests()
|
||||
{
|
||||
return dnnBackendsAndTargets(false, false, true, false); // OCV OpenCL + OCV CPU
|
||||
}
|
||||
};
|
||||
|
||||
typedef TestWithParam<tuple<bool, std::string, tuple<Backend, Target> > > ConvolutionActivationFusion;
|
||||
TEST_P(ConvolutionActivationFusion, Accuracy)
|
||||
{
|
||||
// input
|
||||
// |
|
||||
// -----------------------
|
||||
// | convolution |
|
||||
// -----------------------
|
||||
// |
|
||||
// -----------------------
|
||||
// | activation |
|
||||
// -----------------------
|
||||
// |
|
||||
// output
|
||||
|
||||
const int batch_size = 2, in_channels = 16;
|
||||
const int in_height = 16, in_width = 16;
|
||||
int inputShape[] = {batch_size, in_channels, in_height, in_width};
|
||||
Mat input(4, &inputShape[0], CV_32F);
|
||||
randu(input, 1.0f, 2.0f);
|
||||
|
||||
bool bias_term = get<0>(GetParam());
|
||||
LayerParams convParams;
|
||||
TestLayerFusion::makeDefaultTestConvolutionLayer(convParams, in_channels, in_channels, bias_term);
|
||||
|
||||
std::string actType = get<1>(GetParam());
|
||||
LayerParams activationParams;
|
||||
TestLayerFusion::makeDefaultTestActivationLayer(activationParams, actType, in_channels);
|
||||
|
||||
Backend backendId = get<0>(get<2>(GetParam()));
|
||||
Target targetId = get<1>(get<2>(GetParam()));
|
||||
|
||||
Net net;
|
||||
int convId = net.addLayer(convParams.name, convParams.type, convParams);
|
||||
int activId = net.addLayerToPrev(activationParams.name, activationParams.type, activationParams);
|
||||
net.connect(0, 0, convId, 0);
|
||||
|
||||
std::vector<int> expectedFusedLayers;
|
||||
if (backendId == DNN_BACKEND_OPENCV)
|
||||
{
|
||||
if (targetId == DNN_TARGET_CPU)
|
||||
expectedFusedLayers.push_back(activId); // all activations are fused
|
||||
else if (targetId == DNN_TARGET_OPENCL || targetId == DNN_TARGET_OPENCL_FP16)
|
||||
{
|
||||
if (actType == "ReLU" || actType == "ChannelsPReLU" || actType == "ReLU6" || actType == "TanH" /*|| actType == "Power"*/)
|
||||
expectedFusedLayers.push_back(activId);
|
||||
}
|
||||
}
|
||||
|
||||
TestLayerFusion::test(input, net, backendId, targetId, expectedFusedLayers);
|
||||
}
|
||||
INSTANTIATE_TEST_CASE_P(TestLayerFusion, ConvolutionActivationFusion, Combine(
|
||||
/* bias */ testing::Bool(),
|
||||
/* activation */ TestLayerFusion::activationLayersList(),
|
||||
TestLayerFusion::dnnBackendsAndTargetsForFusionTests()
|
||||
));
|
||||
|
||||
typedef TestWithParam<tuple<bool, std::string, bool, tuple<Backend, Target> > > ConvolutionEltwiseFusion;
|
||||
TEST_P(ConvolutionEltwiseFusion, Accuracy)
|
||||
{
|
||||
// input
|
||||
// |
|
||||
// -------------------------------
|
||||
// | |
|
||||
// | ---------------
|
||||
// | | convolution |
|
||||
// | ---------------
|
||||
// | |
|
||||
// | ---------------- |
|
||||
// --------| eltwise op |-------
|
||||
// ----------------
|
||||
// |
|
||||
// output
|
||||
|
||||
const int batch_size = 2, in_channels = 16;
|
||||
const int in_height = 16, in_width = 16;
|
||||
int inputShape[] = {batch_size, in_channels, in_height, in_width};
|
||||
Mat input(4, &inputShape[0], CV_32F);
|
||||
randu(input, 1.0f, 2.0f); // avoid small values to test eltwise div
|
||||
|
||||
bool bias_term = get<0>(GetParam());
|
||||
LayerParams convParams;
|
||||
TestLayerFusion::makeDefaultTestConvolutionLayer(convParams, in_channels, in_channels, bias_term);
|
||||
|
||||
std::string eltwiseOp = get<1>(GetParam());
|
||||
bool weightedEltwise = get<2>(GetParam());
|
||||
if (eltwiseOp != "sum" && weightedEltwise)
|
||||
throw SkipTestException("weighted eltwise not supported");
|
||||
LayerParams eltwiseParams;
|
||||
TestLayerFusion::makeDefaultTestEltwiseLayer(eltwiseParams, eltwiseOp, weightedEltwise);
|
||||
|
||||
Net net;
|
||||
int convId = net.addLayer(convParams.name, convParams.type, convParams);
|
||||
int eltwiseId = net.addLayer(eltwiseParams.name, eltwiseParams.type, eltwiseParams);
|
||||
net.connect(0, 0, convId, 0);
|
||||
net.connect(convId, 0, eltwiseId, 0);
|
||||
net.connect(0, 0, eltwiseId, 1);
|
||||
|
||||
Backend backendId = get<0>(get<3>(GetParam()));
|
||||
Target targetId = get<1>(get<3>(GetParam()));
|
||||
TestLayerFusion::test(input, net, backendId, targetId);
|
||||
}
|
||||
INSTANTIATE_TEST_CASE_P(TestLayerFusion, ConvolutionEltwiseFusion, Combine(
|
||||
/* bias */ testing::Bool(),
|
||||
/* eltwise op */ TestLayerFusion::eltwiseOpList(),
|
||||
/* eltwise weighted */ testing::Bool(),
|
||||
TestLayerFusion::dnnBackendsAndTargetsForFusionTests()
|
||||
));
|
||||
|
||||
typedef TestWithParam<tuple<bool, std::string, bool, std::string, tuple<Backend, Target> > > ConvolutionEltwiseActivationFusion;
|
||||
TEST_P(ConvolutionEltwiseActivationFusion, Accuracy)
|
||||
{
|
||||
// input
|
||||
// |
|
||||
// -------------------------------
|
||||
// | |
|
||||
// | ---------------
|
||||
// | | convolution |
|
||||
// | ---------------
|
||||
// | |
|
||||
// | ---------------- |
|
||||
// --------| eltwise op |-------
|
||||
// ----------------
|
||||
// |
|
||||
// ----------------
|
||||
// | activation |
|
||||
// ----------------
|
||||
// |
|
||||
// output
|
||||
|
||||
const int batch_size = 2, in_channels = 16;
|
||||
const int in_height = 16, in_width = 16;
|
||||
int inputShape[] = {batch_size, in_channels, in_height, in_width};
|
||||
Mat input(4, &inputShape[0], CV_32F);
|
||||
randu(input, 1.0f, 2.0f); // avoid small values to test eltwise div
|
||||
|
||||
bool bias_term = get<0>(GetParam());
|
||||
LayerParams convParams;
|
||||
TestLayerFusion::makeDefaultTestConvolutionLayer(convParams, in_channels, in_channels, bias_term);
|
||||
|
||||
std::string eltwiseOp = get<1>(GetParam());
|
||||
bool weightedEltwise = get<2>(GetParam());
|
||||
if (eltwiseOp != "sum" && weightedEltwise)
|
||||
throw SkipTestException("weighted eltwise not supported");
|
||||
LayerParams eltwiseParams;
|
||||
TestLayerFusion::makeDefaultTestEltwiseLayer(eltwiseParams, eltwiseOp, weightedEltwise);
|
||||
|
||||
std::string actType = get<3>(GetParam());
|
||||
LayerParams activationParams;
|
||||
TestLayerFusion::makeDefaultTestActivationLayer(activationParams, actType, in_channels);
|
||||
|
||||
Backend backendId = get<0>(get<4>(GetParam()));
|
||||
Target targetId = get<1>(get<4>(GetParam()));
|
||||
|
||||
Net net;
|
||||
int convId = net.addLayer(convParams.name, convParams.type, convParams);
|
||||
int eltwiseId = net.addLayer(eltwiseParams.name, eltwiseParams.type, eltwiseParams);
|
||||
int activId = net.addLayer(activationParams.name, activationParams.type, activationParams);
|
||||
net.connect(0, 0, convId, 0);
|
||||
net.connect(convId, 0, eltwiseId, 0);
|
||||
net.connect(0, 0, eltwiseId, 1);
|
||||
net.connect(eltwiseId, 0, activId, 0);
|
||||
|
||||
std::vector<int> expectedFusedLayers;
|
||||
if (backendId == DNN_BACKEND_OPENCV)
|
||||
{
|
||||
if (targetId == DNN_TARGET_CPU)
|
||||
expectedFusedLayers.push_back(activId); // activation is fused with eltwise layer
|
||||
else if (targetId == DNN_TARGET_OPENCL || targetId == DNN_TARGET_OPENCL_FP16)
|
||||
{
|
||||
if (eltwiseOp == "sum" && !weightedEltwise &&
|
||||
(actType == "ReLU" || actType == "ChannelsPReLU" /*|| actType == "Power"*/)
|
||||
)
|
||||
{
|
||||
expectedFusedLayers.push_back(eltwiseId);
|
||||
expectedFusedLayers.push_back(activId);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
TestLayerFusion::test(input, net, backendId, targetId, expectedFusedLayers);
|
||||
}
|
||||
INSTANTIATE_TEST_CASE_P(TestLayerFusion, ConvolutionEltwiseActivationFusion, Combine(
|
||||
/* bias */ testing::Bool(),
|
||||
/* eltwise op */ TestLayerFusion::eltwiseOpList(),
|
||||
/* eltwise weighted */ testing::Bool(),
|
||||
/* activation */ TestLayerFusion::activationLayersList(),
|
||||
TestLayerFusion::dnnBackendsAndTargetsForFusionTests()
|
||||
));
|
||||
|
||||
typedef TestWithParam<tuple<bool, std::string, std::string, bool, tuple<Backend, Target> > > ConvolutionActivationEltwiseFusion;
|
||||
TEST_P(ConvolutionActivationEltwiseFusion, Accuracy)
|
||||
{
|
||||
// input
|
||||
// |
|
||||
// -------------------------------
|
||||
// | |
|
||||
// | ----------------
|
||||
// | | convolution |
|
||||
// | ----------------
|
||||
// | |
|
||||
// | ----------------
|
||||
// | | activation |
|
||||
// | ----------------
|
||||
// | |
|
||||
// | ---------------- |
|
||||
// --------| eltwise sum |-------
|
||||
// ----------------
|
||||
// |
|
||||
|
||||
const int batch_size = 2, in_channels = 16;
|
||||
const int in_height = 16, in_width = 16;
|
||||
int inputShape[] = {batch_size, in_channels, in_height, in_width};
|
||||
Mat input(4, &inputShape[0], CV_32F);
|
||||
randu(input, 1.0f, 2.0f); // avoid small values to test eltwise div
|
||||
|
||||
bool bias_term = get<0>(GetParam());
|
||||
LayerParams convParams;
|
||||
TestLayerFusion::makeDefaultTestConvolutionLayer(convParams, in_channels, in_channels, bias_term);
|
||||
|
||||
std::string actType = get<1>(GetParam());
|
||||
LayerParams activationParams;
|
||||
TestLayerFusion::makeDefaultTestActivationLayer(activationParams, actType, in_channels);
|
||||
|
||||
std::string eltwiseOp = get<2>(GetParam());
|
||||
bool weightedEltwise = get<3>(GetParam());
|
||||
if (eltwiseOp != "sum" && weightedEltwise)
|
||||
throw SkipTestException("weighted eltwise not supported");
|
||||
LayerParams eltwiseParams;
|
||||
TestLayerFusion::makeDefaultTestEltwiseLayer(eltwiseParams, eltwiseOp, weightedEltwise);
|
||||
|
||||
Backend backendId = get<0>(get<4>(GetParam()));
|
||||
Target targetId = get<1>(get<4>(GetParam()));
|
||||
|
||||
Net net;
|
||||
int convId = net.addLayer(convParams.name, convParams.type, convParams);
|
||||
int activId = net.addLayer(activationParams.name, activationParams.type, activationParams);
|
||||
int eltwiseId = net.addLayer(eltwiseParams.name, eltwiseParams.type, eltwiseParams);
|
||||
net.connect(0, 0, convId, 0);
|
||||
net.connect(convId, 0, activId, 0);
|
||||
net.connect(activId, 0, eltwiseId, 0);
|
||||
net.connect(0, 0, eltwiseId, 1);
|
||||
|
||||
std::vector<int> expectedFusedLayers;
|
||||
if (backendId == DNN_BACKEND_OPENCV)
|
||||
{
|
||||
if (targetId == DNN_TARGET_CPU)
|
||||
expectedFusedLayers.push_back(activId); // activation fused with convolution
|
||||
else if (targetId == DNN_TARGET_OPENCL || targetId == DNN_TARGET_OPENCL_FP16)
|
||||
{
|
||||
if (actType == "ReLU" || actType == "ChannelsPReLU" || actType == "ReLU6" || actType == "TanH" /*|| actType == "Power"*/)
|
||||
expectedFusedLayers.push_back(activId); // activation fused with convolution
|
||||
}
|
||||
}
|
||||
|
||||
TestLayerFusion::test(input, net, backendId, targetId, expectedFusedLayers);
|
||||
}
|
||||
INSTANTIATE_TEST_CASE_P(TestLayerFusion, ConvolutionActivationEltwiseFusion, Combine(
|
||||
/* bias */ testing::Bool(),
|
||||
/* activation */ TestLayerFusion::activationLayersList(),
|
||||
/* eltwise op */ TestLayerFusion::eltwiseOpList(),
|
||||
/* eltwise weighted */ testing::Bool(),
|
||||
TestLayerFusion::dnnBackendsAndTargetsForFusionTests()
|
||||
));
|
||||
|
||||
}} // namespace
|
||||
|
||||
@@ -99,6 +99,15 @@ TEST(readNet, do_not_call_setInput) // https://github.com/opencv/opencv/issues/
|
||||
EXPECT_TRUE(res.empty()) << res.size;
|
||||
}
|
||||
|
||||
TEST(Net, empty_forward_18392)
|
||||
{
|
||||
cv::dnn::Net net;
|
||||
Mat image(Size(512, 512), CV_8UC3, Scalar::all(0));
|
||||
Mat inputBlob = cv::dnn::blobFromImage(image, 1.0, Size(512, 512), Scalar(0,0,0), true, false);
|
||||
net.setInput(inputBlob);
|
||||
EXPECT_ANY_THROW(Mat output = net.forward());
|
||||
}
|
||||
|
||||
#ifdef HAVE_INF_ENGINE
|
||||
static
|
||||
void test_readNet_IE_do_not_call_setInput(Backend backendId)
|
||||
@@ -440,12 +449,14 @@ TEST_P(Async, model_optimizer_pipeline_set_and_forward_single)
|
||||
const Backend backendId = get<0>(get<1>(GetParam()));
|
||||
const Target targetId = get<1>(get<1>(GetParam()));
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && targetId == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
|
||||
if (backendId != DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && backendId != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
throw SkipTestException("No support for async forward");
|
||||
|
||||
const std::string suffix = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD) ? "_fp16" : "";
|
||||
const std::string& model = findDataFile("dnn/layers/layer_convolution" + suffix + ".bin");
|
||||
const std::string& proto = findDataFile("dnn/layers/layer_convolution" + suffix + ".xml");
|
||||
const std::string& model = findDataFile("dnn/layers/layer_convolution.bin");
|
||||
const std::string& proto = findDataFile("dnn/layers/layer_convolution.xml");
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
setInferenceEngineBackendType(CV_DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_API);
|
||||
@@ -499,12 +510,14 @@ TEST_P(Async, model_optimizer_pipeline_set_and_forward_all)
|
||||
const Backend backendId = get<0>(get<1>(GetParam()));
|
||||
const Target targetId = get<1>(get<1>(GetParam()));
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && targetId == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
|
||||
if (backendId != DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && backendId != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
throw SkipTestException("No support for async forward");
|
||||
|
||||
const std::string suffix = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD) ? "_fp16" : "";
|
||||
const std::string& model = findDataFile("dnn/layers/layer_convolution" + suffix + ".bin");
|
||||
const std::string& proto = findDataFile("dnn/layers/layer_convolution" + suffix + ".xml");
|
||||
const std::string& model = findDataFile("dnn/layers/layer_convolution.bin");
|
||||
const std::string& proto = findDataFile("dnn/layers/layer_convolution.xml");
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
setInferenceEngineBackendType(CV_DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_API);
|
||||
@@ -673,9 +686,11 @@ TEST_P(Test_Model_Optimizer, forward_two_nets)
|
||||
const Backend backendId = get<0>(GetParam());
|
||||
const Target targetId = get<1>(GetParam());
|
||||
|
||||
const std::string suffix = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD) ? "_fp16" : "";
|
||||
const std::string& model = findDataFile("dnn/layers/layer_convolution" + suffix + ".bin");
|
||||
const std::string& proto = findDataFile("dnn/layers/layer_convolution" + suffix + ".xml");
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && targetId == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
|
||||
const std::string& model = findDataFile("dnn/layers/layer_convolution.bin");
|
||||
const std::string& proto = findDataFile("dnn/layers/layer_convolution.xml");
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
setInferenceEngineBackendType(CV_DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_API);
|
||||
@@ -712,12 +727,14 @@ TEST_P(Test_Model_Optimizer, readFromBuffer)
|
||||
const Backend backendId = get<0>(GetParam());
|
||||
const Target targetId = get<1>(GetParam());
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && targetId == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
|
||||
if (backendId != DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && backendId != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
throw SkipTestException("No support for async forward");
|
||||
|
||||
const std::string suffix = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD) ? "_fp16" : "";
|
||||
const std::string& weightsFile = findDataFile("dnn/layers/layer_convolution" + suffix + ".bin");
|
||||
const std::string& modelFile = findDataFile("dnn/layers/layer_convolution" + suffix + ".xml");
|
||||
const std::string& weightsFile = findDataFile("dnn/layers/layer_convolution.bin");
|
||||
const std::string& modelFile = findDataFile("dnn/layers/layer_convolution.xml");
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
setInferenceEngineBackendType(CV_DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_API);
|
||||
@@ -765,8 +782,11 @@ TEST_P(Test_Model_Optimizer, flexible_inputs)
|
||||
const Backend backendId = get<0>(GetParam());
|
||||
const Target targetId = get<1>(GetParam());
|
||||
|
||||
const std::string& model = findDataFile("dnn/layers/layer_convolution_fp16.bin");
|
||||
const std::string& proto = findDataFile("dnn/layers/layer_convolution_fp16.xml");
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && targetId == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
|
||||
const std::string& model = findDataFile("dnn/layers/layer_convolution.bin");
|
||||
const std::string& proto = findDataFile("dnn/layers/layer_convolution.xml");
|
||||
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
setInferenceEngineBackendType(CV_DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_API);
|
||||
|
||||
@@ -111,6 +111,73 @@ TEST_P(Test_ONNX_layers, Convolution)
|
||||
testONNXModels("convolution");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Convolution_variable_weight)
|
||||
{
|
||||
if ((backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH ||
|
||||
backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019) && target == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER, CV_TEST_TAG_DNN_SKIP_IE_NGRAPH);
|
||||
|
||||
String basename = "conv_variable_w";
|
||||
Net net = readNetFromONNX(_tf("models/" + basename + ".onnx"));
|
||||
ASSERT_FALSE(net.empty());
|
||||
|
||||
net.setPreferableBackend(backend);
|
||||
net.setPreferableTarget(target);
|
||||
|
||||
for (int i = 0; i < 2; i++)
|
||||
{
|
||||
Mat input = blobFromNPY(_tf("data/input_" + basename + format("_%d", i) + "_0.npy"));
|
||||
Mat weights = blobFromNPY(_tf("data/input_" + basename + format("_%d", i) + "_1.npy"));
|
||||
Mat ref = blobFromNPY(_tf("data/output_" + basename + format("_%d", i) + ".npy"));
|
||||
|
||||
net.setInput(input, "0");
|
||||
net.setInput(weights, "1");
|
||||
|
||||
Mat out = net.forward();
|
||||
normAssert(ref, out, "", default_l1, default_lInf);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Convolution_variable_weight_bias)
|
||||
{
|
||||
if ((backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH ||
|
||||
backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019) && target == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER, CV_TEST_TAG_DNN_SKIP_IE_NGRAPH);
|
||||
|
||||
String basename = "conv_variable_wb";
|
||||
Net net = readNetFromONNX(_tf("models/" + basename + ".onnx"));
|
||||
ASSERT_FALSE(net.empty());
|
||||
|
||||
net.setPreferableBackend(backend);
|
||||
net.setPreferableTarget(target);
|
||||
|
||||
for (int i = 0; i < 2; i++)
|
||||
{
|
||||
Mat input = blobFromNPY(_tf("data/input_" + basename + format("_%d", i) + "_0.npy"));
|
||||
Mat weights = blobFromNPY(_tf("data/input_" + basename + format("_%d", i) + "_1.npy"));
|
||||
Mat bias = blobFromNPY(_tf("data/input_" + basename + format("_%d", i) + "_2.npy"));
|
||||
Mat ref = blobFromNPY(_tf("data/output_" + basename + format("_%d", i) + ".npy"));
|
||||
|
||||
net.setInput(input, "0");
|
||||
net.setInput(weights, "1");
|
||||
net.setInput(bias, "bias");
|
||||
|
||||
Mat out = net.forward();
|
||||
normAssert(ref, out, "", default_l1, default_lInf);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Gather)
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && target == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
testONNXModels("gather");
|
||||
// GPU plugin unsupported slice for constant
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && (target == DNN_TARGET_OPENCL || target == DNN_TARGET_OPENCL_FP16))
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL, CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16, CV_TEST_TAG_DNN_SKIP_IE_NGRAPH);
|
||||
testONNXModels("gather_scalar", npy, 0, 0, false, false);
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Convolution3D)
|
||||
{
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_LT(2019010000)
|
||||
@@ -190,6 +257,23 @@ TEST_P(Test_ONNX_layers, ReduceMean)
|
||||
testONNXModels("reduce_mean_axis2");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, ReduceSum)
|
||||
{
|
||||
testONNXModels("reduce_sum");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, ReduceMaxGlobal)
|
||||
{
|
||||
testONNXModels("reduce_max");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Scale)
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
testONNXModels("scale");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, ReduceMean3D)
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && target != DNN_TARGET_CPU)
|
||||
@@ -211,6 +295,11 @@ TEST_P(Test_ONNX_layers, Cast)
|
||||
testONNXModels("cast");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Power)
|
||||
{
|
||||
testONNXModels("pow2", npy, 0, 0, false, false);
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Concatenation)
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
@@ -337,10 +426,20 @@ TEST_P(Test_ONNX_layers, MatMul)
|
||||
testONNXModels("matmul_4d");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, MatMulAdd)
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
if (backend == DNN_BACKEND_OPENCV && target == DNN_TARGET_OPENCL_FP16)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_OPENCL_FP16);
|
||||
testONNXModels("matmul_add");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Expand)
|
||||
{
|
||||
testONNXModels("expand_batch");
|
||||
testONNXModels("expand_channels");
|
||||
testONNXModels("expand_neg_batch");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, ExpandHW)
|
||||
@@ -524,6 +623,31 @@ TEST_P(Test_ONNX_layers, Pad2d_Unfused)
|
||||
testONNXModels("ZeroPad2d");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, LinearWithConstant)
|
||||
{
|
||||
if (backend == DNN_BACKEND_OPENCV && target == DNN_TARGET_OPENCL_FP16)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_OPENCL_FP16);
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_LT(2020040000)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE);
|
||||
#endif
|
||||
testONNXModels("lin_with_constant");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, MatmulWithTwoInputs)
|
||||
{
|
||||
if (backend == DNN_BACKEND_OPENCV && target == DNN_TARGET_OPENCL_FP16)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_OPENCL_FP16);
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_LT(2020040000)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE);
|
||||
#endif
|
||||
testONNXModels("matmul_with_two_inputs");
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, ResizeOpset11_Torch1_6)
|
||||
{
|
||||
testONNXModels("resize_opset11_torch1.6");
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/*nothing*/, Test_ONNX_layers, dnnBackendsAndTargets());
|
||||
|
||||
class Test_ONNX_nets : public Test_ONNX_layers
|
||||
|
||||
@@ -128,6 +128,13 @@ TEST_P(Test_TensorFlow_layers, reduce_mean)
|
||||
runTensorFlowNet("global_pool_by_axis");
|
||||
}
|
||||
|
||||
TEST_P(Test_TensorFlow_layers, reduce_sum)
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
runTensorFlowNet("sum_pool_by_axis");
|
||||
}
|
||||
|
||||
TEST_P(Test_TensorFlow_layers, conv_single_conv)
|
||||
{
|
||||
runTensorFlowNet("single_conv");
|
||||
@@ -340,6 +347,11 @@ TEST_P(Test_TensorFlow_layers, pooling_reduce_mean)
|
||||
runTensorFlowNet("reduce_mean"); // an average pooling over all spatial dimensions.
|
||||
}
|
||||
|
||||
TEST_P(Test_TensorFlow_layers, pooling_reduce_sum)
|
||||
{
|
||||
runTensorFlowNet("reduce_sum"); // a SUM pooling over all spatial dimensions.
|
||||
}
|
||||
|
||||
TEST_P(Test_TensorFlow_layers, max_pool_grad)
|
||||
{
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||
|
||||
@@ -113,7 +113,7 @@ TEST_P(Test_Torch_layers, run_convolution)
|
||||
{
|
||||
// Output reference values are in range [23.4018, 72.0181]
|
||||
double l1 = (target == DNN_TARGET_OPENCL_FP16 || target == DNN_TARGET_MYRIAD) ? 0.08 : default_l1;
|
||||
double lInf = (target == DNN_TARGET_OPENCL_FP16 || target == DNN_TARGET_MYRIAD) ? 0.42 : default_lInf;
|
||||
double lInf = (target == DNN_TARGET_OPENCL_FP16 || target == DNN_TARGET_MYRIAD) ? 0.43 : default_lInf;
|
||||
runTorchNet("net_conv", "", false, true, true, l1, lInf);
|
||||
}
|
||||
|
||||
@@ -165,7 +165,7 @@ TEST_P(Test_Torch_layers, run_concat)
|
||||
TEST_P(Test_Torch_layers, run_depth_concat)
|
||||
{
|
||||
runTorchNet("net_depth_concat", "", false, true, true, 0.0,
|
||||
target == DNN_TARGET_OPENCL_FP16 ? 0.021 : 0.0);
|
||||
target == DNN_TARGET_OPENCL_FP16 ? 0.032 : 0.0);
|
||||
}
|
||||
|
||||
TEST_P(Test_Torch_layers, run_deconv)
|
||||
@@ -359,6 +359,10 @@ TEST_P(Test_Torch_nets, ENet_accuracy)
|
||||
if (target == DNN_TARGET_MYRIAD) applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||
throw SkipTestException("");
|
||||
}
|
||||
#endif
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_GE(2021010000)
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_NGRAPH);
|
||||
#endif
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && target != DNN_TARGET_CPU)
|
||||
{
|
||||
|
||||
@@ -245,6 +245,31 @@ typedef Feature2D DescriptorExtractor;
|
||||
//! @{
|
||||
|
||||
|
||||
/** @brief Class for implementing the wrapper which makes detectors and extractors to be affine invariant,
|
||||
described as ASIFT in @cite YM11 .
|
||||
*/
|
||||
class CV_EXPORTS_W AffineFeature : public Feature2D
|
||||
{
|
||||
public:
|
||||
/**
|
||||
@param backend The detector/extractor you want to use as backend.
|
||||
@param maxTilt The highest power index of tilt factor. 5 is used in the paper as tilt sampling range n.
|
||||
@param minTilt The lowest power index of tilt factor. 0 is used in the paper.
|
||||
@param tiltStep Tilt sampling step \f$\delta_t\f$ in Algorithm 1 in the paper.
|
||||
@param rotateStepBase Rotation sampling step factor b in Algorithm 1 in the paper.
|
||||
*/
|
||||
CV_WRAP static Ptr<AffineFeature> create(const Ptr<Feature2D>& backend,
|
||||
int maxTilt = 5, int minTilt = 0, float tiltStep = 1.4142135623730951f, float rotateStepBase = 72);
|
||||
|
||||
CV_WRAP virtual void setViewParams(const std::vector<float>& tilts, const std::vector<float>& rolls) = 0;
|
||||
CV_WRAP virtual void getViewParams(std::vector<float>& tilts, std::vector<float>& rolls) const = 0;
|
||||
CV_WRAP virtual String getDefaultName() const CV_OVERRIDE;
|
||||
};
|
||||
|
||||
typedef AffineFeature AffineFeatureDetector;
|
||||
typedef AffineFeature AffineDescriptorExtractor;
|
||||
|
||||
|
||||
/** @brief Class for extracting keypoints and computing descriptors using the Scale Invariant Feature Transform
|
||||
(SIFT) algorithm by D. Lowe @cite Lowe04 .
|
||||
*/
|
||||
@@ -276,6 +301,33 @@ public:
|
||||
double contrastThreshold = 0.04, double edgeThreshold = 10,
|
||||
double sigma = 1.6);
|
||||
|
||||
/** @brief Create SIFT with specified descriptorType.
|
||||
@param nfeatures The number of best features to retain. The features are ranked by their scores
|
||||
(measured in SIFT algorithm as the local contrast)
|
||||
|
||||
@param nOctaveLayers The number of layers in each octave. 3 is the value used in D. Lowe paper. The
|
||||
number of octaves is computed automatically from the image resolution.
|
||||
|
||||
@param contrastThreshold The contrast threshold used to filter out weak features in semi-uniform
|
||||
(low-contrast) regions. The larger the threshold, the less features are produced by the detector.
|
||||
|
||||
@note The contrast threshold will be divided by nOctaveLayers when the filtering is applied. When
|
||||
nOctaveLayers is set to default and if you want to use the value used in D. Lowe paper, 0.03, set
|
||||
this argument to 0.09.
|
||||
|
||||
@param edgeThreshold The threshold used to filter out edge-like features. Note that the its meaning
|
||||
is different from the contrastThreshold, i.e. the larger the edgeThreshold, the less features are
|
||||
filtered out (more features are retained).
|
||||
|
||||
@param sigma The sigma of the Gaussian applied to the input image at the octave \#0. If your image
|
||||
is captured with a weak camera with soft lenses, you might want to reduce the number.
|
||||
|
||||
@param descriptorType The type of descriptors. Only CV_32F and CV_8U are supported.
|
||||
*/
|
||||
CV_WRAP static Ptr<SIFT> create(int nfeatures, int nOctaveLayers,
|
||||
double contrastThreshold, double edgeThreshold,
|
||||
double sigma, int descriptorType);
|
||||
|
||||
CV_WRAP virtual String getDefaultName() const CV_OVERRIDE;
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,358 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
//
|
||||
// This file is based on code issued with the following license.
|
||||
/*********************************************************************
|
||||
* Software License Agreement (BSD License)
|
||||
*
|
||||
* Copyright (C) 2000-2008, Intel Corporation, all rights reserved.
|
||||
* Copyright (C) 2008-2013, Willow Garage Inc., all rights reserved.
|
||||
* Copyright (C) 2013, Evgeny Toropov, all rights reserved.
|
||||
* Third party copyrights are property of their respective owners.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
*
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above
|
||||
* copyright notice, this list of conditions and the following
|
||||
* disclaimer in the documentation and/or other materials provided
|
||||
* with the distribution.
|
||||
* * The name of the copyright holders may not be used to endorse
|
||||
* or promote products derived from this software without specific
|
||||
* prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||
* "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||
* LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
|
||||
* FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE
|
||||
* COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT,
|
||||
* INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
|
||||
* LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
|
||||
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
|
||||
* ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||
* POSSIBILITY OF SUCH DAMAGE.
|
||||
*********************************************************************/
|
||||
|
||||
/*
|
||||
Guoshen Yu, Jean-Michel Morel, ASIFT: An Algorithm for Fully Affine
|
||||
Invariant Comparison, Image Processing On Line, 1 (2011), pp. 11–38.
|
||||
https://doi.org/10.5201/ipol.2011.my-asift
|
||||
*/
|
||||
|
||||
#include "precomp.hpp"
|
||||
#include <iostream>
|
||||
namespace cv {
|
||||
|
||||
class AffineFeature_Impl CV_FINAL : public AffineFeature
|
||||
{
|
||||
public:
|
||||
explicit AffineFeature_Impl(const Ptr<Feature2D>& backend,
|
||||
int maxTilt, int minTilt, float tiltStep, float rotateStepBase);
|
||||
|
||||
int descriptorSize() const CV_OVERRIDE
|
||||
{
|
||||
return backend_->descriptorSize();
|
||||
}
|
||||
|
||||
int descriptorType() const CV_OVERRIDE
|
||||
{
|
||||
return backend_->descriptorType();
|
||||
}
|
||||
|
||||
int defaultNorm() const CV_OVERRIDE
|
||||
{
|
||||
return backend_->defaultNorm();
|
||||
}
|
||||
|
||||
void detectAndCompute(InputArray image, InputArray mask, std::vector<KeyPoint>& keypoints,
|
||||
OutputArray descriptors, bool useProvidedKeypoints=false) CV_OVERRIDE;
|
||||
|
||||
void setViewParams(const std::vector<float>& tilts, const std::vector<float>& rolls) CV_OVERRIDE;
|
||||
void getViewParams(std::vector<float>& tilts, std::vector<float>& rolls) const CV_OVERRIDE;
|
||||
|
||||
protected:
|
||||
void splitKeypointsByView(const std::vector<KeyPoint>& keypoints_,
|
||||
std::vector< std::vector<KeyPoint> >& keypointsByView) const;
|
||||
|
||||
const Ptr<Feature2D> backend_;
|
||||
int maxTilt_;
|
||||
int minTilt_;
|
||||
float tiltStep_;
|
||||
float rotateStepBase_;
|
||||
|
||||
// Tilt factors.
|
||||
std::vector<float> tilts_;
|
||||
// Roll factors.
|
||||
std::vector<float> rolls_;
|
||||
|
||||
private:
|
||||
AffineFeature_Impl(const AffineFeature_Impl &); // copy disabled
|
||||
AffineFeature_Impl& operator=(const AffineFeature_Impl &); // assign disabled
|
||||
};
|
||||
|
||||
AffineFeature_Impl::AffineFeature_Impl(const Ptr<FeatureDetector>& backend,
|
||||
int maxTilt, int minTilt, float tiltStep, float rotateStepBase)
|
||||
: backend_(backend), maxTilt_(maxTilt), minTilt_(minTilt), tiltStep_(tiltStep), rotateStepBase_(rotateStepBase)
|
||||
{
|
||||
int i = minTilt_;
|
||||
if( i == 0 )
|
||||
{
|
||||
tilts_.push_back(1);
|
||||
rolls_.push_back(0);
|
||||
i++;
|
||||
}
|
||||
float tilt = 1;
|
||||
for( ; i <= maxTilt_; i++ )
|
||||
{
|
||||
tilt *= tiltStep_;
|
||||
float rotateStep = rotateStepBase_ / tilt;
|
||||
int rollN = cvFloor(180.0f / rotateStep);
|
||||
if( rollN * rotateStep == 180.0f )
|
||||
rollN--;
|
||||
for( int j = 0; j <= rollN; j++ )
|
||||
{
|
||||
tilts_.push_back(tilt);
|
||||
rolls_.push_back(rotateStep * j);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void AffineFeature_Impl::setViewParams(const std::vector<float>& tilts,
|
||||
const std::vector<float>& rolls)
|
||||
{
|
||||
CV_Assert(tilts.size() == rolls.size());
|
||||
tilts_ = tilts;
|
||||
rolls_ = rolls;
|
||||
}
|
||||
|
||||
void AffineFeature_Impl::getViewParams(std::vector<float>& tilts,
|
||||
std::vector<float>& rolls) const
|
||||
{
|
||||
tilts = tilts_;
|
||||
rolls = rolls_;
|
||||
}
|
||||
|
||||
void AffineFeature_Impl::splitKeypointsByView(const std::vector<KeyPoint>& keypoints_,
|
||||
std::vector< std::vector<KeyPoint> >& keypointsByView) const
|
||||
{
|
||||
for( size_t i = 0; i < keypoints_.size(); i++ )
|
||||
{
|
||||
const KeyPoint& kp = keypoints_[i];
|
||||
CV_Assert( kp.class_id >= 0 && kp.class_id < (int)tilts_.size() );
|
||||
keypointsByView[kp.class_id].push_back(kp);
|
||||
}
|
||||
}
|
||||
|
||||
class skewedDetectAndCompute : public ParallelLoopBody
|
||||
{
|
||||
public:
|
||||
skewedDetectAndCompute(
|
||||
const std::vector<float>& _tilts,
|
||||
const std::vector<float>& _rolls,
|
||||
std::vector< std::vector<KeyPoint> >& _keypointsCollection,
|
||||
std::vector<Mat>& _descriptorCollection,
|
||||
const Mat& _image,
|
||||
const Mat& _mask,
|
||||
const bool _do_keypoints,
|
||||
const bool _do_descriptors,
|
||||
const Ptr<Feature2D>& _backend)
|
||||
: tilts(_tilts),
|
||||
rolls(_rolls),
|
||||
keypointsCollection(_keypointsCollection),
|
||||
descriptorCollection(_descriptorCollection),
|
||||
image(_image),
|
||||
mask(_mask),
|
||||
do_keypoints(_do_keypoints),
|
||||
do_descriptors(_do_descriptors),
|
||||
backend(_backend) {}
|
||||
|
||||
void operator()( const cv::Range& range ) const CV_OVERRIDE
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
|
||||
const int begin = range.start;
|
||||
const int end = range.end;
|
||||
|
||||
for( int a = begin; a < end; a++ )
|
||||
{
|
||||
Mat warpedImage, warpedMask;
|
||||
Matx23f pose, invPose;
|
||||
affineSkew(tilts[a], rolls[a], warpedImage, warpedMask, pose);
|
||||
invertAffineTransform(pose, invPose);
|
||||
|
||||
std::vector<KeyPoint> wKeypoints;
|
||||
Mat wDescriptors;
|
||||
if( !do_keypoints )
|
||||
{
|
||||
const std::vector<KeyPoint>& keypointsInView = keypointsCollection[a];
|
||||
if( keypointsInView.size() == 0 ) // when there are no keypoints in this affine view
|
||||
continue;
|
||||
|
||||
std::vector<Point2f> pts_, pts;
|
||||
KeyPoint::convert(keypointsInView, pts_);
|
||||
transform(pts_, pts, pose);
|
||||
wKeypoints.resize(keypointsInView.size());
|
||||
for( size_t wi = 0; wi < wKeypoints.size(); wi++ )
|
||||
{
|
||||
wKeypoints[wi] = keypointsInView[wi];
|
||||
wKeypoints[wi].pt = pts[wi];
|
||||
}
|
||||
}
|
||||
backend->detectAndCompute(warpedImage, warpedMask, wKeypoints, wDescriptors, !do_keypoints);
|
||||
if( do_keypoints )
|
||||
{
|
||||
// KeyPointsFilter::runByPixelsMask( wKeypoints, warpedMask );
|
||||
if( wKeypoints.size() == 0 )
|
||||
{
|
||||
keypointsCollection[a].clear();
|
||||
continue;
|
||||
}
|
||||
std::vector<Point2f> pts_, pts;
|
||||
KeyPoint::convert(wKeypoints, pts_);
|
||||
transform(pts_, pts, invPose);
|
||||
|
||||
keypointsCollection[a].resize(wKeypoints.size());
|
||||
for( size_t wi = 0; wi < wKeypoints.size(); wi++ )
|
||||
{
|
||||
keypointsCollection[a][wi] = wKeypoints[wi];
|
||||
keypointsCollection[a][wi].pt = pts[wi];
|
||||
keypointsCollection[a][wi].class_id = a;
|
||||
}
|
||||
}
|
||||
if( do_descriptors )
|
||||
wDescriptors.copyTo(descriptorCollection[a]);
|
||||
}
|
||||
}
|
||||
private:
|
||||
void affineSkew(float tilt, float phi,
|
||||
Mat& warpedImage, Mat& warpedMask, Matx23f& pose) const
|
||||
{
|
||||
int h = image.size().height;
|
||||
int w = image.size().width;
|
||||
Mat rotImage;
|
||||
|
||||
Mat mask0;
|
||||
if( mask.empty() )
|
||||
mask0 = Mat(h, w, CV_8UC1, 255);
|
||||
else
|
||||
mask0 = mask;
|
||||
pose = Matx23f(1,0,0,
|
||||
0,1,0);
|
||||
|
||||
if( phi == 0 )
|
||||
image.copyTo(rotImage);
|
||||
else
|
||||
{
|
||||
phi = phi * (float)CV_PI / 180;
|
||||
float s = std::sin(phi);
|
||||
float c = std::cos(phi);
|
||||
Matx22f A(c, -s, s, c);
|
||||
Matx<float, 4, 2> corners(0, 0, (float)w, 0, (float)w,(float)h, 0, (float)h);
|
||||
Mat tf(corners * A.t());
|
||||
Mat tcorners;
|
||||
tf.convertTo(tcorners, CV_32S);
|
||||
Rect rect = boundingRect(tcorners);
|
||||
h = rect.height; w = rect.width;
|
||||
pose = Matx23f(c, -s, -(float)rect.x,
|
||||
s, c, -(float)rect.y);
|
||||
warpAffine(image, rotImage, pose, Size(w, h), INTER_LINEAR, BORDER_REPLICATE);
|
||||
}
|
||||
if( tilt == 1 )
|
||||
warpedImage = rotImage;
|
||||
else
|
||||
{
|
||||
float s = 0.8f * sqrt(tilt * tilt - 1);
|
||||
GaussianBlur(rotImage, rotImage, Size(0, 0), s, 0.01);
|
||||
resize(rotImage, warpedImage, Size(0, 0), 1.0/tilt, 1.0, INTER_NEAREST);
|
||||
pose(0, 0) /= tilt;
|
||||
pose(0, 1) /= tilt;
|
||||
pose(0, 2) /= tilt;
|
||||
}
|
||||
if( phi != 0 || tilt != 1 )
|
||||
warpAffine(mask0, warpedMask, pose, warpedImage.size(), INTER_NEAREST);
|
||||
}
|
||||
|
||||
|
||||
const std::vector<float>& tilts;
|
||||
const std::vector<float>& rolls;
|
||||
std::vector< std::vector<KeyPoint> >& keypointsCollection;
|
||||
std::vector<Mat>& descriptorCollection;
|
||||
const Mat& image;
|
||||
const Mat& mask;
|
||||
const bool do_keypoints;
|
||||
const bool do_descriptors;
|
||||
const Ptr<Feature2D>& backend;
|
||||
};
|
||||
|
||||
void AffineFeature_Impl::detectAndCompute(InputArray _image, InputArray _mask,
|
||||
std::vector<KeyPoint>& keypoints,
|
||||
OutputArray _descriptors,
|
||||
bool useProvidedKeypoints)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
|
||||
bool do_keypoints = !useProvidedKeypoints;
|
||||
bool do_descriptors = _descriptors.needed();
|
||||
Mat image = _image.getMat(), mask = _mask.getMat();
|
||||
Mat descriptors;
|
||||
|
||||
if( (!do_keypoints && !do_descriptors) || _image.empty() )
|
||||
return;
|
||||
|
||||
std::vector< std::vector<KeyPoint> > keypointsCollection(tilts_.size());
|
||||
std::vector< Mat > descriptorCollection(tilts_.size());
|
||||
|
||||
if( do_keypoints )
|
||||
keypoints.clear();
|
||||
else
|
||||
splitKeypointsByView(keypoints, keypointsCollection);
|
||||
|
||||
parallel_for_(Range(0, (int)tilts_.size()), skewedDetectAndCompute(tilts_, rolls_, keypointsCollection, descriptorCollection,
|
||||
image, mask, do_keypoints, do_descriptors, backend_));
|
||||
|
||||
if( do_keypoints )
|
||||
for( size_t i = 0; i < keypointsCollection.size(); i++ )
|
||||
{
|
||||
const std::vector<KeyPoint>& keys = keypointsCollection[i];
|
||||
keypoints.insert(keypoints.end(), keys.begin(), keys.end());
|
||||
}
|
||||
|
||||
if( do_descriptors )
|
||||
{
|
||||
_descriptors.create((int)keypoints.size(), backend_->descriptorSize(), backend_->descriptorType());
|
||||
descriptors = _descriptors.getMat();
|
||||
int iter = 0;
|
||||
for( size_t i = 0; i < descriptorCollection.size(); i++ )
|
||||
{
|
||||
const Mat& descs = descriptorCollection[i];
|
||||
if( descs.empty() )
|
||||
continue;
|
||||
Mat roi(descriptors, Rect(0, iter, descriptors.cols, descs.rows));
|
||||
descs.copyTo(roi);
|
||||
iter += descs.rows;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Ptr<AffineFeature> AffineFeature::create(const Ptr<Feature2D>& backend,
|
||||
int maxTilt, int minTilt, float tiltStep, float rotateStepBase)
|
||||
{
|
||||
CV_Assert(minTilt < maxTilt);
|
||||
CV_Assert(tiltStep > 0);
|
||||
CV_Assert(rotateStepBase > 0);
|
||||
return makePtr<AffineFeature_Impl>(backend, maxTilt, minTilt, tiltStep, rotateStepBase);
|
||||
}
|
||||
|
||||
String AffineFeature::getDefaultName() const
|
||||
{
|
||||
return (Feature2D::getDefaultName() + ".AffineFeature");
|
||||
}
|
||||
|
||||
} // namespace
|
||||
@@ -353,13 +353,30 @@ BRISK_Impl::generateKernel(const std::vector<float> &radiusList,
|
||||
const int rings = (int)radiusList.size();
|
||||
CV_Assert(radiusList.size() != 0 && radiusList.size() == numberList.size());
|
||||
points_ = 0; // remember the total number of points
|
||||
double sineThetaLookupTable[n_rot_];
|
||||
double cosThetaLookupTable[n_rot_];
|
||||
for (int ring = 0; ring < rings; ring++)
|
||||
{
|
||||
points_ += numberList[ring];
|
||||
}
|
||||
|
||||
// using a sine/cosine approximation for the lookup table
|
||||
// utilizes the trig identities:
|
||||
// sin(a + b) = sin(a)cos(b) + cos(a)sin(b)
|
||||
// cos(a + b) = cos(a)cos(b) - sin(a)sin(b)
|
||||
// and the fact that sin(0) = 0, cos(0) = 1
|
||||
double cosval = 1., sinval = 0.;
|
||||
double dcos = cos(2*CV_PI/double(n_rot_)), dsin = sin(2*CV_PI/double(n_rot_));
|
||||
for( size_t rot = 0; rot < n_rot_; ++rot)
|
||||
{
|
||||
sineThetaLookupTable[rot] = sinval;
|
||||
cosThetaLookupTable[rot] = cosval;
|
||||
double t = sinval*dcos + cosval*dsin;
|
||||
cosval = cosval*dcos - sinval*dsin;
|
||||
sinval = t;
|
||||
}
|
||||
// set up the patterns
|
||||
patternPoints_ = new BriskPatternPoint[points_ * scales_ * n_rot_];
|
||||
BriskPatternPoint* patternIterator = patternPoints_;
|
||||
|
||||
// define the scale discretization:
|
||||
static const float lb_scale = (float)(std::log(scalerange_) / std::log(2.0));
|
||||
@@ -370,46 +387,51 @@ BRISK_Impl::generateKernel(const std::vector<float> &radiusList,
|
||||
|
||||
const float sigma_scale = 1.3f;
|
||||
|
||||
for (unsigned int scale = 0; scale < scales_; ++scale)
|
||||
{
|
||||
scaleList_[scale] = (float)std::pow((double) 2.0, (double) (scale * lb_scale_step));
|
||||
sizeList_[scale] = 0;
|
||||
|
||||
// generate the pattern points look-up
|
||||
double alpha, theta;
|
||||
for (size_t rot = 0; rot < n_rot_; ++rot)
|
||||
{
|
||||
theta = double(rot) * 2 * CV_PI / double(n_rot_); // this is the rotation of the feature
|
||||
for (int ring = 0; ring < rings; ++ring)
|
||||
{
|
||||
for (int num = 0; num < numberList[ring]; ++num)
|
||||
{
|
||||
// the actual coordinates on the circle
|
||||
alpha = (double(num)) * 2 * CV_PI / double(numberList[ring]);
|
||||
patternIterator->x = (float)(scaleList_[scale] * radiusList[ring] * cos(alpha + theta)); // feature rotation plus angle of the point
|
||||
patternIterator->y = (float)(scaleList_[scale] * radiusList[ring] * sin(alpha + theta));
|
||||
// and the gaussian kernel sigma
|
||||
if (ring == 0)
|
||||
{
|
||||
patternIterator->sigma = sigma_scale * scaleList_[scale] * 0.5f;
|
||||
}
|
||||
else
|
||||
{
|
||||
patternIterator->sigma = (float)(sigma_scale * scaleList_[scale] * (double(radiusList[ring]))
|
||||
* sin(CV_PI / numberList[ring]));
|
||||
for (unsigned int scale = 0; scale < scales_; ++scale) {
|
||||
scaleList_[scale] = (float) std::pow((double) 2.0, (double) (scale * lb_scale_step));
|
||||
sizeList_[scale] = 0;
|
||||
BriskPatternPoint *patternIteratorOuter = patternPoints_ + (scale * n_rot_ * points_);
|
||||
// generate the pattern points look-up
|
||||
for (int ring = 0; ring < rings; ++ring) {
|
||||
double scaleRadiusProduct = scaleList_[scale] * radiusList[ring];
|
||||
float patternSigma = 0.0f;
|
||||
if (ring == 0) {
|
||||
patternSigma = sigma_scale * scaleList_[scale] * 0.5f;
|
||||
} else {
|
||||
patternSigma = (float) (sigma_scale * scaleList_[scale] * (double(radiusList[ring]))
|
||||
* sin(CV_PI / numberList[ring]));
|
||||
}
|
||||
// adapt the sizeList if necessary
|
||||
const unsigned int size = cvCeil(((scaleList_[scale] * radiusList[ring]) + patternIterator->sigma)) + 1;
|
||||
if (sizeList_[scale] < size)
|
||||
{
|
||||
sizeList_[scale] = size;
|
||||
const unsigned int size = cvCeil(((scaleList_[scale] * radiusList[ring]) + patternSigma)) + 1;
|
||||
if (sizeList_[scale] < size) {
|
||||
sizeList_[scale] = size;
|
||||
}
|
||||
for (int num = 0; num < numberList[ring]; ++num) {
|
||||
BriskPatternPoint *patternIterator = patternIteratorOuter;
|
||||
double alpha = (double(num)) * 2 * CV_PI / double(numberList[ring]);
|
||||
double sine_alpha = sin(alpha);
|
||||
double cosine_alpha = cos(alpha);
|
||||
|
||||
// increment the iterator
|
||||
++patternIterator;
|
||||
}
|
||||
for (size_t rot = 0; rot < n_rot_; ++rot) {
|
||||
double cosine_theta = cosThetaLookupTable[rot];
|
||||
double sine_theta = sineThetaLookupTable[rot];
|
||||
|
||||
// the actual coordinates on the circle
|
||||
// sin(a + b) = sin(a) cos(b) + cos(a) sin(b)
|
||||
// cos(a + b) = cos(a) cos(b) - sin(a) sin(b)
|
||||
patternIterator->x = (float) (scaleRadiusProduct *
|
||||
(cosine_theta * cosine_alpha -
|
||||
sine_theta * sine_alpha)); // feature rotation plus angle of the point
|
||||
patternIterator->y = (float) (scaleRadiusProduct *
|
||||
(sine_theta * cosine_alpha + cosine_theta * sine_alpha));
|
||||
patternIterator->sigma = patternSigma;
|
||||
// and the gaussian kernel sigma
|
||||
// increment the iterator
|
||||
patternIterator += points_;
|
||||
}
|
||||
++patternIteratorOuter;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// now also generate pairings
|
||||
|
||||
@@ -983,7 +983,11 @@ void ORB_Impl::detectAndCompute( InputArray _image, InputArray _mask,
|
||||
int descPatchSize = cvCeil(halfPatchSize*sqrt(2.0));
|
||||
int border = std::max(edgeThreshold, std::max(descPatchSize, HARRIS_BLOCK_SIZE/2))+1;
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
bool useOCL = ocl::isOpenCLActivated() && OCL_FORCE_CHECK(_image.isUMat() || _descriptors.isUMat());
|
||||
#else
|
||||
bool useOCL = false;
|
||||
#endif
|
||||
|
||||
Mat image = _image.getMat(), mask = _mask.getMat();
|
||||
if( image.type() != CV_8UC1 )
|
||||
|
||||
@@ -88,7 +88,7 @@ class SIFT_Impl : public SIFT
|
||||
public:
|
||||
explicit SIFT_Impl( int nfeatures = 0, int nOctaveLayers = 3,
|
||||
double contrastThreshold = 0.04, double edgeThreshold = 10,
|
||||
double sigma = 1.6);
|
||||
double sigma = 1.6, int descriptorType = CV_32F );
|
||||
|
||||
//! returns the descriptor size in floats (128)
|
||||
int descriptorSize() const CV_OVERRIDE;
|
||||
@@ -117,13 +117,25 @@ protected:
|
||||
CV_PROP_RW double contrastThreshold;
|
||||
CV_PROP_RW double edgeThreshold;
|
||||
CV_PROP_RW double sigma;
|
||||
CV_PROP_RW int descriptor_type;
|
||||
};
|
||||
|
||||
Ptr<SIFT> SIFT::create( int _nfeatures, int _nOctaveLayers,
|
||||
double _contrastThreshold, double _edgeThreshold, double _sigma )
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
return makePtr<SIFT_Impl>(_nfeatures, _nOctaveLayers, _contrastThreshold, _edgeThreshold, _sigma);
|
||||
|
||||
return makePtr<SIFT_Impl>(_nfeatures, _nOctaveLayers, _contrastThreshold, _edgeThreshold, _sigma, CV_32F);
|
||||
}
|
||||
|
||||
Ptr<SIFT> SIFT::create( int _nfeatures, int _nOctaveLayers,
|
||||
double _contrastThreshold, double _edgeThreshold, double _sigma, int _descriptorType )
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
|
||||
// SIFT descriptor supports 32bit floating point and 8bit unsigned int.
|
||||
CV_Assert(_descriptorType == CV_32F || _descriptorType == CV_8U);
|
||||
return makePtr<SIFT_Impl>(_nfeatures, _nOctaveLayers, _contrastThreshold, _edgeThreshold, _sigma, _descriptorType);
|
||||
}
|
||||
|
||||
String SIFT::getDefaultName() const
|
||||
@@ -362,12 +374,12 @@ void SIFT_Impl::findScaleSpaceExtrema( const std::vector<Mat>& gauss_pyr, const
|
||||
static
|
||||
void calcSIFTDescriptor(
|
||||
const Mat& img, Point2f ptf, float ori, float scl,
|
||||
int d, int n, float* dst
|
||||
int d, int n, Mat& dst, int row
|
||||
)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
|
||||
CV_CPU_DISPATCH(calcSIFTDescriptor, (img, ptf, ori, scl, d, n, dst),
|
||||
CV_CPU_DISPATCH(calcSIFTDescriptor, (img, ptf, ori, scl, d, n, dst, row),
|
||||
CV_CPU_DISPATCH_MODES_ALL);
|
||||
}
|
||||
|
||||
@@ -408,7 +420,7 @@ public:
|
||||
float angle = 360.f - kpt.angle;
|
||||
if(std::abs(angle - 360.f) < FLT_EPSILON)
|
||||
angle = 0.f;
|
||||
calcSIFTDescriptor(img, ptf, angle, size*0.5f, d, n, descriptors.ptr<float>((int)i));
|
||||
calcSIFTDescriptor(img, ptf, angle, size*0.5f, d, n, descriptors, i);
|
||||
}
|
||||
}
|
||||
private:
|
||||
@@ -429,9 +441,9 @@ static void calcDescriptors(const std::vector<Mat>& gpyr, const std::vector<KeyP
|
||||
//////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
SIFT_Impl::SIFT_Impl( int _nfeatures, int _nOctaveLayers,
|
||||
double _contrastThreshold, double _edgeThreshold, double _sigma )
|
||||
double _contrastThreshold, double _edgeThreshold, double _sigma, int _descriptorType )
|
||||
: nfeatures(_nfeatures), nOctaveLayers(_nOctaveLayers),
|
||||
contrastThreshold(_contrastThreshold), edgeThreshold(_edgeThreshold), sigma(_sigma)
|
||||
contrastThreshold(_contrastThreshold), edgeThreshold(_edgeThreshold), sigma(_sigma), descriptor_type(_descriptorType)
|
||||
{
|
||||
}
|
||||
|
||||
@@ -442,7 +454,7 @@ int SIFT_Impl::descriptorSize() const
|
||||
|
||||
int SIFT_Impl::descriptorType() const
|
||||
{
|
||||
return CV_32F;
|
||||
return descriptor_type;
|
||||
}
|
||||
|
||||
int SIFT_Impl::defaultNorm() const
|
||||
@@ -533,9 +545,9 @@ void SIFT_Impl::detectAndCompute(InputArray _image, InputArray _mask,
|
||||
{
|
||||
//t = (double)getTickCount();
|
||||
int dsize = descriptorSize();
|
||||
_descriptors.create((int)keypoints.size(), dsize, CV_32F);
|
||||
Mat descriptors = _descriptors.getMat();
|
||||
_descriptors.create((int)keypoints.size(), dsize, descriptor_type);
|
||||
|
||||
Mat descriptors = _descriptors.getMat();
|
||||
calcDescriptors(gpyr, keypoints, descriptors, nOctaveLayers, firstOctave);
|
||||
//t = (double)getTickCount() - t;
|
||||
//printf("descriptor extraction time: %g\n", t*1000./tf);
|
||||
|
||||
@@ -150,7 +150,7 @@ void findScaleSpaceExtrema(
|
||||
|
||||
void calcSIFTDescriptor(
|
||||
const Mat& img, Point2f ptf, float ori, float scl,
|
||||
int d, int n, float* dst
|
||||
int d, int n, Mat& dst, int row
|
||||
);
|
||||
|
||||
|
||||
@@ -555,7 +555,7 @@ void findScaleSpaceExtrema(
|
||||
|
||||
void calcSIFTDescriptor(
|
||||
const Mat& img, Point2f ptf, float ori, float scl,
|
||||
int d, int n, float* dst
|
||||
int d, int n, Mat& dstMat, int row
|
||||
)
|
||||
{
|
||||
CV_TRACE_FUNCTION();
|
||||
@@ -575,9 +575,18 @@ void calcSIFTDescriptor(
|
||||
int i, j, k, len = (radius*2+1)*(radius*2+1), histlen = (d+2)*(d+2)*(n+2);
|
||||
int rows = img.rows, cols = img.cols;
|
||||
|
||||
AutoBuffer<float> buf(len*6 + histlen);
|
||||
float *X = buf.data(), *Y = X + len, *Mag = Y, *Ori = Mag + len, *W = Ori + len;
|
||||
float *RBin = W + len, *CBin = RBin + len, *hist = CBin + len;
|
||||
cv::utils::BufferArea area;
|
||||
float *X = 0, *Y = 0, *Mag, *Ori = 0, *W = 0, *RBin = 0, *CBin = 0, *hist = 0, *rawDst = 0;
|
||||
area.allocate(X, len, CV_SIMD_WIDTH);
|
||||
area.allocate(Y, len, CV_SIMD_WIDTH);
|
||||
area.allocate(Ori, len, CV_SIMD_WIDTH);
|
||||
area.allocate(W, len, CV_SIMD_WIDTH);
|
||||
area.allocate(RBin, len, CV_SIMD_WIDTH);
|
||||
area.allocate(CBin, len, CV_SIMD_WIDTH);
|
||||
area.allocate(hist, histlen, CV_SIMD_WIDTH);
|
||||
area.allocate(rawDst, len, CV_SIMD_WIDTH);
|
||||
area.commit();
|
||||
Mag = Y;
|
||||
|
||||
for( i = 0; i < d+2; i++ )
|
||||
{
|
||||
@@ -628,10 +637,10 @@ void calcSIFTDescriptor(
|
||||
const v_int32 __n_plus_2 = vx_setall_s32(n+2);
|
||||
for( ; k <= len - vecsize; k += vecsize )
|
||||
{
|
||||
v_float32 rbin = vx_load(RBin + k);
|
||||
v_float32 cbin = vx_load(CBin + k);
|
||||
v_float32 obin = (vx_load(Ori + k) - __ori) * __bins_per_rad;
|
||||
v_float32 mag = vx_load(Mag + k) * vx_load(W + k);
|
||||
v_float32 rbin = vx_load_aligned(RBin + k);
|
||||
v_float32 cbin = vx_load_aligned(CBin + k);
|
||||
v_float32 obin = (vx_load_aligned(Ori + k) - __ori) * __bins_per_rad;
|
||||
v_float32 mag = vx_load_aligned(Mag + k) * vx_load_aligned(W + k);
|
||||
|
||||
v_int32 r0 = v_floor(rbin);
|
||||
v_int32 c0 = v_floor(cbin);
|
||||
@@ -723,7 +732,7 @@ void calcSIFTDescriptor(
|
||||
hist[idx] += hist[idx+n];
|
||||
hist[idx+1] += hist[idx+n+1];
|
||||
for( k = 0; k < n; k++ )
|
||||
dst[(i*d + j)*n + k] = hist[idx+k];
|
||||
rawDst[(i*d + j)*n + k] = hist[idx+k];
|
||||
}
|
||||
// copy histogram to the descriptor,
|
||||
// apply hysteresis thresholding
|
||||
@@ -735,17 +744,17 @@ void calcSIFTDescriptor(
|
||||
#if CV_SIMD
|
||||
{
|
||||
v_float32 __nrm2 = vx_setzero_f32();
|
||||
v_float32 __dst;
|
||||
v_float32 __rawDst;
|
||||
for( ; k <= len - v_float32::nlanes; k += v_float32::nlanes )
|
||||
{
|
||||
__dst = vx_load(dst + k);
|
||||
__nrm2 = v_fma(__dst, __dst, __nrm2);
|
||||
__rawDst = vx_load_aligned(rawDst + k);
|
||||
__nrm2 = v_fma(__rawDst, __rawDst, __nrm2);
|
||||
}
|
||||
nrm2 = (float)v_reduce_sum(__nrm2);
|
||||
}
|
||||
#endif
|
||||
for( ; k < len; k++ )
|
||||
nrm2 += dst[k]*dst[k];
|
||||
nrm2 += rawDst[k]*rawDst[k];
|
||||
|
||||
float thr = std::sqrt(nrm2)*SIFT_DESCR_MAG_THR;
|
||||
|
||||
@@ -760,9 +769,9 @@ void calcSIFTDescriptor(
|
||||
__m256 __thr = _mm256_set1_ps(thr);
|
||||
for( ; i <= len - 8; i += 8 )
|
||||
{
|
||||
__dst = _mm256_loadu_ps(&dst[i]);
|
||||
__dst = _mm256_loadu_ps(&rawDst[i]);
|
||||
__dst = _mm256_min_ps(__dst, __thr);
|
||||
_mm256_storeu_ps(&dst[i], __dst);
|
||||
_mm256_storeu_ps(&rawDst[i], __dst);
|
||||
#if CV_FMA3
|
||||
__nrm2 = _mm256_fmadd_ps(__dst, __dst, __nrm2);
|
||||
#else
|
||||
@@ -776,44 +785,78 @@ void calcSIFTDescriptor(
|
||||
#endif
|
||||
for( ; i < len; i++ )
|
||||
{
|
||||
float val = std::min(dst[i], thr);
|
||||
dst[i] = val;
|
||||
float val = std::min(rawDst[i], thr);
|
||||
rawDst[i] = val;
|
||||
nrm2 += val*val;
|
||||
}
|
||||
nrm2 = SIFT_INT_DESCR_FCTR/std::max(std::sqrt(nrm2), FLT_EPSILON);
|
||||
|
||||
#if 1
|
||||
k = 0;
|
||||
if( dstMat.type() == CV_32F )
|
||||
{
|
||||
float* dst = dstMat.ptr<float>(row);
|
||||
#if CV_SIMD
|
||||
v_float32 __dst;
|
||||
v_float32 __min = vx_setzero_f32();
|
||||
v_float32 __max = vx_setall_f32(255.0f); // max of uchar
|
||||
v_float32 __nrm2 = vx_setall_f32(nrm2);
|
||||
for( k = 0; k <= len - v_float32::nlanes; k += v_float32::nlanes )
|
||||
{
|
||||
v_float32 __dst;
|
||||
v_float32 __min = vx_setzero_f32();
|
||||
v_float32 __max = vx_setall_f32(255.0f); // max of uchar
|
||||
v_float32 __nrm2 = vx_setall_f32(nrm2);
|
||||
for( k = 0; k <= len - v_float32::nlanes; k += v_float32::nlanes )
|
||||
{
|
||||
__dst = vx_load(dst + k);
|
||||
__dst = v_min(v_max(v_cvt_f32(v_round(__dst * __nrm2)), __min), __max);
|
||||
v_store(dst + k, __dst);
|
||||
}
|
||||
__dst = vx_load_aligned(rawDst + k);
|
||||
__dst = v_min(v_max(v_cvt_f32(v_round(__dst * __nrm2)), __min), __max);
|
||||
v_store(dst + k, __dst);
|
||||
}
|
||||
#endif
|
||||
for( ; k < len; k++ )
|
||||
{
|
||||
dst[k] = saturate_cast<uchar>(dst[k]*nrm2);
|
||||
dst[k] = saturate_cast<uchar>(rawDst[k]*nrm2);
|
||||
}
|
||||
}
|
||||
else // CV_8U
|
||||
{
|
||||
uint8_t* dst = dstMat.ptr<uint8_t>(row);
|
||||
#if CV_SIMD
|
||||
v_float32 __dst0, __dst1;
|
||||
v_uint16 __pack01;
|
||||
v_float32 __nrm2 = vx_setall_f32(nrm2);
|
||||
for( k = 0; k <= len - v_float32::nlanes * 2; k += v_float32::nlanes * 2 )
|
||||
{
|
||||
__dst0 = vx_load_aligned(rawDst + k);
|
||||
__dst1 = vx_load_aligned(rawDst + k + v_float32::nlanes);
|
||||
|
||||
__pack01 = v_pack_u(v_round(__dst0 * __nrm2), v_round(__dst1 * __nrm2));
|
||||
v_pack_store(dst + k, __pack01);
|
||||
}
|
||||
#endif
|
||||
for( ; k < len; k++ )
|
||||
{
|
||||
dst[k] = saturate_cast<uchar>(rawDst[k]*nrm2);
|
||||
}
|
||||
}
|
||||
#else
|
||||
float* dst = dstMat.ptr<float>(row);
|
||||
float nrm1 = 0;
|
||||
for( k = 0; k < len; k++ )
|
||||
{
|
||||
dst[k] *= nrm2;
|
||||
nrm1 += dst[k];
|
||||
rawDst[k] *= nrm2;
|
||||
nrm1 += rawDst[k];
|
||||
}
|
||||
nrm1 = 1.f/std::max(nrm1, FLT_EPSILON);
|
||||
if( dstMat.type() == CV_32F )
|
||||
{
|
||||
for( k = 0; k < len; k++ )
|
||||
{
|
||||
dst[k] = std::sqrt(dst[k] * nrm1);//saturate_cast<uchar>(std::sqrt(dst[k] * nrm1)*SIFT_INT_DESCR_FCTR);
|
||||
dst[k] = std::sqrt(rawDst[k] * nrm1);
|
||||
}
|
||||
}
|
||||
else // CV_8U
|
||||
{
|
||||
for( k = 0; k < len; k++ )
|
||||
{
|
||||
dst[k] = saturate_cast<uchar>(std::sqrt(rawDst[k] * nrm1)*SIFT_INT_DESCR_FCTR);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#include "test_precomp.hpp"
|
||||
|
||||
// #define GENERATE_DATA // generate data in debug mode
|
||||
|
||||
namespace opencv_test { namespace {
|
||||
|
||||
#ifndef GENERATE_DATA
|
||||
static bool isSimilarKeypoints( const KeyPoint& p1, const KeyPoint& p2 )
|
||||
{
|
||||
const float maxPtDif = 1.f;
|
||||
const float maxSizeDif = 1.f;
|
||||
const float maxAngleDif = 2.f;
|
||||
const float maxResponseDif = 0.1f;
|
||||
|
||||
float dist = (float)cv::norm( p1.pt - p2.pt );
|
||||
return (dist < maxPtDif &&
|
||||
fabs(p1.size - p2.size) < maxSizeDif &&
|
||||
abs(p1.angle - p2.angle) < maxAngleDif &&
|
||||
abs(p1.response - p2.response) < maxResponseDif &&
|
||||
(p1.octave & 0xffff) == (p2.octave & 0xffff) // do not care about sublayers and class_id
|
||||
);
|
||||
}
|
||||
#endif
|
||||
|
||||
TEST(Features2d_AFFINE_FEATURE, regression)
|
||||
{
|
||||
Mat image = imread(cvtest::findDataFile("features2d/tsukuba.png"));
|
||||
string xml = cvtest::TS::ptr()->get_data_path() + "asift/regression_cpp.xml.gz";
|
||||
ASSERT_FALSE(image.empty());
|
||||
|
||||
Mat gray;
|
||||
cvtColor(image, gray, COLOR_BGR2GRAY);
|
||||
|
||||
// Default ASIFT generates too large descriptors. This test uses small maxTilt to suppress the size of testdata.
|
||||
Ptr<AffineFeature> ext = AffineFeature::create(SIFT::create(), 2, 0, 1.4142135623730951f, 144.0f);
|
||||
Mat mpt, msize, mangle, mresponse, moctave, mclass_id;
|
||||
#ifdef GENERATE_DATA
|
||||
// calculate
|
||||
vector<KeyPoint> calcKeypoints;
|
||||
Mat calcDescriptors;
|
||||
ext->detectAndCompute(gray, Mat(), calcKeypoints, calcDescriptors, false);
|
||||
|
||||
// create keypoints XML
|
||||
FileStorage fs(xml, FileStorage::WRITE);
|
||||
ASSERT_TRUE(fs.isOpened()) << xml;
|
||||
std::cout << "Creating keypoints XML..." << std::endl;
|
||||
|
||||
mpt = Mat(calcKeypoints.size(), 2, CV_32F);
|
||||
msize = Mat(calcKeypoints.size(), 1, CV_32F);
|
||||
mangle = Mat(calcKeypoints.size(), 1, CV_32F);
|
||||
mresponse = Mat(calcKeypoints.size(), 1, CV_32F);
|
||||
moctave = Mat(calcKeypoints.size(), 1, CV_32S);
|
||||
mclass_id = Mat(calcKeypoints.size(), 1, CV_32S);
|
||||
|
||||
for( size_t i = 0; i < calcKeypoints.size(); i++ )
|
||||
{
|
||||
const KeyPoint& key = calcKeypoints[i];
|
||||
mpt.at<float>(i, 0) = key.pt.x;
|
||||
mpt.at<float>(i, 1) = key.pt.y;
|
||||
msize.at<float>(i, 0) = key.size;
|
||||
mangle.at<float>(i, 0) = key.angle;
|
||||
mresponse.at<float>(i, 0) = key.response;
|
||||
moctave.at<int>(i, 0) = key.octave;
|
||||
mclass_id.at<int>(i, 0) = key.class_id;
|
||||
}
|
||||
|
||||
fs << "keypoints_pt" << mpt;
|
||||
fs << "keypoints_size" << msize;
|
||||
fs << "keypoints_angle" << mangle;
|
||||
fs << "keypoints_response" << mresponse;
|
||||
fs << "keypoints_octave" << moctave;
|
||||
fs << "keypoints_class_id" << mclass_id;
|
||||
|
||||
// create descriptor XML
|
||||
fs << "descriptors" << calcDescriptors;
|
||||
fs.release();
|
||||
#else
|
||||
const float badCountsRatio = 0.01f;
|
||||
const float badDescriptorDist = 1.0f;
|
||||
const float maxBadKeypointsRatio = 0.15f;
|
||||
const float maxBadDescriptorRatio = 0.15f;
|
||||
|
||||
// read keypoints
|
||||
vector<KeyPoint> validKeypoints;
|
||||
Mat validDescriptors;
|
||||
FileStorage fs(xml, FileStorage::READ);
|
||||
ASSERT_TRUE(fs.isOpened()) << xml;
|
||||
|
||||
fs["keypoints_pt"] >> mpt;
|
||||
ASSERT_EQ(mpt.type(), CV_32F);
|
||||
fs["keypoints_size"] >> msize;
|
||||
ASSERT_EQ(msize.type(), CV_32F);
|
||||
fs["keypoints_angle"] >> mangle;
|
||||
ASSERT_EQ(mangle.type(), CV_32F);
|
||||
fs["keypoints_response"] >> mresponse;
|
||||
ASSERT_EQ(mresponse.type(), CV_32F);
|
||||
fs["keypoints_octave"] >> moctave;
|
||||
ASSERT_EQ(moctave.type(), CV_32S);
|
||||
fs["keypoints_class_id"] >> mclass_id;
|
||||
ASSERT_EQ(mclass_id.type(), CV_32S);
|
||||
|
||||
validKeypoints.resize(mpt.rows);
|
||||
for( int i = 0; i < (int)validKeypoints.size(); i++ )
|
||||
{
|
||||
validKeypoints[i].pt.x = mpt.at<float>(i, 0);
|
||||
validKeypoints[i].pt.y = mpt.at<float>(i, 1);
|
||||
validKeypoints[i].size = msize.at<float>(i, 0);
|
||||
validKeypoints[i].angle = mangle.at<float>(i, 0);
|
||||
validKeypoints[i].response = mresponse.at<float>(i, 0);
|
||||
validKeypoints[i].octave = moctave.at<int>(i, 0);
|
||||
validKeypoints[i].class_id = mclass_id.at<int>(i, 0);
|
||||
}
|
||||
|
||||
// read descriptors
|
||||
fs["descriptors"] >> validDescriptors;
|
||||
fs.release();
|
||||
|
||||
// calc and compare keypoints
|
||||
vector<KeyPoint> calcKeypoints;
|
||||
ext->detectAndCompute(gray, Mat(), calcKeypoints, noArray(), false);
|
||||
|
||||
float countRatio = (float)validKeypoints.size() / (float)calcKeypoints.size();
|
||||
ASSERT_LT(countRatio, 1 + badCountsRatio) << "Bad keypoints count ratio.";
|
||||
ASSERT_GT(countRatio, 1 - badCountsRatio) << "Bad keypoints count ratio.";
|
||||
|
||||
int badPointCount = 0, commonPointCount = max((int)validKeypoints.size(), (int)calcKeypoints.size());
|
||||
for( size_t v = 0; v < validKeypoints.size(); v++ )
|
||||
{
|
||||
int nearestIdx = -1;
|
||||
float minDist = std::numeric_limits<float>::max();
|
||||
float angleDistOfNearest = std::numeric_limits<float>::max();
|
||||
|
||||
for( size_t c = 0; c < calcKeypoints.size(); c++ )
|
||||
{
|
||||
if( validKeypoints[v].class_id != calcKeypoints[c].class_id )
|
||||
continue;
|
||||
float curDist = (float)cv::norm( calcKeypoints[c].pt - validKeypoints[v].pt );
|
||||
if( curDist < minDist )
|
||||
{
|
||||
minDist = curDist;
|
||||
nearestIdx = (int)c;
|
||||
angleDistOfNearest = abs( calcKeypoints[c].angle - validKeypoints[v].angle );
|
||||
}
|
||||
else if( curDist == minDist ) // the keypoints whose positions are same but angles are different
|
||||
{
|
||||
float angleDist = abs( calcKeypoints[c].angle - validKeypoints[v].angle );
|
||||
if( angleDist < angleDistOfNearest )
|
||||
{
|
||||
nearestIdx = (int)c;
|
||||
angleDistOfNearest = angleDist;
|
||||
}
|
||||
}
|
||||
}
|
||||
if( nearestIdx == -1 || !isSimilarKeypoints( validKeypoints[v], calcKeypoints[nearestIdx] ) )
|
||||
badPointCount++;
|
||||
}
|
||||
float badKeypointsRatio = (float)badPointCount / (float)commonPointCount;
|
||||
std::cout << "badKeypointsRatio: " << badKeypointsRatio << std::endl;
|
||||
ASSERT_LT( badKeypointsRatio , maxBadKeypointsRatio ) << "Bad accuracy!";
|
||||
|
||||
// Calc and compare descriptors. This uses validKeypoints for extraction.
|
||||
Mat calcDescriptors;
|
||||
ext->detectAndCompute(gray, Mat(), validKeypoints, calcDescriptors, true);
|
||||
|
||||
int dim = validDescriptors.cols;
|
||||
int badDescriptorCount = 0;
|
||||
L1<float> distance;
|
||||
|
||||
for( int i = 0; i < (int)validKeypoints.size(); i++ )
|
||||
{
|
||||
float dist = distance( validDescriptors.ptr<float>(i), calcDescriptors.ptr<float>(i), dim );
|
||||
if( dist > badDescriptorDist )
|
||||
badDescriptorCount++;
|
||||
}
|
||||
float badDescriptorRatio = (float)badDescriptorCount / (float)validKeypoints.size();
|
||||
std::cout << "badDescriptorRatio: " << badDescriptorRatio << std::endl;
|
||||
ASSERT_LT( badDescriptorRatio, maxBadDescriptorRatio ) << "Too many descriptors mismatched.";
|
||||
#endif
|
||||
}
|
||||
|
||||
}} // namespace
|
||||
@@ -0,0 +1,34 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html
|
||||
|
||||
#include "test_precomp.hpp"
|
||||
|
||||
namespace opencv_test { namespace {
|
||||
|
||||
TEST(Features2d_SIFT, descriptor_type)
|
||||
{
|
||||
Mat image = imread(cvtest::findDataFile("features2d/tsukuba.png"));
|
||||
ASSERT_FALSE(image.empty());
|
||||
|
||||
Mat gray;
|
||||
cvtColor(image, gray, COLOR_BGR2GRAY);
|
||||
|
||||
vector<KeyPoint> keypoints;
|
||||
Mat descriptorsFloat, descriptorsUchar;
|
||||
Ptr<SIFT> siftFloat = cv::SIFT::create(0, 3, 0.04, 10, 1.6, CV_32F);
|
||||
siftFloat->detectAndCompute(gray, Mat(), keypoints, descriptorsFloat, false);
|
||||
ASSERT_EQ(descriptorsFloat.type(), CV_32F) << "type mismatch";
|
||||
|
||||
Ptr<SIFT> siftUchar = cv::SIFT::create(0, 3, 0.04, 10, 1.6, CV_8U);
|
||||
siftUchar->detectAndCompute(gray, Mat(), keypoints, descriptorsUchar, false);
|
||||
ASSERT_EQ(descriptorsUchar.type(), CV_8U) << "type mismatch";
|
||||
|
||||
Mat descriptorsFloat2;
|
||||
descriptorsUchar.assignTo(descriptorsFloat2, CV_32F);
|
||||
Mat diff = descriptorsFloat != descriptorsFloat2;
|
||||
ASSERT_EQ(countNonZero(diff), 0) << "descriptors are not identical";
|
||||
}
|
||||
|
||||
|
||||
}} // namespace
|
||||
@@ -95,6 +95,8 @@ using ::cvflann::MaxDistance;
|
||||
using ::cvflann::HammingLUT;
|
||||
using ::cvflann::Hamming;
|
||||
using ::cvflann::Hamming2;
|
||||
using ::cvflann::DNAmmingLUT;
|
||||
using ::cvflann::DNAmming2;
|
||||
using ::cvflann::HistIntersectionDistance;
|
||||
using ::cvflann::HellingerDistance;
|
||||
using ::cvflann::ChiSquareDistance;
|
||||
@@ -131,6 +133,14 @@ performed using library calls, if available. Lookup table implementation is used
|
||||
cv::flann::Hamming2 - %Hamming distance functor. Population count is
|
||||
implemented in 12 arithmetic operations (one of which is multiplication).
|
||||
|
||||
cv::flann::DNAmmingLUT - %Adaptation of the Hamming distance functor to DNA comparison.
|
||||
As the four bases A, C, G, T of the DNA (or A, G, C, U for RNA) can be coded on 2 bits,
|
||||
it counts the bits pairs differences between two sequences using a lookup table implementation.
|
||||
|
||||
cv::flann::DNAmming2 - %Adaptation of the Hamming distance functor to DNA comparison.
|
||||
Bases differences count are vectorised thanks to arithmetic operations using standard
|
||||
registers (AVX2 and AVX-512 should come in a near future).
|
||||
|
||||
cv::flann::HistIntersectionDistance - The histogram
|
||||
intersection distance functor.
|
||||
|
||||
@@ -191,8 +201,28 @@ public:
|
||||
KDTreeIndexParams( int trees = 4 );
|
||||
};
|
||||
@endcode
|
||||
- **HierarchicalClusteringIndexParams** When passing an object of this type the index constructed
|
||||
will be a hierarchical tree of clusters, dividing each set of points into n clusters whose centers
|
||||
are picked among the points without further refinement of their position.
|
||||
This algorithm fits both floating, integer and binary vectors. :
|
||||
@code
|
||||
struct HierarchicalClusteringIndexParams : public IndexParams
|
||||
{
|
||||
HierarchicalClusteringIndexParams(
|
||||
int branching = 32,
|
||||
flann_centers_init_t centers_init = CENTERS_RANDOM,
|
||||
int trees = 4,
|
||||
int leaf_size = 100);
|
||||
|
||||
};
|
||||
@endcode
|
||||
- **KMeansIndexParams** When passing an object of this type the index constructed will be a
|
||||
hierarchical k-means tree. :
|
||||
hierarchical k-means tree (one tree by default), dividing each set of points into n clusters
|
||||
whose barycenters are refined iteratively.
|
||||
Note that this algorithm has been extended to the support of binary vectors as an alternative
|
||||
to LSH when knn search speed is the criterium. It will also outperform LSH when processing
|
||||
directly (i.e. without the use of MCA/PCA) datasets whose points share mostly the same values
|
||||
for most of the dimensions. It is recommended to set more than one tree with binary data. :
|
||||
@code
|
||||
struct KMeansIndexParams : public IndexParams
|
||||
{
|
||||
@@ -201,6 +231,13 @@ public:
|
||||
int iterations = 11,
|
||||
flann_centers_init_t centers_init = CENTERS_RANDOM,
|
||||
float cb_index = 0.2 );
|
||||
|
||||
KMeansIndexParams(
|
||||
int branching,
|
||||
int iterations,
|
||||
flann_centers_init_t centers_init,
|
||||
float cb_index,
|
||||
int trees );
|
||||
};
|
||||
@endcode
|
||||
- **CompositeIndexParams** When using a parameters object of this type the index created
|
||||
@@ -219,7 +256,8 @@ public:
|
||||
- **LshIndexParams** When using a parameters object of this type the index created uses
|
||||
multi-probe LSH (by Multi-Probe LSH: Efficient Indexing for High-Dimensional Similarity Search
|
||||
by Qin Lv, William Josephson, Zhe Wang, Moses Charikar, Kai Li., Proceedings of the 33rd
|
||||
International Conference on Very Large Data Bases (VLDB). Vienna, Austria. September 2007) :
|
||||
International Conference on Very Large Data Bases (VLDB). Vienna, Austria. September 2007).
|
||||
This algorithm is designed for binary vectors. :
|
||||
@code
|
||||
struct LshIndexParams : public IndexParams
|
||||
{
|
||||
|
||||