mirror of
https://github.com/opencv/opencv.git
synced 2026-07-31 00:03:03 +04:00
Compare commits
190 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| b1cf550123 | |||
| 17bd9a1fa1 | |||
| 0321644bbd | |||
| 81e7988eb9 | |||
| 4985311d46 | |||
| 05348f3250 | |||
| 003609e565 | |||
| 9537a909f7 | |||
| 8c2dd5fb9a | |||
| 27545dcc86 | |||
| 724e04e979 | |||
| dfc94c58f0 | |||
| fac895d7ba | |||
| 04b40ff221 | |||
| a3d7811f24 | |||
| 755e0143fb | |||
| dfa48094dc | |||
| e585192eeb | |||
| 646924fce8 | |||
| c63aa7f085 | |||
| c54abde1bd | |||
| 95c1d2a887 | |||
| ebef84e9ea | |||
| b1a772d194 | |||
| f11f2bfb56 | |||
| 24f43e7ae9 | |||
| d94d469c86 | |||
| d20c9bde7e | |||
| 62414e3073 | |||
| 48c985e775 | |||
| 7358fffb0f | |||
| de5b6386e0 | |||
| 1de2d5c2b6 | |||
| ef53a9229f | |||
| 259c39a63a | |||
| 327b98eb13 | |||
| f977d10a19 | |||
| a0cf8c322d | |||
| 1e74f5850b | |||
| 9b76872708 | |||
| 4d587c341b | |||
| 846317ef37 | |||
| 7e62789edf | |||
| 852663f6d2 | |||
| a9b30984a3 | |||
| 2f83c3b689 | |||
| b25ad12f1a | |||
| d95e43a6a1 | |||
| 2cc14bd0fb | |||
| fdc8ed8d05 | |||
| 236c64a17d | |||
| f96569da1e | |||
| 91ff45fbde | |||
| 45aabc5d0d | |||
| f9bd83c854 | |||
| 86a51015b1 | |||
| 998406d20e | |||
| f7b4b750d8 | |||
| a4e2c56317 | |||
| 9c5d7716e2 | |||
| 46fd26e366 | |||
| c410d7a97d | |||
| 6fa63dcc0c | |||
| 96f25332ea | |||
| 3385d38648 | |||
| 1618c963e4 | |||
| 1e3be09b3b | |||
| 6e66a9222a | |||
| 9d1e8b1e1d | |||
| f605373a2b | |||
| 56b7622612 | |||
| 696a6ccd57 | |||
| 51b03b87e6 | |||
| aa7ba0bc1a | |||
| 07e4076585 | |||
| de1a459879 | |||
| 1aacb9bb15 | |||
| 9b4ecc96f6 | |||
| e3f4f874c5 | |||
| 6ace801418 | |||
| d31b93b513 | |||
| ac0fd6aa9a | |||
| 068f33cfdf | |||
| 4807cd8a6e | |||
| 35e824c287 | |||
| 1e0d290f2e | |||
| 0097a8d097 | |||
| 36cc43170d | |||
| 5578ad5e14 | |||
| d11f0a709d | |||
| 0a43b23275 | |||
| 7967683296 | |||
| 5b2c016834 | |||
| aaff125608 | |||
| 407adc7061 | |||
| d0e612dc36 | |||
| 7c23ec90a9 | |||
| 390957fec4 | |||
| 060a76dc3e | |||
| 6625810d2a | |||
| edc442afdb | |||
| 16b9514543 | |||
| 95c7f4a7f0 | |||
| ae6fabc6fe | |||
| 7eaadf616c | |||
| 8fed5fc5ae | |||
| f25951c412 | |||
| 1259a474ba | |||
| 3995deaf76 | |||
| 076587425e | |||
| da6aeaca46 | |||
| 8ee33ca551 | |||
| ea7f13922b | |||
| 38d0063c36 | |||
| df83459721 | |||
| 54a9e00970 | |||
| 9bda96d39e | |||
| 77a5c43d50 | |||
| f28e4b86fb | |||
| b675e6ab77 | |||
| d6306f8ccb | |||
| a9817e9127 | |||
| c08897cd10 | |||
| 384875f4fc | |||
| 9cfa84313c | |||
| fe625a558e | |||
| 9ef41f68fb | |||
| cfb36443fb | |||
| 9d3826c676 | |||
| 4300bb2e1f | |||
| 6edc438789 | |||
| 25cd7c7c50 | |||
| 732cb6c45c | |||
| 266a868ba9 | |||
| 8199967b31 | |||
| 9d61c18143 | |||
| 221bfa4c67 | |||
| 1a8b7f7513 | |||
| 992b47b991 | |||
| 739ff84732 | |||
| 2a177052de | |||
| f7c82baee9 | |||
| 2b34e0abdc | |||
| 633fedaa96 | |||
| 65134c793b | |||
| d5f34cf34c | |||
| cefa602601 | |||
| d773691848 | |||
| f40707dc68 | |||
| 2d8ce500fa | |||
| ddf1b04cce | |||
| ad5b3a4753 | |||
| 531ea5b3a2 | |||
| b468468e7e | |||
| cff0168f3a | |||
| d83901e665 | |||
| 1e1984a586 | |||
| 06dcc5a2c6 | |||
| 000f762fb9 | |||
| 3bc3eeb5da | |||
| 4e5699fa71 | |||
| a76274b549 | |||
| 803ff8ebb9 | |||
| b42152ffeb | |||
| 024b43ca06 | |||
| 9448fe3db4 | |||
| 3817f3a89b | |||
| 2062a7ca8f | |||
| 8334ee18e6 | |||
| cc2592f582 | |||
| 96d35f7c54 | |||
| 98c5fc6ae2 | |||
| fbde0c6c96 | |||
| 602e7c83e2 | |||
| eb9218a86b | |||
| 9f2dcc3f13 | |||
| bc210b292b | |||
| 2113af9c52 | |||
| 6f417b57c1 | |||
| 0acd06d13c | |||
| fd22e98298 | |||
| 167a12028d | |||
| 6af2faebd2 | |||
| fd16222613 | |||
| 1ba6cd8423 | |||
| 4788de784b | |||
| 926535469d | |||
| 5627a0cbdf | |||
| 9a9954a036 | |||
| 591708903b |
Vendored
+22
-17
@@ -1778,30 +1778,30 @@ TegraCvtColor_Invoker(bgrx2hsvf, bgrx2hsv, src_data + static_cast<size_t>(range.
|
|||||||
: CV_HAL_ERROR_NOT_IMPLEMENTED \
|
: CV_HAL_ERROR_NOT_IMPLEMENTED \
|
||||||
)
|
)
|
||||||
|
|
||||||
#define TEGRA_CVT2PYUVTOBGR(src_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx) \
|
#define TEGRA_CVT2PYUVTOBGR_EX(y_data, y_step, uv_data, uv_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx) \
|
||||||
( \
|
( \
|
||||||
CAROTENE_NS::isSupportedConfiguration() ? \
|
CAROTENE_NS::isSupportedConfiguration() ? \
|
||||||
dcn == 3 ? \
|
dcn == 3 ? \
|
||||||
uIdx == 0 ? \
|
uIdx == 0 ? \
|
||||||
(swapBlue ? \
|
(swapBlue ? \
|
||||||
CAROTENE_NS::yuv420i2rgb(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
CAROTENE_NS::yuv420i2rgb(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||||
src_data, src_step, \
|
y_data, y_step, \
|
||||||
src_data + src_step * dst_height, src_step, \
|
uv_data, uv_step, \
|
||||||
dst_data, dst_step) : \
|
dst_data, dst_step) : \
|
||||||
CAROTENE_NS::yuv420i2bgr(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
CAROTENE_NS::yuv420i2bgr(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||||
src_data, src_step, \
|
y_data, y_step, \
|
||||||
src_data + src_step * dst_height, src_step, \
|
uv_data, uv_step, \
|
||||||
dst_data, dst_step)), \
|
dst_data, dst_step)), \
|
||||||
CV_HAL_ERROR_OK : \
|
CV_HAL_ERROR_OK : \
|
||||||
uIdx == 1 ? \
|
uIdx == 1 ? \
|
||||||
(swapBlue ? \
|
(swapBlue ? \
|
||||||
CAROTENE_NS::yuv420sp2rgb(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
CAROTENE_NS::yuv420sp2rgb(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||||
src_data, src_step, \
|
y_data, y_step, \
|
||||||
src_data + src_step * dst_height, src_step, \
|
uv_data, uv_step, \
|
||||||
dst_data, dst_step) : \
|
dst_data, dst_step) : \
|
||||||
CAROTENE_NS::yuv420sp2bgr(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
CAROTENE_NS::yuv420sp2bgr(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||||
src_data, src_step, \
|
y_data, y_step, \
|
||||||
src_data + src_step * dst_height, src_step, \
|
uv_data, uv_step, \
|
||||||
dst_data, dst_step)), \
|
dst_data, dst_step)), \
|
||||||
CV_HAL_ERROR_OK : \
|
CV_HAL_ERROR_OK : \
|
||||||
CV_HAL_ERROR_NOT_IMPLEMENTED : \
|
CV_HAL_ERROR_NOT_IMPLEMENTED : \
|
||||||
@@ -1809,29 +1809,32 @@ TegraCvtColor_Invoker(bgrx2hsvf, bgrx2hsv, src_data + static_cast<size_t>(range.
|
|||||||
uIdx == 0 ? \
|
uIdx == 0 ? \
|
||||||
(swapBlue ? \
|
(swapBlue ? \
|
||||||
CAROTENE_NS::yuv420i2rgbx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
CAROTENE_NS::yuv420i2rgbx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||||
src_data, src_step, \
|
y_data, y_step, \
|
||||||
src_data + src_step * dst_height, src_step, \
|
uv_data, uv_step, \
|
||||||
dst_data, dst_step) : \
|
dst_data, dst_step) : \
|
||||||
CAROTENE_NS::yuv420i2bgrx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
CAROTENE_NS::yuv420i2bgrx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||||
src_data, src_step, \
|
y_data, y_step, \
|
||||||
src_data + src_step * dst_height, src_step, \
|
uv_data, uv_step, \
|
||||||
dst_data, dst_step)), \
|
dst_data, dst_step)), \
|
||||||
CV_HAL_ERROR_OK : \
|
CV_HAL_ERROR_OK : \
|
||||||
uIdx == 1 ? \
|
uIdx == 1 ? \
|
||||||
(swapBlue ? \
|
(swapBlue ? \
|
||||||
CAROTENE_NS::yuv420sp2rgbx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
CAROTENE_NS::yuv420sp2rgbx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||||
src_data, src_step, \
|
y_data, y_step, \
|
||||||
src_data + src_step * dst_height, src_step, \
|
uv_data, uv_step, \
|
||||||
dst_data, dst_step) : \
|
dst_data, dst_step) : \
|
||||||
CAROTENE_NS::yuv420sp2bgrx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
CAROTENE_NS::yuv420sp2bgrx(CAROTENE_NS::Size2D(dst_width, dst_height), \
|
||||||
src_data, src_step, \
|
y_data, y_step, \
|
||||||
src_data + src_step * dst_height, src_step, \
|
uv_data, uv_step, \
|
||||||
dst_data, dst_step)), \
|
dst_data, dst_step)), \
|
||||||
CV_HAL_ERROR_OK : \
|
CV_HAL_ERROR_OK : \
|
||||||
CV_HAL_ERROR_NOT_IMPLEMENTED : \
|
CV_HAL_ERROR_NOT_IMPLEMENTED : \
|
||||||
CV_HAL_ERROR_NOT_IMPLEMENTED \
|
CV_HAL_ERROR_NOT_IMPLEMENTED \
|
||||||
: CV_HAL_ERROR_NOT_IMPLEMENTED \
|
: CV_HAL_ERROR_NOT_IMPLEMENTED \
|
||||||
)
|
)
|
||||||
|
#define TEGRA_CVT2PYUVTOBGR(src_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx) \
|
||||||
|
TEGRA_CVT2PYUVTOBGR_EX(src_data, src_step, src_data + src_step * dst_height, src_step, dst_data, dst_step, \
|
||||||
|
dst_width, dst_height, dcn, swapBlue, uIdx);
|
||||||
|
|
||||||
#undef cv_hal_cvtBGRtoBGR
|
#undef cv_hal_cvtBGRtoBGR
|
||||||
#define cv_hal_cvtBGRtoBGR TEGRA_CVTBGRTOBGR
|
#define cv_hal_cvtBGRtoBGR TEGRA_CVTBGRTOBGR
|
||||||
@@ -1847,6 +1850,8 @@ TegraCvtColor_Invoker(bgrx2hsvf, bgrx2hsv, src_data + static_cast<size_t>(range.
|
|||||||
#define cv_hal_cvtBGRtoHSV TEGRA_CVTBGRTOHSV
|
#define cv_hal_cvtBGRtoHSV TEGRA_CVTBGRTOHSV
|
||||||
#undef cv_hal_cvtTwoPlaneYUVtoBGR
|
#undef cv_hal_cvtTwoPlaneYUVtoBGR
|
||||||
#define cv_hal_cvtTwoPlaneYUVtoBGR TEGRA_CVT2PYUVTOBGR
|
#define cv_hal_cvtTwoPlaneYUVtoBGR TEGRA_CVT2PYUVTOBGR
|
||||||
|
#undef cv_hal_cvtTwoPlaneYUVtoBGREx
|
||||||
|
#define cv_hal_cvtTwoPlaneYUVtoBGREx TEGRA_CVT2PYUVTOBGR_EX
|
||||||
|
|
||||||
#endif // OPENCV_IMGPROC_HAL_INTERFACE_H
|
#endif // OPENCV_IMGPROC_HAL_INTERFACE_H
|
||||||
|
|
||||||
|
|||||||
Vendored
+5
-5
@@ -1,8 +1,8 @@
|
|||||||
# Binaries branch name: ffmpeg/3.4_20210302
|
# Binaries branch name: ffmpeg/3.4_20211005
|
||||||
# Binaries were created for OpenCV: 2ab1f3f166fccc3a01497209cc01c5cea44ff201
|
# Binaries were created for OpenCV: 95c1d2a8872b222f32bd88db9f1efcbd9f70a9cf
|
||||||
ocv_update(FFMPEG_BINARIES_COMMIT "e99214251d9f3cde7c48abd46b2259bddc9885b6")
|
ocv_update(FFMPEG_BINARIES_COMMIT "0bf6c0753d435d2c82c03c48db0c6e18ac79976c")
|
||||||
ocv_update(FFMPEG_FILE_HASH_BIN32 "fad5ada9be36120bba8966709e7953a8")
|
ocv_update(FFMPEG_FILE_HASH_BIN32 "55c25bbc13e4a12d4339b70d3b76987f")
|
||||||
ocv_update(FFMPEG_FILE_HASH_BIN64 "650e2272728491923e566f784f79cfef")
|
ocv_update(FFMPEG_FILE_HASH_BIN64 "67caee9231c6843483b4de9815d6526e")
|
||||||
ocv_update(FFMPEG_FILE_HASH_CMAKE "3b90f67f4b429e77d3da36698cef700c")
|
ocv_update(FFMPEG_FILE_HASH_CMAKE "3b90f67f4b429e77d3da36698cef700c")
|
||||||
|
|
||||||
function(download_win_ffmpeg script_var)
|
function(download_win_ffmpeg script_var)
|
||||||
|
|||||||
Vendored
+9
-5
@@ -923,6 +923,11 @@ int ovx_hal_cvtGraytoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep,
|
|||||||
}
|
}
|
||||||
|
|
||||||
int ovx_hal_cvtTwoPlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int bcn, bool swapBlue, int uIdx)
|
int ovx_hal_cvtTwoPlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int bcn, bool swapBlue, int uIdx)
|
||||||
|
{
|
||||||
|
return ovx_hal_cvtTwoPlaneYUVtoBGREx(a, astep, a + h * astep, astep, b, bstep, w, h, bcn, swapBlue, uIdx);
|
||||||
|
}
|
||||||
|
|
||||||
|
int ovx_hal_cvtTwoPlaneYUVtoBGREx(const uchar * a, size_t astep, const uchar * b, size_t bstep, uchar * c, size_t cstep, int w, int h, int bcn, bool swapBlue, int uIdx)
|
||||||
{
|
{
|
||||||
if (skipSmallImages<VX_KERNEL_COLOR_CONVERT>(w, h))
|
if (skipSmallImages<VX_KERNEL_COLOR_CONVERT>(w, h))
|
||||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||||
@@ -933,8 +938,7 @@ int ovx_hal_cvtTwoPlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t
|
|||||||
|
|
||||||
if (w & 1 || h & 1) // It's not described in spec but sample implementation unable to convert odd sized images
|
if (w & 1 || h & 1) // It's not described in spec but sample implementation unable to convert odd sized images
|
||||||
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
return CV_HAL_ERROR_NOT_IMPLEMENTED;
|
||||||
refineStep(w, h, uIdx ? VX_DF_IMAGE_NV21 : VX_DF_IMAGE_NV12, astep);
|
|
||||||
refineStep(w, h, bcn == 3 ? VX_DF_IMAGE_RGB : VX_DF_IMAGE_RGBX, bstep);
|
|
||||||
try
|
try
|
||||||
{
|
{
|
||||||
ivx::Context ctx = getOpenVXHALContext();
|
ivx::Context ctx = getOpenVXHALContext();
|
||||||
@@ -943,8 +947,8 @@ int ovx_hal_cvtTwoPlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t
|
|||||||
std::vector<void *> ptrs;
|
std::vector<void *> ptrs;
|
||||||
addr.push_back(ivx::Image::createAddressing(w, h, 1, (vx_int32)astep));
|
addr.push_back(ivx::Image::createAddressing(w, h, 1, (vx_int32)astep));
|
||||||
ptrs.push_back((void*)a);
|
ptrs.push_back((void*)a);
|
||||||
addr.push_back(ivx::Image::createAddressing(w / 2, h / 2, 2, (vx_int32)astep));
|
addr.push_back(ivx::Image::createAddressing(w / 2, h / 2, 2, (vx_int32)bstep));
|
||||||
ptrs.push_back((void*)(a + h * astep));
|
ptrs.push_back((void*)b);
|
||||||
|
|
||||||
vxImage
|
vxImage
|
||||||
ia = ivx::Image::createFromHandle(ctx, uIdx ? VX_DF_IMAGE_NV21 : VX_DF_IMAGE_NV12, addr, ptrs);
|
ia = ivx::Image::createFromHandle(ctx, uIdx ? VX_DF_IMAGE_NV21 : VX_DF_IMAGE_NV12, addr, ptrs);
|
||||||
@@ -952,7 +956,7 @@ int ovx_hal_cvtTwoPlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t
|
|||||||
return CV_HAL_ERROR_NOT_IMPLEMENTED; // OpenCV store NV12/NV21 as RANGE_RESTRICTED while OpenVX expect RANGE_FULL
|
return CV_HAL_ERROR_NOT_IMPLEMENTED; // OpenCV store NV12/NV21 as RANGE_RESTRICTED while OpenVX expect RANGE_FULL
|
||||||
vxImage
|
vxImage
|
||||||
ib = ivx::Image::createFromHandle(ctx, bcn == 3 ? VX_DF_IMAGE_RGB : VX_DF_IMAGE_RGBX,
|
ib = ivx::Image::createFromHandle(ctx, bcn == 3 ? VX_DF_IMAGE_RGB : VX_DF_IMAGE_RGBX,
|
||||||
ivx::Image::createAddressing(w, h, bcn, (vx_int32)bstep), b);
|
ivx::Image::createAddressing(w, h, bcn, (vx_int32)cstep), c);
|
||||||
ivx::IVX_CHECK_STATUS(vxuColorConvert(ctx, ia, ib));
|
ivx::IVX_CHECK_STATUS(vxuColorConvert(ctx, ia, ib));
|
||||||
}
|
}
|
||||||
catch (ivx::RuntimeError & e)
|
catch (ivx::RuntimeError & e)
|
||||||
|
|||||||
Vendored
+3
@@ -49,6 +49,7 @@ int ovx_hal_morph(cvhalFilter2D *filter_context, uchar *a, size_t astep, uchar *
|
|||||||
int ovx_hal_cvtBGRtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int depth, int acn, int bcn, bool swapBlue);
|
int ovx_hal_cvtBGRtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int depth, int acn, int bcn, bool swapBlue);
|
||||||
int ovx_hal_cvtGraytoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int depth, int bcn);
|
int ovx_hal_cvtGraytoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int depth, int bcn);
|
||||||
int ovx_hal_cvtTwoPlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int bcn, bool swapBlue, int uIdx);
|
int ovx_hal_cvtTwoPlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int bcn, bool swapBlue, int uIdx);
|
||||||
|
int ovx_hal_cvtTwoPlaneYUVtoBGREx(const uchar * a, size_t astep, const uchar * b, size_t bstep, uchar * c, size_t cstep, int w, int h, int bcn, bool swapBlue, int uIdx);
|
||||||
int ovx_hal_cvtThreePlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int bcn, bool swapBlue, int uIdx);
|
int ovx_hal_cvtThreePlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int bcn, bool swapBlue, int uIdx);
|
||||||
int ovx_hal_cvtBGRtoThreePlaneYUV(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int acn, bool swapBlue, int uIdx);
|
int ovx_hal_cvtBGRtoThreePlaneYUV(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int acn, bool swapBlue, int uIdx);
|
||||||
int ovx_hal_cvtOnePlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int bcn, bool swapBlue, int uIdx, int ycn);
|
int ovx_hal_cvtOnePlaneYUVtoBGR(const uchar * a, size_t astep, uchar * b, size_t bstep, int w, int h, int bcn, bool swapBlue, int uIdx, int ycn);
|
||||||
@@ -130,6 +131,8 @@ int ovx_hal_integral(int depth, int sdepth, int, const uchar * a, size_t astep,
|
|||||||
#define cv_hal_cvtGraytoBGR ovx_hal_cvtGraytoBGR
|
#define cv_hal_cvtGraytoBGR ovx_hal_cvtGraytoBGR
|
||||||
#undef cv_hal_cvtTwoPlaneYUVtoBGR
|
#undef cv_hal_cvtTwoPlaneYUVtoBGR
|
||||||
#define cv_hal_cvtTwoPlaneYUVtoBGR ovx_hal_cvtTwoPlaneYUVtoBGR
|
#define cv_hal_cvtTwoPlaneYUVtoBGR ovx_hal_cvtTwoPlaneYUVtoBGR
|
||||||
|
#undef cv_hal_cvtTwoPlaneYUVtoBGREx
|
||||||
|
#define cv_hal_cvtTwoPlaneYUVtoBGREx ovx_hal_cvtTwoPlaneYUVtoBGREx
|
||||||
#undef cv_hal_cvtThreePlaneYUVtoBGR
|
#undef cv_hal_cvtThreePlaneYUVtoBGR
|
||||||
#define cv_hal_cvtThreePlaneYUVtoBGR ovx_hal_cvtThreePlaneYUVtoBGR
|
#define cv_hal_cvtThreePlaneYUVtoBGR ovx_hal_cvtThreePlaneYUVtoBGR
|
||||||
#undef cv_hal_cvtBGRtoThreePlaneYUV
|
#undef cv_hal_cvtBGRtoThreePlaneYUV
|
||||||
|
|||||||
Vendored
+4
-2
@@ -31,7 +31,7 @@ libpng Portable Network Graphics library.
|
|||||||
libtiff Tag Image File Format (TIFF) Software
|
libtiff Tag Image File Format (TIFF) Software
|
||||||
Copyright (c) 1988-1997 Sam Leffler
|
Copyright (c) 1988-1997 Sam Leffler
|
||||||
Copyright (c) 1991-1997 Silicon Graphics, Inc.
|
Copyright (c) 1991-1997 Silicon Graphics, Inc.
|
||||||
See libtiff home page http://www.remotesensing.org/libtiff/
|
See libtiff home page http://www.libtiff.org/
|
||||||
for details and links to the source code
|
for details and links to the source code
|
||||||
|
|
||||||
WITH_TIFF CMake option must be ON to add libtiff & zlib support to imgcodecs.
|
WITH_TIFF CMake option must be ON to add libtiff & zlib support to imgcodecs.
|
||||||
@@ -51,7 +51,9 @@ jasper JasPer is a collection of software
|
|||||||
Copyright (c) 1999-2000 The University of British Columbia
|
Copyright (c) 1999-2000 The University of British Columbia
|
||||||
Copyright (c) 2001-2003 Michael David Adams
|
Copyright (c) 2001-2003 Michael David Adams
|
||||||
|
|
||||||
The JasPer license can be found in libjasper.
|
See JasPer official GitHub repository
|
||||||
|
https://github.com/jasper-software/jasper.git
|
||||||
|
for details and links to source code
|
||||||
------------------------------------------------------------------------------------
|
------------------------------------------------------------------------------------
|
||||||
openexr OpenEXR is a high dynamic-range (HDR) image file format developed
|
openexr OpenEXR is a high dynamic-range (HDR) image file format developed
|
||||||
by Industrial Light & Magic for use in computer imaging applications.
|
by Industrial Light & Magic for use in computer imaging applications.
|
||||||
|
|||||||
+12
-5
@@ -648,6 +648,8 @@ if(UNIX)
|
|||||||
set(OPENCV_LINKER_LIBS ${OPENCV_LINKER_LIBS} m pthread)
|
set(OPENCV_LINKER_LIBS ${OPENCV_LINKER_LIBS} m pthread)
|
||||||
elseif(EMSCRIPTEN)
|
elseif(EMSCRIPTEN)
|
||||||
# no need to link to system libs with emscripten
|
# no need to link to system libs with emscripten
|
||||||
|
elseif(QNXNTO)
|
||||||
|
set(OPENCV_LINKER_LIBS ${OPENCV_LINKER_LIBS} m)
|
||||||
else()
|
else()
|
||||||
set(OPENCV_LINKER_LIBS ${OPENCV_LINKER_LIBS} dl m pthread rt)
|
set(OPENCV_LINKER_LIBS ${OPENCV_LINKER_LIBS} dl m pthread rt)
|
||||||
endif()
|
endif()
|
||||||
@@ -1230,12 +1232,17 @@ status("")
|
|||||||
status(" GUI: ")
|
status(" GUI: ")
|
||||||
|
|
||||||
if(WITH_QT OR HAVE_QT)
|
if(WITH_QT OR HAVE_QT)
|
||||||
if(HAVE_QT5)
|
if(HAVE_QT)
|
||||||
status(" QT:" "YES (ver ${Qt5Core_VERSION_STRING})")
|
|
||||||
status(" QT OpenGL support:" HAVE_QT_OPENGL THEN "YES (${Qt5OpenGL_LIBRARIES} ${Qt5OpenGL_VERSION_STRING})" ELSE NO)
|
|
||||||
elseif(HAVE_QT)
|
|
||||||
status(" QT:" "YES (ver ${QT_VERSION_MAJOR}.${QT_VERSION_MINOR}.${QT_VERSION_PATCH} ${QT_EDITION})")
|
status(" QT:" "YES (ver ${QT_VERSION_MAJOR}.${QT_VERSION_MINOR}.${QT_VERSION_PATCH} ${QT_EDITION})")
|
||||||
status(" QT OpenGL support:" HAVE_QT_OPENGL THEN "YES (${QT_QTOPENGL_LIBRARY})" ELSE NO)
|
if(HAVE_QT_OPENGL)
|
||||||
|
if(Qt${QT_VERSION_MAJOR}OpenGL_LIBRARIES)
|
||||||
|
status(" QT OpenGL support:" HAVE_QT_OPENGL THEN "YES (${Qt${QT_VERSION_MAJOR}OpenGL_LIBRARIES} ${Qt${QT_VERSION_MAJOR}OpenGL_VERSION_STRING})" ELSE NO)
|
||||||
|
else()
|
||||||
|
status(" QT OpenGL support:" HAVE_QT_OPENGL THEN "YES (${QT_QTOPENGL_LIBRARY})" ELSE NO)
|
||||||
|
endif()
|
||||||
|
else()
|
||||||
|
status(" QT OpenGL support:" "NO")
|
||||||
|
endif()
|
||||||
else()
|
else()
|
||||||
status(" QT:" "NO")
|
status(" QT:" "NO")
|
||||||
endif()
|
endif()
|
||||||
|
|||||||
@@ -400,6 +400,9 @@ if(MSVC)
|
|||||||
endif()
|
endif()
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
|
# Enable [[attribute]] syntax checking to prevent silent failure: "attribute is ignored in this syntactic position"
|
||||||
|
add_extra_compiler_option("/w15240")
|
||||||
|
|
||||||
if(NOT ENABLE_NOISY_WARNINGS)
|
if(NOT ENABLE_NOISY_WARNINGS)
|
||||||
ocv_warnings_disable(CMAKE_CXX_FLAGS /wd4127) # conditional expression is constant
|
ocv_warnings_disable(CMAKE_CXX_FLAGS /wd4127) # conditional expression is constant
|
||||||
ocv_warnings_disable(CMAKE_CXX_FLAGS /wd4251) # class 'std::XXX' needs to have dll-interface to be used by clients of YYY
|
ocv_warnings_disable(CMAKE_CXX_FLAGS /wd4251) # class 'std::XXX' needs to have dll-interface to be used by clients of YYY
|
||||||
|
|||||||
@@ -102,7 +102,7 @@ if(CUDA_FOUND)
|
|||||||
if(CUDA_GENERATION)
|
if(CUDA_GENERATION)
|
||||||
if(NOT ";${_generations};" MATCHES ";${CUDA_GENERATION};")
|
if(NOT ";${_generations};" MATCHES ";${CUDA_GENERATION};")
|
||||||
string(REPLACE ";" ", " _generations "${_generations}")
|
string(REPLACE ";" ", " _generations "${_generations}")
|
||||||
message(FATAL_ERROR "ERROR: ${_generations} Generations are suppered.")
|
message(FATAL_ERROR "ERROR: ${_generations} Generations are supported.")
|
||||||
endif()
|
endif()
|
||||||
unset(CUDA_ARCH_BIN CACHE)
|
unset(CUDA_ARCH_BIN CACHE)
|
||||||
unset(CUDA_ARCH_PTX CACHE)
|
unset(CUDA_ARCH_PTX CACHE)
|
||||||
|
|||||||
@@ -169,6 +169,8 @@ elseif(MSVC)
|
|||||||
set(OpenCV_RUNTIME vc15)
|
set(OpenCV_RUNTIME vc15)
|
||||||
elseif(MSVC_VERSION MATCHES "^192[0-9]$")
|
elseif(MSVC_VERSION MATCHES "^192[0-9]$")
|
||||||
set(OpenCV_RUNTIME vc16)
|
set(OpenCV_RUNTIME vc16)
|
||||||
|
elseif(MSVC_VERSION MATCHES "^193[0-9]$")
|
||||||
|
set(OpenCV_RUNTIME vc17)
|
||||||
else()
|
else()
|
||||||
message(WARNING "OpenCV does not recognize MSVC_VERSION \"${MSVC_VERSION}\". Cannot set OpenCV_RUNTIME")
|
message(WARNING "OpenCV does not recognize MSVC_VERSION \"${MSVC_VERSION}\". Cannot set OpenCV_RUNTIME")
|
||||||
endif()
|
endif()
|
||||||
|
|||||||
@@ -105,6 +105,20 @@ if(InferenceEngine_FOUND)
|
|||||||
message(STATUS "Detected InferenceEngine: cmake package (${InferenceEngine_VERSION})")
|
message(STATUS "Detected InferenceEngine: cmake package (${InferenceEngine_VERSION})")
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
|
if(DEFINED InferenceEngine_VERSION)
|
||||||
|
message(STATUS "InferenceEngine: ${InferenceEngine_VERSION}")
|
||||||
|
if(NOT INF_ENGINE_RELEASE AND NOT (InferenceEngine_VERSION VERSION_LESS "2021.4"))
|
||||||
|
math(EXPR INF_ENGINE_RELEASE_INIT "${InferenceEngine_VERSION_MAJOR} * 1000000 + ${InferenceEngine_VERSION_MINOR} * 10000 + ${InferenceEngine_VERSION_PATCH} * 100")
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
|
if(NOT INF_ENGINE_RELEASE AND NOT INF_ENGINE_RELEASE_INIT)
|
||||||
|
message(STATUS "WARNING: InferenceEngine version has not been set, 2021.4.1 will be used by default. Set INF_ENGINE_RELEASE variable if you experience build errors.")
|
||||||
|
set(INF_ENGINE_RELEASE_INIT "2021040100")
|
||||||
|
elseif(DEFINED INF_ENGINE_RELEASE)
|
||||||
|
set(INF_ENGINE_RELEASE_INIT "${INF_ENGINE_RELEASE}")
|
||||||
|
endif()
|
||||||
|
set(INF_ENGINE_RELEASE "${INF_ENGINE_RELEASE_INIT}" CACHE STRING "Force IE version, should be in form YYYYAABBCC (e.g. 2020.1.0.2 -> 2020010002)")
|
||||||
|
|
||||||
if(NOT INF_ENGINE_TARGET AND INF_ENGINE_LIB_DIRS AND INF_ENGINE_INCLUDE_DIRS)
|
if(NOT INF_ENGINE_TARGET AND INF_ENGINE_LIB_DIRS AND INF_ENGINE_INCLUDE_DIRS)
|
||||||
find_path(ie_custom_inc "inference_engine.hpp" PATHS "${INF_ENGINE_INCLUDE_DIRS}" NO_DEFAULT_PATH)
|
find_path(ie_custom_inc "inference_engine.hpp" PATHS "${INF_ENGINE_INCLUDE_DIRS}" NO_DEFAULT_PATH)
|
||||||
if(CMAKE_BUILD_TYPE STREQUAL "Debug")
|
if(CMAKE_BUILD_TYPE STREQUAL "Debug")
|
||||||
@@ -140,19 +154,6 @@ endif()
|
|||||||
# Add more features to the target
|
# Add more features to the target
|
||||||
|
|
||||||
if(INF_ENGINE_TARGET)
|
if(INF_ENGINE_TARGET)
|
||||||
if(DEFINED InferenceEngine_VERSION)
|
|
||||||
message(STATUS "InferenceEngine: ${InferenceEngine_VERSION}")
|
|
||||||
if(NOT INF_ENGINE_RELEASE AND NOT (InferenceEngine_VERSION VERSION_LESS "2021.4"))
|
|
||||||
math(EXPR INF_ENGINE_RELEASE_INIT "${InferenceEngine_VERSION_MAJOR} * 1000000 + ${InferenceEngine_VERSION_MINOR} * 10000 + ${InferenceEngine_VERSION_PATCH} * 100")
|
|
||||||
endif()
|
|
||||||
endif()
|
|
||||||
if(NOT INF_ENGINE_RELEASE AND NOT INF_ENGINE_RELEASE_INIT)
|
|
||||||
message(WARNING "InferenceEngine version has not been set, 2021.4 will be used by default. Set INF_ENGINE_RELEASE variable if you experience build errors.")
|
|
||||||
set(INF_ENGINE_RELEASE_INIT "2021040000")
|
|
||||||
elseif(DEFINED INF_ENGINE_RELEASE)
|
|
||||||
set(INF_ENGINE_RELEASE_INIT "${INF_ENGINE_RELEASE}")
|
|
||||||
endif()
|
|
||||||
set(INF_ENGINE_RELEASE "${INF_ENGINE_RELEASE_INIT}" CACHE STRING "Force IE version, should be in form YYYYAABBCC (e.g. 2020.1.0.2 -> 2020010002)")
|
|
||||||
set_target_properties(${INF_ENGINE_TARGET} PROPERTIES
|
set_target_properties(${INF_ENGINE_TARGET} PROPERTIES
|
||||||
INTERFACE_COMPILE_DEFINITIONS "HAVE_INF_ENGINE=1;INF_ENGINE_RELEASE=${INF_ENGINE_RELEASE}"
|
INTERFACE_COMPILE_DEFINITIONS "HAVE_INF_ENGINE=1;INF_ENGINE_RELEASE=${INF_ENGINE_RELEASE}"
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -26,31 +26,14 @@ if(VTK_VERSION VERSION_LESS "5.8.0")
|
|||||||
endif()
|
endif()
|
||||||
|
|
||||||
# Different Qt versions can't be linked together
|
# Different Qt versions can't be linked together
|
||||||
if(HAVE_QT5 AND VTK_VERSION VERSION_LESS "6.0.0")
|
if((HAVE_QT AND VTK_USE_QT)
|
||||||
if(VTK_USE_QT)
|
AND NOT DEFINED FORCE_VTK # deprecated
|
||||||
message(STATUS "VTK support is disabled. Incompatible combination: OpenCV + Qt5 and VTK ver.${VTK_VERSION} + Qt4")
|
AND NOT DEFINED OPENCV_FORCE_VTK
|
||||||
endif()
|
)
|
||||||
endif()
|
message(STATUS "VTK support is disabled. Possible incompatible combination: OpenCV+Qt, and VTK ver.${VTK_VERSION} with Qt")
|
||||||
|
message(STATUS "If it is known that VTK was compiled without Qt4, please define '-DOPENCV_FORCE_VTK=TRUE' flag in CMake")
|
||||||
# Different Qt versions can't be linked together. VTK 6.0.0 doesn't provide a way to get Qt version it was linked with
|
|
||||||
if(HAVE_QT5 AND VTK_VERSION VERSION_EQUAL "6.0.0" AND NOT DEFINED FORCE_VTK)
|
|
||||||
message(STATUS "VTK support is disabled. Possible incompatible combination: OpenCV+Qt5, and VTK ver.${VTK_VERSION} with Qt4")
|
|
||||||
message(STATUS "If it is known that VTK was compiled without Qt4, please define '-DFORCE_VTK=TRUE' flag in CMake")
|
|
||||||
return()
|
return()
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
# Different Qt versions can't be linked together
|
|
||||||
if(HAVE_QT AND VTK_VERSION VERSION_GREATER "6.0.0" AND NOT ${VTK_QT_VERSION} STREQUAL "")
|
|
||||||
if(HAVE_QT5 AND ${VTK_QT_VERSION} EQUAL "4")
|
|
||||||
message(STATUS "VTK support is disabled. Incompatible combination: OpenCV + Qt5 and VTK ver.${VTK_VERSION} + Qt4")
|
|
||||||
return()
|
|
||||||
endif()
|
|
||||||
|
|
||||||
if(NOT HAVE_QT5 AND ${VTK_QT_VERSION} EQUAL "5")
|
|
||||||
message(STATUS "VTK support is disabled. Incompatible combination: OpenCV + Qt4 and VTK ver.${VTK_VERSION} + Qt5")
|
|
||||||
return()
|
|
||||||
endif()
|
|
||||||
endif()
|
|
||||||
|
|
||||||
set(HAVE_VTK ON)
|
set(HAVE_VTK ON)
|
||||||
message(STATUS "Found VTK ${VTK_VERSION} (${VTK_USE_FILE})")
|
message(STATUS "Found VTK ${VTK_VERSION} (${VTK_USE_FILE})")
|
||||||
|
|||||||
@@ -11,25 +11,50 @@ if(WITH_WIN32UI)
|
|||||||
CMAKE_FLAGS "-DLINK_LIBRARIES:STRING=user32;gdi32")
|
CMAKE_FLAGS "-DLINK_LIBRARIES:STRING=user32;gdi32")
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
# --- QT4 ---
|
macro(ocv_find_package_Qt4)
|
||||||
ocv_clear_vars(HAVE_QT HAVE_QT5)
|
find_package(Qt4 COMPONENTS QtCore QtGui QtTest ${ARGN})
|
||||||
if(WITH_QT)
|
if(QT4_FOUND)
|
||||||
if(NOT WITH_QT EQUAL 4)
|
set(QT_FOUND 1)
|
||||||
find_package(Qt5 COMPONENTS Core Gui Widgets Test Concurrent REQUIRED NO_MODULE)
|
ocv_assert(QT_VERSION_MAJOR EQUAL 4)
|
||||||
if(Qt5_FOUND)
|
|
||||||
set(HAVE_QT5 ON)
|
|
||||||
set(HAVE_QT ON)
|
|
||||||
find_package(Qt5 COMPONENTS OpenGL QUIET)
|
|
||||||
if(Qt5OpenGL_FOUND)
|
|
||||||
set(QT_QTOPENGL_FOUND ON)
|
|
||||||
endif()
|
|
||||||
endif()
|
|
||||||
endif()
|
endif()
|
||||||
|
endmacro()
|
||||||
|
|
||||||
if(NOT HAVE_QT)
|
macro(ocv_find_package_Qt OCV_QT_VER)
|
||||||
find_package(Qt4 REQUIRED QtCore QtGui QtTest)
|
find_package(Qt${OCV_QT_VER} COMPONENTS Core Gui Widgets Test Concurrent ${ARGN} NO_MODULE)
|
||||||
if(QT4_FOUND)
|
if(Qt${OCV_QT_VER}_FOUND)
|
||||||
set(HAVE_QT TRUE)
|
set(QT_FOUND 1)
|
||||||
|
set(QT_VERSION "${Qt${OCV_QT_VER}_VERSION}")
|
||||||
|
set(QT_VERSION_MAJOR "${Qt${OCV_QT_VER}_VERSION_MAJOR}")
|
||||||
|
set(QT_VERSION_MINOR "${Qt${OCV_QT_VER}_VERSION_MINOR}")
|
||||||
|
set(QT_VERSION_PATCH "${Qt${OCV_QT_VER}_VERSION_PATCH}")
|
||||||
|
set(QT_VERSION_TWEAK "${Qt${OCV_QT_VER}_VERSION_TWEAK}")
|
||||||
|
set(QT_VERSION_COUNT "${Qt${OCV_QT_VER}_VERSION_COUNT}")
|
||||||
|
endif()
|
||||||
|
endmacro()
|
||||||
|
|
||||||
|
# --- QT4 ---
|
||||||
|
if(WITH_QT)
|
||||||
|
if(NOT WITH_QT GREATER 0)
|
||||||
|
# BUG: Qt5Config.cmake script can't handle components properly: find_package(QT NAMES Qt6 Qt5 REQUIRED NO_MODULE COMPONENTS Core Gui Widgets Test Concurrent)
|
||||||
|
ocv_find_package_Qt(6 QUIET)
|
||||||
|
if(NOT QT_FOUND)
|
||||||
|
ocv_find_package_Qt(5 QUIET)
|
||||||
|
endif()
|
||||||
|
if(NOT QT_FOUND)
|
||||||
|
ocv_find_package_Qt4(QUIET)
|
||||||
|
endif()
|
||||||
|
elseif(WITH_QT EQUAL 4)
|
||||||
|
ocv_find_package_Qt4(REQUIRED)
|
||||||
|
else() # WITH_QT=<major version>
|
||||||
|
ocv_find_package_Qt("${WITH_QT}" REQUIRED)
|
||||||
|
endif()
|
||||||
|
if(QT_FOUND)
|
||||||
|
set(HAVE_QT ON)
|
||||||
|
if(QT_VERSION_MAJOR GREATER 4)
|
||||||
|
find_package(Qt${QT_VERSION_MAJOR} COMPONENTS OpenGL QUIET)
|
||||||
|
if(Qt${QT_VERSION_MAJOR}OpenGL_FOUND)
|
||||||
|
set(QT_QTOPENGL_FOUND ON) # HAVE_QT_OPENGL is defined below
|
||||||
|
endif()
|
||||||
endif()
|
endif()
|
||||||
endif()
|
endif()
|
||||||
endif()
|
endif()
|
||||||
|
|||||||
@@ -1488,8 +1488,8 @@ function(ocv_target_link_libraries target)
|
|||||||
if(NOT LINK_PENDING STREQUAL "")
|
if(NOT LINK_PENDING STREQUAL "")
|
||||||
__ocv_push_target_link_libraries(${LINK_MODE} ${LINK_PENDING})
|
__ocv_push_target_link_libraries(${LINK_MODE} ${LINK_PENDING})
|
||||||
set(LINK_PENDING "")
|
set(LINK_PENDING "")
|
||||||
set(LINK_MODE "${dep}")
|
|
||||||
endif()
|
endif()
|
||||||
|
set(LINK_MODE "${dep}")
|
||||||
else()
|
else()
|
||||||
if(BUILD_opencv_world)
|
if(BUILD_opencv_world)
|
||||||
if(OPENCV_MODULE_${dep}_IS_PART_OF_WORLD)
|
if(OPENCV_MODULE_${dep}_IS_PART_OF_WORLD)
|
||||||
|
|||||||
@@ -137,6 +137,20 @@ elseif(MSVC)
|
|||||||
set(OpenCV_RUNTIME vc14) # selecting previous compatible runtime version
|
set(OpenCV_RUNTIME vc14) # selecting previous compatible runtime version
|
||||||
endif()
|
endif()
|
||||||
endif()
|
endif()
|
||||||
|
elseif(MSVC_VERSION MATCHES "^193[0-9]$")
|
||||||
|
set(OpenCV_RUNTIME vc17)
|
||||||
|
check_one_config(has_VS2022)
|
||||||
|
if(NOT has_VS2022)
|
||||||
|
set(OpenCV_RUNTIME vc16)
|
||||||
|
check_one_config(has_VS2019)
|
||||||
|
if(NOT has_VS2019)
|
||||||
|
set(OpenCV_RUNTIME vc15) # selecting previous compatible runtime version
|
||||||
|
check_one_config(has_VS2017)
|
||||||
|
if(NOT has_VS2017)
|
||||||
|
set(OpenCV_RUNTIME vc14) # selecting previous compatible runtime version
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
|
endif()
|
||||||
endif()
|
endif()
|
||||||
elseif(MINGW)
|
elseif(MINGW)
|
||||||
set(OpenCV_RUNTIME mingw)
|
set(OpenCV_RUNTIME mingw)
|
||||||
|
|||||||
+3
@@ -1,6 +1,9 @@
|
|||||||
Contour Features {#tutorial_js_contour_features}
|
Contour Features {#tutorial_js_contour_features}
|
||||||
================
|
================
|
||||||
|
|
||||||
|
@prev_tutorial{tutorial_js_contours_begin}
|
||||||
|
@next_tutorial{tutorial_js_contour_properties}
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
|
|||||||
+3
@@ -1,6 +1,9 @@
|
|||||||
Contour Properties {#tutorial_js_contour_properties}
|
Contour Properties {#tutorial_js_contour_properties}
|
||||||
==================
|
==================
|
||||||
|
|
||||||
|
@prev_tutorial{tutorial_js_contour_features}
|
||||||
|
@next_tutorial{tutorial_js_contours_more_functions}
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
Contours : Getting Started {#tutorial_js_contours_begin}
|
Contours : Getting Started {#tutorial_js_contours_begin}
|
||||||
==========================
|
==========================
|
||||||
|
|
||||||
|
@next_tutorial{tutorial_js_contour_features}
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
|
|||||||
+2
@@ -1,6 +1,8 @@
|
|||||||
Contours Hierarchy {#tutorial_js_contours_hierarchy}
|
Contours Hierarchy {#tutorial_js_contours_hierarchy}
|
||||||
==================
|
==================
|
||||||
|
|
||||||
|
@prev_tutorial{tutorial_js_contours_more_functions}
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
|
|||||||
+3
@@ -1,6 +1,9 @@
|
|||||||
Contours : More Functions {#tutorial_js_contours_more_functions}
|
Contours : More Functions {#tutorial_js_contours_more_functions}
|
||||||
=========================
|
=========================
|
||||||
|
|
||||||
|
@prev_tutorial{tutorial_js_contour_properties}
|
||||||
|
@next_tutorial{tutorial_js_contours_hierarchy}
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
|
|||||||
@@ -98,7 +98,7 @@ import numpy as np
|
|||||||
import cv2 as cv
|
import cv2 as cv
|
||||||
from matplotlib import pyplot as plt
|
from matplotlib import pyplot as plt
|
||||||
|
|
||||||
img = cv.imread('simple.jpg',0)
|
img = cv.imread('blox.jpg',0) # `<opencv_root>/samples/data/blox.jpg`
|
||||||
|
|
||||||
# Initiate FAST object with default values
|
# Initiate FAST object with default values
|
||||||
fast = cv.FastFeatureDetector_create()
|
fast = cv.FastFeatureDetector_create()
|
||||||
@@ -113,17 +113,17 @@ print( "nonmaxSuppression:{}".format(fast.getNonmaxSuppression()) )
|
|||||||
print( "neighborhood: {}".format(fast.getType()) )
|
print( "neighborhood: {}".format(fast.getType()) )
|
||||||
print( "Total Keypoints with nonmaxSuppression: {}".format(len(kp)) )
|
print( "Total Keypoints with nonmaxSuppression: {}".format(len(kp)) )
|
||||||
|
|
||||||
cv.imwrite('fast_true.png',img2)
|
cv.imwrite('fast_true.png', img2)
|
||||||
|
|
||||||
# Disable nonmaxSuppression
|
# Disable nonmaxSuppression
|
||||||
fast.setNonmaxSuppression(0)
|
fast.setNonmaxSuppression(0)
|
||||||
kp = fast.detect(img,None)
|
kp = fast.detect(img, None)
|
||||||
|
|
||||||
print( "Total Keypoints without nonmaxSuppression: {}".format(len(kp)) )
|
print( "Total Keypoints without nonmaxSuppression: {}".format(len(kp)) )
|
||||||
|
|
||||||
img3 = cv.drawKeypoints(img, kp, None, color=(255,0,0))
|
img3 = cv.drawKeypoints(img, kp, None, color=(255,0,0))
|
||||||
|
|
||||||
cv.imwrite('fast_false.png',img3)
|
cv.imwrite('fast_false.png', img3)
|
||||||
@endcode
|
@endcode
|
||||||
See the results. First image shows FAST with nonmaxSuppression and second one without
|
See the results. First image shows FAST with nonmaxSuppression and second one without
|
||||||
nonmaxSuppression:
|
nonmaxSuppression:
|
||||||
|
|||||||
@@ -74,7 +74,7 @@ Canny Edge Detection in OpenCV
|
|||||||
|
|
||||||
OpenCV puts all the above in single function, **cv.Canny()**. We will see how to use it. First
|
OpenCV puts all the above in single function, **cv.Canny()**. We will see how to use it. First
|
||||||
argument is our input image. Second and third arguments are our minVal and maxVal respectively.
|
argument is our input image. Second and third arguments are our minVal and maxVal respectively.
|
||||||
Third argument is aperture_size. It is the size of Sobel kernel used for find image gradients. By
|
Fourth argument is aperture_size. It is the size of Sobel kernel used for find image gradients. By
|
||||||
default it is 3. Last argument is L2gradient which specifies the equation for finding gradient
|
default it is 3. Last argument is L2gradient which specifies the equation for finding gradient
|
||||||
magnitude. If it is True, it uses the equation mentioned above which is more accurate, otherwise it
|
magnitude. If it is True, it uses the equation mentioned above which is more accurate, otherwise it
|
||||||
uses this function: \f$Edge\_Gradient \; (G) = |G_x| + |G_y|\f$. By default, it is False.
|
uses this function: \f$Edge\_Gradient \; (G) = |G_x| + |G_y|\f$. By default, it is False.
|
||||||
|
|||||||
+4
-1
@@ -1,6 +1,9 @@
|
|||||||
Contour Features {#tutorial_py_contour_features}
|
Contour Features {#tutorial_py_contour_features}
|
||||||
================
|
================
|
||||||
|
|
||||||
|
@prev_tutorial{tutorial_py_contours_begin}
|
||||||
|
@next_tutorial{tutorial_py_contour_properties}
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
@@ -91,7 +94,7 @@ convexity defects, which are the local maximum deviations of hull from contours.
|
|||||||
|
|
||||||
There is a little bit things to discuss about it its syntax:
|
There is a little bit things to discuss about it its syntax:
|
||||||
@code{.py}
|
@code{.py}
|
||||||
hull = cv.convexHull(points[, hull[, clockwise[, returnPoints]]
|
hull = cv.convexHull(points[, hull[, clockwise[, returnPoints]]])
|
||||||
@endcode
|
@endcode
|
||||||
Arguments details:
|
Arguments details:
|
||||||
|
|
||||||
|
|||||||
+3
@@ -1,6 +1,9 @@
|
|||||||
Contour Properties {#tutorial_py_contour_properties}
|
Contour Properties {#tutorial_py_contour_properties}
|
||||||
==================
|
==================
|
||||||
|
|
||||||
|
@prev_tutorial{tutorial_py_contour_features}
|
||||||
|
@next_tutorial{tutorial_py_contours_more_functions}
|
||||||
|
|
||||||
Here we will learn to extract some frequently used properties of objects like Solidity, Equivalent
|
Here we will learn to extract some frequently used properties of objects like Solidity, Equivalent
|
||||||
Diameter, Mask image, Mean Intensity etc. More features can be found at [Matlab regionprops
|
Diameter, Mask image, Mean Intensity etc. More features can be found at [Matlab regionprops
|
||||||
documentation](http://www.mathworks.in/help/images/ref/regionprops.html).
|
documentation](http://www.mathworks.in/help/images/ref/regionprops.html).
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
Contours : Getting Started {#tutorial_py_contours_begin}
|
Contours : Getting Started {#tutorial_py_contours_begin}
|
||||||
==========================
|
==========================
|
||||||
|
|
||||||
|
@next_tutorial{tutorial_py_contour_features}
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
|
|||||||
+2
@@ -1,6 +1,8 @@
|
|||||||
Contours Hierarchy {#tutorial_py_contours_hierarchy}
|
Contours Hierarchy {#tutorial_py_contours_hierarchy}
|
||||||
==================
|
==================
|
||||||
|
|
||||||
|
@prev_tutorial{tutorial_py_contours_more_functions}
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
|
|||||||
+4
@@ -1,6 +1,10 @@
|
|||||||
Contours : More Functions {#tutorial_py_contours_more_functions}
|
Contours : More Functions {#tutorial_py_contours_more_functions}
|
||||||
=========================
|
=========================
|
||||||
|
|
||||||
|
@prev_tutorial{tutorial_py_contour_properties}
|
||||||
|
@next_tutorial{tutorial_py_contours_hierarchy}
|
||||||
|
|
||||||
|
|
||||||
Goal
|
Goal
|
||||||
----
|
----
|
||||||
|
|
||||||
|
|||||||
+9
-9
@@ -84,8 +84,8 @@ a new header with the new boundaries:
|
|||||||
Mat D (A, Rect(10, 10, 100, 100) ); // using a rectangle
|
Mat D (A, Rect(10, 10, 100, 100) ); // using a rectangle
|
||||||
Mat E = A(Range::all(), Range(1,3)); // using row and column boundaries
|
Mat E = A(Range::all(), Range(1,3)); // using row and column boundaries
|
||||||
@endcode
|
@endcode
|
||||||
Now you may ask -- if the matrix itself may belong to multiple *Mat* objects who takes responsibility
|
Now you may ask -- if the matrix itself may belong to multiple *Mat* objects, who takes responsibility
|
||||||
for cleaning it up when it's no longer needed. The short answer is: the last object that used it.
|
for cleaning it up when it's no longer needed? The short answer is: the last object that used it.
|
||||||
This is handled by using a reference counting mechanism. Whenever somebody copies a header of a
|
This is handled by using a reference counting mechanism. Whenever somebody copies a header of a
|
||||||
*Mat* object, a counter is increased for the matrix. Whenever a header is cleaned, this counter
|
*Mat* object, a counter is increased for the matrix. Whenever a header is cleaned, this counter
|
||||||
is decreased. When the counter reaches zero the matrix is freed. Sometimes you will want to copy
|
is decreased. When the counter reaches zero the matrix is freed. Sometimes you will want to copy
|
||||||
@@ -95,12 +95,12 @@ Mat F = A.clone();
|
|||||||
Mat G;
|
Mat G;
|
||||||
A.copyTo(G);
|
A.copyTo(G);
|
||||||
@endcode
|
@endcode
|
||||||
Now modifying *F* or *G* will not affect the matrix pointed by the *A*'s header. What you need to
|
Now modifying *F* or *G* will not affect the matrix pointed to by the *A*'s header. What you need to
|
||||||
remember from all this is that:
|
remember from all this is that:
|
||||||
|
|
||||||
- Output image allocation for OpenCV functions is automatic (unless specified otherwise).
|
- Output image allocation for OpenCV functions is automatic (unless specified otherwise).
|
||||||
- You do not need to think about memory management with OpenCV's C++ interface.
|
- You do not need to think about memory management with OpenCV's C++ interface.
|
||||||
- The assignment operator and the copy constructor only copies the header.
|
- The assignment operator and the copy constructor only copy the header.
|
||||||
- The underlying matrix of an image may be copied using the @ref cv::Mat::clone() and @ref cv::Mat::copyTo()
|
- The underlying matrix of an image may be copied using the @ref cv::Mat::clone() and @ref cv::Mat::copyTo()
|
||||||
functions.
|
functions.
|
||||||
|
|
||||||
@@ -115,10 +115,10 @@ of these allows us to create many shades of gray.
|
|||||||
For *colorful* ways we have a lot more methods to choose from. Each of them breaks it down to three
|
For *colorful* ways we have a lot more methods to choose from. Each of them breaks it down to three
|
||||||
or four basic components and we can use the combination of these to create the others. The most
|
or four basic components and we can use the combination of these to create the others. The most
|
||||||
popular one is RGB, mainly because this is also how our eye builds up colors. Its base colors are
|
popular one is RGB, mainly because this is also how our eye builds up colors. Its base colors are
|
||||||
red, green and blue. To code the transparency of a color sometimes a fourth element: alpha (A) is
|
red, green and blue. To code the transparency of a color sometimes a fourth element, alpha (A), is
|
||||||
added.
|
added.
|
||||||
|
|
||||||
There are, however, many other color systems each with their own advantages:
|
There are, however, many other color systems, each with their own advantages:
|
||||||
|
|
||||||
- RGB is the most common as our eyes use something similar, however keep in mind that OpenCV standard display
|
- RGB is the most common as our eyes use something similar, however keep in mind that OpenCV standard display
|
||||||
system composes colors using the BGR color space (red and blue channels are swapped places).
|
system composes colors using the BGR color space (red and blue channels are swapped places).
|
||||||
@@ -132,11 +132,11 @@ There are, however, many other color systems each with their own advantages:
|
|||||||
Each of the building components has its own valid domains. This leads to the data type used. How
|
Each of the building components has its own valid domains. This leads to the data type used. How
|
||||||
we store a component defines the control we have over its domain. The smallest data type possible is
|
we store a component defines the control we have over its domain. The smallest data type possible is
|
||||||
*char*, which means one byte or 8 bits. This may be unsigned (so can store values from 0 to 255) or
|
*char*, which means one byte or 8 bits. This may be unsigned (so can store values from 0 to 255) or
|
||||||
signed (values from -127 to +127). Although in case of three components this already gives 16
|
signed (values from -127 to +127). Although this width, in the case of three components (like RGB), already gives 16
|
||||||
million possible colors to represent (like in case of RGB) we may acquire an even finer control by
|
million possible colors to represent, we may acquire an even finer control by
|
||||||
using the float (4 byte = 32 bit) or double (8 byte = 64 bit) data types for each component.
|
using the float (4 byte = 32 bit) or double (8 byte = 64 bit) data types for each component.
|
||||||
Nevertheless, remember that increasing the size of a component also increases the size of the whole
|
Nevertheless, remember that increasing the size of a component also increases the size of the whole
|
||||||
picture in the memory.
|
picture in memory.
|
||||||
|
|
||||||
Creating a Mat object explicitly
|
Creating a Mat object explicitly
|
||||||
----------------------------------
|
----------------------------------
|
||||||
|
|||||||
@@ -39,14 +39,14 @@ Open your Doxyfile using your favorite text editor and search for the key
|
|||||||
`TAGFILES`. Change it as follows:
|
`TAGFILES`. Change it as follows:
|
||||||
|
|
||||||
@code
|
@code
|
||||||
TAGFILES = ./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/3.4.15
|
TAGFILES = ./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/3.4.16
|
||||||
@endcode
|
@endcode
|
||||||
|
|
||||||
If you had other definitions already, you can append the line using a `\`:
|
If you had other definitions already, you can append the line using a `\`:
|
||||||
|
|
||||||
@code
|
@code
|
||||||
TAGFILES = ./docs/doxygen-tags/libstdc++.tag=https://gcc.gnu.org/onlinedocs/libstdc++/latest-doxygen \
|
TAGFILES = ./docs/doxygen-tags/libstdc++.tag=https://gcc.gnu.org/onlinedocs/libstdc++/latest-doxygen \
|
||||||
./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/3.4.15
|
./docs/doxygen-tags/opencv.tag=http://docs.opencv.org/3.4.16
|
||||||
@endcode
|
@endcode
|
||||||
|
|
||||||
Doxygen can now use the information from the tag file to link to the OpenCV
|
Doxygen can now use the information from the tag file to link to the OpenCV
|
||||||
|
|||||||
@@ -10,6 +10,8 @@ Working with a boosted cascade of weak classifiers includes two major stages: th
|
|||||||
|
|
||||||
To support this tutorial, several official OpenCV applications will be used: [opencv_createsamples](https://github.com/opencv/opencv/tree/3.4/apps/createsamples), [opencv_annotation](https://github.com/opencv/opencv/tree/3.4/apps/annotation), [opencv_traincascade](https://github.com/opencv/opencv/tree/3.4/apps/traincascade) and [opencv_visualisation](https://github.com/opencv/opencv/tree/3.4/apps/visualisation).
|
To support this tutorial, several official OpenCV applications will be used: [opencv_createsamples](https://github.com/opencv/opencv/tree/3.4/apps/createsamples), [opencv_annotation](https://github.com/opencv/opencv/tree/3.4/apps/annotation), [opencv_traincascade](https://github.com/opencv/opencv/tree/3.4/apps/traincascade) and [opencv_visualisation](https://github.com/opencv/opencv/tree/3.4/apps/visualisation).
|
||||||
|
|
||||||
|
@note Createsamples and traincascade are disabled since OpenCV 4.0. Consider using these apps for training from 3.4 branch for Cascade Classifier. Model format is the same between 3.4 and 4.x.
|
||||||
|
|
||||||
### Important notes
|
### Important notes
|
||||||
|
|
||||||
- If you come across any tutorial mentioning the old opencv_haartraining tool <i>(which is deprecated and still using the OpenCV1.x interface)</i>, then please ignore that tutorial and stick to the opencv_traincascade tool. This tool is a newer version, written in C++ in accordance to the OpenCV 2.x and OpenCV 3.x API. The opencv_traincascade supports both HAAR like wavelet features @cite Viola01 and LBP (Local Binary Patterns) @cite Liao2007 features. LBP features yield integer precision in contrast to HAAR features, yielding floating point precision, so both training and detection with LBP are several times faster then with HAAR features. Regarding the LBP and HAAR detection quality, it mainly depends on the training data used and the training parameters selected. It's possible to train a LBP-based classifier that will provide almost the same quality as HAAR-based one, within a percentage of the training time.
|
- If you come across any tutorial mentioning the old opencv_haartraining tool <i>(which is deprecated and still using the OpenCV1.x interface)</i>, then please ignore that tutorial and stick to the opencv_traincascade tool. This tool is a newer version, written in C++ in accordance to the OpenCV 2.x and OpenCV 3.x API. The opencv_traincascade supports both HAAR like wavelet features @cite Viola01 and LBP (Local Binary Patterns) @cite Liao2007 features. LBP features yield integer precision in contrast to HAAR features, yielding floating point precision, so both training and detection with LBP are several times faster then with HAAR features. Regarding the LBP and HAAR detection quality, it mainly depends on the training data used and the training parameters selected. It's possible to train a LBP-based classifier that will provide almost the same quality as HAAR-based one, within a percentage of the training time.
|
||||||
|
|||||||
@@ -180,7 +180,7 @@ implementation below.
|
|||||||
|
|
||||||
This will return a similarity index for each channel of the image. This value is between zero and
|
This will return a similarity index for each channel of the image. This value is between zero and
|
||||||
one, where one corresponds to perfect fit. Unfortunately, the many Gaussian blurring is quite
|
one, where one corresponds to perfect fit. Unfortunately, the many Gaussian blurring is quite
|
||||||
costly, so while the PSNR may work in a real time like environment (24 frame per second) this will
|
costly, so while the PSNR may work in a real time like environment (24 frames per second) this will
|
||||||
take significantly more than to accomplish similar performance results.
|
take significantly more than to accomplish similar performance results.
|
||||||
|
|
||||||
Therefore, the source code presented at the start of the tutorial will perform the PSNR measurement
|
Therefore, the source code presented at the start of the tutorial will perform the PSNR measurement
|
||||||
|
|||||||
@@ -116,6 +116,59 @@ String dumpRange(const Range& argument)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
CV_WRAP static inline
|
||||||
|
String testReservedKeywordConversion(int positional_argument, int lambda = 2, int from = 3)
|
||||||
|
{
|
||||||
|
return format("arg=%d, lambda=%d, from=%d", positional_argument, lambda, from);
|
||||||
|
}
|
||||||
|
|
||||||
|
CV_EXPORTS_W String dumpVectorOfInt(const std::vector<int>& vec);
|
||||||
|
|
||||||
|
CV_EXPORTS_W String dumpVectorOfDouble(const std::vector<double>& vec);
|
||||||
|
|
||||||
|
CV_EXPORTS_W String dumpVectorOfRect(const std::vector<Rect>& vec);
|
||||||
|
|
||||||
|
CV_WRAP static inline
|
||||||
|
void generateVectorOfRect(size_t len, CV_OUT std::vector<Rect>& vec)
|
||||||
|
{
|
||||||
|
vec.resize(len);
|
||||||
|
if (len > 0)
|
||||||
|
{
|
||||||
|
RNG rng(12345);
|
||||||
|
Mat tmp(static_cast<int>(len), 1, CV_32SC4);
|
||||||
|
rng.fill(tmp, RNG::UNIFORM, 10, 20);
|
||||||
|
tmp.copyTo(vec);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
CV_WRAP static inline
|
||||||
|
void generateVectorOfInt(size_t len, CV_OUT std::vector<int>& vec)
|
||||||
|
{
|
||||||
|
vec.resize(len);
|
||||||
|
if (len > 0)
|
||||||
|
{
|
||||||
|
RNG rng(554433);
|
||||||
|
Mat tmp(static_cast<int>(len), 1, CV_32SC1);
|
||||||
|
rng.fill(tmp, RNG::UNIFORM, -10, 10);
|
||||||
|
tmp.copyTo(vec);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
CV_WRAP static inline
|
||||||
|
void generateVectorOfMat(size_t len, int rows, int cols, int dtype, CV_OUT std::vector<Mat>& vec)
|
||||||
|
{
|
||||||
|
vec.resize(len);
|
||||||
|
if (len > 0)
|
||||||
|
{
|
||||||
|
RNG rng(65431);
|
||||||
|
for (size_t i = 0; i < len; ++i)
|
||||||
|
{
|
||||||
|
vec[i].create(rows, cols, dtype);
|
||||||
|
rng.fill(vec[i], RNG::UNIFORM, 0, 10);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
CV_WRAP static inline
|
CV_WRAP static inline
|
||||||
void testRaiseGeneralException()
|
void testRaiseGeneralException()
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -575,14 +575,47 @@ Cv64suf;
|
|||||||
# endif
|
# endif
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
/****************************************************************************************\
|
||||||
|
* CV_NODISCARD_STD attribute (C++17) *
|
||||||
|
* encourages the compiler to issue a warning if the return value is discarded *
|
||||||
|
\****************************************************************************************/
|
||||||
|
#ifndef CV_NODISCARD_STD
|
||||||
|
# ifndef __has_cpp_attribute
|
||||||
|
// workaround preprocessor non-compliance https://reviews.llvm.org/D57851
|
||||||
|
# define __has_cpp_attribute(__x) 0
|
||||||
|
# endif
|
||||||
|
# if __has_cpp_attribute(nodiscard)
|
||||||
|
# define CV_NODISCARD_STD [[nodiscard]]
|
||||||
|
# elif __cplusplus >= 201703L
|
||||||
|
// available when compiler is C++17 compliant
|
||||||
|
# define CV_NODISCARD_STD [[nodiscard]]
|
||||||
|
# elif defined(_MSC_VER) && _MSC_VER >= 1911 && _MSVC_LANG >= 201703L
|
||||||
|
// available with VS2017 v15.3+ with /std:c++17 or higher; works on functions and classes
|
||||||
|
# define CV_NODISCARD_STD [[nodiscard]]
|
||||||
|
# elif defined(__GNUC__) && (((__GNUC__ * 100) + __GNUC_MINOR__) >= 700) && (__cplusplus >= 201103L)
|
||||||
|
// available with GCC 7.0+; works on functions, works or silently fails on classes
|
||||||
|
# define CV_NODISCARD_STD [[nodiscard]]
|
||||||
|
# elif defined(__GNUC__) && (((__GNUC__ * 100) + __GNUC_MINOR__) >= 408) && (__cplusplus >= 201103L)
|
||||||
|
// available with GCC 4.8+ but it usually does nothing and can fail noisily -- therefore not used
|
||||||
|
// define CV_NODISCARD_STD [[gnu::warn_unused_result]]
|
||||||
|
# endif
|
||||||
|
#endif
|
||||||
|
#ifndef CV_NODISCARD_STD
|
||||||
|
# define CV_NODISCARD_STD /* nothing by default */
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
/****************************************************************************************\
|
/****************************************************************************************\
|
||||||
* CV_NODISCARD attribute *
|
* CV_NODISCARD attribute (deprecated, GCC only) *
|
||||||
* encourages the compiler to issue a warning if the return value is discarded (C++17) *
|
* DONT USE: use instead the standard CV_NODISCARD_STD macro above *
|
||||||
|
* this legacy method silently fails to issue warning until some version *
|
||||||
|
* after gcc 6.3.0. Yet with gcc 7+ you can use the above standard method *
|
||||||
|
* which makes this method useless. Don't use it. *
|
||||||
|
* @deprecated use instead CV_NODISCARD_STD *
|
||||||
\****************************************************************************************/
|
\****************************************************************************************/
|
||||||
#ifndef CV_NODISCARD
|
#ifndef CV_NODISCARD
|
||||||
# if defined(__GNUC__)
|
# if defined(__GNUC__)
|
||||||
# define CV_NODISCARD __attribute__((__warn_unused_result__)) // at least available with GCC 3.4
|
# define CV_NODISCARD __attribute__((__warn_unused_result__))
|
||||||
# elif defined(__clang__) && defined(__has_attribute)
|
# elif defined(__clang__) && defined(__has_attribute)
|
||||||
# if __has_attribute(__warn_unused_result__)
|
# if __has_attribute(__warn_unused_result__)
|
||||||
# define CV_NODISCARD __attribute__((__warn_unused_result__))
|
# define CV_NODISCARD __attribute__((__warn_unused_result__))
|
||||||
|
|||||||
@@ -1204,14 +1204,14 @@ public:
|
|||||||
The method creates a square diagonal matrix from specified main diagonal.
|
The method creates a square diagonal matrix from specified main diagonal.
|
||||||
@param d One-dimensional matrix that represents the main diagonal.
|
@param d One-dimensional matrix that represents the main diagonal.
|
||||||
*/
|
*/
|
||||||
static Mat diag(const Mat& d);
|
CV_NODISCARD_STD static Mat diag(const Mat& d);
|
||||||
|
|
||||||
/** @brief Creates a full copy of the array and the underlying data.
|
/** @brief Creates a full copy of the array and the underlying data.
|
||||||
|
|
||||||
The method creates a full copy of the array. The original step[] is not taken into account. So, the
|
The method creates a full copy of the array. The original step[] is not taken into account. So, the
|
||||||
array copy is a continuous array occupying total()*elemSize() bytes.
|
array copy is a continuous array occupying total()*elemSize() bytes.
|
||||||
*/
|
*/
|
||||||
Mat clone() const CV_NODISCARD;
|
CV_NODISCARD_STD Mat clone() const;
|
||||||
|
|
||||||
/** @brief Copies the matrix to another one.
|
/** @brief Copies the matrix to another one.
|
||||||
|
|
||||||
@@ -1375,20 +1375,20 @@ public:
|
|||||||
@param cols Number of columns.
|
@param cols Number of columns.
|
||||||
@param type Created matrix type.
|
@param type Created matrix type.
|
||||||
*/
|
*/
|
||||||
static MatExpr zeros(int rows, int cols, int type);
|
CV_NODISCARD_STD static MatExpr zeros(int rows, int cols, int type);
|
||||||
|
|
||||||
/** @overload
|
/** @overload
|
||||||
@param size Alternative to the matrix size specification Size(cols, rows) .
|
@param size Alternative to the matrix size specification Size(cols, rows) .
|
||||||
@param type Created matrix type.
|
@param type Created matrix type.
|
||||||
*/
|
*/
|
||||||
static MatExpr zeros(Size size, int type);
|
CV_NODISCARD_STD static MatExpr zeros(Size size, int type);
|
||||||
|
|
||||||
/** @overload
|
/** @overload
|
||||||
@param ndims Array dimensionality.
|
@param ndims Array dimensionality.
|
||||||
@param sz Array of integers specifying the array shape.
|
@param sz Array of integers specifying the array shape.
|
||||||
@param type Created matrix type.
|
@param type Created matrix type.
|
||||||
*/
|
*/
|
||||||
static MatExpr zeros(int ndims, const int* sz, int type);
|
CV_NODISCARD_STD static MatExpr zeros(int ndims, const int* sz, int type);
|
||||||
|
|
||||||
/** @brief Returns an array of all 1's of the specified size and type.
|
/** @brief Returns an array of all 1's of the specified size and type.
|
||||||
|
|
||||||
@@ -1406,20 +1406,20 @@ public:
|
|||||||
@param cols Number of columns.
|
@param cols Number of columns.
|
||||||
@param type Created matrix type.
|
@param type Created matrix type.
|
||||||
*/
|
*/
|
||||||
static MatExpr ones(int rows, int cols, int type);
|
CV_NODISCARD_STD static MatExpr ones(int rows, int cols, int type);
|
||||||
|
|
||||||
/** @overload
|
/** @overload
|
||||||
@param size Alternative to the matrix size specification Size(cols, rows) .
|
@param size Alternative to the matrix size specification Size(cols, rows) .
|
||||||
@param type Created matrix type.
|
@param type Created matrix type.
|
||||||
*/
|
*/
|
||||||
static MatExpr ones(Size size, int type);
|
CV_NODISCARD_STD static MatExpr ones(Size size, int type);
|
||||||
|
|
||||||
/** @overload
|
/** @overload
|
||||||
@param ndims Array dimensionality.
|
@param ndims Array dimensionality.
|
||||||
@param sz Array of integers specifying the array shape.
|
@param sz Array of integers specifying the array shape.
|
||||||
@param type Created matrix type.
|
@param type Created matrix type.
|
||||||
*/
|
*/
|
||||||
static MatExpr ones(int ndims, const int* sz, int type);
|
CV_NODISCARD_STD static MatExpr ones(int ndims, const int* sz, int type);
|
||||||
|
|
||||||
/** @brief Returns an identity matrix of the specified size and type.
|
/** @brief Returns an identity matrix of the specified size and type.
|
||||||
|
|
||||||
@@ -1435,13 +1435,13 @@ public:
|
|||||||
@param cols Number of columns.
|
@param cols Number of columns.
|
||||||
@param type Created matrix type.
|
@param type Created matrix type.
|
||||||
*/
|
*/
|
||||||
static MatExpr eye(int rows, int cols, int type);
|
CV_NODISCARD_STD static MatExpr eye(int rows, int cols, int type);
|
||||||
|
|
||||||
/** @overload
|
/** @overload
|
||||||
@param size Alternative matrix size specification as Size(cols, rows) .
|
@param size Alternative matrix size specification as Size(cols, rows) .
|
||||||
@param type Created matrix type.
|
@param type Created matrix type.
|
||||||
*/
|
*/
|
||||||
static MatExpr eye(Size size, int type);
|
CV_NODISCARD_STD static MatExpr eye(Size size, int type);
|
||||||
|
|
||||||
/** @brief Allocates new array data if needed.
|
/** @brief Allocates new array data if needed.
|
||||||
|
|
||||||
@@ -2302,7 +2302,7 @@ public:
|
|||||||
Mat_ row(int y) const;
|
Mat_ row(int y) const;
|
||||||
Mat_ col(int x) const;
|
Mat_ col(int x) const;
|
||||||
Mat_ diag(int d=0) const;
|
Mat_ diag(int d=0) const;
|
||||||
Mat_ clone() const CV_NODISCARD;
|
CV_NODISCARD_STD Mat_ clone() const;
|
||||||
|
|
||||||
//! overridden forms of Mat::elemSize() etc.
|
//! overridden forms of Mat::elemSize() etc.
|
||||||
size_t elemSize() const;
|
size_t elemSize() const;
|
||||||
@@ -2315,14 +2315,14 @@ public:
|
|||||||
size_t stepT(int i=0) const;
|
size_t stepT(int i=0) const;
|
||||||
|
|
||||||
//! overridden forms of Mat::zeros() etc. Data type is omitted, of course
|
//! overridden forms of Mat::zeros() etc. Data type is omitted, of course
|
||||||
static MatExpr zeros(int rows, int cols);
|
CV_NODISCARD_STD static MatExpr zeros(int rows, int cols);
|
||||||
static MatExpr zeros(Size size);
|
CV_NODISCARD_STD static MatExpr zeros(Size size);
|
||||||
static MatExpr zeros(int _ndims, const int* _sizes);
|
CV_NODISCARD_STD static MatExpr zeros(int _ndims, const int* _sizes);
|
||||||
static MatExpr ones(int rows, int cols);
|
CV_NODISCARD_STD static MatExpr ones(int rows, int cols);
|
||||||
static MatExpr ones(Size size);
|
CV_NODISCARD_STD static MatExpr ones(Size size);
|
||||||
static MatExpr ones(int _ndims, const int* _sizes);
|
CV_NODISCARD_STD static MatExpr ones(int _ndims, const int* _sizes);
|
||||||
static MatExpr eye(int rows, int cols);
|
CV_NODISCARD_STD static MatExpr eye(int rows, int cols);
|
||||||
static MatExpr eye(Size size);
|
CV_NODISCARD_STD static MatExpr eye(Size size);
|
||||||
|
|
||||||
//! some more overridden methods
|
//! some more overridden methods
|
||||||
Mat_& adjustROI( int dtop, int dbottom, int dleft, int dright );
|
Mat_& adjustROI( int dtop, int dbottom, int dleft, int dright );
|
||||||
@@ -2469,10 +2469,10 @@ public:
|
|||||||
//! <0 - a diagonal from the lower half)
|
//! <0 - a diagonal from the lower half)
|
||||||
UMat diag(int d=0) const;
|
UMat diag(int d=0) const;
|
||||||
//! constructs a square diagonal matrix which main diagonal is vector "d"
|
//! constructs a square diagonal matrix which main diagonal is vector "d"
|
||||||
static UMat diag(const UMat& d);
|
CV_NODISCARD_STD static UMat diag(const UMat& d);
|
||||||
|
|
||||||
//! returns deep copy of the matrix, i.e. the data is copied
|
//! returns deep copy of the matrix, i.e. the data is copied
|
||||||
UMat clone() const CV_NODISCARD;
|
CV_NODISCARD_STD UMat clone() const;
|
||||||
//! copies the matrix content to "m".
|
//! copies the matrix content to "m".
|
||||||
// It calls m.create(this->size(), this->type()).
|
// It calls m.create(this->size(), this->type()).
|
||||||
void copyTo( OutputArray m ) const;
|
void copyTo( OutputArray m ) const;
|
||||||
@@ -2503,14 +2503,14 @@ public:
|
|||||||
double dot(InputArray m) const;
|
double dot(InputArray m) const;
|
||||||
|
|
||||||
//! Matlab-style matrix initialization
|
//! Matlab-style matrix initialization
|
||||||
static UMat zeros(int rows, int cols, int type);
|
CV_NODISCARD_STD static UMat zeros(int rows, int cols, int type);
|
||||||
static UMat zeros(Size size, int type);
|
CV_NODISCARD_STD static UMat zeros(Size size, int type);
|
||||||
static UMat zeros(int ndims, const int* sz, int type);
|
CV_NODISCARD_STD static UMat zeros(int ndims, const int* sz, int type);
|
||||||
static UMat ones(int rows, int cols, int type);
|
CV_NODISCARD_STD static UMat ones(int rows, int cols, int type);
|
||||||
static UMat ones(Size size, int type);
|
CV_NODISCARD_STD static UMat ones(Size size, int type);
|
||||||
static UMat ones(int ndims, const int* sz, int type);
|
CV_NODISCARD_STD static UMat ones(int ndims, const int* sz, int type);
|
||||||
static UMat eye(int rows, int cols, int type);
|
CV_NODISCARD_STD static UMat eye(int rows, int cols, int type);
|
||||||
static UMat eye(Size size, int type);
|
CV_NODISCARD_STD static UMat eye(Size size, int type);
|
||||||
|
|
||||||
//! allocates new matrix data unless the matrix already has specified size and type.
|
//! allocates new matrix data unless the matrix already has specified size and type.
|
||||||
// previous data is unreferenced if needed.
|
// previous data is unreferenced if needed.
|
||||||
@@ -2767,7 +2767,7 @@ public:
|
|||||||
SparseMat& operator = (const Mat& m);
|
SparseMat& operator = (const Mat& m);
|
||||||
|
|
||||||
//! creates full copy of the matrix
|
//! creates full copy of the matrix
|
||||||
SparseMat clone() const CV_NODISCARD;
|
CV_NODISCARD_STD SparseMat clone() const;
|
||||||
|
|
||||||
//! copies all the data to the destination matrix. All the previous content of m is erased
|
//! copies all the data to the destination matrix. All the previous content of m is erased
|
||||||
void copyTo( SparseMat& m ) const;
|
void copyTo( SparseMat& m ) const;
|
||||||
@@ -3004,7 +3004,7 @@ public:
|
|||||||
SparseMat_& operator = (const Mat& m);
|
SparseMat_& operator = (const Mat& m);
|
||||||
|
|
||||||
//! makes full copy of the matrix. All the elements are duplicated
|
//! makes full copy of the matrix. All the elements are duplicated
|
||||||
SparseMat_ clone() const CV_NODISCARD;
|
CV_NODISCARD_STD SparseMat_ clone() const;
|
||||||
//! equivalent to cv::SparseMat::create(dims, _sizes, DataType<_Tp>::type)
|
//! equivalent to cv::SparseMat::create(dims, _sizes, DataType<_Tp>::type)
|
||||||
void create(int dims, const int* _sizes);
|
void create(int dims, const int* _sizes);
|
||||||
//! converts sparse matrix to the old-style CvSparseMat. All the elements are copied
|
//! converts sparse matrix to the old-style CvSparseMat. All the elements are copied
|
||||||
|
|||||||
@@ -146,22 +146,22 @@ public:
|
|||||||
Matx(std::initializer_list<_Tp>); //!< initialize from an initializer list
|
Matx(std::initializer_list<_Tp>); //!< initialize from an initializer list
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
static Matx all(_Tp alpha);
|
CV_NODISCARD_STD static Matx all(_Tp alpha);
|
||||||
static Matx zeros();
|
CV_NODISCARD_STD static Matx zeros();
|
||||||
static Matx ones();
|
CV_NODISCARD_STD static Matx ones();
|
||||||
static Matx eye();
|
CV_NODISCARD_STD static Matx eye();
|
||||||
static Matx diag(const diag_type& d);
|
CV_NODISCARD_STD static Matx diag(const diag_type& d);
|
||||||
/** @brief Generates uniformly distributed random numbers
|
/** @brief Generates uniformly distributed random numbers
|
||||||
@param a Range boundary.
|
@param a Range boundary.
|
||||||
@param b The other range boundary (boundaries don't have to be ordered, the lower boundary is inclusive,
|
@param b The other range boundary (boundaries don't have to be ordered, the lower boundary is inclusive,
|
||||||
the upper one is exclusive).
|
the upper one is exclusive).
|
||||||
*/
|
*/
|
||||||
static Matx randu(_Tp a, _Tp b);
|
CV_NODISCARD_STD static Matx randu(_Tp a, _Tp b);
|
||||||
/** @brief Generates normally distributed random numbers
|
/** @brief Generates normally distributed random numbers
|
||||||
@param a Mean value.
|
@param a Mean value.
|
||||||
@param b Standard deviation.
|
@param b Standard deviation.
|
||||||
*/
|
*/
|
||||||
static Matx randn(_Tp a, _Tp b);
|
CV_NODISCARD_STD static Matx randn(_Tp a, _Tp b);
|
||||||
|
|
||||||
//! dot product computed with the default precision
|
//! dot product computed with the default precision
|
||||||
_Tp dot(const Matx<_Tp, m, n>& v) const;
|
_Tp dot(const Matx<_Tp, m, n>& v) const;
|
||||||
|
|||||||
@@ -562,7 +562,9 @@ public:
|
|||||||
i = set(i, a6); i = set(i, a7); i = set(i, a8); i = set(i, a9); i = set(i, a10); i = set(i, a11);
|
i = set(i, a6); i = set(i, a7); i = set(i, a8); i = set(i, a9); i = set(i, a10); i = set(i, a11);
|
||||||
i = set(i, a12); i = set(i, a13); i = set(i, a14); set(i, a15); return *this;
|
i = set(i, a12); i = set(i, a13); i = set(i, a14); set(i, a15); return *this;
|
||||||
}
|
}
|
||||||
/** @brief Run the OpenCL kernel.
|
|
||||||
|
/** @brief Run the OpenCL kernel (globalsize value may be adjusted)
|
||||||
|
|
||||||
@param dims the work problem dimensions. It is the length of globalsize and localsize. It can be either 1, 2 or 3.
|
@param dims the work problem dimensions. It is the length of globalsize and localsize. It can be either 1, 2 or 3.
|
||||||
@param globalsize work items for each dimension. It is not the final globalsize passed to
|
@param globalsize work items for each dimension. It is not the final globalsize passed to
|
||||||
OpenCL. Each dimension will be adjusted to the nearest integer divisible by the corresponding
|
OpenCL. Each dimension will be adjusted to the nearest integer divisible by the corresponding
|
||||||
@@ -571,12 +573,26 @@ public:
|
|||||||
@param localsize work-group size for each dimension.
|
@param localsize work-group size for each dimension.
|
||||||
@param sync specify whether to wait for OpenCL computation to finish before return.
|
@param sync specify whether to wait for OpenCL computation to finish before return.
|
||||||
@param q command queue
|
@param q command queue
|
||||||
|
|
||||||
|
@note Use run_() if your kernel code doesn't support adjusted globalsize.
|
||||||
*/
|
*/
|
||||||
bool run(int dims, size_t globalsize[],
|
bool run(int dims, size_t globalsize[],
|
||||||
size_t localsize[], bool sync, const Queue& q=Queue());
|
size_t localsize[], bool sync, const Queue& q=Queue());
|
||||||
|
|
||||||
|
/** @brief Run the OpenCL kernel
|
||||||
|
*
|
||||||
|
* @param dims the work problem dimensions. It is the length of globalsize and localsize. It can be either 1, 2 or 3.
|
||||||
|
* @param globalsize work items for each dimension. This value is passed to OpenCL without changes.
|
||||||
|
* @param localsize work-group size for each dimension.
|
||||||
|
* @param sync specify whether to wait for OpenCL computation to finish before return.
|
||||||
|
* @param q command queue
|
||||||
|
*/
|
||||||
|
bool run_(int dims, size_t globalsize[], size_t localsize[], bool sync, const Queue& q=Queue());
|
||||||
|
|
||||||
bool runTask(bool sync, const Queue& q=Queue());
|
bool runTask(bool sync, const Queue& q=Queue());
|
||||||
|
|
||||||
/** @brief Similar to synchronized run() call with returning of kernel execution time
|
/** @brief Similar to synchronized run_() call with returning of kernel execution time
|
||||||
|
*
|
||||||
* Separate OpenCL command queue may be used (with CL_QUEUE_PROFILING_ENABLE)
|
* Separate OpenCL command queue may be used (with CL_QUEUE_PROFILING_ENABLE)
|
||||||
* @return Execution time in nanoseconds or negative number on error
|
* @return Execution time in nanoseconds or negative number on error
|
||||||
*/
|
*/
|
||||||
|
|||||||
@@ -7,8 +7,8 @@
|
|||||||
|
|
||||||
#define CV_VERSION_MAJOR 3
|
#define CV_VERSION_MAJOR 3
|
||||||
#define CV_VERSION_MINOR 4
|
#define CV_VERSION_MINOR 4
|
||||||
#define CV_VERSION_REVISION 15
|
#define CV_VERSION_REVISION 16
|
||||||
#define CV_VERSION_STATUS "-pre"
|
#define CV_VERSION_STATUS ""
|
||||||
|
|
||||||
#define CVAUX_STR_EXP(__A) #__A
|
#define CVAUX_STR_EXP(__A) #__A
|
||||||
#define CVAUX_STR(__A) CVAUX_STR_EXP(__A)
|
#define CVAUX_STR(__A) CVAUX_STR_EXP(__A)
|
||||||
|
|||||||
@@ -5,6 +5,7 @@
|
|||||||
#include "precomp.hpp"
|
#include "precomp.hpp"
|
||||||
#include "opencv2/core/bindings_utils.hpp"
|
#include "opencv2/core/bindings_utils.hpp"
|
||||||
#include <sstream>
|
#include <sstream>
|
||||||
|
#include <iomanip>
|
||||||
|
|
||||||
namespace cv { namespace utils {
|
namespace cv { namespace utils {
|
||||||
|
|
||||||
@@ -208,4 +209,50 @@ CV_EXPORTS_W String dumpInputOutputArrayOfArrays(InputOutputArrayOfArrays argume
|
|||||||
return ss.str();
|
return ss.str();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static inline std::ostream& operator<<(std::ostream& os, const cv::Rect& rect)
|
||||||
|
{
|
||||||
|
return os << "[x=" << rect.x << ", y=" << rect.y << ", w=" << rect.width << ", h=" << rect.height << ']';
|
||||||
|
}
|
||||||
|
|
||||||
|
template <class T, class Formatter>
|
||||||
|
static inline String dumpVector(const std::vector<T>& vec, Formatter format)
|
||||||
|
{
|
||||||
|
std::ostringstream oss("[", std::ios::ate);
|
||||||
|
if (!vec.empty())
|
||||||
|
{
|
||||||
|
oss << format << vec[0];
|
||||||
|
for (std::size_t i = 1; i < vec.size(); ++i)
|
||||||
|
{
|
||||||
|
oss << ", " << format << vec[i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
oss << "]";
|
||||||
|
return oss.str();
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline std::ostream& noFormat(std::ostream& os)
|
||||||
|
{
|
||||||
|
return os;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline std::ostream& floatFormat(std::ostream& os)
|
||||||
|
{
|
||||||
|
return os << std::fixed << std::setprecision(2);
|
||||||
|
}
|
||||||
|
|
||||||
|
String dumpVectorOfInt(const std::vector<int>& vec)
|
||||||
|
{
|
||||||
|
return dumpVector(vec, &noFormat);
|
||||||
|
}
|
||||||
|
|
||||||
|
String dumpVectorOfDouble(const std::vector<double>& vec)
|
||||||
|
{
|
||||||
|
return dumpVector(vec, &floatFormat);
|
||||||
|
}
|
||||||
|
|
||||||
|
String dumpVectorOfRect(const std::vector<Rect>& vec)
|
||||||
|
{
|
||||||
|
return dumpVector(vec, &noFormat);
|
||||||
|
}
|
||||||
|
|
||||||
}} // namespace
|
}} // namespace
|
||||||
|
|||||||
@@ -24,11 +24,6 @@
|
|||||||
|
|
||||||
#ifdef HAVE_OPENCL
|
#ifdef HAVE_OPENCL
|
||||||
|
|
||||||
#include <sstream>
|
|
||||||
#include "opencl_kernels_core.hpp"
|
|
||||||
#include "opencv2/core/opencl/runtime/opencl_clamdblas.hpp"
|
|
||||||
#include "opencv2/core/opencl/runtime/opencl_core.hpp"
|
|
||||||
|
|
||||||
namespace cv
|
namespace cv
|
||||||
{
|
{
|
||||||
|
|
||||||
@@ -37,52 +32,75 @@ static bool intel_gpu_gemm(
|
|||||||
UMat B, Size sizeB,
|
UMat B, Size sizeB,
|
||||||
UMat D, Size sizeD,
|
UMat D, Size sizeD,
|
||||||
double alpha, double beta,
|
double alpha, double beta,
|
||||||
bool atrans, bool btrans)
|
bool atrans, bool btrans,
|
||||||
|
bool& isPropagatedC2D
|
||||||
|
)
|
||||||
{
|
{
|
||||||
CV_UNUSED(sizeB);
|
CV_UNUSED(sizeB);
|
||||||
|
|
||||||
int M = sizeD.height, N = sizeD.width, K = ((atrans)? sizeA.height : sizeA.width);
|
int M = sizeD.height, N = sizeD.width, K = ((atrans)? sizeA.height : sizeA.width);
|
||||||
|
|
||||||
std::string kernelName;
|
if (M < 4 || N < 4 || K < 4) // vload4
|
||||||
bool ret = true;
|
return false;
|
||||||
|
|
||||||
size_t lx = 8, ly = 4;
|
CV_LOG_VERBOSE(NULL, 0, "M=" << M << " N=" << N << " K=" << K);
|
||||||
size_t dx = 4, dy = 8;
|
|
||||||
|
std::string kernelName;
|
||||||
|
|
||||||
|
unsigned int lx = 8, ly = 4;
|
||||||
|
unsigned int dx = 4, dy = 8;
|
||||||
|
|
||||||
if(!atrans && !btrans)
|
if(!atrans && !btrans)
|
||||||
{
|
{
|
||||||
|
|
||||||
if (M % 32 == 0 && N % 32 == 0 && K % 16 == 0)
|
if (M % 32 == 0 && N % 32 == 0 && K % 16 == 0)
|
||||||
{
|
{
|
||||||
kernelName = "intelblas_gemm_buffer_NN_sp";
|
kernelName = "intelblas_gemm_buffer_NN_sp";
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
if (M % 2 != 0)
|
||||||
|
return false;
|
||||||
|
// vload4(0, dst_write0) - 4 cols
|
||||||
|
// multiply by lx: 8
|
||||||
|
if (N % (4*8) != 0)
|
||||||
|
return false;
|
||||||
kernelName = "intelblas_gemm_buffer_NN";
|
kernelName = "intelblas_gemm_buffer_NN";
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
else if(atrans && !btrans)
|
else if(atrans && !btrans)
|
||||||
{
|
{
|
||||||
|
if (M % 32 != 0)
|
||||||
|
return false;
|
||||||
|
if (N % 32 != 0)
|
||||||
|
return false;
|
||||||
kernelName = "intelblas_gemm_buffer_TN";
|
kernelName = "intelblas_gemm_buffer_TN";
|
||||||
}
|
}
|
||||||
else if(!atrans && btrans)
|
else if(!atrans && btrans)
|
||||||
{
|
{
|
||||||
|
if (K % 4 != 0)
|
||||||
|
return false;
|
||||||
kernelName = "intelblas_gemm_buffer_NT";
|
kernelName = "intelblas_gemm_buffer_NT";
|
||||||
ly = 16;
|
ly = 16;
|
||||||
dx = 1;
|
dx = 1;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
if (M % 32 != 0)
|
||||||
|
return false;
|
||||||
|
if (N % 32 != 0)
|
||||||
|
return false;
|
||||||
|
if (K % 16 != 0)
|
||||||
|
return false;
|
||||||
kernelName = "intelblas_gemm_buffer_TT";
|
kernelName = "intelblas_gemm_buffer_TT";
|
||||||
}
|
}
|
||||||
|
|
||||||
const size_t gx = (size_t)(N + dx - 1) / dx;
|
CV_LOG_DEBUG(NULL, "kernel: " << kernelName << " (M=" << M << " N=" << N << " K=" << K << ")");
|
||||||
const size_t gy = (size_t)(M + dy - 1) / dy;
|
|
||||||
|
const size_t gx = divUp((size_t)N, dx);
|
||||||
|
const size_t gy = divUp((size_t)M, dy);
|
||||||
|
|
||||||
size_t local[] = {lx, ly, 1};
|
size_t local[] = {lx, ly, 1};
|
||||||
size_t global[] = {(gx + lx - 1) / lx * lx, (gy + ly - 1) / ly * ly, 1};
|
size_t global[] = {roundUp(gx, lx), roundUp(gy, ly), 1};
|
||||||
|
|
||||||
int stride = (M * N < 1024 * 1024) ? 10000000 : 256;
|
|
||||||
|
|
||||||
ocl::Queue q;
|
ocl::Queue q;
|
||||||
String errmsg;
|
String errmsg;
|
||||||
@@ -110,10 +128,13 @@ static bool intel_gpu_gemm(
|
|||||||
(int)(D.step / sizeof(float))
|
(int)(D.step / sizeof(float))
|
||||||
);
|
);
|
||||||
|
|
||||||
ret = k.run(2, global, local, false, q);
|
bool ret = k.run(2, global, local, false, q);
|
||||||
|
return ret;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
int stride = (M * N < 1024 * 1024) ? 10000000 : 256;
|
||||||
|
|
||||||
for(int start_index = 0; start_index < K; start_index += stride)
|
for(int start_index = 0; start_index < K; start_index += stride)
|
||||||
{
|
{
|
||||||
ocl::Kernel k(kernelName.c_str(), program);
|
ocl::Kernel k(kernelName.c_str(), program);
|
||||||
@@ -132,12 +153,16 @@ static bool intel_gpu_gemm(
|
|||||||
(int) start_index, // 14 start_index
|
(int) start_index, // 14 start_index
|
||||||
stride);
|
stride);
|
||||||
|
|
||||||
ret = k.run(2, global, local, false, q);
|
bool ret = k.run(2, global, local, false, q);
|
||||||
if (!ret) return ret;
|
if (!ret)
|
||||||
|
{
|
||||||
|
if (start_index != 0)
|
||||||
|
isPropagatedC2D = false; // D array content is changed, need to rewrite
|
||||||
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
return ret;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
} // namespace cv
|
} // namespace cv
|
||||||
|
|||||||
@@ -42,6 +42,8 @@
|
|||||||
//M*/
|
//M*/
|
||||||
|
|
||||||
#include "precomp.hpp"
|
#include "precomp.hpp"
|
||||||
|
#include <opencv2/core/utils/logger.hpp>
|
||||||
|
|
||||||
#include "opencl_kernels_core.hpp"
|
#include "opencl_kernels_core.hpp"
|
||||||
#include "opencv2/core/opencl/runtime/opencl_clamdblas.hpp"
|
#include "opencv2/core/opencl/runtime/opencl_clamdblas.hpp"
|
||||||
#include "opencv2/core/opencl/runtime/opencl_core.hpp"
|
#include "opencv2/core/opencl/runtime/opencl_core.hpp"
|
||||||
@@ -155,10 +157,12 @@ static bool ocl_gemm_amdblas( InputArray matA, InputArray matB, double alpha,
|
|||||||
static bool ocl_gemm( InputArray matA, InputArray matB, double alpha,
|
static bool ocl_gemm( InputArray matA, InputArray matB, double alpha,
|
||||||
InputArray matC, double beta, OutputArray matD, int flags )
|
InputArray matC, double beta, OutputArray matD, int flags )
|
||||||
{
|
{
|
||||||
int depth = matA.depth(), cn = matA.channels();
|
int type = matA.type();
|
||||||
int type = CV_MAKETYPE(depth, cn);
|
int depth = CV_MAT_DEPTH(type);
|
||||||
|
int cn = CV_MAT_CN(type);
|
||||||
|
|
||||||
CV_Assert_N( type == matB.type(), (type == CV_32FC1 || type == CV_64FC1 || type == CV_32FC2 || type == CV_64FC2) );
|
CV_CheckTypeEQ(type, matB.type(), "");
|
||||||
|
CV_CheckType(type, type == CV_32FC1 || type == CV_64FC1 || type == CV_32FC2 || type == CV_64FC2, "");
|
||||||
|
|
||||||
const ocl::Device & dev = ocl::Device::getDefault();
|
const ocl::Device & dev = ocl::Device::getDefault();
|
||||||
bool doubleSupport = dev.doubleFPConfig() > 0;
|
bool doubleSupport = dev.doubleFPConfig() > 0;
|
||||||
@@ -170,88 +174,103 @@ static bool ocl_gemm( InputArray matA, InputArray matB, double alpha,
|
|||||||
Size sizeA = matA.size(), sizeB = matB.size(), sizeC = haveC ? matC.size() : Size(0, 0);
|
Size sizeA = matA.size(), sizeB = matB.size(), sizeC = haveC ? matC.size() : Size(0, 0);
|
||||||
bool atrans = (flags & GEMM_1_T) != 0, btrans = (flags & GEMM_2_T) != 0, ctrans = (flags & GEMM_3_T) != 0;
|
bool atrans = (flags & GEMM_1_T) != 0, btrans = (flags & GEMM_2_T) != 0, ctrans = (flags & GEMM_3_T) != 0;
|
||||||
|
|
||||||
CV_Assert( !haveC || matC.type() == type );
|
if (haveC)
|
||||||
|
CV_CheckTypeEQ(type, matC.type(), "");
|
||||||
|
|
||||||
|
Size sizeD(((btrans) ? sizeB.height : sizeB.width),
|
||||||
|
((atrans) ? sizeA.width : sizeA.height));
|
||||||
|
|
||||||
|
if (atrans)
|
||||||
|
sizeA = Size(sizeA.height, sizeA.width);
|
||||||
|
if (btrans)
|
||||||
|
sizeB = Size(sizeB.height, sizeB.width);
|
||||||
|
if (haveC && ctrans)
|
||||||
|
sizeC = Size(sizeC.height, sizeC.width);
|
||||||
|
|
||||||
|
CV_CheckEQ(sizeA.width, sizeB.height, "");
|
||||||
|
if (haveC)
|
||||||
|
CV_CheckEQ(sizeC, sizeD, "");
|
||||||
|
|
||||||
|
UMat A = matA.getUMat();
|
||||||
|
UMat B = matB.getUMat();
|
||||||
|
|
||||||
Size sizeD(((btrans)? sizeB.height : sizeB.width),
|
|
||||||
((atrans)? sizeA.width : sizeA.height));
|
|
||||||
matD.create(sizeD, type);
|
matD.create(sizeD, type);
|
||||||
|
UMat D = matD.getUMat();
|
||||||
|
|
||||||
UMat A = matA.getUMat(), B = matB.getUMat(), D = matD.getUMat();
|
bool isPropagatedC2D = false; // D content is updated with C / C.t()
|
||||||
|
|
||||||
|
if (dev.intelSubgroupsSupport() && (depth == CV_32F) && cn == 1)
|
||||||
if (!dev.intelSubgroupsSupport() || (depth == CV_64F) || cn != 1)
|
|
||||||
{
|
|
||||||
String opts;
|
|
||||||
|
|
||||||
if (atrans)
|
|
||||||
sizeA = Size(sizeA.height, sizeA.width);
|
|
||||||
if (btrans)
|
|
||||||
sizeB = Size(sizeB.height, sizeB.width);
|
|
||||||
if (haveC && ctrans)
|
|
||||||
sizeC = Size(sizeC.height, sizeC.width);
|
|
||||||
|
|
||||||
CV_Assert( sizeA.width == sizeB.height && (!haveC || sizeC == sizeD) );
|
|
||||||
|
|
||||||
int max_wg_size = (int)dev.maxWorkGroupSize();
|
|
||||||
int block_size = (max_wg_size / (32*cn) < 32) ? (max_wg_size / (16*cn) < 16) ? (max_wg_size / (8*cn) < 8) ? 1 : 8 : 16 : 32;
|
|
||||||
|
|
||||||
if (atrans)
|
|
||||||
A = A.t();
|
|
||||||
|
|
||||||
if (btrans)
|
|
||||||
B = B.t();
|
|
||||||
|
|
||||||
if (haveC)
|
|
||||||
ctrans ? transpose(matC, D) : matC.copyTo(D);
|
|
||||||
|
|
||||||
int vectorWidths[] = { 4, 4, 2, 2, 1, 4, cn, -1 };
|
|
||||||
int kercn = ocl::checkOptimalVectorWidth(vectorWidths, B, D);
|
|
||||||
|
|
||||||
opts += format(" -D T=%s -D T1=%s -D WT=%s -D cn=%d -D kercn=%d -D LOCAL_SIZE=%d%s%s%s",
|
|
||||||
ocl::typeToStr(type), ocl::typeToStr(depth), ocl::typeToStr(CV_MAKETYPE(depth, kercn)),
|
|
||||||
cn, kercn, block_size,
|
|
||||||
(sizeA.width % block_size !=0) ? " -D NO_MULT" : "",
|
|
||||||
haveC ? " -D HAVE_C" : "",
|
|
||||||
doubleSupport ? " -D DOUBLE_SUPPORT" : "");
|
|
||||||
|
|
||||||
ocl::Kernel k("gemm", cv::ocl::core::gemm_oclsrc, opts);
|
|
||||||
if (k.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
if (depth == CV_64F)
|
|
||||||
k.args(ocl::KernelArg::ReadOnlyNoSize(A),
|
|
||||||
ocl::KernelArg::ReadOnlyNoSize(B, cn, kercn),
|
|
||||||
ocl::KernelArg::ReadWrite(D, cn, kercn),
|
|
||||||
sizeA.width, alpha, beta);
|
|
||||||
else
|
|
||||||
k.args(ocl::KernelArg::ReadOnlyNoSize(A),
|
|
||||||
ocl::KernelArg::ReadOnlyNoSize(B, cn, kercn),
|
|
||||||
ocl::KernelArg::ReadWrite(D, cn, kercn),
|
|
||||||
sizeA.width, (float)alpha, (float)beta);
|
|
||||||
|
|
||||||
size_t globalsize[2] = { (size_t)sizeD.width * cn / kercn, (size_t)sizeD.height};
|
|
||||||
size_t localsize[2] = { (size_t)block_size, (size_t)block_size};
|
|
||||||
|
|
||||||
return k.run(2, globalsize, block_size!=1 ? localsize : NULL, false);
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
{
|
||||||
if (haveC && beta != 0.0)
|
if (haveC && beta != 0.0)
|
||||||
{
|
{
|
||||||
ctrans ? transpose(matC, D) : matC.copyTo(D);
|
ctrans ? transpose(matC, D) : matC.copyTo(D);
|
||||||
|
isPropagatedC2D = true;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
beta = 0.0;
|
beta = 0.0;
|
||||||
}
|
}
|
||||||
|
|
||||||
return intel_gpu_gemm(A, sizeA,
|
bool res = intel_gpu_gemm(A, matA.size(),
|
||||||
B, sizeB,
|
B, matB.size(),
|
||||||
D, sizeD,
|
D, sizeD,
|
||||||
alpha,
|
alpha,
|
||||||
beta,
|
beta,
|
||||||
atrans, btrans);
|
atrans, btrans,
|
||||||
|
isPropagatedC2D);
|
||||||
|
if (res)
|
||||||
|
return true;
|
||||||
|
// fallback on generic OpenCL code
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (sizeD.width < 8 || sizeD.height < 8)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
String opts;
|
||||||
|
|
||||||
|
int wg_size = (int)dev.maxWorkGroupSize();
|
||||||
|
int sizeDmin = std::min(sizeD.width, sizeD.height);
|
||||||
|
wg_size = std::min(wg_size, sizeDmin * sizeDmin);
|
||||||
|
int block_size = (wg_size / (32*cn) < 32) ? (wg_size / (16*cn) < 16) ? (wg_size / (8*cn) < 8) ? 1 : 8 : 16 : 32;
|
||||||
|
|
||||||
|
if (atrans)
|
||||||
|
A = A.t();
|
||||||
|
|
||||||
|
if (btrans)
|
||||||
|
B = B.t();
|
||||||
|
|
||||||
|
if (haveC && !isPropagatedC2D)
|
||||||
|
ctrans ? transpose(matC, D) : matC.copyTo(D);
|
||||||
|
|
||||||
|
int vectorWidths[] = { 4, 4, 2, 2, 1, 4, cn, -1 };
|
||||||
|
int kercn = ocl::checkOptimalVectorWidth(vectorWidths, B, D);
|
||||||
|
|
||||||
|
opts += format(" -D T=%s -D T1=%s -D WT=%s -D cn=%d -D kercn=%d -D LOCAL_SIZE=%d%s%s%s",
|
||||||
|
ocl::typeToStr(type), ocl::typeToStr(depth), ocl::typeToStr(CV_MAKETYPE(depth, kercn)),
|
||||||
|
cn, kercn, block_size,
|
||||||
|
(sizeA.width % block_size !=0) ? " -D NO_MULT" : "",
|
||||||
|
haveC ? " -D HAVE_C" : "",
|
||||||
|
doubleSupport ? " -D DOUBLE_SUPPORT" : "");
|
||||||
|
|
||||||
|
ocl::Kernel k("gemm", cv::ocl::core::gemm_oclsrc, opts);
|
||||||
|
if (k.empty())
|
||||||
|
return false;
|
||||||
|
|
||||||
|
if (depth == CV_64F)
|
||||||
|
k.args(ocl::KernelArg::ReadOnlyNoSize(A),
|
||||||
|
ocl::KernelArg::ReadOnlyNoSize(B, cn, kercn),
|
||||||
|
ocl::KernelArg::ReadWrite(D, cn, kercn),
|
||||||
|
sizeA.width, alpha, beta);
|
||||||
|
else
|
||||||
|
k.args(ocl::KernelArg::ReadOnlyNoSize(A),
|
||||||
|
ocl::KernelArg::ReadOnlyNoSize(B, cn, kercn),
|
||||||
|
ocl::KernelArg::ReadWrite(D, cn, kercn),
|
||||||
|
sizeA.width, (float)alpha, (float)beta);
|
||||||
|
|
||||||
|
size_t globalsize[2] = { (size_t)sizeD.width * cn / kercn, (size_t)sizeD.height};
|
||||||
|
size_t localsize[2] = { (size_t)block_size, (size_t)block_size};
|
||||||
|
|
||||||
|
return k.run(2, globalsize, block_size !=1 ? localsize : NULL, false);
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
|||||||
@@ -749,18 +749,17 @@ Mat::Mat(const Mat& m, const Rect& roi)
|
|||||||
data += roi.x*esz;
|
data += roi.x*esz;
|
||||||
CV_Assert( 0 <= roi.x && 0 <= roi.width && roi.x + roi.width <= m.cols &&
|
CV_Assert( 0 <= roi.x && 0 <= roi.width && roi.x + roi.width <= m.cols &&
|
||||||
0 <= roi.y && 0 <= roi.height && roi.y + roi.height <= m.rows );
|
0 <= roi.y && 0 <= roi.height && roi.y + roi.height <= m.rows );
|
||||||
if( u )
|
|
||||||
CV_XADD(&u->refcount, 1);
|
|
||||||
if( roi.width < m.cols || roi.height < m.rows )
|
if( roi.width < m.cols || roi.height < m.rows )
|
||||||
flags |= SUBMATRIX_FLAG;
|
flags |= SUBMATRIX_FLAG;
|
||||||
|
|
||||||
step[0] = m.step[0]; step[1] = esz;
|
step[0] = m.step[0]; step[1] = esz;
|
||||||
updateContinuityFlag();
|
updateContinuityFlag();
|
||||||
|
|
||||||
|
addref();
|
||||||
if( rows <= 0 || cols <= 0 )
|
if( rows <= 0 || cols <= 0 )
|
||||||
{
|
{
|
||||||
release();
|
|
||||||
rows = cols = 0;
|
rows = cols = 0;
|
||||||
|
release();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+94
-32
@@ -76,8 +76,11 @@
|
|||||||
#undef CV__ALLOCATOR_STATS_LOG
|
#undef CV__ALLOCATOR_STATS_LOG
|
||||||
|
|
||||||
#define CV_OPENCL_ALWAYS_SHOW_BUILD_LOG 0
|
#define CV_OPENCL_ALWAYS_SHOW_BUILD_LOG 0
|
||||||
|
#define CV_OPENCL_SHOW_BUILD_OPTIONS 0
|
||||||
|
#define CV_OPENCL_SHOW_BUILD_KERNELS 0
|
||||||
|
|
||||||
#define CV_OPENCL_SHOW_RUN_KERNELS 0
|
#define CV_OPENCL_SHOW_RUN_KERNELS 0
|
||||||
|
#define CV_OPENCL_SYNC_RUN_KERNELS 0
|
||||||
#define CV_OPENCL_TRACE_CHECK 0
|
#define CV_OPENCL_TRACE_CHECK 0
|
||||||
|
|
||||||
#define CV_OPENCL_VALIDATE_BINARY_PROGRAMS 1
|
#define CV_OPENCL_VALIDATE_BINARY_PROGRAMS 1
|
||||||
@@ -1667,7 +1670,7 @@ static bool parseOpenCLDeviceConfiguration(const std::string& configurationStr,
|
|||||||
split(configurationStr, ':', parts);
|
split(configurationStr, ':', parts);
|
||||||
if (parts.size() > 3)
|
if (parts.size() > 3)
|
||||||
{
|
{
|
||||||
std::cerr << "ERROR: Invalid configuration string for OpenCL device" << std::endl;
|
CV_LOG_ERROR(NULL, "OpenCL: Invalid configuration string for OpenCL device: " << configurationStr);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
if (parts.size() > 2)
|
if (parts.size() > 2)
|
||||||
@@ -1684,22 +1687,20 @@ static bool parseOpenCLDeviceConfiguration(const std::string& configurationStr,
|
|||||||
}
|
}
|
||||||
|
|
||||||
#if defined WINRT || defined _WIN32_WCE
|
#if defined WINRT || defined _WIN32_WCE
|
||||||
static cl_device_id selectOpenCLDevice()
|
static cl_device_id selectOpenCLDevice(const char* configuration = NULL)
|
||||||
{
|
{
|
||||||
|
CV_UNUSED(configuration)
|
||||||
return NULL;
|
return NULL;
|
||||||
}
|
}
|
||||||
#else
|
#else
|
||||||
// std::tolower is int->int
|
static cl_device_id selectOpenCLDevice(const char* configuration = NULL)
|
||||||
static char char_tolower(char ch)
|
|
||||||
{
|
|
||||||
return (char)std::tolower((int)ch);
|
|
||||||
}
|
|
||||||
static cl_device_id selectOpenCLDevice()
|
|
||||||
{
|
{
|
||||||
std::string platform, deviceName;
|
std::string platform, deviceName;
|
||||||
std::vector<std::string> deviceTypes;
|
std::vector<std::string> deviceTypes;
|
||||||
|
|
||||||
const char* configuration = getenv("OPENCV_OPENCL_DEVICE");
|
if (!configuration)
|
||||||
|
configuration = getenv("OPENCV_OPENCL_DEVICE");
|
||||||
|
|
||||||
if (configuration &&
|
if (configuration &&
|
||||||
(strcmp(configuration, "disabled") == 0 ||
|
(strcmp(configuration, "disabled") == 0 ||
|
||||||
!parseOpenCLDeviceConfiguration(std::string(configuration), platform, deviceTypes, deviceName)
|
!parseOpenCLDeviceConfiguration(std::string(configuration), platform, deviceTypes, deviceName)
|
||||||
@@ -1744,22 +1745,24 @@ static cl_device_id selectOpenCLDevice()
|
|||||||
platforms.resize(numPlatforms);
|
platforms.resize(numPlatforms);
|
||||||
}
|
}
|
||||||
|
|
||||||
int selectedPlatform = -1;
|
|
||||||
if (platform.length() > 0)
|
if (platform.length() > 0)
|
||||||
{
|
{
|
||||||
for (size_t i = 0; i < platforms.size(); i++)
|
for (std::vector<cl_platform_id>::iterator currentPlatform = platforms.begin(); currentPlatform != platforms.end();)
|
||||||
{
|
{
|
||||||
std::string name;
|
std::string name;
|
||||||
CV_OCL_DBG_CHECK(getStringInfo(clGetPlatformInfo, platforms[i], CL_PLATFORM_NAME, name));
|
CV_OCL_DBG_CHECK(getStringInfo(clGetPlatformInfo, *currentPlatform, CL_PLATFORM_NAME, name));
|
||||||
if (name.find(platform) != std::string::npos)
|
if (name.find(platform) != std::string::npos)
|
||||||
{
|
{
|
||||||
selectedPlatform = (int)i;
|
++currentPlatform;
|
||||||
break;
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
currentPlatform = platforms.erase(currentPlatform);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (selectedPlatform == -1)
|
if (platforms.size() == 0)
|
||||||
{
|
{
|
||||||
std::cerr << "ERROR: Can't find OpenCL platform by name: " << platform << std::endl;
|
CV_LOG_ERROR(NULL, "OpenCL: Can't find OpenCL platform by name: " << platform);
|
||||||
goto not_found;
|
goto not_found;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1778,7 +1781,7 @@ static cl_device_id selectOpenCLDevice()
|
|||||||
{
|
{
|
||||||
int deviceType = 0;
|
int deviceType = 0;
|
||||||
std::string tempStrDeviceType = deviceTypes[t];
|
std::string tempStrDeviceType = deviceTypes[t];
|
||||||
std::transform(tempStrDeviceType.begin(), tempStrDeviceType.end(), tempStrDeviceType.begin(), char_tolower);
|
std::transform(tempStrDeviceType.begin(), tempStrDeviceType.end(), tempStrDeviceType.begin(), details::char_tolower);
|
||||||
|
|
||||||
if (tempStrDeviceType == "gpu" || tempStrDeviceType == "dgpu" || tempStrDeviceType == "igpu")
|
if (tempStrDeviceType == "gpu" || tempStrDeviceType == "dgpu" || tempStrDeviceType == "igpu")
|
||||||
deviceType = Device::TYPE_GPU;
|
deviceType = Device::TYPE_GPU;
|
||||||
@@ -1790,17 +1793,15 @@ static cl_device_id selectOpenCLDevice()
|
|||||||
deviceType = Device::TYPE_ALL;
|
deviceType = Device::TYPE_ALL;
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
std::cerr << "ERROR: Unsupported device type for OpenCL device (GPU, CPU, ACCELERATOR): " << deviceTypes[t] << std::endl;
|
CV_LOG_ERROR(NULL, "OpenCL: Unsupported device type for OpenCL device (GPU, CPU, ACCELERATOR): " << deviceTypes[t]);
|
||||||
goto not_found;
|
goto not_found;
|
||||||
}
|
}
|
||||||
|
|
||||||
std::vector<cl_device_id> devices; // TODO Use clReleaseDevice to cleanup
|
std::vector<cl_device_id> devices;
|
||||||
for (int i = selectedPlatform >= 0 ? selectedPlatform : 0;
|
for (std::vector<cl_platform_id>::iterator currentPlatform = platforms.begin(); currentPlatform != platforms.end(); ++currentPlatform)
|
||||||
(selectedPlatform >= 0 ? i == selectedPlatform : true) && (i < (int)platforms.size());
|
|
||||||
i++)
|
|
||||||
{
|
{
|
||||||
cl_uint count = 0;
|
cl_uint count = 0;
|
||||||
cl_int status = clGetDeviceIDs(platforms[i], deviceType, 0, NULL, &count);
|
cl_int status = clGetDeviceIDs(*currentPlatform, deviceType, 0, NULL, &count);
|
||||||
if (!(status == CL_SUCCESS || status == CL_DEVICE_NOT_FOUND))
|
if (!(status == CL_SUCCESS || status == CL_DEVICE_NOT_FOUND))
|
||||||
{
|
{
|
||||||
CV_OCL_DBG_CHECK_RESULT(status, "clGetDeviceIDs get count");
|
CV_OCL_DBG_CHECK_RESULT(status, "clGetDeviceIDs get count");
|
||||||
@@ -1809,7 +1810,7 @@ static cl_device_id selectOpenCLDevice()
|
|||||||
continue;
|
continue;
|
||||||
size_t base = devices.size();
|
size_t base = devices.size();
|
||||||
devices.resize(base + count);
|
devices.resize(base + count);
|
||||||
status = clGetDeviceIDs(platforms[i], deviceType, count, &devices[base], &count);
|
status = clGetDeviceIDs(*currentPlatform, deviceType, count, &devices[base], &count);
|
||||||
if (!(status == CL_SUCCESS || status == CL_DEVICE_NOT_FOUND))
|
if (!(status == CL_SUCCESS || status == CL_DEVICE_NOT_FOUND))
|
||||||
{
|
{
|
||||||
CV_OCL_DBG_CHECK_RESULT(status, "clGetDeviceIDs get IDs");
|
CV_OCL_DBG_CHECK_RESULT(status, "clGetDeviceIDs get IDs");
|
||||||
@@ -1841,13 +1842,16 @@ not_found:
|
|||||||
if (!configuration)
|
if (!configuration)
|
||||||
return NULL; // suppress messages on stderr
|
return NULL; // suppress messages on stderr
|
||||||
|
|
||||||
std::cerr << "ERROR: Requested OpenCL device not found, check configuration: " << configuration << std::endl
|
std::ostringstream msg;
|
||||||
<< " Platform: " << (platform.length() == 0 ? "any" : platform) << std::endl
|
msg << "ERROR: Requested OpenCL device not found, check configuration: '" << configuration << "'" << std::endl
|
||||||
<< " Device types: ";
|
<< " Platform: " << (platform.length() == 0 ? "any" : platform) << std::endl
|
||||||
|
<< " Device types:";
|
||||||
for (size_t t = 0; t < deviceTypes.size(); t++)
|
for (size_t t = 0; t < deviceTypes.size(); t++)
|
||||||
std::cerr << deviceTypes[t] << " ";
|
msg << ' ' << deviceTypes[t];
|
||||||
|
|
||||||
std::cerr << std::endl << " Device name: " << (deviceName.length() == 0 ? "any" : deviceName) << std::endl;
|
msg << std::endl << " Device name: " << (deviceName.length() == 0 ? "any" : deviceName);
|
||||||
|
|
||||||
|
CV_LOG_ERROR(NULL, msg.str());
|
||||||
return NULL;
|
return NULL;
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
@@ -2774,19 +2778,33 @@ struct Kernel::Impl
|
|||||||
|
|
||||||
void cleanupUMats()
|
void cleanupUMats()
|
||||||
{
|
{
|
||||||
|
bool exceptionOccurred = false;
|
||||||
for( int i = 0; i < MAX_ARRS; i++ )
|
for( int i = 0; i < MAX_ARRS; i++ )
|
||||||
|
{
|
||||||
if( u[i] )
|
if( u[i] )
|
||||||
{
|
{
|
||||||
if( CV_XADD(&u[i]->urefcount, -1) == 1 )
|
if( CV_XADD(&u[i]->urefcount, -1) == 1 )
|
||||||
{
|
{
|
||||||
u[i]->flags |= UMatData::ASYNC_CLEANUP;
|
u[i]->flags |= UMatData::ASYNC_CLEANUP;
|
||||||
u[i]->currAllocator->deallocate(u[i]);
|
try
|
||||||
|
{
|
||||||
|
u[i]->currAllocator->deallocate(u[i]);
|
||||||
|
}
|
||||||
|
catch(const std::exception& exc)
|
||||||
|
{
|
||||||
|
// limited by legacy before C++11, therefore log and
|
||||||
|
// remember some exception occurred to throw below
|
||||||
|
CV_LOG_ERROR(NULL, "OCL: Unexpected C++ exception in OpenCL Kernel::Impl::cleanupUMats(): " << exc.what());
|
||||||
|
exceptionOccurred = true;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
u[i] = 0;
|
u[i] = 0;
|
||||||
}
|
}
|
||||||
|
}
|
||||||
nu = 0;
|
nu = 0;
|
||||||
haveTempDstUMats = false;
|
haveTempDstUMats = false;
|
||||||
haveTempSrcUMats = false;
|
haveTempSrcUMats = false;
|
||||||
|
CV_Assert(!exceptionOccurred);
|
||||||
}
|
}
|
||||||
|
|
||||||
void addUMat(const UMat& m, bool dst)
|
void addUMat(const UMat& m, bool dst)
|
||||||
@@ -2817,8 +2835,16 @@ struct Kernel::Impl
|
|||||||
void finit(cl_event e)
|
void finit(cl_event e)
|
||||||
{
|
{
|
||||||
CV_UNUSED(e);
|
CV_UNUSED(e);
|
||||||
cleanupUMats();
|
|
||||||
isInProgress = false;
|
isInProgress = false;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
cleanupUMats();
|
||||||
|
}
|
||||||
|
catch(...)
|
||||||
|
{
|
||||||
|
release();
|
||||||
|
throw;
|
||||||
|
}
|
||||||
release();
|
release();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2959,6 +2985,10 @@ bool Kernel::empty() const
|
|||||||
|
|
||||||
static cv::String dumpValue(size_t sz, const void* p)
|
static cv::String dumpValue(size_t sz, const void* p)
|
||||||
{
|
{
|
||||||
|
if (!p)
|
||||||
|
return "NULL";
|
||||||
|
if (sz == 2)
|
||||||
|
return cv::format("%d / %uu / 0x%04x", *(short*)p, *(unsigned short*)p, *(short*)p);
|
||||||
if (sz == 4)
|
if (sz == 4)
|
||||||
return cv::format("%d / %uu / 0x%08x / %g", *(int*)p, *(int*)p, *(int*)p, *(float*)p);
|
return cv::format("%d / %uu / 0x%08x / %g", *(int*)p, *(int*)p, *(int*)p, *(float*)p);
|
||||||
if (sz == 8)
|
if (sz == 8)
|
||||||
@@ -3131,6 +3161,14 @@ bool Kernel::run(int dims, size_t _globalsize[], size_t _localsize[],
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
bool Kernel::run_(int dims, size_t _globalsize[], size_t _localsize[],
|
||||||
|
bool sync, const Queue& q)
|
||||||
|
{
|
||||||
|
CV_Assert(p);
|
||||||
|
return p->run(dims, _globalsize, _localsize, sync, NULL, q);
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
static bool isRaiseErrorOnReuseAsyncKernel()
|
static bool isRaiseErrorOnReuseAsyncKernel()
|
||||||
{
|
{
|
||||||
static bool initialized = false;
|
static bool initialized = false;
|
||||||
@@ -3171,6 +3209,10 @@ bool Kernel::Impl::run(int dims, size_t globalsize[], size_t localsize[],
|
|||||||
return false; // OpenCV 5.0: raise error
|
return false; // OpenCV 5.0: raise error
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#if CV_OPENCL_SYNC_RUN_KERNELS
|
||||||
|
sync = true;
|
||||||
|
#endif
|
||||||
|
|
||||||
cl_command_queue qq = getQueue(q);
|
cl_command_queue qq = getQueue(q);
|
||||||
if (haveTempDstUMats)
|
if (haveTempDstUMats)
|
||||||
sync = true;
|
sync = true;
|
||||||
@@ -3601,7 +3643,28 @@ struct Program::Impl
|
|||||||
if (!param_buildExtraOptions.empty())
|
if (!param_buildExtraOptions.empty())
|
||||||
buildflags = joinBuildOptions(buildflags, param_buildExtraOptions);
|
buildflags = joinBuildOptions(buildflags, param_buildExtraOptions);
|
||||||
}
|
}
|
||||||
|
#if CV_OPENCL_SHOW_BUILD_OPTIONS
|
||||||
|
CV_LOG_INFO(NULL, "OpenCL program '" << sourceModule_ << "/" << sourceName_ << "' options:" << buildflags);
|
||||||
|
#endif
|
||||||
compile(ctx, src_, errmsg);
|
compile(ctx, src_, errmsg);
|
||||||
|
#if CV_OPENCL_SHOW_BUILD_KERNELS
|
||||||
|
if (handle)
|
||||||
|
{
|
||||||
|
size_t retsz = 0;
|
||||||
|
char kernels_buffer[4096] = {0};
|
||||||
|
cl_int result = clGetProgramInfo(handle, CL_PROGRAM_KERNEL_NAMES, sizeof(kernels_buffer), &kernels_buffer[0], &retsz);
|
||||||
|
CV_OCL_DBG_CHECK_RESULT(result, cv::format("clGetProgramInfo(CL_PROGRAM_KERNEL_NAMES: %s/%s)", sourceModule_.c_str(), sourceName_.c_str()).c_str());
|
||||||
|
if (result == CL_SUCCESS && retsz < sizeof(kernels_buffer))
|
||||||
|
{
|
||||||
|
kernels_buffer[retsz] = 0;
|
||||||
|
CV_LOG_INFO(NULL, "OpenCL program '" << sourceModule_ << "/" << sourceName_ << "' kernels: '" << kernels_buffer << "'");
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
CV_LOG_ERROR(NULL, "OpenCL program '" << sourceModule_ << "/" << sourceName_ << "' can't retrieve kernel names!");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
bool compile(const Context& ctx, const ProgramSource::Impl* src_, String& errmsg)
|
bool compile(const Context& ctx, const ProgramSource::Impl* src_, String& errmsg)
|
||||||
@@ -3833,7 +3896,6 @@ struct Program::Impl
|
|||||||
CV_LOG_INFO(NULL, result << ": Kernels='" << kernels_buffer << "'");
|
CV_LOG_INFO(NULL, result << ": Kernels='" << kernels_buffer << "'");
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
}
|
}
|
||||||
return handle != NULL;
|
return handle != NULL;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -392,6 +392,15 @@ __kernel void intelblas_gemm_buffer_NN(
|
|||||||
#define TILE_N 8
|
#define TILE_N 8
|
||||||
#define SLM_BLOCK 512
|
#define SLM_BLOCK 512
|
||||||
|
|
||||||
|
/*
|
||||||
|
A K B.t() K D N
|
||||||
|
----------- ----------- -----------
|
||||||
|
| | | | | |
|
||||||
|
M | | x N | | => M | |
|
||||||
|
| | | | | |
|
||||||
|
----------- ----------- -----------
|
||||||
|
*/
|
||||||
|
|
||||||
__attribute__((reqd_work_group_size(8, LWG_HEIGHT, 1)))
|
__attribute__((reqd_work_group_size(8, LWG_HEIGHT, 1)))
|
||||||
__kernel void intelblas_gemm_buffer_NT(
|
__kernel void intelblas_gemm_buffer_NT(
|
||||||
const __global float *src0, int off0,
|
const __global float *src0, int off0,
|
||||||
@@ -422,59 +431,79 @@ __kernel void intelblas_gemm_buffer_NT(
|
|||||||
float8 dot06 = 0.f;
|
float8 dot06 = 0.f;
|
||||||
float8 dot07 = 0.f;
|
float8 dot07 = 0.f;
|
||||||
|
|
||||||
float4 brow0;
|
const int dst_row = (global_y * TILE_M);
|
||||||
float4 brow1;
|
__global float *dst_write0 = dst + global_x + dst_row * ldC + offd;
|
||||||
float4 brow2;
|
|
||||||
float4 brow3;
|
|
||||||
float4 brow4;
|
|
||||||
float4 brow5;
|
|
||||||
float4 brow6;
|
|
||||||
float4 brow7;
|
|
||||||
|
|
||||||
__global float *dst_write0 = dst + local_x * VEC_SIZE + ( group_x * TILE_N ) + ( group_y * LWG_HEIGHT * TILE_M + local_y * TILE_M) * ldC + offd;
|
const __global float *src0_read00 = src0 + off0;
|
||||||
|
const int a_row_base = global_y * TILE_M;
|
||||||
|
const int a_col_base = local_x * (TILE_K / 8); // <= TILE_K - 4
|
||||||
|
|
||||||
const __global float *src0_read = src0 + local_x * ( TILE_K / 8 ) + ( group_y * LWG_HEIGHT * TILE_M + local_y * TILE_M ) * ldA + off0;
|
const __global float *src1_read00 = src1 + off1;
|
||||||
|
const int b_row_base = (group_x * TILE_N);
|
||||||
const __global float *src1_read0 = src1 + ( group_x * TILE_N ) * ldB + off1;
|
//const int b_col_base = 0;
|
||||||
|
|
||||||
__local float slm_brow[8 * SLM_BLOCK];
|
__local float slm_brow[8 * SLM_BLOCK];
|
||||||
__local float* slm_brow0;
|
|
||||||
|
|
||||||
int local_index = mad24(local_y, 8, local_x) * 4;
|
int local_index = mad24(local_y, 8, local_x) * 4;
|
||||||
int w;
|
int w = 0;
|
||||||
for(int b_tile = 0; b_tile < K; b_tile += SLM_BLOCK) {
|
for (int b_tile = 0; b_tile < K; b_tile += SLM_BLOCK)
|
||||||
|
{
|
||||||
|
#define UPDATE_BROW(_row) \
|
||||||
|
{ \
|
||||||
|
float4 brow; \
|
||||||
|
int b_row = b_row_base + _row; \
|
||||||
|
int b_col = b_tile + local_index; \
|
||||||
|
if (b_row < N && b_col <= K - 4 /*vload4*/) \
|
||||||
|
brow = vload4(0, src1_read00 + mad24(b_row, ldB, b_col)); \
|
||||||
|
else \
|
||||||
|
brow = (float4)0; \
|
||||||
|
vstore4(brow, 0, slm_brow + mad24(_row, SLM_BLOCK, local_index)); \
|
||||||
|
}
|
||||||
|
|
||||||
barrier(CLK_LOCAL_MEM_FENCE);
|
barrier(CLK_LOCAL_MEM_FENCE);
|
||||||
vstore4(vload4(0, src1_read0 + mad24(0, ldB, local_index)), 0, slm_brow + mad24(0, SLM_BLOCK, local_index));
|
UPDATE_BROW(0);
|
||||||
vstore4(vload4(0, src1_read0 + mad24(1, ldB, local_index)), 0, slm_brow + mad24(1, SLM_BLOCK, local_index));
|
UPDATE_BROW(1);
|
||||||
vstore4(vload4(0, src1_read0 + mad24(2, ldB, local_index)), 0, slm_brow + mad24(2, SLM_BLOCK, local_index));
|
UPDATE_BROW(2);
|
||||||
vstore4(vload4(0, src1_read0 + mad24(3, ldB, local_index)), 0, slm_brow + mad24(3, SLM_BLOCK, local_index));
|
UPDATE_BROW(3);
|
||||||
vstore4(vload4(0, src1_read0 + mad24(4, ldB, local_index)), 0, slm_brow + mad24(4, SLM_BLOCK, local_index));
|
UPDATE_BROW(4);
|
||||||
vstore4(vload4(0, src1_read0 + mad24(5, ldB, local_index)), 0, slm_brow + mad24(5, SLM_BLOCK, local_index));
|
UPDATE_BROW(5);
|
||||||
vstore4(vload4(0, src1_read0 + mad24(6, ldB, local_index)), 0, slm_brow + mad24(6, SLM_BLOCK, local_index));
|
UPDATE_BROW(6);
|
||||||
vstore4(vload4(0, src1_read0 + mad24(7, ldB, local_index)), 0, slm_brow + mad24(7, SLM_BLOCK, local_index));
|
UPDATE_BROW(7);
|
||||||
barrier(CLK_LOCAL_MEM_FENCE);
|
barrier(CLK_LOCAL_MEM_FENCE);
|
||||||
|
#undef UPDATE_BROW
|
||||||
|
|
||||||
slm_brow0 = slm_brow + local_x * (TILE_K / 8);
|
for (int k_tile_offset = 0; k_tile_offset < SLM_BLOCK; k_tile_offset += TILE_K)
|
||||||
w = b_tile;
|
{
|
||||||
int end_w = min(b_tile + SLM_BLOCK, K);
|
int a_col = a_col_base + b_tile + k_tile_offset;
|
||||||
while( w + TILE_K <= end_w ) {
|
|
||||||
float4 arow;
|
|
||||||
|
|
||||||
brow0 = vload4(0, slm_brow0 + 0 * SLM_BLOCK);
|
if (a_col > K - 4 /*vload4*/)
|
||||||
brow1 = vload4(0, slm_brow0 + 1 * SLM_BLOCK);
|
break;
|
||||||
brow2 = vload4(0, slm_brow0 + 2 * SLM_BLOCK);
|
|
||||||
brow3 = vload4(0, slm_brow0 + 3 * SLM_BLOCK);
|
|
||||||
brow4 = vload4(0, slm_brow0 + 4 * SLM_BLOCK);
|
|
||||||
brow5 = vload4(0, slm_brow0 + 5 * SLM_BLOCK);
|
|
||||||
brow6 = vload4(0, slm_brow0 + 6 * SLM_BLOCK);
|
|
||||||
brow7 = vload4(0, slm_brow0 + 7 * SLM_BLOCK);
|
|
||||||
|
|
||||||
#define MM_DOT_PRODUCT(_row,_dot) \
|
int slm_brow_col = a_col_base + k_tile_offset; // <= SLM_BLOCK - 4
|
||||||
arow = vload4(0, src0_read + _row * ldA); \
|
#define READ_SLM_BROW(_row) \
|
||||||
_dot = mad( (float8)(arow.x), (float8)(brow0.x, brow1.x, brow2.x, brow3.x, brow4.x, brow5.x, brow6.x, brow7.x), _dot ); \
|
float4 brow##_row = vload4(0, slm_brow + mad24(_row, SLM_BLOCK, slm_brow_col));
|
||||||
_dot = mad( (float8)(arow.y), (float8)(brow0.y, brow1.y, brow2.y, brow3.y, brow4.y, brow5.y, brow6.y, brow7.y), _dot ); \
|
|
||||||
_dot = mad( (float8)(arow.z), (float8)(brow0.z, brow1.z, brow2.z, brow3.z, brow4.z, brow5.z, brow6.z, brow7.z), _dot ); \
|
READ_SLM_BROW(0);
|
||||||
_dot = mad( (float8)(arow.w), (float8)(brow0.w, brow1.w, brow2.w, brow3.w, brow4.w, brow5.w, brow6.w, brow7.w), _dot );
|
READ_SLM_BROW(1);
|
||||||
|
READ_SLM_BROW(2);
|
||||||
|
READ_SLM_BROW(3);
|
||||||
|
READ_SLM_BROW(4);
|
||||||
|
READ_SLM_BROW(5);
|
||||||
|
READ_SLM_BROW(6);
|
||||||
|
READ_SLM_BROW(7);
|
||||||
|
#undef READ_SLM_BROW
|
||||||
|
|
||||||
|
#define MM_DOT_PRODUCT(_row,_dot) \
|
||||||
|
{ \
|
||||||
|
int a_row = a_row_base + _row; \
|
||||||
|
if (a_row < M) { \
|
||||||
|
float4 arow = vload4(0, src0_read00 + mad24(a_row, ldA, a_col)); \
|
||||||
|
_dot = mad( (float8)(arow.x), (float8)(brow0.x, brow1.x, brow2.x, brow3.x, brow4.x, brow5.x, brow6.x, brow7.x), _dot ); \
|
||||||
|
_dot = mad( (float8)(arow.y), (float8)(brow0.y, brow1.y, brow2.y, brow3.y, brow4.y, brow5.y, brow6.y, brow7.y), _dot ); \
|
||||||
|
_dot = mad( (float8)(arow.z), (float8)(brow0.z, brow1.z, brow2.z, brow3.z, brow4.z, brow5.z, brow6.z, brow7.z), _dot ); \
|
||||||
|
_dot = mad( (float8)(arow.w), (float8)(brow0.w, brow1.w, brow2.w, brow3.w, brow4.w, brow5.w, brow6.w, brow7.w), _dot ); \
|
||||||
|
} \
|
||||||
|
}
|
||||||
|
|
||||||
MM_DOT_PRODUCT(0,dot00);
|
MM_DOT_PRODUCT(0,dot00);
|
||||||
MM_DOT_PRODUCT(1,dot01);
|
MM_DOT_PRODUCT(1,dot01);
|
||||||
@@ -485,53 +514,7 @@ __kernel void intelblas_gemm_buffer_NT(
|
|||||||
MM_DOT_PRODUCT(6,dot06);
|
MM_DOT_PRODUCT(6,dot06);
|
||||||
MM_DOT_PRODUCT(7,dot07);
|
MM_DOT_PRODUCT(7,dot07);
|
||||||
#undef MM_DOT_PRODUCT
|
#undef MM_DOT_PRODUCT
|
||||||
|
|
||||||
src0_read += TILE_K;
|
|
||||||
slm_brow0 += TILE_K;
|
|
||||||
w += TILE_K;
|
|
||||||
}
|
}
|
||||||
src1_read0 += SLM_BLOCK;
|
|
||||||
}
|
|
||||||
|
|
||||||
if(w < K) {
|
|
||||||
float4 arow;
|
|
||||||
|
|
||||||
#define READ_BROW(_brow,_row) \
|
|
||||||
_brow = vload4(0, slm_brow0 + _row * SLM_BLOCK); \
|
|
||||||
_brow.x = (mad24(local_x, 4, w) < K) ? _brow.x : 0.0f; \
|
|
||||||
_brow.y = (mad24(local_x, 4, w + 1) < K) ? _brow.y : 0.0f; \
|
|
||||||
_brow.z = (mad24(local_x, 4, w + 2) < K) ? _brow.z : 0.0f; \
|
|
||||||
_brow.w = (mad24(local_x, 4, w + 3) < K) ? _brow.w : 0.0f;
|
|
||||||
|
|
||||||
READ_BROW(brow0,0);
|
|
||||||
READ_BROW(brow1,1);
|
|
||||||
READ_BROW(brow2,2);
|
|
||||||
READ_BROW(brow3,3);
|
|
||||||
READ_BROW(brow4,4);
|
|
||||||
READ_BROW(brow5,5);
|
|
||||||
READ_BROW(brow6,6);
|
|
||||||
READ_BROW(brow7,7);
|
|
||||||
|
|
||||||
#define MM_DOT_PRODUCT(_row,_dot) \
|
|
||||||
arow = vload4(0, src0_read + _row * ldA); \
|
|
||||||
arow.x = (mad24(local_x, 4, w) < K) ? arow.x : 0.0f; \
|
|
||||||
arow.y = (mad24(local_x, 4, w + 1) < K) ? arow.y : 0.0f; \
|
|
||||||
arow.z = (mad24(local_x, 4, w + 2) < K) ? arow.z : 0.0f; \
|
|
||||||
arow.w = (mad24(local_x, 4, w + 3) < K) ? arow.w : 0.0f; \
|
|
||||||
_dot = mad( (float8)(arow.x), (float8)(brow0.x, brow1.x, brow2.x, brow3.x, brow4.x, brow5.x, brow6.x, brow7.x), _dot ); \
|
|
||||||
_dot = mad( (float8)(arow.y), (float8)(brow0.y, brow1.y, brow2.y, brow3.y, brow4.y, brow5.y, brow6.y, brow7.y), _dot ); \
|
|
||||||
_dot = mad( (float8)(arow.z), (float8)(brow0.z, brow1.z, brow2.z, brow3.z, brow4.z, brow5.z, brow6.z, brow7.z), _dot ); \
|
|
||||||
_dot = mad( (float8)(arow.w), (float8)(brow0.w, brow1.w, brow2.w, brow3.w, brow4.w, brow5.w, brow6.w, brow7.w), _dot );
|
|
||||||
|
|
||||||
MM_DOT_PRODUCT(0,dot00);
|
|
||||||
MM_DOT_PRODUCT(1,dot01);
|
|
||||||
MM_DOT_PRODUCT(2,dot02);
|
|
||||||
MM_DOT_PRODUCT(3,dot03);
|
|
||||||
MM_DOT_PRODUCT(4,dot04);
|
|
||||||
MM_DOT_PRODUCT(5,dot05);
|
|
||||||
MM_DOT_PRODUCT(6,dot06);
|
|
||||||
MM_DOT_PRODUCT(7,dot07);
|
|
||||||
#undef MM_DOT_PRODUCT
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#define REDUCE(_dot) \
|
#define REDUCE(_dot) \
|
||||||
@@ -572,21 +555,22 @@ __kernel void intelblas_gemm_buffer_NT(
|
|||||||
output = (local_x == 5) ? _dot.s5 : output; \
|
output = (local_x == 5) ? _dot.s5 : output; \
|
||||||
output = (local_x == 6) ? _dot.s6 : output; \
|
output = (local_x == 6) ? _dot.s6 : output; \
|
||||||
output = (local_x == 7) ? _dot.s7 : output; \
|
output = (local_x == 7) ? _dot.s7 : output; \
|
||||||
if (beta != 0.0) \
|
if (beta != 0.0f) \
|
||||||
dst_write0[0] = mad(output, (float)alpha, ((float)beta * dst_write0[0])); \
|
dst_write0[0] = mad(output, (float)alpha, ((float)beta * dst_write0[0])); \
|
||||||
else \
|
else \
|
||||||
dst_write0[0] = output * (float)alpha; \
|
dst_write0[0] = output * (float)alpha; \
|
||||||
dst_write0 += ldC;
|
dst_write0 += ldC;
|
||||||
|
|
||||||
if(global_x < N && global_y * 8 < M) {
|
if (global_x < N && dst_row < M)
|
||||||
OUTPUT(dot00);
|
{
|
||||||
if(mad24(global_y, 8, 1) < M) { OUTPUT(dot01); }
|
/*if (dst_row + 0 < M)*/ { OUTPUT(dot00); }
|
||||||
if(mad24(global_y, 8, 2) < M) { OUTPUT(dot02); }
|
if (dst_row + 1 < M) { OUTPUT(dot01); }
|
||||||
if(mad24(global_y, 8, 3) < M) { OUTPUT(dot03); }
|
if (dst_row + 2 < M) { OUTPUT(dot02); }
|
||||||
if(mad24(global_y, 8, 4) < M) { OUTPUT(dot04); }
|
if (dst_row + 3 < M) { OUTPUT(dot03); }
|
||||||
if(mad24(global_y, 8, 5) < M) { OUTPUT(dot05); }
|
if (dst_row + 4 < M) { OUTPUT(dot04); }
|
||||||
if(mad24(global_y, 8, 6) < M) { OUTPUT(dot06); }
|
if (dst_row + 5 < M) { OUTPUT(dot05); }
|
||||||
if(mad24(global_y, 8, 7) < M) { OUTPUT(dot07); }
|
if (dst_row + 6 < M) { OUTPUT(dot06); }
|
||||||
|
if (dst_row + 7 < M) { OUTPUT(dot07); }
|
||||||
}
|
}
|
||||||
#undef OUTPUT
|
#undef OUTPUT
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -53,9 +53,9 @@
|
|||||||
#undef abs
|
#undef abs
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if defined __linux__ || defined __APPLE__ || defined __GLIBC__ \
|
#if defined __unix__ || defined __APPLE__ || defined __GLIBC__ \
|
||||||
|| defined __HAIKU__ || defined __EMSCRIPTEN__ || defined __FreeBSD__ \
|
|| defined __HAIKU__ || defined __EMSCRIPTEN__ \
|
||||||
|| defined __OpenBSD__
|
|| defined __FreeBSD__ || defined __NetBSD__ || defined __OpenBSD__
|
||||||
#include <unistd.h>
|
#include <unistd.h>
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <sys/types.h>
|
#include <sys/types.h>
|
||||||
|
|||||||
@@ -615,6 +615,11 @@ void ThreadPool::run(const Range& range, const ParallelLoopBody& body, double ns
|
|||||||
{
|
{
|
||||||
WorkerThread& thread = *(threads[i].get());
|
WorkerThread& thread = *(threads[i].get());
|
||||||
if (
|
if (
|
||||||
|
#if defined(__clang__) && defined(__has_feature)
|
||||||
|
#if __has_feature(thread_sanitizer)
|
||||||
|
1 || // Robust workaround to avoid data race warning of `thread.job`
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
#if !defined(CV_USE_GLOBAL_WORKERS_COND_VAR)
|
#if !defined(CV_USE_GLOBAL_WORKERS_COND_VAR)
|
||||||
thread.isActive ||
|
thread.isActive ||
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
+102
-61
@@ -53,6 +53,18 @@
|
|||||||
#include <opencv2/core/utils/tls.hpp>
|
#include <opencv2/core/utils/tls.hpp>
|
||||||
#include <opencv2/core/utils/instrumentation.hpp>
|
#include <opencv2/core/utils/instrumentation.hpp>
|
||||||
|
|
||||||
|
#ifndef OPENCV_WITH_THREAD_SANITIZER
|
||||||
|
#if defined(__clang__) && defined(__has_feature)
|
||||||
|
#if __has_feature(thread_sanitizer)
|
||||||
|
#define OPENCV_WITH_THREAD_SANITIZER 1
|
||||||
|
#include <atomic> // assume C++11
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
#ifndef OPENCV_WITH_THREAD_SANITIZER
|
||||||
|
#define OPENCV_WITH_THREAD_SANITIZER 0
|
||||||
|
#endif
|
||||||
|
|
||||||
namespace cv {
|
namespace cv {
|
||||||
|
|
||||||
static void _initSystem()
|
static void _initSystem()
|
||||||
@@ -114,10 +126,14 @@ void* allocSingletonNewBuffer(size_t size) { return malloc(size); }
|
|||||||
#include <cstdlib> // std::abort
|
#include <cstdlib> // std::abort
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if defined __ANDROID__ || defined __linux__ || defined __FreeBSD__ || defined __OpenBSD__ || defined __HAIKU__
|
#if defined __ANDROID__ || defined __unix__ || defined __FreeBSD__ || defined __OpenBSD__ || defined __HAIKU__
|
||||||
# include <unistd.h>
|
# include <unistd.h>
|
||||||
# include <fcntl.h>
|
# include <fcntl.h>
|
||||||
|
#if defined __QNXNTO__
|
||||||
|
# include <sys/elf.h>
|
||||||
|
#else
|
||||||
# include <elf.h>
|
# include <elf.h>
|
||||||
|
#endif
|
||||||
#if defined __ANDROID__ || defined __linux__
|
#if defined __ANDROID__ || defined __linux__
|
||||||
# include <linux/auxvec.h>
|
# include <linux/auxvec.h>
|
||||||
#endif
|
#endif
|
||||||
@@ -128,7 +144,7 @@ void* allocSingletonNewBuffer(size_t size) { return malloc(size); }
|
|||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
|
||||||
#if (defined __ppc64__ || defined __PPC64__) && defined __linux__
|
#if (defined __ppc64__ || defined __PPC64__) && defined __unix__
|
||||||
# include "sys/auxv.h"
|
# include "sys/auxv.h"
|
||||||
# ifndef AT_HWCAP2
|
# ifndef AT_HWCAP2
|
||||||
# define AT_HWCAP2 26
|
# define AT_HWCAP2 26
|
||||||
@@ -229,7 +245,7 @@ std::wstring GetTempFileNameWinRT(std::wstring prefix)
|
|||||||
#include "omp.h"
|
#include "omp.h"
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if defined __linux__ || defined __APPLE__ || defined __EMSCRIPTEN__ || defined __FreeBSD__ || defined __GLIBC__ || defined __HAIKU__
|
#if defined __unix__ || defined __APPLE__ || defined __EMSCRIPTEN__ || defined __FreeBSD__ || defined __GLIBC__ || defined __HAIKU__
|
||||||
#include <unistd.h>
|
#include <unistd.h>
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <sys/types.h>
|
#include <sys/types.h>
|
||||||
@@ -529,7 +545,7 @@ struct HWFeatures
|
|||||||
}
|
}
|
||||||
#endif // CV_CPUID_X86
|
#endif // CV_CPUID_X86
|
||||||
|
|
||||||
#if defined __ANDROID__ || defined __linux__
|
#if defined __ANDROID__ || defined __linux__ || defined __FreeBSD__
|
||||||
#ifdef __aarch64__
|
#ifdef __aarch64__
|
||||||
have[CV_CPU_NEON] = true;
|
have[CV_CPU_NEON] = true;
|
||||||
have[CV_CPU_FP16] = true;
|
have[CV_CPU_FP16] = true;
|
||||||
@@ -555,7 +571,7 @@ struct HWFeatures
|
|||||||
CV_LOG_INFO(NULL, "- FP16 instructions is NOT enabled via build flags");
|
CV_LOG_INFO(NULL, "- FP16 instructions is NOT enabled via build flags");
|
||||||
#endif
|
#endif
|
||||||
#endif
|
#endif
|
||||||
#elif defined __arm__
|
#elif defined __arm__ && !defined __FreeBSD__
|
||||||
int cpufile = open("/proc/self/auxv", O_RDONLY);
|
int cpufile = open("/proc/self/auxv", O_RDONLY);
|
||||||
|
|
||||||
if (cpufile >= 0)
|
if (cpufile >= 0)
|
||||||
@@ -591,7 +607,7 @@ struct HWFeatures
|
|||||||
have[CV_CPU_MSA] = true;
|
have[CV_CPU_MSA] = true;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if (defined __ppc64__ || defined __PPC64__) && defined __linux__
|
#if (defined __ppc64__ || defined __PPC64__) && defined __unix__
|
||||||
unsigned int hwcap = getauxval(AT_HWCAP);
|
unsigned int hwcap = getauxval(AT_HWCAP);
|
||||||
if (hwcap & PPC_FEATURE_HAS_VSX) {
|
if (hwcap & PPC_FEATURE_HAS_VSX) {
|
||||||
hwcap = getauxval(AT_HWCAP2);
|
hwcap = getauxval(AT_HWCAP2);
|
||||||
@@ -804,12 +820,12 @@ int64 getTickCount(void)
|
|||||||
LARGE_INTEGER counter;
|
LARGE_INTEGER counter;
|
||||||
QueryPerformanceCounter( &counter );
|
QueryPerformanceCounter( &counter );
|
||||||
return (int64)counter.QuadPart;
|
return (int64)counter.QuadPart;
|
||||||
#elif defined __linux || defined __linux__
|
#elif defined __MACH__ && defined __APPLE__
|
||||||
|
return (int64)mach_absolute_time();
|
||||||
|
#elif defined __unix__
|
||||||
struct timespec tp;
|
struct timespec tp;
|
||||||
clock_gettime(CLOCK_MONOTONIC, &tp);
|
clock_gettime(CLOCK_MONOTONIC, &tp);
|
||||||
return (int64)tp.tv_sec*1000000000 + tp.tv_nsec;
|
return (int64)tp.tv_sec*1000000000 + tp.tv_nsec;
|
||||||
#elif defined __MACH__ && defined __APPLE__
|
|
||||||
return (int64)mach_absolute_time();
|
|
||||||
#else
|
#else
|
||||||
struct timeval tv;
|
struct timeval tv;
|
||||||
gettimeofday(&tv, NULL);
|
gettimeofday(&tv, NULL);
|
||||||
@@ -823,8 +839,6 @@ double getTickFrequency(void)
|
|||||||
LARGE_INTEGER freq;
|
LARGE_INTEGER freq;
|
||||||
QueryPerformanceFrequency(&freq);
|
QueryPerformanceFrequency(&freq);
|
||||||
return (double)freq.QuadPart;
|
return (double)freq.QuadPart;
|
||||||
#elif defined __linux || defined __linux__
|
|
||||||
return 1e9;
|
|
||||||
#elif defined __MACH__ && defined __APPLE__
|
#elif defined __MACH__ && defined __APPLE__
|
||||||
static double freq = 0;
|
static double freq = 0;
|
||||||
if( freq == 0 )
|
if( freq == 0 )
|
||||||
@@ -834,6 +848,8 @@ double getTickFrequency(void)
|
|||||||
freq = sTimebaseInfo.denom*1e9/sTimebaseInfo.numer;
|
freq = sTimebaseInfo.denom*1e9/sTimebaseInfo.numer;
|
||||||
}
|
}
|
||||||
return freq;
|
return freq;
|
||||||
|
#elif defined __unix__
|
||||||
|
return 1e9;
|
||||||
#else
|
#else
|
||||||
return 1e6;
|
return 1e6;
|
||||||
#endif
|
#endif
|
||||||
@@ -1422,74 +1438,75 @@ namespace details {
|
|||||||
#endif
|
#endif
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
template <class T>
|
|
||||||
class DisposedSingletonMark
|
|
||||||
{
|
|
||||||
private:
|
|
||||||
static bool mark;
|
|
||||||
protected:
|
|
||||||
DisposedSingletonMark() {}
|
|
||||||
~DisposedSingletonMark()
|
|
||||||
{
|
|
||||||
mark = true;
|
|
||||||
}
|
|
||||||
public:
|
|
||||||
static bool isDisposed() { return mark; }
|
|
||||||
};
|
|
||||||
|
|
||||||
// TLS platform abstraction layer
|
// TLS platform abstraction layer
|
||||||
class TlsAbstraction : public DisposedSingletonMark<TlsAbstraction>
|
class TlsAbstraction
|
||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
TlsAbstraction();
|
TlsAbstraction();
|
||||||
~TlsAbstraction();
|
~TlsAbstraction()
|
||||||
void* getData() const
|
|
||||||
{
|
{
|
||||||
if (isDisposed()) // guard: static initialization order fiasco
|
// TlsAbstraction singleton should not be released
|
||||||
return NULL;
|
// There is no reliable way to avoid problems caused by static initialization order fiasco
|
||||||
return getData_();
|
// NB: Do NOT use logging here
|
||||||
}
|
fprintf(stderr, "OpenCV FATAL: TlsAbstraction::~TlsAbstraction() call is not expected\n");
|
||||||
void setData(void *pData)
|
fflush(stderr);
|
||||||
{
|
|
||||||
if (isDisposed()) // guard: static initialization order fiasco
|
|
||||||
return;
|
|
||||||
return setData_(pData);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void* getData() const;
|
||||||
|
void setData(void *pData);
|
||||||
|
|
||||||
|
void releaseSystemResources();
|
||||||
|
|
||||||
private:
|
private:
|
||||||
void* getData_() const;
|
|
||||||
void setData_(void *pData);
|
|
||||||
|
|
||||||
#ifdef _WIN32
|
#ifdef _WIN32
|
||||||
#ifndef WINRT
|
#ifndef WINRT
|
||||||
DWORD tlsKey;
|
DWORD tlsKey;
|
||||||
|
bool disposed;
|
||||||
#endif
|
#endif
|
||||||
#else // _WIN32
|
#else // _WIN32
|
||||||
pthread_key_t tlsKey;
|
pthread_key_t tlsKey;
|
||||||
|
#if OPENCV_WITH_THREAD_SANITIZER
|
||||||
|
std::atomic<bool> disposed;
|
||||||
|
#else
|
||||||
|
bool disposed;
|
||||||
|
#endif
|
||||||
#endif
|
#endif
|
||||||
};
|
};
|
||||||
|
|
||||||
template<> bool DisposedSingletonMark<TlsAbstraction>::mark = false;
|
class TlsAbstractionReleaseGuard
|
||||||
|
|
||||||
static TlsAbstraction& getTlsAbstraction_()
|
|
||||||
{
|
{
|
||||||
static TlsAbstraction g_tls; // disposed in atexit() handlers (required for unregistering our callbacks)
|
TlsAbstraction& tls_;
|
||||||
return g_tls;
|
public:
|
||||||
}
|
TlsAbstractionReleaseGuard(TlsAbstraction& tls) : tls_(tls)
|
||||||
|
{
|
||||||
|
/* nothing */
|
||||||
|
}
|
||||||
|
~TlsAbstractionReleaseGuard()
|
||||||
|
{
|
||||||
|
tls_.releaseSystemResources();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// TODO use reference
|
||||||
static TlsAbstraction* getTlsAbstraction()
|
static TlsAbstraction* getTlsAbstraction()
|
||||||
{
|
{
|
||||||
#ifdef CV_CXX11
|
#ifdef CV_CXX11
|
||||||
static TlsAbstraction* instance = &getTlsAbstraction_();
|
static TlsAbstraction *g_tls = new TlsAbstraction(); // memory leak is intended here to avoid disposing of TLS container
|
||||||
|
static TlsAbstractionReleaseGuard g_tlsReleaseGuard(*g_tls);
|
||||||
#else
|
#else
|
||||||
static TlsAbstraction* volatile instance = NULL;
|
static TlsAbstraction* volatile g_tls = NULL;
|
||||||
if (instance == NULL)
|
if (g_tls == NULL)
|
||||||
{
|
{
|
||||||
cv::AutoLock lock(cv::getInitializationMutex());
|
cv::AutoLock lock(cv::getInitializationMutex());
|
||||||
if (instance == NULL)
|
if (g_tls == NULL)
|
||||||
instance = &getTlsAbstraction_();
|
{
|
||||||
|
g_tls = new TlsAbstraction();
|
||||||
|
static TlsAbstractionReleaseGuard g_tlsReleaseGuard(*g_tls);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
return DisposedSingletonMark<TlsAbstraction>::isDisposed() ? NULL : instance;
|
return g_tls;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -1497,12 +1514,15 @@ static TlsAbstraction* getTlsAbstraction()
|
|||||||
#ifdef WINRT
|
#ifdef WINRT
|
||||||
static __declspec( thread ) void* tlsData = NULL; // using C++11 thread attribute for local thread data
|
static __declspec( thread ) void* tlsData = NULL; // using C++11 thread attribute for local thread data
|
||||||
TlsAbstraction::TlsAbstraction() {}
|
TlsAbstraction::TlsAbstraction() {}
|
||||||
TlsAbstraction::~TlsAbstraction() {}
|
void TlsAbstraction::releaseSystemResources()
|
||||||
void* TlsAbstraction::getData_() const
|
{
|
||||||
|
cv::__termination = true; // DllMain is missing in static builds
|
||||||
|
}
|
||||||
|
void* TlsAbstraction::getData() const
|
||||||
{
|
{
|
||||||
return tlsData;
|
return tlsData;
|
||||||
}
|
}
|
||||||
void TlsAbstraction::setData_(void *pData)
|
void TlsAbstraction::setData(void *pData)
|
||||||
{
|
{
|
||||||
tlsData = pData;
|
tlsData = pData;
|
||||||
}
|
}
|
||||||
@@ -1511,6 +1531,7 @@ void TlsAbstraction::setData_(void *pData)
|
|||||||
static void NTAPI opencv_fls_destructor(void* pData);
|
static void NTAPI opencv_fls_destructor(void* pData);
|
||||||
#endif // CV_USE_FLS
|
#endif // CV_USE_FLS
|
||||||
TlsAbstraction::TlsAbstraction()
|
TlsAbstraction::TlsAbstraction()
|
||||||
|
: disposed(false)
|
||||||
{
|
{
|
||||||
#ifndef CV_USE_FLS
|
#ifndef CV_USE_FLS
|
||||||
tlsKey = TlsAlloc();
|
tlsKey = TlsAlloc();
|
||||||
@@ -1519,8 +1540,10 @@ TlsAbstraction::TlsAbstraction()
|
|||||||
#endif // CV_USE_FLS
|
#endif // CV_USE_FLS
|
||||||
CV_Assert(tlsKey != TLS_OUT_OF_INDEXES);
|
CV_Assert(tlsKey != TLS_OUT_OF_INDEXES);
|
||||||
}
|
}
|
||||||
TlsAbstraction::~TlsAbstraction()
|
void TlsAbstraction::releaseSystemResources()
|
||||||
{
|
{
|
||||||
|
cv::__termination = true; // DllMain is missing in static builds
|
||||||
|
disposed = true;
|
||||||
#ifndef CV_USE_FLS
|
#ifndef CV_USE_FLS
|
||||||
TlsFree(tlsKey);
|
TlsFree(tlsKey);
|
||||||
#else // CV_USE_FLS
|
#else // CV_USE_FLS
|
||||||
@@ -1528,16 +1551,20 @@ TlsAbstraction::~TlsAbstraction()
|
|||||||
#endif // CV_USE_FLS
|
#endif // CV_USE_FLS
|
||||||
tlsKey = TLS_OUT_OF_INDEXES;
|
tlsKey = TLS_OUT_OF_INDEXES;
|
||||||
}
|
}
|
||||||
void* TlsAbstraction::getData_() const
|
void* TlsAbstraction::getData() const
|
||||||
{
|
{
|
||||||
|
if (disposed)
|
||||||
|
return NULL;
|
||||||
#ifndef CV_USE_FLS
|
#ifndef CV_USE_FLS
|
||||||
return TlsGetValue(tlsKey);
|
return TlsGetValue(tlsKey);
|
||||||
#else // CV_USE_FLS
|
#else // CV_USE_FLS
|
||||||
return FlsGetValue(tlsKey);
|
return FlsGetValue(tlsKey);
|
||||||
#endif // CV_USE_FLS
|
#endif // CV_USE_FLS
|
||||||
}
|
}
|
||||||
void TlsAbstraction::setData_(void *pData)
|
void TlsAbstraction::setData(void *pData)
|
||||||
{
|
{
|
||||||
|
if (disposed)
|
||||||
|
return; // no-op
|
||||||
#ifndef CV_USE_FLS
|
#ifndef CV_USE_FLS
|
||||||
CV_Assert(TlsSetValue(tlsKey, pData) == TRUE);
|
CV_Assert(TlsSetValue(tlsKey, pData) == TRUE);
|
||||||
#else // CV_USE_FLS
|
#else // CV_USE_FLS
|
||||||
@@ -1548,11 +1575,14 @@ void TlsAbstraction::setData_(void *pData)
|
|||||||
#else // _WIN32
|
#else // _WIN32
|
||||||
static void opencv_tls_destructor(void* pData);
|
static void opencv_tls_destructor(void* pData);
|
||||||
TlsAbstraction::TlsAbstraction()
|
TlsAbstraction::TlsAbstraction()
|
||||||
|
: disposed(false)
|
||||||
{
|
{
|
||||||
CV_Assert(pthread_key_create(&tlsKey, opencv_tls_destructor) == 0);
|
CV_Assert(pthread_key_create(&tlsKey, opencv_tls_destructor) == 0);
|
||||||
}
|
}
|
||||||
TlsAbstraction::~TlsAbstraction()
|
void TlsAbstraction::releaseSystemResources()
|
||||||
{
|
{
|
||||||
|
cv::__termination = true; // DllMain is missing in static builds
|
||||||
|
disposed = true;
|
||||||
if (pthread_key_delete(tlsKey) != 0)
|
if (pthread_key_delete(tlsKey) != 0)
|
||||||
{
|
{
|
||||||
// Don't use logging here
|
// Don't use logging here
|
||||||
@@ -1560,12 +1590,16 @@ TlsAbstraction::~TlsAbstraction()
|
|||||||
fflush(stderr);
|
fflush(stderr);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
void* TlsAbstraction::getData_() const
|
void* TlsAbstraction::getData() const
|
||||||
{
|
{
|
||||||
|
if (disposed)
|
||||||
|
return NULL;
|
||||||
return pthread_getspecific(tlsKey);
|
return pthread_getspecific(tlsKey);
|
||||||
}
|
}
|
||||||
void TlsAbstraction::setData_(void *pData)
|
void TlsAbstraction::setData(void *pData)
|
||||||
{
|
{
|
||||||
|
if (disposed)
|
||||||
|
return; // no-op
|
||||||
CV_Assert(pthread_setspecific(tlsKey, pData) == 0);
|
CV_Assert(pthread_setspecific(tlsKey, pData) == 0);
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
@@ -1593,6 +1627,7 @@ public:
|
|||||||
TlsStorage() :
|
TlsStorage() :
|
||||||
tlsSlotsSize(0)
|
tlsSlotsSize(0)
|
||||||
{
|
{
|
||||||
|
(void)getTlsAbstraction(); // ensure singeton initialization (for correct order of atexit calls)
|
||||||
tlsSlots.reserve(32);
|
tlsSlots.reserve(32);
|
||||||
threads.reserve(32);
|
threads.reserve(32);
|
||||||
g_isTlsStorageInitialized = true;
|
g_isTlsStorageInitialized = true;
|
||||||
@@ -1830,6 +1865,12 @@ static void WINAPI opencv_fls_destructor(void* pData)
|
|||||||
#endif // CV_USE_FLS
|
#endif // CV_USE_FLS
|
||||||
#endif // _WIN32
|
#endif // _WIN32
|
||||||
|
|
||||||
|
static TlsStorage* const g_force_initialization_of_TlsStorage
|
||||||
|
#if defined __GNUC__
|
||||||
|
__attribute__((unused))
|
||||||
|
#endif
|
||||||
|
= &getTlsStorage();
|
||||||
|
|
||||||
} // namespace details
|
} // namespace details
|
||||||
using namespace details;
|
using namespace details;
|
||||||
|
|
||||||
|
|||||||
@@ -540,13 +540,26 @@ UMat Mat::getUMat(int accessFlags, UMatUsageFlags usageFlags) const
|
|||||||
CV_XADD(&(u->refcount), 1);
|
CV_XADD(&(u->refcount), 1);
|
||||||
CV_XADD(&(u->urefcount), 1);
|
CV_XADD(&(u->urefcount), 1);
|
||||||
}
|
}
|
||||||
hdr.flags = flags;
|
try
|
||||||
setSize(hdr, dims, size.p, step.p);
|
{
|
||||||
finalizeHdr(hdr);
|
hdr.flags = flags;
|
||||||
hdr.u = new_u;
|
setSize(hdr, dims, size.p, step.p);
|
||||||
hdr.offset = 0; //data - datastart;
|
finalizeHdr(hdr);
|
||||||
hdr.addref();
|
hdr.u = new_u;
|
||||||
return hdr;
|
hdr.offset = 0; //data - datastart;
|
||||||
|
hdr.addref();
|
||||||
|
return hdr;
|
||||||
|
}
|
||||||
|
catch(...)
|
||||||
|
{
|
||||||
|
if (u != NULL)
|
||||||
|
{
|
||||||
|
CV_XADD(&(u->refcount), -1);
|
||||||
|
CV_XADD(&(u->urefcount), -1);
|
||||||
|
}
|
||||||
|
new_u->currAllocator->deallocate(new_u);
|
||||||
|
throw;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void UMat::create(int d, const int* _sizes, int _type, UMatUsageFlags _usageFlags)
|
void UMat::create(int d, const int* _sizes, int _type, UMatUsageFlags _usageFlags)
|
||||||
@@ -692,18 +705,17 @@ UMat::UMat(const UMat& m, const Rect& roi)
|
|||||||
offset += roi.x*esz;
|
offset += roi.x*esz;
|
||||||
CV_Assert( 0 <= roi.x && 0 <= roi.width && roi.x + roi.width <= m.cols &&
|
CV_Assert( 0 <= roi.x && 0 <= roi.width && roi.x + roi.width <= m.cols &&
|
||||||
0 <= roi.y && 0 <= roi.height && roi.y + roi.height <= m.rows );
|
0 <= roi.y && 0 <= roi.height && roi.y + roi.height <= m.rows );
|
||||||
if( u )
|
|
||||||
CV_XADD(&(u->urefcount), 1);
|
|
||||||
if( roi.width < m.cols || roi.height < m.rows )
|
if( roi.width < m.cols || roi.height < m.rows )
|
||||||
flags |= SUBMATRIX_FLAG;
|
flags |= SUBMATRIX_FLAG;
|
||||||
|
|
||||||
step[0] = m.step[0]; step[1] = esz;
|
step[0] = m.step[0]; step[1] = esz;
|
||||||
updateContinuityFlag();
|
updateContinuityFlag();
|
||||||
|
|
||||||
|
addref();
|
||||||
if( rows <= 0 || cols <= 0 )
|
if( rows <= 0 || cols <= 0 )
|
||||||
{
|
{
|
||||||
release();
|
|
||||||
rows = cols = 0;
|
rows = cols = 0;
|
||||||
|
release();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -969,24 +981,29 @@ Mat UMat::getMat(int accessFlags) const
|
|||||||
// TODO Support ACCESS_READ (ACCESS_WRITE) without unnecessary data transfers
|
// TODO Support ACCESS_READ (ACCESS_WRITE) without unnecessary data transfers
|
||||||
accessFlags |= ACCESS_RW;
|
accessFlags |= ACCESS_RW;
|
||||||
UMatDataAutoLock autolock(u);
|
UMatDataAutoLock autolock(u);
|
||||||
if(CV_XADD(&u->refcount, 1) == 0)
|
try
|
||||||
u->currAllocator->map(u, accessFlags);
|
|
||||||
if (u->data != 0)
|
|
||||||
{
|
{
|
||||||
Mat hdr(dims, size.p, type(), u->data + offset, step.p);
|
if(CV_XADD(&u->refcount, 1) == 0)
|
||||||
hdr.flags = flags;
|
u->currAllocator->map(u, accessFlags);
|
||||||
hdr.u = u;
|
if (u->data != 0)
|
||||||
hdr.datastart = u->data;
|
{
|
||||||
hdr.data = u->data + offset;
|
Mat hdr(dims, size.p, type(), u->data + offset, step.p);
|
||||||
hdr.datalimit = hdr.dataend = u->data + u->size;
|
hdr.flags = flags;
|
||||||
return hdr;
|
hdr.u = u;
|
||||||
|
hdr.datastart = u->data;
|
||||||
|
hdr.data = u->data + offset;
|
||||||
|
hdr.datalimit = hdr.dataend = u->data + u->size;
|
||||||
|
return hdr;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
else
|
catch(...)
|
||||||
{
|
{
|
||||||
CV_XADD(&u->refcount, -1);
|
CV_XADD(&u->refcount, -1);
|
||||||
CV_Assert(u->data != 0 && "Error mapping of UMat to host memory.");
|
throw;
|
||||||
return Mat();
|
|
||||||
}
|
}
|
||||||
|
CV_XADD(&u->refcount, -1);
|
||||||
|
CV_Assert(u->data != 0 && "Error mapping of UMat to host memory.");
|
||||||
|
return Mat();
|
||||||
}
|
}
|
||||||
|
|
||||||
void* UMat::handle(int accessFlags) const
|
void* UMat::handle(int accessFlags) const
|
||||||
|
|||||||
@@ -67,6 +67,8 @@ PARAM_TEST_CASE(Gemm,
|
|||||||
|
|
||||||
double alpha, beta;
|
double alpha, beta;
|
||||||
|
|
||||||
|
int M, N, K;
|
||||||
|
|
||||||
TEST_DECLARE_INPUT_PARAMETER(A);
|
TEST_DECLARE_INPUT_PARAMETER(A);
|
||||||
TEST_DECLARE_INPUT_PARAMETER(B);
|
TEST_DECLARE_INPUT_PARAMETER(B);
|
||||||
TEST_DECLARE_INPUT_PARAMETER(C);
|
TEST_DECLARE_INPUT_PARAMETER(C);
|
||||||
@@ -90,30 +92,27 @@ PARAM_TEST_CASE(Gemm,
|
|||||||
|
|
||||||
void generateTestData()
|
void generateTestData()
|
||||||
{
|
{
|
||||||
// set minimum size to 20, since testing less sizes doesn't make sense
|
M = (int)randomDoubleLog(1, 100);
|
||||||
Size ARoiSize = randomSize(20, MAX_VALUE);
|
N = (int)randomDoubleLog(1, 100);
|
||||||
|
K = (int)randomDoubleLog(1, 1200);
|
||||||
|
|
||||||
|
M = roundUp(M, 1);
|
||||||
|
N = roundUp(N, 1);
|
||||||
|
K = roundUp(K, 1);
|
||||||
|
|
||||||
|
Size ARoiSize = (atrans) ? Size(M, K) : Size(K, M);
|
||||||
Border ABorder = randomBorder(0, use_roi ? MAX_VALUE : 0);
|
Border ABorder = randomBorder(0, use_roi ? MAX_VALUE : 0);
|
||||||
randomSubMat(A, A_roi, ARoiSize, ABorder, type, -11, 11);
|
randomSubMat(A, A_roi, ARoiSize, ABorder, type, -11, 11);
|
||||||
|
|
||||||
if (atrans)
|
Size BRoiSize = (btrans) ? Size(K, N) : Size(N, K);
|
||||||
ARoiSize = Size(ARoiSize.height, ARoiSize.width);
|
|
||||||
|
|
||||||
Size BRoiSize = randomSize(20, MAX_VALUE);
|
|
||||||
if (btrans)
|
|
||||||
BRoiSize.width = ARoiSize.width;
|
|
||||||
else
|
|
||||||
BRoiSize.height = ARoiSize.width;
|
|
||||||
|
|
||||||
Border BBorder = randomBorder(0, use_roi ? MAX_VALUE : 0);
|
Border BBorder = randomBorder(0, use_roi ? MAX_VALUE : 0);
|
||||||
randomSubMat(B, B_roi, BRoiSize, BBorder, type, -11, 11);
|
randomSubMat(B, B_roi, BRoiSize, BBorder, type, -11, 11);
|
||||||
|
|
||||||
if (btrans)
|
Size CRoiSize = (ctrans) ? Size(M, N) : Size(N, M);
|
||||||
BRoiSize = Size(BRoiSize.height, BRoiSize.width);
|
|
||||||
|
|
||||||
Size DRoiSize = Size(BRoiSize.width, ARoiSize.height), CRoiSizeT(DRoiSize.height, DRoiSize.width);
|
|
||||||
Border CBorder = randomBorder(0, use_roi ? MAX_VALUE : 0);
|
Border CBorder = randomBorder(0, use_roi ? MAX_VALUE : 0);
|
||||||
randomSubMat(C, C_roi, ctrans ? CRoiSizeT : DRoiSize, CBorder, type, -11, 11);
|
randomSubMat(C, C_roi, CRoiSize, CBorder, type, -11, 11);
|
||||||
|
|
||||||
|
Size DRoiSize = Size(N, M);
|
||||||
Border DBorder = randomBorder(0, use_roi ? MAX_VALUE : 0);
|
Border DBorder = randomBorder(0, use_roi ? MAX_VALUE : 0);
|
||||||
randomSubMat(D, D_roi, DRoiSize, DBorder, type, -11, 11);
|
randomSubMat(D, D_roi, DRoiSize, DBorder, type, -11, 11);
|
||||||
|
|
||||||
@@ -132,11 +131,12 @@ OCL_TEST_P(Gemm, Accuracy)
|
|||||||
for (int i = 0; i < test_loop_times; ++i)
|
for (int i = 0; i < test_loop_times; ++i)
|
||||||
{
|
{
|
||||||
generateTestData();
|
generateTestData();
|
||||||
|
SCOPED_TRACE(cv::format("i=%d: M=%d N=%d K=%d", i, M, N, K));
|
||||||
|
|
||||||
OCL_OFF(cv::gemm(A_roi, B_roi, alpha, C_roi, beta, D_roi, flags));
|
OCL_OFF(cv::gemm(A_roi, B_roi, alpha, C_roi, beta, D_roi, flags));
|
||||||
OCL_ON(cv::gemm(uA_roi, uB_roi, alpha, uC_roi, beta, uD_roi, flags));
|
OCL_ON(cv::gemm(uA_roi, uB_roi, alpha, uC_roi, beta, uD_roi, flags));
|
||||||
|
|
||||||
double eps = D_roi.size().area() * 1e-4;
|
double eps = D_roi.size().area() * (1e-5 * K);
|
||||||
OCL_EXPECT_MATS_NEAR(D, eps);
|
OCL_EXPECT_MATS_NEAR(D, eps);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1419,4 +1419,37 @@ TEST(UMat, resize_Mat_issue_13577)
|
|||||||
cv::ocl::setUseOpenCL(useOCL); // restore state
|
cv::ocl::setUseOpenCL(useOCL); // restore state
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST(UMat, exceptions_refcounts_issue_20594)
|
||||||
|
{
|
||||||
|
if (!cv::ocl::useOpenCL())
|
||||||
|
{
|
||||||
|
// skip test, difficult to create exception scenario without OpenCL
|
||||||
|
std::cout << "OpenCL is not enabled. Skip test" << std::endl;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
UMat umat1(10, 10, CV_8UC1);
|
||||||
|
EXPECT_EQ(0, umat1.u->refcount);
|
||||||
|
|
||||||
|
// cause exception in underlying allocator
|
||||||
|
void* const original_handle = umat1.u->handle;
|
||||||
|
umat1.u->handle = NULL;
|
||||||
|
try
|
||||||
|
{
|
||||||
|
Mat mat1 = umat1.getMat(ACCESS_RW);
|
||||||
|
}
|
||||||
|
catch (...)
|
||||||
|
{
|
||||||
|
// nothing
|
||||||
|
}
|
||||||
|
|
||||||
|
// check for correct refcount, and no change of intentional bad handle
|
||||||
|
EXPECT_EQ(0, umat1.u->refcount);
|
||||||
|
EXPECT_EQ(NULL, umat1.u->handle);
|
||||||
|
|
||||||
|
// reset UMat to good state
|
||||||
|
umat1.u->refcount = 0;
|
||||||
|
umat1.u->handle = original_handle;
|
||||||
|
}
|
||||||
|
|
||||||
} } // namespace opencv_test::ocl
|
} } // namespace opencv_test::ocl
|
||||||
|
|||||||
@@ -445,8 +445,8 @@ PARAM_TEST_CASE(GaussianBlur, cv::cuda::DeviceInfo, cv::Size, MatDepth, Channels
|
|||||||
CUDA_TEST_P(GaussianBlur, Accuracy)
|
CUDA_TEST_P(GaussianBlur, Accuracy)
|
||||||
{
|
{
|
||||||
cv::Mat src = randomMat(size, type);
|
cv::Mat src = randomMat(size, type);
|
||||||
double sigma1 = randomDouble(0.1, 1.0);
|
double sigma1 = randomDouble(0.0, 1.0);
|
||||||
double sigma2 = randomDouble(0.1, 1.0);
|
double sigma2 = randomDouble(0.0, 1.0);
|
||||||
|
|
||||||
cv::Ptr<cv::cuda::Filter> gauss = cv::cuda::createGaussianFilter(src.type(), -1, ksize, sigma1, sigma2, borderType);
|
cv::Ptr<cv::cuda::Filter> gauss = cv::cuda::createGaussianFilter(src.type(), -1, ksize, sigma1, sigma2, borderType);
|
||||||
|
|
||||||
|
|||||||
@@ -47,9 +47,9 @@
|
|||||||
#include "opencv2/core/async.hpp"
|
#include "opencv2/core/async.hpp"
|
||||||
|
|
||||||
#if !defined CV_DOXYGEN && !defined CV_STATIC_ANALYSIS && !defined CV_DNN_DONT_ADD_EXPERIMENTAL_NS
|
#if !defined CV_DOXYGEN && !defined CV_STATIC_ANALYSIS && !defined CV_DNN_DONT_ADD_EXPERIMENTAL_NS
|
||||||
#define CV__DNN_EXPERIMENTAL_NS_BEGIN namespace experimental_dnn_34_v22 {
|
#define CV__DNN_EXPERIMENTAL_NS_BEGIN namespace experimental_dnn_34_v23 {
|
||||||
#define CV__DNN_EXPERIMENTAL_NS_END }
|
#define CV__DNN_EXPERIMENTAL_NS_END }
|
||||||
namespace cv { namespace dnn { namespace experimental_dnn_34_v22 { } using namespace experimental_dnn_34_v22; }}
|
namespace cv { namespace dnn { namespace experimental_dnn_34_v23 { } using namespace experimental_dnn_34_v23; }}
|
||||||
#else
|
#else
|
||||||
#define CV__DNN_EXPERIMENTAL_NS_BEGIN
|
#define CV__DNN_EXPERIMENTAL_NS_BEGIN
|
||||||
#define CV__DNN_EXPERIMENTAL_NS_END
|
#define CV__DNN_EXPERIMENTAL_NS_END
|
||||||
|
|||||||
@@ -62,6 +62,12 @@ def printParams(backend, target):
|
|||||||
}
|
}
|
||||||
print('%s/%s' % (backendNames[backend], targetNames[target]))
|
print('%s/%s' % (backendNames[backend], targetNames[target]))
|
||||||
|
|
||||||
|
def getDefaultThreshold(target):
|
||||||
|
if target == cv.dnn.DNN_TARGET_OPENCL_FP16 or target == cv.dnn.DNN_TARGET_MYRIAD:
|
||||||
|
return 4e-3
|
||||||
|
else:
|
||||||
|
return 1e-5
|
||||||
|
|
||||||
testdata_required = bool(os.environ.get('OPENCV_DNN_TEST_REQUIRE_TESTDATA', False))
|
testdata_required = bool(os.environ.get('OPENCV_DNN_TEST_REQUIRE_TESTDATA', False))
|
||||||
|
|
||||||
g_dnnBackendsAndTargets = None
|
g_dnnBackendsAndTargets = None
|
||||||
@@ -305,5 +311,37 @@ class dnn_test(NewOpenCVTests):
|
|||||||
|
|
||||||
cv.dnn_unregisterLayer('CropCaffe')
|
cv.dnn_unregisterLayer('CropCaffe')
|
||||||
|
|
||||||
|
# check that dnn module can work with 3D tensor as input for network
|
||||||
|
def test_input_3d(self):
|
||||||
|
model = self.find_dnn_file('dnn/onnx/models/hidden_lstm.onnx')
|
||||||
|
input_file = self.find_dnn_file('dnn/onnx/data/input_hidden_lstm.npy')
|
||||||
|
output_file = self.find_dnn_file('dnn/onnx/data/output_hidden_lstm.npy')
|
||||||
|
if model is None:
|
||||||
|
raise unittest.SkipTest("Missing DNN test files (dnn/onnx/models/hidden_lstm.onnx). "
|
||||||
|
"Verify OPENCV_DNN_TEST_DATA_PATH configuration parameter.")
|
||||||
|
if input_file is None or output_file is None:
|
||||||
|
raise unittest.SkipTest("Missing DNN test files (dnn/onnx/data/{input/output}_hidden_lstm.npy). "
|
||||||
|
"Verify OPENCV_DNN_TEST_DATA_PATH configuration parameter.")
|
||||||
|
|
||||||
|
input = np.load(input_file)
|
||||||
|
# we have to expand the shape of input tensor because Python bindings cut 3D tensors to 2D
|
||||||
|
# it should be fixed in future. see : https://github.com/opencv/opencv/issues/19091
|
||||||
|
# please remove `expand_dims` after that
|
||||||
|
input = np.expand_dims(input, axis=3)
|
||||||
|
gold_output = np.load(output_file)
|
||||||
|
|
||||||
|
for backend, target in self.dnnBackendsAndTargets:
|
||||||
|
printParams(backend, target)
|
||||||
|
|
||||||
|
net = cv.dnn.readNet(model)
|
||||||
|
|
||||||
|
net.setPreferableBackend(backend)
|
||||||
|
net.setPreferableTarget(target)
|
||||||
|
|
||||||
|
net.setInput(input)
|
||||||
|
real_output = net.forward()
|
||||||
|
|
||||||
|
normAssert(self, real_output, gold_output, "", getDefaultThreshold(target))
|
||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
NewOpenCVTests.bootstrap()
|
NewOpenCVTests.bootstrap()
|
||||||
|
|||||||
@@ -26,25 +26,90 @@ struct ConvParam_t {
|
|||||||
double declared_flops;
|
double declared_flops;
|
||||||
};
|
};
|
||||||
// Details: #12142
|
// Details: #12142
|
||||||
|
// Last update: 2021-09
|
||||||
static const ConvParam_t testConvolutionConfigs[] = {
|
static const ConvParam_t testConvolutionConfigs[] = {
|
||||||
|
/* GFLOPS 3.398 x 20 = 67.956 */ {{7, 7}, {{1, 128, 46, 46}}, 128, 1, {1, 1}, {1, 1}, {3, 3}, {0, 0}, "", true, 3397788160.},
|
||||||
|
/* GFLOPS 16.987 x 3 = 50.962 */ {{5, 5}, {{1, 1152, 16, 16}}, 1152, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 16987226112.},
|
||||||
|
/* GFLOPS 23.122 x 2 = 46.244 */ {{5, 5}, {{1, 672, 32, 32}}, 672, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 23121788928.},
|
||||||
|
/* GFLOPS 9.987 x 3 = 29.960 */ {{3, 3}, {{1, 256, 92, 92}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 9986707456.},
|
||||||
|
/* GFLOPS 1.595 x 16 = 25.524 */ {{3, 3}, {{1, 256, 26, 26}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 1595230208.},
|
||||||
|
/* GFLOPS 4.566 x 5 = 22.828 */ {{7, 7}, {{1, 172, 46, 46}}, 128, 1, {1, 1}, {1, 1}, {3, 3}, {0, 0}, "", true, 4565684736.},
|
||||||
|
/* GFLOPS 1.596 x 14 = 22.338 */ {{3, 3}, {{1, 128, 52, 52}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 1595576320.},
|
||||||
|
/* GFLOPS 1.595 x 12 = 19.141 */ {{3, 3}, {{1, 512, 13, 13}}, 1024, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 1595057152.},
|
||||||
|
/* GFLOPS 6.814 x 2 = 13.629 */ {{3, 3}, {{1, 512, 38, 38}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 6814386176.},
|
||||||
|
/* GFLOPS 6.637 x 2 = 13.274 */ {{3, 3}, {{1, 256, 75, 75}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 6636960000.},
|
||||||
|
/* GFLOPS 11.797 x 1 = 11.797 */ {{5, 5}, {{1, 240, 64, 64}}, 240, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 11797463040.},
|
||||||
|
/* GFLOPS 11.797 x 1 = 11.797 */ {{5, 5}, {{1, 480, 32, 32}}, 480, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 11796971520.},
|
||||||
|
/* GFLOPS 10.701 x 1 = 10.701 */ {{3, 3}, {{1, 512, 38, 38}}, 804, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 10700715792.},
|
||||||
/* GFLOPS 10.087 x 1 = 10.087 */ {{3, 3}, {{1, 576, 38, 50}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 10086963200.},
|
/* GFLOPS 10.087 x 1 = 10.087 */ {{3, 3}, {{1, 576, 38, 50}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 10086963200.},
|
||||||
|
/* GFLOPS 9.993 x 1 = 9.993 */ {{3, 3}, {{1, 64, 368, 368}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 9993207808.},
|
||||||
|
/* GFLOPS 9.989 x 1 = 9.989 */ {{3, 3}, {{1, 128, 184, 184}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 9988874240.},
|
||||||
|
/* GFLOPS 9.986 x 1 = 9.986 */ {{3, 3}, {{1, 512, 46, 46}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 9985624064.},
|
||||||
/* GFLOPS 1.704 x 5 = 8.518 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 512, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1703596544.},
|
/* GFLOPS 1.704 x 5 = 8.518 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 512, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1703596544.},
|
||||||
/* GFLOPS 1.704 x 5 = 8.518 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1703596544.},
|
/* GFLOPS 1.704 x 5 = 8.518 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1703596544.},
|
||||||
|
/* GFLOPS 4.247 x 2 = 8.494 */ {{3, 3}, {{1, 480, 32, 32}}, 480, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 4247224320.},
|
||||||
|
/* GFLOPS 8.025 x 1 = 8.025 */ {{3, 3}, {{1, 1024, 19, 19}}, 1206, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 8025101478.},
|
||||||
|
/* GFLOPS 0.798 x 9 = 7.180 */ {{3, 3}, {{1, 128, 52, 52}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 797788160.},
|
||||||
|
/* GFLOPS 0.798 x 9 = 7.179 */ {{3, 3}, {{1, 256, 26, 26}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 797615104.},
|
||||||
|
/* GFLOPS 6.641 x 1 = 6.641 */ {{3, 3}, {{1, 64, 300, 300}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 6641280000.},
|
||||||
/* GFLOPS 6.641 x 1 = 6.641 */ {{3, 3}, {{1, 64, 150, 200}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 6641280000.},
|
/* GFLOPS 6.641 x 1 = 6.641 */ {{3, 3}, {{1, 64, 150, 200}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 6641280000.},
|
||||||
|
/* GFLOPS 6.638 x 1 = 6.638 */ {{3, 3}, {{1, 128, 150, 150}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 6638400000.},
|
||||||
|
/* GFLOPS 6.118 x 1 = 6.118 */ {{3, 3}, {{1, 144, 128, 128}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 6117654528.},
|
||||||
|
/* GFLOPS 6.116 x 1 = 6.116 */ {{3, 3}, {{1, 1152, 16, 16}}, 1152, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 6115590144.},
|
||||||
|
/* GFLOPS 5.780 x 1 = 5.780 */ {{5, 5}, {{1, 672, 32, 32}}, 672, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 5780447232.},
|
||||||
|
/* GFLOPS 1.704 x 3 = 5.111 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1703596544.},
|
||||||
|
/* GFLOPS 4.997 x 1 = 4.997 */ {{3, 3}, {{1, 64, 184, 184}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 4996603904.},
|
||||||
|
/* GFLOPS 4.994 x 1 = 4.994 */ {{3, 3}, {{1, 128, 92, 92}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 4994437120.},
|
||||||
|
/* GFLOPS 4.993 x 1 = 4.993 */ {{3, 3}, {{1, 256, 46, 46}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 4993353728.},
|
||||||
|
/* GFLOPS 4.993 x 1 = 4.993 */ {{3, 3}, {{1, 512, 46, 46}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 4992812032.},
|
||||||
/* GFLOPS 1.659 x 3 = 4.977 */ {{3, 3}, {{1, 960, 10, 10}}, 960, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1658976000.},
|
/* GFLOPS 1.659 x 3 = 4.977 */ {{3, 3}, {{1, 960, 10, 10}}, 960, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1658976000.},
|
||||||
/* GFLOPS 2.156 x 2 = 4.312 */ {{3, 3}, {{1, 576, 19, 19}}, 576, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 2156088384.},
|
/* GFLOPS 2.156 x 2 = 4.312 */ {{3, 3}, {{1, 576, 19, 19}}, 576, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 2156088384.},
|
||||||
|
/* GFLOPS 4.247 x 1 = 4.247 */ {{5, 5}, {{1, 144, 128, 128}}, 144, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 4247322624.},
|
||||||
|
/* GFLOPS 0.798 x 5 = 3.988 */ {{3, 3}, {{1, 512, 13, 13}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 797528576.},
|
||||||
/* GFLOPS 0.958 x 4 = 3.833 */ {{3, 3}, {{1, 384, 19, 19}}, 384, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 958307712.},
|
/* GFLOPS 0.958 x 4 = 3.833 */ {{3, 3}, {{1, 384, 19, 19}}, 384, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 958307712.},
|
||||||
|
/* GFLOPS 0.624 x 6 = 3.746 */ {{3, 3}, {{1, 128, 46, 46}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 624304640.},
|
||||||
|
/* GFLOPS 3.408 x 1 = 3.408 */ {{3, 3}, {{1, 256, 38, 38}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 3407562752.},
|
||||||
|
/* GFLOPS 3.407 x 1 = 3.407 */ {{3, 3}, {{1, 512, 19, 19}}, 1024, 1, {1, 1}, {6, 6}, {6, 6}, {0, 0}, "", true, 3407193088.},
|
||||||
|
/* GFLOPS 0.177 x 19 = 3.370 */ {{1, 1}, {{1, 512, 26, 26}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 177382400.},
|
||||||
|
/* GFLOPS 0.302 x 11 = 3.325 */ {{3, 3}, {{1, 64, 64, 64}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 302252032.},
|
||||||
|
/* GFLOPS 3.321 x 1 = 3.321 */ {{3, 3}, {{1, 64, 150, 150}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 3320640000.},
|
||||||
/* GFLOPS 0.830 x 4 = 3.321 */ {{3, 3}, {{1, 64, 75, 100}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 830160000.},
|
/* GFLOPS 0.830 x 4 = 3.321 */ {{3, 3}, {{1, 64, 75, 100}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 830160000.},
|
||||||
|
/* GFLOPS 3.319 x 1 = 3.319 */ {{3, 3}, {{1, 128, 75, 75}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 3319200000.},
|
||||||
|
/* GFLOPS 1.598 x 2 = 3.195 */ {{3, 3}, {{1, 32, 416, 416}}, 64, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 1597652992.},
|
||||||
|
/* GFLOPS 1.598 x 2 = 3.195 */ {{3, 3}, {{1, 32, 208, 208}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 1597652992.},
|
||||||
|
/* GFLOPS 1.596 x 2 = 3.193 */ {{3, 3}, {{1, 64, 208, 208}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 1596268544.},
|
||||||
|
/* GFLOPS 1.596 x 2 = 3.193 */ {{3, 3}, {{1, 64, 104, 104}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 1596268544.},
|
||||||
|
/* GFLOPS 1.596 x 2 = 3.191 */ {{3, 3}, {{1, 128, 104, 104}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 1595576320.},
|
||||||
|
/* GFLOPS 1.595 x 2 = 3.190 */ {{3, 3}, {{1, 256, 52, 52}}, 512, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 1595230208.},
|
||||||
|
/* GFLOPS 1.595 x 2 = 3.190 */ {{3, 3}, {{1, 512, 26, 26}}, 1024, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 1595057152.},
|
||||||
|
/* GFLOPS 0.178 x 16 = 2.841 */ {{1, 1}, {{1, 256, 52, 52}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 177555456.},
|
||||||
|
/* GFLOPS 2.719 x 1 = 2.719 */ {{3, 3}, {{1, 96, 256, 256}}, 96, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 2719481856.},
|
||||||
|
/* GFLOPS 0.177 x 15 = 2.659 */ {{1, 1}, {{1, 1024, 13, 13}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 177295872.},
|
||||||
/* GFLOPS 1.245 x 2 = 2.490 */ {{3, 3}, {{1, 96, 75, 100}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1244880000.},
|
/* GFLOPS 1.245 x 2 = 2.490 */ {{3, 3}, {{1, 96, 75, 100}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1244880000.},
|
||||||
|
/* GFLOPS 0.798 x 3 = 2.394 */ {{3, 3}, {{1, 64, 104, 104}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 798134272.},
|
||||||
|
/* GFLOPS 0.472 x 5 = 2.360 */ {{3, 3}, {{1, 256, 20, 20}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 471961600.},
|
||||||
|
/* GFLOPS 2.255 x 1 = 2.255 */ {{3, 3}, {{1, 128, 80, 100}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2255285760.},
|
||||||
|
/* GFLOPS 2.153 x 1 = 2.153 */ {{3, 3}, {{1, 128, 78, 98}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2152611840.},
|
||||||
/* GFLOPS 2.100 x 1 = 2.100 */ {{3, 3}, {{1, 144, 75, 75}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 2100330000.},
|
/* GFLOPS 2.100 x 1 = 2.100 */ {{3, 3}, {{1, 144, 75, 75}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 2100330000.},
|
||||||
|
/* GFLOPS 2.052 x 1 = 2.052 */ {{3, 3}, {{1, 128, 76, 96}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2052298240.},
|
||||||
/* GFLOPS 1.022 x 2 = 2.044 */ {{3, 3}, {{1, 576, 19, 19}}, 273, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1021896057.},
|
/* GFLOPS 1.022 x 2 = 2.044 */ {{3, 3}, {{1, 576, 19, 19}}, 273, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1021896057.},
|
||||||
|
/* GFLOPS 1.995 x 1 = 1.995 */ {{9, 9}, {{1, 3, 320, 400}}, 32, 1, {1, 1}, {1, 1}, {4, 4}, {0, 0}, "", true, 1994752000.},
|
||||||
|
/* GFLOPS 1.954 x 1 = 1.954 */ {{3, 3}, {{1, 128, 74, 94}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1954344960.},
|
||||||
/* GFLOPS 0.958 x 2 = 1.917 */ {{3, 3}, {{1, 192, 38, 38}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 958446336.},
|
/* GFLOPS 0.958 x 2 = 1.917 */ {{3, 3}, {{1, 192, 38, 38}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 958446336.},
|
||||||
/* GFLOPS 1.888 x 1 = 1.888 */ {{3, 3}, {{1, 1024, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1887539200.},
|
/* GFLOPS 1.888 x 1 = 1.888 */ {{3, 3}, {{1, 1024, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1887539200.},
|
||||||
/* GFLOPS 1.888 x 1 = 1.888 */ {{3, 3}, {{1, 1024, 10, 10}}, 1024, 1024, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1887539200.},
|
/* GFLOPS 1.888 x 1 = 1.888 */ {{3, 3}, {{1, 1024, 10, 10}}, 1024, 1024, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1887539200.},
|
||||||
|
/* GFLOPS 1.859 x 1 = 1.859 */ {{3, 3}, {{1, 128, 72, 92}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1858752000.},
|
||||||
|
/* GFLOPS 1.766 x 1 = 1.766 */ {{3, 3}, {{1, 128, 70, 90}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1765519360.},
|
||||||
/* GFLOPS 1.704 x 1 = 1.704 */ {{3, 3}, {{1, 256, 38, 38}}, 256, 256, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1703781376.},
|
/* GFLOPS 1.704 x 1 = 1.704 */ {{3, 3}, {{1, 256, 38, 38}}, 256, 256, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1703781376.},
|
||||||
/* GFLOPS 1.704 x 1 = 1.704 */ {{3, 3}, {{1, 256, 38, 38}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1703781376.},
|
/* GFLOPS 1.704 x 1 = 1.704 */ {{3, 3}, {{1, 256, 38, 38}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1703781376.},
|
||||||
|
/* GFLOPS 1.675 x 1 = 1.675 */ {{3, 3}, {{1, 128, 68, 88}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1674647040.},
|
||||||
/* GFLOPS 1.660 x 1 = 1.660 */ {{3, 3}, {{1, 128, 75, 75}}, 128, 128, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1659600000.},
|
/* GFLOPS 1.660 x 1 = 1.660 */ {{3, 3}, {{1, 128, 75, 75}}, 128, 128, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1659600000.},
|
||||||
/* GFLOPS 1.660 x 1 = 1.660 */ {{3, 3}, {{1, 128, 75, 75}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1659600000.},
|
/* GFLOPS 1.660 x 1 = 1.660 */ {{3, 3}, {{1, 128, 75, 75}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1659600000.},
|
||||||
|
/* GFLOPS 1.586 x 1 = 1.586 */ {{3, 3}, {{1, 128, 66, 86}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1586135040.},
|
||||||
|
/* GFLOPS 1.500 x 1 = 1.500 */ {{3, 3}, {{1, 128, 64, 84}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1499983360.},
|
||||||
|
/* GFLOPS 1.416 x 1 = 1.416 */ {{3, 3}, {{1, 128, 62, 82}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1416192000.},
|
||||||
|
/* GFLOPS 0.472 x 3 = 1.416 */ {{3, 3}, {{1, 128, 40, 40}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 472064000.},
|
||||||
|
/* GFLOPS 0.472 x 3 = 1.416 */ {{3, 3}, {{1, 512, 10, 10}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 471910400.},
|
||||||
/* GFLOPS 0.280 x 5 = 1.402 */ {{1, 1}, {{1, 576, 38, 50}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 280409600.},
|
/* GFLOPS 0.280 x 5 = 1.402 */ {{1, 1}, {{1, 576, 38, 50}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 280409600.},
|
||||||
/* GFLOPS 0.701 x 2 = 1.401 */ {{3, 3}, {{1, 128, 38, 50}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 700720000.},
|
/* GFLOPS 0.701 x 2 = 1.401 */ {{3, 3}, {{1, 128, 38, 50}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 700720000.},
|
||||||
/* GFLOPS 0.231 x 6 = 1.388 */ {{3, 3}, {{1, 128, 56, 56}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 231311360.},
|
/* GFLOPS 0.231 x 6 = 1.388 */ {{3, 3}, {{1, 128, 56, 56}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 231311360.},
|
||||||
@@ -53,20 +118,39 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.420 x 3 = 1.261 */ {{3, 3}, {{1, 96, 38, 50}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 420492800.},
|
/* GFLOPS 0.420 x 3 = 1.261 */ {{3, 3}, {{1, 96, 38, 50}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 420492800.},
|
||||||
/* GFLOPS 1.261 x 1 = 1.261 */ {{3, 3}, {{1, 192, 38, 50}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1261113600.},
|
/* GFLOPS 1.261 x 1 = 1.261 */ {{3, 3}, {{1, 192, 38, 50}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1261113600.},
|
||||||
/* GFLOPS 1.258 x 1 = 1.258 */ {{3, 3}, {{1, 1280, 10, 10}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1258038600.},
|
/* GFLOPS 1.258 x 1 = 1.258 */ {{3, 3}, {{1, 1280, 10, 10}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1258038600.},
|
||||||
|
/* GFLOPS 1.248 x 1 = 1.248 */ {{3, 3}, {{1, 256, 46, 46}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1248338432.},
|
||||||
/* GFLOPS 1.245 x 1 = 1.245 */ {{3, 3}, {{1, 64, 75, 75}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1245240000.},
|
/* GFLOPS 1.245 x 1 = 1.245 */ {{3, 3}, {{1, 64, 75, 75}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1245240000.},
|
||||||
|
/* GFLOPS 1.210 x 1 = 1.210 */ {{3, 3}, {{1, 32, 256, 256}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1210056704.},
|
||||||
|
/* GFLOPS 1.196 x 1 = 1.196 */ {{3, 3}, {{1, 384, 26, 26}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 1196336128.},
|
||||||
|
/* GFLOPS 1.195 x 1 = 1.195 */ {{9, 9}, {{1, 32, 240, 320}}, 3, 1, {1, 1}, {1, 1}, {4, 4}, {0, 0}, "", true, 1194624000.},
|
||||||
|
/* GFLOPS 1.182 x 1 = 1.182 */ {{3, 3}, {{1, 32, 320, 400}}, 64, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 1181696000.},
|
||||||
|
/* GFLOPS 1.181 x 1 = 1.181 */ {{3, 3}, {{1, 64, 160, 200}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 1180672000.},
|
||||||
/* GFLOPS 0.561 x 2 = 1.121 */ {{3, 3}, {{1, 128, 38, 50}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 560576000.},
|
/* GFLOPS 0.561 x 2 = 1.121 */ {{3, 3}, {{1, 128, 38, 50}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 560576000.},
|
||||||
|
/* GFLOPS 1.112 x 1 = 1.112 */ {{3, 3}, {{1, 512, 10, 10}}, 1206, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1111570200.},
|
||||||
|
/* GFLOPS 0.357 x 3 = 1.072 */ {{1, 1}, {{1, 64, 208, 208}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 357187584.},
|
||||||
|
/* GFLOPS 1.062 x 1 = 1.062 */ {{3, 3}, {{1, 240, 64, 64}}, 240, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1061928960.},
|
||||||
|
/* GFLOPS 0.076 x 14 = 1.058 */ {{3, 3}, {{1, 64, 32, 32}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 75563008.},
|
||||||
/* GFLOPS 1.051 x 1 = 1.051 */ {{3, 3}, {{1, 160, 38, 50}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1050988800.},
|
/* GFLOPS 1.051 x 1 = 1.051 */ {{3, 3}, {{1, 160, 38, 50}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1050988800.},
|
||||||
|
/* GFLOPS 0.210 x 5 = 1.051 */ {{1, 1}, {{1, 256, 20, 20}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 210124800.},
|
||||||
|
/* GFLOPS 0.210 x 5 = 1.049 */ {{1, 1}, {{1, 1024, 20, 20}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 209817600.},
|
||||||
/* GFLOPS 1.006 x 1 = 1.006 */ {{3, 3}, {{1, 1024, 10, 10}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1006441800.},
|
/* GFLOPS 1.006 x 1 = 1.006 */ {{3, 3}, {{1, 1024, 10, 10}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1006441800.},
|
||||||
/* GFLOPS 0.246 x 4 = 0.985 */ {{1, 1}, {{1, 256, 75, 100}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 246240000.},
|
/* GFLOPS 0.246 x 4 = 0.985 */ {{1, 1}, {{1, 256, 75, 100}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 246240000.},
|
||||||
/* GFLOPS 0.189 x 5 = 0.947 */ {{1, 1}, {{1, 512, 19, 19}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 189452800.},
|
/* GFLOPS 0.189 x 5 = 0.947 */ {{1, 1}, {{1, 512, 19, 19}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 189452800.},
|
||||||
/* GFLOPS 0.189 x 5 = 0.947 */ {{1, 1}, {{1, 512, 19, 19}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 189452800.},
|
/* GFLOPS 0.189 x 5 = 0.947 */ {{1, 1}, {{1, 512, 19, 19}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 189452800.},
|
||||||
|
/* GFLOPS 0.472 x 2 = 0.945 */ {{3, 3}, {{1, 64, 80, 80}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 472268800.},
|
||||||
/* GFLOPS 0.934 x 1 = 0.934 */ {{3, 3}, {{1, 96, 150, 150}}, 96, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 933660000.},
|
/* GFLOPS 0.934 x 1 = 0.934 */ {{3, 3}, {{1, 96, 150, 150}}, 96, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 933660000.},
|
||||||
/* GFLOPS 0.231 x 4 = 0.925 */ {{3, 3}, {{1, 128, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 231311360.},
|
/* GFLOPS 0.231 x 4 = 0.925 */ {{3, 3}, {{1, 128, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 231311360.},
|
||||||
/* GFLOPS 0.896 x 1 = 0.896 */ {{5, 5}, {{1, 96, 27, 27}}, 256, 2, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 895981824.},
|
/* GFLOPS 0.896 x 1 = 0.896 */ {{5, 5}, {{1, 96, 27, 27}}, 256, 2, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 895981824.},
|
||||||
|
/* GFLOPS 0.089 x 10 = 0.890 */ {{1, 1}, {{1, 128, 52, 52}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 88950784.},
|
||||||
|
/* GFLOPS 0.089 x 10 = 0.888 */ {{1, 1}, {{1, 256, 26, 26}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 88777728.},
|
||||||
/* GFLOPS 0.876 x 1 = 0.876 */ {{3, 3}, {{1, 160, 38, 50}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 875824000.},
|
/* GFLOPS 0.876 x 1 = 0.876 */ {{3, 3}, {{1, 160, 38, 50}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 875824000.},
|
||||||
/* GFLOPS 0.850 x 1 = 0.850 */ {{7, 7}, {{1, 3, 600, 800}}, 24, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 849600000.},
|
/* GFLOPS 0.850 x 1 = 0.850 */ {{7, 7}, {{1, 3, 600, 800}}, 24, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 849600000.},
|
||||||
/* GFLOPS 0.841 x 1 = 0.841 */ {{3, 3}, {{1, 128, 38, 50}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 840864000.},
|
/* GFLOPS 0.841 x 1 = 0.841 */ {{3, 3}, {{1, 128, 38, 50}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 840864000.},
|
||||||
/* GFLOPS 0.415 x 2 = 0.831 */ {{3, 3}, {{1, 32, 150, 150}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 415440000.},
|
/* GFLOPS 0.415 x 2 = 0.831 */ {{3, 3}, {{1, 32, 150, 150}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 415440000.},
|
||||||
|
/* GFLOPS 0.757 x 1 = 0.757 */ {{1, 1}, {{1, 1024, 19, 19}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 757441536.},
|
||||||
|
/* GFLOPS 0.712 x 1 = 0.712 */ {{1, 1}, {{1, 128, 208, 208}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 711606272.},
|
||||||
|
/* GFLOPS 0.178 x 4 = 0.712 */ {{1, 1}, {{1, 128, 104, 104}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 177901568.},
|
||||||
|
/* GFLOPS 0.354 x 2 = 0.707 */ {{1, 1}, {{1, 256, 52, 52}}, 255, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 353723760.},
|
||||||
/* GFLOPS 0.351 x 2 = 0.701 */ {{1, 1}, {{1, 576, 38, 50}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 350512000.},
|
/* GFLOPS 0.351 x 2 = 0.701 */ {{1, 1}, {{1, 576, 38, 50}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 350512000.},
|
||||||
/* GFLOPS 0.701 x 1 = 0.701 */ {{3, 3}, {{1, 128, 75, 100}}, 160, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 700720000.},
|
/* GFLOPS 0.701 x 1 = 0.701 */ {{3, 3}, {{1, 128, 75, 100}}, 160, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 700720000.},
|
||||||
/* GFLOPS 0.694 x 1 = 0.694 */ {{3, 3}, {{1, 64, 56, 56}}, 192, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 694235136.},
|
/* GFLOPS 0.694 x 1 = 0.694 */ {{3, 3}, {{1, 64, 56, 56}}, 192, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 694235136.},
|
||||||
@@ -75,19 +159,31 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.058 x 12 = 0.694 */ {{3, 3}, {{1, 128, 28, 28}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 57827840.},
|
/* GFLOPS 0.058 x 12 = 0.694 */ {{3, 3}, {{1, 128, 28, 28}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 57827840.},
|
||||||
/* GFLOPS 0.231 x 3 = 0.694 */ {{3, 3}, {{1, 512, 7, 7}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 231236096.},
|
/* GFLOPS 0.231 x 3 = 0.694 */ {{3, 3}, {{1, 512, 7, 7}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 231236096.},
|
||||||
/* GFLOPS 0.160 x 4 = 0.639 */ {{3, 3}, {{1, 64, 38, 38}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 159833472.},
|
/* GFLOPS 0.160 x 4 = 0.639 */ {{3, 3}, {{1, 64, 38, 38}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 159833472.},
|
||||||
|
/* GFLOPS 0.211 x 3 = 0.634 */ {{1, 1}, {{1, 64, 80, 80}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 211353600.},
|
||||||
|
/* GFLOPS 0.211 x 3 = 0.632 */ {{1, 1}, {{1, 128, 40, 40}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 210534400.},
|
||||||
|
/* GFLOPS 0.210 x 3 = 0.630 */ {{1, 1}, {{1, 512, 40, 40}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 209920000.},
|
||||||
|
/* GFLOPS 0.210 x 3 = 0.630 */ {{1, 1}, {{1, 512, 10, 10}}, 2048, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 209920000.},
|
||||||
/* GFLOPS 0.103 x 6 = 0.618 */ {{1, 1}, {{1, 256, 14, 14}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102961152.},
|
/* GFLOPS 0.103 x 6 = 0.618 */ {{1, 1}, {{1, 256, 14, 14}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102961152.},
|
||||||
/* GFLOPS 0.615 x 1 = 0.615 */ {{1, 1}, {{1, 320, 75, 100}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 615360000.},
|
/* GFLOPS 0.615 x 1 = 0.615 */ {{1, 1}, {{1, 320, 75, 100}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 615360000.},
|
||||||
|
/* GFLOPS 0.305 x 2 = 0.609 */ {{3, 3}, {{1, 3, 416, 416}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 304578560.},
|
||||||
/* GFLOPS 0.597 x 1 = 0.597 */ {{3, 3}, {{1, 576, 19, 19}}, 576, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 597254400.},
|
/* GFLOPS 0.597 x 1 = 0.597 */ {{3, 3}, {{1, 576, 19, 19}}, 576, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 597254400.},
|
||||||
|
/* GFLOPS 0.278 x 2 = 0.557 */ {{1, 1}, {{1, 128, 46, 46}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 278431744.},
|
||||||
/* GFLOPS 0.185 x 3 = 0.554 */ {{1, 1}, {{1, 192, 75, 100}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 184800000.},
|
/* GFLOPS 0.185 x 3 = 0.554 */ {{1, 1}, {{1, 192, 75, 100}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 184800000.},
|
||||||
/* GFLOPS 0.553 x 1 = 0.553 */ {{3, 3}, {{1, 64, 75, 100}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 553440000.},
|
/* GFLOPS 0.553 x 1 = 0.553 */ {{3, 3}, {{1, 64, 75, 100}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 553440000.},
|
||||||
/* GFLOPS 0.539 x 1 = 0.539 */ {{3, 3}, {{1, 144, 75, 75}}, 144, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 539178048.},
|
/* GFLOPS 0.539 x 1 = 0.539 */ {{3, 3}, {{1, 144, 75, 75}}, 144, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 539178048.},
|
||||||
/* GFLOPS 0.103 x 5 = 0.514 */ {{1, 1}, {{1, 1024, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102810624.},
|
/* GFLOPS 0.103 x 5 = 0.514 */ {{1, 1}, {{1, 1024, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102810624.},
|
||||||
/* GFLOPS 0.491 x 1 = 0.491 */ {{1, 1}, {{1, 576, 38, 50}}, 224, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 490716800.},
|
/* GFLOPS 0.491 x 1 = 0.491 */ {{1, 1}, {{1, 576, 38, 50}}, 224, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 490716800.},
|
||||||
|
/* GFLOPS 0.483 x 1 = 0.483 */ {{7, 7}, {{1, 3, 320, 320}}, 64, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", false, 483328000.},
|
||||||
/* GFLOPS 0.240 x 2 = 0.479 */ {{3, 3}, {{1, 96, 38, 38}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 239680896.},
|
/* GFLOPS 0.240 x 2 = 0.479 */ {{3, 3}, {{1, 96, 38, 38}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 239680896.},
|
||||||
|
/* GFLOPS 0.477 x 1 = 0.477 */ {{3, 3}, {{1, 3, 368, 368}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 476692480.},
|
||||||
/* GFLOPS 0.237 x 2 = 0.474 */ {{7, 7}, {{1, 3, 224, 224}}, 64, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", true, 236830720.},
|
/* GFLOPS 0.237 x 2 = 0.474 */ {{7, 7}, {{1, 3, 224, 224}}, 64, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", true, 236830720.},
|
||||||
/* GFLOPS 0.472 x 1 = 0.472 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 512, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 471910400.},
|
/* GFLOPS 0.472 x 1 = 0.472 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 512, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 471910400.},
|
||||||
/* GFLOPS 0.472 x 1 = 0.472 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 471910400.},
|
/* GFLOPS 0.472 x 1 = 0.472 */ {{3, 3}, {{1, 512, 19, 19}}, 512, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 471910400.},
|
||||||
|
/* GFLOPS 0.155 x 3 = 0.464 */ {{1, 1}, {{1, 112, 32, 32}}, 672, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 154828800.},
|
||||||
|
/* GFLOPS 0.114 x 4 = 0.454 */ {{1, 1}, {{1, 192, 16, 16}}, 1152, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 113541120.},
|
||||||
/* GFLOPS 0.449 x 1 = 0.449 */ {{3, 3}, {{1, 384, 13, 13}}, 384, 2, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 448626048.},
|
/* GFLOPS 0.449 x 1 = 0.449 */ {{3, 3}, {{1, 384, 13, 13}}, 384, 2, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 448626048.},
|
||||||
|
/* GFLOPS 0.089 x 5 = 0.443 */ {{1, 1}, {{1, 512, 13, 13}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 88691200.},
|
||||||
|
/* GFLOPS 0.428 x 1 = 0.428 */ {{1, 1}, {{1, 64, 64, 64}}, 810, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 427991040.},
|
||||||
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 128, 75, 75}}, 128, 128, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 426037760.},
|
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 128, 75, 75}}, 128, 128, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 426037760.},
|
||||||
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 128, 75, 75}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 426037760.},
|
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 128, 75, 75}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 426037760.},
|
||||||
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 128, 38, 38}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 426037760.},
|
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 128, 38, 38}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 426037760.},
|
||||||
@@ -95,46 +191,81 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 256, 38, 38}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 425945344.},
|
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 256, 38, 38}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 425945344.},
|
||||||
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 256, 19, 19}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 425945344.},
|
/* GFLOPS 0.426 x 1 = 0.426 */ {{3, 3}, {{1, 256, 19, 19}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 425945344.},
|
||||||
/* GFLOPS 0.421 x 1 = 0.421 */ {{1, 1}, {{1, 576, 38, 50}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 420614400.},
|
/* GFLOPS 0.421 x 1 = 0.421 */ {{1, 1}, {{1, 576, 38, 50}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 420614400.},
|
||||||
|
/* GFLOPS 0.420 x 1 = 0.420 */ {{1, 1}, {{1, 256, 40, 40}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 420249600.},
|
||||||
|
/* GFLOPS 0.210 x 2 = 0.420 */ {{1, 1}, {{1, 256, 80, 80}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 210124800.},
|
||||||
|
/* GFLOPS 0.420 x 1 = 0.420 */ {{1, 1}, {{1, 512, 20, 20}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 419840000.},
|
||||||
|
/* GFLOPS 0.420 x 1 = 0.420 */ {{1, 1}, {{1, 1024, 10, 10}}, 2048, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 419635200.},
|
||||||
|
/* GFLOPS 0.210 x 2 = 0.420 */ {{1, 1}, {{1, 2048, 10, 10}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 209766400.},
|
||||||
/* GFLOPS 0.415 x 1 = 0.415 */ {{3, 3}, {{1, 32, 150, 150}}, 32, 32, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 415440000.},
|
/* GFLOPS 0.415 x 1 = 0.415 */ {{3, 3}, {{1, 32, 150, 150}}, 32, 32, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 415440000.},
|
||||||
/* GFLOPS 0.415 x 1 = 0.415 */ {{3, 3}, {{1, 64, 150, 150}}, 64, 64, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 415080000.},
|
/* GFLOPS 0.415 x 1 = 0.415 */ {{3, 3}, {{1, 64, 150, 150}}, 64, 64, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 415080000.},
|
||||||
/* GFLOPS 0.415 x 1 = 0.415 */ {{3, 3}, {{1, 64, 150, 150}}, 64, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 415080000.},
|
/* GFLOPS 0.415 x 1 = 0.415 */ {{3, 3}, {{1, 64, 150, 150}}, 64, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 415080000.},
|
||||||
/* GFLOPS 0.104 x 4 = 0.414 */ {{1, 1}, {{1, 64, 56, 56}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 103563264.},
|
/* GFLOPS 0.104 x 4 = 0.414 */ {{1, 1}, {{1, 64, 56, 56}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 103563264.},
|
||||||
/* GFLOPS 0.103 x 4 = 0.413 */ {{1, 1}, {{1, 128, 28, 28}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 103161856.},
|
/* GFLOPS 0.103 x 4 = 0.413 */ {{1, 1}, {{1, 128, 28, 28}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 103161856.},
|
||||||
|
/* GFLOPS 0.399 x 1 = 0.399 */ {{3, 3}, {{1, 32, 208, 208}}, 64, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 399413248.},
|
||||||
|
/* GFLOPS 0.200 x 2 = 0.399 */ {{3, 3}, {{1, 32, 104, 104}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 199706624.},
|
||||||
|
/* GFLOPS 0.200 x 2 = 0.399 */ {{3, 3}, {{1, 64, 52, 52}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 199533568.},
|
||||||
|
/* GFLOPS 0.399 x 1 = 0.399 */ {{3, 3}, {{1, 128, 52, 52}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 398894080.},
|
||||||
|
/* GFLOPS 0.199 x 2 = 0.399 */ {{3, 3}, {{1, 128, 26, 26}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 199447040.},
|
||||||
|
/* GFLOPS 0.399 x 1 = 0.399 */ {{3, 3}, {{1, 256, 26, 26}}, 512, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 398807552.},
|
||||||
|
/* GFLOPS 0.399 x 1 = 0.399 */ {{3, 3}, {{1, 256, 13, 13}}, 512, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 398807552.},
|
||||||
/* GFLOPS 0.376 x 1 = 0.376 */ {{1, 1}, {{1, 24, 300, 400}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 376320000.},
|
/* GFLOPS 0.376 x 1 = 0.376 */ {{1, 1}, {{1, 24, 300, 400}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 376320000.},
|
||||||
|
/* GFLOPS 0.179 x 2 = 0.357 */ {{1, 1}, {{1, 64, 208, 208}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 178593792.},
|
||||||
|
/* GFLOPS 0.089 x 4 = 0.357 */ {{1, 1}, {{1, 64, 104, 104}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 89296896.},
|
||||||
|
/* GFLOPS 0.356 x 1 = 0.356 */ {{1, 1}, {{1, 128, 104, 104}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 355803136.},
|
||||||
|
/* GFLOPS 0.355 x 1 = 0.355 */ {{1, 1}, {{1, 256, 52, 52}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 355110912.},
|
||||||
|
/* GFLOPS 0.355 x 1 = 0.355 */ {{1, 1}, {{1, 512, 26, 26}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 354764800.},
|
||||||
|
/* GFLOPS 0.355 x 1 = 0.355 */ {{1, 1}, {{1, 1024, 13, 13}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 354591744.},
|
||||||
|
/* GFLOPS 0.355 x 1 = 0.355 */ {{1, 1}, {{1, 2048, 13, 13}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 354505216.},
|
||||||
|
/* GFLOPS 0.177 x 2 = 0.353 */ {{1, 1}, {{1, 512, 26, 26}}, 255, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 176689500.},
|
||||||
|
/* GFLOPS 0.070 x 5 = 0.348 */ {{1, 1}, {{1, 128, 46, 46}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 69607936.},
|
||||||
/* GFLOPS 0.347 x 1 = 0.347 */ {{3, 3}, {{1, 128, 28, 28}}, 192, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 346967040.},
|
/* GFLOPS 0.347 x 1 = 0.347 */ {{3, 3}, {{1, 128, 28, 28}}, 192, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 346967040.},
|
||||||
/* GFLOPS 0.347 x 1 = 0.347 */ {{3, 3}, {{1, 128, 28, 28}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 346967040.},
|
/* GFLOPS 0.347 x 1 = 0.347 */ {{3, 3}, {{1, 128, 28, 28}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 346967040.},
|
||||||
/* GFLOPS 0.014 x 24 = 0.347 */ {{3, 3}, {{1, 128, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 14456960.},
|
/* GFLOPS 0.014 x 24 = 0.347 */ {{3, 3}, {{1, 128, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 14456960.},
|
||||||
|
/* GFLOPS 0.113 x 3 = 0.340 */ {{1, 1}, {{1, 1152, 16, 16}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 113295360.},
|
||||||
/* GFLOPS 0.053 x 6 = 0.320 */ {{1, 1}, {{1, 576, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 53277824.},
|
/* GFLOPS 0.053 x 6 = 0.320 */ {{1, 1}, {{1, 576, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 53277824.},
|
||||||
/* GFLOPS 0.319 x 1 = 0.319 */ {{3, 3}, {{1, 192, 19, 19}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 319482112.},
|
/* GFLOPS 0.319 x 1 = 0.319 */ {{3, 3}, {{1, 192, 19, 19}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 319482112.},
|
||||||
|
/* GFLOPS 0.317 x 1 = 0.317 */ {{3, 3}, {{1, 3, 300, 300}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 316800000.},
|
||||||
/* GFLOPS 0.315 x 1 = 0.315 */ {{3, 3}, {{1, 96, 75, 100}}, 96, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 315369600.},
|
/* GFLOPS 0.315 x 1 = 0.315 */ {{3, 3}, {{1, 96, 75, 100}}, 96, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 315369600.},
|
||||||
/* GFLOPS 0.103 x 3 = 0.309 */ {{1, 1}, {{1, 512, 7, 7}}, 2048, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102860800.},
|
/* GFLOPS 0.103 x 3 = 0.309 */ {{1, 1}, {{1, 512, 7, 7}}, 2048, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102860800.},
|
||||||
/* GFLOPS 0.103 x 3 = 0.309 */ {{1, 1}, {{1, 512, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102860800.},
|
/* GFLOPS 0.103 x 3 = 0.309 */ {{1, 1}, {{1, 512, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102860800.},
|
||||||
|
/* GFLOPS 0.154 x 2 = 0.309 */ {{1, 1}, {{1, 672, 32, 32}}, 112, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 154255360.},
|
||||||
/* GFLOPS 0.308 x 1 = 0.308 */ {{1, 1}, {{1, 320, 75, 100}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 307680000.},
|
/* GFLOPS 0.308 x 1 = 0.308 */ {{1, 1}, {{1, 320, 75, 100}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 307680000.},
|
||||||
|
/* GFLOPS 0.034 x 9 = 0.304 */ {{1, 1}, {{1, 64, 64, 64}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 33816576.},
|
||||||
/* GFLOPS 0.299 x 1 = 0.299 */ {{3, 3}, {{1, 256, 13, 13}}, 384, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 299105664.},
|
/* GFLOPS 0.299 x 1 = 0.299 */ {{3, 3}, {{1, 256, 13, 13}}, 384, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 299105664.},
|
||||||
/* GFLOPS 0.299 x 1 = 0.299 */ {{3, 3}, {{1, 384, 13, 13}}, 256, 2, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 299084032.},
|
/* GFLOPS 0.299 x 1 = 0.299 */ {{3, 3}, {{1, 384, 13, 13}}, 256, 2, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 299084032.},
|
||||||
/* GFLOPS 0.017 x 17 = 0.290 */ {{1, 1}, {{1, 32, 32, 64}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 17039360.},
|
/* GFLOPS 0.017 x 17 = 0.290 */ {{1, 1}, {{1, 32, 32, 64}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 17039360.},
|
||||||
/* GFLOPS 0.017 x 16 = 0.269 */ {{1, 1}, {{1, 128, 32, 64}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 16842752.},
|
/* GFLOPS 0.017 x 16 = 0.269 */ {{1, 1}, {{1, 128, 32, 64}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 16842752.},
|
||||||
/* GFLOPS 0.133 x 2 = 0.266 */ {{3, 3}, {{1, 128, 19, 19}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 133136800.},
|
/* GFLOPS 0.133 x 2 = 0.266 */ {{3, 3}, {{1, 128, 19, 19}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 133136800.},
|
||||||
|
/* GFLOPS 0.266 x 1 = 0.266 */ {{1, 1}, {{1, 384, 52, 52}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 266160128.},
|
||||||
|
/* GFLOPS 0.266 x 1 = 0.266 */ {{1, 1}, {{1, 768, 26, 26}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 265987072.},
|
||||||
/* GFLOPS 0.038 x 7 = 0.265 */ {{3, 3}, {{1, 16, 64, 128}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 37879808.},
|
/* GFLOPS 0.038 x 7 = 0.265 */ {{3, 3}, {{1, 16, 64, 128}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 37879808.},
|
||||||
|
/* GFLOPS 0.019 x 14 = 0.264 */ {{3, 3}, {{1, 64, 16, 16}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 18890752.},
|
||||||
|
/* GFLOPS 0.262 x 1 = 0.262 */ {{1, 1}, {{1, 2560, 20, 20}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 262195200.},
|
||||||
/* GFLOPS 0.126 x 2 = 0.252 */ {{3, 3}, {{1, 512, 5, 5}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 125812050.},
|
/* GFLOPS 0.126 x 2 = 0.252 */ {{3, 3}, {{1, 512, 5, 5}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 125812050.},
|
||||||
/* GFLOPS 0.248 x 1 = 0.248 */ {{1, 1}, {{1, 64, 150, 200}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 247680000.},
|
/* GFLOPS 0.248 x 1 = 0.248 */ {{1, 1}, {{1, 64, 150, 200}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 247680000.},
|
||||||
/* GFLOPS 0.040 x 6 = 0.240 */ {{1, 1}, {{1, 576, 19, 19}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 39958368.},
|
/* GFLOPS 0.040 x 6 = 0.240 */ {{1, 1}, {{1, 576, 19, 19}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 39958368.},
|
||||||
/* GFLOPS 0.080 x 3 = 0.240 */ {{3, 3}, {{1, 96, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 79893632.},
|
/* GFLOPS 0.080 x 3 = 0.240 */ {{3, 3}, {{1, 96, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 79893632.},
|
||||||
/* GFLOPS 0.240 x 1 = 0.240 */ {{3, 3}, {{1, 192, 38, 38}}, 192, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 239611584.},
|
/* GFLOPS 0.240 x 1 = 0.240 */ {{3, 3}, {{1, 192, 38, 38}}, 192, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 239611584.},
|
||||||
/* GFLOPS 0.240 x 1 = 0.240 */ {{3, 3}, {{1, 192, 19, 19}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 239611584.},
|
/* GFLOPS 0.240 x 1 = 0.240 */ {{3, 3}, {{1, 192, 19, 19}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 239611584.},
|
||||||
|
/* GFLOPS 0.079 x 3 = 0.237 */ {{1, 1}, {{1, 80, 32, 32}}, 480, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 79134720.},
|
||||||
/* GFLOPS 0.237 x 1 = 0.237 */ {{7, 7}, {{1, 3, 224, 224}}, 64, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", false, 236830720.},
|
/* GFLOPS 0.237 x 1 = 0.237 */ {{7, 7}, {{1, 3, 224, 224}}, 64, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", false, 236830720.},
|
||||||
/* GFLOPS 0.237 x 1 = 0.237 */ {{7, 7}, {{1, 3, 224, 224}}, 64, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 236830720.},
|
/* GFLOPS 0.237 x 1 = 0.237 */ {{7, 7}, {{1, 3, 224, 224}}, 64, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 236830720.},
|
||||||
|
/* GFLOPS 0.118 x 2 = 0.236 */ {{3, 3}, {{1, 32, 80, 80}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 118169600.},
|
||||||
|
/* GFLOPS 0.236 x 1 = 0.236 */ {{3, 3}, {{1, 256, 19, 19}}, 512, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 235980800.},
|
||||||
|
/* GFLOPS 0.116 x 2 = 0.231 */ {{1, 1}, {{1, 24, 128, 128}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 115605504.},
|
||||||
/* GFLOPS 0.111 x 2 = 0.221 */ {{3, 3}, {{1, 192, 10, 10}}, 320, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 110624000.},
|
/* GFLOPS 0.111 x 2 = 0.221 */ {{3, 3}, {{1, 192, 10, 10}}, 320, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 110624000.},
|
||||||
/* GFLOPS 0.213 x 1 = 0.213 */ {{3, 3}, {{1, 128, 38, 38}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 213018880.},
|
/* GFLOPS 0.213 x 1 = 0.213 */ {{3, 3}, {{1, 128, 38, 38}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 213018880.},
|
||||||
/* GFLOPS 0.213 x 1 = 0.213 */ {{3, 3}, {{1, 128, 19, 19}}, 256, 1, {1, 1}, {2, 2}, {2, 2}, {0, 0}, "", false, 213018880.},
|
/* GFLOPS 0.213 x 1 = 0.213 */ {{3, 3}, {{1, 128, 19, 19}}, 256, 1, {1, 1}, {2, 2}, {2, 2}, {0, 0}, "", false, 213018880.},
|
||||||
/* GFLOPS 0.107 x 2 = 0.213 */ {{3, 3}, {{1, 128, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 106509440.},
|
/* GFLOPS 0.107 x 2 = 0.213 */ {{3, 3}, {{1, 128, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 106509440.},
|
||||||
/* GFLOPS 0.213 x 1 = 0.213 */ {{3, 3}, {{1, 256, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 212972672.},
|
/* GFLOPS 0.213 x 1 = 0.213 */ {{3, 3}, {{1, 256, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 212972672.},
|
||||||
|
/* GFLOPS 0.213 x 1 = 0.213 */ {{3, 3}, {{1, 512, 38, 38}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 212949568.},
|
||||||
/* GFLOPS 0.212 x 1 = 0.212 */ {{7, 7}, {{1, 3, 300, 300}}, 32, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", true, 212400000.},
|
/* GFLOPS 0.212 x 1 = 0.212 */ {{7, 7}, {{1, 3, 300, 300}}, 32, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", true, 212400000.},
|
||||||
/* GFLOPS 0.211 x 1 = 0.211 */ {{11, 11}, {{1, 3, 227, 227}}, 96, 1, {4, 4}, {1, 1}, {0, 0}, {0, 0}, "", true, 211120800.},
|
/* GFLOPS 0.211 x 1 = 0.211 */ {{11, 11}, {{1, 3, 227, 227}}, 96, 1, {4, 4}, {1, 1}, {0, 0}, {0, 0}, "", true, 211120800.},
|
||||||
/* GFLOPS 0.210 x 1 = 0.210 */ {{3, 3}, {{1, 64, 38, 50}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 210307200.},
|
/* GFLOPS 0.210 x 1 = 0.210 */ {{3, 3}, {{1, 64, 38, 50}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 210307200.},
|
||||||
/* GFLOPS 0.210 x 1 = 0.210 */ {{1, 1}, {{1, 1024, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 209817600.},
|
/* GFLOPS 0.210 x 1 = 0.210 */ {{1, 1}, {{1, 1024, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 209817600.},
|
||||||
/* GFLOPS 0.210 x 1 = 0.210 */ {{1, 1}, {{1, 1024, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 209817600.},
|
/* GFLOPS 0.210 x 1 = 0.210 */ {{1, 1}, {{1, 1024, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 209817600.},
|
||||||
/* GFLOPS 0.104 x 2 = 0.208 */ {{3, 3}, {{1, 32, 75, 75}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 103860000.},
|
/* GFLOPS 0.104 x 2 = 0.208 */ {{3, 3}, {{1, 32, 75, 75}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", false, 103860000.},
|
||||||
|
/* GFLOPS 0.208 x 1 = 0.208 */ {{1, 1}, {{1, 16, 256, 256}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 207618048.},
|
||||||
/* GFLOPS 0.206 x 1 = 0.206 */ {{1, 1}, {{1, 256, 56, 56}}, 512, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "", false, 205922304.},
|
/* GFLOPS 0.206 x 1 = 0.206 */ {{1, 1}, {{1, 256, 56, 56}}, 512, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "", false, 205922304.},
|
||||||
/* GFLOPS 0.206 x 1 = 0.206 */ {{1, 1}, {{1, 256, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 205922304.},
|
/* GFLOPS 0.206 x 1 = 0.206 */ {{1, 1}, {{1, 256, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 205922304.},
|
||||||
/* GFLOPS 0.103 x 2 = 0.206 */ {{1, 1}, {{1, 256, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102961152.},
|
/* GFLOPS 0.103 x 2 = 0.206 */ {{1, 1}, {{1, 256, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 102961152.},
|
||||||
@@ -148,27 +279,35 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.190 x 1 = 0.190 */ {{1, 1}, {{1, 256, 38, 38}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 189637632.},
|
/* GFLOPS 0.190 x 1 = 0.190 */ {{1, 1}, {{1, 256, 38, 38}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 189637632.},
|
||||||
/* GFLOPS 0.190 x 1 = 0.190 */ {{1, 1}, {{1, 256, 38, 38}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 189637632.},
|
/* GFLOPS 0.190 x 1 = 0.190 */ {{1, 1}, {{1, 256, 38, 38}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 189637632.},
|
||||||
/* GFLOPS 0.047 x 4 = 0.190 */ {{1, 1}, {{1, 256, 38, 38}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 47409408.},
|
/* GFLOPS 0.047 x 4 = 0.190 */ {{1, 1}, {{1, 256, 38, 38}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 47409408.},
|
||||||
|
/* GFLOPS 0.189 x 1 = 0.189 */ {{1, 1}, {{1, 1024, 19, 19}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 189360384.},
|
||||||
/* GFLOPS 0.038 x 5 = 0.189 */ {{3, 3}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 37814272.},
|
/* GFLOPS 0.038 x 5 = 0.189 */ {{3, 3}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 37814272.},
|
||||||
|
/* GFLOPS 0.189 x 1 = 0.189 */ {{1, 1}, {{1, 1152, 16, 16}}, 320, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 188825600.},
|
||||||
/* GFLOPS 0.185 x 1 = 0.185 */ {{1, 1}, {{1, 128, 75, 75}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 185040000.},
|
/* GFLOPS 0.185 x 1 = 0.185 */ {{1, 1}, {{1, 128, 75, 75}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 185040000.},
|
||||||
/* GFLOPS 0.185 x 1 = 0.185 */ {{1, 1}, {{1, 128, 75, 75}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 185040000.},
|
/* GFLOPS 0.185 x 1 = 0.185 */ {{1, 1}, {{1, 128, 75, 75}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 185040000.},
|
||||||
/* GFLOPS 0.181 x 1 = 0.181 */ {{3, 3}, {{1, 160, 14, 14}}, 320, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 180696320.},
|
/* GFLOPS 0.181 x 1 = 0.181 */ {{3, 3}, {{1, 160, 14, 14}}, 320, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 180696320.},
|
||||||
/* GFLOPS 0.181 x 1 = 0.181 */ {{3, 3}, {{1, 160, 14, 14}}, 320, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 180696320.},
|
/* GFLOPS 0.181 x 1 = 0.181 */ {{3, 3}, {{1, 160, 14, 14}}, 320, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 180696320.},
|
||||||
/* GFLOPS 0.090 x 2 = 0.181 */ {{3, 3}, {{1, 224, 10, 10}}, 224, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 90339200.},
|
/* GFLOPS 0.090 x 2 = 0.181 */ {{3, 3}, {{1, 224, 10, 10}}, 224, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 90339200.},
|
||||||
/* GFLOPS 0.180 x 1 = 0.180 */ {{1, 1}, {{1, 224, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 180232192.},
|
/* GFLOPS 0.180 x 1 = 0.180 */ {{1, 1}, {{1, 224, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 180232192.},
|
||||||
|
/* GFLOPS 0.088 x 2 = 0.177 */ {{1, 1}, {{1, 1024, 13, 13}}, 255, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 88301655.},
|
||||||
/* GFLOPS 0.174 x 1 = 0.174 */ {{3, 3}, {{1, 96, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 173508608.},
|
/* GFLOPS 0.174 x 1 = 0.174 */ {{3, 3}, {{1, 96, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 173508608.},
|
||||||
/* GFLOPS 0.174 x 1 = 0.174 */ {{3, 3}, {{1, 96, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 173508608.},
|
/* GFLOPS 0.174 x 1 = 0.174 */ {{3, 3}, {{1, 96, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 173508608.},
|
||||||
/* GFLOPS 0.166 x 1 = 0.166 */ {{3, 3}, {{1, 160, 19, 19}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 166406560.},
|
/* GFLOPS 0.166 x 1 = 0.166 */ {{3, 3}, {{1, 160, 19, 19}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 166406560.},
|
||||||
/* GFLOPS 0.080 x 2 = 0.160 */ {{1, 1}, {{1, 576, 19, 19}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 79916736.},
|
/* GFLOPS 0.080 x 2 = 0.160 */ {{1, 1}, {{1, 576, 19, 19}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 79916736.},
|
||||||
/* GFLOPS 0.160 x 1 = 0.160 */ {{3, 3}, {{1, 128, 19, 19}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 159764160.},
|
/* GFLOPS 0.160 x 1 = 0.160 */ {{3, 3}, {{1, 128, 19, 19}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 159764160.},
|
||||||
|
/* GFLOPS 0.160 x 1 = 0.160 */ {{3, 3}, {{1, 1024, 19, 19}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 159703512.},
|
||||||
/* GFLOPS 0.159 x 1 = 0.159 */ {{7, 7}, {{1, 3, 300, 300}}, 24, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 159300000.},
|
/* GFLOPS 0.159 x 1 = 0.159 */ {{7, 7}, {{1, 3, 300, 300}}, 24, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 159300000.},
|
||||||
|
/* GFLOPS 0.080 x 2 = 0.159 */ {{1, 1}, {{1, 40, 64, 64}}, 240, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 79626240.},
|
||||||
|
/* GFLOPS 0.079 x 2 = 0.157 */ {{1, 1}, {{1, 480, 32, 32}}, 80, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 78725120.},
|
||||||
/* GFLOPS 0.155 x 1 = 0.155 */ {{1, 1}, {{1, 192, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 154542080.},
|
/* GFLOPS 0.155 x 1 = 0.155 */ {{1, 1}, {{1, 192, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 154542080.},
|
||||||
/* GFLOPS 0.146 x 1 = 0.146 */ {{3, 3}, {{1, 144, 14, 14}}, 288, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 146369664.},
|
/* GFLOPS 0.146 x 1 = 0.146 */ {{3, 3}, {{1, 144, 14, 14}}, 288, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 146369664.},
|
||||||
/* GFLOPS 0.146 x 1 = 0.146 */ {{3, 3}, {{1, 144, 14, 14}}, 288, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 146369664.},
|
/* GFLOPS 0.146 x 1 = 0.146 */ {{3, 3}, {{1, 144, 14, 14}}, 288, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 146369664.},
|
||||||
/* GFLOPS 0.072 x 2 = 0.144 */ {{1, 1}, {{1, 1024, 10, 10}}, 352, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 72124800.},
|
/* GFLOPS 0.072 x 2 = 0.144 */ {{1, 1}, {{1, 1024, 10, 10}}, 352, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 72124800.},
|
||||||
/* GFLOPS 0.140 x 1 = 0.140 */ {{1, 1}, {{1, 576, 38, 50}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 140204800.},
|
/* GFLOPS 0.140 x 1 = 0.140 */ {{1, 1}, {{1, 576, 38, 50}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 140204800.},
|
||||||
|
/* GFLOPS 0.139 x 1 = 0.139 */ {{3, 3}, {{1, 256, 5, 5}}, 1206, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 138961350.},
|
||||||
/* GFLOPS 0.017 x 8 = 0.138 */ {{1, 1}, {{1, 16, 64, 128}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 17301504.},
|
/* GFLOPS 0.017 x 8 = 0.138 */ {{1, 1}, {{1, 16, 64, 128}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 17301504.},
|
||||||
/* GFLOPS 0.067 x 2 = 0.133 */ {{1, 1}, {{1, 576, 19, 19}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 66597280.},
|
/* GFLOPS 0.067 x 2 = 0.133 */ {{1, 1}, {{1, 576, 19, 19}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 66597280.},
|
||||||
/* GFLOPS 0.133 x 1 = 0.133 */ {{3, 3}, {{1, 128, 38, 38}}, 160, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 133136800.},
|
/* GFLOPS 0.133 x 1 = 0.133 */ {{3, 3}, {{1, 128, 38, 38}}, 160, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 133136800.},
|
||||||
|
/* GFLOPS 0.044 x 3 = 0.133 */ {{1, 1}, {{1, 512, 13, 13}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 44345600.},
|
||||||
/* GFLOPS 0.129 x 1 = 0.129 */ {{1, 1}, {{1, 160, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 128851968.},
|
/* GFLOPS 0.129 x 1 = 0.129 */ {{1, 1}, {{1, 160, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 128851968.},
|
||||||
/* GFLOPS 0.128 x 1 = 0.128 */ {{3, 3}, {{1, 64, 24, 24}}, 192, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 127512576.},
|
/* GFLOPS 0.128 x 1 = 0.128 */ {{3, 3}, {{1, 64, 24, 24}}, 192, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 127512576.},
|
||||||
/* GFLOPS 0.120 x 1 = 0.120 */ {{5, 5}, {{1, 32, 28, 28}}, 96, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 120497664.},
|
/* GFLOPS 0.120 x 1 = 0.120 */ {{5, 5}, {{1, 32, 28, 28}}, 96, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 120497664.},
|
||||||
@@ -176,22 +315,35 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.040 x 3 = 0.120 */ {{1, 1}, {{1, 96, 19, 19}}, 576, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 40131648.},
|
/* GFLOPS 0.040 x 3 = 0.120 */ {{1, 1}, {{1, 96, 19, 19}}, 576, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 40131648.},
|
||||||
/* GFLOPS 0.118 x 1 = 0.118 */ {{1, 1}, {{1, 320, 38, 38}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 118477312.},
|
/* GFLOPS 0.118 x 1 = 0.118 */ {{1, 1}, {{1, 320, 38, 38}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 118477312.},
|
||||||
/* GFLOPS 0.017 x 7 = 0.118 */ {{1, 1}, {{1, 64, 64, 128}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 16908288.},
|
/* GFLOPS 0.017 x 7 = 0.118 */ {{1, 1}, {{1, 64, 64, 128}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 16908288.},
|
||||||
|
/* GFLOPS 0.118 x 1 = 0.118 */ {{3, 3}, {{1, 64, 80, 80}}, 64, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 118067200.},
|
||||||
|
/* GFLOPS 0.118 x 1 = 0.118 */ {{3, 3}, {{1, 64, 40, 40}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 118067200.},
|
||||||
/* GFLOPS 0.039 x 3 = 0.118 */ {{1, 1}, {{1, 1024, 10, 10}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 39340800.},
|
/* GFLOPS 0.039 x 3 = 0.118 */ {{1, 1}, {{1, 1024, 10, 10}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 39340800.},
|
||||||
|
/* GFLOPS 0.118 x 1 = 0.118 */ {{3, 3}, {{1, 128, 40, 40}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 118016000.},
|
||||||
|
/* GFLOPS 0.118 x 1 = 0.118 */ {{3, 3}, {{1, 128, 20, 20}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 118016000.},
|
||||||
|
/* GFLOPS 0.118 x 1 = 0.118 */ {{3, 3}, {{1, 256, 20, 20}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 117990400.},
|
||||||
/* GFLOPS 0.118 x 1 = 0.118 */ {{3, 3}, {{1, 256, 19, 19}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 117990400.},
|
/* GFLOPS 0.118 x 1 = 0.118 */ {{3, 3}, {{1, 256, 19, 19}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 117990400.},
|
||||||
/* GFLOPS 0.058 x 2 = 0.116 */ {{3, 3}, {{1, 16, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 58003456.},
|
/* GFLOPS 0.058 x 2 = 0.116 */ {{3, 3}, {{1, 16, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 58003456.},
|
||||||
/* GFLOPS 0.058 x 2 = 0.116 */ {{3, 3}, {{1, 32, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 57903104.},
|
/* GFLOPS 0.058 x 2 = 0.116 */ {{3, 3}, {{1, 32, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 57903104.},
|
||||||
/* GFLOPS 0.058 x 2 = 0.116 */ {{3, 3}, {{1, 64, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 57852928.},
|
/* GFLOPS 0.058 x 2 = 0.116 */ {{3, 3}, {{1, 64, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 57852928.},
|
||||||
/* GFLOPS 0.116 x 1 = 0.116 */ {{3, 3}, {{1, 128, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 115655680.},
|
/* GFLOPS 0.116 x 1 = 0.116 */ {{3, 3}, {{1, 128, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 115655680.},
|
||||||
/* GFLOPS 0.116 x 1 = 0.116 */ {{3, 3}, {{1, 128, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 115655680.},
|
/* GFLOPS 0.116 x 1 = 0.116 */ {{3, 3}, {{1, 128, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 115655680.},
|
||||||
|
/* GFLOPS 0.115 x 1 = 0.115 */ {{3, 3}, {{1, 3, 512, 512}}, 32, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 115343360.},
|
||||||
|
/* GFLOPS 0.114 x 1 = 0.114 */ {{1, 1}, {{1, 144, 128, 128}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 113639424.},
|
||||||
/* GFLOPS 0.112 x 1 = 0.112 */ {{1, 1}, {{1, 1024, 10, 10}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 111875400.},
|
/* GFLOPS 0.112 x 1 = 0.112 */ {{1, 1}, {{1, 1024, 10, 10}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 111875400.},
|
||||||
|
/* GFLOPS 0.110 x 1 = 0.110 */ {{1, 1}, {{1, 480, 32, 32}}, 112, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 110215168.},
|
||||||
|
/* GFLOPS 0.107 x 1 = 0.107 */ {{1, 1}, {{1, 64, 32, 32}}, 810, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 106997760.},
|
||||||
/* GFLOPS 0.036 x 3 = 0.107 */ {{1, 1}, {{1, 192, 38, 38}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 35580160.},
|
/* GFLOPS 0.036 x 3 = 0.107 */ {{1, 1}, {{1, 192, 38, 38}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 35580160.},
|
||||||
/* GFLOPS 0.107 x 1 = 0.107 */ {{3, 3}, {{1, 32, 75, 75}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 106648064.},
|
/* GFLOPS 0.107 x 1 = 0.107 */ {{3, 3}, {{1, 32, 75, 75}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 106648064.},
|
||||||
/* GFLOPS 0.107 x 1 = 0.107 */ {{3, 3}, {{1, 64, 38, 38}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 106555648.},
|
/* GFLOPS 0.107 x 1 = 0.107 */ {{3, 3}, {{1, 64, 38, 38}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 106555648.},
|
||||||
|
/* GFLOPS 0.105 x 1 = 0.105 */ {{1, 1}, {{1, 256, 40, 40}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 105062400.},
|
||||||
|
/* GFLOPS 0.105 x 1 = 0.105 */ {{1, 1}, {{1, 512, 20, 20}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 104960000.},
|
||||||
/* GFLOPS 0.105 x 1 = 0.105 */ {{1, 1}, {{1, 512, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 104960000.},
|
/* GFLOPS 0.105 x 1 = 0.105 */ {{1, 1}, {{1, 512, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 104960000.},
|
||||||
/* GFLOPS 0.105 x 1 = 0.105 */ {{1, 1}, {{1, 512, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 104960000.},
|
/* GFLOPS 0.105 x 1 = 0.105 */ {{1, 1}, {{1, 512, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 104960000.},
|
||||||
|
/* GFLOPS 0.105 x 1 = 0.105 */ {{1, 1}, {{1, 1024, 10, 10}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 104908800.},
|
||||||
/* GFLOPS 0.103 x 1 = 0.103 */ {{1, 1}, {{1, 128, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 103161856.},
|
/* GFLOPS 0.103 x 1 = 0.103 */ {{1, 1}, {{1, 128, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 103161856.},
|
||||||
/* GFLOPS 0.051 x 2 = 0.103 */ {{1, 1}, {{1, 256, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 51480576.},
|
/* GFLOPS 0.051 x 2 = 0.103 */ {{1, 1}, {{1, 256, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 51480576.},
|
||||||
/* GFLOPS 0.051 x 2 = 0.103 */ {{1, 1}, {{1, 256, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 51480576.},
|
/* GFLOPS 0.051 x 2 = 0.103 */ {{1, 1}, {{1, 256, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 51480576.},
|
||||||
|
/* GFLOPS 0.008 x 12 = 0.101 */ {{1, 1}, {{1, 64, 32, 32}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 8454144.},
|
||||||
/* GFLOPS 0.101 x 1 = 0.101 */ {{1, 1}, {{1, 512, 19, 19}}, 273, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 101016825.},
|
/* GFLOPS 0.101 x 1 = 0.101 */ {{1, 1}, {{1, 512, 19, 19}}, 273, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 101016825.},
|
||||||
/* GFLOPS 0.096 x 1 = 0.096 */ {{1, 1}, {{1, 480, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 96438272.},
|
/* GFLOPS 0.096 x 1 = 0.096 */ {{1, 1}, {{1, 480, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 96438272.},
|
||||||
/* GFLOPS 0.095 x 1 = 0.095 */ {{1, 1}, {{1, 128, 38, 38}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 95003648.},
|
/* GFLOPS 0.095 x 1 = 0.095 */ {{1, 1}, {{1, 128, 38, 38}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 95003648.},
|
||||||
@@ -208,8 +360,10 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.092 x 1 = 0.092 */ {{1, 1}, {{1, 192, 75, 100}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 92400000.},
|
/* GFLOPS 0.092 x 1 = 0.092 */ {{1, 1}, {{1, 192, 75, 100}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 92400000.},
|
||||||
/* GFLOPS 0.090 x 1 = 0.090 */ {{1, 1}, {{1, 448, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 90015744.},
|
/* GFLOPS 0.090 x 1 = 0.090 */ {{1, 1}, {{1, 448, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 90015744.},
|
||||||
/* GFLOPS 0.045 x 2 = 0.090 */ {{3, 3}, {{1, 576, 19, 19}}, 12, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 44918508.},
|
/* GFLOPS 0.045 x 2 = 0.090 */ {{3, 3}, {{1, 576, 19, 19}}, 12, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 44918508.},
|
||||||
|
/* GFLOPS 0.044 x 2 = 0.089 */ {{1, 1}, {{1, 256, 26, 26}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 44388864.},
|
||||||
/* GFLOPS 0.089 x 1 = 0.089 */ {{3, 3}, {{1, 112, 14, 14}}, 224, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 88554368.},
|
/* GFLOPS 0.089 x 1 = 0.089 */ {{3, 3}, {{1, 112, 14, 14}}, 224, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 88554368.},
|
||||||
/* GFLOPS 0.089 x 1 = 0.089 */ {{3, 3}, {{1, 112, 14, 14}}, 224, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 88554368.},
|
/* GFLOPS 0.089 x 1 = 0.089 */ {{3, 3}, {{1, 112, 14, 14}}, 224, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 88554368.},
|
||||||
|
/* GFLOPS 0.088 x 1 = 0.088 */ {{1, 1}, {{1, 256, 26, 26}}, 255, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 88430940.},
|
||||||
/* GFLOPS 0.021 x 4 = 0.084 */ {{5, 1}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {1, 1}, {2, 0}, {0, 0}, "", false, 21037056.},
|
/* GFLOPS 0.021 x 4 = 0.084 */ {{5, 1}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {1, 1}, {2, 0}, {0, 0}, "", false, 21037056.},
|
||||||
/* GFLOPS 0.021 x 4 = 0.084 */ {{1, 5}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {1, 1}, {0, 2}, {0, 0}, "", true, 21037056.},
|
/* GFLOPS 0.021 x 4 = 0.084 */ {{1, 5}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {1, 1}, {0, 2}, {0, 0}, "", true, 21037056.},
|
||||||
/* GFLOPS 0.084 x 1 = 0.084 */ {{1, 1}, {{1, 416, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 83593216.},
|
/* GFLOPS 0.084 x 1 = 0.084 */ {{1, 1}, {{1, 416, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 83593216.},
|
||||||
@@ -217,9 +371,13 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.040 x 2 = 0.080 */ {{1, 1}, {{1, 576, 19, 19}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39958368.},
|
/* GFLOPS 0.040 x 2 = 0.080 */ {{1, 1}, {{1, 576, 19, 19}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39958368.},
|
||||||
/* GFLOPS 0.040 x 2 = 0.079 */ {{1, 1}, {{1, 24, 75, 75}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39690000.},
|
/* GFLOPS 0.040 x 2 = 0.079 */ {{1, 1}, {{1, 24, 75, 75}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39690000.},
|
||||||
/* GFLOPS 0.040 x 2 = 0.079 */ {{3, 3}, {{1, 3, 300, 300}}, 32, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39600000.},
|
/* GFLOPS 0.040 x 2 = 0.079 */ {{3, 3}, {{1, 3, 300, 300}}, 32, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39600000.},
|
||||||
|
/* GFLOPS 0.079 x 1 = 0.079 */ {{1, 1}, {{1, 240, 64, 64}}, 40, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 78807040.},
|
||||||
|
/* GFLOPS 0.079 x 1 = 0.079 */ {{1, 1}, {{1, 384, 40, 40}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 78745600.},
|
||||||
/* GFLOPS 0.077 x 1 = 0.077 */ {{1, 1}, {{1, 96, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 77471744.},
|
/* GFLOPS 0.077 x 1 = 0.077 */ {{1, 1}, {{1, 96, 56, 56}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 77471744.},
|
||||||
/* GFLOPS 0.077 x 1 = 0.077 */ {{3, 3}, {{1, 192, 10, 10}}, 224, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 77436800.},
|
/* GFLOPS 0.077 x 1 = 0.077 */ {{3, 3}, {{1, 192, 10, 10}}, 224, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 77436800.},
|
||||||
/* GFLOPS 0.077 x 1 = 0.077 */ {{1, 1}, {{1, 384, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 77170688.},
|
/* GFLOPS 0.077 x 1 = 0.077 */ {{1, 1}, {{1, 384, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 77170688.},
|
||||||
|
/* GFLOPS 0.076 x 1 = 0.076 */ {{3, 3}, {{1, 3, 416, 416}}, 32, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", false, 76144640.},
|
||||||
|
/* GFLOPS 0.076 x 1 = 0.076 */ {{1, 1}, {{1, 96, 128, 128}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 75890688.},
|
||||||
/* GFLOPS 0.038 x 2 = 0.076 */ {{3, 3}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {8, 8}, {8, 8}, {0, 0}, "", true, 37814272.},
|
/* GFLOPS 0.038 x 2 = 0.076 */ {{3, 3}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {8, 8}, {8, 8}, {0, 0}, "", true, 37814272.},
|
||||||
/* GFLOPS 0.038 x 2 = 0.076 */ {{3, 3}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {4, 4}, {4, 4}, {0, 0}, "", true, 37814272.},
|
/* GFLOPS 0.038 x 2 = 0.076 */ {{3, 3}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {4, 4}, {4, 4}, {0, 0}, "", true, 37814272.},
|
||||||
/* GFLOPS 0.038 x 2 = 0.076 */ {{3, 3}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {2, 2}, {2, 2}, {0, 0}, "", true, 37814272.},
|
/* GFLOPS 0.038 x 2 = 0.076 */ {{3, 3}, {{1, 32, 32, 64}}, 32, 1, {1, 1}, {2, 2}, {2, 2}, {0, 0}, "", true, 37814272.},
|
||||||
@@ -230,6 +388,9 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.071 x 1 = 0.071 */ {{1, 1}, {{1, 24, 150, 150}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 70560000.},
|
/* GFLOPS 0.071 x 1 = 0.071 */ {{1, 1}, {{1, 24, 150, 150}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 70560000.},
|
||||||
/* GFLOPS 0.070 x 1 = 0.070 */ {{3, 3}, {{1, 96, 14, 14}}, 208, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 70487872.},
|
/* GFLOPS 0.070 x 1 = 0.070 */ {{3, 3}, {{1, 96, 14, 14}}, 208, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 70487872.},
|
||||||
/* GFLOPS 0.069 x 1 = 0.069 */ {{3, 3}, {{1, 96, 14, 14}}, 204, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 69132336.},
|
/* GFLOPS 0.069 x 1 = 0.069 */ {{3, 3}, {{1, 96, 14, 14}}, 204, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 69132336.},
|
||||||
|
/* GFLOPS 0.068 x 1 = 0.068 */ {{1, 1}, {{1, 32, 256, 256}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 68157440.},
|
||||||
|
/* GFLOPS 0.005 x 14 = 0.066 */ {{3, 3}, {{1, 64, 8, 8}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 4722688.},
|
||||||
|
/* GFLOPS 0.066 x 1 = 0.066 */ {{1, 1}, {{1, 672, 16, 16}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 66109440.},
|
||||||
/* GFLOPS 0.066 x 1 = 0.066 */ {{1, 1}, {{1, 1280, 10, 10}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 65561600.},
|
/* GFLOPS 0.066 x 1 = 0.066 */ {{1, 1}, {{1, 1280, 10, 10}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 65561600.},
|
||||||
/* GFLOPS 0.033 x 2 = 0.065 */ {{3, 3}, {{1, 48, 14, 14}}, 192, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 32551680.},
|
/* GFLOPS 0.033 x 2 = 0.065 */ {{3, 3}, {{1, 48, 14, 14}}, 192, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 32551680.},
|
||||||
/* GFLOPS 0.065 x 1 = 0.065 */ {{3, 3}, {{1, 192, 7, 7}}, 384, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 65046912.},
|
/* GFLOPS 0.065 x 1 = 0.065 */ {{3, 3}, {{1, 192, 7, 7}}, 384, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 65046912.},
|
||||||
@@ -239,6 +400,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.032 x 2 = 0.064 */ {{3, 3}, {{1, 96, 12, 12}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 31868928.},
|
/* GFLOPS 0.032 x 2 = 0.064 */ {{3, 3}, {{1, 96, 12, 12}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 31868928.},
|
||||||
/* GFLOPS 0.061 x 1 = 0.061 */ {{1, 1}, {{1, 960, 10, 10}}, 320, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 61472000.},
|
/* GFLOPS 0.061 x 1 = 0.061 */ {{1, 1}, {{1, 960, 10, 10}}, 320, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 61472000.},
|
||||||
/* GFLOPS 0.031 x 2 = 0.061 */ {{1, 1}, {{1, 960, 10, 10}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 30736000.},
|
/* GFLOPS 0.031 x 2 = 0.061 */ {{1, 1}, {{1, 960, 10, 10}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 30736000.},
|
||||||
|
/* GFLOPS 0.061 x 1 = 0.061 */ {{1, 1}, {{1, 512, 46, 46}}, 28, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 60729200.},
|
||||||
/* GFLOPS 0.060 x 1 = 0.060 */ {{3, 3}, {{1, 96, 38, 38}}, 96, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 59920224.},
|
/* GFLOPS 0.060 x 1 = 0.060 */ {{3, 3}, {{1, 96, 38, 38}}, 96, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 59920224.},
|
||||||
/* GFLOPS 0.059 x 1 = 0.059 */ {{1, 1}, {{1, 320, 38, 38}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 59238656.},
|
/* GFLOPS 0.059 x 1 = 0.059 */ {{1, 1}, {{1, 320, 38, 38}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 59238656.},
|
||||||
/* GFLOPS 0.059 x 1 = 0.059 */ {{3, 3}, {{1, 128, 19, 19}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 59008000.},
|
/* GFLOPS 0.059 x 1 = 0.059 */ {{3, 3}, {{1, 128, 19, 19}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 59008000.},
|
||||||
@@ -253,6 +415,11 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.053 x 1 = 0.053 */ {{3, 3}, {{1, 128, 38, 38}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 53254720.},
|
/* GFLOPS 0.053 x 1 = 0.053 */ {{3, 3}, {{1, 128, 38, 38}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 53254720.},
|
||||||
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 528, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 53036032.},
|
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 528, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 53036032.},
|
||||||
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 528, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 53036032.},
|
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 528, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 53036032.},
|
||||||
|
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 64, 80, 80}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 52838400.},
|
||||||
|
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 64, 40, 40}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 52838400.},
|
||||||
|
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 128, 80, 80}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 52633600.},
|
||||||
|
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 128, 20, 20}}, 512, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 52633600.},
|
||||||
|
/* GFLOPS 0.053 x 1 = 0.053 */ {{1, 1}, {{1, 256, 10, 10}}, 1024, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 52531200.},
|
||||||
/* GFLOPS 0.052 x 1 = 0.052 */ {{1, 1}, {{1, 1024, 10, 10}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 52454400.},
|
/* GFLOPS 0.052 x 1 = 0.052 */ {{1, 1}, {{1, 1024, 10, 10}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 52454400.},
|
||||||
/* GFLOPS 0.052 x 1 = 0.052 */ {{1, 1}, {{1, 1024, 10, 10}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 52454400.},
|
/* GFLOPS 0.052 x 1 = 0.052 */ {{1, 1}, {{1, 1024, 10, 10}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 52454400.},
|
||||||
/* GFLOPS 0.052 x 1 = 0.052 */ {{1, 1}, {{1, 1024, 10, 10}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 52454400.},
|
/* GFLOPS 0.052 x 1 = 0.052 */ {{1, 1}, {{1, 1024, 10, 10}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 52454400.},
|
||||||
@@ -268,6 +435,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.050 x 1 = 0.050 */ {{1, 1}, {{1, 992, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 49799680.},
|
/* GFLOPS 0.050 x 1 = 0.050 */ {{1, 1}, {{1, 992, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 49799680.},
|
||||||
/* GFLOPS 0.048 x 1 = 0.048 */ {{1, 1}, {{1, 960, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 48194048.},
|
/* GFLOPS 0.048 x 1 = 0.048 */ {{1, 1}, {{1, 960, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 48194048.},
|
||||||
/* GFLOPS 0.047 x 1 = 0.047 */ {{1, 1}, {{1, 256, 19, 19}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 47409408.},
|
/* GFLOPS 0.047 x 1 = 0.047 */ {{1, 1}, {{1, 256, 19, 19}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 47409408.},
|
||||||
|
/* GFLOPS 0.047 x 1 = 0.047 */ {{1, 1}, {{1, 144, 64, 64}}, 40, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 47349760.},
|
||||||
/* GFLOPS 0.047 x 1 = 0.047 */ {{1, 1}, {{1, 512, 38, 50}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 46740000.},
|
/* GFLOPS 0.047 x 1 = 0.047 */ {{1, 1}, {{1, 512, 38, 50}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 46740000.},
|
||||||
/* GFLOPS 0.047 x 1 = 0.047 */ {{1, 1}, {{1, 928, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 46588416.},
|
/* GFLOPS 0.047 x 1 = 0.047 */ {{1, 1}, {{1, 928, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 46588416.},
|
||||||
/* GFLOPS 0.046 x 1 = 0.046 */ {{1, 1}, {{1, 64, 75, 75}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 46440000.},
|
/* GFLOPS 0.046 x 1 = 0.046 */ {{1, 1}, {{1, 64, 75, 75}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 46440000.},
|
||||||
@@ -280,6 +448,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.045 x 1 = 0.045 */ {{3, 3}, {{1, 3, 227, 227}}, 64, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "", true, 44946880.},
|
/* GFLOPS 0.045 x 1 = 0.045 */ {{3, 3}, {{1, 3, 227, 227}}, 64, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "", true, 44946880.},
|
||||||
/* GFLOPS 0.044 x 1 = 0.044 */ {{3, 3}, {{1, 128, 19, 19}}, 192, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 44256000.},
|
/* GFLOPS 0.044 x 1 = 0.044 */ {{3, 3}, {{1, 128, 19, 19}}, 192, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 44256000.},
|
||||||
/* GFLOPS 0.044 x 1 = 0.044 */ {{3, 3}, {{1, 1024, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 44239200.},
|
/* GFLOPS 0.044 x 1 = 0.044 */ {{3, 3}, {{1, 1024, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 44239200.},
|
||||||
|
/* GFLOPS 0.044 x 1 = 0.044 */ {{1, 1}, {{1, 512, 13, 13}}, 255, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 44172375.},
|
||||||
/* GFLOPS 0.043 x 1 = 0.043 */ {{7, 7}, {{1, 3, 96, 96}}, 64, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", true, 43499520.},
|
/* GFLOPS 0.043 x 1 = 0.043 */ {{7, 7}, {{1, 3, 96, 96}}, 64, 1, {2, 2}, {1, 1}, {3, 3}, {0, 0}, "", true, 43499520.},
|
||||||
/* GFLOPS 0.043 x 1 = 0.043 */ {{1, 1}, {{1, 864, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 43377152.},
|
/* GFLOPS 0.043 x 1 = 0.043 */ {{1, 1}, {{1, 864, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 43377152.},
|
||||||
/* GFLOPS 0.042 x 1 = 0.042 */ {{1, 1}, {{1, 832, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 41771520.},
|
/* GFLOPS 0.042 x 1 = 0.042 */ {{1, 1}, {{1, 832, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 41771520.},
|
||||||
@@ -289,6 +458,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.040 x 1 = 0.040 */ {{3, 3}, {{1, 64, 19, 19}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 39958368.},
|
/* GFLOPS 0.040 x 1 = 0.040 */ {{3, 3}, {{1, 64, 19, 19}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 39958368.},
|
||||||
/* GFLOPS 0.040 x 1 = 0.040 */ {{3, 3}, {{1, 256, 19, 19}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 39932376.},
|
/* GFLOPS 0.040 x 1 = 0.040 */ {{3, 3}, {{1, 256, 19, 19}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 39932376.},
|
||||||
/* GFLOPS 0.040 x 1 = 0.040 */ {{3, 3}, {{1, 3, 300, 300}}, 32, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 39600000.},
|
/* GFLOPS 0.040 x 1 = 0.040 */ {{3, 3}, {{1, 3, 300, 300}}, 32, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 39600000.},
|
||||||
|
/* GFLOPS 0.039 x 1 = 0.039 */ {{1, 1}, {{1, 240, 32, 32}}, 80, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39403520.},
|
||||||
/* GFLOPS 0.039 x 1 = 0.039 */ {{1, 1}, {{1, 144, 75, 75}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39015000.},
|
/* GFLOPS 0.039 x 1 = 0.039 */ {{1, 1}, {{1, 144, 75, 75}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 39015000.},
|
||||||
/* GFLOPS 0.039 x 1 = 0.039 */ {{1, 1}, {{1, 192, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 38635520.},
|
/* GFLOPS 0.039 x 1 = 0.039 */ {{1, 1}, {{1, 192, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 38635520.},
|
||||||
/* GFLOPS 0.039 x 1 = 0.039 */ {{1, 1}, {{1, 768, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 38560256.},
|
/* GFLOPS 0.039 x 1 = 0.039 */ {{1, 1}, {{1, 768, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 38560256.},
|
||||||
@@ -297,9 +467,11 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.036 x 1 = 0.036 */ {{1, 1}, {{1, 480, 14, 14}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 36164352.},
|
/* GFLOPS 0.036 x 1 = 0.036 */ {{1, 1}, {{1, 480, 14, 14}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 36164352.},
|
||||||
/* GFLOPS 0.018 x 2 = 0.036 */ {{1, 1}, {{1, 192, 38, 38}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 17790080.},
|
/* GFLOPS 0.018 x 2 = 0.036 */ {{1, 1}, {{1, 192, 38, 38}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 17790080.},
|
||||||
/* GFLOPS 0.035 x 1 = 0.035 */ {{1, 1}, {{1, 704, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 35348992.},
|
/* GFLOPS 0.035 x 1 = 0.035 */ {{1, 1}, {{1, 704, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 35348992.},
|
||||||
|
/* GFLOPS 0.035 x 1 = 0.035 */ {{1, 1}, {{1, 512, 46, 46}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 34702400.},
|
||||||
/* GFLOPS 0.034 x 1 = 0.034 */ {{1, 1}, {{1, 672, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 33743360.},
|
/* GFLOPS 0.034 x 1 = 0.034 */ {{1, 1}, {{1, 672, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 33743360.},
|
||||||
/* GFLOPS 0.034 x 1 = 0.034 */ {{1, 1}, {{1, 128, 32, 64}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 33685504.},
|
/* GFLOPS 0.034 x 1 = 0.034 */ {{1, 1}, {{1, 128, 32, 64}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 33685504.},
|
||||||
/* GFLOPS 0.034 x 1 = 0.034 */ {{2, 2}, {{1, 64, 64, 128}}, 32, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "", false, 33619968.},
|
/* GFLOPS 0.034 x 1 = 0.034 */ {{2, 2}, {{1, 64, 64, 128}}, 32, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "", false, 33619968.},
|
||||||
|
/* GFLOPS 0.033 x 1 = 0.033 */ {{3, 3}, {{1, 256, 3, 3}}, 804, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 33350724.},
|
||||||
/* GFLOPS 0.033 x 1 = 0.033 */ {{1, 1}, {{1, 528, 14, 14}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 33147520.},
|
/* GFLOPS 0.033 x 1 = 0.033 */ {{1, 1}, {{1, 528, 14, 14}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 33147520.},
|
||||||
/* GFLOPS 0.033 x 1 = 0.033 */ {{1, 1}, {{1, 528, 14, 14}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 33147520.},
|
/* GFLOPS 0.033 x 1 = 0.033 */ {{1, 1}, {{1, 528, 14, 14}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 33147520.},
|
||||||
/* GFLOPS 0.033 x 1 = 0.033 */ {{1, 1}, {{1, 1024, 10, 10}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 32784000.},
|
/* GFLOPS 0.033 x 1 = 0.033 */ {{1, 1}, {{1, 1024, 10, 10}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 32784000.},
|
||||||
@@ -307,24 +479,29 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.032 x 1 = 0.032 */ {{1, 1}, {{1, 512, 14, 14}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 32144000.},
|
/* GFLOPS 0.032 x 1 = 0.032 */ {{1, 1}, {{1, 512, 14, 14}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 32144000.},
|
||||||
/* GFLOPS 0.032 x 1 = 0.032 */ {{1, 1}, {{1, 640, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 32137728.},
|
/* GFLOPS 0.032 x 1 = 0.032 */ {{1, 1}, {{1, 640, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 32137728.},
|
||||||
/* GFLOPS 0.032 x 1 = 0.032 */ {{1, 1}, {{1, 508, 14, 14}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 31893120.},
|
/* GFLOPS 0.032 x 1 = 0.032 */ {{1, 1}, {{1, 508, 14, 14}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 31893120.},
|
||||||
|
/* GFLOPS 0.011 x 3 = 0.032 */ {{1, 1}, {{1, 320, 16, 16}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 10502144.},
|
||||||
/* GFLOPS 0.031 x 1 = 0.031 */ {{1, 1}, {{1, 832, 7, 7}}, 384, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 31328640.},
|
/* GFLOPS 0.031 x 1 = 0.031 */ {{1, 1}, {{1, 832, 7, 7}}, 384, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 31328640.},
|
||||||
/* GFLOPS 0.031 x 1 = 0.031 */ {{1, 1}, {{1, 832, 7, 7}}, 384, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 31328640.},
|
/* GFLOPS 0.031 x 1 = 0.031 */ {{1, 1}, {{1, 832, 7, 7}}, 384, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 31328640.},
|
||||||
/* GFLOPS 0.031 x 1 = 0.031 */ {{1, 1}, {{1, 608, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 30532096.},
|
/* GFLOPS 0.031 x 1 = 0.031 */ {{1, 1}, {{1, 608, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 30532096.},
|
||||||
|
/* GFLOPS 0.015 x 2 = 0.030 */ {{1, 1}, {{1, 128, 46, 46}}, 28, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 15226736.},
|
||||||
/* GFLOPS 0.015 x 2 = 0.030 */ {{5, 5}, {{1, 24, 14, 14}}, 64, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 15065344.},
|
/* GFLOPS 0.015 x 2 = 0.030 */ {{5, 5}, {{1, 24, 14, 14}}, 64, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 15065344.},
|
||||||
/* GFLOPS 0.015 x 2 = 0.030 */ {{5, 5}, {{1, 24, 14, 14}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 15065344.},
|
/* GFLOPS 0.015 x 2 = 0.030 */ {{5, 5}, {{1, 24, 14, 14}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 15065344.},
|
||||||
/* GFLOPS 0.015 x 2 = 0.030 */ {{5, 5}, {{1, 48, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 15059072.},
|
/* GFLOPS 0.015 x 2 = 0.030 */ {{5, 5}, {{1, 48, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 15059072.},
|
||||||
/* GFLOPS 0.029 x 1 = 0.029 */ {{3, 3}, {{1, 256, 10, 10}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 29497600.},
|
/* GFLOPS 0.029 x 1 = 0.029 */ {{3, 3}, {{1, 256, 10, 10}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 29497600.},
|
||||||
|
/* GFLOPS 0.015 x 2 = 0.029 */ {{1, 1}, {{1, 112, 32, 32}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 14745600.},
|
||||||
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 192, 28, 28}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 28976640.},
|
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 192, 28, 28}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 28976640.},
|
||||||
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 192, 28, 28}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 28976640.},
|
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 192, 28, 28}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 28976640.},
|
||||||
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 512, 14, 14}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 28929600.},
|
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 512, 14, 14}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 28929600.},
|
||||||
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 512, 14, 14}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 28929600.},
|
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 512, 14, 14}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 28929600.},
|
||||||
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 576, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 28926464.},
|
/* GFLOPS 0.029 x 1 = 0.029 */ {{1, 1}, {{1, 576, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 28926464.},
|
||||||
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 544, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 27320832.},
|
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 544, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 27320832.},
|
||||||
|
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 64, 16, 16}}, 810, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 26749440.},
|
||||||
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 384, 19, 19}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 26650464.},
|
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 384, 19, 19}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 26650464.},
|
||||||
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 576, 19, 19}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 26638912.},
|
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 576, 19, 19}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 26638912.},
|
||||||
/* GFLOPS 0.027 x 1 = 0.027 */ {{3, 3}, {{1, 128, 38, 38}}, 8, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 26627360.},
|
/* GFLOPS 0.027 x 1 = 0.027 */ {{3, 3}, {{1, 128, 38, 38}}, 8, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 26627360.},
|
||||||
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 528, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 26518016.},
|
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 528, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 26518016.},
|
||||||
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 528, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 26518016.},
|
/* GFLOPS 0.027 x 1 = 0.027 */ {{1, 1}, {{1, 528, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 26518016.},
|
||||||
|
/* GFLOPS 0.009 x 3 = 0.026 */ {{1, 1}, {{1, 128, 46, 46}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 8700992.},
|
||||||
/* GFLOPS 0.026 x 1 = 0.026 */ {{1, 1}, {{1, 96, 75, 75}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 26055000.},
|
/* GFLOPS 0.026 x 1 = 0.026 */ {{1, 1}, {{1, 96, 75, 75}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 26055000.},
|
||||||
/* GFLOPS 0.026 x 1 = 0.026 */ {{1, 1}, {{1, 64, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 25890816.},
|
/* GFLOPS 0.026 x 1 = 0.026 */ {{1, 1}, {{1, 64, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 25890816.},
|
||||||
/* GFLOPS 0.026 x 1 = 0.026 */ {{1, 1}, {{1, 64, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 25890816.},
|
/* GFLOPS 0.026 x 1 = 0.026 */ {{1, 1}, {{1, 64, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 25890816.},
|
||||||
@@ -336,6 +513,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.013 x 2 = 0.026 */ {{1, 1}, {{1, 256, 28, 28}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 12870144.},
|
/* GFLOPS 0.013 x 2 = 0.026 */ {{1, 1}, {{1, 256, 28, 28}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 12870144.},
|
||||||
/* GFLOPS 0.026 x 1 = 0.026 */ {{1, 1}, {{1, 512, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 25715200.},
|
/* GFLOPS 0.026 x 1 = 0.026 */ {{1, 1}, {{1, 512, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 25715200.},
|
||||||
/* GFLOPS 0.013 x 2 = 0.026 */ {{1, 1}, {{1, 512, 14, 14}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 12857600.},
|
/* GFLOPS 0.013 x 2 = 0.026 */ {{1, 1}, {{1, 512, 14, 14}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 12857600.},
|
||||||
|
/* GFLOPS 0.002 x 12 = 0.025 */ {{1, 1}, {{1, 64, 16, 16}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 2113536.},
|
||||||
/* GFLOPS 0.024 x 1 = 0.024 */ {{1, 1}, {{1, 480, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 24109568.},
|
/* GFLOPS 0.024 x 1 = 0.024 */ {{1, 1}, {{1, 480, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 24109568.},
|
||||||
/* GFLOPS 0.024 x 1 = 0.024 */ {{1, 1}, {{1, 128, 38, 38}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "", false, 23750912.},
|
/* GFLOPS 0.024 x 1 = 0.024 */ {{1, 1}, {{1, 128, 38, 38}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "", false, 23750912.},
|
||||||
/* GFLOPS 0.024 x 1 = 0.024 */ {{1, 1}, {{1, 256, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 23704704.},
|
/* GFLOPS 0.024 x 1 = 0.024 */ {{1, 1}, {{1, 256, 19, 19}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 23704704.},
|
||||||
@@ -345,7 +523,9 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.023 x 1 = 0.023 */ {{1, 1}, {{1, 448, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 22503936.},
|
/* GFLOPS 0.023 x 1 = 0.023 */ {{1, 1}, {{1, 448, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 22503936.},
|
||||||
/* GFLOPS 0.023 x 1 = 0.023 */ {{1, 1}, {{1, 512, 14, 14}}, 112, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 22500800.},
|
/* GFLOPS 0.023 x 1 = 0.023 */ {{1, 1}, {{1, 512, 14, 14}}, 112, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 22500800.},
|
||||||
/* GFLOPS 0.022 x 1 = 0.022 */ {{1, 1}, {{1, 508, 14, 14}}, 112, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 22325184.},
|
/* GFLOPS 0.022 x 1 = 0.022 */ {{1, 1}, {{1, 508, 14, 14}}, 112, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 22325184.},
|
||||||
|
/* GFLOPS 0.022 x 1 = 0.022 */ {{3, 3}, {{1, 512, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 22120800.},
|
||||||
/* GFLOPS 0.021 x 1 = 0.021 */ {{3, 3}, {{1, 128, 12, 12}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 21242880.},
|
/* GFLOPS 0.021 x 1 = 0.021 */ {{3, 3}, {{1, 128, 12, 12}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 21242880.},
|
||||||
|
/* GFLOPS 0.021 x 1 = 0.021 */ {{1, 1}, {{1, 40, 64, 64}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 21233664.},
|
||||||
/* GFLOPS 0.021 x 1 = 0.021 */ {{1, 1}, {{1, 416, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 20898304.},
|
/* GFLOPS 0.021 x 1 = 0.021 */ {{1, 1}, {{1, 416, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 20898304.},
|
||||||
/* GFLOPS 0.021 x 1 = 0.021 */ {{1, 1}, {{1, 832, 7, 7}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 20885760.},
|
/* GFLOPS 0.021 x 1 = 0.021 */ {{1, 1}, {{1, 832, 7, 7}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 20885760.},
|
||||||
/* GFLOPS 0.021 x 1 = 0.021 */ {{1, 1}, {{1, 832, 7, 7}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 20885760.},
|
/* GFLOPS 0.021 x 1 = 0.021 */ {{1, 1}, {{1, 832, 7, 7}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 20885760.},
|
||||||
@@ -360,6 +540,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.019 x 1 = 0.019 */ {{1, 1}, {{1, 192, 28, 28}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 19317760.},
|
/* GFLOPS 0.019 x 1 = 0.019 */ {{1, 1}, {{1, 192, 28, 28}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 19317760.},
|
||||||
/* GFLOPS 0.019 x 1 = 0.019 */ {{1, 1}, {{1, 192, 28, 28}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 19317760.},
|
/* GFLOPS 0.019 x 1 = 0.019 */ {{1, 1}, {{1, 192, 28, 28}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 19317760.},
|
||||||
/* GFLOPS 0.019 x 1 = 0.019 */ {{1, 1}, {{1, 384, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 19292672.},
|
/* GFLOPS 0.019 x 1 = 0.019 */ {{1, 1}, {{1, 384, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 19292672.},
|
||||||
|
/* GFLOPS 0.019 x 1 = 0.019 */ {{1, 1}, {{1, 64, 64, 64}}, 36, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 19021824.},
|
||||||
/* GFLOPS 0.018 x 1 = 0.018 */ {{1, 1}, {{1, 576, 10, 10}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 18448000.},
|
/* GFLOPS 0.018 x 1 = 0.018 */ {{1, 1}, {{1, 576, 10, 10}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 18448000.},
|
||||||
/* GFLOPS 0.018 x 1 = 0.018 */ {{1, 1}, {{1, 480, 14, 14}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 18082176.},
|
/* GFLOPS 0.018 x 1 = 0.018 */ {{1, 1}, {{1, 480, 14, 14}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 18082176.},
|
||||||
/* GFLOPS 0.018 x 1 = 0.018 */ {{1, 1}, {{1, 480, 14, 14}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 18082176.},
|
/* GFLOPS 0.018 x 1 = 0.018 */ {{1, 1}, {{1, 480, 14, 14}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 18082176.},
|
||||||
@@ -371,13 +552,16 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.016 x 1 = 0.016 */ {{1, 1}, {{1, 832, 7, 7}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 15664320.},
|
/* GFLOPS 0.016 x 1 = 0.016 */ {{1, 1}, {{1, 832, 7, 7}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 15664320.},
|
||||||
/* GFLOPS 0.015 x 1 = 0.015 */ {{5, 5}, {{1, 48, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 15059072.},
|
/* GFLOPS 0.015 x 1 = 0.015 */ {{5, 5}, {{1, 48, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 15059072.},
|
||||||
/* GFLOPS 0.015 x 1 = 0.015 */ {{5, 5}, {{1, 32, 12, 12}}, 64, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 14754816.},
|
/* GFLOPS 0.015 x 1 = 0.015 */ {{5, 5}, {{1, 32, 12, 12}}, 64, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 14754816.},
|
||||||
|
/* GFLOPS 0.015 x 1 = 0.015 */ {{3, 3}, {{1, 128, 10, 10}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 14752000.},
|
||||||
/* GFLOPS 0.014 x 1 = 0.014 */ {{1, 1}, {{1, 288, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 14475776.},
|
/* GFLOPS 0.014 x 1 = 0.014 */ {{1, 1}, {{1, 288, 14, 14}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 14475776.},
|
||||||
/* GFLOPS 0.014 x 1 = 0.014 */ {{1, 1}, {{1, 512, 5, 5}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 13991250.},
|
/* GFLOPS 0.014 x 1 = 0.014 */ {{1, 1}, {{1, 512, 5, 5}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 13991250.},
|
||||||
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 144, 38, 38}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 13354112.},
|
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 144, 38, 38}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 13354112.},
|
||||||
/* GFLOPS 0.007 x 2 = 0.013 */ {{1, 1}, {{1, 16, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6623232.},
|
/* GFLOPS 0.007 x 2 = 0.013 */ {{1, 1}, {{1, 16, 56, 56}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6623232.},
|
||||||
|
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 512, 10, 10}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 13120000.},
|
||||||
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 832, 7, 7}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 13053600.},
|
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 832, 7, 7}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 13053600.},
|
||||||
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 832, 7, 7}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 13053600.},
|
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 832, 7, 7}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 13053600.},
|
||||||
/* GFLOPS 0.007 x 2 = 0.013 */ {{1, 1}, {{1, 32, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6522880.},
|
/* GFLOPS 0.007 x 2 = 0.013 */ {{1, 1}, {{1, 32, 28, 28}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6522880.},
|
||||||
|
/* GFLOPS 0.001 x 11 = 0.013 */ {{3, 3}, {{1, 64, 4, 4}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1180672.},
|
||||||
/* GFLOPS 0.006 x 2 = 0.013 */ {{1, 1}, {{1, 64, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6472704.},
|
/* GFLOPS 0.006 x 2 = 0.013 */ {{1, 1}, {{1, 64, 14, 14}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6472704.},
|
||||||
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 128, 56, 56}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 12895232.},
|
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 128, 56, 56}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 12895232.},
|
||||||
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 256, 28, 28}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 12870144.},
|
/* GFLOPS 0.013 x 1 = 0.013 */ {{1, 1}, {{1, 256, 28, 28}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 12870144.},
|
||||||
@@ -394,6 +578,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.012 x 1 = 0.012 */ {{1, 1}, {{1, 640, 6, 6}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 11805696.},
|
/* GFLOPS 0.012 x 1 = 0.012 */ {{1, 1}, {{1, 640, 6, 6}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 11805696.},
|
||||||
/* GFLOPS 0.012 x 1 = 0.012 */ {{1, 1}, {{1, 928, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 11647104.},
|
/* GFLOPS 0.012 x 1 = 0.012 */ {{1, 1}, {{1, 928, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 11647104.},
|
||||||
/* GFLOPS 0.011 x 1 = 0.011 */ {{1, 1}, {{1, 896, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 11245696.},
|
/* GFLOPS 0.011 x 1 = 0.011 */ {{1, 1}, {{1, 896, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 11245696.},
|
||||||
|
/* GFLOPS 0.011 x 1 = 0.011 */ {{1, 1}, {{1, 256, 13, 13}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 11097216.},
|
||||||
/* GFLOPS 0.011 x 1 = 0.011 */ {{3, 3}, {{1, 256, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 11061600.},
|
/* GFLOPS 0.011 x 1 = 0.011 */ {{3, 3}, {{1, 256, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 11061600.},
|
||||||
/* GFLOPS 0.006 x 2 = 0.011 */ {{3, 3}, {{1, 512, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 5530200.},
|
/* GFLOPS 0.006 x 2 = 0.011 */ {{3, 3}, {{1, 512, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 5530200.},
|
||||||
/* GFLOPS 0.011 x 1 = 0.011 */ {{1, 1}, {{1, 864, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 10844288.},
|
/* GFLOPS 0.011 x 1 = 0.011 */ {{1, 1}, {{1, 864, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 10844288.},
|
||||||
@@ -417,13 +602,13 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.008 x 1 = 0.008 */ {{1, 1}, {{1, 608, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 7633024.},
|
/* GFLOPS 0.008 x 1 = 0.008 */ {{1, 1}, {{1, 608, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 7633024.},
|
||||||
/* GFLOPS 0.008 x 1 = 0.008 */ {{5, 5}, {{1, 16, 14, 14}}, 48, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 7535808.},
|
/* GFLOPS 0.008 x 1 = 0.008 */ {{5, 5}, {{1, 16, 14, 14}}, 48, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 7535808.},
|
||||||
/* GFLOPS 0.008 x 1 = 0.008 */ {{5, 5}, {{1, 16, 14, 14}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 7535808.},
|
/* GFLOPS 0.008 x 1 = 0.008 */ {{5, 5}, {{1, 16, 14, 14}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 7535808.},
|
||||||
/* GFLOPS 0.004 x 2 = 0.007 */ {{3, 3}, {{1, 64, 5, 5}}, 128, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 3689600.},
|
|
||||||
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 640, 6, 6}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 7378560.},
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 640, 6, 6}}, 160, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 7378560.},
|
||||||
/* GFLOPS 0.004 x 2 = 0.007 */ {{1, 1}, {{1, 48, 14, 14}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 3650304.},
|
/* GFLOPS 0.004 x 2 = 0.007 */ {{1, 1}, {{1, 48, 14, 14}}, 192, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 3650304.},
|
||||||
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 384, 14, 14}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 7234752.},
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 384, 14, 14}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 7234752.},
|
||||||
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 576, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 7231616.},
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 576, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 7231616.},
|
||||||
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 256, 12, 12}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 7091712.},
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 256, 12, 12}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 7091712.},
|
||||||
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 544, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 6830208.},
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 544, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 6830208.},
|
||||||
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 64, 8, 8}}, 810, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 6687360.},
|
||||||
/* GFLOPS 0.007 x 1 = 0.007 */ {{3, 3}, {{1, 160, 6, 6}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 6637824.},
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{3, 3}, {{1, 160, 6, 6}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 6637824.},
|
||||||
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 528, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6629504.},
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 528, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6629504.},
|
||||||
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 528, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 6629504.},
|
/* GFLOPS 0.007 x 1 = 0.007 */ {{1, 1}, {{1, 528, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 6629504.},
|
||||||
@@ -434,11 +619,13 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.006 x 1 = 0.006 */ {{1, 1}, {{1, 512, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 6428800.},
|
/* GFLOPS 0.006 x 1 = 0.006 */ {{1, 1}, {{1, 512, 7, 7}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 6428800.},
|
||||||
/* GFLOPS 0.006 x 1 = 0.006 */ {{1, 1}, {{1, 512, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6428800.},
|
/* GFLOPS 0.006 x 1 = 0.006 */ {{1, 1}, {{1, 512, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6428800.},
|
||||||
/* GFLOPS 0.006 x 1 = 0.006 */ {{1, 1}, {{1, 512, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 6428800.},
|
/* GFLOPS 0.006 x 1 = 0.006 */ {{1, 1}, {{1, 512, 14, 14}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 6428800.},
|
||||||
|
/* GFLOPS 0.001 x 12 = 0.006 */ {{1, 1}, {{1, 64, 8, 8}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 528384.},
|
||||||
/* GFLOPS 0.006 x 1 = 0.006 */ {{3, 3}, {{1, 256, 10, 10}}, 12, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 5530800.},
|
/* GFLOPS 0.006 x 1 = 0.006 */ {{3, 3}, {{1, 256, 10, 10}}, 12, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 5530800.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 192, 12, 12}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 5322240.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 192, 12, 12}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 5322240.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{3, 3}, {{1, 128, 5, 5}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 5310720.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{3, 3}, {{1, 128, 5, 5}}, 256, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 5310720.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{3, 3}, {{1, 128, 5, 5}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 5310720.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{3, 3}, {{1, 128, 5, 5}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 5310720.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{3, 3}, {{1, 128, 5, 5}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 5310720.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{3, 3}, {{1, 128, 5, 5}}, 256, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 5310720.},
|
||||||
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{3, 3}, {{1, 128, 5, 5}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 5310720.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 1024, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4917600.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 1024, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4917600.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 1024, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 4917600.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 1024, 10, 10}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 4917600.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 192, 28, 28}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4829440.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 192, 28, 28}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4829440.},
|
||||||
@@ -446,6 +633,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 256, 14, 14}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4826304.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 256, 14, 14}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4826304.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 512, 14, 14}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 4821600.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 512, 14, 14}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 4821600.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 508, 14, 14}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 4783968.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 508, 14, 14}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 4783968.},
|
||||||
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 64, 32, 32}}, 36, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 4755456.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 64, 24, 24}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4755456.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 64, 24, 24}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4755456.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 256, 12, 12}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4727808.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 256, 12, 12}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4727808.},
|
||||||
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 1024, 3, 3}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4720896.},
|
/* GFLOPS 0.005 x 1 = 0.005 */ {{1, 1}, {{1, 1024, 3, 3}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 4720896.},
|
||||||
@@ -455,6 +643,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.004 x 1 = 0.004 */ {{1, 1}, {{1, 16, 128, 256}}, 4, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 4325376.},
|
/* GFLOPS 0.004 x 1 = 0.004 */ {{1, 1}, {{1, 16, 128, 256}}, 4, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 4325376.},
|
||||||
/* GFLOPS 0.004 x 1 = 0.004 */ {{1, 1}, {{1, 64, 64, 128}}, 4, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 4227072.},
|
/* GFLOPS 0.004 x 1 = 0.004 */ {{1, 1}, {{1, 64, 64, 128}}, 4, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", false, 4227072.},
|
||||||
/* GFLOPS 0.004 x 1 = 0.004 */ {{1, 1}, {{1, 832, 7, 7}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 3916080.},
|
/* GFLOPS 0.004 x 1 = 0.004 */ {{1, 1}, {{1, 832, 7, 7}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 3916080.},
|
||||||
|
/* GFLOPS 0.004 x 1 = 0.004 */ {{3, 3}, {{1, 256, 1, 1}}, 804, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 3705636.},
|
||||||
/* GFLOPS 0.004 x 1 = 0.004 */ {{5, 5}, {{1, 16, 12, 12}}, 32, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 3691008.},
|
/* GFLOPS 0.004 x 1 = 0.004 */ {{5, 5}, {{1, 16, 12, 12}}, 32, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 3691008.},
|
||||||
/* GFLOPS 0.004 x 1 = 0.004 */ {{3, 3}, {{1, 64, 10, 10}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 3689600.},
|
/* GFLOPS 0.004 x 1 = 0.004 */ {{3, 3}, {{1, 64, 10, 10}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 3689600.},
|
||||||
/* GFLOPS 0.004 x 1 = 0.004 */ {{5, 5}, {{1, 32, 6, 6}}, 64, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 3688704.},
|
/* GFLOPS 0.004 x 1 = 0.004 */ {{5, 5}, {{1, 32, 6, 6}}, 64, 1, {1, 1}, {1, 1}, {2, 2}, {0, 0}, "", true, 3688704.},
|
||||||
@@ -470,6 +659,7 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 480, 14, 14}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 3013696.},
|
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 480, 14, 14}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 3013696.},
|
||||||
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 320, 12, 12}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2953728.},
|
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 320, 12, 12}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2953728.},
|
||||||
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 640, 6, 6}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2951424.},
|
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 640, 6, 6}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2951424.},
|
||||||
|
/* GFLOPS 0.003 x 1 = 0.003 */ {{3, 3}, {{1, 256, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 2765400.},
|
||||||
/* GFLOPS 0.003 x 1 = 0.003 */ {{3, 3}, {{1, 128, 5, 5}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 2655360.},
|
/* GFLOPS 0.003 x 1 = 0.003 */ {{3, 3}, {{1, 128, 5, 5}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 2655360.},
|
||||||
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 832, 7, 7}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2610720.},
|
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 832, 7, 7}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2610720.},
|
||||||
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 256, 3, 3}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 2520882.},
|
/* GFLOPS 0.003 x 1 = 0.003 */ {{1, 1}, {{1, 256, 3, 3}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 2520882.},
|
||||||
@@ -482,32 +672,46 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 508, 4, 4}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 2082816.},
|
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 508, 4, 4}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 2082816.},
|
||||||
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 1024, 1, 1}}, 1000, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2049000.},
|
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 1024, 1, 1}}, 1000, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 2049000.},
|
||||||
/* GFLOPS 0.001 x 2 = 0.002 */ {{3, 3}, {{1, 256, 3, 3}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 995544.},
|
/* GFLOPS 0.001 x 2 = 0.002 */ {{3, 3}, {{1, 256, 3, 3}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 995544.},
|
||||||
/* GFLOPS 0.001 x 2 = 0.002 */ {{3, 3}, {{1, 128, 5, 5}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 922000.},
|
|
||||||
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 1024, 3, 3}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1770336.},
|
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 1024, 3, 3}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1770336.},
|
||||||
|
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 64, 4, 4}}, 810, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 1671840.},
|
||||||
|
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 32, 80, 80}}, 4, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1664000.},
|
||||||
|
/* GFLOPS 0.002 x 1 = 0.002 */ {{1, 1}, {{1, 256, 5, 5}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1641600.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 640, 6, 6}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1475712.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 640, 6, 6}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1475712.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 128, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1383000.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 128, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 1383000.},
|
||||||
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 64, 5, 5}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1328256.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 736, 3, 3}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1272672.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 736, 3, 3}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 1272672.},
|
||||||
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 64, 16, 16}}, 36, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 1188864.},
|
||||||
|
/* GFLOPS 0.000 x 9 = 0.001 */ {{1, 1}, {{1, 64, 4, 4}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 132096.},
|
||||||
|
/* GFLOPS 0.001 x 2 = 0.001 */ {{1, 1}, {{1, 256, 3, 3}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 590976.},
|
||||||
/* GFLOPS 0.001 x 2 = 0.001 */ {{1, 1}, {{1, 256, 3, 3}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 590976.},
|
/* GFLOPS 0.001 x 2 = 0.001 */ {{1, 1}, {{1, 256, 3, 3}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 590976.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 128, 3, 3}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1180160.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 128, 3, 3}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 1180160.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 2, 2}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1120392.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 2, 2}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1120392.},
|
||||||
/* GFLOPS 0.000 x 2 = 0.001 */ {{3, 3}, {{1, 128, 5, 5}}, 8, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 461000.},
|
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 192, 12, 12}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 887040.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 192, 12, 12}}, 16, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 887040.},
|
||||||
/* GFLOPS 0.000 x 2 = 0.001 */ {{3, 3}, {{1, 256, 2, 2}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 442464.},
|
/* GFLOPS 0.000 x 2 = 0.001 */ {{3, 3}, {{1, 256, 2, 2}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 442464.},
|
||||||
/* GFLOPS 0.000 x 2 = 0.001 */ {{1, 1}, {{1, 128, 5, 5}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 411200.},
|
/* GFLOPS 0.000 x 2 = 0.001 */ {{1, 1}, {{1, 32, 80, 80}}, 1, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 416000.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 128, 5, 5}}, 12, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 691500.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 128, 5, 5}}, 12, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 691500.},
|
||||||
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 256, 3, 3}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 663696.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 640, 2, 2}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 655872.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 640, 2, 2}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 655872.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 512, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 615000.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 512, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 615000.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 512, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 615000.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 512, 5, 5}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 615000.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 128, 3, 3}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 592128.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 128, 3, 3}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 592128.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 3, 3}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 590976.},
|
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 3, 3}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 590976.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 3, 3}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 590976.},
|
||||||
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{3, 3}, {{1, 128, 3, 3}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 590080.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 3, 3}}, 126, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 581742.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 3, 3}}, 126, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 581742.},
|
||||||
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 4, 4}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 525312.},
|
/* GFLOPS 0.001 x 1 = 0.001 */ {{1, 1}, {{1, 256, 4, 4}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 525312.},
|
||||||
|
/* GFLOPS 0.000 x 4 = 0.000 */ {{1, 1}, {{1, 48, 1, 1}}, 1152, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 111744.},
|
||||||
|
/* GFLOPS 0.000 x 4 = 0.000 */ {{1, 1}, {{1, 1152, 1, 1}}, 48, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 110640.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 5, 5}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 411200.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 128, 3, 3}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 331920.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 192, 5, 5}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 308000.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 192, 5, 5}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 308000.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 64, 8, 8}}, 36, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 297216.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 2, 2}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 263168.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 2, 2}}, 256, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 263168.},
|
||||||
/* GFLOPS 0.000 x 2 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 131328.},
|
/* GFLOPS 0.000 x 2 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 131328.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 126, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 258552.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 126, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 258552.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 1024, 1, 1}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 196704.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 1024, 1, 1}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 196704.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 128, 3, 3}}, 8, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 165960.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 3, 3}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 148032.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 3, 3}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 147584.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 2, 2}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 147584.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 2, 2}}, 128, 1, {2, 2}, {1, 1}, {1, 1}, {0, 0}, "", true, 147584.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 2, 2}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 147584.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 2, 2}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 147584.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 2, 2}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 147584.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 2, 2}}, 128, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 147584.},
|
||||||
@@ -515,16 +719,32 @@ static const ConvParam_t testConvolutionConfigs[] = {
|
|||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 1, 1}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 140322.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 1, 1}}, 546, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 140322.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 131328.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 131328.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 131328.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 64, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 131328.},
|
||||||
|
/* GFLOPS 0.000 x 3 = 0.000 */ {{1, 1}, {{1, 28, 1, 1}}, 672, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 38304.},
|
||||||
|
/* GFLOPS 0.000 x 3 = 0.000 */ {{1, 1}, {{1, 672, 1, 1}}, 28, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 37660.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 3, 3}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 110808.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 3, 3}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 110808.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 3, 3}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 110808.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 3, 3}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 110808.},
|
||||||
/* GFLOPS 0.000 x 2 = 0.000 */ {{3, 3}, {{1, 128, 1, 1}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 55320.},
|
/* GFLOPS 0.000 x 2 = 0.000 */ {{3, 3}, {{1, 128, 1, 1}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 55320.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 64, 4, 4}}, 36, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "VALID", true, 74304.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 2, 2}}, 64, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 73792.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 64, 2, 2}}, 64, 1, {2, 2}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 73792.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 256, 1, 1}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 73744.},
|
||||||
|
/* GFLOPS 0.000 x 3 = 0.000 */ {{1, 1}, {{1, 20, 1, 1}}, 480, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 19680.},
|
||||||
|
/* GFLOPS 0.000 x 3 = 0.000 */ {{1, 1}, {{1, 480, 1, 1}}, 20, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 19220.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 49248.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 49248.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 49248.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 256, 2, 2}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 49248.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 128, 1, 1}}, 16, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 36880.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 1, 1}}, 126, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 32382.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 1, 1}}, 126, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 32382.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{3, 3}, {{1, 128, 1, 1}}, 8, 1, {1, 1}, {1, 1}, {1, 1}, {0, 0}, "", true, 18440.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 64, 1, 1}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 16512.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 64, 1, 1}}, 128, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", false, 16512.},
|
||||||
|
/* GFLOPS 0.000 x 2 = 0.000 */ {{1, 1}, {{1, 10, 1, 1}}, 240, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 5040.},
|
||||||
|
/* GFLOPS 0.000 x 2 = 0.000 */ {{1, 1}, {{1, 240, 1, 1}}, 10, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 4810.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 1, 1}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6168.},
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 1, 1}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "", true, 6168.},
|
||||||
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 1, 1}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 6168.}
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 128, 1, 1}}, 24, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 6168.},
|
||||||
|
/* GFLOPS 0.000 x 2 = 0.000 */ {{1, 1}, {{1, 6, 1, 1}}, 144, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1872.},
|
||||||
|
/* GFLOPS 0.000 x 2 = 0.000 */ {{1, 1}, {{1, 144, 1, 1}}, 6, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 1734.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 4, 1, 1}}, 96, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 864.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 96, 1, 1}}, 4, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 772.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 8, 1, 1}}, 32, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 544.},
|
||||||
|
/* GFLOPS 0.000 x 1 = 0.000 */ {{1, 1}, {{1, 32, 1, 1}}, 8, 1, {1, 1}, {1, 1}, {0, 0}, {0, 0}, "SAME", true, 520.}
|
||||||
};
|
};
|
||||||
struct ConvParamID
|
struct ConvParamID
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -306,16 +306,13 @@ public:
|
|||||||
|
|
||||||
caffe::LayerParameter* binLayer = netBinary.mutable_layer(li);
|
caffe::LayerParameter* binLayer = netBinary.mutable_layer(li);
|
||||||
const int numBlobs = binLayer->blobs_size();
|
const int numBlobs = binLayer->blobs_size();
|
||||||
|
std::vector<caffe::BlobProto*> blobs(numBlobs);
|
||||||
|
binLayer->mutable_blobs()->ExtractSubrange(0, numBlobs, blobs.data());
|
||||||
layerParams.blobs.resize(numBlobs);
|
layerParams.blobs.resize(numBlobs);
|
||||||
for (int bi = 0; bi < numBlobs; bi++)
|
for (int bi = 0; bi < numBlobs; bi++)
|
||||||
{
|
{
|
||||||
blobFromProto(binLayer->blobs(bi), layerParams.blobs[bi]);
|
blobFromProto(*blobs[bi], layerParams.blobs[bi]);
|
||||||
}
|
delete blobs[bi];
|
||||||
binLayer->clear_blobs();
|
|
||||||
CV_Assert(numBlobs == binLayer->blobs().ClearedCount());
|
|
||||||
for (int bi = 0; bi < numBlobs; bi++)
|
|
||||||
{
|
|
||||||
delete binLayer->mutable_blobs()->ReleaseCleared();
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -92,6 +92,7 @@
|
|||||||
#ifdef HAVE_PROTOBUF
|
#ifdef HAVE_PROTOBUF
|
||||||
#include <google/protobuf/io/coded_stream.h>
|
#include <google/protobuf/io/coded_stream.h>
|
||||||
#include <google/protobuf/io/zero_copy_stream_impl.h>
|
#include <google/protobuf/io/zero_copy_stream_impl.h>
|
||||||
|
#include <google/protobuf/stubs/common.h>
|
||||||
#include <google/protobuf/text_format.h>
|
#include <google/protobuf/text_format.h>
|
||||||
|
|
||||||
#include <opencv2/core.hpp>
|
#include <opencv2/core.hpp>
|
||||||
@@ -1111,7 +1112,11 @@ static const int kProtoReadBytesLimit = INT_MAX; // Max size of 2 GB minus 1 by
|
|||||||
|
|
||||||
bool ReadProtoFromBinary(ZeroCopyInputStream* input, Message *proto) {
|
bool ReadProtoFromBinary(ZeroCopyInputStream* input, Message *proto) {
|
||||||
CodedInputStream coded_input(input);
|
CodedInputStream coded_input(input);
|
||||||
|
#if GOOGLE_PROTOBUF_VERSION >= 3006000
|
||||||
|
coded_input.SetTotalBytesLimit(kProtoReadBytesLimit);
|
||||||
|
#else
|
||||||
coded_input.SetTotalBytesLimit(kProtoReadBytesLimit, 536870912);
|
coded_input.SetTotalBytesLimit(kProtoReadBytesLimit, 536870912);
|
||||||
|
#endif
|
||||||
|
|
||||||
return proto->ParseFromCodedStream(&coded_input);
|
return proto->ParseFromCodedStream(&coded_input);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -470,7 +470,7 @@ namespace cv {
|
|||||||
fused_layer_names.push_back(last_layer);
|
fused_layer_names.push_back(last_layer);
|
||||||
}
|
}
|
||||||
|
|
||||||
void setYolo(int classes, const std::vector<int>& mask, const std::vector<float>& anchors, float thresh, float nms_threshold, float scale_x_y)
|
void setYolo(int classes, const std::vector<int>& mask, const std::vector<float>& anchors, float thresh, float nms_threshold, float scale_x_y, int new_coords)
|
||||||
{
|
{
|
||||||
cv::dnn::LayerParams region_param;
|
cv::dnn::LayerParams region_param;
|
||||||
region_param.name = "Region-name";
|
region_param.name = "Region-name";
|
||||||
@@ -484,6 +484,7 @@ namespace cv {
|
|||||||
region_param.set<float>("thresh", thresh);
|
region_param.set<float>("thresh", thresh);
|
||||||
region_param.set<float>("nms_threshold", nms_threshold);
|
region_param.set<float>("nms_threshold", nms_threshold);
|
||||||
region_param.set<float>("scale_x_y", scale_x_y);
|
region_param.set<float>("scale_x_y", scale_x_y);
|
||||||
|
region_param.set<int>("new_coords", new_coords);
|
||||||
|
|
||||||
std::vector<float> usedAnchors(numAnchors * 2);
|
std::vector<float> usedAnchors(numAnchors * 2);
|
||||||
for (int i = 0; i < numAnchors; ++i)
|
for (int i = 0; i < numAnchors; ++i)
|
||||||
@@ -882,6 +883,7 @@ namespace cv {
|
|||||||
float thresh = getParam<float>(layer_params, "thresh", 0.2);
|
float thresh = getParam<float>(layer_params, "thresh", 0.2);
|
||||||
float nms_threshold = getParam<float>(layer_params, "nms_threshold", 0.0);
|
float nms_threshold = getParam<float>(layer_params, "nms_threshold", 0.0);
|
||||||
float scale_x_y = getParam<float>(layer_params, "scale_x_y", 1.0);
|
float scale_x_y = getParam<float>(layer_params, "scale_x_y", 1.0);
|
||||||
|
int new_coords = getParam<int>(layer_params, "new_coords", 0);
|
||||||
|
|
||||||
std::string anchors_values = getParam<std::string>(layer_params, "anchors", std::string());
|
std::string anchors_values = getParam<std::string>(layer_params, "anchors", std::string());
|
||||||
CV_Assert(!anchors_values.empty());
|
CV_Assert(!anchors_values.empty());
|
||||||
@@ -894,7 +896,7 @@ namespace cv {
|
|||||||
CV_Assert(classes > 0 && num_of_anchors > 0 && (num_of_anchors * 2) == anchors_vec.size());
|
CV_Assert(classes > 0 && num_of_anchors > 0 && (num_of_anchors * 2) == anchors_vec.size());
|
||||||
|
|
||||||
setParams.setPermute(false);
|
setParams.setPermute(false);
|
||||||
setParams.setYolo(classes, mask_vec, anchors_vec, thresh, nms_threshold, scale_x_y);
|
setParams.setYolo(classes, mask_vec, anchors_vec, thresh, nms_threshold, scale_x_y, new_coords);
|
||||||
}
|
}
|
||||||
else {
|
else {
|
||||||
CV_Error(cv::Error::StsParseError, "Unknown layer type: " + layer_type);
|
CV_Error(cv::Error::StsParseError, "Unknown layer type: " + layer_type);
|
||||||
|
|||||||
+21
-8
@@ -1944,7 +1944,10 @@ struct Net::Impl : public detail::NetImplBase
|
|||||||
|
|
||||||
Ptr<InfEngineNgraphNode> ieNode = node.dynamicCast<InfEngineNgraphNode>();
|
Ptr<InfEngineNgraphNode> ieNode = node.dynamicCast<InfEngineNgraphNode>();
|
||||||
CV_Assert(!ieNode.empty());
|
CV_Assert(!ieNode.empty());
|
||||||
ieNode->net->reset();
|
|
||||||
|
CV_Assert(ieNode->net);
|
||||||
|
InfEngineNgraphNet& ienet = *ieNode->net;
|
||||||
|
ienet.reset();
|
||||||
|
|
||||||
for (it = layers.begin(); it != layers.end(); ++it)
|
for (it = layers.begin(); it != layers.end(); ++it)
|
||||||
{
|
{
|
||||||
@@ -1961,16 +1964,26 @@ struct Net::Impl : public detail::NetImplBase
|
|||||||
{
|
{
|
||||||
for (int i = 0; i < ld.outputBlobsWrappers.size(); ++i)
|
for (int i = 0; i < ld.outputBlobsWrappers.size(); ++i)
|
||||||
{
|
{
|
||||||
InferenceEngine::DataPtr dataPtr = ngraphDataNode(ld.outputBlobsWrappers[i]);
|
auto it = ienet.outputsDesc.find(ld.name);
|
||||||
dataPtr->setName(ld.name);
|
if (it != ienet.outputsDesc.end())
|
||||||
|
{
|
||||||
|
const InferenceEngine::TensorDesc& descriptor = it->second;
|
||||||
|
InferenceEngine::DataPtr dataPtr = ngraphDataOutputNode(ld.outputBlobsWrappers[i], descriptor, ld.name);
|
||||||
|
dataPtr->setName(ld.name);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
InferenceEngine::DataPtr dataPtr = ngraphDataNode(ld.outputBlobsWrappers[i]);
|
||||||
|
dataPtr->setName(ld.name);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
ieNode->net->addBlobs(ld.inputBlobsWrappers);
|
ienet.addBlobs(ld.inputBlobsWrappers);
|
||||||
ieNode->net->addBlobs(ld.outputBlobsWrappers);
|
ienet.addBlobs(ld.outputBlobsWrappers);
|
||||||
ld.skip = true;
|
ld.skip = true;
|
||||||
}
|
}
|
||||||
layers[lastLayerId].skip = false;
|
layers[lastLayerId].skip = false;
|
||||||
ieNode->net->init((Target)preferableTarget);
|
ienet.init((Target)preferableTarget);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -3719,8 +3732,8 @@ void Net::forward(OutputArrayOfArrays outputBlobs,
|
|||||||
matvec.push_back(impl->getBlob(pins[i]));
|
matvec.push_back(impl->getBlob(pins[i]));
|
||||||
}
|
}
|
||||||
|
|
||||||
std::vector<Mat> & outputvec = *(std::vector<Mat> *)outputBlobs.getObj();
|
outputBlobs.create((int)matvec.size(), 1, CV_32F/*FIXIT*/, -1); // allocate vector
|
||||||
outputvec = matvec;
|
outputBlobs.assign(matvec);
|
||||||
}
|
}
|
||||||
|
|
||||||
void Net::forward(std::vector<std::vector<Mat> >& outputBlobs,
|
void Net::forward(std::vector<std::vector<Mat> >& outputBlobs,
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ Mutex& getInitializationMutex();
|
|||||||
void initializeLayerFactory();
|
void initializeLayerFactory();
|
||||||
|
|
||||||
namespace detail {
|
namespace detail {
|
||||||
|
#define CALL_MEMBER_FN(object, ptrToMemFn) ((object).*(ptrToMemFn))
|
||||||
|
|
||||||
struct NetImplBase
|
struct NetImplBase
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -789,21 +789,32 @@ void NgraphBackendLayer::forward(InputArrayOfArrays inputs, OutputArrayOfArrays
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
static InferenceEngine::Layout estimateLayout(const Mat& m)
|
static InferenceEngine::Layout estimateLayout(int dims)
|
||||||
{
|
{
|
||||||
if (m.dims == 4)
|
if (dims == 4)
|
||||||
return InferenceEngine::Layout::NCHW;
|
return InferenceEngine::Layout::NCHW;
|
||||||
else if (m.dims == 3)
|
else if (dims == 3)
|
||||||
return InferenceEngine::Layout::CHW;
|
return InferenceEngine::Layout::CHW;
|
||||||
else if (m.dims == 2)
|
else if (dims == 2)
|
||||||
return InferenceEngine::Layout::NC;
|
return InferenceEngine::Layout::NC;
|
||||||
else if (m.dims == 1)
|
else if (dims == 1)
|
||||||
return InferenceEngine::Layout::C;
|
return InferenceEngine::Layout::C;
|
||||||
else if (m.dims == 5)
|
else if (dims == 5)
|
||||||
return InferenceEngine::Layout::NCDHW;
|
return InferenceEngine::Layout::NCDHW;
|
||||||
else
|
else
|
||||||
return InferenceEngine::Layout::ANY;
|
return InferenceEngine::Layout::ANY;
|
||||||
}
|
}
|
||||||
|
static inline
|
||||||
|
InferenceEngine::Layout estimateLayout(size_t dims)
|
||||||
|
{
|
||||||
|
return estimateLayout((int)dims);
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline
|
||||||
|
InferenceEngine::Layout estimateLayout(const Mat& m)
|
||||||
|
{
|
||||||
|
return estimateLayout(m.dims);
|
||||||
|
}
|
||||||
|
|
||||||
static InferenceEngine::DataPtr wrapToInfEngineDataNode(const Mat& m, const std::string& name = "")
|
static InferenceEngine::DataPtr wrapToInfEngineDataNode(const Mat& m, const std::string& name = "")
|
||||||
{
|
{
|
||||||
@@ -839,6 +850,7 @@ InferenceEngine::Blob::Ptr wrapToNgraphBlob(const Mat& m, InferenceEngine::Layou
|
|||||||
|
|
||||||
NgraphBackendWrapper::NgraphBackendWrapper(int targetId, const cv::Mat& m)
|
NgraphBackendWrapper::NgraphBackendWrapper(int targetId, const cv::Mat& m)
|
||||||
: BackendWrapper(DNN_BACKEND_INFERENCE_ENGINE_NGRAPH, targetId)
|
: BackendWrapper(DNN_BACKEND_INFERENCE_ENGINE_NGRAPH, targetId)
|
||||||
|
, host((Mat*)&m)
|
||||||
{
|
{
|
||||||
dataPtr = wrapToInfEngineDataNode(m);
|
dataPtr = wrapToInfEngineDataNode(m);
|
||||||
blob = wrapToNgraphBlob(m, estimateLayout(m));
|
blob = wrapToNgraphBlob(m, estimateLayout(m));
|
||||||
@@ -890,7 +902,11 @@ InferenceEngine::Blob::Ptr copyBlob(const InferenceEngine::Blob::Ptr& blob)
|
|||||||
copy = InferenceEngine::make_shared_blob<uint8_t>(description);
|
copy = InferenceEngine::make_shared_blob<uint8_t>(description);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
CV_Error(Error::StsNotImplemented, "Unsupported blob precision");
|
{
|
||||||
|
std::ostringstream msg;
|
||||||
|
msg << precision;
|
||||||
|
CV_Error_(Error::StsNotImplemented, ("Unsupported blob precision: %s", msg.str().c_str()));
|
||||||
|
}
|
||||||
copy->allocate();
|
copy->allocate();
|
||||||
return copy;
|
return copy;
|
||||||
}
|
}
|
||||||
@@ -903,6 +919,66 @@ InferenceEngine::DataPtr ngraphDataNode(const Ptr<BackendWrapper>& ptr)
|
|||||||
return p->dataPtr;
|
return p->dataPtr;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static
|
||||||
|
InferenceEngine::Blob::Ptr reallocateBlob(Mat &m, const InferenceEngine::TensorDesc& description)
|
||||||
|
{
|
||||||
|
auto dims = description.getDims();
|
||||||
|
auto layout = estimateLayout(dims.size());
|
||||||
|
MatShape matShape(dims.begin(), dims.end());
|
||||||
|
if (description.getPrecision() == InferenceEngine::Precision::FP32)
|
||||||
|
{
|
||||||
|
m.create(matShape, CV_32FC1);
|
||||||
|
return InferenceEngine::make_shared_blob<float>(
|
||||||
|
{description.getPrecision(), dims, layout}, (float*)m.data);
|
||||||
|
}
|
||||||
|
else if (description.getPrecision() == InferenceEngine::Precision::I32)
|
||||||
|
{
|
||||||
|
m.create(matShape, CV_32SC1);
|
||||||
|
return InferenceEngine::make_shared_blob<int>(
|
||||||
|
{description.getPrecision(), dims, layout}, (int*)m.data);
|
||||||
|
}
|
||||||
|
else if (description.getPrecision() == InferenceEngine::Precision::U8)
|
||||||
|
{
|
||||||
|
m.create(matShape, CV_8UC1);
|
||||||
|
return InferenceEngine::make_shared_blob<uchar>(
|
||||||
|
{description.getPrecision(), dims, layout}, (uchar*)m.data);
|
||||||
|
}
|
||||||
|
std::ostringstream msg;
|
||||||
|
msg << "Unsupported IE precision: " << description.getPrecision();
|
||||||
|
CV_Error(Error::StsNotImplemented, msg.str());
|
||||||
|
}
|
||||||
|
|
||||||
|
InferenceEngine::DataPtr ngraphDataOutputNode(
|
||||||
|
const Ptr<BackendWrapper>& ptr,
|
||||||
|
const InferenceEngine::TensorDesc& description,
|
||||||
|
const std::string name)
|
||||||
|
{
|
||||||
|
CV_Assert(!ptr.empty());
|
||||||
|
Ptr<NgraphBackendWrapper> p = ptr.dynamicCast<NgraphBackendWrapper>();
|
||||||
|
CV_Assert(!p.empty());
|
||||||
|
NgraphBackendWrapper& w = *p;
|
||||||
|
const InferenceEngine::TensorDesc& blobDesc = w.blob.get()->getTensorDesc();
|
||||||
|
auto dims = description.getDims();
|
||||||
|
bool reallocate = false;
|
||||||
|
if (blobDesc.getPrecision() != description.getPrecision())
|
||||||
|
{
|
||||||
|
reallocate = true;
|
||||||
|
CV_LOG_WARNING(NULL, "Reallocate output '" << name << "' blob due to wrong precision: " << blobDesc.getPrecision() << " => " << description.getPrecision() << " ndims=" << dims.size());
|
||||||
|
}
|
||||||
|
if (dims.size() != blobDesc.getDims().size())
|
||||||
|
{
|
||||||
|
reallocate = true;
|
||||||
|
CV_LOG_WARNING(NULL, "Reallocate output '" << name << "' blob due to wrong dims: " << blobDesc.getDims().size() << " => " << dims.size());
|
||||||
|
}
|
||||||
|
if (reallocate)
|
||||||
|
{
|
||||||
|
auto layout = estimateLayout(dims.size());
|
||||||
|
w.dataPtr = InferenceEngine::DataPtr(new InferenceEngine::Data(name,
|
||||||
|
{description.getPrecision(), dims, layout}));
|
||||||
|
w.blob = reallocateBlob(*w.host, description);
|
||||||
|
}
|
||||||
|
return w.dataPtr;
|
||||||
|
}
|
||||||
|
|
||||||
void forwardNgraph(const std::vector<Ptr<BackendWrapper> >& outBlobsWrappers,
|
void forwardNgraph(const std::vector<Ptr<BackendWrapper> >& outBlobsWrappers,
|
||||||
Ptr<BackendNode>& node, bool isAsync)
|
Ptr<BackendNode>& node, bool isAsync)
|
||||||
@@ -918,6 +994,13 @@ void InfEngineNgraphNet::reset()
|
|||||||
allBlobs.clear();
|
allBlobs.clear();
|
||||||
infRequests.clear();
|
infRequests.clear();
|
||||||
isInit = false;
|
isInit = false;
|
||||||
|
|
||||||
|
outputsDesc.clear();
|
||||||
|
for (const auto& it : cnn.getOutputsInfo())
|
||||||
|
{
|
||||||
|
const std::string& name = it.first;
|
||||||
|
outputsDesc.insert({name, it.second->getTensorDesc()});
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void InfEngineNgraphNet::addBlobs(const std::vector<cv::Ptr<BackendWrapper> >& ptrs)
|
void InfEngineNgraphNet::addBlobs(const std::vector<cv::Ptr<BackendWrapper> >& ptrs)
|
||||||
|
|||||||
@@ -54,7 +54,8 @@ public:
|
|||||||
void setNodePtr(std::shared_ptr<ngraph::Node>* ptr);
|
void setNodePtr(std::shared_ptr<ngraph::Node>* ptr);
|
||||||
|
|
||||||
void reset();
|
void reset();
|
||||||
private:
|
|
||||||
|
//private:
|
||||||
detail::NetImplBase& netImpl_;
|
detail::NetImplBase& netImpl_;
|
||||||
|
|
||||||
void release();
|
void release();
|
||||||
@@ -89,6 +90,8 @@ private:
|
|||||||
bool hasNetOwner;
|
bool hasNetOwner;
|
||||||
std::vector<std::string> requestedOutputs;
|
std::vector<std::string> requestedOutputs;
|
||||||
std::unordered_set<std::shared_ptr<ngraph::Node>> unconnectedNodes;
|
std::unordered_set<std::shared_ptr<ngraph::Node>> unconnectedNodes;
|
||||||
|
|
||||||
|
std::map<std::string, InferenceEngine::TensorDesc> outputsDesc;
|
||||||
};
|
};
|
||||||
|
|
||||||
class InfEngineNgraphNode : public BackendNode
|
class InfEngineNgraphNode : public BackendNode
|
||||||
@@ -121,12 +124,17 @@ public:
|
|||||||
virtual void copyToHost() CV_OVERRIDE;
|
virtual void copyToHost() CV_OVERRIDE;
|
||||||
virtual void setHostDirty() CV_OVERRIDE;
|
virtual void setHostDirty() CV_OVERRIDE;
|
||||||
|
|
||||||
|
Mat* host;
|
||||||
InferenceEngine::DataPtr dataPtr;
|
InferenceEngine::DataPtr dataPtr;
|
||||||
InferenceEngine::Blob::Ptr blob;
|
InferenceEngine::Blob::Ptr blob;
|
||||||
AsyncArray futureMat;
|
AsyncArray futureMat;
|
||||||
};
|
};
|
||||||
|
|
||||||
InferenceEngine::DataPtr ngraphDataNode(const Ptr<BackendWrapper>& ptr);
|
InferenceEngine::DataPtr ngraphDataNode(const Ptr<BackendWrapper>& ptr);
|
||||||
|
InferenceEngine::DataPtr ngraphDataOutputNode(
|
||||||
|
const Ptr<BackendWrapper>& ptr,
|
||||||
|
const InferenceEngine::TensorDesc& description,
|
||||||
|
const std::string name);
|
||||||
|
|
||||||
// This is a fake class to run networks from Model Optimizer. Objects of that
|
// This is a fake class to run networks from Model Optimizer. Objects of that
|
||||||
// class simulate responses of layers are imported by OpenCV and supported by
|
// class simulate responses of layers are imported by OpenCV and supported by
|
||||||
|
|||||||
@@ -231,7 +231,7 @@ public:
|
|||||||
kernel.set(4, ocl::KernelArg::PtrReadOnly(umat_weight));
|
kernel.set(4, ocl::KernelArg::PtrReadOnly(umat_weight));
|
||||||
kernel.set(5, ocl::KernelArg::PtrReadOnly(umat_bias));
|
kernel.set(5, ocl::KernelArg::PtrReadOnly(umat_bias));
|
||||||
kernel.set(6, ocl::KernelArg::PtrWriteOnly(dst));
|
kernel.set(6, ocl::KernelArg::PtrWriteOnly(dst));
|
||||||
bool ret = kernel.run(2, global, NULL, false);
|
bool ret = kernel.run_(2, global, NULL, false);
|
||||||
if (!ret)
|
if (!ret)
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -46,6 +46,7 @@
|
|||||||
#include "../op_inf_engine.hpp"
|
#include "../op_inf_engine.hpp"
|
||||||
#include "../ie_ngraph.hpp"
|
#include "../ie_ngraph.hpp"
|
||||||
|
|
||||||
|
#include <opencv2/core/utils/configuration.private.hpp>
|
||||||
#include <opencv2/core/utils/logger.hpp>
|
#include <opencv2/core/utils/logger.hpp>
|
||||||
|
|
||||||
#include "opencv2/core/hal/hal.hpp"
|
#include "opencv2/core/hal/hal.hpp"
|
||||||
@@ -1494,7 +1495,26 @@ public:
|
|||||||
config.pad = pad;
|
config.pad = pad;
|
||||||
config.stride = stride;
|
config.stride = stride;
|
||||||
config.dilation = dilation;
|
config.dilation = dilation;
|
||||||
|
if (inputs[0].dims != 4 && inputs[0].dims != umat_blobs[0].dims)
|
||||||
|
{
|
||||||
|
static bool bypassCheck = utils::getConfigurationParameterBool("OPENCV_OCL4DNN_CONVOLUTION_IGNORE_INPUT_DIMS_4_CHECK", false);
|
||||||
|
if (!bypassCheck)
|
||||||
|
{
|
||||||
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: Unsupported configuration: inputs[0].dims=" << inputs[0].dims << " umat_blobs[0].dims=" << umat_blobs[0].dims
|
||||||
|
<< ". Consider reporting complete reproducer to https://github.com/opencv/opencv/issues/20833."
|
||||||
|
<< " You can skip this check temporary through OPENCV_OCL4DNN_CONVOLUTION_IGNORE_INPUT_DIMS_4_CHECK=1"
|
||||||
|
);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
config.group = inputs[0].size[1] / umat_blobs[0].size[1];
|
config.group = inputs[0].size[1] / umat_blobs[0].size[1];
|
||||||
|
if (config.group < 1) // config.group == 0 causes div by zero in ocl4dnn code
|
||||||
|
{
|
||||||
|
CV_LOG_WARNING(NULL, "DNN/OpenCL: Unsupported config.group=" << config.group
|
||||||
|
<< ". Consider reporting complete reproducer to https://github.com/opencv/opencv/issues/20833"
|
||||||
|
);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
config.bias_term = umat_blobs.size() == 2;
|
config.bias_term = umat_blobs.size() == 2;
|
||||||
config.use_half = use_half;
|
config.use_half = use_half;
|
||||||
|
|
||||||
|
|||||||
@@ -456,7 +456,7 @@ public:
|
|||||||
// Retrieve all prior bboxes
|
// Retrieve all prior bboxes
|
||||||
std::vector<util::NormalizedBBox> priorBBoxes;
|
std::vector<util::NormalizedBBox> priorBBoxes;
|
||||||
std::vector<std::vector<float> > priorVariances;
|
std::vector<std::vector<float> > priorVariances;
|
||||||
GetPriorBBoxes(priorData, numPriors, _bboxesNormalized, priorBBoxes, priorVariances);
|
GetPriorBBoxes(priorData, numPriors, _bboxesNormalized, _varianceEncodedInTarget, priorBBoxes, priorVariances);
|
||||||
|
|
||||||
// Decode all loc predictions to bboxes
|
// Decode all loc predictions to bboxes
|
||||||
util::NormalizedBBox clipBounds;
|
util::NormalizedBBox clipBounds;
|
||||||
@@ -750,7 +750,7 @@ public:
|
|||||||
CV_Assert(prior_bboxes.size() == prior_variances.size());
|
CV_Assert(prior_bboxes.size() == prior_variances.size());
|
||||||
CV_Assert(prior_bboxes.size() == bboxes.size());
|
CV_Assert(prior_bboxes.size() == bboxes.size());
|
||||||
size_t num_bboxes = prior_bboxes.size();
|
size_t num_bboxes = prior_bboxes.size();
|
||||||
CV_Assert(num_bboxes == 0 || prior_variances[0].size() == 4);
|
CV_Assert(num_bboxes == 0 || prior_variances[0].size() == 4 || variance_encoded_in_target);
|
||||||
decode_bboxes.clear(); decode_bboxes.resize(num_bboxes);
|
decode_bboxes.clear(); decode_bboxes.resize(num_bboxes);
|
||||||
if(variance_encoded_in_target)
|
if(variance_encoded_in_target)
|
||||||
{
|
{
|
||||||
@@ -802,12 +802,13 @@ public:
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Get prior bounding boxes from prior_data
|
// Get prior bounding boxes from prior_data
|
||||||
// prior_data: 1 x 2 x num_priors * 4 x 1 blob.
|
// prior_data: 1 x 1 x num_priors * 4 x 1 blob or 1 x 2 x num_priors * 4 x 1 blob.
|
||||||
// num_priors: number of priors.
|
// num_priors: number of priors.
|
||||||
// prior_bboxes: stores all the prior bboxes in the format of util::NormalizedBBox.
|
// prior_bboxes: stores all the prior bboxes in the format of util::NormalizedBBox.
|
||||||
// prior_variances: stores all the variances needed by prior bboxes.
|
// prior_variances: stores all the variances needed by prior bboxes.
|
||||||
static void GetPriorBBoxes(const float* priorData, const int& numPriors,
|
static void GetPriorBBoxes(const float* priorData, const int& numPriors,
|
||||||
bool normalized_bbox, std::vector<util::NormalizedBBox>& priorBBoxes,
|
bool normalized_bbox, bool variance_encoded_in_target,
|
||||||
|
std::vector<util::NormalizedBBox>& priorBBoxes,
|
||||||
std::vector<std::vector<float> >& priorVariances)
|
std::vector<std::vector<float> >& priorVariances)
|
||||||
{
|
{
|
||||||
priorBBoxes.clear(); priorBBoxes.resize(numPriors);
|
priorBBoxes.clear(); priorBBoxes.resize(numPriors);
|
||||||
@@ -823,13 +824,16 @@ public:
|
|||||||
bbox.set_size(BBoxSize(bbox, normalized_bbox));
|
bbox.set_size(BBoxSize(bbox, normalized_bbox));
|
||||||
}
|
}
|
||||||
|
|
||||||
for (int i = 0; i < numPriors; ++i)
|
if (!variance_encoded_in_target)
|
||||||
{
|
{
|
||||||
int startIdx = (numPriors + i) * 4;
|
for (int i = 0; i < numPriors; ++i)
|
||||||
// not needed here: priorVariances[i].clear();
|
|
||||||
for (int j = 0; j < 4; ++j)
|
|
||||||
{
|
{
|
||||||
priorVariances[i].push_back(priorData[startIdx + j]);
|
int startIdx = (numPriors + i) * 4;
|
||||||
|
// not needed here: priorVariances[i].clear();
|
||||||
|
for (int j = 0; j < 4; ++j)
|
||||||
|
{
|
||||||
|
priorVariances[i].push_back(priorData[startIdx + j]);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1542,7 +1542,7 @@ Ptr<Layer> ChannelsPReLULayer::create(const LayerParams& params)
|
|||||||
if (params.blobs[0].total() == 1)
|
if (params.blobs[0].total() == 1)
|
||||||
{
|
{
|
||||||
LayerParams reluParams = params;
|
LayerParams reluParams = params;
|
||||||
reluParams.set("negative_slope", params.blobs[0].at<float>(0));
|
reluParams.set("negative_slope", *params.blobs[0].ptr<float>());
|
||||||
return ReLULayer::create(reluParams);
|
return ReLULayer::create(reluParams);
|
||||||
}
|
}
|
||||||
Ptr<ChannelsPReLULayer> l(new ElementWiseLayer<ChannelsPReLUFunctor>(ChannelsPReLUFunctor(params.blobs[0])));
|
Ptr<ChannelsPReLULayer> l(new ElementWiseLayer<ChannelsPReLUFunctor>(ChannelsPReLUFunctor(params.blobs[0])));
|
||||||
|
|||||||
@@ -191,7 +191,7 @@ public:
|
|||||||
k1.set(argId++, ocl::KernelArg::PtrReadOnly(bnorm_weight));
|
k1.set(argId++, ocl::KernelArg::PtrReadOnly(bnorm_weight));
|
||||||
k1.set(argId++, ocl::KernelArg::PtrReadOnly(bnorm_bias));
|
k1.set(argId++, ocl::KernelArg::PtrReadOnly(bnorm_bias));
|
||||||
k1.set(argId++, ocl::KernelArg::PtrWriteOnly(outMat));
|
k1.set(argId++, ocl::KernelArg::PtrWriteOnly(outMat));
|
||||||
ret = k1.run(1, globalsize, localsize, false);
|
ret = k1.run_(1, globalsize, localsize, false);
|
||||||
if (!ret)
|
if (!ret)
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -80,12 +80,31 @@ static void sigmoid(const Mat &src, Mat &dst)
|
|||||||
cv::pow(1 + dst, -1, dst);
|
cv::pow(1 + dst, -1, dst);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
typedef void (*ActivationFunction)(const Mat &src, Mat &dst);
|
||||||
|
static ActivationFunction get_activation_function(const String& activation) {
|
||||||
|
// most used activations for PyTorch and TF : Tanh, Sigmoid
|
||||||
|
// if you need to support more optional activations use std::map instead
|
||||||
|
if (activation == "Tanh")
|
||||||
|
{
|
||||||
|
return tanh;
|
||||||
|
}
|
||||||
|
else if (activation == "Sigmoid")
|
||||||
|
{
|
||||||
|
return sigmoid;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
CV_Error(Error::StsNotImplemented,
|
||||||
|
cv::format("Activation function [%s] for layer LSTM is not supported", activation.c_str()));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
class LSTMLayerImpl CV_FINAL : public LSTMLayer
|
class LSTMLayerImpl CV_FINAL : public LSTMLayer
|
||||||
{
|
{
|
||||||
int numTimeStamps, numSamples;
|
int numTimeStamps, numSamples;
|
||||||
bool allocated;
|
bool allocated;
|
||||||
|
|
||||||
MatShape outTailShape; //shape of single output sample
|
MatShape outTailShape; //shape of single output sample
|
||||||
MatShape outTsShape; //shape of N output samples
|
MatShape outTsShape; //shape of N output samples
|
||||||
|
|
||||||
bool useTimestampDim;
|
bool useTimestampDim;
|
||||||
@@ -95,6 +114,10 @@ class LSTMLayerImpl CV_FINAL : public LSTMLayer
|
|||||||
bool reverse; // If true, go in negative direction along the time axis
|
bool reverse; // If true, go in negative direction along the time axis
|
||||||
bool bidirectional; // If true, produces both forward and reversed directions along time axis
|
bool bidirectional; // If true, produces both forward and reversed directions along time axis
|
||||||
|
|
||||||
|
ActivationFunction f_activation;
|
||||||
|
ActivationFunction g_activation;
|
||||||
|
ActivationFunction h_activation;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
LSTMLayerImpl(const LayerParams& params)
|
LSTMLayerImpl(const LayerParams& params)
|
||||||
@@ -112,19 +135,24 @@ public:
|
|||||||
const Mat& Wh = blobs[0];
|
const Mat& Wh = blobs[0];
|
||||||
const Mat& Wx = blobs[1];
|
const Mat& Wx = blobs[1];
|
||||||
const Mat& bias = blobs[2];
|
const Mat& bias = blobs[2];
|
||||||
|
const Mat& hInternal = blobs[3];
|
||||||
|
const Mat& cInternal = blobs[4];
|
||||||
CV_CheckEQ(Wh.dims, 2, "");
|
CV_CheckEQ(Wh.dims, 2, "");
|
||||||
CV_CheckEQ(Wx.dims, 2, "");
|
CV_CheckEQ(Wx.dims, 2, "");
|
||||||
CV_CheckEQ(Wh.rows, Wx.rows, "");
|
CV_CheckEQ(Wh.rows, Wx.rows, "");
|
||||||
CV_CheckEQ(Wh.rows, (1 + static_cast<int>(bidirectional))*4*Wh.cols, "");
|
CV_CheckEQ(Wh.rows, (1 + static_cast<int>(bidirectional))*4*Wh.cols, "");
|
||||||
CV_CheckEQ(Wh.rows, (int)bias.total(), "");
|
CV_CheckEQ(Wh.rows, (int)bias.total(), "");
|
||||||
|
CV_CheckEQ(hInternal.cols, Wh.cols, "");
|
||||||
|
CV_CheckEQ(hInternal.cols, cInternal.cols, "");
|
||||||
|
CV_CheckEQ(hInternal.rows, cInternal.rows, "");
|
||||||
CV_Assert(Wh.type() == Wx.type() && Wx.type() == bias.type());
|
CV_Assert(Wh.type() == Wx.type() && Wx.type() == bias.type());
|
||||||
|
|
||||||
// Peephole weights.
|
// Peephole weights.
|
||||||
if (blobs.size() > 3)
|
if (blobs.size() > 5)
|
||||||
{
|
{
|
||||||
CV_Assert(blobs.size() == 6);
|
CV_Assert(blobs.size() == 8);
|
||||||
const int N = Wh.cols;
|
const int N = Wh.cols;
|
||||||
for (int i = 3; i < 6; ++i)
|
for (int i = 5; i < 8; ++i)
|
||||||
{
|
{
|
||||||
CV_Assert(blobs[i].rows == N && blobs[i].cols == N);
|
CV_Assert(blobs[i].rows == N && blobs[i].cols == N);
|
||||||
CV_Assert(blobs[i].type() == bias.type());
|
CV_Assert(blobs[i].type() == bias.type());
|
||||||
@@ -140,6 +168,20 @@ public:
|
|||||||
reverse = params.get<bool>("reverse", false);
|
reverse = params.get<bool>("reverse", false);
|
||||||
CV_Assert(!reverse || !bidirectional);
|
CV_Assert(!reverse || !bidirectional);
|
||||||
|
|
||||||
|
// read activations
|
||||||
|
DictValue activations = params.get<DictValue>("activations", "");
|
||||||
|
if (activations.size() == 1) // if activations wasn't specified use default
|
||||||
|
{
|
||||||
|
f_activation = sigmoid;
|
||||||
|
g_activation = tanh;
|
||||||
|
h_activation = tanh;
|
||||||
|
} else {
|
||||||
|
CV_Assert(activations.size() == 3);
|
||||||
|
f_activation = get_activation_function(activations.getStringValue(0));
|
||||||
|
g_activation = get_activation_function(activations.getStringValue(1));
|
||||||
|
h_activation = get_activation_function(activations.getStringValue(2));
|
||||||
|
}
|
||||||
|
|
||||||
allocated = false;
|
allocated = false;
|
||||||
outTailShape.clear();
|
outTailShape.clear();
|
||||||
}
|
}
|
||||||
@@ -181,7 +223,7 @@ public:
|
|||||||
std::vector<MatShape> &outputs,
|
std::vector<MatShape> &outputs,
|
||||||
std::vector<MatShape> &internals) const CV_OVERRIDE
|
std::vector<MatShape> &internals) const CV_OVERRIDE
|
||||||
{
|
{
|
||||||
CV_Assert((!usePeephole && blobs.size() == 3) || (usePeephole && blobs.size() == 6));
|
CV_Assert((!usePeephole && blobs.size() == 5) || (usePeephole && blobs.size() == 8));
|
||||||
CV_Assert(inputs.size() == 1);
|
CV_Assert(inputs.size() == 1);
|
||||||
const MatShape& inp0 = inputs[0];
|
const MatShape& inp0 = inputs[0];
|
||||||
|
|
||||||
@@ -228,7 +270,7 @@ public:
|
|||||||
std::vector<Mat> input;
|
std::vector<Mat> input;
|
||||||
inputs_arr.getMatVector(input);
|
inputs_arr.getMatVector(input);
|
||||||
|
|
||||||
CV_Assert((!usePeephole && blobs.size() == 3) || (usePeephole && blobs.size() == 6));
|
CV_Assert((!usePeephole && blobs.size() == 5) || (usePeephole && blobs.size() == 8));
|
||||||
CV_Assert(input.size() == 1);
|
CV_Assert(input.size() == 1);
|
||||||
const Mat& inp0 = input[0];
|
const Mat& inp0 = input[0];
|
||||||
|
|
||||||
@@ -284,13 +326,14 @@ public:
|
|||||||
const Mat &Wh = blobs[0].rowRange(i * blobs[0].rows / numDirs, (i + 1) * blobs[0].rows / numDirs);
|
const Mat &Wh = blobs[0].rowRange(i * blobs[0].rows / numDirs, (i + 1) * blobs[0].rows / numDirs);
|
||||||
const Mat &Wx = blobs[1].rowRange(i * blobs[1].rows / numDirs, (i + 1) * blobs[1].rows / numDirs);
|
const Mat &Wx = blobs[1].rowRange(i * blobs[1].rows / numDirs, (i + 1) * blobs[1].rows / numDirs);
|
||||||
const Mat &bias = blobs[2].colRange(i * blobs[2].cols / numDirs, (i + 1) * blobs[2].cols / numDirs);
|
const Mat &bias = blobs[2].colRange(i * blobs[2].cols / numDirs, (i + 1) * blobs[2].cols / numDirs);
|
||||||
|
const Mat &h_0 = blobs[3].rowRange(i * blobs[3].rows / numDirs, (i + 1) * blobs[3].rows / numDirs);
|
||||||
|
const Mat &c_0 = blobs[4].rowRange(i * blobs[4].rows / numDirs, (i + 1) * blobs[4].rows / numDirs);
|
||||||
|
|
||||||
int numOut = Wh.size[1];
|
int numOut = Wh.size[1];
|
||||||
|
|
||||||
Mat hInternal = internals[0], cInternal = internals[1],
|
Mat hInternal = internals[0], cInternal = internals[1],
|
||||||
dummyOnes = internals[2], gates = internals[3];
|
dummyOnes = internals[2], gates = internals[3];
|
||||||
hInternal.setTo(0.);
|
h_0.copyTo(hInternal);
|
||||||
cInternal.setTo(0.);
|
c_0.copyTo(cInternal);
|
||||||
dummyOnes.setTo(1.);
|
dummyOnes.setTo(1.);
|
||||||
|
|
||||||
int numSamplesTotal = numTimeStamps*numSamples;
|
int numSamplesTotal = numTimeStamps*numSamples;
|
||||||
@@ -331,17 +374,17 @@ public:
|
|||||||
if (usePeephole)
|
if (usePeephole)
|
||||||
{
|
{
|
||||||
Mat gatesIF = gates.colRange(0, 2*numOut);
|
Mat gatesIF = gates.colRange(0, 2*numOut);
|
||||||
gemm(cInternal, blobs[3], 1, gateI, 1, gateI);
|
gemm(cInternal, blobs[5], 1, gateI, 1, gateI);
|
||||||
gemm(cInternal, blobs[4], 1, gateF, 1, gateF);
|
gemm(cInternal, blobs[6], 1, gateF, 1, gateF);
|
||||||
sigmoid(gatesIF, gatesIF);
|
f_activation(gatesIF, gatesIF);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
Mat gatesIFO = gates.colRange(0, 3*numOut);
|
Mat gatesIFO = gates.colRange(0, 3*numOut);
|
||||||
sigmoid(gatesIFO, gatesIFO);
|
f_activation(gatesIFO, gatesIFO);
|
||||||
}
|
}
|
||||||
|
|
||||||
tanh(gateG, gateG);
|
g_activation(gateG, gateG);
|
||||||
|
|
||||||
//compute c_t
|
//compute c_t
|
||||||
multiply(gateF, cInternal, gateF); // f_t (*) c_{t-1}
|
multiply(gateF, cInternal, gateF); // f_t (*) c_{t-1}
|
||||||
@@ -355,12 +398,12 @@ public:
|
|||||||
}
|
}
|
||||||
if (usePeephole)
|
if (usePeephole)
|
||||||
{
|
{
|
||||||
gemm(cInternal, blobs[5], 1, gateO, 1, gateO);
|
gemm(cInternal, blobs[7], 1, gateO, 1, gateO);
|
||||||
sigmoid(gateO, gateO);
|
f_activation(gateO, gateO);
|
||||||
}
|
}
|
||||||
|
|
||||||
//compute h_t
|
//compute h_t
|
||||||
tanh(cInternal, hInternal);
|
h_activation(cInternal, hInternal);
|
||||||
multiply(gateO, hInternal, hInternal);
|
multiply(gateO, hInternal, hInternal);
|
||||||
|
|
||||||
//save results in output blobs
|
//save results in output blobs
|
||||||
|
|||||||
@@ -64,6 +64,7 @@ class RegionLayerImpl CV_FINAL : public RegionLayer
|
|||||||
public:
|
public:
|
||||||
int coords, classes, anchors, classfix;
|
int coords, classes, anchors, classfix;
|
||||||
float thresh, nmsThreshold, scale_x_y;
|
float thresh, nmsThreshold, scale_x_y;
|
||||||
|
int new_coords;
|
||||||
bool useSoftmax, useLogistic;
|
bool useSoftmax, useLogistic;
|
||||||
#ifdef HAVE_OPENCL
|
#ifdef HAVE_OPENCL
|
||||||
UMat blob_umat;
|
UMat blob_umat;
|
||||||
@@ -83,6 +84,7 @@ public:
|
|||||||
useLogistic = params.get<bool>("logistic", false);
|
useLogistic = params.get<bool>("logistic", false);
|
||||||
nmsThreshold = params.get<float>("nms_threshold", 0.4);
|
nmsThreshold = params.get<float>("nms_threshold", 0.4);
|
||||||
scale_x_y = params.get<float>("scale_x_y", 1.0); // Yolov4
|
scale_x_y = params.get<float>("scale_x_y", 1.0); // Yolov4
|
||||||
|
new_coords = params.get<int>("new_coords", 0); // Yolov4x-mish
|
||||||
|
|
||||||
CV_Assert(nmsThreshold >= 0.);
|
CV_Assert(nmsThreshold >= 0.);
|
||||||
CV_Assert(coords == 4);
|
CV_Assert(coords == 4);
|
||||||
@@ -113,7 +115,7 @@ public:
|
|||||||
{
|
{
|
||||||
#ifdef HAVE_DNN_NGRAPH
|
#ifdef HAVE_DNN_NGRAPH
|
||||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
if (backendId == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH)
|
||||||
return INF_ENGINE_VER_MAJOR_GE(INF_ENGINE_RELEASE_2020_2) && preferableTarget != DNN_TARGET_MYRIAD;
|
return INF_ENGINE_VER_MAJOR_GE(INF_ENGINE_RELEASE_2020_2) && preferableTarget != DNN_TARGET_MYRIAD && new_coords == 0;
|
||||||
#endif
|
#endif
|
||||||
return backendId == DNN_BACKEND_OPENCV;
|
return backendId == DNN_BACKEND_OPENCV;
|
||||||
}
|
}
|
||||||
@@ -259,26 +261,28 @@ public:
|
|||||||
const float *srcData = inpBlob.ptr<float>();
|
const float *srcData = inpBlob.ptr<float>();
|
||||||
float *dstData = outBlob.ptr<float>();
|
float *dstData = outBlob.ptr<float>();
|
||||||
|
|
||||||
// logistic activation for t0, for each grid cell (X x Y x Anchor-index)
|
if (new_coords == 0) {
|
||||||
for (int i = 0; i < batch_size*rows*cols*anchors; ++i) {
|
// logistic activation for t0, for each grid cell (X x Y x Anchor-index)
|
||||||
int index = cell_size*i;
|
|
||||||
float x = srcData[index + 4];
|
|
||||||
dstData[index + 4] = logistic_activate(x); // logistic activation
|
|
||||||
}
|
|
||||||
|
|
||||||
if (useSoftmax) { // Yolo v2
|
|
||||||
for (int i = 0; i < batch_size*rows*cols*anchors; ++i) {
|
for (int i = 0; i < batch_size*rows*cols*anchors; ++i) {
|
||||||
int index = cell_size*i;
|
int index = cell_size*i;
|
||||||
softmax_activate(srcData + index + 5, classes, 1, dstData + index + 5);
|
float x = srcData[index + 4];
|
||||||
|
dstData[index + 4] = logistic_activate(x); // logistic activation
|
||||||
}
|
}
|
||||||
}
|
|
||||||
else if (useLogistic) { // Yolo v3
|
if (useSoftmax) { // Yolo v2
|
||||||
for (int i = 0; i < batch_size*rows*cols*anchors; ++i){
|
for (int i = 0; i < batch_size*rows*cols*anchors; ++i) {
|
||||||
int index = cell_size*i;
|
int index = cell_size*i;
|
||||||
const float* input = srcData + index + 5;
|
softmax_activate(srcData + index + 5, classes, 1, dstData + index + 5);
|
||||||
float* output = dstData + index + 5;
|
}
|
||||||
for (int c = 0; c < classes; ++c)
|
}
|
||||||
output[c] = logistic_activate(input[c]);
|
else if (useLogistic) { // Yolo v3
|
||||||
|
for (int i = 0; i < batch_size*rows*cols*anchors; ++i){
|
||||||
|
int index = cell_size*i;
|
||||||
|
const float* input = srcData + index + 5;
|
||||||
|
float* output = dstData + index + 5;
|
||||||
|
for (int c = 0; c < classes; ++c)
|
||||||
|
output[c] = logistic_activate(input[c]);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
for (int b = 0; b < batch_size; ++b)
|
for (int b = 0; b < batch_size; ++b)
|
||||||
@@ -290,20 +294,46 @@ public:
|
|||||||
int index = (y*cols + x)*anchors + a; // index for each grid-cell & anchor
|
int index = (y*cols + x)*anchors + a; // index for each grid-cell & anchor
|
||||||
int p_index = index_sample_offset + index * cell_size + 4;
|
int p_index = index_sample_offset + index * cell_size + 4;
|
||||||
float scale = dstData[p_index];
|
float scale = dstData[p_index];
|
||||||
if (classfix == -1 && scale < .5) scale = 0; // if(t0 < 0.5) t0 = 0;
|
if (classfix == -1 && scale < .5)
|
||||||
|
{
|
||||||
|
scale = 0; // if(t0 < 0.5) t0 = 0;
|
||||||
|
}
|
||||||
int box_index = index_sample_offset + index * cell_size;
|
int box_index = index_sample_offset + index * cell_size;
|
||||||
|
|
||||||
float x_tmp = (logistic_activate(srcData[box_index + 0]) - 0.5f) * scale_x_y + 0.5f;
|
if (new_coords == 1) {
|
||||||
float y_tmp = (logistic_activate(srcData[box_index + 1]) - 0.5f) * scale_x_y + 0.5f;
|
float x_tmp = (srcData[box_index + 0] - 0.5f) * scale_x_y + 0.5f;
|
||||||
dstData[box_index + 0] = (x + x_tmp) / cols;
|
float y_tmp = (srcData[box_index + 1] - 0.5f) * scale_x_y + 0.5f;
|
||||||
dstData[box_index + 1] = (y + y_tmp) / rows;
|
dstData[box_index + 0] = (x + x_tmp) / cols;
|
||||||
dstData[box_index + 2] = exp(srcData[box_index + 2]) * biasData[2 * a] / wNorm;
|
dstData[box_index + 1] = (y + y_tmp) / rows;
|
||||||
dstData[box_index + 3] = exp(srcData[box_index + 3]) * biasData[2 * a + 1] / hNorm;
|
dstData[box_index + 2] = (srcData[box_index + 2]) * (srcData[box_index + 2]) * 4 * biasData[2 * a] / wNorm;
|
||||||
|
dstData[box_index + 3] = (srcData[box_index + 3]) * (srcData[box_index + 3]) * 4 * biasData[2 * a + 1] / hNorm;
|
||||||
|
|
||||||
int class_index = index_sample_offset + index * cell_size + 5;
|
scale = srcData[p_index];
|
||||||
for (int j = 0; j < classes; ++j) {
|
if (classfix == -1 && scale < thresh)
|
||||||
float prob = scale*dstData[class_index + j]; // prob = IoU(box, object) = t0 * class-probability
|
{
|
||||||
dstData[class_index + j] = (prob > thresh) ? prob : 0; // if (IoU < threshold) IoU = 0;
|
scale = 0; // if(t0 < 0.5) t0 = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
int class_index = index_sample_offset + index * cell_size + 5;
|
||||||
|
for (int j = 0; j < classes; ++j) {
|
||||||
|
float prob = scale*srcData[class_index + j]; // prob = IoU(box, object) = t0 * class-probability
|
||||||
|
dstData[class_index + j] = (prob > thresh) ? prob : 0; // if (IoU < threshold) IoU = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
float x_tmp = (logistic_activate(srcData[box_index + 0]) - 0.5f) * scale_x_y + 0.5f;
|
||||||
|
float y_tmp = (logistic_activate(srcData[box_index + 1]) - 0.5f) * scale_x_y + 0.5f;
|
||||||
|
dstData[box_index + 0] = (x + x_tmp) / cols;
|
||||||
|
dstData[box_index + 1] = (y + y_tmp) / rows;
|
||||||
|
dstData[box_index + 2] = exp(srcData[box_index + 2]) * biasData[2 * a] / wNorm;
|
||||||
|
dstData[box_index + 3] = exp(srcData[box_index + 3]) * biasData[2 * a + 1] / hNorm;
|
||||||
|
|
||||||
|
int class_index = index_sample_offset + index * cell_size + 5;
|
||||||
|
for (int j = 0; j < classes; ++j) {
|
||||||
|
float prob = scale*dstData[class_index + j]; // prob = IoU(box, object) = t0 * class-probability
|
||||||
|
dstData[class_index + j] = (prob > thresh) ? prob : 0; // if (IoU < threshold) IoU = 0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (nmsThreshold > 0) {
|
if (nmsThreshold > 0) {
|
||||||
|
|||||||
@@ -111,7 +111,14 @@ public:
|
|||||||
internals_arr.getMatVector(internals);
|
internals_arr.getMatVector(internals);
|
||||||
|
|
||||||
if (outHeight == inputs[0].size[2] && outWidth == inputs[0].size[3])
|
if (outHeight == inputs[0].size[2] && outWidth == inputs[0].size[3])
|
||||||
|
{
|
||||||
|
// outputs[0] = inputs[0] doesn't work due to BlobManager optimizations
|
||||||
|
if (inputs[0].data != outputs[0].data)
|
||||||
|
{
|
||||||
|
inputs[0].copyTo(outputs[0]);
|
||||||
|
}
|
||||||
return;
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
Mat& inp = inputs[0];
|
Mat& inp = inputs[0];
|
||||||
Mat& out = outputs[0];
|
Mat& out = outputs[0];
|
||||||
|
|||||||
@@ -58,6 +58,31 @@ namespace cv
|
|||||||
namespace dnn
|
namespace dnn
|
||||||
{
|
{
|
||||||
|
|
||||||
|
void sliceRangesFromShape(const MatShape& inpShape, int& axis, std::vector<std::vector<cv::Range> >& sliceRanges)
|
||||||
|
{
|
||||||
|
CV_Assert(inpShape.size() > 0);
|
||||||
|
bool axisNeg = (axis < 0);
|
||||||
|
axis = (axis + static_cast<int>(inpShape.size())) % inpShape.size();
|
||||||
|
int n = inpShape[axis];
|
||||||
|
|
||||||
|
for (size_t i = 0; i < sliceRanges.size(); ++i){
|
||||||
|
std::vector<Range>& ranges = sliceRanges[i];
|
||||||
|
if (axisNeg)
|
||||||
|
{
|
||||||
|
ranges.insert(ranges.begin(), axis, Range::all());
|
||||||
|
}
|
||||||
|
Range& range = ranges.back();
|
||||||
|
|
||||||
|
if (range.start >= 0)
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
CV_Assert(n != 0);
|
||||||
|
range.start = (n + range.start) % n;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
class SliceLayerImpl : public SliceLayer
|
class SliceLayerImpl : public SliceLayer
|
||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
@@ -69,20 +94,22 @@ public:
|
|||||||
num_split = params.get<int>("num_split", 0);
|
num_split = params.get<int>("num_split", 0);
|
||||||
hasDynamicShapes = params.get<bool>("has_dynamic_shapes", false);
|
hasDynamicShapes = params.get<bool>("has_dynamic_shapes", false);
|
||||||
shapesInitialized = !hasDynamicShapes;
|
shapesInitialized = !hasDynamicShapes;
|
||||||
|
|
||||||
if (params.has("slice_point"))
|
if (params.has("slice_point"))
|
||||||
{
|
{
|
||||||
CV_Assert(!params.has("begin") && !params.has("size") && !params.has("end"));
|
CV_Assert(!params.has("begin") && !params.has("size") && !params.has("end"));
|
||||||
const DictValue &indicesValue = params.get("slice_point");
|
const DictValue &indicesValue = params.get("slice_point");
|
||||||
|
int size = axis > 0 ? axis + 1 : 1;
|
||||||
sliceRanges.resize(indicesValue.size() + 1,
|
sliceRanges.resize(indicesValue.size() + 1,
|
||||||
std::vector<Range>(axis + 1, Range::all()));
|
std::vector<Range>(size, Range::all()));
|
||||||
int prevSlice = 0;
|
int prevSlice = 0;
|
||||||
for (int i = 0; i < indicesValue.size(); ++i)
|
for (int i = 0; i < indicesValue.size(); ++i)
|
||||||
{
|
{
|
||||||
sliceRanges[i][axis].start = prevSlice;
|
sliceRanges[i][size - 1].start = prevSlice;
|
||||||
sliceRanges[i][axis].end = indicesValue.get<int>(i);
|
sliceRanges[i][size - 1].end = indicesValue.get<int>(i);
|
||||||
prevSlice = sliceRanges[i][axis].end;
|
prevSlice = sliceRanges[i][size - 1].end;
|
||||||
}
|
}
|
||||||
sliceRanges.back()[axis].start = prevSlice;
|
sliceRanges.back()[size - 1].start = prevSlice;
|
||||||
}
|
}
|
||||||
else if (params.has("begin"))
|
else if (params.has("begin"))
|
||||||
{
|
{
|
||||||
@@ -97,7 +124,6 @@ public:
|
|||||||
{
|
{
|
||||||
int start = begins.get<int>(i);
|
int start = begins.get<int>(i);
|
||||||
int sizeOrEnd = sizesOrEnds.get<int>(i); // It may be negative to reverse indexation.
|
int sizeOrEnd = sizesOrEnds.get<int>(i); // It may be negative to reverse indexation.
|
||||||
CV_Assert(start >= 0);
|
|
||||||
|
|
||||||
sliceRanges[0][i].start = start;
|
sliceRanges[0][i].start = start;
|
||||||
if (params.has("size"))
|
if (params.has("size"))
|
||||||
@@ -154,16 +180,20 @@ public:
|
|||||||
CV_Assert(inputs.size() == 1);
|
CV_Assert(inputs.size() == 1);
|
||||||
MatShape inpShape = inputs[0];
|
MatShape inpShape = inputs[0];
|
||||||
|
|
||||||
if (!sliceRanges.empty())
|
int axis_rw = axis;
|
||||||
|
std::vector<std::vector<cv::Range> > sliceRanges_rw = sliceRanges;
|
||||||
|
sliceRangesFromShape(inpShape, axis_rw, sliceRanges_rw);
|
||||||
|
|
||||||
|
if (!sliceRanges_rw.empty())
|
||||||
{
|
{
|
||||||
outputs.resize(sliceRanges.size(), inpShape);
|
outputs.resize(sliceRanges_rw.size(), inpShape);
|
||||||
for (int i = 0; i < outputs.size(); ++i)
|
for (int i = 0; i < outputs.size(); ++i)
|
||||||
{
|
{
|
||||||
CV_Assert(sliceRanges[i].size() <= inpShape.size());
|
CV_Assert(sliceRanges_rw[i].size() <= inpShape.size());
|
||||||
for (int j = 0; j < sliceRanges[i].size(); ++j)
|
for (int j = 0; j < sliceRanges_rw[i].size(); ++j)
|
||||||
{
|
{
|
||||||
if (shapesInitialized || inpShape[j] > 0)
|
if (shapesInitialized || inpShape[j] > 0)
|
||||||
outputs[i][j] = normalize_axis_range(sliceRanges[i][j], inpShape[j]).size();
|
outputs[i][j] = normalize_axis_range(sliceRanges_rw[i][j], inpShape[j]).size();
|
||||||
|
|
||||||
if (!sliceSteps.empty() && (i < sliceSteps.size()) && (j < sliceSteps[i].size()) && (sliceSteps[i][j] > 1))
|
if (!sliceSteps.empty() && (i < sliceSteps.size()) && (j < sliceSteps[i].size()) && (sliceSteps[i][j] > 1))
|
||||||
outputs[i][j] = (outputs[i][j] + sliceSteps[i][j] - 1) / sliceSteps[i][j];
|
outputs[i][j] = (outputs[i][j] + sliceSteps[i][j] - 1) / sliceSteps[i][j];
|
||||||
@@ -172,10 +202,10 @@ public:
|
|||||||
}
|
}
|
||||||
else // Divide input blob on equal parts by axis.
|
else // Divide input blob on equal parts by axis.
|
||||||
{
|
{
|
||||||
CV_Assert(0 <= axis && axis < inpShape.size());
|
CV_Assert(0 <= axis_rw && axis_rw < inpShape.size());
|
||||||
int splits = num_split ? num_split : requiredOutputs;
|
int splits = num_split ? num_split : requiredOutputs;
|
||||||
CV_Assert(splits > 0 && inpShape[axis] % splits == 0);
|
CV_Assert(splits > 0 && inpShape[axis_rw] % splits == 0);
|
||||||
inpShape[axis] /= splits;
|
inpShape[axis_rw] /= splits;
|
||||||
outputs.resize(splits, inpShape);
|
outputs.resize(splits, inpShape);
|
||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
@@ -200,6 +230,7 @@ public:
|
|||||||
CV_Assert(inputs.size() == 1);
|
CV_Assert(inputs.size() == 1);
|
||||||
const MatSize& inpShape = inputs[0].size;
|
const MatSize& inpShape = inputs[0].size;
|
||||||
|
|
||||||
|
sliceRangesFromShape(shape(inputs[0]), axis, sliceRanges);
|
||||||
finalSliceRanges = sliceRanges;
|
finalSliceRanges = sliceRanges;
|
||||||
|
|
||||||
if (sliceRanges.empty())
|
if (sliceRanges.empty())
|
||||||
@@ -482,7 +513,7 @@ public:
|
|||||||
ocl::KernelArg::PtrReadOnly(input),
|
ocl::KernelArg::PtrReadOnly(input),
|
||||||
ocl::KernelArg::PtrWriteOnly(output)
|
ocl::KernelArg::PtrWriteOnly(output)
|
||||||
)
|
)
|
||||||
.run(2, (size_t*)ocl.global_size, (size_t*)ocl.local_size, false);
|
.run_(2, (size_t*)ocl.global_size, (size_t*)ocl.local_size, false);
|
||||||
if (!ret)
|
if (!ret)
|
||||||
return false;
|
return false;
|
||||||
} // for outputs.size()
|
} // for outputs.size()
|
||||||
|
|||||||
@@ -222,8 +222,6 @@ class OCL4DNNConvSpatial
|
|||||||
bool createDWConvKernel(int32_t blockWidth,
|
bool createDWConvKernel(int32_t blockWidth,
|
||||||
int32_t blockHeight,
|
int32_t blockHeight,
|
||||||
int32_t blockDepth);
|
int32_t blockDepth);
|
||||||
void CreateSubBuffer(const UMat& buffer, UMat& sub_buffer,
|
|
||||||
int32_t offset, int32_t size, bool write_only);
|
|
||||||
bool convolve(const UMat &bottom, UMat &top,
|
bool convolve(const UMat &bottom, UMat &top,
|
||||||
const UMat &weight, const UMat &bias,
|
const UMat &weight, const UMat &bias,
|
||||||
int32_t numImages,
|
int32_t numImages,
|
||||||
@@ -269,7 +267,7 @@ class OCL4DNNConvSpatial
|
|||||||
void generate_idlf_tuneritems(std::vector< cv::Ptr<tunerParam> > &tunerItems,
|
void generate_idlf_tuneritems(std::vector< cv::Ptr<tunerParam> > &tunerItems,
|
||||||
int blockM, int blockK, int simd_size);
|
int blockM, int blockK, int simd_size);
|
||||||
void setFusionDefine(ocl4dnnFusedActiv_t fused_activ, bool fused_eltwise);
|
void setFusionDefine(ocl4dnnFusedActiv_t fused_activ, bool fused_eltwise);
|
||||||
void setFusionArg(ocl4dnnFusedActiv_t fused_activ, bool fused_eltwise, ocl::Kernel &kernel, cl_uint &argIdx);
|
void setFusionArg(ocl4dnnFusedActiv_t fused_activ, bool fused_eltwise, int fused_eltwise_offset, ocl::Kernel &kernel, cl_uint &argIdx);
|
||||||
|
|
||||||
int32_t group_;
|
int32_t group_;
|
||||||
bool bias_term_;
|
bool bias_term_;
|
||||||
|
|||||||
@@ -116,6 +116,7 @@ ocl::Image2D ocl4dnnGEMMCopyBufferToImage(UMat buffer, int offset,
|
|||||||
.args(
|
.args(
|
||||||
ocl::KernelArg::PtrReadOnly(buffer),
|
ocl::KernelArg::PtrReadOnly(buffer),
|
||||||
image, offset,
|
image, offset,
|
||||||
|
padded_width, padded_height,
|
||||||
width, height,
|
width, height,
|
||||||
ld)
|
ld)
|
||||||
.run(2, global_copy, NULL, false);
|
.run(2, global_copy, NULL, false);
|
||||||
|
|||||||
@@ -167,6 +167,7 @@ OCL4DNNConvSpatial<Dtype>::OCL4DNNConvSpatial(OCL4DNNConvConfig config)
|
|||||||
channels_ = config.in_shape[dims - spatial_dims - 1];
|
channels_ = config.in_shape[dims - spatial_dims - 1];
|
||||||
num_output_ = config.out_shape[dims - spatial_dims - 1];
|
num_output_ = config.out_shape[dims - spatial_dims - 1];
|
||||||
group_ = config.group;
|
group_ = config.group;
|
||||||
|
CV_CheckGT(group_, 0, ""); // avoid div by zero below
|
||||||
|
|
||||||
fused_activ_ = OCL4DNN_CONV_FUSED_ACTIV_NONE;
|
fused_activ_ = OCL4DNN_CONV_FUSED_ACTIV_NONE;
|
||||||
fused_eltwise_ = false;
|
fused_eltwise_ = false;
|
||||||
@@ -218,14 +219,7 @@ OCL4DNNConvSpatial<Dtype>::OCL4DNNConvSpatial(OCL4DNNConvConfig config)
|
|||||||
#endif
|
#endif
|
||||||
if (!use_cache_path_)
|
if (!use_cache_path_)
|
||||||
{
|
{
|
||||||
static int warn_ = 0;
|
CV_LOG_ONCE_ERROR(NULL, "OpenCV(ocl4dnn): Kernel configuration cache directory doesn't exist: " << cache_path_);
|
||||||
if (!warn_)
|
|
||||||
{
|
|
||||||
std::cerr
|
|
||||||
<< "OpenCV(ocl4dnn): Kernel configuration cache directory doesn't exist: " << cache_path_ << std::endl
|
|
||||||
<< std::endl;
|
|
||||||
warn_ = true;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -270,17 +264,21 @@ void OCL4DNNConvSpatial<Dtype>::setFusionDefine(ocl4dnnFusedActiv_t fused_activ,
|
|||||||
}
|
}
|
||||||
|
|
||||||
template<typename Dtype>
|
template<typename Dtype>
|
||||||
void OCL4DNNConvSpatial<Dtype>::setFusionArg(ocl4dnnFusedActiv_t fused_activ, bool fused_eltwise, ocl::Kernel &kernel, cl_uint &argIdx)
|
void OCL4DNNConvSpatial<Dtype>::setFusionArg(ocl4dnnFusedActiv_t fused_activ, bool fused_eltwise, int fused_eltwise_offset, ocl::Kernel &kernel, cl_uint &argIdx)
|
||||||
{
|
{
|
||||||
if (fused_eltwise)
|
if (fused_eltwise)
|
||||||
kernel.set(argIdx++, (cl_mem)bottom_data2_.handle(ACCESS_READ));
|
{
|
||||||
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom_data2_));
|
||||||
|
if (fused_eltwise_offset >= 0)
|
||||||
|
kernel.set(argIdx++, fused_eltwise_offset);
|
||||||
|
}
|
||||||
|
|
||||||
switch (fused_activ) {
|
switch (fused_activ) {
|
||||||
case OCL4DNN_CONV_FUSED_ACTIV_RELU:
|
case OCL4DNN_CONV_FUSED_ACTIV_RELU:
|
||||||
kernel.set(argIdx++, (float)negative_slope_);
|
kernel.set(argIdx++, (float)negative_slope_);
|
||||||
break;
|
break;
|
||||||
case OCL4DNN_CONV_FUSED_ACTIV_PRELU:
|
case OCL4DNN_CONV_FUSED_ACTIV_PRELU:
|
||||||
kernel.set(argIdx++, (cl_mem)negative_slope_umat_.handle(ACCESS_READ));
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(negative_slope_umat_));
|
||||||
break;
|
break;
|
||||||
case OCL4DNN_CONV_FUSED_ACTIV_POWER:
|
case OCL4DNN_CONV_FUSED_ACTIV_POWER:
|
||||||
kernel.set(argIdx++, (float)power_);
|
kernel.set(argIdx++, (float)power_);
|
||||||
@@ -414,7 +412,6 @@ void OCL4DNNConvSpatial<Dtype>::setupKernelDetails(int32_t kernelType,
|
|||||||
addDef("CHANNELS", channels_ / group_);
|
addDef("CHANNELS", channels_ / group_);
|
||||||
addDef("APPLY_BIAS", bias_term_);
|
addDef("APPLY_BIAS", bias_term_);
|
||||||
addDef("OUTPUT_Z", M_);
|
addDef("OUTPUT_Z", M_);
|
||||||
addDef("ZPAR", 1);
|
|
||||||
setFusionDefine(fused_activ_, fused_eltwise_);
|
setFusionDefine(fused_activ_, fused_eltwise_);
|
||||||
|
|
||||||
src_ = cv::ocl::dnn::conv_layer_spatial_oclsrc;
|
src_ = cv::ocl::dnn::conv_layer_spatial_oclsrc;
|
||||||
@@ -668,8 +665,7 @@ void interleaveMatrix(Dtype* mem_dst, const Dtype *mem,
|
|||||||
int r, int c, int interleavedRows, int nonInterleavedRows,
|
int r, int c, int interleavedRows, int nonInterleavedRows,
|
||||||
int blockWidth, int rowAlignment )
|
int blockWidth, int rowAlignment )
|
||||||
{
|
{
|
||||||
CHECK_EQ(interleavedRows % 2, 0) <<
|
CV_Check(interleavedRows, interleavedRows % 2 == 0, "interleaveMatrix only supports even values for interleavedRows.");
|
||||||
"interleaveMatrix only supports even values for interleavedRows.";
|
|
||||||
|
|
||||||
size_t memSize = r * c * sizeof(float);
|
size_t memSize = r * c * sizeof(float);
|
||||||
size_t dstSize = memSize *
|
size_t dstSize = memSize *
|
||||||
@@ -681,9 +677,12 @@ void interleaveMatrix(Dtype* mem_dst, const Dtype *mem,
|
|||||||
const int yStride = c * 2;
|
const int yStride = c * 2;
|
||||||
const Dtype *pSrc = mem;
|
const Dtype *pSrc = mem;
|
||||||
Dtype* pDst = mem_dst;
|
Dtype* pDst = mem_dst;
|
||||||
for (int y = 0; y < r;) {
|
for (int y = 0; y < r;)
|
||||||
for (int rows = 0; rows < interleavedRows; rows += 2) {
|
{
|
||||||
if ( y >= r ) break;
|
for (int rows = 0; rows < interleavedRows; rows += 2)
|
||||||
|
{
|
||||||
|
if (y >= r)
|
||||||
|
break;
|
||||||
if ((c % xStride) == 0) {
|
if ((c % xStride) == 0) {
|
||||||
for (int x = 0; x < c / xStride; x++) {
|
for (int x = 0; x < c / xStride; x++) {
|
||||||
memcpy(pDst + x * xStride * 2, // NOLINT
|
memcpy(pDst + x * xStride * 2, // NOLINT
|
||||||
@@ -708,11 +707,14 @@ void interleaveMatrix(Dtype* mem_dst, const Dtype *mem,
|
|||||||
y += 2;
|
y += 2;
|
||||||
}
|
}
|
||||||
|
|
||||||
for (int rows = 0; rows < nonInterleavedRows; rows++) {
|
for (int rows = 0; rows < nonInterleavedRows; rows++)
|
||||||
if (y >= r) break;
|
{
|
||||||
|
if (y >= r)
|
||||||
|
break;
|
||||||
const int stride = rowAlignment;
|
const int stride = rowAlignment;
|
||||||
int remaining = c;
|
int remaining = c;
|
||||||
for (int x = 0; x < c; x += stride) {
|
for (int x = 0; x < c; x += stride)
|
||||||
|
{
|
||||||
if (remaining >= stride) {
|
if (remaining >= stride) {
|
||||||
memcpy(pDst + x * 2, pSrc + x, stride * sizeof(Dtype)); // NOLINT
|
memcpy(pDst + x * 2, pSrc + x, stride * sizeof(Dtype)); // NOLINT
|
||||||
remaining -=stride;
|
remaining -=stride;
|
||||||
@@ -765,12 +767,11 @@ bool OCL4DNNConvSpatial<Dtype>::swizzleWeight(const UMat &weight,
|
|||||||
swizzled_factor
|
swizzled_factor
|
||||||
);
|
);
|
||||||
|
|
||||||
size_t global_work_size_copy[3] = {
|
size_t global_work_size_copy[1] = { (size_t)(alignSize(num_output_, swizzled_factor) * channels * kernel_w_ * kernel_h_) };
|
||||||
(size_t) (alignSize(num_output_, swizzled_factor) * channels * kernel_w_ * kernel_h_), 1, 1 };
|
|
||||||
|
|
||||||
if (!oclk_copy_weight.run(3, global_work_size_copy, NULL, false))
|
if (!oclk_copy_weight.run_(1, global_work_size_copy, NULL, false))
|
||||||
{
|
{
|
||||||
std::cout << "Swizzle kernel run failed." << std::endl;
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: Swizzle kernel run failed");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
@@ -849,34 +850,6 @@ bool OCL4DNNConvSpatial<float>::createBasicKernel(int32_t blockWidth,
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
template<>
|
|
||||||
void OCL4DNNConvSpatial<float>::CreateSubBuffer(const UMat& buffer, UMat& sub_buffer,
|
|
||||||
int32_t offset, int32_t size, bool write_only)
|
|
||||||
{
|
|
||||||
cl_mem sub_mem;
|
|
||||||
cl_buffer_region region;
|
|
||||||
cl_int err;
|
|
||||||
size_t element_size = (use_half_) ? sizeof(short) : sizeof(float);
|
|
||||||
|
|
||||||
region.origin = offset * element_size + buffer.offset;
|
|
||||||
region.size = size * element_size;
|
|
||||||
sub_mem = clCreateSubBuffer((cl_mem)buffer.handle(ACCESS_READ),
|
|
||||||
write_only ? CL_MEM_WRITE_ONLY : CL_MEM_READ_ONLY,
|
|
||||||
CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err);
|
|
||||||
if (err)
|
|
||||||
{
|
|
||||||
std::cout << "Failed to create sub buffer." << std::endl;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
int step = element_size, rows = size, cols = 1;
|
|
||||||
ocl::convertFromBuffer(sub_mem, step, rows, cols,
|
|
||||||
(use_half_) ? CV_16SC1 : CV_32FC1, sub_buffer);
|
|
||||||
|
|
||||||
//decrease ocl mem refcount
|
|
||||||
clReleaseMemObject(sub_mem);
|
|
||||||
}
|
|
||||||
|
|
||||||
template<>
|
template<>
|
||||||
bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
||||||
const UMat &weight, const UMat &bias,
|
const UMat &weight, const UMat &bias,
|
||||||
@@ -895,10 +868,12 @@ bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
|||||||
if (config->kernelType == KERNEL_TYPE_INTEL_IDLF) {
|
if (config->kernelType == KERNEL_TYPE_INTEL_IDLF) {
|
||||||
if (!swizzleWeight(weight, config->workItem_output[2], false))
|
if (!swizzleWeight(weight, config->workItem_output[2], false))
|
||||||
return false;
|
return false;
|
||||||
|
#if 0
|
||||||
size_t total_bottom_size = bottom_dim_ * numImages;
|
size_t total_bottom_size = bottom_dim_ * numImages;
|
||||||
size_t total_kernel_size = kernel_h_ * kernel_w_ * channels_ * M_;
|
size_t total_kernel_size = kernel_h_ * kernel_w_ * channels_ * M_;
|
||||||
size_t total_bias_size = M_ * group_;
|
size_t total_bias_size = M_ * group_;
|
||||||
size_t total_top_size = top_dim_ * numImages;
|
size_t total_top_size = top_dim_ * numImages;
|
||||||
|
#endif
|
||||||
for (int32_t g = 0; g < group_; ++g) {
|
for (int32_t g = 0; g < group_; ++g) {
|
||||||
bias_offset = M_ * g;
|
bias_offset = M_ * g;
|
||||||
int32_t image_offset = width_ * height_ * (channels_ / group_) * g;
|
int32_t image_offset = width_ * height_ * (channels_ / group_) * g;
|
||||||
@@ -910,89 +885,41 @@ bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
|||||||
return false;
|
return false;
|
||||||
|
|
||||||
cl_uint argIdx = 0;
|
cl_uint argIdx = 0;
|
||||||
setFusionArg(fused_activ_, fused_eltwise_, kernel, argIdx);
|
setFusionArg(fused_activ_, fused_eltwise_, output_image_offset, kernel, argIdx);
|
||||||
|
|
||||||
UMat img_buffer;
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
||||||
if (image_offset)
|
kernel.set(argIdx++, image_offset);
|
||||||
{
|
|
||||||
CreateSubBuffer(bottom, img_buffer, image_offset,
|
|
||||||
total_bottom_size - image_offset, false);
|
|
||||||
if (img_buffer.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(img_buffer));
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(swizzled_weights_umat));
|
||||||
}
|
kernel.set(argIdx++, kernel_offset);
|
||||||
else
|
|
||||||
{
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
|
||||||
}
|
|
||||||
|
|
||||||
UMat kernel_buffer;
|
|
||||||
if (kernel_offset)
|
|
||||||
{
|
|
||||||
CreateSubBuffer(swizzled_weights_umat, kernel_buffer, kernel_offset,
|
|
||||||
total_kernel_size - kernel_offset, false);
|
|
||||||
if (kernel_buffer.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(kernel_buffer));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(swizzled_weights_umat));
|
|
||||||
}
|
|
||||||
|
|
||||||
UMat bias_buffer;
|
|
||||||
if (bias_term_)
|
if (bias_term_)
|
||||||
{
|
{
|
||||||
if (bias_offset)
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bias));
|
||||||
{
|
kernel.set(argIdx++, bias_offset);
|
||||||
CreateSubBuffer(bias, bias_buffer, bias_offset,
|
|
||||||
total_bias_size - bias_offset, false);
|
|
||||||
if (bias_buffer.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bias_buffer));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bias));
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
UMat out_buffer;
|
kernel.set(argIdx++, ocl::KernelArg::PtrWriteOnly(top));
|
||||||
if (output_image_offset)
|
kernel.set(argIdx++, (int)(top.offset / element_size) + output_image_offset);
|
||||||
{
|
|
||||||
CreateSubBuffer(top, out_buffer, output_image_offset,
|
|
||||||
total_top_size - output_image_offset, true);
|
|
||||||
if (out_buffer.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrWriteOnly(out_buffer));
|
|
||||||
kernel.set(argIdx++, (int)(out_buffer.offset / element_size));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrWriteOnly(top));
|
|
||||||
kernel.set(argIdx++, (int)(top.offset / element_size));
|
|
||||||
}
|
|
||||||
|
|
||||||
kernel.set(argIdx++, (uint16_t)width_);
|
kernel.set(argIdx++, (uint16_t)width_);
|
||||||
kernel.set(argIdx++, (uint16_t)height_);
|
kernel.set(argIdx++, (uint16_t)height_);
|
||||||
kernel.set(argIdx++, (uint16_t)output_w_);
|
kernel.set(argIdx++, (uint16_t)output_w_);
|
||||||
kernel.set(argIdx++, (uint16_t)output_h_);
|
kernel.set(argIdx++, (uint16_t)output_h_);
|
||||||
if (!kernel.run(3, config->global_work_size, config->local_work_size, false))
|
if (!kernel.run_(3, config->global_work_size, config->local_work_size, false))
|
||||||
{
|
{
|
||||||
std::cout << "IDLF kernel run failed." << std::endl;
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: IDLF kernel run failed");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else if (config->kernelType == KERNEL_TYPE_GEMM_LIKE) {
|
} else if (config->kernelType == KERNEL_TYPE_GEMM_LIKE) {
|
||||||
if (!swizzleWeight(weight, config->workItem_output[1], true))
|
if (!swizzleWeight(weight, config->workItem_output[1], true))
|
||||||
return false;
|
return false;
|
||||||
|
#if 0
|
||||||
size_t total_bottom_size = bottom_dim_ * numImages;
|
size_t total_bottom_size = bottom_dim_ * numImages;
|
||||||
size_t total_kernel_size = kernel_h_ * kernel_w_ * channels_ * M_;
|
size_t total_kernel_size = kernel_h_ * kernel_w_ * channels_ * M_;
|
||||||
size_t total_bias_size = M_ * group_;
|
size_t total_bias_size = M_ * group_;
|
||||||
|
#endif
|
||||||
size_t total_top_size = top_dim_ * numImages;
|
size_t total_top_size = top_dim_ * numImages;
|
||||||
for (int32_t g = 0; g < group_; ++g) {
|
for (int32_t g = 0; g < group_; ++g) {
|
||||||
bias_offset = M_ * g;
|
bias_offset = M_ * g;
|
||||||
@@ -1005,72 +932,25 @@ bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
|||||||
return false;
|
return false;
|
||||||
|
|
||||||
cl_uint argIdx = 0;
|
cl_uint argIdx = 0;
|
||||||
setFusionArg(fused_activ_, fused_eltwise_, kernel, argIdx);
|
setFusionArg(fused_activ_, fused_eltwise_, output_image_offset, kernel, argIdx);
|
||||||
|
|
||||||
UMat img_buffer;
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
||||||
if (image_offset)
|
kernel.set(argIdx++, (int)image_offset);
|
||||||
{
|
kernel.set(argIdx++, (int)(bottom.total() - image_offset));
|
||||||
CreateSubBuffer(bottom, img_buffer, image_offset,
|
|
||||||
total_bottom_size - image_offset, false);
|
|
||||||
if (img_buffer.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(img_buffer));
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(swizzled_weights_umat));
|
||||||
}
|
kernel.set(argIdx++, (int)kernel_offset);
|
||||||
else
|
kernel.set(argIdx++, (int)(swizzled_weights_umat.total() - kernel_offset));
|
||||||
{
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
|
||||||
}
|
|
||||||
|
|
||||||
UMat kernel_buffer;
|
|
||||||
if (kernel_offset)
|
|
||||||
{
|
|
||||||
CreateSubBuffer(swizzled_weights_umat, kernel_buffer, kernel_offset,
|
|
||||||
total_kernel_size - kernel_offset, false);
|
|
||||||
if (kernel_buffer.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(kernel_buffer));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(swizzled_weights_umat));
|
|
||||||
}
|
|
||||||
|
|
||||||
UMat bias_buffer;
|
|
||||||
if (bias_term_)
|
if (bias_term_)
|
||||||
{
|
{
|
||||||
if (bias_offset)
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bias));
|
||||||
{
|
kernel.set(argIdx++, (int)bias_offset);
|
||||||
CreateSubBuffer(bias, bias_buffer, bias_offset,
|
|
||||||
total_bias_size - bias_offset, false);
|
|
||||||
if (bias_buffer.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bias_buffer));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bias));
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
UMat out_buffer;
|
kernel.set(argIdx++, ocl::KernelArg::PtrWriteOnly(top));
|
||||||
if (output_image_offset)
|
kernel.set(argIdx++, (int)(top.offset / element_size) + output_image_offset);
|
||||||
{
|
kernel.set(argIdx++, (int)total_top_size - (int)(top.offset / element_size));
|
||||||
CreateSubBuffer(top, out_buffer, output_image_offset,
|
|
||||||
total_top_size - output_image_offset, true);
|
|
||||||
if (out_buffer.empty())
|
|
||||||
return false;
|
|
||||||
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrWriteOnly(out_buffer));
|
|
||||||
kernel.set(argIdx++, (int)(out_buffer.offset / element_size));
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrWriteOnly(top));
|
|
||||||
kernel.set(argIdx++, (int)(top.offset / element_size));
|
|
||||||
}
|
|
||||||
|
|
||||||
kernel.set(argIdx++, (uint16_t)width_);
|
kernel.set(argIdx++, (uint16_t)width_);
|
||||||
kernel.set(argIdx++, (uint16_t)height_);
|
kernel.set(argIdx++, (uint16_t)height_);
|
||||||
@@ -1100,9 +980,9 @@ bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
|||||||
gy = alignSize(gy, blockK);
|
gy = alignSize(gy, blockK);
|
||||||
size_t global_size[3] = { gx, gy, config->global_work_size[2] };
|
size_t global_size[3] = { gx, gy, config->global_work_size[2] };
|
||||||
|
|
||||||
if (!kernel.run(3, global_size, config->local_work_size, false))
|
if (!kernel.run_(3, global_size, config->local_work_size, false))
|
||||||
{
|
{
|
||||||
std::cout << "GEMM like kernel run failed." << std::endl;
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: GEMM like kernel run failed");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1112,7 +992,7 @@ bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
|||||||
return false;
|
return false;
|
||||||
|
|
||||||
cl_uint argIdx = 0;
|
cl_uint argIdx = 0;
|
||||||
setFusionArg(fused_activ_, fused_eltwise_, kernel, argIdx);
|
setFusionArg(fused_activ_, fused_eltwise_, -1, kernel, argIdx);
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(weight));
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(weight));
|
||||||
if (bias_term_)
|
if (bias_term_)
|
||||||
@@ -1124,14 +1004,17 @@ bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
|||||||
kernel.set(argIdx++, (uint16_t)output_w_);
|
kernel.set(argIdx++, (uint16_t)output_w_);
|
||||||
kernel.set(argIdx++, (uint16_t)output_h_);
|
kernel.set(argIdx++, (uint16_t)output_h_);
|
||||||
|
|
||||||
size_t global_size[3];
|
size_t wgs = kernel.workGroupSize();
|
||||||
global_size[0] = output_w_;
|
if (!wgs)
|
||||||
global_size[1] = output_h_;
|
|
||||||
global_size[2] = num_output_ * num_;
|
|
||||||
|
|
||||||
if (!kernel.run(3, global_size, NULL, false))
|
|
||||||
{
|
{
|
||||||
std::cout << "DWCONV kernel run failed." << std::endl;
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: Can't query workGroupSize of DWCONV kernel");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
size_t lws[1] = { wgs };
|
||||||
|
size_t gws[1] = { roundUp((size_t)output_w_ * output_h_ * num_output_ * num_, (unsigned)lws[0]) };
|
||||||
|
if (!kernel.run_(1, gws, lws, false))
|
||||||
|
{
|
||||||
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: DWCONV kernel run failed");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
@@ -1152,7 +1035,7 @@ bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
|||||||
return false;
|
return false;
|
||||||
|
|
||||||
cl_uint argIdx = 0;
|
cl_uint argIdx = 0;
|
||||||
setFusionArg(fused_activ_, fused_eltwise_, kernel, argIdx);
|
setFusionArg(fused_activ_, fused_eltwise_, -1, kernel, argIdx);
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
||||||
kernel.set(argIdx++, image_offset);
|
kernel.set(argIdx++, image_offset);
|
||||||
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(weight));
|
kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(weight));
|
||||||
@@ -1171,11 +1054,18 @@ bool OCL4DNNConvSpatial<float>::convolve(const UMat &bottom, UMat &top,
|
|||||||
kernel.set(argIdx++, (uint16_t)output_h_);
|
kernel.set(argIdx++, (uint16_t)output_h_);
|
||||||
kernel.set(argIdx++, (uint16_t)pad_w_);
|
kernel.set(argIdx++, (uint16_t)pad_w_);
|
||||||
kernel.set(argIdx++, (uint16_t)pad_h_);
|
kernel.set(argIdx++, (uint16_t)pad_h_);
|
||||||
if (!kernel.run(3, config->global_work_size,
|
|
||||||
(config->use_null_local) ? NULL : config->local_work_size,
|
size_t wgs = kernel.workGroupSize();
|
||||||
false))
|
if (!wgs)
|
||||||
{
|
{
|
||||||
std::cout << "Basic kernel run failed." << std::endl;
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: Can't query workGroupSize of Basic kernel");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
size_t lws[1] = { wgs };
|
||||||
|
size_t gws[1] = { roundUp((size_t)output_w_ * output_h_ * M_, (unsigned)lws[0]) };
|
||||||
|
if (!kernel.run_(1, gws, lws, false))
|
||||||
|
{
|
||||||
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: Basic kernel run failed");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1195,14 +1085,9 @@ float OCL4DNNConvSpatial<float>::timedConvolve(const UMat &bottom, UMat &top,
|
|||||||
{
|
{
|
||||||
queue = cv::ocl::Queue::getDefault();
|
queue = cv::ocl::Queue::getDefault();
|
||||||
}
|
}
|
||||||
catch (const cv::Exception&)
|
catch (const std::exception& e)
|
||||||
{
|
{
|
||||||
static int warn_ = 0;
|
CV_LOG_ONCE_ERROR(NULL, "OpenCV(ocl4dnn): Can't get OpenCL default queue for auto-tuning: " << e.what());
|
||||||
if (!warn_)
|
|
||||||
{
|
|
||||||
std::cout << "OpenCV(ocl4dnn): Can't get OpenCL default queue for auto-tuning." << std::endl;
|
|
||||||
warn_ = true;
|
|
||||||
}
|
|
||||||
return 1e6;
|
return 1e6;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1257,8 +1142,11 @@ bool OCL4DNNConvSpatial<float>::verifyResult(const UMat &bottom,
|
|||||||
else if (config->tested)
|
else if (config->tested)
|
||||||
return false;
|
return false;
|
||||||
|
|
||||||
int32_t sz[4] = {numImages, num_output_, output_h_, output_w_};
|
//int32_t sz[4] = {numImages, num_output_, output_h_, output_w_};
|
||||||
top.zeros(4, sz, (use_half_) ? CV_16SC1 : CV_32FC1);
|
CV_CheckEQ(top.total(), (size_t)numImages * num_output_ * output_h_ * output_w_, "");
|
||||||
|
CV_CheckTypeEQ(top.type(), (use_half_) ? CV_16SC1 : CV_32FC1, "");
|
||||||
|
top.setTo(Scalar::all(0));
|
||||||
|
|
||||||
bool saved_tuned = tuned_;
|
bool saved_tuned = tuned_;
|
||||||
tuned_ = false;
|
tuned_ = false;
|
||||||
convolve(bottom, top, weight, bias, numImages, config);
|
convolve(bottom, top, weight, bias, numImages, config);
|
||||||
@@ -1403,9 +1291,9 @@ ocl::Program OCL4DNNConvSpatial<Dtype>::compileKernel()
|
|||||||
phash.insert(std::pair<std::string, ocl::Program>(kernel_name_, program));
|
phash.insert(std::pair<std::string, ocl::Program>(kernel_name_, program));
|
||||||
if (!program.ptr())
|
if (!program.ptr())
|
||||||
{
|
{
|
||||||
std::cout << "Failed to compile kernel: " << kernel_name_
|
CV_LOG_WARNING(NULL, "DNN/OpenCL: Failed to compile kernel: " << kernel_name_
|
||||||
<< ", buildflags: " << options
|
<< ", buildflags: '" << options << "', errmsg: '" << errmsg << "'"
|
||||||
<< ", errmsg: " << errmsg << std::endl;
|
);
|
||||||
}
|
}
|
||||||
return program;
|
return program;
|
||||||
}
|
}
|
||||||
@@ -1434,26 +1322,13 @@ bool OCL4DNNConvSpatial<float>::createGEMMLikeConvKernel(int32_t blockM,
|
|||||||
ocl::Program program = compileKernel();
|
ocl::Program program = compileKernel();
|
||||||
if (program.ptr())
|
if (program.ptr())
|
||||||
{
|
{
|
||||||
size_t workgroupSize_used;
|
|
||||||
ocl::Kernel kernel(kernel_name_.c_str(), program);
|
ocl::Kernel kernel(kernel_name_.c_str(), program);
|
||||||
if (kernel.empty())
|
if (kernel.empty())
|
||||||
return false;
|
return false;
|
||||||
|
|
||||||
workgroupSize_used = kernel.preferedWorkGroupSizeMultiple();
|
kernelQueue.push_back(makePtr<kernelConfig>(kernel_name_, &global_size[0], &local_size[0], &workItemOutput[0],
|
||||||
if (workgroupSize_used != simd_size)
|
true, KERNEL_TYPE_GEMM_LIKE));
|
||||||
{
|
return true;
|
||||||
std::cerr << "OpenCV(ocl4dnn): The OpenCL compiler chose a simd size (" << workgroupSize_used << ") that " << std::endl;
|
|
||||||
std::cerr << " does not equal the size (" << simd_size << ") kernel source required." << std::endl;
|
|
||||||
std::cerr << " Skip this kernel " << kernel_name_ << std::endl;
|
|
||||||
unloadProgram(kernel_name_);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
kernelQueue.push_back(makePtr<kernelConfig>(kernel_name_, &global_size[0], &local_size[0], &workItemOutput[0],
|
|
||||||
true, KERNEL_TYPE_GEMM_LIKE));
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
return false;
|
return false;
|
||||||
@@ -1499,26 +1374,13 @@ bool OCL4DNNConvSpatial<float>::createIDLFKernel(int32_t blockWidth,
|
|||||||
ocl::Program program = compileKernel();
|
ocl::Program program = compileKernel();
|
||||||
if (program.ptr())
|
if (program.ptr())
|
||||||
{
|
{
|
||||||
size_t workgroupSize_used;
|
|
||||||
ocl::Kernel kernel(kernel_name_.c_str(), program);
|
ocl::Kernel kernel(kernel_name_.c_str(), program);
|
||||||
if (kernel.empty())
|
if (kernel.empty())
|
||||||
return false;
|
return false;
|
||||||
|
|
||||||
workgroupSize_used = kernel.preferedWorkGroupSizeMultiple();
|
kernelQueue.push_back(makePtr<kernelConfig>(kernel_name_, &global_size[0], &local_size[0], &workItemOutput[0],
|
||||||
if (workgroupSize_used != simd_size)
|
true, KERNEL_TYPE_INTEL_IDLF));
|
||||||
{
|
return true;
|
||||||
std::cerr << "OpenCV(ocl4dnn): The OpenCL compiler chose a simd size (" << workgroupSize_used << ") that " << std::endl;
|
|
||||||
std::cerr << " does not equal the size (" << simd_size << ") kernel source required." << std::endl;
|
|
||||||
std::cerr << " Skip this kernel " << kernel_name_ << std::endl;
|
|
||||||
unloadProgram(kernel_name_);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
kernelQueue.push_back(makePtr<kernelConfig>(kernel_name_, &global_size[0], &local_size[0], &workItemOutput[0],
|
|
||||||
true, KERNEL_TYPE_INTEL_IDLF));
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
return false;
|
return false;
|
||||||
@@ -1857,7 +1719,8 @@ void OCL4DNNConvSpatial<float>::setupConvolution(const UMat &bottom,
|
|||||||
fastestTime = kernelQueue[x]->executionTime;
|
fastestTime = kernelQueue[x]->executionTime;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (fastestKernel < 0) break;
|
if (fastestKernel < 0)
|
||||||
|
break;
|
||||||
// Test fastest kernel
|
// Test fastest kernel
|
||||||
bool verified = verifyResult(bottom, top, weight, bias, numImages, kernelQueue[fastestKernel], verifyTop);
|
bool verified = verifyResult(bottom, top, weight, bias, numImages, kernelQueue[fastestKernel], verifyTop);
|
||||||
if (verified == true) {
|
if (verified == true) {
|
||||||
@@ -2016,17 +1879,18 @@ bool OCL4DNNConvSpatial<Dtype>::setupKernelByConfig(int x, int y, int z, int typ
|
|||||||
{
|
{
|
||||||
if (z == 1)
|
if (z == 1)
|
||||||
z = 16;
|
z = 16;
|
||||||
CHECK_EQ(z == 16 || z == 8, true) << "invalid SIMD size" << std::endl;
|
CV_Check(z, z == 16 || z == 8, "DNN/OpenCL: IDLF - invalid SIMD size");
|
||||||
}
|
}
|
||||||
kernelQueue.clear();
|
kernelQueue.clear();
|
||||||
createConvolutionKernel(type, x, y, z);
|
createConvolutionKernel(type, x, y, z);
|
||||||
if (kernelQueue.size() != 1) {
|
if (kernelQueue.size() != 1)
|
||||||
std::cerr << "Failed setup kernel by config:"
|
{
|
||||||
|
CV_LOG_ERROR(NULL, "DNN/OpenCL: Failed setup kernel by config: "
|
||||||
<< " x = " << x
|
<< " x = " << x
|
||||||
<< " y = " << y
|
<< " y = " << y
|
||||||
<< " z = " << z
|
<< " z = " << z
|
||||||
<< " type = " << type
|
<< " type = " << type
|
||||||
<< std::endl;
|
);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
bestKernelConfig = kernelQueue[0];
|
bestKernelConfig = kernelQueue[0];
|
||||||
@@ -2058,13 +1922,9 @@ bool OCL4DNNConvSpatial<Dtype>::loadTunedConfig()
|
|||||||
{
|
{
|
||||||
if (cache_path_.empty())
|
if (cache_path_.empty())
|
||||||
{
|
{
|
||||||
static int warn_ = 0;
|
CV_LOG_ONCE_WARNING(NULL, "OpenCV(ocl4dnn): consider to specify kernel configuration cache directory "
|
||||||
if (!warn_)
|
"through OPENCV_OCL4DNN_CONFIG_PATH parameter."
|
||||||
{
|
);
|
||||||
std::cout << "OpenCV(ocl4dnn): consider to specify kernel configuration cache directory " << std::endl
|
|
||||||
<< " via OPENCV_OCL4DNN_CONFIG_PATH parameter." << std::endl;
|
|
||||||
warn_ = true;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -127,7 +127,7 @@ bool OCL4DNNSoftmax<Dtype>::Forward(const UMat& bottom, UMat& top)
|
|||||||
oclk_softmax_forward_kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
oclk_softmax_forward_kernel.set(argIdx++, ocl::KernelArg::PtrReadOnly(bottom));
|
||||||
oclk_softmax_forward_kernel.set(argIdx++, ocl::KernelArg::PtrWriteOnly(top));
|
oclk_softmax_forward_kernel.set(argIdx++, ocl::KernelArg::PtrWriteOnly(top));
|
||||||
}
|
}
|
||||||
ret = oclk_softmax_forward_kernel.run(3, global_size, local_size, false);
|
ret = oclk_softmax_forward_kernel.run_(3, global_size, local_size, false);
|
||||||
}
|
}
|
||||||
return ret;
|
return ret;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -231,6 +231,27 @@ public:
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
class NormalizeSubgraph2_2 : public NormalizeSubgraphBase
|
||||||
|
{
|
||||||
|
public:
|
||||||
|
NormalizeSubgraph2_2()
|
||||||
|
{
|
||||||
|
int input = addNodeToMatch("");
|
||||||
|
int norm = addNodeToMatch("ReduceL2", input);
|
||||||
|
|
||||||
|
int min = addNodeToMatch("");
|
||||||
|
int max = addNodeToMatch("");
|
||||||
|
int clip = addNodeToMatch("Clip", norm, min, max);
|
||||||
|
|
||||||
|
int shape = addNodeToMatch("");
|
||||||
|
int expand = addNodeToMatch("Expand", clip, shape);
|
||||||
|
|
||||||
|
addNodeToMatch("Div", input, expand);
|
||||||
|
|
||||||
|
setFusedNode("Normalize", input);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
class NormalizeSubgraph3 : public NormalizeSubgraphBase
|
class NormalizeSubgraph3 : public NormalizeSubgraphBase
|
||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
@@ -555,6 +576,7 @@ void simplifySubgraphs(opencv_onnx::GraphProto& net)
|
|||||||
subgraphs.push_back(makePtr<SoftMaxSubgraph>());
|
subgraphs.push_back(makePtr<SoftMaxSubgraph>());
|
||||||
subgraphs.push_back(makePtr<NormalizeSubgraph1>());
|
subgraphs.push_back(makePtr<NormalizeSubgraph1>());
|
||||||
subgraphs.push_back(makePtr<NormalizeSubgraph2>());
|
subgraphs.push_back(makePtr<NormalizeSubgraph2>());
|
||||||
|
subgraphs.push_back(makePtr<NormalizeSubgraph2_2>());
|
||||||
subgraphs.push_back(makePtr<NormalizeSubgraph3>());
|
subgraphs.push_back(makePtr<NormalizeSubgraph3>());
|
||||||
subgraphs.push_back(makePtr<BatchNormalizationSubgraph1>());
|
subgraphs.push_back(makePtr<BatchNormalizationSubgraph1>());
|
||||||
subgraphs.push_back(makePtr<BatchNormalizationSubgraph2>());
|
subgraphs.push_back(makePtr<BatchNormalizationSubgraph2>());
|
||||||
|
|||||||
+1837
-1470
File diff suppressed because it is too large
Load Diff
@@ -74,18 +74,22 @@
|
|||||||
(_dst_)[(_offset_)] = ACTIVATION_RELU_FUNCTION(_x_, _channel_); \
|
(_dst_)[(_offset_)] = ACTIVATION_RELU_FUNCTION(_x_, _channel_); \
|
||||||
} while(0)
|
} while(0)
|
||||||
#define ELTWISE_DATA_ARG __global Dtype* eltwise_data,
|
#define ELTWISE_DATA_ARG __global Dtype* eltwise_data,
|
||||||
|
#define ELTWISE_DATA_ARG_WITH_OFFSET __global Dtype* eltwise_ptr, int eltwise_offset,
|
||||||
#else
|
#else
|
||||||
#define ACTIVATION_FUNCTION(_dst_, _offset_, _data_, _channel_) do { \
|
#define ACTIVATION_FUNCTION(_dst_, _offset_, _data_, _channel_) do { \
|
||||||
const Dtype _x_ = (_data_); \
|
const Dtype _x_ = (_data_); \
|
||||||
(_dst_)[(_offset_)] = ACTIVATION_RELU_FUNCTION(_x_, _channel_); \
|
(_dst_)[(_offset_)] = ACTIVATION_RELU_FUNCTION(_x_, _channel_); \
|
||||||
} while(0)
|
} while(0)
|
||||||
#define ELTWISE_DATA_ARG
|
#define ELTWISE_DATA_ARG
|
||||||
|
#define ELTWISE_DATA_ARG_WITH_OFFSET
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if APPLY_BIAS
|
#if APPLY_BIAS
|
||||||
#define BIAS_KERNEL_ARG __global Dtype * biases_base,
|
#define BIAS_KERNEL_ARG __global Dtype * biases_base,
|
||||||
|
#define BIAS_KERNEL_ARG_WITH_OFFSET __global Dtype * biases_base_ptr, int biases_base_offset,
|
||||||
#else
|
#else
|
||||||
#define BIAS_KERNEL_ARG
|
#define BIAS_KERNEL_ARG
|
||||||
|
#define BIAS_KERNEL_ARG_WITH_OFFSET
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define __CAT(x, y) x##y
|
#define __CAT(x, y) x##y
|
||||||
@@ -154,22 +158,18 @@ __kernel void ConvolveBasic(
|
|||||||
)
|
)
|
||||||
{
|
{
|
||||||
__global Dtype* convolved_image = convolved_image_base + convolved_image_base_offset;
|
__global Dtype* convolved_image = convolved_image_base + convolved_image_base_offset;
|
||||||
const int outputX = get_global_id(0);
|
const int out_idx = get_global_id(0); // 1D task layout: [output_width * output_height * OUTPUT_Z]
|
||||||
const int outputY = get_global_id(1);
|
const int plane_size = output_width * output_height;
|
||||||
const int kernelNum = get_global_id(2) * ZPAR;
|
const int out_plane_idx = out_idx % plane_size;
|
||||||
if (outputX < output_width && outputY < output_height)
|
const int outputZ = out_idx / plane_size; // kernelNum
|
||||||
|
const int outputY = out_plane_idx / output_width;
|
||||||
|
const int outputX = out_plane_idx % output_width;
|
||||||
|
if (outputZ < OUTPUT_Z)
|
||||||
{
|
{
|
||||||
Dtype sum[ZPAR];
|
Dtype sum = 0.0f;
|
||||||
for (int kern = 0; kern < ZPAR; kern++)
|
|
||||||
{
|
|
||||||
sum[kern] = 0.0f;
|
|
||||||
}
|
|
||||||
const int org_y = outputY * STRIDE_Y - pad_h;
|
const int org_y = outputY * STRIDE_Y - pad_h;
|
||||||
const int org_x = outputX * STRIDE_X - pad_w;
|
const int org_x = outputX * STRIDE_X - pad_w;
|
||||||
const int currentKernelOffset = kernel_offset + kernelNum*KERNEL_HEIGHT*KERNEL_WIDTH*CHANNELS;
|
const int currentKernelOffset = kernel_offset + outputZ*KERNEL_HEIGHT*KERNEL_WIDTH*CHANNELS;
|
||||||
#if APPLY_BIAS
|
|
||||||
const int biasIndex = bias_offset + kernelNum;
|
|
||||||
#endif
|
|
||||||
const int local_image_offset = org_y * input_width + org_x;
|
const int local_image_offset = org_y * input_width + org_x;
|
||||||
const int imageSize = input_width * input_height;
|
const int imageSize = input_width * input_height;
|
||||||
__global Dtype* image_dataPtr = (image_data + (image_offset + local_image_offset));
|
__global Dtype* image_dataPtr = (image_data + (image_offset + local_image_offset));
|
||||||
@@ -178,17 +178,13 @@ __kernel void ConvolveBasic(
|
|||||||
{
|
{
|
||||||
for (int y = 0; y < KERNEL_HEIGHT; y++)
|
for (int y = 0; y < KERNEL_HEIGHT; y++)
|
||||||
{
|
{
|
||||||
|
int y_ = org_y + y * DILATION_Y;
|
||||||
for (int x = 0; x < KERNEL_WIDTH; x++)
|
for (int x = 0; x < KERNEL_WIDTH; x++)
|
||||||
{
|
{
|
||||||
int y_ = org_y + y * DILATION_Y;
|
|
||||||
int x_ = org_x + x * DILATION_X;
|
int x_ = org_x + x * DILATION_X;
|
||||||
if (!(y_ >= 0 && y_ < input_height && x_ >= 0 && x_ < input_width))
|
if (y_ >= 0 && y_ < input_height && x_ >= 0 && x_ < input_width)
|
||||||
{
|
{
|
||||||
continue;
|
sum = mad(image_dataPtr[x * DILATION_X], kernel_dataPtr[x], sum);
|
||||||
}
|
|
||||||
for (int kern = 0; kern < ZPAR; kern++)
|
|
||||||
{
|
|
||||||
sum[kern] += image_dataPtr[x * DILATION_X] * kernel_dataPtr[kern*KERNEL_HEIGHT*KERNEL_WIDTH*CHANNELS + x];
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
image_dataPtr += input_width * DILATION_Y;
|
image_dataPtr += input_width * DILATION_Y;
|
||||||
@@ -197,18 +193,13 @@ __kernel void ConvolveBasic(
|
|||||||
image_dataPtr += imageSize - input_width*KERNEL_HEIGHT*DILATION_Y;
|
image_dataPtr += imageSize - input_width*KERNEL_HEIGHT*DILATION_Y;
|
||||||
}
|
}
|
||||||
|
|
||||||
for (int kern = 0; kern < ZPAR; kern++)
|
int offset = convolved_image_offset + out_idx;
|
||||||
{
|
|
||||||
if (kernelNum + kern < OUTPUT_Z)
|
|
||||||
{
|
|
||||||
int offset = convolved_image_offset + (kernelNum+kern)*output_height*output_width + outputY*output_width + outputX;
|
|
||||||
#if APPLY_BIAS
|
#if APPLY_BIAS
|
||||||
ACTIVATION_FUNCTION(convolved_image, offset, sum[kern] + bias[biasIndex + kern], biasIndex + kern);
|
int biasIndex = bias_offset + outputZ;
|
||||||
|
ACTIVATION_FUNCTION(convolved_image, offset, sum + bias[biasIndex], biasIndex);
|
||||||
#else
|
#else
|
||||||
ACTIVATION_FUNCTION(convolved_image, offset, sum[kern], kernelNum + kern);
|
ACTIVATION_FUNCTION(convolved_image, offset, sum, outputZ);
|
||||||
#endif
|
#endif
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -223,19 +214,28 @@ __attribute__((reqd_work_group_size(1, 1, SIMD_SIZE)))
|
|||||||
__attribute__((intel_reqd_sub_group_size(SIMD_SIZE)))
|
__attribute__((intel_reqd_sub_group_size(SIMD_SIZE)))
|
||||||
__kernel void
|
__kernel void
|
||||||
convolve_simd(
|
convolve_simd(
|
||||||
ELTWISE_DATA_ARG
|
ELTWISE_DATA_ARG_WITH_OFFSET
|
||||||
FUSED_ARG
|
FUSED_ARG
|
||||||
__global Dtype* inputs,
|
__global Dtype* inputs_ptr, const int inputs_offset,
|
||||||
__global Dtype* weights,
|
__global Dtype* weights_ptr, const int weights_offset,
|
||||||
BIAS_KERNEL_ARG
|
BIAS_KERNEL_ARG_WITH_OFFSET
|
||||||
__global Dtype* outputs_base,
|
__global Dtype* outputs_base, const int outputs_offset,
|
||||||
const int outputs_offset,
|
|
||||||
const ushort input_width,
|
const ushort input_width,
|
||||||
const ushort input_height,
|
const ushort input_height,
|
||||||
const ushort output_width,
|
const ushort output_width,
|
||||||
const ushort output_height)
|
const ushort output_height)
|
||||||
{
|
{
|
||||||
|
__global Dtype* inputs = inputs_ptr + inputs_offset;
|
||||||
|
__global Dtype* weights = weights_ptr + weights_offset;
|
||||||
|
#if APPLY_BIAS
|
||||||
|
__global Dtype* biases_base = biases_base_ptr + biases_base_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
__global Dtype* outputs = outputs_base + outputs_offset;
|
__global Dtype* outputs = outputs_base + outputs_offset;
|
||||||
|
#ifdef FUSED_CONV_ELTWISE
|
||||||
|
__global Dtype* eltwise_data = eltwise_ptr + eltwise_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
unsigned int oc = get_global_id(0) * OUT_BLOCK_WIDTH; // oc = Output Column
|
unsigned int oc = get_global_id(0) * OUT_BLOCK_WIDTH; // oc = Output Column
|
||||||
unsigned int or = get_global_id(1) * OUT_BLOCK_HEIGHT; // or = Output Row
|
unsigned int or = get_global_id(1) * OUT_BLOCK_HEIGHT; // or = Output Row
|
||||||
unsigned int fm = get_global_id(2); // fm = Feature Map = od = Output Depth
|
unsigned int fm = get_global_id(2); // fm = Feature Map = od = Output Depth
|
||||||
@@ -388,13 +388,12 @@ typedef struct half0 { half s0; } half0; //never used but makes compiler happy.
|
|||||||
#define ROW_PITCH input_width
|
#define ROW_PITCH input_width
|
||||||
|
|
||||||
#define GEMM_LIKE_KERNEL_ARGS \
|
#define GEMM_LIKE_KERNEL_ARGS \
|
||||||
ELTWISE_DATA_ARG \
|
ELTWISE_DATA_ARG_WITH_OFFSET \
|
||||||
FUSED_ARG \
|
FUSED_ARG \
|
||||||
const __global Dtype *src0, \
|
const __global Dtype *src0_ptr, const unsigned int src0_offset, const unsigned int src0_limit, \
|
||||||
const __global Dtype *src1, \
|
const __global Dtype *src1_ptr, const unsigned int src1_offset, const unsigned int src1_limit, \
|
||||||
BIAS_KERNEL_ARG \
|
BIAS_KERNEL_ARG_WITH_OFFSET \
|
||||||
__global Dtype *dst_base, \
|
__global Dtype *dst_base, const unsigned int dst_offset, const unsigned int dst_limit, \
|
||||||
const int dst_offset, \
|
|
||||||
const ushort input_width, \
|
const ushort input_width, \
|
||||||
const ushort input_height, \
|
const ushort input_height, \
|
||||||
const ushort output_width, \
|
const ushort output_width, \
|
||||||
@@ -424,7 +423,17 @@ typedef struct half0 { half s0; } half0; //never used but makes compiler happy.
|
|||||||
__attribute__((intel_reqd_sub_group_size(8)))
|
__attribute__((intel_reqd_sub_group_size(8)))
|
||||||
__kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
__kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
||||||
{
|
{
|
||||||
|
const __global Dtype *src0 = src0_ptr + src0_offset;
|
||||||
|
const __global Dtype *src1 = src1_ptr + src1_offset;
|
||||||
|
#if APPLY_BIAS
|
||||||
|
__global Dtype* biases_base = biases_base_ptr + biases_base_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
__global Dtype *dst = dst_base + dst_offset;
|
__global Dtype *dst = dst_base + dst_offset;
|
||||||
|
#ifdef FUSED_CONV_ELTWISE
|
||||||
|
__global Dtype* eltwise_data = eltwise_ptr + eltwise_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
const int group_x = get_group_id(0);
|
const int group_x = get_group_id(0);
|
||||||
const int group_y = get_group_id(1);
|
const int group_y = get_group_id(1);
|
||||||
const int global_x = get_global_id(0);
|
const int global_x = get_global_id(0);
|
||||||
@@ -447,6 +456,14 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
}
|
}
|
||||||
typedef CAT( Dtype, KERNEL_WIDTH ) Dtype_t;
|
typedef CAT( Dtype, KERNEL_WIDTH ) Dtype_t;
|
||||||
|
|
||||||
|
// U_GEMM_LIKE_CONV_k11x11_cn3_g1_s4x4_d1x1_b1_in240x240_p0x0_num1_M96_activ1_eltwise0_FP32_5_1_8_32_SIMD8 doesn't run properly (src0_read out of bounds)
|
||||||
|
// Test: DNNTestNetwork.AlexNet/0 (to run all kernels use OPENCV_OCL4DNN_FORCE_AUTO_TUNING=1)
|
||||||
|
#if 0 // INPUT_PAD_W == 0 && INPUT_PAD_H == 0 && DILATION_X == 1 && DILATION_Y == 1 && INPUT_PAD_BOTTOM == 0 && INPUT_PAD_RIGHT == 0
|
||||||
|
#define OPTIMIZE_READ 1
|
||||||
|
#else
|
||||||
|
#define OPTIMIZE_READ 0
|
||||||
|
#endif
|
||||||
|
|
||||||
// True for all threads if filter_width is multiple of TILE_N
|
// True for all threads if filter_width is multiple of TILE_N
|
||||||
// else, true for all but right-most column of threads.
|
// else, true for all but right-most column of threads.
|
||||||
if( TILE_N_LAST == 0 || global_x < WIDTH1 / TILE_N )
|
if( TILE_N_LAST == 0 || global_x < WIDTH1 / TILE_N )
|
||||||
@@ -463,7 +480,7 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
// atile is M rows x K columns.
|
// atile is M rows x K columns.
|
||||||
int curr_x = ( global_y % output_width ) * STRIDE_X;
|
int curr_x = ( global_y % output_width ) * STRIDE_X;
|
||||||
int curr_y = ( global_y / output_width ) * STRIDE_Y;
|
int curr_y = ( global_y / output_width ) * STRIDE_Y;
|
||||||
#if INPUT_PAD_H != 0 || INPUT_PAD_W != 0 || DILATION_X != 1 || DILATION_Y != 1 || INPUT_PAD_BOTTOM != 0 || INPUT_PAD_RIGHT != 0
|
#if !OPTIMIZE_READ
|
||||||
int saved_y = curr_y;
|
int saved_y = curr_y;
|
||||||
#endif
|
#endif
|
||||||
const __global Dtype *src0_read = src0
|
const __global Dtype *src0_read = src0
|
||||||
@@ -483,7 +500,7 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
do
|
do
|
||||||
{
|
{
|
||||||
int patch_row = 0;
|
int patch_row = 0;
|
||||||
#if INPUT_PAD_H != 0 || INPUT_PAD_W != 0 || DILATION_X != 1 || DILATION_Y != 1 || INPUT_PAD_BOTTOM != 0 || INPUT_PAD_RIGHT != 0
|
#if !OPTIMIZE_READ
|
||||||
curr_y = saved_y;
|
curr_y = saved_y;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
@@ -501,11 +518,17 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
// ...
|
// ...
|
||||||
const bool kernel_width_is_odd = KERNEL_WIDTH % 2 == 1;
|
const bool kernel_width_is_odd = KERNEL_WIDTH % 2 == 1;
|
||||||
|
|
||||||
#if INPUT_PAD_W == 0 && INPUT_PAD_H == 0 && DILATION_X == 1 && DILATION_Y == 1 && INPUT_PAD_BOTTOM == 0 && INPUT_PAD_RIGHT == 0
|
#if OPTIMIZE_READ
|
||||||
#if KERNEL_WIDTH == 3
|
#if KERNEL_WIDTH == 3
|
||||||
Dtype_t blockA00 = vload3(0, src0_read);
|
Dtype_t blockA00 = vload3(0, src0_read);
|
||||||
Dtype* pblockA00 = (Dtype*)(&blockA00);
|
Dtype* pblockA00 = (Dtype*)(&blockA00);
|
||||||
#else
|
#else
|
||||||
|
#if 0 // debug
|
||||||
|
if ((int)(src0_read - src0) >= src0_limit - KERNEL_WIDTH)
|
||||||
|
{
|
||||||
|
printf("CATCH: src0_read-src0: %d limit=%d curr_y,curr_x=%d,%d\n", (int)(src0_read - src0), src0_limit, curr_y, curr_x);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
Dtype_t blockA00 = ( (const __global Dtype_t*)src0_read )[ 0 ];
|
Dtype_t blockA00 = ( (const __global Dtype_t*)src0_read )[ 0 ];
|
||||||
Dtype* pblockA00 = (Dtype*)(&blockA00);
|
Dtype* pblockA00 = (Dtype*)(&blockA00);
|
||||||
#endif
|
#endif
|
||||||
@@ -626,7 +649,7 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
// atile is M rows x K columns.
|
// atile is M rows x K columns.
|
||||||
int curr_x = ( global_y % output_width ) * STRIDE_X;
|
int curr_x = ( global_y % output_width ) * STRIDE_X;
|
||||||
int curr_y = ( global_y / output_width ) * STRIDE_Y;
|
int curr_y = ( global_y / output_width ) * STRIDE_Y;
|
||||||
#if INPUT_PAD_H != 0 || INPUT_PAD_W != 0 || DILATION_X != 1 || DILATION_Y != 1 || INPUT_PAD_BOTTOM != 0 || INPUT_PAD_RIGHT != 0
|
#if !OPTIMIZE_READ
|
||||||
int saved_y = curr_y;
|
int saved_y = curr_y;
|
||||||
#endif
|
#endif
|
||||||
const __global Dtype *src0_read = src0
|
const __global Dtype *src0_read = src0
|
||||||
@@ -646,14 +669,14 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
do
|
do
|
||||||
{
|
{
|
||||||
int patch_row = 0;
|
int patch_row = 0;
|
||||||
#if INPUT_PAD_H != 0 || INPUT_PAD_W != 0 || DILATION_X != 1 || DILATION_Y != 1 || INPUT_PAD_BOTTOM != 0 || INPUT_PAD_RIGHT != 0
|
#if !OPTIMIZE_READ
|
||||||
curr_y = saved_y;
|
curr_y = saved_y;
|
||||||
#endif
|
#endif
|
||||||
do
|
do
|
||||||
{
|
{
|
||||||
// Load atile and interleaved btile.
|
// Load atile and interleaved btile.
|
||||||
const bool kernel_width_is_odd = KERNEL_WIDTH % 2 == 1;
|
const bool kernel_width_is_odd = KERNEL_WIDTH % 2 == 1;
|
||||||
#if INPUT_PAD_W == 0 && INPUT_PAD_H == 0 && DILATION_X == 1 && DILATION_Y == 1 && INPUT_PAD_BOTTOM == 0 && INPUT_PAD_RIGHT == 0
|
#if OPTIMIZE_READ
|
||||||
Dtype_t blockA00 = ( (const __global Dtype_t*)src0_read )[ 0 ];
|
Dtype_t blockA00 = ( (const __global Dtype_t*)src0_read )[ 0 ];
|
||||||
Dtype* pblockA00 = (Dtype*)(&blockA00);
|
Dtype* pblockA00 = (Dtype*)(&blockA00);
|
||||||
#else
|
#else
|
||||||
@@ -790,7 +813,7 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
#endif
|
#endif // TILE_N_LAST > 0
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
#ifdef GEMM_LIKE_CONV_32_2
|
#ifdef GEMM_LIKE_CONV_32_2
|
||||||
@@ -813,7 +836,17 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
__attribute__((intel_reqd_sub_group_size(8)))
|
__attribute__((intel_reqd_sub_group_size(8)))
|
||||||
__kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
__kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
||||||
{
|
{
|
||||||
|
const __global Dtype *src0 = src0_ptr + src0_offset;
|
||||||
|
const __global Dtype *src1 = src1_ptr + src1_offset;
|
||||||
|
#if APPLY_BIAS
|
||||||
|
__global Dtype* biases_base = biases_base_ptr + biases_base_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
__global Dtype *dst = dst_base + dst_offset;
|
__global Dtype *dst = dst_base + dst_offset;
|
||||||
|
#ifdef FUSED_CONV_ELTWISE
|
||||||
|
__global Dtype* eltwise_data = eltwise_ptr + eltwise_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
const int group_x = get_group_id(0);
|
const int group_x = get_group_id(0);
|
||||||
const int group_y = get_group_id(1);
|
const int group_y = get_group_id(1);
|
||||||
const int global_x = get_global_id(0);
|
const int global_x = get_global_id(0);
|
||||||
@@ -1375,7 +1408,17 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
__attribute__((intel_reqd_sub_group_size(16)))
|
__attribute__((intel_reqd_sub_group_size(16)))
|
||||||
__kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
__kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
||||||
{
|
{
|
||||||
|
const __global Dtype *src0 = src0_ptr + src0_offset;
|
||||||
|
const __global Dtype *src1 = src1_ptr + src1_offset;
|
||||||
|
#if APPLY_BIAS
|
||||||
|
__global Dtype* biases_base = biases_base_ptr + biases_base_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
__global Dtype *dst = dst_base + dst_offset;
|
__global Dtype *dst = dst_base + dst_offset;
|
||||||
|
#ifdef FUSED_CONV_ELTWISE
|
||||||
|
__global Dtype* eltwise_data = eltwise_ptr + eltwise_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
const int group_x = get_group_id(0);
|
const int group_x = get_group_id(0);
|
||||||
const int group_y = get_group_id(1);
|
const int group_y = get_group_id(1);
|
||||||
const int global_x = get_global_id(0);
|
const int global_x = get_global_id(0);
|
||||||
@@ -1561,7 +1604,17 @@ __kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
|||||||
__attribute__((intel_reqd_sub_group_size(16)))
|
__attribute__((intel_reqd_sub_group_size(16)))
|
||||||
__kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
__kernel void Conv_Interleaved(GEMM_LIKE_KERNEL_ARGS)
|
||||||
{
|
{
|
||||||
|
const __global Dtype *src0 = src0_ptr + src0_offset;
|
||||||
|
const __global Dtype *src1 = src1_ptr + src1_offset;
|
||||||
|
#if APPLY_BIAS
|
||||||
|
__global Dtype* biases_base = biases_base_ptr + biases_base_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
__global Dtype *dst = dst_base + dst_offset;
|
__global Dtype *dst = dst_base + dst_offset;
|
||||||
|
#ifdef FUSED_CONV_ELTWISE
|
||||||
|
__global Dtype* eltwise_data = eltwise_ptr + eltwise_offset;
|
||||||
|
#endif
|
||||||
|
|
||||||
const int group_x = get_group_id(0);
|
const int group_x = get_group_id(0);
|
||||||
const int group_y = get_group_id(1);
|
const int group_y = get_group_id(1);
|
||||||
const int global_x = get_global_id(0);
|
const int global_x = get_global_id(0);
|
||||||
@@ -1780,10 +1833,13 @@ __kernel void DWCONV(
|
|||||||
const ushort output_width,
|
const ushort output_width,
|
||||||
const ushort output_height) {
|
const ushort output_height) {
|
||||||
__global Dtype* convolved_image = convolved_image_base + convolved_image_offset;
|
__global Dtype* convolved_image = convolved_image_base + convolved_image_offset;
|
||||||
const int outputX = get_global_id(0);
|
const int out_idx = get_global_id(0); // 1D task layout: [output_width * output_height * OUTPUT_Z]
|
||||||
const int outputY = get_global_id(1);
|
const int plane_size = output_width * output_height;
|
||||||
const int outputZ = get_global_id(2);
|
const int out_plane_idx = out_idx % plane_size;
|
||||||
if(outputX < output_width && outputY < output_height)
|
const int outputZ = out_idx / plane_size;
|
||||||
|
const int outputY = out_plane_idx / output_width;
|
||||||
|
const int outputX = out_plane_idx % output_width;
|
||||||
|
if (outputZ < OUTPUT_Z)
|
||||||
{
|
{
|
||||||
Dtype sum = 0.;
|
Dtype sum = 0.;
|
||||||
|
|
||||||
|
|||||||
@@ -62,8 +62,8 @@ __kernel void TEMPLATE(copyWeightsSwizzled, Dtype)
|
|||||||
//Original location
|
//Original location
|
||||||
|
|
||||||
//Output location
|
//Output location
|
||||||
int outputSublayer = channels / swizzleFactor;
|
//int outputSublayer = channels / swizzleFactor;
|
||||||
int outputSublayerIndex = channels % swizzleFactor;
|
//int outputSublayerIndex = channels % swizzleFactor;
|
||||||
|
|
||||||
int filter = sX / (kernel_w*kernel_h*channels);
|
int filter = sX / (kernel_w*kernel_h*channels);
|
||||||
int kernel_X = sX % kernel_w;
|
int kernel_X = sX % kernel_w;
|
||||||
@@ -73,6 +73,10 @@ __kernel void TEMPLATE(copyWeightsSwizzled, Dtype)
|
|||||||
int FP = filter / swizzleFactor;
|
int FP = filter / swizzleFactor;
|
||||||
int F1 = filter % swizzleFactor;
|
int F1 = filter % swizzleFactor;
|
||||||
|
|
||||||
weightOut[FP*(kernel_w*kernel_h*channels*swizzleFactor) + kernel_C*(kernel_w*kernel_h*swizzleFactor) + kernel_Y*(kernel_w*swizzleFactor) + kernel_X*swizzleFactor + F1]
|
int idxOut = FP*(kernel_w*kernel_h*channels*swizzleFactor) + kernel_C*(kernel_w*kernel_h*swizzleFactor) + kernel_Y*(kernel_w*swizzleFactor) + kernel_X*swizzleFactor + F1;
|
||||||
= weightIn[filter*(kernel_w*kernel_h*channels) + kernel_C*(kernel_w*kernel_h) + kernel_Y*kernel_w + kernel_X];
|
int idxIn = filter*(kernel_w*kernel_h*channels) + kernel_C*(kernel_w*kernel_h) + kernel_Y*kernel_w + kernel_X;
|
||||||
|
|
||||||
|
// idxIn is not valid if (filter >= outputs) - no data for these elements. Output alignment gaps are filled by zeros
|
||||||
|
Dtype v = (filter < outputs) ? weightIn[idxIn] : (Dtype)0;
|
||||||
|
weightOut[idxOut] = v;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -954,6 +954,10 @@ __kernel void TEMPLATE(gemm_buffer_copy_image_transpose, Dtype)(
|
|||||||
{
|
{
|
||||||
const int gidx = get_global_id(0);
|
const int gidx = get_global_id(0);
|
||||||
const int gidy = get_global_id(1);
|
const int gidy = get_global_id(1);
|
||||||
|
|
||||||
|
if (gidx >= width || gidy >= height)
|
||||||
|
return;
|
||||||
|
|
||||||
int2 coord_dst = (int2)(gidx, gidy);
|
int2 coord_dst = (int2)(gidx, gidy);
|
||||||
__global Dtype* A_off = A + offA;
|
__global Dtype* A_off = A + offA;
|
||||||
Dtype srcA = A_off[gidy * ldA + gidx];
|
Dtype srcA = A_off[gidy * ldA + gidx];
|
||||||
@@ -968,12 +972,18 @@ __kernel void TEMPLATE(gemm_buffer_copy_image_no_transpose, Dtype)(
|
|||||||
__global Dtype* A,
|
__global Dtype* A,
|
||||||
__write_only image2d_t ImA,
|
__write_only image2d_t ImA,
|
||||||
int offA,
|
int offA,
|
||||||
|
int padded_width,
|
||||||
|
int padded_height,
|
||||||
int width,
|
int width,
|
||||||
int height,
|
int height,
|
||||||
int ldA)
|
int ldA)
|
||||||
{
|
{
|
||||||
const int gidx = get_global_id(0);
|
const int gidx = get_global_id(0);
|
||||||
const int gidy = get_global_id(1);
|
const int gidy = get_global_id(1);
|
||||||
|
|
||||||
|
if (gidx >= padded_width || gidy >= padded_height)
|
||||||
|
return;
|
||||||
|
|
||||||
int2 coord_dst = (int2)(gidx, gidy);
|
int2 coord_dst = (int2)(gidx, gidy);
|
||||||
#if TYPE == TYPE_HALF
|
#if TYPE == TYPE_HALF
|
||||||
if (gidx >= width || gidy >= height) {
|
if (gidx >= width || gidy >= height) {
|
||||||
|
|||||||
@@ -19,6 +19,16 @@ CV__DNN_EXPERIMENTAL_NS_BEGIN
|
|||||||
using ::google::protobuf::RepeatedField;
|
using ::google::protobuf::RepeatedField;
|
||||||
using ::google::protobuf::MapPair;
|
using ::google::protobuf::MapPair;
|
||||||
|
|
||||||
|
static Mat getTensorContentRef_(const tensorflow::TensorProto& tensor);
|
||||||
|
static inline
|
||||||
|
bool isAlignedMat(const Mat& m)
|
||||||
|
{
|
||||||
|
int depth = m.depth();
|
||||||
|
int alignment = CV_ELEM_SIZE1(depth);
|
||||||
|
return (((size_t)m.data) & (alignment - 1)) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
class TFNodeWrapper : public ImportNodeWrapper
|
class TFNodeWrapper : public ImportNodeWrapper
|
||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
@@ -719,8 +729,19 @@ public:
|
|||||||
{
|
{
|
||||||
if (!negativeScales)
|
if (!negativeScales)
|
||||||
{
|
{
|
||||||
Mat scales = getTensorContent(inputNodes[1]->attr().at("value").tensor(), /*copy*/false);
|
Mat scalesRef = getTensorContentRef_(inputNodes[1]->attr().at("value").tensor());
|
||||||
scales *= -1;
|
// FIXME: This breaks the const guarantees of tensor() by writing to scalesRef
|
||||||
|
if (isAlignedMat(scalesRef))
|
||||||
|
{
|
||||||
|
scalesRef *= -1;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
Mat scales = scalesRef.clone() * -1;
|
||||||
|
CV_Assert(scalesRef.isContinuous());
|
||||||
|
CV_Assert(scales.isContinuous());
|
||||||
|
memcpy(scalesRef.data, scales.data, scales.total() * scales.elemSize());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -832,7 +853,8 @@ void RemoveIdentityOps(tensorflow::GraphDef& net)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Mat getTensorContent(const tensorflow::TensorProto &tensor, bool copy)
|
// NB: returned Mat::data pointer may be unaligned
|
||||||
|
Mat getTensorContentRef_(const tensorflow::TensorProto& tensor)
|
||||||
{
|
{
|
||||||
const std::string& content = tensor.tensor_content();
|
const std::string& content = tensor.tensor_content();
|
||||||
Mat m;
|
Mat m;
|
||||||
@@ -904,7 +926,18 @@ Mat getTensorContent(const tensorflow::TensorProto &tensor, bool copy)
|
|||||||
CV_Error(Error::StsError, "Tensor's data type is not supported");
|
CV_Error(Error::StsError, "Tensor's data type is not supported");
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
return copy ? m.clone() : m;
|
|
||||||
|
return m;
|
||||||
|
}
|
||||||
|
|
||||||
|
Mat getTensorContent(const tensorflow::TensorProto& tensor, bool forceCopy)
|
||||||
|
{
|
||||||
|
// If necessary clone m to have aligned data pointer
|
||||||
|
Mat m = getTensorContentRef_(tensor);
|
||||||
|
if (forceCopy || !isAlignedMat(m))
|
||||||
|
return m.clone();
|
||||||
|
else
|
||||||
|
return m;
|
||||||
}
|
}
|
||||||
|
|
||||||
void releaseTensor(tensorflow::TensorProto* tensor)
|
void releaseTensor(tensorflow::TensorProto* tensor)
|
||||||
|
|||||||
@@ -21,7 +21,7 @@ void RemoveIdentityOps(tensorflow::GraphDef& net);
|
|||||||
|
|
||||||
void simplifySubgraphs(tensorflow::GraphDef& net);
|
void simplifySubgraphs(tensorflow::GraphDef& net);
|
||||||
|
|
||||||
Mat getTensorContent(const tensorflow::TensorProto &tensor, bool copy = true);
|
Mat getTensorContent(const tensorflow::TensorProto& tensor, bool forceCopy = true);
|
||||||
|
|
||||||
void releaseTensor(tensorflow::TensorProto* tensor);
|
void releaseTensor(tensorflow::TensorProto* tensor);
|
||||||
|
|
||||||
|
|||||||
@@ -122,8 +122,10 @@ void parseTensor(const tensorflow::TensorProto &tensor, Mat &dstBlob)
|
|||||||
}
|
}
|
||||||
|
|
||||||
dstBlob.create(shape, CV_32F);
|
dstBlob.create(shape, CV_32F);
|
||||||
|
CV_Assert(dstBlob.isContinuous());
|
||||||
|
|
||||||
Mat tensorContent = getTensorContent(tensor, /*no copy*/false);
|
Mat tensorContent = getTensorContent(tensor, /*no copy*/false);
|
||||||
|
CV_Assert(tensorContent.isContinuous());
|
||||||
int size = tensorContent.total();
|
int size = tensorContent.total();
|
||||||
CV_Assert(size == (int)dstBlob.total());
|
CV_Assert(size == (int)dstBlob.total());
|
||||||
|
|
||||||
@@ -404,12 +406,53 @@ void setKSize(LayerParams &layerParams, const tensorflow::NodeDef &layer)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void setPadding(LayerParams &layerParams, const tensorflow::NodeDef &layer)
|
void setPadMode(LayerParams &layerParams, const tensorflow::NodeDef &layer)
|
||||||
{
|
{
|
||||||
if (hasLayerAttr(layer, "padding"))
|
if (hasLayerAttr(layer, "padding"))
|
||||||
layerParams.set("pad_mode", getLayerAttr(layer, "padding").s());
|
layerParams.set("pad_mode", getLayerAttr(layer, "padding").s());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool getExplicitPadding(LayerParams &layerParams, const tensorflow::NodeDef &layer, int64_t (&pads)[8])
|
||||||
|
{
|
||||||
|
if (!layerParams.has("pad_mode") ||
|
||||||
|
layerParams.get("pad_mode").getStringValue() != "EXPLICIT")
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
CV_Assert(hasLayerAttr(layer, "explicit_paddings"));
|
||||||
|
|
||||||
|
const tensorflow::AttrValue& protoPads = getLayerAttr(layer, "explicit_paddings");
|
||||||
|
if (protoPads.list().i_size() != 8)
|
||||||
|
{
|
||||||
|
CV_Error(Error::StsNotImplemented, "Unsupported asymmetric padding configuration.");
|
||||||
|
}
|
||||||
|
|
||||||
|
int n = sizeof(pads) / sizeof(pads[0]);
|
||||||
|
for (int i = 0; i < n; ++i)
|
||||||
|
{
|
||||||
|
pads[i] = protoPads.list().i(i);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (getDataLayout(layer) != DATA_LAYOUT_NCHW)
|
||||||
|
{
|
||||||
|
CV_LOG_DEBUG(NULL, "DNN/TF: Data format " << getLayerAttr(layer, "data_format").s() << ", assuming NHWC.");
|
||||||
|
// Perhaps, we have NHWC padding dimensions order.
|
||||||
|
// N H W C
|
||||||
|
// 0 1 2 3 4 5 6 7
|
||||||
|
std::swap(pads[2], pads[6]);
|
||||||
|
std::swap(pads[3], pads[7]);
|
||||||
|
// N C W H
|
||||||
|
// 0 1 2 3 4 5 6 7
|
||||||
|
std::swap(pads[4], pads[6]);
|
||||||
|
std::swap(pads[5], pads[7]);
|
||||||
|
// N C H W
|
||||||
|
// 0 1 2 3 4 5 6 7
|
||||||
|
}
|
||||||
|
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
Pin parsePin(const std::string &name)
|
Pin parsePin(const std::string &name)
|
||||||
{
|
{
|
||||||
Pin pin(name);
|
Pin pin(name);
|
||||||
@@ -510,6 +553,7 @@ protected:
|
|||||||
|
|
||||||
private:
|
private:
|
||||||
void addPermuteLayer(const int* order, const std::string& permName, Pin& inpId);
|
void addPermuteLayer(const int* order, const std::string& permName, Pin& inpId);
|
||||||
|
void setPadding(LayerParams &layerParams, const tensorflow::NodeDef &layer, std::string& inputName, float value = 0.);
|
||||||
|
|
||||||
typedef void (TFImporter::*TFImporterNodeParser)(tensorflow::GraphDef&, const tensorflow::NodeDef&, LayerParams&);
|
typedef void (TFImporter::*TFImporterNodeParser)(tensorflow::GraphDef&, const tensorflow::NodeDef&, LayerParams&);
|
||||||
typedef std::map<std::string, TFImporterNodeParser> DispatchMap;
|
typedef std::map<std::string, TFImporterNodeParser> DispatchMap;
|
||||||
@@ -551,6 +595,31 @@ private:
|
|||||||
void parseCustomLayer (tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams);
|
void parseCustomLayer (tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams);
|
||||||
};
|
};
|
||||||
|
|
||||||
|
void TFImporter::setPadding(LayerParams &layerParams, const tensorflow::NodeDef &layer, std::string& inputName, float value)
|
||||||
|
{
|
||||||
|
setPadMode(layerParams, layer);
|
||||||
|
int64_t pads[8];
|
||||||
|
|
||||||
|
if (!getExplicitPadding(layerParams, layer, pads))
|
||||||
|
{
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
LayerParams padLp;
|
||||||
|
padLp.name = layer.name() + "/pad";
|
||||||
|
padLp.type = "Padding";
|
||||||
|
padLp.set("paddings", DictValue::arrayInt(pads, sizeof(pads) / sizeof(pads[0])));
|
||||||
|
padLp.set("value", value);
|
||||||
|
|
||||||
|
int id = dstNet.addLayer(padLp.name, padLp.type, padLp);
|
||||||
|
layer_id[padLp.name] = id;
|
||||||
|
|
||||||
|
connect(layer_id, dstNet, parsePin(inputName), id, 0);
|
||||||
|
inputName = padLp.name;
|
||||||
|
|
||||||
|
layerParams.set("pad_mode", "VALID");
|
||||||
|
}
|
||||||
|
|
||||||
const TFImporter::DispatchMap TFImporter::buildDispatchMap()
|
const TFImporter::DispatchMap TFImporter::buildDispatchMap()
|
||||||
{
|
{
|
||||||
static DispatchMap dispatch;
|
static DispatchMap dispatch;
|
||||||
@@ -580,7 +649,7 @@ const TFImporter::DispatchMap TFImporter::buildDispatchMap()
|
|||||||
dispatch["PriorBox"] = &TFImporter::parsePriorBox;
|
dispatch["PriorBox"] = &TFImporter::parsePriorBox;
|
||||||
dispatch["Softmax"] = &TFImporter::parseSoftmax;
|
dispatch["Softmax"] = &TFImporter::parseSoftmax;
|
||||||
dispatch["CropAndResize"] = &TFImporter::parseCropAndResize;
|
dispatch["CropAndResize"] = &TFImporter::parseCropAndResize;
|
||||||
dispatch["Mean"] = dispatch["Sum"] = &TFImporter::parseMean;
|
dispatch["Mean"] = dispatch["Sum"] = dispatch["Max"] = &TFImporter::parseMean;
|
||||||
dispatch["Pack"] = &TFImporter::parsePack;
|
dispatch["Pack"] = &TFImporter::parsePack;
|
||||||
dispatch["ClipByValue"] = &TFImporter::parseClipByValue;
|
dispatch["ClipByValue"] = &TFImporter::parseClipByValue;
|
||||||
dispatch["LeakyRelu"] = &TFImporter::parseLeakyRelu;
|
dispatch["LeakyRelu"] = &TFImporter::parseLeakyRelu;
|
||||||
@@ -590,6 +659,7 @@ const TFImporter::DispatchMap TFImporter::buildDispatchMap()
|
|||||||
return dispatch;
|
return dispatch;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "Conv2D" "SpaceToBatchND" "DepthwiseConv2dNative" "Pad" "MirrorPad" "Conv3D"
|
||||||
void TFImporter::parseConvolution(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer_, LayerParams& layerParams)
|
void TFImporter::parseConvolution(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer_, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
tensorflow::NodeDef layer = layer_;
|
tensorflow::NodeDef layer = layer_;
|
||||||
@@ -787,7 +857,7 @@ void TFImporter::parseConvolution(tensorflow::GraphDef& net, const tensorflow::N
|
|||||||
|
|
||||||
setStrides(layerParams, layer);
|
setStrides(layerParams, layer);
|
||||||
if (!layerParams.has("pad_w") && !layerParams.has("pad_h"))
|
if (!layerParams.has("pad_w") && !layerParams.has("pad_h"))
|
||||||
setPadding(layerParams, layer);
|
setPadding(layerParams, layer, input);
|
||||||
|
|
||||||
// The final node of dilated convolution subgraph.
|
// The final node of dilated convolution subgraph.
|
||||||
next_layers = getNextLayers(net, name, "BatchToSpaceND");
|
next_layers = getNextLayers(net, name, "BatchToSpaceND");
|
||||||
@@ -809,6 +879,7 @@ void TFImporter::parseConvolution(tensorflow::GraphDef& net, const tensorflow::N
|
|||||||
data_layouts[name] = DATA_LAYOUT_NHWC;
|
data_layouts[name] = DATA_LAYOUT_NHWC;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "BiasAdd" "Add" "AddV2" "Sub" "AddN"
|
||||||
void TFImporter::parseBias(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseBias(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
const std::string& name = layer.name();
|
const std::string& name = layer.name();
|
||||||
@@ -845,7 +916,12 @@ void TFImporter::parseBias(tensorflow::GraphDef& net, const tensorflow::NodeDef&
|
|||||||
layer_id[name] = id;
|
layer_id[name] = id;
|
||||||
|
|
||||||
// one input only
|
// one input only
|
||||||
connect(layer_id, dstNet, parsePin(layer.input(0)), id, 0);
|
Pin inp0 = parsePin(layer.input(0));
|
||||||
|
if (layer_id.find(inp0.name) != layer_id.end())
|
||||||
|
// First operand is a constant.
|
||||||
|
connect(layer_id, dstNet, parsePin(layer.input(0)), id, 0);
|
||||||
|
else
|
||||||
|
connect(layer_id, dstNet, parsePin(layer.input(1)), id, 0);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
@@ -1020,6 +1096,7 @@ void TFImporter::parseReshape(tensorflow::GraphDef& net, const tensorflow::NodeD
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "Flatten" "Squeeze"
|
||||||
void TFImporter::parseFlatten(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseFlatten(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
const std::string& name = layer.name();
|
const std::string& name = layer.name();
|
||||||
@@ -1178,6 +1255,7 @@ void TFImporter::parseLrn(tensorflow::GraphDef& net, const tensorflow::NodeDef&
|
|||||||
connectToAllBlobs(layer_id, dstNet, parsePin(layer.input(0)), id, num_inputs);
|
connectToAllBlobs(layer_id, dstNet, parsePin(layer.input(0)), id, num_inputs);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "Concat" "ConcatV2"
|
||||||
void TFImporter::parseConcat(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseConcat(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
const std::string& name = layer.name();
|
const std::string& name = layer.name();
|
||||||
@@ -1228,26 +1306,29 @@ void TFImporter::parseConcat(tensorflow::GraphDef& net, const tensorflow::NodeDe
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "MaxPool" "MaxPool3D"
|
||||||
void TFImporter::parseMaxPool(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseMaxPool(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
const std::string& name = layer.name();
|
const std::string& name = layer.name();
|
||||||
const int num_inputs = layer.input_size();
|
const int num_inputs = layer.input_size();
|
||||||
|
std::string inputName = layer.input(0);
|
||||||
|
|
||||||
CV_CheckGT(num_inputs, 0, "");
|
CV_CheckGT(num_inputs, 0, "");
|
||||||
layerParams.set("pool", "max");
|
layerParams.set("pool", "max");
|
||||||
|
|
||||||
setKSize(layerParams, layer);
|
setKSize(layerParams, layer);
|
||||||
setStrides(layerParams, layer);
|
setStrides(layerParams, layer);
|
||||||
setPadding(layerParams, layer);
|
setPadding(layerParams, layer, inputName, -std::numeric_limits<float>::infinity());
|
||||||
// Test_TensorFlow_nets.EAST_text_detection/1, NGRAPH/CPU
|
// Test_TensorFlow_nets.EAST_text_detection/1, NGRAPH/CPU
|
||||||
layerParams.set("ceil_mode", false);
|
layerParams.set("ceil_mode", false);
|
||||||
|
|
||||||
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
||||||
layer_id[name] = id;
|
layer_id[name] = id;
|
||||||
|
|
||||||
connectToAllBlobs(layer_id, dstNet, parsePin(layer.input(0)), id, num_inputs);
|
connectToAllBlobs(layer_id, dstNet, parsePin(inputName), id, num_inputs);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "AvgPool" "AvgPool3D"
|
||||||
void TFImporter::parseAvgPool(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseAvgPool(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
const std::string& name = layer.name();
|
const std::string& name = layer.name();
|
||||||
@@ -1258,7 +1339,7 @@ void TFImporter::parseAvgPool(tensorflow::GraphDef& net, const tensorflow::NodeD
|
|||||||
layerParams.set("ave_pool_padded_area", false);
|
layerParams.set("ave_pool_padded_area", false);
|
||||||
setKSize(layerParams, layer);
|
setKSize(layerParams, layer);
|
||||||
setStrides(layerParams, layer);
|
setStrides(layerParams, layer);
|
||||||
setPadding(layerParams, layer);
|
setPadMode(layerParams, layer);
|
||||||
|
|
||||||
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
||||||
layer_id[name] = id;
|
layer_id[name] = id;
|
||||||
@@ -1434,6 +1515,7 @@ void TFImporter::parseStridedSlice(tensorflow::GraphDef& net, const tensorflow::
|
|||||||
connect(layer_id, dstNet, parsePin(layer.input(0)), id, 0);
|
connect(layer_id, dstNet, parsePin(layer.input(0)), id, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "Mul" "RealDiv"
|
||||||
void TFImporter::parseMul(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseMul(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
const std::string& name = layer.name();
|
const std::string& name = layer.name();
|
||||||
@@ -1591,6 +1673,7 @@ void TFImporter::parseMul(tensorflow::GraphDef& net, const tensorflow::NodeDef&
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "FusedBatchNorm" "FusedBatchNormV3"
|
||||||
void TFImporter::parseFusedBatchNorm(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseFusedBatchNorm(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
// op: "FusedBatchNorm"
|
// op: "FusedBatchNorm"
|
||||||
@@ -1673,7 +1756,7 @@ void TFImporter::parseConv2DBackpropInput(tensorflow::GraphDef& net, const tenso
|
|||||||
// input: "weights"
|
// input: "weights"
|
||||||
// input: "input"
|
// input: "input"
|
||||||
|
|
||||||
const std::string& name = layer.name();
|
std::string name = layer.name();
|
||||||
const int num_inputs = layer.input_size();
|
const int num_inputs = layer.input_size();
|
||||||
|
|
||||||
CV_CheckEQ(num_inputs, 3, "Expected output shape, weights and input nodes");
|
CV_CheckEQ(num_inputs, 3, "Expected output shape, weights and input nodes");
|
||||||
@@ -1704,7 +1787,21 @@ void TFImporter::parseConv2DBackpropInput(tensorflow::GraphDef& net, const tenso
|
|||||||
layerParams.set("num_output", kshape[1]);
|
layerParams.set("num_output", kshape[1]);
|
||||||
|
|
||||||
setStrides(layerParams, layer);
|
setStrides(layerParams, layer);
|
||||||
setPadding(layerParams, layer);
|
setPadMode(layerParams, layer);
|
||||||
|
int64_t pads[8];
|
||||||
|
bool explicit_pads = getExplicitPadding(layerParams, layer, pads);
|
||||||
|
int64_t begs[4] = {};
|
||||||
|
int64_t ends[4] = {-1, -1, -1, -1};
|
||||||
|
if (explicit_pads)
|
||||||
|
{
|
||||||
|
name += "/deconv";
|
||||||
|
layerParams.set("pad_mode", "VALID");
|
||||||
|
for (int i = 2; i < 4; ++i) // begins=[0, 0, a, b], ends=[-1, -1, c, d]
|
||||||
|
{
|
||||||
|
begs[i] = pads[2*i];
|
||||||
|
ends[i] = -1 - pads[2*i + 1];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// For convolution layer, output shape computes as
|
// For convolution layer, output shape computes as
|
||||||
// o = 1 + (i - k + 2*p) / s
|
// o = 1 + (i - k + 2*p) / s
|
||||||
@@ -1721,8 +1818,9 @@ void TFImporter::parseConv2DBackpropInput(tensorflow::GraphDef& net, const tenso
|
|||||||
const int strideY = layerParams.get<int>("stride_h");
|
const int strideY = layerParams.get<int>("stride_h");
|
||||||
const int strideX = layerParams.get<int>("stride_w");
|
const int strideX = layerParams.get<int>("stride_w");
|
||||||
Mat outShape = getTensorContent(getConstBlob(layer, value_id, 0));
|
Mat outShape = getTensorContent(getConstBlob(layer, value_id, 0));
|
||||||
const int outH = outShape.at<int>(1);
|
int shift = (getDataLayout(layer) == DATA_LAYOUT_NCHW);
|
||||||
const int outW = outShape.at<int>(2);
|
const int outH = outShape.at<int>(1 + shift) + begs[2] - 1 - ends[2];
|
||||||
|
const int outW = outShape.at<int>(2 + shift) + begs[3] - 1 - ends[3];
|
||||||
if (layerParams.get<String>("pad_mode") == "SAME")
|
if (layerParams.get<String>("pad_mode") == "SAME")
|
||||||
{
|
{
|
||||||
layerParams.set("adj_w", (outW - 1) % strideX);
|
layerParams.set("adj_w", (outW - 1) % strideX);
|
||||||
@@ -1738,6 +1836,16 @@ void TFImporter::parseConv2DBackpropInput(tensorflow::GraphDef& net, const tenso
|
|||||||
|
|
||||||
// one input only
|
// one input only
|
||||||
connect(layer_id, dstNet, parsePin(layer.input(2)), id, 0);
|
connect(layer_id, dstNet, parsePin(layer.input(2)), id, 0);
|
||||||
|
if (explicit_pads) // If we have explicit paddings, remove extra data
|
||||||
|
{
|
||||||
|
layerParams.set("begin", DictValue::arrayInt(begs, sizeof(begs) / sizeof(begs[0])));
|
||||||
|
layerParams.set("end", DictValue::arrayInt(ends, sizeof(ends) / sizeof(ends[0])));
|
||||||
|
|
||||||
|
int id = dstNet.addLayer(layer.name(), "Slice", layerParams);
|
||||||
|
layer_id[layer.name()] = id;
|
||||||
|
|
||||||
|
connect(layer_id, dstNet, parsePin(name), id, 0);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void TFImporter::parseBlockLSTM(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseBlockLSTM(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
@@ -1745,8 +1853,8 @@ void TFImporter::parseBlockLSTM(tensorflow::GraphDef& net, const tensorflow::Nod
|
|||||||
// op: "BlockLSTM"
|
// op: "BlockLSTM"
|
||||||
// input: "lstm_block_wrapper/ToInt64/x" (ignore, number of time stamps)
|
// input: "lstm_block_wrapper/ToInt64/x" (ignore, number of time stamps)
|
||||||
// input: "input"
|
// input: "input"
|
||||||
// input: "lstm_block_wrapper/zeros" (ignore)
|
// input: "lstm_block_wrapper/zeros"
|
||||||
// input: "lstm_block_wrapper/zeros" (ignore)
|
// input: "lstm_block_wrapper/zeros"
|
||||||
// input: "lstm_block_wrapper/kernel"
|
// input: "lstm_block_wrapper/kernel"
|
||||||
// input: "lstm_block_wrapper/w_i_diag"
|
// input: "lstm_block_wrapper/w_i_diag"
|
||||||
// input: "lstm_block_wrapper/w_f_diag"
|
// input: "lstm_block_wrapper/w_f_diag"
|
||||||
@@ -1772,9 +1880,11 @@ void TFImporter::parseBlockLSTM(tensorflow::GraphDef& net, const tensorflow::Nod
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Mat W, Wh, Wx, b;
|
Mat W, Wh, Wx, b, cs_prev, h_prev;
|
||||||
blobFromTensor(getConstBlob(layer, value_id, 4), W);
|
blobFromTensor(getConstBlob(layer, value_id, 4), W);
|
||||||
blobFromTensor(getConstBlob(layer, value_id, 8), b);
|
blobFromTensor(getConstBlob(layer, value_id, 8), b);
|
||||||
|
blobFromTensor(getConstBlob(layer, value_id, 2), cs_prev);
|
||||||
|
blobFromTensor(getConstBlob(layer, value_id, 3), h_prev);
|
||||||
const int outSize = W.cols / 4;
|
const int outSize = W.cols / 4;
|
||||||
|
|
||||||
// IGFO->IFOG
|
// IGFO->IFOG
|
||||||
@@ -1790,10 +1900,12 @@ void TFImporter::parseBlockLSTM(tensorflow::GraphDef& net, const tensorflow::Nod
|
|||||||
Wx = W.rowRange(0, W.rows - outSize).t();
|
Wx = W.rowRange(0, W.rows - outSize).t();
|
||||||
Wh = W.rowRange(W.rows - outSize, W.rows).t();
|
Wh = W.rowRange(W.rows - outSize, W.rows).t();
|
||||||
|
|
||||||
layerParams.blobs.resize(3);
|
layerParams.blobs.resize(5);
|
||||||
layerParams.blobs[0] = Wh;
|
layerParams.blobs[0] = Wh;
|
||||||
layerParams.blobs[1] = Wx;
|
layerParams.blobs[1] = Wx;
|
||||||
layerParams.blobs[2] = b;
|
layerParams.blobs[2] = b;
|
||||||
|
layerParams.blobs[3] = h_prev;
|
||||||
|
layerParams.blobs[4] = cs_prev;
|
||||||
|
|
||||||
if (hasLayerAttr(layer, "use_peephole"))
|
if (hasLayerAttr(layer, "use_peephole"))
|
||||||
{
|
{
|
||||||
@@ -1801,14 +1913,14 @@ void TFImporter::parseBlockLSTM(tensorflow::GraphDef& net, const tensorflow::Nod
|
|||||||
if (usePeephole)
|
if (usePeephole)
|
||||||
{
|
{
|
||||||
layerParams.set("use_peephole", true);
|
layerParams.set("use_peephole", true);
|
||||||
layerParams.blobs.resize(6);
|
layerParams.blobs.resize(8);
|
||||||
for (int i = 0; i < 3; ++i)
|
for (int i = 0; i < 3; ++i)
|
||||||
{
|
{
|
||||||
Mat w;
|
Mat w;
|
||||||
blobFromTensor(getConstBlob(layer, value_id, 5 + i), w);
|
blobFromTensor(getConstBlob(layer, value_id, 5 + i), w);
|
||||||
w = w.reshape(1, w.total()); // Single column.
|
w = w.reshape(1, w.total()); // Single column.
|
||||||
w = Mat::diag(w); // Make a diagonal matrix.
|
w = Mat::diag(w); // Make a diagonal matrix.
|
||||||
layerParams.blobs[3 + i] = w;
|
layerParams.blobs[5 + i] = w;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1821,6 +1933,7 @@ void TFImporter::parseBlockLSTM(tensorflow::GraphDef& net, const tensorflow::Nod
|
|||||||
data_layouts[name] = DATA_LAYOUT_UNKNOWN;
|
data_layouts[name] = DATA_LAYOUT_UNKNOWN;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "ResizeNearestNeighbor" "ResizeBilinear" "FusedResizeAndPadConv2D"
|
||||||
void TFImporter::parseResize(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer_, LayerParams& layerParams)
|
void TFImporter::parseResize(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer_, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
tensorflow::NodeDef layer = layer_;
|
tensorflow::NodeDef layer = layer_;
|
||||||
@@ -2009,6 +2122,7 @@ void TFImporter::parseCropAndResize(tensorflow::GraphDef& net, const tensorflow:
|
|||||||
connect(layer_id, dstNet, parsePin(layer.input(1)), id, 1);
|
connect(layer_id, dstNet, parsePin(layer.input(1)), id, 1);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "Mean" "Sum" "Max"
|
||||||
void TFImporter::parseMean(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseMean(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
// Computes the mean of elements across dimensions of a tensor.
|
// Computes the mean of elements across dimensions of a tensor.
|
||||||
@@ -2027,7 +2141,12 @@ void TFImporter::parseMean(tensorflow::GraphDef& net, const tensorflow::NodeDef&
|
|||||||
const std::string& name = layer.name();
|
const std::string& name = layer.name();
|
||||||
const std::string& type = layer.op();
|
const std::string& type = layer.op();
|
||||||
const int num_inputs = layer.input_size();
|
const int num_inputs = layer.input_size();
|
||||||
|
std::string pool_type = cv::toLowerCase(type);
|
||||||
|
|
||||||
|
if (pool_type == "mean")
|
||||||
|
{
|
||||||
|
pool_type = "ave";
|
||||||
|
}
|
||||||
CV_CheckGT(num_inputs, 0, "");
|
CV_CheckGT(num_inputs, 0, "");
|
||||||
|
|
||||||
Mat indices = getTensorContent(getConstBlob(layer, value_id, 1));
|
Mat indices = getTensorContent(getConstBlob(layer, value_id, 1));
|
||||||
@@ -2064,7 +2183,7 @@ void TFImporter::parseMean(tensorflow::GraphDef& net, const tensorflow::NodeDef&
|
|||||||
LayerParams avgLp;
|
LayerParams avgLp;
|
||||||
std::string avgName = name + "/avg";
|
std::string avgName = name + "/avg";
|
||||||
CV_Assert(layer_id.find(avgName) == layer_id.end());
|
CV_Assert(layer_id.find(avgName) == layer_id.end());
|
||||||
avgLp.set("pool", type == "Mean" ? "ave" : "sum");
|
avgLp.set("pool", pool_type);
|
||||||
// pooling kernel H x 1
|
// pooling kernel H x 1
|
||||||
avgLp.set("global_pooling_h", true);
|
avgLp.set("global_pooling_h", true);
|
||||||
avgLp.set("kernel_w", 1);
|
avgLp.set("kernel_w", 1);
|
||||||
@@ -2105,7 +2224,7 @@ void TFImporter::parseMean(tensorflow::GraphDef& net, const tensorflow::NodeDef&
|
|||||||
int axis = toNCHW(indices.at<int>(0));
|
int axis = toNCHW(indices.at<int>(0));
|
||||||
if (axis == 2 || axis == 3)
|
if (axis == 2 || axis == 3)
|
||||||
{
|
{
|
||||||
layerParams.set("pool", type == "Mean" ? "ave" : "sum");
|
layerParams.set("pool", pool_type);
|
||||||
layerParams.set(axis == 2 ? "kernel_w" : "kernel_h", 1);
|
layerParams.set(axis == 2 ? "kernel_w" : "kernel_h", 1);
|
||||||
layerParams.set(axis == 2 ? "global_pooling_h" : "global_pooling_w", true);
|
layerParams.set(axis == 2 ? "global_pooling_h" : "global_pooling_w", true);
|
||||||
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
||||||
@@ -2137,7 +2256,7 @@ void TFImporter::parseMean(tensorflow::GraphDef& net, const tensorflow::NodeDef&
|
|||||||
Pin inpId = parsePin(layer.input(0));
|
Pin inpId = parsePin(layer.input(0));
|
||||||
addPermuteLayer(order, name + "/nhwc", inpId);
|
addPermuteLayer(order, name + "/nhwc", inpId);
|
||||||
|
|
||||||
layerParams.set("pool", type == "Mean" ? "ave" : "sum");
|
layerParams.set("pool", pool_type);
|
||||||
layerParams.set("kernel_h", 1);
|
layerParams.set("kernel_h", 1);
|
||||||
layerParams.set("global_pooling_w", true);
|
layerParams.set("global_pooling_w", true);
|
||||||
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
||||||
@@ -2167,7 +2286,7 @@ void TFImporter::parseMean(tensorflow::GraphDef& net, const tensorflow::NodeDef&
|
|||||||
if (indices.total() != 2 || indices.at<int>(0) != 1 || indices.at<int>(1) != 2)
|
if (indices.total() != 2 || indices.at<int>(0) != 1 || indices.at<int>(1) != 2)
|
||||||
CV_Error(Error::StsNotImplemented, "Unsupported mode of reduce_mean or reduce_sum operation.");
|
CV_Error(Error::StsNotImplemented, "Unsupported mode of reduce_mean or reduce_sum operation.");
|
||||||
|
|
||||||
layerParams.set("pool", type == "Mean" ? "ave" : "sum");
|
layerParams.set("pool", pool_type);
|
||||||
layerParams.set("global_pooling", true);
|
layerParams.set("global_pooling", true);
|
||||||
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
int id = dstNet.addLayer(name, "Pooling", layerParams);
|
||||||
layer_id[name] = id;
|
layer_id[name] = id;
|
||||||
@@ -2271,6 +2390,7 @@ void TFImporter::parseLeakyRelu(tensorflow::GraphDef& net, const tensorflow::Nod
|
|||||||
connectToAllBlobs(layer_id, dstNet, parsePin(layer.input(0)), id, num_inputs);
|
connectToAllBlobs(layer_id, dstNet, parsePin(layer.input(0)), id, num_inputs);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// "Abs" "Tanh" "Sigmoid" "Relu" "Elu" "Exp" "Identity" "Relu6"
|
||||||
void TFImporter::parseActivation(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
void TFImporter::parseActivation(tensorflow::GraphDef& net, const tensorflow::NodeDef& layer, LayerParams& layerParams)
|
||||||
{
|
{
|
||||||
const std::string& name = layer.name();
|
const std::string& name = layer.name();
|
||||||
@@ -2404,8 +2524,10 @@ void TFImporter::kernelFromTensor(const tensorflow::TensorProto &tensor, Mat &ds
|
|||||||
out_c = shape[0]; input_c = shape[1];
|
out_c = shape[0]; input_c = shape[1];
|
||||||
|
|
||||||
dstBlob.create(shape, CV_32F);
|
dstBlob.create(shape, CV_32F);
|
||||||
|
CV_Assert(dstBlob.isContinuous());
|
||||||
|
|
||||||
Mat tensorContent = getTensorContent(tensor, /*no copy*/false);
|
Mat tensorContent = getTensorContent(tensor, /*no copy*/false);
|
||||||
|
CV_Assert(tensorContent.isContinuous());
|
||||||
int size = tensorContent.total();
|
int size = tensorContent.total();
|
||||||
CV_Assert(size == (int)dstBlob.total());
|
CV_Assert(size == (int)dstBlob.total());
|
||||||
|
|
||||||
@@ -2717,7 +2839,6 @@ void TFImporter::populateNet()
|
|||||||
addConstNodes(netBin, value_id, layers_to_ignore);
|
addConstNodes(netBin, value_id, layers_to_ignore);
|
||||||
addConstNodes(netTxt, value_id, layers_to_ignore);
|
addConstNodes(netTxt, value_id, layers_to_ignore);
|
||||||
|
|
||||||
|
|
||||||
for (int li = 0; li < layersSize; li++)
|
for (int li = 0; li < layersSize; li++)
|
||||||
{
|
{
|
||||||
const tensorflow::NodeDef& layer = net.node(li);
|
const tensorflow::NodeDef& layer = net.node(li);
|
||||||
@@ -2773,7 +2894,7 @@ void TFImporter::parseNode(const tensorflow::NodeDef& layer)
|
|||||||
DispatchMap::const_iterator iter = dispatch.find(type);
|
DispatchMap::const_iterator iter = dispatch.find(type);
|
||||||
if (iter != dispatch.end())
|
if (iter != dispatch.end())
|
||||||
{
|
{
|
||||||
((*this).*(iter->second))(net, layer, layerParams);
|
CALL_MEMBER_FN(*this, iter->second)(net, layer, layerParams);
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -681,6 +681,78 @@ TEST_P(Test_Darknet_nets, YOLOv4_tiny)
|
|||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_Darknet_nets, YOLOv4x_mish)
|
||||||
|
{
|
||||||
|
applyTestTag(CV_TEST_TAG_LONG, (target == DNN_TARGET_CPU ? CV_TEST_TAG_MEMORY_1GB : CV_TEST_TAG_MEMORY_2GB));
|
||||||
|
|
||||||
|
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2020040000) // nGraph compilation failure
|
||||||
|
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && target == DNN_TARGET_OPENCL)
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
|
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH && target == DNN_TARGET_OPENCL_FP16)
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
|
#endif
|
||||||
|
#if defined(INF_ENGINE_RELEASE)
|
||||||
|
if (target == DNN_TARGET_MYRIAD) // NC_OUT_OF_MEMORY
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// batchId, classId, confidence, left, top, right, bottom
|
||||||
|
const int N0 = 3;
|
||||||
|
const int N1 = 5;
|
||||||
|
static const float ref_[/* (N0 + N1) * 7 */] = {
|
||||||
|
0, 16, 0.925536f, 0.17188f, 0.386832f, 0.406138f, 0.941696f,
|
||||||
|
0, 1, 0.912028f, 0.162125f, 0.208863f, 0.741316f, 0.729332f,
|
||||||
|
0, 7, 0.841018f, 0.608953f, 0.128653f, 0.900692f, 0.295657f,
|
||||||
|
|
||||||
|
1, 2, 0.925697f, 0.650438f, 0.458118f, 0.813927f, 0.661775f,
|
||||||
|
1, 0, 0.882156f, 0.203644f, 0.365763f, 0.265473f, 0.632195f,
|
||||||
|
1, 2, 0.848857f, 0.451044f, 0.462997f, 0.496629f, 0.522719f,
|
||||||
|
1, 9, 0.736015f, 0.374503f, 0.316029f, 0.399358f, 0.392883f,
|
||||||
|
1, 9, 0.727129f, 0.662469f, 0.373687f, 0.687877f, 0.441335f,
|
||||||
|
};
|
||||||
|
Mat ref(N0 + N1, 7, CV_32FC1, (void*)ref_);
|
||||||
|
|
||||||
|
double scoreDiff = (target == DNN_TARGET_OPENCL_FP16 || target == DNN_TARGET_MYRIAD) ? 0.006 : 8e-5;
|
||||||
|
double iouDiff = (target == DNN_TARGET_OPENCL_FP16 || target == DNN_TARGET_MYRIAD) ? 0.042 : 3e-4;
|
||||||
|
|
||||||
|
std::string config_file = "yolov4x-mish.cfg";
|
||||||
|
std::string weights_file = "yolov4x-mish.weights";
|
||||||
|
|
||||||
|
#if defined(INF_ENGINE_RELEASE)
|
||||||
|
if ((backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 ||
|
||||||
|
backend == DNN_BACKEND_INFERENCE_ENGINE_NGRAPH) && target == DNN_TARGET_MYRIAD &&
|
||||||
|
getInferenceEngineVPUType() == CV_DNN_INFERENCE_ENGINE_VPU_TYPE_MYRIAD_X)
|
||||||
|
{
|
||||||
|
scoreDiff = 0.04;
|
||||||
|
iouDiff = 0.2;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
{
|
||||||
|
SCOPED_TRACE("batch size 1");
|
||||||
|
testDarknetModel(config_file, weights_file, ref.rowRange(0, N0), scoreDiff, iouDiff);
|
||||||
|
}
|
||||||
|
|
||||||
|
{
|
||||||
|
SCOPED_TRACE("batch size 2");
|
||||||
|
|
||||||
|
#if defined(INF_ENGINE_RELEASE)
|
||||||
|
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||||
|
{
|
||||||
|
if (target == DNN_TARGET_OPENCL)
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
|
else if (target == DNN_TARGET_OPENCL_FP16 && INF_ENGINE_VER_MAJOR_LE(202010000))
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
|
else if (target == DNN_TARGET_MYRIAD &&
|
||||||
|
getInferenceEngineVPUType() == CV_DNN_INFERENCE_ENGINE_VPU_TYPE_MYRIAD_X)
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD_X);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
testDarknetModel(config_file, weights_file, ref, scoreDiff, iouDiff);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
INSTANTIATE_TEST_CASE_P(/**/, Test_Darknet_nets, dnnBackendsAndTargets());
|
INSTANTIATE_TEST_CASE_P(/**/, Test_Darknet_nets, dnnBackendsAndTargets());
|
||||||
|
|
||||||
|
|||||||
@@ -103,11 +103,34 @@ static const std::map<std::string, OpenVINOModelTestCaseInfo>& getOpenVINOTestMo
|
|||||||
#if INF_ENGINE_RELEASE >= 2020010000
|
#if INF_ENGINE_RELEASE >= 2020010000
|
||||||
// Downloaded using these parameters for Open Model Zoo downloader (2020.1):
|
// Downloaded using these parameters for Open Model Zoo downloader (2020.1):
|
||||||
// ./downloader.py -o ${OPENCV_DNN_TEST_DATA_PATH}/omz_intel_models --cache_dir ${OPENCV_DNN_TEST_DATA_PATH}/.omz_cache/ \
|
// ./downloader.py -o ${OPENCV_DNN_TEST_DATA_PATH}/omz_intel_models --cache_dir ${OPENCV_DNN_TEST_DATA_PATH}/.omz_cache/ \
|
||||||
// --name person-detection-retail-0013
|
// --name person-detection-retail-0013,age-gender-recognition-retail-0013
|
||||||
{ "person-detection-retail-0013", { // IRv10
|
{ "person-detection-retail-0013", { // IRv10
|
||||||
"intel/person-detection-retail-0013/FP32/person-detection-retail-0013",
|
"intel/person-detection-retail-0013/FP32/person-detection-retail-0013",
|
||||||
"intel/person-detection-retail-0013/FP16/person-detection-retail-0013"
|
"intel/person-detection-retail-0013/FP16/person-detection-retail-0013"
|
||||||
}},
|
}},
|
||||||
|
{ "age-gender-recognition-retail-0013", {
|
||||||
|
"intel/age-gender-recognition-retail-0013/FP16/age-gender-recognition-retail-0013",
|
||||||
|
"intel/age-gender-recognition-retail-0013/FP32/age-gender-recognition-retail-0013"
|
||||||
|
}},
|
||||||
|
#endif
|
||||||
|
#if INF_ENGINE_RELEASE >= 2021020000
|
||||||
|
// OMZ: 2020.2
|
||||||
|
{ "face-detection-0105", {
|
||||||
|
"intel/face-detection-0105/FP32/face-detection-0105",
|
||||||
|
"intel/face-detection-0105/FP16/face-detection-0105"
|
||||||
|
}},
|
||||||
|
{ "face-detection-0106", {
|
||||||
|
"intel/face-detection-0106/FP32/face-detection-0106",
|
||||||
|
"intel/face-detection-0106/FP16/face-detection-0106"
|
||||||
|
}},
|
||||||
|
#endif
|
||||||
|
#if INF_ENGINE_RELEASE >= 2021040000
|
||||||
|
// OMZ: 2021.4
|
||||||
|
{ "person-vehicle-bike-detection-2004", {
|
||||||
|
"intel/person-vehicle-bike-detection-2004/FP32/person-vehicle-bike-detection-2004",
|
||||||
|
"intel/person-vehicle-bike-detection-2004/FP16/person-vehicle-bike-detection-2004"
|
||||||
|
//"intel/person-vehicle-bike-detection-2004/FP16-INT8/person-vehicle-bike-detection-2004"
|
||||||
|
}},
|
||||||
#endif
|
#endif
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -123,13 +146,40 @@ static const std::vector<std::string> getOpenVINOTestModelsList()
|
|||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline static std::string getOpenVINOModel(const std::string &modelName, bool isFP16)
|
||||||
|
{
|
||||||
|
const std::map<std::string, OpenVINOModelTestCaseInfo>& models = getOpenVINOTestModels();
|
||||||
|
const auto it = models.find(modelName);
|
||||||
|
if (it != models.end())
|
||||||
|
{
|
||||||
|
OpenVINOModelTestCaseInfo modelInfo = it->second;
|
||||||
|
if (isFP16 && modelInfo.modelPathFP16)
|
||||||
|
return std::string(modelInfo.modelPathFP16);
|
||||||
|
else if (!isFP16 && modelInfo.modelPathFP32)
|
||||||
|
return std::string(modelInfo.modelPathFP32);
|
||||||
|
}
|
||||||
|
return std::string();
|
||||||
|
}
|
||||||
|
|
||||||
static inline void genData(const InferenceEngine::TensorDesc& desc, Mat& m, Blob::Ptr& dataPtr)
|
static inline void genData(const InferenceEngine::TensorDesc& desc, Mat& m, Blob::Ptr& dataPtr)
|
||||||
{
|
{
|
||||||
const std::vector<size_t>& dims = desc.getDims();
|
const std::vector<size_t>& dims = desc.getDims();
|
||||||
m.create(std::vector<int>(dims.begin(), dims.end()), CV_32F);
|
if (desc.getPrecision() == InferenceEngine::Precision::FP32)
|
||||||
randu(m, -1, 1);
|
{
|
||||||
|
m.create(std::vector<int>(dims.begin(), dims.end()), CV_32F);
|
||||||
dataPtr = make_shared_blob<float>(desc, (float*)m.data);
|
randu(m, -1, 1);
|
||||||
|
dataPtr = make_shared_blob<float>(desc, (float*)m.data);
|
||||||
|
}
|
||||||
|
else if (desc.getPrecision() == InferenceEngine::Precision::I32)
|
||||||
|
{
|
||||||
|
m.create(std::vector<int>(dims.begin(), dims.end()), CV_32S);
|
||||||
|
randu(m, -100, 100);
|
||||||
|
dataPtr = make_shared_blob<int>(desc, (int*)m.data);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
FAIL() << "Unsupported precision: " << desc.getPrecision();
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void runIE(Target target, const std::string& xmlPath, const std::string& binPath,
|
void runIE(Target target, const std::string& xmlPath, const std::string& binPath,
|
||||||
@@ -235,7 +285,16 @@ void runIE(Target target, const std::string& xmlPath, const std::string& binPath
|
|||||||
BlobMap inputBlobs;
|
BlobMap inputBlobs;
|
||||||
for (auto& it : net.getInputsInfo())
|
for (auto& it : net.getInputsInfo())
|
||||||
{
|
{
|
||||||
genData(it.second->getTensorDesc(), inputsMap[it.first], inputBlobs[it.first]);
|
const InferenceEngine::TensorDesc& desc = it.second->getTensorDesc();
|
||||||
|
genData(desc, inputsMap[it.first], inputBlobs[it.first]);
|
||||||
|
if (cvtest::debugLevel > 0)
|
||||||
|
{
|
||||||
|
const std::vector<size_t>& dims = desc.getDims();
|
||||||
|
std::cout << "Input: '" << it.first << "' precison=" << desc.getPrecision() << " dims=" << dims.size() << " [";
|
||||||
|
for (auto d : dims)
|
||||||
|
std::cout << " " << d;
|
||||||
|
std::cout << "] ocv_mat=" << inputsMap[it.first].size << " of " << typeToString(inputsMap[it.first].type()) << std::endl;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
infRequest.SetInput(inputBlobs);
|
infRequest.SetInput(inputBlobs);
|
||||||
|
|
||||||
@@ -244,7 +303,16 @@ void runIE(Target target, const std::string& xmlPath, const std::string& binPath
|
|||||||
BlobMap outputBlobs;
|
BlobMap outputBlobs;
|
||||||
for (auto& it : net.getOutputsInfo())
|
for (auto& it : net.getOutputsInfo())
|
||||||
{
|
{
|
||||||
genData(it.second->getTensorDesc(), outputsMap[it.first], outputBlobs[it.first]);
|
const InferenceEngine::TensorDesc& desc = it.second->getTensorDesc();
|
||||||
|
genData(desc, outputsMap[it.first], outputBlobs[it.first]);
|
||||||
|
if (cvtest::debugLevel > 0)
|
||||||
|
{
|
||||||
|
const std::vector<size_t>& dims = desc.getDims();
|
||||||
|
std::cout << "Output: '" << it.first << "' precison=" << desc.getPrecision() << " dims=" << dims.size() << " [";
|
||||||
|
for (auto d : dims)
|
||||||
|
std::cout << " " << d;
|
||||||
|
std::cout << "] ocv_mat=" << outputsMap[it.first].size << " of " << typeToString(outputsMap[it.first].type()) << std::endl;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
infRequest.SetOutput(outputBlobs);
|
infRequest.SetOutput(outputBlobs);
|
||||||
|
|
||||||
@@ -265,6 +333,12 @@ void runCV(Backend backendId, Target targetId, const std::string& xmlPath, const
|
|||||||
net.setPreferableTarget(targetId);
|
net.setPreferableTarget(targetId);
|
||||||
|
|
||||||
std::vector<String> outNames = net.getUnconnectedOutLayersNames();
|
std::vector<String> outNames = net.getUnconnectedOutLayersNames();
|
||||||
|
if (cvtest::debugLevel > 0)
|
||||||
|
{
|
||||||
|
std::cout << "OpenCV output names: " << outNames.size() << std::endl;
|
||||||
|
for (auto name : outNames)
|
||||||
|
std::cout << "- " << name << std::endl;
|
||||||
|
}
|
||||||
std::vector<Mat> outs;
|
std::vector<Mat> outs;
|
||||||
net.forward(outs, outNames);
|
net.forward(outs, outNames);
|
||||||
|
|
||||||
@@ -288,13 +362,26 @@ TEST_P(DNNTestOpenVINO, models)
|
|||||||
ASSERT_FALSE(backendId != DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && backendId != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH) <<
|
ASSERT_FALSE(backendId != DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && backendId != DNN_BACKEND_INFERENCE_ENGINE_NGRAPH) <<
|
||||||
"Inference Engine backend is required";
|
"Inference Engine backend is required";
|
||||||
|
|
||||||
#if INF_ENGINE_VER_MAJOR_EQ(2021040000)
|
#if INF_ENGINE_VER_MAJOR_GE(2021030000)
|
||||||
if (targetId == DNN_TARGET_MYRIAD && (
|
if (targetId == DNN_TARGET_MYRIAD && (false
|
||||||
modelName == "person-detection-retail-0013" || // ncDeviceOpen:1013 Failed to find booted device after boot
|
|| modelName == "person-detection-retail-0013" // ncDeviceOpen:1013 Failed to find booted device after boot
|
||||||
modelName == "age-gender-recognition-retail-0013" // ncDeviceOpen:1013 Failed to find booted device after boot
|
|| modelName == "age-gender-recognition-retail-0013" // ncDeviceOpen:1013 Failed to find booted device after boot
|
||||||
|
|| modelName == "face-detection-0105" // get_element_type() must be called on a node with exactly one output
|
||||||
|
|| modelName == "face-detection-0106" // get_element_type() must be called on a node with exactly one output
|
||||||
|
|| modelName == "person-vehicle-bike-detection-2004" // 2021.4+: ncDeviceOpen:1013 Failed to find booted device after boot
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_DNN_BACKEND_INFERENCE_ENGINE_NGRAPH, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_DNN_BACKEND_INFERENCE_ENGINE_NGRAPH, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
|
if (targetId == DNN_TARGET_OPENCL && (false
|
||||||
|
|| modelName == "face-detection-0106" // Operation: 2278 of type ExperimentalDetectronPriorGridGenerator(op::v6) is not supported
|
||||||
|
)
|
||||||
|
)
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL, CV_DNN_BACKEND_INFERENCE_ENGINE_NGRAPH, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
|
if (targetId == DNN_TARGET_OPENCL_FP16 && (false
|
||||||
|
|| modelName == "face-detection-0106" // Operation: 2278 of type ExperimentalDetectronPriorGridGenerator(op::v6) is not supported
|
||||||
|
)
|
||||||
|
)
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16, CV_DNN_BACKEND_INFERENCE_ENGINE_NGRAPH, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if INF_ENGINE_VER_MAJOR_GE(2020020000)
|
#if INF_ENGINE_VER_MAJOR_GE(2020020000)
|
||||||
@@ -319,11 +406,8 @@ TEST_P(DNNTestOpenVINO, models)
|
|||||||
|
|
||||||
bool isFP16 = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD);
|
bool isFP16 = (targetId == DNN_TARGET_OPENCL_FP16 || targetId == DNN_TARGET_MYRIAD);
|
||||||
|
|
||||||
const std::map<std::string, OpenVINOModelTestCaseInfo>& models = getOpenVINOTestModels();
|
const std::string modelPath = getOpenVINOModel(modelName, isFP16);
|
||||||
const auto it = models.find(modelName);
|
ASSERT_FALSE(modelPath.empty()) << modelName;
|
||||||
ASSERT_TRUE(it != models.end()) << modelName;
|
|
||||||
OpenVINOModelTestCaseInfo modelInfo = it->second;
|
|
||||||
std::string modelPath = isFP16 ? modelInfo.modelPathFP16 : modelInfo.modelPathFP32;
|
|
||||||
|
|
||||||
std::string xmlPath = findDataFile(modelPath + ".xml", false);
|
std::string xmlPath = findDataFile(modelPath + ".xml", false);
|
||||||
std::string binPath = findDataFile(modelPath + ".bin", false);
|
std::string binPath = findDataFile(modelPath + ".bin", false);
|
||||||
@@ -334,6 +418,8 @@ TEST_P(DNNTestOpenVINO, models)
|
|||||||
if (targetId == DNN_TARGET_MYRIAD)
|
if (targetId == DNN_TARGET_MYRIAD)
|
||||||
resetMyriadDevice();
|
resetMyriadDevice();
|
||||||
EXPECT_NO_THROW(runIE(targetId, xmlPath, binPath, inputsMap, ieOutputsMap)) << "runIE";
|
EXPECT_NO_THROW(runIE(targetId, xmlPath, binPath, inputsMap, ieOutputsMap)) << "runIE";
|
||||||
|
if (targetId == DNN_TARGET_MYRIAD)
|
||||||
|
resetMyriadDevice();
|
||||||
EXPECT_NO_THROW(runCV(backendId, targetId, xmlPath, binPath, inputsMap, cvOutputsMap)) << "runCV";
|
EXPECT_NO_THROW(runCV(backendId, targetId, xmlPath, binPath, inputsMap, cvOutputsMap)) << "runCV";
|
||||||
|
|
||||||
double eps = 0;
|
double eps = 0;
|
||||||
@@ -341,6 +427,14 @@ TEST_P(DNNTestOpenVINO, models)
|
|||||||
if (targetId == DNN_TARGET_CPU && checkHardwareSupport(CV_CPU_AVX_512F))
|
if (targetId == DNN_TARGET_CPU && checkHardwareSupport(CV_CPU_AVX_512F))
|
||||||
eps = 1e-5;
|
eps = 1e-5;
|
||||||
#endif
|
#endif
|
||||||
|
#if INF_ENGINE_VER_MAJOR_GE(2021030000)
|
||||||
|
if (targetId == DNN_TARGET_CPU && modelName == "face-detection-0105")
|
||||||
|
eps = 2e-4;
|
||||||
|
#endif
|
||||||
|
#if INF_ENGINE_VER_MAJOR_GE(2021040000)
|
||||||
|
if (targetId == DNN_TARGET_CPU && modelName == "person-vehicle-bike-detection-2004")
|
||||||
|
eps = 1e-6;
|
||||||
|
#endif
|
||||||
|
|
||||||
EXPECT_EQ(ieOutputsMap.size(), cvOutputsMap.size());
|
EXPECT_EQ(ieOutputsMap.size(), cvOutputsMap.size());
|
||||||
for (auto& srcIt : ieOutputsMap)
|
for (auto& srcIt : ieOutputsMap)
|
||||||
|
|||||||
@@ -434,7 +434,7 @@ class Layer_LSTM_Test : public ::testing::Test
|
|||||||
{
|
{
|
||||||
public:
|
public:
|
||||||
int numInp, numOut;
|
int numInp, numOut;
|
||||||
Mat Wh, Wx, b;
|
Mat Wh, Wx, b, h, c;
|
||||||
Ptr<LSTMLayer> layer;
|
Ptr<LSTMLayer> layer;
|
||||||
std::vector<Mat> inputs, outputs;
|
std::vector<Mat> inputs, outputs;
|
||||||
|
|
||||||
@@ -449,12 +449,17 @@ public:
|
|||||||
Wh = Mat::ones(4 * numOut, numOut, CV_32F);
|
Wh = Mat::ones(4 * numOut, numOut, CV_32F);
|
||||||
Wx = Mat::ones(4 * numOut, numInp, CV_32F);
|
Wx = Mat::ones(4 * numOut, numInp, CV_32F);
|
||||||
b = Mat::ones(4 * numOut, 1, CV_32F);
|
b = Mat::ones(4 * numOut, 1, CV_32F);
|
||||||
|
h = Mat::ones(4, numOut, CV_32F);
|
||||||
|
c = Mat::ones(4, numOut, CV_32F);
|
||||||
|
|
||||||
LayerParams lp;
|
LayerParams lp;
|
||||||
lp.blobs.resize(3);
|
lp.blobs.resize(5);
|
||||||
lp.blobs[0] = Wh;
|
lp.blobs[0] = Wh;
|
||||||
lp.blobs[1] = Wx;
|
lp.blobs[1] = Wx;
|
||||||
lp.blobs[2] = b;
|
lp.blobs[2] = b;
|
||||||
|
lp.blobs[3] = h;
|
||||||
|
lp.blobs[4] = c;
|
||||||
|
|
||||||
lp.set<bool>("produce_cell_output", produceCellOutput);
|
lp.set<bool>("produce_cell_output", produceCellOutput);
|
||||||
lp.set<bool>("use_timestamp_dim", useTimestampDim);
|
lp.set<bool>("use_timestamp_dim", useTimestampDim);
|
||||||
|
|
||||||
@@ -502,10 +507,12 @@ TEST_F(Layer_LSTM_Test, get_set_test)
|
|||||||
TEST(Layer_LSTM_Test_Accuracy_with_, CaffeRecurrent)
|
TEST(Layer_LSTM_Test_Accuracy_with_, CaffeRecurrent)
|
||||||
{
|
{
|
||||||
LayerParams lp;
|
LayerParams lp;
|
||||||
lp.blobs.resize(3);
|
lp.blobs.resize(5);
|
||||||
lp.blobs[0] = blobFromNPY(_tf("lstm.prototxt.w_2.npy")); // Wh
|
lp.blobs[0] = blobFromNPY(_tf("lstm.prototxt.w_2.npy")); // Wh
|
||||||
lp.blobs[1] = blobFromNPY(_tf("lstm.prototxt.w_0.npy")); // Wx
|
lp.blobs[1] = blobFromNPY(_tf("lstm.prototxt.w_0.npy")); // Wx
|
||||||
lp.blobs[2] = blobFromNPY(_tf("lstm.prototxt.w_1.npy")); // bias
|
lp.blobs[2] = blobFromNPY(_tf("lstm.prototxt.w_1.npy")); // bias
|
||||||
|
lp.blobs[3] = Mat::zeros(2, 17, CV_32F); // h_0
|
||||||
|
lp.blobs[4] = Mat::zeros(2, 17, CV_32F); // c_0
|
||||||
Ptr<LSTMLayer> layer = LSTMLayer::create(lp);
|
Ptr<LSTMLayer> layer = LSTMLayer::create(lp);
|
||||||
|
|
||||||
Mat inp = blobFromNPY(_tf("recurrent.input.npy"));
|
Mat inp = blobFromNPY(_tf("recurrent.input.npy"));
|
||||||
@@ -516,6 +523,68 @@ TEST(Layer_LSTM_Test_Accuracy_with_, CaffeRecurrent)
|
|||||||
normAssert(h_t_reference, outputs[0]);
|
normAssert(h_t_reference, outputs[0]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST(Layer_LSTM_Test_Accuracy_with_, HiddenParams)
|
||||||
|
{
|
||||||
|
Mat Wx = blobFromNPY(_tf("lstm.hidden.W.npy"));
|
||||||
|
Mat Wh = blobFromNPY(_tf("lstm.hidden.R.npy"));
|
||||||
|
Mat b = blobFromNPY(_tf("lstm.hidden.B.npy"));
|
||||||
|
Mat h0 = blobFromNPY(_tf("lstm.hidden.h0.npy"));
|
||||||
|
Mat c0 = blobFromNPY(_tf("lstm.hidden.c0.npy"));
|
||||||
|
|
||||||
|
const int numHidden = 3;
|
||||||
|
const int numDirs = Wx.size[0];
|
||||||
|
const int numFeatures = Wx.size[2];
|
||||||
|
|
||||||
|
b = b.reshape(1, b.size[0]);
|
||||||
|
Mat bx = b.colRange(0, b.cols / 2);
|
||||||
|
Mat bh = b.colRange(b.cols / 2, b.cols);
|
||||||
|
b = bx + bh;
|
||||||
|
|
||||||
|
// IFGO->IGFO
|
||||||
|
for (int k = 0; k < numDirs; ++k)
|
||||||
|
{
|
||||||
|
float* WxData = Wx.ptr<float>(k);
|
||||||
|
float* WhData = Wh.ptr<float>(k);
|
||||||
|
float* biasData = b.ptr<float>(k);
|
||||||
|
for (int j = 0; j < numHidden; ++j)
|
||||||
|
{
|
||||||
|
for (int i = 0; i < numFeatures; ++i)
|
||||||
|
{
|
||||||
|
std::swap(WxData[(numHidden + j) * numFeatures + i],
|
||||||
|
WxData[(numHidden * 2 + j) * numFeatures + i]);
|
||||||
|
}
|
||||||
|
for (int i = 0; i < numHidden; ++i)
|
||||||
|
{
|
||||||
|
std::swap(WhData[(numHidden + j) * numHidden + i],
|
||||||
|
WhData[(numHidden * 2 + j) * numHidden + i]);
|
||||||
|
}
|
||||||
|
std::swap(biasData[numHidden + j], biasData[numHidden * 2 + j]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Wx = Wx.reshape(1, Wx.size[0] * Wx.size[1]);
|
||||||
|
Wh = Wh.reshape(1, Wh.size[0] * Wh.size[1]);
|
||||||
|
h0 = h0.reshape(1, h0.size[0] * h0.size[1]);
|
||||||
|
c0 = c0.reshape(1, c0.size[0] * c0.size[1]);
|
||||||
|
|
||||||
|
LayerParams lstmParams;
|
||||||
|
lstmParams.blobs.resize(5);
|
||||||
|
lstmParams.blobs[0] = Wh;
|
||||||
|
lstmParams.blobs[1] = Wx;
|
||||||
|
lstmParams.blobs[2] = b;
|
||||||
|
lstmParams.blobs[3] = h0;
|
||||||
|
lstmParams.blobs[4] = c0;
|
||||||
|
lstmParams.set("bidirectional", false);
|
||||||
|
Ptr<LSTMLayer> layer = LSTMLayer::create(lstmParams);
|
||||||
|
|
||||||
|
Mat inp = blobFromNPY(_tf("lstm.hidden.input.npy"));
|
||||||
|
std::vector<Mat> inputs(1, inp), outputs;
|
||||||
|
runLayer(layer, inputs, outputs);
|
||||||
|
|
||||||
|
Mat h_t_reference = blobFromNPY(_tf("lstm.hidden.output.npy"));
|
||||||
|
normAssert(h_t_reference, outputs[0]);
|
||||||
|
}
|
||||||
|
|
||||||
TEST(Layer_RNN_Test_Accuracy_with_, CaffeRecurrent)
|
TEST(Layer_RNN_Test_Accuracy_with_, CaffeRecurrent)
|
||||||
{
|
{
|
||||||
Ptr<RNNLayer> layer = RNNLayer::create(LayerParams());
|
Ptr<RNNLayer> layer = RNNLayer::create(LayerParams());
|
||||||
@@ -560,6 +629,9 @@ TEST(Layer_LSTM_Test_Accuracy_, Reverse)
|
|||||||
bias.at<float>(2, 0) = 1e10f; // Output gate - always output everything
|
bias.at<float>(2, 0) = 1e10f; // Output gate - always output everything
|
||||||
bias.at<float>(3, 0) = 0.f; // Update signal
|
bias.at<float>(3, 0) = 0.f; // Update signal
|
||||||
|
|
||||||
|
cv::Mat hInternal = cv::Mat::zeros(1, 1, CV_32FC1);
|
||||||
|
cv::Mat cInternal = cv::Mat::zeros(1, 1, CV_32FC1);
|
||||||
|
|
||||||
LayerParams lp;
|
LayerParams lp;
|
||||||
lp.set("reverse", true);
|
lp.set("reverse", true);
|
||||||
lp.set("use_timestamp_dim", true);
|
lp.set("use_timestamp_dim", true);
|
||||||
@@ -567,6 +639,8 @@ TEST(Layer_LSTM_Test_Accuracy_, Reverse)
|
|||||||
lp.blobs.push_back(Wh);
|
lp.blobs.push_back(Wh);
|
||||||
lp.blobs.push_back(Wx);
|
lp.blobs.push_back(Wx);
|
||||||
lp.blobs.push_back(bias);
|
lp.blobs.push_back(bias);
|
||||||
|
lp.blobs.push_back(hInternal);
|
||||||
|
lp.blobs.push_back(cInternal);
|
||||||
|
|
||||||
cv::Ptr<cv::dnn::LSTMLayer> layer = LSTMLayer::create(lp);
|
cv::Ptr<cv::dnn::LSTMLayer> layer = LSTMLayer::create(lp);
|
||||||
std::vector<cv::Mat> outputs;
|
std::vector<cv::Mat> outputs;
|
||||||
|
|||||||
@@ -109,6 +109,7 @@ TEST_P(Test_ONNX_layers, MaxPooling_2)
|
|||||||
TEST_P(Test_ONNX_layers, Convolution)
|
TEST_P(Test_ONNX_layers, Convolution)
|
||||||
{
|
{
|
||||||
testONNXModels("convolution");
|
testONNXModels("convolution");
|
||||||
|
testONNXModels("conv_asymmetric_pads");
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, Convolution_variable_weight)
|
TEST_P(Test_ONNX_layers, Convolution_variable_weight)
|
||||||
@@ -249,6 +250,11 @@ TEST_P(Test_ONNX_layers, ReLU)
|
|||||||
testONNXModels("ReLU");
|
testONNXModels("ReLU");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_ONNX_layers, PReLU)
|
||||||
|
{
|
||||||
|
testONNXModels("PReLU_slope");
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, Clip)
|
TEST_P(Test_ONNX_layers, Clip)
|
||||||
{
|
{
|
||||||
testONNXModels("clip", npy);
|
testONNXModels("clip", npy);
|
||||||
@@ -283,6 +289,8 @@ TEST_P(Test_ONNX_layers, Scale)
|
|||||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||||
testONNXModels("scale");
|
testONNXModels("scale");
|
||||||
|
testONNXModels("scale_broadcast", npy, 0, 0, false, true, 3);
|
||||||
|
testONNXModels("scale_broadcast_mid", npy, 0, 0, false, true, 2);
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, ReduceMean3D)
|
TEST_P(Test_ONNX_layers, ReduceMean3D)
|
||||||
@@ -469,6 +477,8 @@ TEST_P(Test_ONNX_layers, MatMulAdd)
|
|||||||
|
|
||||||
TEST_P(Test_ONNX_layers, Expand)
|
TEST_P(Test_ONNX_layers, Expand)
|
||||||
{
|
{
|
||||||
|
testONNXModels("expand");
|
||||||
|
testONNXModels("expand_identity");
|
||||||
testONNXModels("expand_batch");
|
testONNXModels("expand_batch");
|
||||||
testONNXModels("expand_channels");
|
testONNXModels("expand_channels");
|
||||||
testONNXModels("expand_neg_batch");
|
testONNXModels("expand_neg_batch");
|
||||||
@@ -551,6 +561,11 @@ TEST_P(Test_ONNX_layers, DynamicResize)
|
|||||||
testONNXModels("dynamic_resize_scale_11", npy, 0, 0, false, true, 2);
|
testONNXModels("dynamic_resize_scale_11", npy, 0, 0, false, true, 2);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_ONNX_layers, Resize_HumanSeg)
|
||||||
|
{
|
||||||
|
testONNXModels("resize_humanseg");
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, Div)
|
TEST_P(Test_ONNX_layers, Div)
|
||||||
{
|
{
|
||||||
const String model = _tf("models/div.onnx");
|
const String model = _tf("models/div.onnx");
|
||||||
@@ -590,6 +605,7 @@ TEST_P(Test_ONNX_layers, DynamicReshape)
|
|||||||
TEST_P(Test_ONNX_layers, Reshape)
|
TEST_P(Test_ONNX_layers, Reshape)
|
||||||
{
|
{
|
||||||
testONNXModels("unsqueeze");
|
testONNXModels("unsqueeze");
|
||||||
|
testONNXModels("unsqueeze_opset_13");
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, Squeeze)
|
TEST_P(Test_ONNX_layers, Squeeze)
|
||||||
@@ -604,6 +620,7 @@ TEST_P(Test_ONNX_layers, ReduceL2)
|
|||||||
testONNXModels("reduceL2");
|
testONNXModels("reduceL2");
|
||||||
testONNXModels("reduceL2_subgraph");
|
testONNXModels("reduceL2_subgraph");
|
||||||
testONNXModels("reduceL2_subgraph_2");
|
testONNXModels("reduceL2_subgraph_2");
|
||||||
|
testONNXModels("reduceL2_subgraph2_2");
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, Split)
|
TEST_P(Test_ONNX_layers, Split)
|
||||||
@@ -616,6 +633,8 @@ TEST_P(Test_ONNX_layers, Split)
|
|||||||
testONNXModels("split_2");
|
testONNXModels("split_2");
|
||||||
testONNXModels("split_3");
|
testONNXModels("split_3");
|
||||||
testONNXModels("split_4");
|
testONNXModels("split_4");
|
||||||
|
testONNXModels("split_sizes");
|
||||||
|
testONNXModels("split_neg_axis");
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, Slice)
|
TEST_P(Test_ONNX_layers, Slice)
|
||||||
@@ -624,6 +643,7 @@ TEST_P(Test_ONNX_layers, Slice)
|
|||||||
testONNXModels("slice", npy, 0, 0, false, false);
|
testONNXModels("slice", npy, 0, 0, false, false);
|
||||||
#else
|
#else
|
||||||
testONNXModels("slice");
|
testONNXModels("slice");
|
||||||
|
testONNXModels("slice_neg_starts");
|
||||||
testONNXModels("slice_opset_11");
|
testONNXModels("slice_opset_11");
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
@@ -664,6 +684,11 @@ TEST_P(Test_ONNX_layers, Split_EltwiseMax)
|
|||||||
testONNXModels("split_max");
|
testONNXModels("split_max");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_ONNX_layers, LSTM_Activations)
|
||||||
|
{
|
||||||
|
testONNXModels("lstm_cntk_tanh", pb, 0, 0, false, false);
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, LSTM)
|
TEST_P(Test_ONNX_layers, LSTM)
|
||||||
{
|
{
|
||||||
testONNXModels("lstm", npy, 0, 0, false, false);
|
testONNXModels("lstm", npy, 0, 0, false, false);
|
||||||
@@ -674,6 +699,16 @@ TEST_P(Test_ONNX_layers, LSTM_bidirectional)
|
|||||||
testONNXModels("lstm_bidirectional", npy, 0, 0, false, false);
|
testONNXModels("lstm_bidirectional", npy, 0, 0, false, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_ONNX_layers, LSTM_hidden)
|
||||||
|
{
|
||||||
|
testONNXModels("hidden_lstm", npy, 0, 0, false, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_ONNX_layers, LSTM_hidden_bidirectional)
|
||||||
|
{
|
||||||
|
testONNXModels("hidden_lstm_bi", npy, 0, 0, false, false);
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, Pad2d_Unfused)
|
TEST_P(Test_ONNX_layers, Pad2d_Unfused)
|
||||||
{
|
{
|
||||||
testONNXModels("ReflectionPad2d");
|
testONNXModels("ReflectionPad2d");
|
||||||
@@ -809,6 +844,7 @@ TEST_P(Test_ONNX_layers, DynamicAxes)
|
|||||||
testONNXModels("resize_opset11_torch1.6_dynamic_axes");
|
testONNXModels("resize_opset11_torch1.6_dynamic_axes");
|
||||||
testONNXModels("average_pooling_dynamic_axes");
|
testONNXModels("average_pooling_dynamic_axes");
|
||||||
testONNXModels("maxpooling_sigmoid_dynamic_axes");
|
testONNXModels("maxpooling_sigmoid_dynamic_axes");
|
||||||
|
testONNXModels("dynamic_batch");
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_P(Test_ONNX_layers, MaxPool1d)
|
TEST_P(Test_ONNX_layers, MaxPool1d)
|
||||||
|
|||||||
@@ -128,6 +128,13 @@ TEST_P(Test_TensorFlow_layers, reduce_mean)
|
|||||||
runTensorFlowNet("global_pool_by_axis");
|
runTensorFlowNet("global_pool_by_axis");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_TensorFlow_layers, reduce_max)
|
||||||
|
{
|
||||||
|
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER);
|
||||||
|
runTensorFlowNet("max_pool_by_axis");
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_TensorFlow_layers, reduce_sum)
|
TEST_P(Test_TensorFlow_layers, reduce_sum)
|
||||||
{
|
{
|
||||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019)
|
||||||
@@ -135,11 +142,21 @@ TEST_P(Test_TensorFlow_layers, reduce_sum)
|
|||||||
runTensorFlowNet("sum_pool_by_axis");
|
runTensorFlowNet("sum_pool_by_axis");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_TensorFlow_layers, reduce_max_channel)
|
||||||
|
{
|
||||||
|
runTensorFlowNet("reduce_max_channel");
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_TensorFlow_layers, reduce_sum_channel)
|
TEST_P(Test_TensorFlow_layers, reduce_sum_channel)
|
||||||
{
|
{
|
||||||
runTensorFlowNet("reduce_sum_channel");
|
runTensorFlowNet("reduce_sum_channel");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_TensorFlow_layers, reduce_max_channel_keep_dims)
|
||||||
|
{
|
||||||
|
runTensorFlowNet("reduce_max_channel", false, 0.0, 0.0, false, "_keep_dims");
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_TensorFlow_layers, reduce_sum_channel_keep_dims)
|
TEST_P(Test_TensorFlow_layers, reduce_sum_channel_keep_dims)
|
||||||
{
|
{
|
||||||
runTensorFlowNet("reduce_sum_channel", false, 0.0, 0.0, false, "_keep_dims");
|
runTensorFlowNet("reduce_sum_channel", false, 0.0, 0.0, false, "_keep_dims");
|
||||||
@@ -203,6 +220,16 @@ TEST_P(Test_TensorFlow_layers, padding)
|
|||||||
runTensorFlowNet("keras_pad_concat");
|
runTensorFlowNet("keras_pad_concat");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_TensorFlow_layers, padding_asymmetric)
|
||||||
|
{
|
||||||
|
runTensorFlowNet("conv2d_asymmetric_pads_nchw");
|
||||||
|
runTensorFlowNet("conv2d_asymmetric_pads_nhwc");
|
||||||
|
runTensorFlowNet("max_pool2d_asymmetric_pads_nchw");
|
||||||
|
runTensorFlowNet("max_pool2d_asymmetric_pads_nhwc");
|
||||||
|
runTensorFlowNet("conv2d_backprop_input_asymmetric_pads_nchw");
|
||||||
|
runTensorFlowNet("conv2d_backprop_input_asymmetric_pads_nhwc");
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_TensorFlow_layers, padding_same)
|
TEST_P(Test_TensorFlow_layers, padding_same)
|
||||||
{
|
{
|
||||||
// Reference output values are in range [0.0006, 2.798]
|
// Reference output values are in range [0.0006, 2.798]
|
||||||
@@ -376,6 +403,11 @@ TEST_P(Test_TensorFlow_layers, pooling_reduce_mean)
|
|||||||
runTensorFlowNet("reduce_mean"); // an average pooling over all spatial dimensions.
|
runTensorFlowNet("reduce_mean"); // an average pooling over all spatial dimensions.
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_TensorFlow_layers, pooling_reduce_max)
|
||||||
|
{
|
||||||
|
runTensorFlowNet("reduce_max"); // a MAX pooling over all spatial dimensions.
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_TensorFlow_layers, pooling_reduce_sum)
|
TEST_P(Test_TensorFlow_layers, pooling_reduce_sum)
|
||||||
{
|
{
|
||||||
runTensorFlowNet("reduce_sum"); // a SUM pooling over all spatial dimensions.
|
runTensorFlowNet("reduce_sum"); // a SUM pooling over all spatial dimensions.
|
||||||
@@ -527,6 +559,18 @@ TEST_P(Test_TensorFlow_layers, l2_normalize)
|
|||||||
runTensorFlowNet("l2_normalize");
|
runTensorFlowNet("l2_normalize");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_TensorFlow_layers, BiasAdd)
|
||||||
|
{
|
||||||
|
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_GE(2019010000)
|
||||||
|
if (backend == DNN_BACKEND_INFERENCE_ENGINE_NN_BUILDER_2019 && target == DNN_TARGET_MYRIAD
|
||||||
|
&& getInferenceEngineVPUType() == CV_DNN_INFERENCE_ENGINE_VPU_TYPE_MYRIAD_X
|
||||||
|
)
|
||||||
|
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD_X, CV_TEST_TAG_DNN_SKIP_IE_NN_BUILDER, CV_TEST_TAG_DNN_SKIP_IE_VERSION);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
runTensorFlowNet("bias_add_1");
|
||||||
|
}
|
||||||
|
|
||||||
// TODO: fix it and add to l2_normalize
|
// TODO: fix it and add to l2_normalize
|
||||||
TEST_P(Test_TensorFlow_layers, l2_normalize_3d)
|
TEST_P(Test_TensorFlow_layers, l2_normalize_3d)
|
||||||
{
|
{
|
||||||
@@ -1093,6 +1137,11 @@ TEST_P(Test_TensorFlow_layers, resize_bilinear_down)
|
|||||||
runTensorFlowNet("resize_bilinear_down");
|
runTensorFlowNet("resize_bilinear_down");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_P(Test_TensorFlow_layers, resize_concat_optimization)
|
||||||
|
{
|
||||||
|
runTensorFlowNet("resize_concat_optimization");
|
||||||
|
}
|
||||||
|
|
||||||
TEST_P(Test_TensorFlow_layers, tf2_dense)
|
TEST_P(Test_TensorFlow_layers, tf2_dense)
|
||||||
{
|
{
|
||||||
runTensorFlowNet("tf2_dense");
|
runTensorFlowNet("tf2_dense");
|
||||||
|
|||||||
@@ -1082,7 +1082,7 @@ public:
|
|||||||
that is, copies both parameters and train data. If emptyTrainData is true, the method creates an
|
that is, copies both parameters and train data. If emptyTrainData is true, the method creates an
|
||||||
object copy with the current parameters but with empty train data.
|
object copy with the current parameters but with empty train data.
|
||||||
*/
|
*/
|
||||||
CV_WRAP virtual Ptr<DescriptorMatcher> clone( bool emptyTrainData=false ) const = 0;
|
CV_WRAP CV_NODISCARD_STD virtual Ptr<DescriptorMatcher> clone( bool emptyTrainData=false ) const = 0;
|
||||||
|
|
||||||
/** @brief Creates a descriptor matcher of a given type with the default parameters (using default
|
/** @brief Creates a descriptor matcher of a given type with the default parameters (using default
|
||||||
constructor).
|
constructor).
|
||||||
@@ -1142,7 +1142,7 @@ protected:
|
|||||||
static bool isPossibleMatch( InputArray mask, int queryIdx, int trainIdx );
|
static bool isPossibleMatch( InputArray mask, int queryIdx, int trainIdx );
|
||||||
static bool isMaskedOut( InputArrayOfArrays masks, int queryIdx );
|
static bool isMaskedOut( InputArrayOfArrays masks, int queryIdx );
|
||||||
|
|
||||||
static Mat clone_op( Mat m ) { return m.clone(); }
|
CV_NODISCARD_STD static Mat clone_op( Mat m ) { return m.clone(); }
|
||||||
void checkMasks( InputArrayOfArrays masks, int queryDescriptorsCount ) const;
|
void checkMasks( InputArrayOfArrays masks, int queryDescriptorsCount ) const;
|
||||||
|
|
||||||
//! Collection of descriptors from train images.
|
//! Collection of descriptors from train images.
|
||||||
@@ -1183,7 +1183,7 @@ public:
|
|||||||
*/
|
*/
|
||||||
CV_WRAP static Ptr<BFMatcher> create( int normType=NORM_L2, bool crossCheck=false ) ;
|
CV_WRAP static Ptr<BFMatcher> create( int normType=NORM_L2, bool crossCheck=false ) ;
|
||||||
|
|
||||||
virtual Ptr<DescriptorMatcher> clone( bool emptyTrainData=false ) const CV_OVERRIDE;
|
CV_NODISCARD_STD virtual Ptr<DescriptorMatcher> clone( bool emptyTrainData=false ) const CV_OVERRIDE;
|
||||||
protected:
|
protected:
|
||||||
virtual void knnMatchImpl( InputArray queryDescriptors, std::vector<std::vector<DMatch> >& matches, int k,
|
virtual void knnMatchImpl( InputArray queryDescriptors, std::vector<std::vector<DMatch> >& matches, int k,
|
||||||
InputArrayOfArrays masks=noArray(), bool compactResult=false ) CV_OVERRIDE;
|
InputArrayOfArrays masks=noArray(), bool compactResult=false ) CV_OVERRIDE;
|
||||||
@@ -1222,7 +1222,7 @@ public:
|
|||||||
|
|
||||||
CV_WRAP static Ptr<FlannBasedMatcher> create();
|
CV_WRAP static Ptr<FlannBasedMatcher> create();
|
||||||
|
|
||||||
virtual Ptr<DescriptorMatcher> clone( bool emptyTrainData=false ) const CV_OVERRIDE;
|
CV_NODISCARD_STD virtual Ptr<DescriptorMatcher> clone( bool emptyTrainData=false ) const CV_OVERRIDE;
|
||||||
protected:
|
protected:
|
||||||
static void convertToDMatches( const DescriptorCollection& descriptors,
|
static void convertToDMatches( const DescriptorCollection& descriptors,
|
||||||
const Mat& indices, const Mat& distances,
|
const Mat& indices, const Mat& distances,
|
||||||
|
|||||||
@@ -44,6 +44,8 @@
|
|||||||
#include <iterator>
|
#include <iterator>
|
||||||
#include <limits>
|
#include <limits>
|
||||||
|
|
||||||
|
#include <opencv2/core/utils/logger.hpp>
|
||||||
|
|
||||||
// Requires CMake flag: DEBUG_opencv_features2d=ON
|
// Requires CMake flag: DEBUG_opencv_features2d=ON
|
||||||
//#define DEBUG_BLOB_DETECTOR
|
//#define DEBUG_BLOB_DETECTOR
|
||||||
|
|
||||||
@@ -317,6 +319,19 @@ void SimpleBlobDetectorImpl::detect(InputArray image, std::vector<cv::KeyPoint>&
|
|||||||
CV_Error(Error::StsUnsupportedFormat, "Blob detector only supports 8-bit images!");
|
CV_Error(Error::StsUnsupportedFormat, "Blob detector only supports 8-bit images!");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
CV_CheckGT(params.thresholdStep, 0.0f, "");
|
||||||
|
if (params.minThreshold + params.thresholdStep >= params.maxThreshold)
|
||||||
|
{
|
||||||
|
// https://github.com/opencv/opencv/issues/6667
|
||||||
|
CV_LOG_ONCE_INFO(NULL, "SimpleBlobDetector: params.minDistBetweenBlobs is ignored for case with single threshold");
|
||||||
|
#if 0 // OpenCV 5.0
|
||||||
|
CV_CheckEQ(params.minRepeatability, 1u, "Incompatible parameters for case with single threshold");
|
||||||
|
#else
|
||||||
|
if (params.minRepeatability != 1)
|
||||||
|
CV_LOG_WARNING(NULL, "SimpleBlobDetector: params.minRepeatability=" << params.minRepeatability << " is incompatible for case with single threshold. Empty result is expected.");
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
std::vector < std::vector<Center> > centers;
|
std::vector < std::vector<Center> > centers;
|
||||||
for (double thresh = params.minThreshold; thresh < params.maxThreshold; thresh += params.thresholdStep)
|
for (double thresh = params.minThreshold; thresh < params.maxThreshold; thresh += params.thresholdStep)
|
||||||
{
|
{
|
||||||
@@ -325,19 +340,13 @@ void SimpleBlobDetectorImpl::detect(InputArray image, std::vector<cv::KeyPoint>&
|
|||||||
|
|
||||||
std::vector < Center > curCenters;
|
std::vector < Center > curCenters;
|
||||||
findBlobs(grayscaleImage, binarizedImage, curCenters);
|
findBlobs(grayscaleImage, binarizedImage, curCenters);
|
||||||
if(params.maxThreshold - params.minThreshold <= params.thresholdStep) {
|
|
||||||
// if the difference between min and max threshold is less than the threshold step
|
|
||||||
// we're only going to enter the loop once, so we need to add curCenters
|
|
||||||
// to ensure we still use minDistBetweenBlobs
|
|
||||||
centers.push_back(curCenters);
|
|
||||||
}
|
|
||||||
std::vector < std::vector<Center> > newCenters;
|
std::vector < std::vector<Center> > newCenters;
|
||||||
for (size_t i = 0; i < curCenters.size(); i++)
|
for (size_t i = 0; i < curCenters.size(); i++)
|
||||||
{
|
{
|
||||||
bool isNew = true;
|
bool isNew = true;
|
||||||
for (size_t j = 0; j < centers.size(); j++)
|
for (size_t j = 0; j < centers.size(); j++)
|
||||||
{
|
{
|
||||||
double dist = norm(centers[j][centers[j].size() / 2 ].location - curCenters[i].location);
|
double dist = norm(centers[j][ centers[j].size() / 2 ].location - curCenters[i].location);
|
||||||
isNew = dist >= params.minDistBetweenBlobs && dist >= centers[j][ centers[j].size() / 2 ].radius && dist >= curCenters[i].radius;
|
isNew = dist >= params.minDistBetweenBlobs && dist >= centers[j][ centers[j].size() / 2 ].radius && dist >= curCenters[i].radius;
|
||||||
if (!isNew)
|
if (!isNew)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -311,7 +311,7 @@ private:
|
|||||||
void KAZEFeatures::Determinant_Hessian(std::vector<KeyPoint>& kpts)
|
void KAZEFeatures::Determinant_Hessian(std::vector<KeyPoint>& kpts)
|
||||||
{
|
{
|
||||||
int level = 0;
|
int level = 0;
|
||||||
float dist = 0.0, smax = 3.0;
|
float smax = 3.0;
|
||||||
int npoints = 0, id_repeated = 0;
|
int npoints = 0, id_repeated = 0;
|
||||||
int left_x = 0, right_x = 0, up_y = 0, down_y = 0;
|
int left_x = 0, right_x = 0, up_y = 0, down_y = 0;
|
||||||
bool is_extremum = false, is_repeated = false, is_out = false;
|
bool is_extremum = false, is_repeated = false, is_out = false;
|
||||||
@@ -338,17 +338,24 @@ void KAZEFeatures::Determinant_Hessian(std::vector<KeyPoint>& kpts)
|
|||||||
for (int j = 0; j < (int)kpts_par_[i].size(); j++)
|
for (int j = 0; j < (int)kpts_par_[i].size(); j++)
|
||||||
{
|
{
|
||||||
level = i + 1;
|
level = i + 1;
|
||||||
|
const TEvolution& evolution_level = evolution_[level];
|
||||||
|
|
||||||
is_extremum = true;
|
is_extremum = true;
|
||||||
is_repeated = false;
|
is_repeated = false;
|
||||||
is_out = false;
|
is_out = false;
|
||||||
|
|
||||||
// Check in case we have the same point as maxima in previous evolution levels
|
const KeyPoint& kpts_par_ij = kpts_par_[i][j];
|
||||||
for (int ik = 0; ik < (int)kpts.size(); ik++) {
|
|
||||||
if (kpts[ik].class_id == level || kpts[ik].class_id == level + 1 || kpts[ik].class_id == level - 1) {
|
|
||||||
dist = pow(kpts_par_[i][j].pt.x - kpts[ik].pt.x, 2) + pow(kpts_par_[i][j].pt.y - kpts[ik].pt.y, 2);
|
|
||||||
|
|
||||||
if (dist < evolution_[level].sigma_size*evolution_[level].sigma_size) {
|
// Check in case we have the same point as maxima in previous evolution levels
|
||||||
if (kpts_par_[i][j].response > kpts[ik].response) {
|
for (int ik = 0; ik < (int)kpts.size(); ik++)
|
||||||
|
{
|
||||||
|
const KeyPoint& kpts_ik = kpts[ik];
|
||||||
|
if (kpts_ik.class_id == level || kpts_ik.class_id == level + 1 || kpts_ik.class_id == level - 1) {
|
||||||
|
Point2f diff = kpts_par_ij.pt - kpts_ik.pt;
|
||||||
|
float dist = diff.dot(diff);
|
||||||
|
|
||||||
|
if (dist < evolution_level.sigma_size*evolution_level.sigma_size) {
|
||||||
|
if (kpts_par_ij.response > kpts_ik.response) {
|
||||||
id_repeated = ik;
|
id_repeated = ik;
|
||||||
is_repeated = true;
|
is_repeated = true;
|
||||||
}
|
}
|
||||||
@@ -363,23 +370,23 @@ void KAZEFeatures::Determinant_Hessian(std::vector<KeyPoint>& kpts)
|
|||||||
|
|
||||||
if (is_extremum == true) {
|
if (is_extremum == true) {
|
||||||
// Check that the point is under the image limits for the descriptor computation
|
// Check that the point is under the image limits for the descriptor computation
|
||||||
left_x = cvRound(kpts_par_[i][j].pt.x - smax*kpts_par_[i][j].size);
|
left_x = cvRound(kpts_par_ij.pt.x - smax*kpts_par_ij.size);
|
||||||
right_x = cvRound(kpts_par_[i][j].pt.x + smax*kpts_par_[i][j].size);
|
right_x = cvRound(kpts_par_ij.pt.x + smax*kpts_par_ij.size);
|
||||||
up_y = cvRound(kpts_par_[i][j].pt.y - smax*kpts_par_[i][j].size);
|
up_y = cvRound(kpts_par_ij.pt.y - smax*kpts_par_ij.size);
|
||||||
down_y = cvRound(kpts_par_[i][j].pt.y + smax*kpts_par_[i][j].size);
|
down_y = cvRound(kpts_par_ij.pt.y + smax*kpts_par_ij.size);
|
||||||
|
|
||||||
if (left_x < 0 || right_x >= evolution_[level].Ldet.cols ||
|
if (left_x < 0 || right_x >= evolution_level.Ldet.cols ||
|
||||||
up_y < 0 || down_y >= evolution_[level].Ldet.rows) {
|
up_y < 0 || down_y >= evolution_level.Ldet.rows) {
|
||||||
is_out = true;
|
is_out = true;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (is_out == false) {
|
if (is_out == false) {
|
||||||
if (is_repeated == false) {
|
if (is_repeated == false) {
|
||||||
kpts.push_back(kpts_par_[i][j]);
|
kpts.push_back(kpts_par_ij);
|
||||||
npoints++;
|
npoints++;
|
||||||
}
|
}
|
||||||
else {
|
else {
|
||||||
kpts[id_repeated] = kpts_par_[i][j];
|
kpts[id_repeated] = kpts_par_ij;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -513,16 +520,16 @@ public:
|
|||||||
if (options_.upright)
|
if (options_.upright)
|
||||||
{
|
{
|
||||||
kpts[i].angle = 0.0;
|
kpts[i].angle = 0.0;
|
||||||
if (options_.extended)
|
if (options_.extended)
|
||||||
Get_KAZE_Upright_Descriptor_128(kpts[i], desc.ptr<float>((int)i));
|
Get_KAZE_Upright_Descriptor_128(kpts[i], desc.ptr<float>((int)i));
|
||||||
else
|
else
|
||||||
Get_KAZE_Upright_Descriptor_64(kpts[i], desc.ptr<float>((int)i));
|
Get_KAZE_Upright_Descriptor_64(kpts[i], desc.ptr<float>((int)i));
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
KAZEFeatures::Compute_Main_Orientation(kpts[i], evolution, options_);
|
KAZEFeatures::Compute_Main_Orientation(kpts[i], evolution, options_);
|
||||||
|
|
||||||
if (options_.extended)
|
if (options_.extended)
|
||||||
Get_KAZE_Descriptor_128(kpts[i], desc.ptr<float>((int)i));
|
Get_KAZE_Descriptor_128(kpts[i], desc.ptr<float>((int)i));
|
||||||
else
|
else
|
||||||
Get_KAZE_Descriptor_64(kpts[i], desc.ptr<float>((int)i));
|
Get_KAZE_Descriptor_64(kpts[i], desc.ptr<float>((int)i));
|
||||||
@@ -712,26 +719,26 @@ void KAZE_Descriptor_Invoker::Get_KAZE_Upright_Descriptor_64(const KeyPoint &kpt
|
|||||||
y1 = (int)(sample_y - 0.5f);
|
y1 = (int)(sample_y - 0.5f);
|
||||||
x1 = (int)(sample_x - 0.5f);
|
x1 = (int)(sample_x - 0.5f);
|
||||||
|
|
||||||
checkDescriptorLimits(x1, y1, options_.img_width, options_.img_height);
|
checkDescriptorLimits(x1, y1, options_.img_width, options_.img_height);
|
||||||
|
|
||||||
y2 = (int)(sample_y + 0.5f);
|
y2 = (int)(sample_y + 0.5f);
|
||||||
x2 = (int)(sample_x + 0.5f);
|
x2 = (int)(sample_x + 0.5f);
|
||||||
|
|
||||||
checkDescriptorLimits(x2, y2, options_.img_width, options_.img_height);
|
checkDescriptorLimits(x2, y2, options_.img_width, options_.img_height);
|
||||||
|
|
||||||
fx = sample_x - x1;
|
fx = sample_x - x1;
|
||||||
fy = sample_y - y1;
|
fy = sample_y - y1;
|
||||||
|
|
||||||
res1 = *(evolution[level].Lx.ptr<float>(y1)+x1);
|
res1 = *(evolution[level].Lx.ptr<float>(y1)+x1);
|
||||||
res2 = *(evolution[level].Lx.ptr<float>(y1)+x2);
|
res2 = *(evolution[level].Lx.ptr<float>(y1)+x2);
|
||||||
res3 = *(evolution[level].Lx.ptr<float>(y2)+x1);
|
res3 = *(evolution[level].Lx.ptr<float>(y2)+x1);
|
||||||
res4 = *(evolution[level].Lx.ptr<float>(y2)+x2);
|
res4 = *(evolution[level].Lx.ptr<float>(y2)+x2);
|
||||||
rx = (1.0f - fx)*(1.0f - fy)*res1 + fx*(1.0f - fy)*res2 + (1.0f - fx)*fy*res3 + fx*fy*res4;
|
rx = (1.0f - fx)*(1.0f - fy)*res1 + fx*(1.0f - fy)*res2 + (1.0f - fx)*fy*res3 + fx*fy*res4;
|
||||||
|
|
||||||
res1 = *(evolution[level].Ly.ptr<float>(y1)+x1);
|
res1 = *(evolution[level].Ly.ptr<float>(y1)+x1);
|
||||||
res2 = *(evolution[level].Ly.ptr<float>(y1)+x2);
|
res2 = *(evolution[level].Ly.ptr<float>(y1)+x2);
|
||||||
res3 = *(evolution[level].Ly.ptr<float>(y2)+x1);
|
res3 = *(evolution[level].Ly.ptr<float>(y2)+x1);
|
||||||
res4 = *(evolution[level].Ly.ptr<float>(y2)+x2);
|
res4 = *(evolution[level].Ly.ptr<float>(y2)+x2);
|
||||||
ry = (1.0f - fx)*(1.0f - fy)*res1 + fx*(1.0f - fy)*res2 + (1.0f - fx)*fy*res3 + fx*fy*res4;
|
ry = (1.0f - fx)*(1.0f - fy)*res1 + fx*(1.0f - fy)*res2 + (1.0f - fx)*fy*res3 + fx*fy*res4;
|
||||||
|
|
||||||
rx = gauss_s1*rx;
|
rx = gauss_s1*rx;
|
||||||
|
|||||||
@@ -131,12 +131,17 @@ static void
|
|||||||
HarrisResponses(const Mat& img, const std::vector<Rect>& layerinfo,
|
HarrisResponses(const Mat& img, const std::vector<Rect>& layerinfo,
|
||||||
std::vector<KeyPoint>& pts, int blockSize, float harris_k)
|
std::vector<KeyPoint>& pts, int blockSize, float harris_k)
|
||||||
{
|
{
|
||||||
CV_Assert( img.type() == CV_8UC1 && blockSize*blockSize <= 2048 );
|
CV_CheckTypeEQ(img.type(), CV_8UC1, "");
|
||||||
|
CV_CheckGT(blockSize, 0, "");
|
||||||
|
CV_CheckLE(blockSize*blockSize, 2048, "");
|
||||||
|
|
||||||
size_t ptidx, ptsize = pts.size();
|
size_t ptidx, ptsize = pts.size();
|
||||||
|
|
||||||
const uchar* ptr00 = img.ptr<uchar>();
|
const uchar* ptr00 = img.ptr<uchar>();
|
||||||
int step = (int)(img.step/img.elemSize1());
|
size_t size_t_step = img.step;
|
||||||
|
CV_CheckLE(size_t_step * blockSize + blockSize + 1, (size_t)INT_MAX, ""); // ofs computation, step+1
|
||||||
|
int step = static_cast<int>(size_t_step);
|
||||||
|
|
||||||
int r = blockSize/2;
|
int r = blockSize/2;
|
||||||
|
|
||||||
float scale = 1.f/((1 << 2) * blockSize * 255.f);
|
float scale = 1.f/((1 << 2) * blockSize * 255.f);
|
||||||
@@ -154,7 +159,7 @@ HarrisResponses(const Mat& img, const std::vector<Rect>& layerinfo,
|
|||||||
int y0 = cvRound(pts[ptidx].pt.y);
|
int y0 = cvRound(pts[ptidx].pt.y);
|
||||||
int z = pts[ptidx].octave;
|
int z = pts[ptidx].octave;
|
||||||
|
|
||||||
const uchar* ptr0 = ptr00 + (y0 - r + layerinfo[z].y)*step + x0 - r + layerinfo[z].x;
|
const uchar* ptr0 = ptr00 + (y0 - r + layerinfo[z].y)*size_t_step + (x0 - r + layerinfo[z].x);
|
||||||
int a = 0, b = 0, c = 0;
|
int a = 0, b = 0, c = 0;
|
||||||
|
|
||||||
for( int k = 0; k < blockSize*blockSize; k++ )
|
for( int k = 0; k < blockSize*blockSize; k++ )
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ TEST(Features2d_BlobDetector, bug_6667)
|
|||||||
SimpleBlobDetector::Params params;
|
SimpleBlobDetector::Params params;
|
||||||
params.minThreshold = 250;
|
params.minThreshold = 250;
|
||||||
params.maxThreshold = 260;
|
params.maxThreshold = 260;
|
||||||
|
params.minRepeatability = 1; // https://github.com/opencv/opencv/issues/6667
|
||||||
std::vector<KeyPoint> keypoints;
|
std::vector<KeyPoint> keypoints;
|
||||||
|
|
||||||
Ptr<SimpleBlobDetector> detector = SimpleBlobDetector::create(params);
|
Ptr<SimpleBlobDetector> detector = SimpleBlobDetector::create(params);
|
||||||
|
|||||||
@@ -141,5 +141,31 @@ TEST(Features2D_ORB, regression_16197)
|
|||||||
ASSERT_NO_THROW(orbPtr->detectAndCompute(img, noArray(), kps, fv));
|
ASSERT_NO_THROW(orbPtr->detectAndCompute(img, noArray(), kps, fv));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// https://github.com/opencv/opencv-python/issues/537
|
||||||
|
BIGDATA_TEST(Features2D_ORB, regression_opencv_python_537) // memory usage: ~3 Gb
|
||||||
|
{
|
||||||
|
applyTestTag(
|
||||||
|
CV_TEST_TAG_LONG,
|
||||||
|
CV_TEST_TAG_DEBUG_VERYLONG,
|
||||||
|
CV_TEST_TAG_MEMORY_6GB
|
||||||
|
);
|
||||||
|
|
||||||
|
const int width = 25000;
|
||||||
|
const int height = 25000;
|
||||||
|
Mat img(Size(width, height), CV_8UC1, Scalar::all(0));
|
||||||
|
|
||||||
|
const int border = 23, num_lines = 23;
|
||||||
|
for (int i = 0; i < num_lines; i++)
|
||||||
|
{
|
||||||
|
cv::Point2i point1(border + i * 100, border + i * 100);
|
||||||
|
cv::Point2i point2(width - border - i * 100, height - border * i * 100);
|
||||||
|
cv::line(img, point1, point2, 255, 1, LINE_AA);
|
||||||
|
}
|
||||||
|
|
||||||
|
Ptr<ORB> orbPtr = ORB::create(31);
|
||||||
|
std::vector<KeyPoint> kps;
|
||||||
|
Mat fv;
|
||||||
|
ASSERT_NO_THROW(orbPtr->detectAndCompute(img, noArray(), kps, fv));
|
||||||
|
}
|
||||||
|
|
||||||
}} // namespace
|
}} // namespace
|
||||||
|
|||||||
@@ -41,46 +41,55 @@ file(GLOB highgui_ext_hdrs
|
|||||||
# Removing WinRT API headers by default
|
# Removing WinRT API headers by default
|
||||||
list(REMOVE_ITEM highgui_ext_hdrs "${CMAKE_CURRENT_LIST_DIR}/include/opencv2/${name}/highgui_winrt.hpp")
|
list(REMOVE_ITEM highgui_ext_hdrs "${CMAKE_CURRENT_LIST_DIR}/include/opencv2/${name}/highgui_winrt.hpp")
|
||||||
|
|
||||||
if(HAVE_QT5)
|
if(HAVE_QT)
|
||||||
# "Automoc" doesn't work properly with opencv_world build, use QT5_WRAP_CPP() directly
|
if(QT_VERSION_MAJOR GREATER 4)
|
||||||
#set(CMAKE_AUTOMOC ON)
|
# "Automoc" doesn't work properly with opencv_world build, use QT<ver>_WRAP_CPP() directly
|
||||||
|
#set(CMAKE_AUTOMOC ON)
|
||||||
|
|
||||||
set(CMAKE_INCLUDE_CURRENT_DIR ON)
|
set(CMAKE_INCLUDE_CURRENT_DIR ON)
|
||||||
|
|
||||||
QT5_ADD_RESOURCES(_RCC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.qrc)
|
if(QT_VERSION_MAJOR EQUAL 6)
|
||||||
QT5_WRAP_CPP(_MOC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.h)
|
QT6_ADD_RESOURCES(_RCC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.qrc)
|
||||||
list(APPEND highgui_srcs
|
QT6_WRAP_CPP(_MOC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.h)
|
||||||
${CMAKE_CURRENT_LIST_DIR}/src/window_QT.cpp
|
elseif(QT_VERSION_MAJOR EQUAL 5)
|
||||||
${CMAKE_CURRENT_LIST_DIR}/src/window_QT.h
|
QT5_ADD_RESOURCES(_RCC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.qrc)
|
||||||
${_MOC_OUTFILES}
|
QT5_WRAP_CPP(_MOC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.h)
|
||||||
${_RCC_OUTFILES})
|
else()
|
||||||
|
message(FATAL_ERROR "Unsuported QT version: ${QT_VERSION_MAJOR}")
|
||||||
|
endif()
|
||||||
|
|
||||||
foreach(dt5_dep Core Gui Widgets Test Concurrent)
|
list(APPEND highgui_srcs
|
||||||
add_definitions(${Qt5${dt5_dep}_DEFINITIONS})
|
${CMAKE_CURRENT_LIST_DIR}/src/window_QT.cpp
|
||||||
include_directories(${Qt5${dt5_dep}_INCLUDE_DIRS})
|
${CMAKE_CURRENT_LIST_DIR}/src/window_QT.h
|
||||||
list(APPEND HIGHGUI_LIBRARIES ${Qt5${dt5_dep}_LIBRARIES})
|
${_MOC_OUTFILES}
|
||||||
endforeach()
|
${_RCC_OUTFILES})
|
||||||
|
|
||||||
if(HAVE_QT_OPENGL)
|
set(qt_deps Core Gui Widgets Test Concurrent)
|
||||||
add_definitions(${Qt5OpenGL_DEFINITIONS})
|
if(HAVE_QT_OPENGL)
|
||||||
include_directories(${Qt5OpenGL_INCLUDE_DIRS})
|
list(APPEND qt_deps OpenGL)
|
||||||
list(APPEND HIGHGUI_LIBRARIES ${Qt5OpenGL_LIBRARIES})
|
endif()
|
||||||
endif()
|
|
||||||
|
|
||||||
elseif(HAVE_QT)
|
foreach(dt_dep ${qt_deps})
|
||||||
if (HAVE_QT_OPENGL)
|
add_definitions(${Qt${QT_VERSION_MAJOR}${dt_dep}_DEFINITIONS})
|
||||||
set(QT_USE_QTOPENGL TRUE)
|
include_directories(${Qt${QT_VERSION_MAJOR}${dt_dep}_INCLUDE_DIRS})
|
||||||
endif()
|
list(APPEND HIGHGUI_LIBRARIES ${Qt${QT_VERSION_MAJOR}${dt_dep}_LIBRARIES})
|
||||||
include(${QT_USE_FILE})
|
endforeach()
|
||||||
|
else()
|
||||||
|
ocv_assert(QT_VERSION_MAJOR EQUAL 4)
|
||||||
|
if (HAVE_QT_OPENGL)
|
||||||
|
set(QT_USE_QTOPENGL TRUE)
|
||||||
|
endif()
|
||||||
|
include(${QT_USE_FILE})
|
||||||
|
|
||||||
QT4_ADD_RESOURCES(_RCC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.qrc)
|
QT4_ADD_RESOURCES(_RCC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.qrc)
|
||||||
QT4_WRAP_CPP(_MOC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.h)
|
QT4_WRAP_CPP(_MOC_OUTFILES ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.h)
|
||||||
|
|
||||||
list(APPEND HIGHGUI_LIBRARIES ${QT_LIBRARIES})
|
list(APPEND HIGHGUI_LIBRARIES ${QT_LIBRARIES})
|
||||||
list(APPEND highgui_srcs ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.cpp ${_MOC_OUTFILES} ${_RCC_OUTFILES})
|
list(APPEND highgui_srcs ${CMAKE_CURRENT_LIST_DIR}/src/window_QT.cpp ${_MOC_OUTFILES} ${_RCC_OUTFILES})
|
||||||
ocv_check_flag_support(CXX -Wno-missing-declarations _have_flag "")
|
ocv_check_flag_support(CXX -Wno-missing-declarations _have_flag "")
|
||||||
if(${_have_flag})
|
if(${_have_flag})
|
||||||
set_source_files_properties(${_RCC_OUTFILES} PROPERTIES COMPILE_FLAGS -Wno-missing-declarations)
|
set_source_files_properties(${_RCC_OUTFILES} PROPERTIES COMPILE_FLAGS -Wno-missing-declarations)
|
||||||
|
endif()
|
||||||
endif()
|
endif()
|
||||||
elseif(WINRT)
|
elseif(WINRT)
|
||||||
if(NOT WINRT_8_0)
|
if(NOT WINRT_8_0)
|
||||||
|
|||||||
@@ -64,6 +64,38 @@
|
|||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
|
||||||
|
#if QT_VERSION >= QT_VERSION_CHECK(5, 15, 0)
|
||||||
|
#define Qt_MiddleButton Qt::MiddleButton
|
||||||
|
inline Qt::Orientation wheelEventOrientation(QWheelEvent *we) {
|
||||||
|
if (std::abs(we->angleDelta().x()) < std::abs(we->angleDelta().y()))
|
||||||
|
return Qt::Vertical;
|
||||||
|
else
|
||||||
|
return Qt::Horizontal;
|
||||||
|
}
|
||||||
|
inline int wheelEventDelta(QWheelEvent *we) {
|
||||||
|
if(wheelEventOrientation(we) == Qt::Vertical)
|
||||||
|
return we->angleDelta().y();
|
||||||
|
else
|
||||||
|
return we->angleDelta().x();
|
||||||
|
}
|
||||||
|
inline QPoint wheelEventPos(QWheelEvent *we) {
|
||||||
|
return we->position().toPoint();
|
||||||
|
}
|
||||||
|
#else
|
||||||
|
#define Qt_MiddleButton Qt::MidButton
|
||||||
|
inline Qt::Orientation wheelEventOrientation(QWheelEvent *we) {
|
||||||
|
return we->orientation();
|
||||||
|
}
|
||||||
|
inline int wheelEventDelta(QWheelEvent *we) {
|
||||||
|
return we->delta();
|
||||||
|
}
|
||||||
|
inline QPoint wheelEventPos(QWheelEvent *we) {
|
||||||
|
return we->pos();
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
//Static and global first
|
//Static and global first
|
||||||
static GuiReceiver *guiMainThread = NULL;
|
static GuiReceiver *guiMainThread = NULL;
|
||||||
static int parameterSystemC = 1;
|
static int parameterSystemC = 1;
|
||||||
@@ -1579,7 +1611,9 @@ CvWinProperties::CvWinProperties(QString name_paraWindow, QObject* /*parent*/)
|
|||||||
myLayout->setObjectName(QString::fromUtf8("boxLayout"));
|
myLayout->setObjectName(QString::fromUtf8("boxLayout"));
|
||||||
myLayout->setContentsMargins(0, 0, 0, 0);
|
myLayout->setContentsMargins(0, 0, 0, 0);
|
||||||
myLayout->setSpacing(0);
|
myLayout->setSpacing(0);
|
||||||
|
#if QT_VERSION < QT_VERSION_CHECK(5, 13, 0)
|
||||||
myLayout->setMargin(0);
|
myLayout->setMargin(0);
|
||||||
|
#endif
|
||||||
myLayout->setSizeConstraint(QLayout::SetFixedSize);
|
myLayout->setSizeConstraint(QLayout::SetFixedSize);
|
||||||
setLayout(myLayout);
|
setLayout(myLayout);
|
||||||
|
|
||||||
@@ -1957,7 +1991,9 @@ void CvWindow::createBarLayout()
|
|||||||
myBarLayout->setObjectName(QString::fromUtf8("barLayout"));
|
myBarLayout->setObjectName(QString::fromUtf8("barLayout"));
|
||||||
myBarLayout->setContentsMargins(0, 0, 0, 0);
|
myBarLayout->setContentsMargins(0, 0, 0, 0);
|
||||||
myBarLayout->setSpacing(0);
|
myBarLayout->setSpacing(0);
|
||||||
|
#if QT_VERSION < QT_VERSION_CHECK(5, 13, 0)
|
||||||
myBarLayout->setMargin(0);
|
myBarLayout->setMargin(0);
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -1967,7 +2003,9 @@ void CvWindow::createGlobalLayout()
|
|||||||
myGlobalLayout->setObjectName(QString::fromUtf8("boxLayout"));
|
myGlobalLayout->setObjectName(QString::fromUtf8("boxLayout"));
|
||||||
myGlobalLayout->setContentsMargins(0, 0, 0, 0);
|
myGlobalLayout->setContentsMargins(0, 0, 0, 0);
|
||||||
myGlobalLayout->setSpacing(0);
|
myGlobalLayout->setSpacing(0);
|
||||||
|
#if QT_VERSION < QT_VERSION_CHECK(5, 13, 0)
|
||||||
myGlobalLayout->setMargin(0);
|
myGlobalLayout->setMargin(0);
|
||||||
|
#endif
|
||||||
setMinimumSize(1, 1);
|
setMinimumSize(1, 1);
|
||||||
|
|
||||||
if (param_flags == CV_WINDOW_AUTOSIZE)
|
if (param_flags == CV_WINDOW_AUTOSIZE)
|
||||||
@@ -2205,7 +2243,7 @@ void CvWindow::icvLoadControlPanel()
|
|||||||
}
|
}
|
||||||
if (t->type == type_CvButtonbar)
|
if (t->type == type_CvButtonbar)
|
||||||
{
|
{
|
||||||
int subsize = settings.beginReadArray(QString("buttonbar")+i);
|
int subsize = settings.beginReadArray(QString("buttonbar%1").arg(i));
|
||||||
|
|
||||||
if ( subsize == ((CvButtonbar*)t)->layout()->count() )
|
if ( subsize == ((CvButtonbar*)t)->layout()->count() )
|
||||||
icvLoadButtonbar((CvButtonbar*)t,&settings);
|
icvLoadButtonbar((CvButtonbar*)t,&settings);
|
||||||
@@ -2236,7 +2274,7 @@ void CvWindow::icvSaveControlPanel()
|
|||||||
}
|
}
|
||||||
if (t->type == type_CvButtonbar)
|
if (t->type == type_CvButtonbar)
|
||||||
{
|
{
|
||||||
settings.beginWriteArray(QString("buttonbar")+i);
|
settings.beginWriteArray(QString("buttonbar%1").arg(i));
|
||||||
icvSaveButtonbar((CvButtonbar*)t,&settings);
|
icvSaveButtonbar((CvButtonbar*)t,&settings);
|
||||||
settings.endArray();
|
settings.endArray();
|
||||||
}
|
}
|
||||||
@@ -2396,14 +2434,14 @@ void OCVViewPort::icvmouseHandler(QMouseEvent* evnt, type_mouse_event category,
|
|||||||
flags |= CV_EVENT_FLAG_LBUTTON;
|
flags |= CV_EVENT_FLAG_LBUTTON;
|
||||||
if(buttons & Qt::RightButton)
|
if(buttons & Qt::RightButton)
|
||||||
flags |= CV_EVENT_FLAG_RBUTTON;
|
flags |= CV_EVENT_FLAG_RBUTTON;
|
||||||
if(buttons & Qt::MidButton)
|
if(buttons & Qt_MiddleButton)
|
||||||
flags |= CV_EVENT_FLAG_MBUTTON;
|
flags |= CV_EVENT_FLAG_MBUTTON;
|
||||||
|
|
||||||
if (cv_event == -1) {
|
if (cv_event == -1) {
|
||||||
if (category == mouse_wheel) {
|
if (category == mouse_wheel) {
|
||||||
QWheelEvent *we = (QWheelEvent *) evnt;
|
QWheelEvent *we = (QWheelEvent *) evnt;
|
||||||
cv_event = ((we->orientation() == Qt::Vertical) ? CV_EVENT_MOUSEWHEEL : CV_EVENT_MOUSEHWHEEL);
|
cv_event = ((wheelEventOrientation(we) == Qt::Vertical) ? CV_EVENT_MOUSEWHEEL : CV_EVENT_MOUSEHWHEEL);
|
||||||
flags |= (we->delta() & 0xffff)<<16;
|
flags |= (wheelEventDelta(we) & 0xffff)<<16;
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
switch(evnt->button())
|
switch(evnt->button())
|
||||||
@@ -2416,7 +2454,7 @@ void OCVViewPort::icvmouseHandler(QMouseEvent* evnt, type_mouse_event category,
|
|||||||
cv_event = tableMouseButtons[category][1];
|
cv_event = tableMouseButtons[category][1];
|
||||||
flags |= CV_EVENT_FLAG_RBUTTON;
|
flags |= CV_EVENT_FLAG_RBUTTON;
|
||||||
break;
|
break;
|
||||||
case Qt::MidButton:
|
case Qt_MiddleButton:
|
||||||
cv_event = tableMouseButtons[category][2];
|
cv_event = tableMouseButtons[category][2];
|
||||||
flags |= CV_EVENT_FLAG_MBUTTON;
|
flags |= CV_EVENT_FLAG_MBUTTON;
|
||||||
break;
|
break;
|
||||||
@@ -2772,7 +2810,7 @@ void DefaultViewPort::wheelEvent(QWheelEvent* evnt)
|
|||||||
{
|
{
|
||||||
icvmouseEvent((QMouseEvent *)evnt, mouse_wheel);
|
icvmouseEvent((QMouseEvent *)evnt, mouse_wheel);
|
||||||
|
|
||||||
scaleView(evnt->delta() / 240.0, evnt->pos());
|
scaleView(wheelEventDelta(evnt) / 240.0, wheelEventPos(evnt));
|
||||||
viewport()->update();
|
viewport()->update();
|
||||||
|
|
||||||
QWidget::wheelEvent(evnt);
|
QWidget::wheelEvent(evnt);
|
||||||
@@ -2883,18 +2921,19 @@ inline bool DefaultViewPort::isSameSize(IplImage* img1, IplImage* img2)
|
|||||||
void DefaultViewPort::controlImagePosition()
|
void DefaultViewPort::controlImagePosition()
|
||||||
{
|
{
|
||||||
qreal left, top, right, bottom;
|
qreal left, top, right, bottom;
|
||||||
|
qreal factor = 1.0 / param_matrixWorld.m11();
|
||||||
|
|
||||||
//after check top-left, bottom right corner to avoid getting "out" during zoom/panning
|
//after check top-left, bottom right corner to avoid getting "out" during zoom/panning
|
||||||
param_matrixWorld.map(0,0,&left,&top);
|
param_matrixWorld.map(0,0,&left,&top);
|
||||||
|
|
||||||
if (left > 0)
|
if (left > 0)
|
||||||
{
|
{
|
||||||
param_matrixWorld.translate(-left,0);
|
param_matrixWorld.translate(-left * factor, 0);
|
||||||
left = 0;
|
left = 0;
|
||||||
}
|
}
|
||||||
if (top > 0)
|
if (top > 0)
|
||||||
{
|
{
|
||||||
param_matrixWorld.translate(0,-top);
|
param_matrixWorld.translate(0, -top * factor);
|
||||||
top = 0;
|
top = 0;
|
||||||
}
|
}
|
||||||
//-------
|
//-------
|
||||||
@@ -2903,12 +2942,12 @@ void DefaultViewPort::controlImagePosition()
|
|||||||
param_matrixWorld.map(sizeImage.width(),sizeImage.height(),&right,&bottom);
|
param_matrixWorld.map(sizeImage.width(),sizeImage.height(),&right,&bottom);
|
||||||
if (right < sizeImage.width())
|
if (right < sizeImage.width())
|
||||||
{
|
{
|
||||||
param_matrixWorld.translate(sizeImage.width()-right,0);
|
param_matrixWorld.translate((sizeImage.width() - right) * factor, 0);
|
||||||
right = sizeImage.width();
|
right = sizeImage.width();
|
||||||
}
|
}
|
||||||
if (bottom < sizeImage.height())
|
if (bottom < sizeImage.height())
|
||||||
{
|
{
|
||||||
param_matrixWorld.translate(0,sizeImage.height()-bottom);
|
param_matrixWorld.translate(0, (sizeImage.height() - bottom) * factor);
|
||||||
bottom = sizeImage.height();
|
bottom = sizeImage.height();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1220,12 +1220,17 @@ protected:
|
|||||||
//! @addtogroup imgproc_feature
|
//! @addtogroup imgproc_feature
|
||||||
//! @{
|
//! @{
|
||||||
|
|
||||||
|
/** @example samples/cpp/lsd_lines.cpp
|
||||||
|
An example using the LineSegmentDetector
|
||||||
|
\image html building_lsd.png "Sample output image" width=434 height=300
|
||||||
|
*/
|
||||||
|
|
||||||
/** @brief Line segment detector class
|
/** @brief Line segment detector class
|
||||||
|
|
||||||
following the algorithm described at @cite Rafael12 .
|
following the algorithm described at @cite Rafael12 .
|
||||||
|
|
||||||
@note Implementation has been removed due original code license conflict
|
@note Implementation has been removed from OpenCV version 3.4.6 to 3.4.15 and version 4.1.0 to 4.5.3 due original code license conflict.
|
||||||
|
restored again after [Computation of a NFA](https://github.com/rafael-grompone-von-gioi/binomial_nfa) code published under the MIT license.
|
||||||
*/
|
*/
|
||||||
class CV_EXPORTS_W LineSegmentDetector : public Algorithm
|
class CV_EXPORTS_W LineSegmentDetector : public Algorithm
|
||||||
{
|
{
|
||||||
@@ -1239,8 +1244,8 @@ public:
|
|||||||
|
|
||||||
@param image A grayscale (CV_8UC1) input image. If only a roi needs to be selected, use:
|
@param image A grayscale (CV_8UC1) input image. If only a roi needs to be selected, use:
|
||||||
`lsd_ptr-\>detect(image(roi), lines, ...); lines += Scalar(roi.x, roi.y, roi.x, roi.y);`
|
`lsd_ptr-\>detect(image(roi), lines, ...); lines += Scalar(roi.x, roi.y, roi.x, roi.y);`
|
||||||
@param lines A vector of Vec4i or Vec4f elements specifying the beginning and ending point of a line. Where
|
@param lines A vector of Vec4f elements specifying the beginning and ending point of a line. Where
|
||||||
Vec4i/Vec4f is (x1, y1, x2, y2), point 1 is the start, point 2 - end. Returned lines are strictly
|
Vec4f is (x1, y1, x2, y2), point 1 is the start, point 2 - end. Returned lines are strictly
|
||||||
oriented depending on the gradient.
|
oriented depending on the gradient.
|
||||||
@param width Vector of widths of the regions, where the lines are found. E.g. Width of line.
|
@param width Vector of widths of the regions, where the lines are found. E.g. Width of line.
|
||||||
@param prec Vector of precisions with which the lines are found.
|
@param prec Vector of precisions with which the lines are found.
|
||||||
@@ -1288,8 +1293,6 @@ to edit those, as to tailor it for their own application.
|
|||||||
@param log_eps Detection threshold: -log10(NFA) \> log_eps. Used only when advance refinement is chosen.
|
@param log_eps Detection threshold: -log10(NFA) \> log_eps. Used only when advance refinement is chosen.
|
||||||
@param density_th Minimal density of aligned region points in the enclosing rectangle.
|
@param density_th Minimal density of aligned region points in the enclosing rectangle.
|
||||||
@param n_bins Number of bins in pseudo-ordering of gradient modulus.
|
@param n_bins Number of bins in pseudo-ordering of gradient modulus.
|
||||||
|
|
||||||
@note Implementation has been removed due original code license conflict
|
|
||||||
*/
|
*/
|
||||||
CV_EXPORTS_W Ptr<LineSegmentDetector> createLineSegmentDetector(
|
CV_EXPORTS_W Ptr<LineSegmentDetector> createLineSegmentDetector(
|
||||||
int refine = LSD_REFINE_STD, double scale = 0.8,
|
int refine = LSD_REFINE_STD, double scale = 0.8,
|
||||||
@@ -2223,7 +2226,7 @@ enlarge an image, it will generally look best with c#INTER_CUBIC (slow) or #INTE
|
|||||||
@param src input image.
|
@param src input image.
|
||||||
@param dst output image; it has the size dsize (when it is non-zero) or the size computed from
|
@param dst output image; it has the size dsize (when it is non-zero) or the size computed from
|
||||||
src.size(), fx, and fy; the type of dst is the same as of src.
|
src.size(), fx, and fy; the type of dst is the same as of src.
|
||||||
@param dsize output image size; if it equals zero, it is computed as:
|
@param dsize output image size; if it equals zero (`None` in Python), it is computed as:
|
||||||
\f[\texttt{dsize = Size(round(fx*src.cols), round(fy*src.rows))}\f]
|
\f[\texttt{dsize = Size(round(fx*src.cols), round(fy*src.rows))}\f]
|
||||||
Either dsize or both fx and fy must be non-zero.
|
Either dsize or both fx and fy must be non-zero.
|
||||||
@param fx scale factor along the horizontal axis; when it equals 0, it is computed as
|
@param fx scale factor along the horizontal axis; when it equals 0, it is computed as
|
||||||
@@ -3951,6 +3954,7 @@ hierarchy[i][0] , hierarchy[i][1] , hierarchy[i][2] , and hierarchy[i][3] are se
|
|||||||
in contours of the next and previous contours at the same hierarchical level, the first child
|
in contours of the next and previous contours at the same hierarchical level, the first child
|
||||||
contour and the parent contour, respectively. If for the contour i there are no next, previous,
|
contour and the parent contour, respectively. If for the contour i there are no next, previous,
|
||||||
parent, or nested contours, the corresponding elements of hierarchy[i] will be negative.
|
parent, or nested contours, the corresponding elements of hierarchy[i] will be negative.
|
||||||
|
@note In Python, hierarchy is nested inside a top level array. Use hierarchy[0][i] to access hierarchical elements of i-th contour.
|
||||||
@param mode Contour retrieval mode, see #RetrievalModes
|
@param mode Contour retrieval mode, see #RetrievalModes
|
||||||
@param method Contour approximation method, see #ContourApproximationModes
|
@param method Contour approximation method, see #ContourApproximationModes
|
||||||
@param offset Optional offset by which every contour point is shifted. This is useful if the
|
@param offset Optional offset by which every contour point is shifted. This is useful if the
|
||||||
|
|||||||
@@ -198,6 +198,11 @@ CV_EXPORTS void cvtTwoPlaneYUVtoBGR(const uchar * y_data, const uchar * uv_data,
|
|||||||
int dst_width, int dst_height,
|
int dst_width, int dst_height,
|
||||||
int dcn, bool swapBlue, int uIdx);
|
int dcn, bool swapBlue, int uIdx);
|
||||||
|
|
||||||
|
CV_EXPORTS void cvtTwoPlaneYUVtoBGR(const uchar * y_data, size_t y_step, const uchar * uv_data, size_t uv_step,
|
||||||
|
uchar * dst_data, size_t dst_step,
|
||||||
|
int dst_width, int dst_height,
|
||||||
|
int dcn, bool swapBlue, int uIdx);
|
||||||
|
|
||||||
CV_EXPORTS void cvtThreePlaneYUVtoBGR(const uchar * src_data, size_t src_step,
|
CV_EXPORTS void cvtThreePlaneYUVtoBGR(const uchar * src_data, size_t src_step,
|
||||||
uchar * dst_data, size_t dst_step,
|
uchar * dst_data, size_t dst_step,
|
||||||
int dst_width, int dst_height,
|
int dst_width, int dst_height,
|
||||||
|
|||||||
@@ -124,8 +124,9 @@ void cvtTwoPlaneYUVtoBGR(const uchar * src_data, size_t src_step,
|
|||||||
|
|
||||||
CALL_HAL(cvtTwoPlaneYUVtoBGR, cv_hal_cvtTwoPlaneYUVtoBGR, src_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx);
|
CALL_HAL(cvtTwoPlaneYUVtoBGR, cv_hal_cvtTwoPlaneYUVtoBGR, src_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx);
|
||||||
|
|
||||||
CV_CPU_DISPATCH(cvtTwoPlaneYUVtoBGR, (src_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx),
|
cvtTwoPlaneYUVtoBGR(
|
||||||
CV_CPU_DISPATCH_MODES_ALL);
|
src_data, src_step, src_data + src_step * dst_height, src_step, dst_data, dst_step,
|
||||||
|
dst_width, dst_height, dcn, swapBlue, uIdx);
|
||||||
}
|
}
|
||||||
|
|
||||||
void cvtTwoPlaneYUVtoBGR(const uchar * y_data, const uchar * uv_data, size_t src_step,
|
void cvtTwoPlaneYUVtoBGR(const uchar * y_data, const uchar * uv_data, size_t src_step,
|
||||||
@@ -135,7 +136,20 @@ void cvtTwoPlaneYUVtoBGR(const uchar * y_data, const uchar * uv_data, size_t src
|
|||||||
{
|
{
|
||||||
CV_INSTRUMENT_REGION();
|
CV_INSTRUMENT_REGION();
|
||||||
|
|
||||||
CV_CPU_DISPATCH(cvtTwoPlaneYUVtoBGR, (y_data, uv_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx),
|
cvtTwoPlaneYUVtoBGR(y_data, src_step, uv_data, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx);
|
||||||
|
}
|
||||||
|
|
||||||
|
void cvtTwoPlaneYUVtoBGR(const uchar * y_data, size_t y_step, const uchar * uv_data, size_t uv_step,
|
||||||
|
uchar * dst_data, size_t dst_step,
|
||||||
|
int dst_width, int dst_height,
|
||||||
|
int dcn, bool swapBlue, int uIdx)
|
||||||
|
{
|
||||||
|
CV_INSTRUMENT_REGION();
|
||||||
|
|
||||||
|
CALL_HAL(cvtTwoPlaneYUVtoBGREx, cv_hal_cvtTwoPlaneYUVtoBGREx,
|
||||||
|
y_data, y_step, uv_data, uv_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx);
|
||||||
|
|
||||||
|
CV_CPU_DISPATCH(cvtTwoPlaneYUVtoBGR, (y_data, y_step, uv_data, uv_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx),
|
||||||
CV_CPU_DISPATCH_MODES_ALL);
|
CV_CPU_DISPATCH_MODES_ALL);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -172,7 +186,8 @@ void cvtBGRtoTwoPlaneYUV(const uchar * src_data, size_t src_step,
|
|||||||
{
|
{
|
||||||
CV_INSTRUMENT_REGION();
|
CV_INSTRUMENT_REGION();
|
||||||
|
|
||||||
// TODO: add hal replacement method
|
CALL_HAL(cvtBGRtoTwoPlaneYUV, cv_hal_cvtBGRtoTwoPlaneYUV,
|
||||||
|
src_data, src_step, y_data, dst_step, uv_data, dst_step, width, height, scn, swapBlue, uIdx);
|
||||||
|
|
||||||
CV_CPU_DISPATCH(cvtBGRtoTwoPlaneYUV, (src_data, src_step, y_data, uv_data, dst_step, width, height, scn, swapBlue, uIdx),
|
CV_CPU_DISPATCH(cvtBGRtoTwoPlaneYUV, (src_data, src_step, y_data, uv_data, dst_step, width, height, scn, swapBlue, uIdx),
|
||||||
CV_CPU_DISPATCH_MODES_ALL);
|
CV_CPU_DISPATCH_MODES_ALL);
|
||||||
@@ -406,14 +421,21 @@ void cvtColorTwoPlaneYUV2BGRpair( InputArray _ysrc, InputArray _uvsrc, OutputArr
|
|||||||
|
|
||||||
Mat ysrc = _ysrc.getMat(), uvsrc = _uvsrc.getMat();
|
Mat ysrc = _ysrc.getMat(), uvsrc = _uvsrc.getMat();
|
||||||
|
|
||||||
CV_CheckEQ(ysrc.step, uvsrc.step, "");
|
|
||||||
|
|
||||||
_dst.create( ysz, CV_MAKETYPE(depth, dcn));
|
_dst.create( ysz, CV_MAKETYPE(depth, dcn));
|
||||||
Mat dst = _dst.getMat();
|
Mat dst = _dst.getMat();
|
||||||
|
|
||||||
hal::cvtTwoPlaneYUVtoBGR(ysrc.data, uvsrc.data, ysrc.step,
|
if(ysrc.step == uvsrc.step)
|
||||||
dst.data, dst.step, dst.cols, dst.rows,
|
{
|
||||||
dcn, swapb, uidx);
|
hal::cvtTwoPlaneYUVtoBGR(ysrc.data, uvsrc.data, ysrc.step,
|
||||||
|
dst.data, dst.step, dst.cols, dst.rows,
|
||||||
|
dcn, swapb, uidx);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
hal::cvtTwoPlaneYUVtoBGR(ysrc.data, ysrc.step, uvsrc.data, uvsrc.step,
|
||||||
|
dst.data, dst.step, dst.cols, dst.rows,
|
||||||
|
dcn, swapb, uidx);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
} // namespace cv
|
} // namespace cv
|
||||||
|
|||||||
@@ -17,11 +17,7 @@ void cvtYUVtoBGR(const uchar * src_data, size_t src_step,
|
|||||||
uchar * dst_data, size_t dst_step,
|
uchar * dst_data, size_t dst_step,
|
||||||
int width, int height,
|
int width, int height,
|
||||||
int depth, int dcn, bool swapBlue, bool isCbCr);
|
int depth, int dcn, bool swapBlue, bool isCbCr);
|
||||||
void cvtTwoPlaneYUVtoBGR(const uchar * src_data, size_t src_step,
|
void cvtTwoPlaneYUVtoBGR(const uchar * y_data, size_t y_step, const uchar * uv_data, size_t uv_step,
|
||||||
uchar * dst_data, size_t dst_step,
|
|
||||||
int dst_width, int dst_height,
|
|
||||||
int dcn, bool swapBlue, int uIdx);
|
|
||||||
void cvtTwoPlaneYUVtoBGR(const uchar * y_data, const uchar * uv_data, size_t src_step,
|
|
||||||
uchar * dst_data, size_t dst_step,
|
uchar * dst_data, size_t dst_step,
|
||||||
int dst_width, int dst_height,
|
int dst_width, int dst_height,
|
||||||
int dcn, bool swapBlue, int uIdx);
|
int dcn, bool swapBlue, int uIdx);
|
||||||
@@ -1177,24 +1173,28 @@ struct YUV420sp2RGB8Invoker : ParallelLoopBody
|
|||||||
uchar * dst_data;
|
uchar * dst_data;
|
||||||
size_t dst_step;
|
size_t dst_step;
|
||||||
int width;
|
int width;
|
||||||
const uchar* my1, *muv;
|
const uchar* my1;
|
||||||
size_t stride;
|
size_t my1_step;
|
||||||
|
const uchar* muv;
|
||||||
|
size_t muv_step;
|
||||||
|
|
||||||
YUV420sp2RGB8Invoker(uchar * _dst_data, size_t _dst_step, int _dst_width, size_t _stride, const uchar* _y1, const uchar* _uv)
|
YUV420sp2RGB8Invoker(uchar * _dst_data, size_t _dst_step, int _dst_width,
|
||||||
: dst_data(_dst_data), dst_step(_dst_step), width(_dst_width), my1(_y1), muv(_uv), stride(_stride) {}
|
const uchar* _y1, size_t _y1_step, const uchar* _uv, size_t _uv_step) :
|
||||||
|
dst_data(_dst_data), dst_step(_dst_step), width(_dst_width),
|
||||||
|
my1(_y1), my1_step(_y1_step), muv(_uv), muv_step(_uv_step) {}
|
||||||
|
|
||||||
void operator()(const Range& range) const CV_OVERRIDE
|
void operator()(const Range& range) const CV_OVERRIDE
|
||||||
{
|
{
|
||||||
const int rangeBegin = range.start * 2;
|
const int rangeBegin = range.start * 2;
|
||||||
const int rangeEnd = range.end * 2;
|
const int rangeEnd = range.end * 2;
|
||||||
|
|
||||||
const uchar* y1 = my1 + rangeBegin * stride, *uv = muv + rangeBegin * stride / 2;
|
const uchar* y1 = my1 + rangeBegin * my1_step, *uv = muv + rangeBegin * muv_step / 2;
|
||||||
|
|
||||||
for (int j = rangeBegin; j < rangeEnd; j += 2, y1 += stride * 2, uv += stride)
|
for (int j = rangeBegin; j < rangeEnd; j += 2, y1 += my1_step * 2, uv += muv_step)
|
||||||
{
|
{
|
||||||
uchar* row1 = dst_data + dst_step * j;
|
uchar* row1 = dst_data + dst_step * j;
|
||||||
uchar* row2 = dst_data + dst_step * (j + 1);
|
uchar* row2 = dst_data + dst_step * (j + 1);
|
||||||
const uchar* y2 = y1 + stride;
|
const uchar* y2 = y1 + my1_step;
|
||||||
|
|
||||||
int i = 0;
|
int i = 0;
|
||||||
#if CV_SIMD
|
#if CV_SIMD
|
||||||
@@ -1395,9 +1395,10 @@ struct YUV420p2RGB8Invoker : ParallelLoopBody
|
|||||||
#define MIN_SIZE_FOR_PARALLEL_YUV420_CONVERSION (320*240)
|
#define MIN_SIZE_FOR_PARALLEL_YUV420_CONVERSION (320*240)
|
||||||
|
|
||||||
template<int bIdx, int uIdx, int dcn>
|
template<int bIdx, int uIdx, int dcn>
|
||||||
inline void cvtYUV420sp2RGB(uchar * dst_data, size_t dst_step, int dst_width, int dst_height, size_t _stride, const uchar* _y1, const uchar* _uv)
|
inline void cvtYUV420sp2RGB(uchar * dst_data, size_t dst_step, int dst_width, int dst_height,
|
||||||
|
const uchar* _y1, size_t _y1_step, const uchar* _uv, size_t _uv_step)
|
||||||
{
|
{
|
||||||
YUV420sp2RGB8Invoker<bIdx, uIdx, dcn> converter(dst_data, dst_step, dst_width, _stride, _y1, _uv);
|
YUV420sp2RGB8Invoker<bIdx, uIdx, dcn> converter(dst_data, dst_step, dst_width, _y1, _y1_step, _uv, _uv_step);
|
||||||
if (dst_width * dst_height >= MIN_SIZE_FOR_PARALLEL_YUV420_CONVERSION)
|
if (dst_width * dst_height >= MIN_SIZE_FOR_PARALLEL_YUV420_CONVERSION)
|
||||||
parallel_for_(Range(0, dst_height/2), converter);
|
parallel_for_(Range(0, dst_height/2), converter);
|
||||||
else
|
else
|
||||||
@@ -1817,26 +1818,16 @@ void cvtYUVtoBGR(const uchar * src_data, size_t src_step,
|
|||||||
CvtColorLoop(src_data, src_step, dst_data, dst_step, width, height, YCrCb2RGB_f<float>(dcn, blueIdx, isCbCr));
|
CvtColorLoop(src_data, src_step, dst_data, dst_step, width, height, YCrCb2RGB_f<float>(dcn, blueIdx, isCbCr));
|
||||||
}
|
}
|
||||||
|
|
||||||
void cvtTwoPlaneYUVtoBGR(const uchar * src_data, size_t src_step,
|
|
||||||
uchar * dst_data, size_t dst_step,
|
|
||||||
int dst_width, int dst_height,
|
|
||||||
int dcn, bool swapBlue, int uIdx)
|
|
||||||
{
|
|
||||||
CV_INSTRUMENT_REGION();
|
|
||||||
|
|
||||||
const uchar* uv = src_data + src_step * static_cast<size_t>(dst_height);
|
|
||||||
cvtTwoPlaneYUVtoBGR(src_data, uv, src_step, dst_data, dst_step, dst_width, dst_height, dcn, swapBlue, uIdx);
|
|
||||||
}
|
|
||||||
|
|
||||||
typedef void (*cvt_2plane_yuv_ptr_t)(uchar * /* dst_data*/,
|
typedef void (*cvt_2plane_yuv_ptr_t)(uchar * /* dst_data*/,
|
||||||
size_t /* dst_step */,
|
size_t /* dst_step */,
|
||||||
int /* dst_width */,
|
int /* dst_width */,
|
||||||
int /* dst_height */,
|
int /* dst_height */,
|
||||||
size_t /* _stride */,
|
|
||||||
const uchar* /* _y1 */,
|
const uchar* /* _y1 */,
|
||||||
const uchar* /* _uv */);
|
size_t /* _y1_step */,
|
||||||
|
const uchar* /* _uv */,
|
||||||
|
size_t /* _uv_step */);
|
||||||
|
|
||||||
void cvtTwoPlaneYUVtoBGR(const uchar * y_data, const uchar * uv_data, size_t src_step,
|
void cvtTwoPlaneYUVtoBGR(const uchar * y_data, size_t y_step, const uchar * uv_data, size_t uv_step,
|
||||||
uchar * dst_data, size_t dst_step,
|
uchar * dst_data, size_t dst_step,
|
||||||
int dst_width, int dst_height,
|
int dst_width, int dst_height,
|
||||||
int dcn, bool swapBlue, int uIdx)
|
int dcn, bool swapBlue, int uIdx)
|
||||||
@@ -1859,7 +1850,7 @@ void cvtTwoPlaneYUVtoBGR(const uchar * y_data, const uchar * uv_data, size_t src
|
|||||||
default: CV_Error( CV_StsBadFlag, "Unknown/unsupported color conversion code" ); break;
|
default: CV_Error( CV_StsBadFlag, "Unknown/unsupported color conversion code" ); break;
|
||||||
};
|
};
|
||||||
|
|
||||||
cvtPtr(dst_data, dst_step, dst_width, dst_height, src_step, y_data, uv_data);
|
cvtPtr(dst_data, dst_step, dst_width, dst_height, y_data, y_step, uv_data, uv_step);
|
||||||
}
|
}
|
||||||
|
|
||||||
typedef void (*cvt_3plane_yuv_ptr_t)(uchar * /* dst_data */,
|
typedef void (*cvt_3plane_yuv_ptr_t)(uchar * /* dst_data */,
|
||||||
|
|||||||
@@ -465,6 +465,49 @@ struct RowVec_8u32s
|
|||||||
bool smallValues;
|
bool smallValues;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
struct RowVec_8u32f
|
||||||
|
{
|
||||||
|
RowVec_8u32f() {}
|
||||||
|
RowVec_8u32f( const Mat& _kernel ) : kernel(_kernel) {}
|
||||||
|
|
||||||
|
int operator()(const uchar* _src, uchar* _dst, int width, int cn) const
|
||||||
|
{
|
||||||
|
CV_INSTRUMENT_REGION();
|
||||||
|
|
||||||
|
int i = 0, k, _ksize = kernel.rows + kernel.cols - 1;
|
||||||
|
float* dst = (float*)_dst;
|
||||||
|
const float* _kx = kernel.ptr<float>();
|
||||||
|
width *= cn;
|
||||||
|
for( ; i <= width - v_uint8::nlanes; i += v_uint8::nlanes )
|
||||||
|
{
|
||||||
|
v_float32 s0 = vx_setzero_f32();
|
||||||
|
v_float32 s1 = vx_setzero_f32();
|
||||||
|
v_float32 s2 = vx_setzero_f32();
|
||||||
|
v_float32 s3 = vx_setzero_f32();
|
||||||
|
k = 0;
|
||||||
|
for( ; k < _ksize ; k++ )
|
||||||
|
{
|
||||||
|
v_float32 f = vx_setall_f32(_kx[k]);
|
||||||
|
const uchar* src = (const uchar*)_src + i + k * cn;
|
||||||
|
v_float32 vs_ll = v_cvt_f32(v_reinterpret_as_s32(vx_load_expand_q(src)));
|
||||||
|
v_float32 vs_lh = v_cvt_f32(v_reinterpret_as_s32(vx_load_expand_q(src + v_float32::nlanes)));
|
||||||
|
v_float32 vs_hl = v_cvt_f32(v_reinterpret_as_s32(vx_load_expand_q(src + 2*v_float32::nlanes)));
|
||||||
|
v_float32 vs_hh = v_cvt_f32(v_reinterpret_as_s32(vx_load_expand_q(src + 3*v_float32::nlanes)));
|
||||||
|
s0 = v_muladd(vs_ll, f, s0);
|
||||||
|
s1 = v_muladd(vs_lh, f, s1);
|
||||||
|
s2 = v_muladd(vs_hl, f, s2);
|
||||||
|
s3 = v_muladd(vs_hh, f, s3);
|
||||||
|
}
|
||||||
|
v_store(dst + i, s0);
|
||||||
|
v_store(dst + i + v_float32::nlanes, s1);
|
||||||
|
v_store(dst + i + 2*v_float32::nlanes, s2);
|
||||||
|
v_store(dst + i + 3*v_float32::nlanes, s3);
|
||||||
|
}
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
|
||||||
|
Mat kernel;
|
||||||
|
};
|
||||||
|
|
||||||
struct SymmRowSmallVec_8u32s
|
struct SymmRowSmallVec_8u32s
|
||||||
{
|
{
|
||||||
@@ -2292,6 +2335,7 @@ struct FilterVec_32f
|
|||||||
#else
|
#else
|
||||||
|
|
||||||
typedef RowNoVec RowVec_8u32s;
|
typedef RowNoVec RowVec_8u32s;
|
||||||
|
typedef RowNoVec RowVec_8u32f;
|
||||||
typedef RowNoVec RowVec_16s32f;
|
typedef RowNoVec RowVec_16s32f;
|
||||||
typedef RowNoVec RowVec_32f;
|
typedef RowNoVec RowVec_32f;
|
||||||
typedef SymmRowSmallNoVec SymmRowSmallVec_8u32s;
|
typedef SymmRowSmallNoVec SymmRowSmallVec_8u32s;
|
||||||
@@ -2899,7 +2943,8 @@ Ptr<BaseRowFilter> getLinearRowFilter(
|
|||||||
return makePtr<RowFilter<uchar, int, RowVec_8u32s> >
|
return makePtr<RowFilter<uchar, int, RowVec_8u32s> >
|
||||||
(kernel, anchor, RowVec_8u32s(kernel));
|
(kernel, anchor, RowVec_8u32s(kernel));
|
||||||
if( sdepth == CV_8U && ddepth == CV_32F )
|
if( sdepth == CV_8U && ddepth == CV_32F )
|
||||||
return makePtr<RowFilter<uchar, float, RowNoVec> >(kernel, anchor);
|
return makePtr<RowFilter<uchar, float, RowVec_8u32f> >
|
||||||
|
(kernel, anchor, RowVec_8u32f(kernel));
|
||||||
if( sdepth == CV_8U && ddepth == CV_64F )
|
if( sdepth == CV_8U && ddepth == CV_64F )
|
||||||
return makePtr<RowFilter<uchar, double, RowNoVec> >(kernel, anchor);
|
return makePtr<RowFilter<uchar, double, RowNoVec> >(kernel, anchor);
|
||||||
if( sdepth == CV_16U && ddepth == CV_32F )
|
if( sdepth == CV_16U && ddepth == CV_32F )
|
||||||
|
|||||||
@@ -498,6 +498,39 @@ inline int hal_ni_cvtLabtoBGR(const uchar * src_data, size_t src_step, uchar * d
|
|||||||
*/
|
*/
|
||||||
inline int hal_ni_cvtTwoPlaneYUVtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int dst_width, int dst_height, int dcn, bool swapBlue, int uIdx) { return CV_HAL_ERROR_NOT_IMPLEMENTED; }
|
inline int hal_ni_cvtTwoPlaneYUVtoBGR(const uchar * src_data, size_t src_step, uchar * dst_data, size_t dst_step, int dst_width, int dst_height, int dcn, bool swapBlue, int uIdx) { return CV_HAL_ERROR_NOT_IMPLEMENTED; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
@brief Extended version of hal_cvtTwoPlaneYUVtoBGR.
|
||||||
|
@param y_data,y_step source image data and step (Y-plane)
|
||||||
|
@param uv_data,uv_step source image data and step (UV-plane)
|
||||||
|
@param dst_data,dst_step destination image data and step
|
||||||
|
@param dst_width,dst_height destination image size
|
||||||
|
@param dcn destination image channels (3 or 4)
|
||||||
|
@param swapBlue if set to true B and R destination channels will be swapped (write RGB)
|
||||||
|
@param uIdx U-channel index in the interleaved U/V plane (0 or 1)
|
||||||
|
Convert from YUV (YUV420sp (or NV12/NV21) - Y plane followed by interleaved U/V plane) to BGR, RGB, BGRA or RGBA.
|
||||||
|
Only for CV_8U.
|
||||||
|
*/
|
||||||
|
inline int hal_ni_cvtTwoPlaneYUVtoBGREx(const uchar * y_data, size_t y_step, const uchar * uv_data, size_t uv_step,
|
||||||
|
uchar * dst_data, size_t dst_step, int dst_width, int dst_height,
|
||||||
|
int dcn, bool swapBlue, int uIdx) { return CV_HAL_ERROR_NOT_IMPLEMENTED; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
@brief hal_cvtBGRtoTwoPlaneYUV
|
||||||
|
@param src_data,src_step source image data and step
|
||||||
|
@param y_data,y_step destination image data and step (Y-plane)
|
||||||
|
@param uv_data,uv_step destination image data and step (UV-plane)
|
||||||
|
@param width,height image size
|
||||||
|
@param scn source image channels (3 or 4)
|
||||||
|
@param swapBlue if set to true B and R source channels will be swapped (treat as RGB)
|
||||||
|
@param uIdx U-channel plane index (0 or 1)
|
||||||
|
Convert from BGR, RGB, BGRA or RGBA to YUV (YUV420sp (or NV12/NV21) - Y plane followed by interleaved U/V plane).
|
||||||
|
Only for CV_8U.
|
||||||
|
*/
|
||||||
|
inline int hal_ni_cvtBGRtoTwoPlaneYUV(const uchar * src_data, size_t src_step,
|
||||||
|
uchar * y_data, size_t y_step, uchar * uv_data, size_t uv_step,
|
||||||
|
int width, int height,
|
||||||
|
int scn, bool swapBlue, int uIdx) { return CV_HAL_ERROR_NOT_IMPLEMENTED; }
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@brief hal_cvtThreePlaneYUVtoBGR
|
@brief hal_cvtThreePlaneYUVtoBGR
|
||||||
@param src_data,src_step source image data and step
|
@param src_data,src_step source image data and step
|
||||||
@@ -576,6 +609,8 @@ inline int hal_ni_cvtMultipliedRGBAtoRGBA(const uchar * src_data, size_t src_ste
|
|||||||
#define cv_hal_cvtBGRtoLab hal_ni_cvtBGRtoLab
|
#define cv_hal_cvtBGRtoLab hal_ni_cvtBGRtoLab
|
||||||
#define cv_hal_cvtLabtoBGR hal_ni_cvtLabtoBGR
|
#define cv_hal_cvtLabtoBGR hal_ni_cvtLabtoBGR
|
||||||
#define cv_hal_cvtTwoPlaneYUVtoBGR hal_ni_cvtTwoPlaneYUVtoBGR
|
#define cv_hal_cvtTwoPlaneYUVtoBGR hal_ni_cvtTwoPlaneYUVtoBGR
|
||||||
|
#define cv_hal_cvtTwoPlaneYUVtoBGREx hal_ni_cvtTwoPlaneYUVtoBGREx
|
||||||
|
#define cv_hal_cvtBGRtoTwoPlaneYUV hal_ni_cvtBGRtoTwoPlaneYUV
|
||||||
#define cv_hal_cvtThreePlaneYUVtoBGR hal_ni_cvtThreePlaneYUVtoBGR
|
#define cv_hal_cvtThreePlaneYUVtoBGR hal_ni_cvtThreePlaneYUVtoBGR
|
||||||
#define cv_hal_cvtBGRtoThreePlaneYUV hal_ni_cvtBGRtoThreePlaneYUV
|
#define cv_hal_cvtBGRtoThreePlaneYUV hal_ni_cvtBGRtoThreePlaneYUV
|
||||||
#define cv_hal_cvtOnePlaneYUVtoBGR hal_ni_cvtOnePlaneYUVtoBGR
|
#define cv_hal_cvtOnePlaneYUVtoBGR hal_ni_cvtOnePlaneYUVtoBGR
|
||||||
|
|||||||
@@ -435,12 +435,14 @@ HoughLinesSDiv( InputArray image, OutputArray lines, int type,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
int pos = (int)(lst.size() - 1);
|
||||||
|
if( pos >= 0 && lst[pos].rho < 0 )
|
||||||
|
lst.pop_back();
|
||||||
|
|
||||||
lines.create((int)lst.size(), 1, type);
|
lines.create((int)lst.size(), 1, type);
|
||||||
Mat _lines = lines.getMat();
|
Mat _lines = lines.getMat();
|
||||||
for( size_t idx = 0; idx < lst.size(); idx++ )
|
for( size_t idx = 0; idx < lst.size(); idx++ )
|
||||||
{
|
{
|
||||||
if( lst[idx].rho < 0 )
|
|
||||||
continue;
|
|
||||||
if (type == CV_32FC2)
|
if (type == CV_32FC2)
|
||||||
{
|
{
|
||||||
_lines.at<Vec2f>((int)idx) = Vec2f(lst[idx].rho, lst[idx].theta);
|
_lines.at<Vec2f>((int)idx) = Vec2f(lst[idx].rho, lst[idx].theta);
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user