mirror of
https://github.com/opencv/opencv.git
synced 2026-07-25 21:33:04 +04:00
Compare commits
152 Commits
4.1.2-openvino
...
4.1.2
| Author | SHA1 | Date | |
|---|---|---|---|
| 4c71dbf0af | |||
| 0cf77c3916 | |||
| 9efafc3e33 | |||
| 65573784c4 | |||
| dd4f591d54 | |||
| 6bdb9ca725 | |||
| 1ca74c3c03 | |||
| c3a588037a | |||
| 40215f99d7 | |||
| 3fd36c1be1 | |||
| 24effe8cd6 | |||
| 2ff1fb60ad | |||
| 374d952e09 | |||
| 6efdfee3f0 | |||
| 42ac089e12 | |||
| 9ab5e52f04 | |||
| 4748aca61f | |||
| f2fe6f40c2 | |||
| 7837ae0e19 | |||
| a007220c52 | |||
| ebb77bb311 | |||
| 53400d86e2 | |||
| c0489963bb | |||
| 626bfbf309 | |||
| 98fc098216 | |||
| 953c60829e | |||
| 6d811f9879 | |||
| 22d0c57a1c | |||
| b1485d0337 | |||
| c54753c185 | |||
| bdc097495a | |||
| 8115b004b9 | |||
| feff8bf972 | |||
| 4429352125 | |||
| f301f17b61 | |||
| c13a5ce229 | |||
| c9b2cedebd | |||
| ab5252c18e | |||
| e35fd463e7 | |||
| fd11e3a81d | |||
| f0058bbed3 | |||
| 22d86116ee | |||
| deea46000e | |||
| c99db2b9db | |||
| c69245da1f | |||
| 3c50d24d8e | |||
| 53c88f0f9e | |||
| d50babfc99 | |||
| 2b66495a9d | |||
| 3fb6617d62 | |||
| a105f56957 | |||
| 23bd1866ca | |||
| 59c182ed2b | |||
| c06115cb3f | |||
| 8b483a91bc | |||
| 8814645c8d | |||
| c92f3072b4 | |||
| ffea927ac2 | |||
| 77346d7286 | |||
| 3133bb49aa | |||
| 440a937d24 | |||
| 06067efa3f | |||
| 6aa689f87c | |||
| ba0b3983c6 | |||
| a3144cbadc | |||
| 5c9a624a85 | |||
| fba9fdfd27 | |||
| d88d1c9935 | |||
| 507ca291e1 | |||
| ed9bca969c | |||
| aa61e79615 | |||
| 25dee8383c | |||
| f81e401cd0 | |||
| bc927f9788 | |||
| e2a5a6a05c | |||
| 7ce9428e96 | |||
| 9f4f9000bc | |||
| 677b94c92e | |||
| b88435fdc2 | |||
| 790927bb55 | |||
| e923712d81 | |||
| eacadf0e73 | |||
| 6e246ee58c | |||
| e3daf489c8 | |||
| 1c17b3281a | |||
| d2cacac07a | |||
| e0be771b64 | |||
| c2096771cb | |||
| 2fef9bc1d8 | |||
| 3cf9185159 | |||
| c8abf2ad14 | |||
| bab8addcbb | |||
| c4d2e3c0b3 | |||
| b40fd6de32 | |||
| 3516d0835e | |||
| fcc69d5a60 | |||
| fef7fc343e | |||
| ad2854c8b3 | |||
| a74fe2ec01 | |||
| eabbe38001 | |||
| b1ea91d8bd | |||
| a052567db8 | |||
| 2d2330382d | |||
| 66842f5a18 | |||
| 33e9fe9312 | |||
| 65c209fad3 | |||
| a9163a53d3 | |||
| c657c6cbac | |||
| ee23a7575d | |||
| 8167d47efc | |||
| 1ecea1f4a6 | |||
| 3ece62cb9c | |||
| 46fd112f9b | |||
| cd7fa04985 | |||
| bcf7d3160c | |||
| d2872afce0 | |||
| 70c88a2087 | |||
| 7770ba1f10 | |||
| 7b3a752012 | |||
| 5a75808516 | |||
| b4c5b50a3e | |||
| 03c4c67dad | |||
| f5f9187720 | |||
| 89b82e796e | |||
| 48fa39f675 | |||
| 69c19dc3d3 | |||
| 5c0502b470 | |||
| f10fce9ab4 | |||
| aaad238c6e | |||
| 3ddb005480 | |||
| 741aee6901 | |||
| 0428f60d66 | |||
| b465c82696 | |||
| 1e410ed826 | |||
| 8609198b05 | |||
| 95a55250a7 | |||
| 349f0cf5e9 | |||
| 57b16a4b6e | |||
| 758156a9b6 | |||
| bf8b5ffeb1 | |||
| 8747082091 | |||
| a17231a6b4 | |||
| b3a0507546 | |||
| 4de115c08c | |||
| f139a0bda8 | |||
| bdf23ce855 | |||
| 60488ffbbd | |||
| 6a49887695 | |||
| e7b6753a10 | |||
| 9d4f01626c | |||
| d974d4c6ce | |||
| f6ee03471d |
Vendored
+9
@@ -46,6 +46,15 @@ if(";${CPU_BASELINE_FINAL};" MATCHES "SSE2"
|
||||
add_definitions(-DPNG_INTEL_SSE)
|
||||
endif()
|
||||
|
||||
# set definitions and sources for MIPS
|
||||
if(";${CPU_BASELINE_FINAL};" MATCHES "MSA")
|
||||
list(APPEND lib_srcs mips/mips_init.c mips/filter_msa_intrinsics.c)
|
||||
add_definitions(-DPNG_MIPS_MSA_OPT=2)
|
||||
ocv_warnings_disable(CMAKE_C_FLAGS -Wshadow)
|
||||
else()
|
||||
add_definitions(-DPNG_MIPS_MSA_OPT=0)
|
||||
endif()
|
||||
|
||||
if(PPC64LE OR PPC64)
|
||||
# VSX3 features are backwards compatible
|
||||
if(";${CPU_BASELINE_FINAL};" MATCHES "VSX.*"
|
||||
|
||||
+808
@@ -0,0 +1,808 @@
|
||||
|
||||
/* filter_msa_intrinsics.c - MSA optimised filter functions
|
||||
*
|
||||
* Copyright (c) 2018 Cosmin Truta
|
||||
* Copyright (c) 2016 Glenn Randers-Pehrson
|
||||
* Written by Mandar Sahastrabuddhe, August 2016.
|
||||
*
|
||||
* This code is released under the libpng license.
|
||||
* For conditions of distribution and use, see the disclaimer
|
||||
* and license in png.h
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdint.h>
|
||||
#include "../pngpriv.h"
|
||||
|
||||
#ifdef PNG_READ_SUPPORTED
|
||||
|
||||
/* This code requires -mfpu=msa on the command line: */
|
||||
#if PNG_MIPS_MSA_IMPLEMENTATION == 1 /* intrinsics code from pngpriv.h */
|
||||
|
||||
#include <msa.h>
|
||||
|
||||
/* libpng row pointers are not necessarily aligned to any particular boundary,
|
||||
* however this code will only work with appropriate alignment. mips/mips_init.c
|
||||
* checks for this (and will not compile unless it is done). This code uses
|
||||
* variants of png_aligncast to avoid compiler warnings.
|
||||
*/
|
||||
#define png_ptr(type,pointer) png_aligncast(type *,pointer)
|
||||
#define png_ptrc(type,pointer) png_aligncastconst(const type *,pointer)
|
||||
|
||||
/* The following relies on a variable 'temp_pointer' being declared with type
|
||||
* 'type'. This is written this way just to hide the GCC strict aliasing
|
||||
* warning; note that the code is safe because there never is an alias between
|
||||
* the input and output pointers.
|
||||
*/
|
||||
#define png_ldr(type,pointer)\
|
||||
(temp_pointer = png_ptr(type,pointer), *temp_pointer)
|
||||
|
||||
#if PNG_MIPS_MSA_OPT > 0
|
||||
|
||||
#ifdef CLANG_BUILD
|
||||
#define MSA_SRLI_B(a, b) __msa_srli_b((v16i8) a, b)
|
||||
|
||||
#define LW(psrc) \
|
||||
( { \
|
||||
uint8_t *psrc_lw_m = (uint8_t *) (psrc); \
|
||||
uint32_t val_m; \
|
||||
\
|
||||
asm volatile ( \
|
||||
"lw %[val_m], %[psrc_lw_m] \n\t" \
|
||||
\
|
||||
: [val_m] "=r" (val_m) \
|
||||
: [psrc_lw_m] "m" (*psrc_lw_m) \
|
||||
); \
|
||||
\
|
||||
val_m; \
|
||||
} )
|
||||
|
||||
#define SH(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sh_m = (uint8_t *) (pdst); \
|
||||
uint16_t val_m = (val); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"sh %[val_m], %[pdst_sh_m] \n\t" \
|
||||
\
|
||||
: [pdst_sh_m] "=m" (*pdst_sh_m) \
|
||||
: [val_m] "r" (val_m) \
|
||||
); \
|
||||
}
|
||||
|
||||
#define SW(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sw_m = (uint8_t *) (pdst); \
|
||||
uint32_t val_m = (val); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"sw %[val_m], %[pdst_sw_m] \n\t" \
|
||||
\
|
||||
: [pdst_sw_m] "=m" (*pdst_sw_m) \
|
||||
: [val_m] "r" (val_m) \
|
||||
); \
|
||||
}
|
||||
|
||||
#if (__mips == 64)
|
||||
#define SD(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sd_m = (uint8_t *) (pdst); \
|
||||
uint64_t val_m = (val); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"sd %[val_m], %[pdst_sd_m] \n\t" \
|
||||
\
|
||||
: [pdst_sd_m] "=m" (*pdst_sd_m) \
|
||||
: [val_m] "r" (val_m) \
|
||||
); \
|
||||
}
|
||||
#else
|
||||
#define SD(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sd_m = (uint8_t *) (pdst); \
|
||||
uint32_t val0_m, val1_m; \
|
||||
\
|
||||
val0_m = (uint32_t) ((val) & 0x00000000FFFFFFFF); \
|
||||
val1_m = (uint32_t) (((val) >> 32) & 0x00000000FFFFFFFF); \
|
||||
\
|
||||
SW(val0_m, pdst_sd_m); \
|
||||
SW(val1_m, pdst_sd_m + 4); \
|
||||
}
|
||||
#endif
|
||||
#else
|
||||
#define MSA_SRLI_B(a, b) (a >> b)
|
||||
|
||||
#if (__mips_isa_rev >= 6)
|
||||
#define LW(psrc) \
|
||||
( { \
|
||||
uint8_t *psrc_lw_m = (uint8_t *) (psrc); \
|
||||
uint32_t val_m; \
|
||||
\
|
||||
asm volatile ( \
|
||||
"lw %[val_m], %[psrc_lw_m] \n\t" \
|
||||
\
|
||||
: [val_m] "=r" (val_m) \
|
||||
: [psrc_lw_m] "m" (*psrc_lw_m) \
|
||||
); \
|
||||
\
|
||||
val_m; \
|
||||
} )
|
||||
|
||||
#define SH(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sh_m = (uint8_t *) (pdst); \
|
||||
uint16_t val_m = (val); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"sh %[val_m], %[pdst_sh_m] \n\t" \
|
||||
\
|
||||
: [pdst_sh_m] "=m" (*pdst_sh_m) \
|
||||
: [val_m] "r" (val_m) \
|
||||
); \
|
||||
}
|
||||
|
||||
#define SW(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sw_m = (uint8_t *) (pdst); \
|
||||
uint32_t val_m = (val); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"sw %[val_m], %[pdst_sw_m] \n\t" \
|
||||
\
|
||||
: [pdst_sw_m] "=m" (*pdst_sw_m) \
|
||||
: [val_m] "r" (val_m) \
|
||||
); \
|
||||
}
|
||||
|
||||
#if (__mips == 64)
|
||||
#define SD(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sd_m = (uint8_t *) (pdst); \
|
||||
uint64_t val_m = (val); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"sd %[val_m], %[pdst_sd_m] \n\t" \
|
||||
\
|
||||
: [pdst_sd_m] "=m" (*pdst_sd_m) \
|
||||
: [val_m] "r" (val_m) \
|
||||
); \
|
||||
}
|
||||
#else
|
||||
#define SD(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sd_m = (uint8_t *) (pdst); \
|
||||
uint32_t val0_m, val1_m; \
|
||||
\
|
||||
val0_m = (uint32_t) ((val) & 0x00000000FFFFFFFF); \
|
||||
val1_m = (uint32_t) (((val) >> 32) & 0x00000000FFFFFFFF); \
|
||||
\
|
||||
SW(val0_m, pdst_sd_m); \
|
||||
SW(val1_m, pdst_sd_m + 4); \
|
||||
}
|
||||
#endif
|
||||
#else // !(__mips_isa_rev >= 6)
|
||||
#define LW(psrc) \
|
||||
( { \
|
||||
uint8_t *psrc_lw_m = (uint8_t *) (psrc); \
|
||||
uint32_t val_m; \
|
||||
\
|
||||
asm volatile ( \
|
||||
"ulw %[val_m], %[psrc_lw_m] \n\t" \
|
||||
\
|
||||
: [val_m] "=r" (val_m) \
|
||||
: [psrc_lw_m] "m" (*psrc_lw_m) \
|
||||
); \
|
||||
\
|
||||
val_m; \
|
||||
} )
|
||||
|
||||
#define SH(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sh_m = (uint8_t *) (pdst); \
|
||||
uint16_t val_m = (val); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"ush %[val_m], %[pdst_sh_m] \n\t" \
|
||||
\
|
||||
: [pdst_sh_m] "=m" (*pdst_sh_m) \
|
||||
: [val_m] "r" (val_m) \
|
||||
); \
|
||||
}
|
||||
|
||||
#define SW(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sw_m = (uint8_t *) (pdst); \
|
||||
uint32_t val_m = (val); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"usw %[val_m], %[pdst_sw_m] \n\t" \
|
||||
\
|
||||
: [pdst_sw_m] "=m" (*pdst_sw_m) \
|
||||
: [val_m] "r" (val_m) \
|
||||
); \
|
||||
}
|
||||
|
||||
#define SD(val, pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_sd_m = (uint8_t *) (pdst); \
|
||||
uint32_t val0_m, val1_m; \
|
||||
\
|
||||
val0_m = (uint32_t) ((val) & 0x00000000FFFFFFFF); \
|
||||
val1_m = (uint32_t) (((val) >> 32) & 0x00000000FFFFFFFF); \
|
||||
\
|
||||
SW(val0_m, pdst_sd_m); \
|
||||
SW(val1_m, pdst_sd_m + 4); \
|
||||
}
|
||||
|
||||
#define SW_ZERO(pdst) \
|
||||
{ \
|
||||
uint8_t *pdst_m = (uint8_t *) (pdst); \
|
||||
\
|
||||
asm volatile ( \
|
||||
"usw $0, %[pdst_m] \n\t" \
|
||||
\
|
||||
: [pdst_m] "=m" (*pdst_m) \
|
||||
: \
|
||||
); \
|
||||
}
|
||||
#endif // (__mips_isa_rev >= 6)
|
||||
#endif
|
||||
|
||||
#define LD_B(RTYPE, psrc) *((RTYPE *) (psrc))
|
||||
#define LD_UB(...) LD_B(v16u8, __VA_ARGS__)
|
||||
#define LD_B2(RTYPE, psrc, stride, out0, out1) \
|
||||
{ \
|
||||
out0 = LD_B(RTYPE, (psrc)); \
|
||||
out1 = LD_B(RTYPE, (psrc) + stride); \
|
||||
}
|
||||
#define LD_UB2(...) LD_B2(v16u8, __VA_ARGS__)
|
||||
#define LD_B4(RTYPE, psrc, stride, out0, out1, out2, out3) \
|
||||
{ \
|
||||
LD_B2(RTYPE, (psrc), stride, out0, out1); \
|
||||
LD_B2(RTYPE, (psrc) + 2 * stride , stride, out2, out3); \
|
||||
}
|
||||
#define LD_UB4(...) LD_B4(v16u8, __VA_ARGS__)
|
||||
|
||||
#define ST_B(RTYPE, in, pdst) *((RTYPE *) (pdst)) = (in)
|
||||
#define ST_UB(...) ST_B(v16u8, __VA_ARGS__)
|
||||
#define ST_B2(RTYPE, in0, in1, pdst, stride) \
|
||||
{ \
|
||||
ST_B(RTYPE, in0, (pdst)); \
|
||||
ST_B(RTYPE, in1, (pdst) + stride); \
|
||||
}
|
||||
#define ST_UB2(...) ST_B2(v16u8, __VA_ARGS__)
|
||||
#define ST_B4(RTYPE, in0, in1, in2, in3, pdst, stride) \
|
||||
{ \
|
||||
ST_B2(RTYPE, in0, in1, (pdst), stride); \
|
||||
ST_B2(RTYPE, in2, in3, (pdst) + 2 * stride, stride); \
|
||||
}
|
||||
#define ST_UB4(...) ST_B4(v16u8, __VA_ARGS__)
|
||||
|
||||
#define ADD2(in0, in1, in2, in3, out0, out1) \
|
||||
{ \
|
||||
out0 = in0 + in1; \
|
||||
out1 = in2 + in3; \
|
||||
}
|
||||
#define ADD3(in0, in1, in2, in3, in4, in5, \
|
||||
out0, out1, out2) \
|
||||
{ \
|
||||
ADD2(in0, in1, in2, in3, out0, out1); \
|
||||
out2 = in4 + in5; \
|
||||
}
|
||||
#define ADD4(in0, in1, in2, in3, in4, in5, in6, in7, \
|
||||
out0, out1, out2, out3) \
|
||||
{ \
|
||||
ADD2(in0, in1, in2, in3, out0, out1); \
|
||||
ADD2(in4, in5, in6, in7, out2, out3); \
|
||||
}
|
||||
|
||||
#define ILVR_B2(RTYPE, in0, in1, in2, in3, out0, out1) \
|
||||
{ \
|
||||
out0 = (RTYPE) __msa_ilvr_b((v16i8) in0, (v16i8) in1); \
|
||||
out1 = (RTYPE) __msa_ilvr_b((v16i8) in2, (v16i8) in3); \
|
||||
}
|
||||
#define ILVR_B2_SH(...) ILVR_B2(v8i16, __VA_ARGS__)
|
||||
|
||||
#define HSUB_UB2(RTYPE, in0, in1, out0, out1) \
|
||||
{ \
|
||||
out0 = (RTYPE) __msa_hsub_u_h((v16u8) in0, (v16u8) in0); \
|
||||
out1 = (RTYPE) __msa_hsub_u_h((v16u8) in1, (v16u8) in1); \
|
||||
}
|
||||
#define HSUB_UB2_SH(...) HSUB_UB2(v8i16, __VA_ARGS__)
|
||||
|
||||
#define SLDI_B2_0(RTYPE, in0, in1, out0, out1, slide_val) \
|
||||
{ \
|
||||
v16i8 zero_m = { 0 }; \
|
||||
out0 = (RTYPE) __msa_sldi_b((v16i8) zero_m, (v16i8) in0, slide_val); \
|
||||
out1 = (RTYPE) __msa_sldi_b((v16i8) zero_m, (v16i8) in1, slide_val); \
|
||||
}
|
||||
#define SLDI_B2_0_UB(...) SLDI_B2_0(v16u8, __VA_ARGS__)
|
||||
|
||||
#define SLDI_B3_0(RTYPE, in0, in1, in2, out0, out1, out2, slide_val) \
|
||||
{ \
|
||||
v16i8 zero_m = { 0 }; \
|
||||
SLDI_B2_0(RTYPE, in0, in1, out0, out1, slide_val); \
|
||||
out2 = (RTYPE) __msa_sldi_b((v16i8) zero_m, (v16i8) in2, slide_val); \
|
||||
}
|
||||
#define SLDI_B3_0_UB(...) SLDI_B3_0(v16u8, __VA_ARGS__)
|
||||
|
||||
#define ILVEV_W2(RTYPE, in0, in1, in2, in3, out0, out1) \
|
||||
{ \
|
||||
out0 = (RTYPE) __msa_ilvev_w((v4i32) in1, (v4i32) in0); \
|
||||
out1 = (RTYPE) __msa_ilvev_w((v4i32) in3, (v4i32) in2); \
|
||||
}
|
||||
#define ILVEV_W2_UB(...) ILVEV_W2(v16u8, __VA_ARGS__)
|
||||
|
||||
#define ADD_ABS_H3(RTYPE, in0, in1, in2, out0, out1, out2) \
|
||||
{ \
|
||||
RTYPE zero = {0}; \
|
||||
\
|
||||
out0 = __msa_add_a_h((v8i16) zero, in0); \
|
||||
out1 = __msa_add_a_h((v8i16) zero, in1); \
|
||||
out2 = __msa_add_a_h((v8i16) zero, in2); \
|
||||
}
|
||||
#define ADD_ABS_H3_SH(...) ADD_ABS_H3(v8i16, __VA_ARGS__)
|
||||
|
||||
#define VSHF_B2(RTYPE, in0, in1, in2, in3, mask0, mask1, out0, out1) \
|
||||
{ \
|
||||
out0 = (RTYPE) __msa_vshf_b((v16i8) mask0, (v16i8) in1, (v16i8) in0); \
|
||||
out1 = (RTYPE) __msa_vshf_b((v16i8) mask1, (v16i8) in3, (v16i8) in2); \
|
||||
}
|
||||
#define VSHF_B2_UB(...) VSHF_B2(v16u8, __VA_ARGS__)
|
||||
|
||||
#define CMP_AND_SELECT(inp0, inp1, inp2, inp3, inp4, inp5, out0) \
|
||||
{ \
|
||||
v8i16 _sel_h0, _sel_h1; \
|
||||
v16u8 _sel_b0, _sel_b1; \
|
||||
_sel_h0 = (v8i16) __msa_clt_u_h((v8u16) inp1, (v8u16) inp0); \
|
||||
_sel_b0 = (v16u8) __msa_pckev_b((v16i8) _sel_h0, (v16i8) _sel_h0); \
|
||||
inp0 = (v8i16) __msa_bmnz_v((v16u8) inp0, (v16u8) inp1, (v16u8) _sel_h0); \
|
||||
inp4 = (v16u8) __msa_bmnz_v(inp3, inp4, _sel_b0); \
|
||||
_sel_h1 = (v8i16) __msa_clt_u_h((v8u16) inp2, (v8u16) inp0); \
|
||||
_sel_b1 = (v16u8) __msa_pckev_b((v16i8) _sel_h1, (v16i8) _sel_h1); \
|
||||
inp4 = (v16u8) __msa_bmnz_v(inp4, inp5, _sel_b1); \
|
||||
out0 += inp4; \
|
||||
}
|
||||
|
||||
void png_read_filter_row_up_msa(png_row_infop row_info, png_bytep row,
|
||||
png_const_bytep prev_row)
|
||||
{
|
||||
size_t i, cnt, cnt16, cnt32;
|
||||
size_t istop = row_info->rowbytes;
|
||||
png_bytep rp = row;
|
||||
png_const_bytep pp = prev_row;
|
||||
v16u8 src0, src1, src2, src3, src4, src5, src6, src7;
|
||||
|
||||
for (i = 0; i < (istop >> 6); i++)
|
||||
{
|
||||
LD_UB4(rp, 16, src0, src1, src2, src3);
|
||||
LD_UB4(pp, 16, src4, src5, src6, src7);
|
||||
pp += 64;
|
||||
|
||||
ADD4(src0, src4, src1, src5, src2, src6, src3, src7,
|
||||
src0, src1, src2, src3);
|
||||
|
||||
ST_UB4(src0, src1, src2, src3, rp, 16);
|
||||
rp += 64;
|
||||
}
|
||||
|
||||
if (istop & 0x3F)
|
||||
{
|
||||
cnt32 = istop & 0x20;
|
||||
cnt16 = istop & 0x10;
|
||||
cnt = istop & 0xF;
|
||||
|
||||
if(cnt32)
|
||||
{
|
||||
if (cnt16 && cnt)
|
||||
{
|
||||
LD_UB4(rp, 16, src0, src1, src2, src3);
|
||||
LD_UB4(pp, 16, src4, src5, src6, src7);
|
||||
|
||||
ADD4(src0, src4, src1, src5, src2, src6, src3, src7,
|
||||
src0, src1, src2, src3);
|
||||
|
||||
ST_UB4(src0, src1, src2, src3, rp, 16);
|
||||
rp += 64;
|
||||
}
|
||||
else if (cnt16 || cnt)
|
||||
{
|
||||
LD_UB2(rp, 16, src0, src1);
|
||||
LD_UB2(pp, 16, src4, src5);
|
||||
pp += 32;
|
||||
src2 = LD_UB(rp + 32);
|
||||
src6 = LD_UB(pp);
|
||||
|
||||
ADD3(src0, src4, src1, src5, src2, src6, src0, src1, src2);
|
||||
|
||||
ST_UB2(src0, src1, rp, 16);
|
||||
rp += 32;
|
||||
ST_UB(src2, rp);
|
||||
rp += 16;
|
||||
}
|
||||
else
|
||||
{
|
||||
LD_UB2(rp, 16, src0, src1);
|
||||
LD_UB2(pp, 16, src4, src5);
|
||||
|
||||
ADD2(src0, src4, src1, src5, src0, src1);
|
||||
|
||||
ST_UB2(src0, src1, rp, 16);
|
||||
rp += 32;
|
||||
}
|
||||
}
|
||||
else if (cnt16 && cnt)
|
||||
{
|
||||
LD_UB2(rp, 16, src0, src1);
|
||||
LD_UB2(pp, 16, src4, src5);
|
||||
|
||||
ADD2(src0, src4, src1, src5, src0, src1);
|
||||
|
||||
ST_UB2(src0, src1, rp, 16);
|
||||
rp += 32;
|
||||
}
|
||||
else if (cnt16 || cnt)
|
||||
{
|
||||
src0 = LD_UB(rp);
|
||||
src4 = LD_UB(pp);
|
||||
pp += 16;
|
||||
|
||||
src0 += src4;
|
||||
|
||||
ST_UB(src0, rp);
|
||||
rp += 16;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void png_read_filter_row_sub4_msa(png_row_infop row_info, png_bytep row,
|
||||
png_const_bytep prev_row)
|
||||
{
|
||||
size_t count;
|
||||
size_t istop = row_info->rowbytes;
|
||||
png_bytep src = row;
|
||||
png_bytep nxt = row + 4;
|
||||
int32_t inp0;
|
||||
v16u8 src0, src1, src2, src3, src4;
|
||||
v16u8 dst0, dst1;
|
||||
v16u8 zero = { 0 };
|
||||
|
||||
istop -= 4;
|
||||
|
||||
inp0 = LW(src);
|
||||
src += 4;
|
||||
src0 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp0);
|
||||
|
||||
for (count = 0; count < istop; count += 16)
|
||||
{
|
||||
src1 = LD_UB(src);
|
||||
src += 16;
|
||||
|
||||
src2 = (v16u8) __msa_sldi_b((v16i8) zero, (v16i8) src1, 4);
|
||||
src3 = (v16u8) __msa_sldi_b((v16i8) zero, (v16i8) src1, 8);
|
||||
src4 = (v16u8) __msa_sldi_b((v16i8) zero, (v16i8) src1, 12);
|
||||
src1 += src0;
|
||||
src2 += src1;
|
||||
src3 += src2;
|
||||
src4 += src3;
|
||||
src0 = src4;
|
||||
ILVEV_W2_UB(src1, src2, src3, src4, dst0, dst1);
|
||||
dst0 = (v16u8) __msa_pckev_d((v2i64) dst1, (v2i64) dst0);
|
||||
|
||||
ST_UB(dst0, nxt);
|
||||
nxt += 16;
|
||||
}
|
||||
}
|
||||
|
||||
void png_read_filter_row_sub3_msa(png_row_infop row_info, png_bytep row,
|
||||
png_const_bytep prev_row)
|
||||
{
|
||||
size_t count;
|
||||
size_t istop = row_info->rowbytes;
|
||||
png_bytep src = row;
|
||||
png_bytep nxt = row + 3;
|
||||
int64_t out0;
|
||||
int32_t inp0, out1;
|
||||
v16u8 src0, src1, src2, src3, src4, dst0, dst1;
|
||||
v16u8 zero = { 0 };
|
||||
v16i8 mask0 = { 0, 1, 2, 16, 17, 18, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 };
|
||||
v16i8 mask1 = { 0, 1, 2, 3, 4, 5, 16, 17, 18, 19, 20, 21, 0, 0, 0, 0 };
|
||||
|
||||
istop -= 3;
|
||||
|
||||
inp0 = LW(src);
|
||||
src += 3;
|
||||
src0 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp0);
|
||||
|
||||
for (count = 0; count < istop; count += 12)
|
||||
{
|
||||
src1 = LD_UB(src);
|
||||
src += 12;
|
||||
|
||||
src2 = (v16u8) __msa_sldi_b((v16i8) zero, (v16i8) src1, 3);
|
||||
src3 = (v16u8) __msa_sldi_b((v16i8) zero, (v16i8) src1, 6);
|
||||
src4 = (v16u8) __msa_sldi_b((v16i8) zero, (v16i8) src1, 9);
|
||||
src1 += src0;
|
||||
src2 += src1;
|
||||
src3 += src2;
|
||||
src4 += src3;
|
||||
src0 = src4;
|
||||
VSHF_B2_UB(src1, src2, src3, src4, mask0, mask0, dst0, dst1);
|
||||
dst0 = (v16u8) __msa_vshf_b(mask1, (v16i8) dst1, (v16i8) dst0);
|
||||
out0 = __msa_copy_s_d((v2i64) dst0, 0);
|
||||
out1 = __msa_copy_s_w((v4i32) dst0, 2);
|
||||
|
||||
SD(out0, nxt);
|
||||
nxt += 8;
|
||||
SW(out1, nxt);
|
||||
nxt += 4;
|
||||
}
|
||||
}
|
||||
|
||||
void png_read_filter_row_avg4_msa(png_row_infop row_info, png_bytep row,
|
||||
png_const_bytep prev_row)
|
||||
{
|
||||
size_t i;
|
||||
png_bytep src = row;
|
||||
png_bytep nxt = row;
|
||||
png_const_bytep pp = prev_row;
|
||||
size_t istop = row_info->rowbytes - 4;
|
||||
int32_t inp0, inp1, out0;
|
||||
v16u8 src0, src1, src2, src3, src4, src5, src6, src7, src8, src9, dst0, dst1;
|
||||
v16u8 zero = { 0 };
|
||||
|
||||
inp0 = LW(pp);
|
||||
pp += 4;
|
||||
inp1 = LW(src);
|
||||
src += 4;
|
||||
src0 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp0);
|
||||
src1 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp1);
|
||||
src0 = (v16u8) MSA_SRLI_B(src0, 1);
|
||||
src1 += src0;
|
||||
out0 = __msa_copy_s_w((v4i32) src1, 0);
|
||||
SW(out0, nxt);
|
||||
nxt += 4;
|
||||
|
||||
for (i = 0; i < istop; i += 16)
|
||||
{
|
||||
src2 = LD_UB(pp);
|
||||
pp += 16;
|
||||
src6 = LD_UB(src);
|
||||
src += 16;
|
||||
|
||||
SLDI_B2_0_UB(src2, src6, src3, src7, 4);
|
||||
SLDI_B2_0_UB(src2, src6, src4, src8, 8);
|
||||
SLDI_B2_0_UB(src2, src6, src5, src9, 12);
|
||||
src2 = __msa_ave_u_b(src2, src1);
|
||||
src6 += src2;
|
||||
src3 = __msa_ave_u_b(src3, src6);
|
||||
src7 += src3;
|
||||
src4 = __msa_ave_u_b(src4, src7);
|
||||
src8 += src4;
|
||||
src5 = __msa_ave_u_b(src5, src8);
|
||||
src9 += src5;
|
||||
src1 = src9;
|
||||
ILVEV_W2_UB(src6, src7, src8, src9, dst0, dst1);
|
||||
dst0 = (v16u8) __msa_pckev_d((v2i64) dst1, (v2i64) dst0);
|
||||
|
||||
ST_UB(dst0, nxt);
|
||||
nxt += 16;
|
||||
}
|
||||
}
|
||||
|
||||
void png_read_filter_row_avg3_msa(png_row_infop row_info, png_bytep row,
|
||||
png_const_bytep prev_row)
|
||||
{
|
||||
size_t i;
|
||||
png_bytep src = row;
|
||||
png_bytep nxt = row;
|
||||
png_const_bytep pp = prev_row;
|
||||
size_t istop = row_info->rowbytes - 3;
|
||||
int64_t out0;
|
||||
int32_t inp0, inp1, out1;
|
||||
int16_t out2;
|
||||
v16u8 src0, src1, src2, src3, src4, src5, src6, src7, src8, src9, dst0, dst1;
|
||||
v16u8 zero = { 0 };
|
||||
v16i8 mask0 = { 0, 1, 2, 16, 17, 18, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 };
|
||||
v16i8 mask1 = { 0, 1, 2, 3, 4, 5, 16, 17, 18, 19, 20, 21, 0, 0, 0, 0 };
|
||||
|
||||
inp0 = LW(pp);
|
||||
pp += 3;
|
||||
inp1 = LW(src);
|
||||
src += 3;
|
||||
src0 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp0);
|
||||
src1 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp1);
|
||||
src0 = (v16u8) MSA_SRLI_B(src0, 1);
|
||||
src1 += src0;
|
||||
out2 = __msa_copy_s_h((v8i16) src1, 0);
|
||||
SH(out2, nxt);
|
||||
nxt += 2;
|
||||
nxt[0] = src1[2];
|
||||
nxt++;
|
||||
|
||||
for (i = 0; i < istop; i += 12)
|
||||
{
|
||||
src2 = LD_UB(pp);
|
||||
pp += 12;
|
||||
src6 = LD_UB(src);
|
||||
src += 12;
|
||||
|
||||
SLDI_B2_0_UB(src2, src6, src3, src7, 3);
|
||||
SLDI_B2_0_UB(src2, src6, src4, src8, 6);
|
||||
SLDI_B2_0_UB(src2, src6, src5, src9, 9);
|
||||
src2 = __msa_ave_u_b(src2, src1);
|
||||
src6 += src2;
|
||||
src3 = __msa_ave_u_b(src3, src6);
|
||||
src7 += src3;
|
||||
src4 = __msa_ave_u_b(src4, src7);
|
||||
src8 += src4;
|
||||
src5 = __msa_ave_u_b(src5, src8);
|
||||
src9 += src5;
|
||||
src1 = src9;
|
||||
VSHF_B2_UB(src6, src7, src8, src9, mask0, mask0, dst0, dst1);
|
||||
dst0 = (v16u8) __msa_vshf_b(mask1, (v16i8) dst1, (v16i8) dst0);
|
||||
out0 = __msa_copy_s_d((v2i64) dst0, 0);
|
||||
out1 = __msa_copy_s_w((v4i32) dst0, 2);
|
||||
|
||||
SD(out0, nxt);
|
||||
nxt += 8;
|
||||
SW(out1, nxt);
|
||||
nxt += 4;
|
||||
}
|
||||
}
|
||||
|
||||
void png_read_filter_row_paeth4_msa(png_row_infop row_info,
|
||||
png_bytep row,
|
||||
png_const_bytep prev_row)
|
||||
{
|
||||
int32_t count, rp_end;
|
||||
png_bytep nxt;
|
||||
png_const_bytep prev_nxt;
|
||||
int32_t inp0, inp1, res0;
|
||||
v16u8 src0, src1, src2, src3, src4, src5, src6, src7, src8, src9;
|
||||
v16u8 src10, src11, src12, src13, dst0, dst1;
|
||||
v8i16 vec0, vec1, vec2;
|
||||
v16u8 zero = { 0 };
|
||||
|
||||
nxt = row;
|
||||
prev_nxt = prev_row;
|
||||
|
||||
inp0 = LW(nxt);
|
||||
inp1 = LW(prev_nxt);
|
||||
prev_nxt += 4;
|
||||
src0 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp0);
|
||||
src1 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp1);
|
||||
|
||||
src1 += src0;
|
||||
res0 = __msa_copy_s_w((v4i32) src1, 0);
|
||||
|
||||
SW(res0, nxt);
|
||||
nxt += 4;
|
||||
|
||||
/* Remainder */
|
||||
rp_end = row_info->rowbytes - 4;
|
||||
|
||||
for (count = 0; count < rp_end; count += 16)
|
||||
{
|
||||
src2 = LD_UB(prev_nxt);
|
||||
prev_nxt += 16;
|
||||
src6 = LD_UB(prev_row);
|
||||
prev_row += 16;
|
||||
src10 = LD_UB(nxt);
|
||||
|
||||
SLDI_B3_0_UB(src2, src6, src10, src3, src7, src11, 4);
|
||||
SLDI_B3_0_UB(src2, src6, src10, src4, src8, src12, 8);
|
||||
SLDI_B3_0_UB(src2, src6, src10, src5, src9, src13, 12);
|
||||
ILVR_B2_SH(src2, src6, src1, src6, vec0, vec1);
|
||||
HSUB_UB2_SH(vec0, vec1, vec0, vec1);
|
||||
vec2 = vec0 + vec1;
|
||||
ADD_ABS_H3_SH(vec0, vec1, vec2, vec0, vec1, vec2);
|
||||
CMP_AND_SELECT(vec0, vec1, vec2, src1, src2, src6, src10);
|
||||
ILVR_B2_SH(src3, src7, src10, src7, vec0, vec1);
|
||||
HSUB_UB2_SH(vec0, vec1, vec0, vec1);
|
||||
vec2 = vec0 + vec1;
|
||||
ADD_ABS_H3_SH(vec0, vec1, vec2, vec0, vec1, vec2);
|
||||
CMP_AND_SELECT(vec0, vec1, vec2, src10, src3, src7, src11);
|
||||
ILVR_B2_SH(src4, src8, src11, src8, vec0, vec1);
|
||||
HSUB_UB2_SH(vec0, vec1, vec0, vec1);
|
||||
vec2 = vec0 + vec1;
|
||||
ADD_ABS_H3_SH(vec0, vec1, vec2, vec0, vec1, vec2);
|
||||
CMP_AND_SELECT(vec0, vec1, vec2, src11, src4, src8, src12);
|
||||
ILVR_B2_SH(src5, src9, src12, src9, vec0, vec1);
|
||||
HSUB_UB2_SH(vec0, vec1, vec0, vec1);
|
||||
vec2 = vec0 + vec1;
|
||||
ADD_ABS_H3_SH(vec0, vec1, vec2, vec0, vec1, vec2);
|
||||
CMP_AND_SELECT(vec0, vec1, vec2, src12, src5, src9, src13);
|
||||
src1 = src13;
|
||||
ILVEV_W2_UB(src10, src11, src12, src1, dst0, dst1);
|
||||
dst0 = (v16u8) __msa_pckev_d((v2i64) dst1, (v2i64) dst0);
|
||||
|
||||
ST_UB(dst0, nxt);
|
||||
nxt += 16;
|
||||
}
|
||||
}
|
||||
|
||||
void png_read_filter_row_paeth3_msa(png_row_infop row_info,
|
||||
png_bytep row,
|
||||
png_const_bytep prev_row)
|
||||
{
|
||||
int32_t count, rp_end;
|
||||
png_bytep nxt;
|
||||
png_const_bytep prev_nxt;
|
||||
int64_t out0;
|
||||
int32_t inp0, inp1, out1;
|
||||
int16_t out2;
|
||||
v16u8 src0, src1, src2, src3, src4, src5, src6, src7, src8, src9, dst0, dst1;
|
||||
v16u8 src10, src11, src12, src13;
|
||||
v8i16 vec0, vec1, vec2;
|
||||
v16u8 zero = { 0 };
|
||||
v16i8 mask0 = { 0, 1, 2, 16, 17, 18, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 };
|
||||
v16i8 mask1 = { 0, 1, 2, 3, 4, 5, 16, 17, 18, 19, 20, 21, 0, 0, 0, 0 };
|
||||
|
||||
nxt = row;
|
||||
prev_nxt = prev_row;
|
||||
|
||||
inp0 = LW(nxt);
|
||||
inp1 = LW(prev_nxt);
|
||||
prev_nxt += 3;
|
||||
src0 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp0);
|
||||
src1 = (v16u8) __msa_insert_w((v4i32) zero, 0, inp1);
|
||||
|
||||
src1 += src0;
|
||||
out2 = __msa_copy_s_h((v8i16) src1, 0);
|
||||
|
||||
SH(out2, nxt);
|
||||
nxt += 2;
|
||||
nxt[0] = src1[2];
|
||||
nxt++;
|
||||
|
||||
/* Remainder */
|
||||
rp_end = row_info->rowbytes - 3;
|
||||
|
||||
for (count = 0; count < rp_end; count += 12)
|
||||
{
|
||||
src2 = LD_UB(prev_nxt);
|
||||
prev_nxt += 12;
|
||||
src6 = LD_UB(prev_row);
|
||||
prev_row += 12;
|
||||
src10 = LD_UB(nxt);
|
||||
|
||||
SLDI_B3_0_UB(src2, src6, src10, src3, src7, src11, 3);
|
||||
SLDI_B3_0_UB(src2, src6, src10, src4, src8, src12, 6);
|
||||
SLDI_B3_0_UB(src2, src6, src10, src5, src9, src13, 9);
|
||||
ILVR_B2_SH(src2, src6, src1, src6, vec0, vec1);
|
||||
HSUB_UB2_SH(vec0, vec1, vec0, vec1);
|
||||
vec2 = vec0 + vec1;
|
||||
ADD_ABS_H3_SH(vec0, vec1, vec2, vec0, vec1, vec2);
|
||||
CMP_AND_SELECT(vec0, vec1, vec2, src1, src2, src6, src10);
|
||||
ILVR_B2_SH(src3, src7, src10, src7, vec0, vec1);
|
||||
HSUB_UB2_SH(vec0, vec1, vec0, vec1);
|
||||
vec2 = vec0 + vec1;
|
||||
ADD_ABS_H3_SH(vec0, vec1, vec2, vec0, vec1, vec2);
|
||||
CMP_AND_SELECT(vec0, vec1, vec2, src10, src3, src7, src11);
|
||||
ILVR_B2_SH(src4, src8, src11, src8, vec0, vec1);
|
||||
HSUB_UB2_SH(vec0, vec1, vec0, vec1);
|
||||
vec2 = vec0 + vec1;
|
||||
ADD_ABS_H3_SH(vec0, vec1, vec2, vec0, vec1, vec2);
|
||||
CMP_AND_SELECT(vec0, vec1, vec2, src11, src4, src8, src12);
|
||||
ILVR_B2_SH(src5, src9, src12, src9, vec0, vec1);
|
||||
HSUB_UB2_SH(vec0, vec1, vec0, vec1);
|
||||
vec2 = vec0 + vec1;
|
||||
ADD_ABS_H3_SH(vec0, vec1, vec2, vec0, vec1, vec2);
|
||||
CMP_AND_SELECT(vec0, vec1, vec2, src12, src5, src9, src13);
|
||||
src1 = src13;
|
||||
VSHF_B2_UB(src10, src11, src12, src13, mask0, mask0, dst0, dst1);
|
||||
dst0 = (v16u8) __msa_vshf_b(mask1, (v16i8) dst1, (v16i8) dst0);
|
||||
out0 = __msa_copy_s_d((v2i64) dst0, 0);
|
||||
out1 = __msa_copy_s_w((v4i32) dst0, 2);
|
||||
|
||||
SD(out0, nxt);
|
||||
nxt += 8;
|
||||
SW(out1, nxt);
|
||||
nxt += 4;
|
||||
}
|
||||
}
|
||||
|
||||
#endif /* PNG_MIPS_MSA_OPT > 0 */
|
||||
#endif /* PNG_MIPS_MSA_IMPLEMENTATION == 1 (intrinsics) */
|
||||
#endif /* READ */
|
||||
Vendored
+127
@@ -0,0 +1,127 @@
|
||||
|
||||
/* mips_init.c - MSA optimised filter functions
|
||||
*
|
||||
* Copyright (c) 2018 Cosmin Truta
|
||||
* Copyright (c) 2016 Glenn Randers-Pehrson
|
||||
* Written by Mandar Sahastrabuddhe, 2016.
|
||||
*
|
||||
* This code is released under the libpng license.
|
||||
* For conditions of distribution and use, see the disclaimer
|
||||
* and license in png.h
|
||||
*/
|
||||
|
||||
/* Below, after checking __linux__, various non-C90 POSIX 1003.1 functions are
|
||||
* called.
|
||||
*/
|
||||
#define _POSIX_SOURCE 1
|
||||
|
||||
#include <stdio.h>
|
||||
#include "../pngpriv.h"
|
||||
|
||||
#ifdef PNG_READ_SUPPORTED
|
||||
|
||||
#if PNG_MIPS_MSA_OPT > 0
|
||||
#ifdef PNG_MIPS_MSA_CHECK_SUPPORTED /* Do run-time checks */
|
||||
/* WARNING: it is strongly recommended that you do not build libpng with
|
||||
* run-time checks for CPU features if at all possible. In the case of the MIPS
|
||||
* MSA instructions there is no processor-specific way of detecting the
|
||||
* presence of the required support, therefore run-time detection is extremely
|
||||
* OS specific.
|
||||
*
|
||||
* You may set the macro PNG_MIPS_MSA_FILE to the file name of file containing
|
||||
* a fragment of C source code which defines the png_have_msa function. There
|
||||
* are a number of implementations in contrib/mips-msa, but the only one that
|
||||
* has partial support is contrib/mips-msa/linux.c - a generic Linux
|
||||
* implementation which reads /proc/cpufino.
|
||||
*/
|
||||
#ifndef PNG_MIPS_MSA_FILE
|
||||
# ifdef __linux__
|
||||
# define PNG_MIPS_MSA_FILE "contrib/mips-msa/linux.c"
|
||||
# endif
|
||||
#endif
|
||||
|
||||
#ifdef PNG_MIPS_MSA_FILE
|
||||
|
||||
#include <signal.h> /* for sig_atomic_t */
|
||||
static int png_have_msa(png_structp png_ptr);
|
||||
#include PNG_MIPS_MSA_FILE
|
||||
|
||||
#else /* PNG_MIPS_MSA_FILE */
|
||||
# error "PNG_MIPS_MSA_FILE undefined: no support for run-time MIPS MSA checks"
|
||||
#endif /* PNG_MIPS_MSA_FILE */
|
||||
#endif /* PNG_MIPS_MSA_CHECK_SUPPORTED */
|
||||
|
||||
#ifndef PNG_ALIGNED_MEMORY_SUPPORTED
|
||||
# error "ALIGNED_MEMORY is required; set: -DPNG_ALIGNED_MEMORY_SUPPORTED"
|
||||
#endif
|
||||
|
||||
void
|
||||
png_init_filter_functions_msa(png_structp pp, unsigned int bpp)
|
||||
{
|
||||
/* The switch statement is compiled in for MIPS_MSA_API, the call to
|
||||
* png_have_msa is compiled in for MIPS_MSA_CHECK. If both are defined
|
||||
* the check is only performed if the API has not set the MSA option on
|
||||
* or off explicitly. In this case the check controls what happens.
|
||||
*/
|
||||
|
||||
#ifdef PNG_MIPS_MSA_API_SUPPORTED
|
||||
switch ((pp->options >> PNG_MIPS_MSA) & 3)
|
||||
{
|
||||
case PNG_OPTION_UNSET:
|
||||
/* Allow the run-time check to execute if it has been enabled -
|
||||
* thus both API and CHECK can be turned on. If it isn't supported
|
||||
* this case will fall through to the 'default' below, which just
|
||||
* returns.
|
||||
*/
|
||||
#ifdef PNG_MIPS_MSA_CHECK_SUPPORTED
|
||||
{
|
||||
static volatile sig_atomic_t no_msa = -1; /* not checked */
|
||||
|
||||
if (no_msa < 0)
|
||||
no_msa = !png_have_msa(pp);
|
||||
|
||||
if (no_msa)
|
||||
return;
|
||||
}
|
||||
#endif /* PNG_MIPS_MSA_CHECK_SUPPORTED */
|
||||
break;
|
||||
|
||||
default: /* OFF or INVALID */
|
||||
return;
|
||||
|
||||
case PNG_OPTION_ON:
|
||||
/* Option turned on */
|
||||
break;
|
||||
}
|
||||
/* IMPORTANT: any new external functions used here must be declared using
|
||||
* PNG_INTERNAL_FUNCTION in ../pngpriv.h. This is required so that the
|
||||
* 'prefix' option to configure works:
|
||||
*
|
||||
* ./configure --with-libpng-prefix=foobar_
|
||||
*
|
||||
* Verify you have got this right by running the above command, doing a build
|
||||
* and examining pngprefix.h; it must contain a #define for every external
|
||||
* function you add. (Notice that this happens automatically for the
|
||||
* initialization function.)
|
||||
*/
|
||||
pp->read_filter[PNG_FILTER_VALUE_UP-1] = png_read_filter_row_up_msa;
|
||||
|
||||
if (bpp == 3)
|
||||
{
|
||||
pp->read_filter[PNG_FILTER_VALUE_SUB-1] = png_read_filter_row_sub3_msa;
|
||||
pp->read_filter[PNG_FILTER_VALUE_AVG-1] = png_read_filter_row_avg3_msa;
|
||||
pp->read_filter[PNG_FILTER_VALUE_PAETH-1] = png_read_filter_row_paeth3_msa;
|
||||
}
|
||||
else if (bpp == 4)
|
||||
{
|
||||
pp->read_filter[PNG_FILTER_VALUE_SUB-1] = png_read_filter_row_sub4_msa;
|
||||
pp->read_filter[PNG_FILTER_VALUE_AVG-1] = png_read_filter_row_avg4_msa;
|
||||
pp->read_filter[PNG_FILTER_VALUE_PAETH-1] = png_read_filter_row_paeth4_msa;
|
||||
}
|
||||
#else
|
||||
(void)pp;
|
||||
(void)bpp;
|
||||
#endif /* PNG_MIPS_MSA_API_SUPPORTED */
|
||||
}
|
||||
#endif /* PNG_MIPS_MSA_OPT > 0 */
|
||||
#endif /* READ */
|
||||
+53
@@ -0,0 +1,53 @@
|
||||
diff --git a/3rdparty/libpng/mips/mips_init.c b/3rdparty/libpng/mips/mips_init.c
|
||||
index 8dd283deef..6a061cccfa 100644
|
||||
--- a/3rdparty/libpng/mips/mips_init.c
|
||||
+++ b/3rdparty/libpng/mips/mips_init.c
|
||||
@@ -73,7 +73,6 @@ png_init_filter_functions_msa(png_structp pp, unsigned int bpp)
|
||||
* this case will fall through to the 'default' below, which just
|
||||
* returns.
|
||||
*/
|
||||
-#endif /* PNG_MIPS_MSA_API_SUPPORTED */
|
||||
#ifdef PNG_MIPS_MSA_CHECK_SUPPORTED
|
||||
{
|
||||
static volatile sig_atomic_t no_msa = -1; /* not checked */
|
||||
@@ -84,12 +83,9 @@ png_init_filter_functions_msa(png_structp pp, unsigned int bpp)
|
||||
if (no_msa)
|
||||
return;
|
||||
}
|
||||
-#ifdef PNG_MIPS_MSA_API_SUPPORTED
|
||||
- break;
|
||||
-#endif
|
||||
#endif /* PNG_MIPS_MSA_CHECK_SUPPORTED */
|
||||
+ break;
|
||||
|
||||
-#ifdef PNG_MIPS_MSA_API_SUPPORTED
|
||||
default: /* OFF or INVALID */
|
||||
return;
|
||||
|
||||
@@ -97,8 +93,6 @@ png_init_filter_functions_msa(png_structp pp, unsigned int bpp)
|
||||
/* Option turned on */
|
||||
break;
|
||||
}
|
||||
-#endif
|
||||
-
|
||||
/* IMPORTANT: any new external functions used here must be declared using
|
||||
* PNG_INTERNAL_FUNCTION in ../pngpriv.h. This is required so that the
|
||||
* 'prefix' option to configure works:
|
||||
@@ -118,13 +112,16 @@ png_init_filter_functions_msa(png_structp pp, unsigned int bpp)
|
||||
pp->read_filter[PNG_FILTER_VALUE_AVG-1] = png_read_filter_row_avg3_msa;
|
||||
pp->read_filter[PNG_FILTER_VALUE_PAETH-1] = png_read_filter_row_paeth3_msa;
|
||||
}
|
||||
-
|
||||
else if (bpp == 4)
|
||||
{
|
||||
pp->read_filter[PNG_FILTER_VALUE_SUB-1] = png_read_filter_row_sub4_msa;
|
||||
pp->read_filter[PNG_FILTER_VALUE_AVG-1] = png_read_filter_row_avg4_msa;
|
||||
pp->read_filter[PNG_FILTER_VALUE_PAETH-1] = png_read_filter_row_paeth4_msa;
|
||||
}
|
||||
+#else
|
||||
+ (void)pp;
|
||||
+ (void)bpp;
|
||||
+#endif /* PNG_MIPS_MSA_API_SUPPORTED */
|
||||
}
|
||||
#endif /* PNG_MIPS_MSA_OPT > 0 */
|
||||
#endif /* READ */
|
||||
@@ -0,0 +1,22 @@
|
||||
diff --git a/3rdparty/libwebp/src/dsp/msa_macro.h b/3rdparty/libwebp/src/dsp/msa_macro.h
|
||||
index de026a1d9e..a16c0bb300 100644
|
||||
--- a/3rdparty/libwebp/src/dsp/msa_macro.h
|
||||
+++ b/3rdparty/libwebp/src/dsp/msa_macro.h
|
||||
@@ -73,7 +73,7 @@
|
||||
static inline TYPE FUNC_NAME(const void* const psrc) { \
|
||||
const uint8_t* const psrc_m = (const uint8_t*)psrc; \
|
||||
TYPE val_m; \
|
||||
- asm volatile ( \
|
||||
+ __asm__ volatile ( \
|
||||
"" #INSTR " %[val_m], %[psrc_m] \n\t" \
|
||||
: [val_m] "=r" (val_m) \
|
||||
: [psrc_m] "m" (*psrc_m)); \
|
||||
@@ -86,7 +86,7 @@
|
||||
static inline void FUNC_NAME(TYPE val, void* const pdst) { \
|
||||
uint8_t* const pdst_m = (uint8_t*)pdst; \
|
||||
TYPE val_m = val; \
|
||||
- asm volatile ( \
|
||||
+ __asm__ volatile ( \
|
||||
" " #INSTR " %[val_m], %[pdst_m] \n\t" \
|
||||
: [pdst_m] "=m" (*pdst_m) \
|
||||
: [val_m] "r" (val_m)); \
|
||||
Vendored
+2
-2
@@ -73,7 +73,7 @@
|
||||
static inline TYPE FUNC_NAME(const void* const psrc) { \
|
||||
const uint8_t* const psrc_m = (const uint8_t*)psrc; \
|
||||
TYPE val_m; \
|
||||
asm volatile ( \
|
||||
__asm__ volatile ( \
|
||||
"" #INSTR " %[val_m], %[psrc_m] \n\t" \
|
||||
: [val_m] "=r" (val_m) \
|
||||
: [psrc_m] "m" (*psrc_m)); \
|
||||
@@ -86,7 +86,7 @@
|
||||
static inline void FUNC_NAME(TYPE val, void* const pdst) { \
|
||||
uint8_t* const pdst_m = (uint8_t*)pdst; \
|
||||
TYPE val_m = val; \
|
||||
asm volatile ( \
|
||||
__asm__ volatile ( \
|
||||
" " #INSTR " %[val_m], %[pdst_m] \n\t" \
|
||||
: [pdst_m] "=m" (*pdst_m) \
|
||||
: [val_m] "r" (val_m)); \
|
||||
|
||||
+11
-5
@@ -477,6 +477,7 @@ OCV_OPTION(CV_DISABLE_OPTIMIZATION "Disable explicit optimized code (dispatch
|
||||
OCV_OPTION(CV_TRACE "Enable OpenCV code trace" ON)
|
||||
OCV_OPTION(OPENCV_GENERATE_SETUPVARS "Generate setup_vars* scripts" ON IF (NOT ANDROID AND NOT APPLE_FRAMEWORK) )
|
||||
OCV_OPTION(ENABLE_CONFIG_VERIFICATION "Fail build if actual configuration doesn't match requested (WITH_XXX != HAVE_XXX)" OFF)
|
||||
OCV_OPTION(OPENCV_ENABLE_MEMALIGN "Enable posix_memalign or memalign usage" ON)
|
||||
|
||||
OCV_OPTION(ENABLE_PYLINT "Add target with Pylint checks" (BUILD_DOCS OR BUILD_EXAMPLES) IF (NOT CMAKE_CROSSCOMPILING AND NOT APPLE_FRAMEWORK) )
|
||||
OCV_OPTION(ENABLE_FLAKE8 "Add target with Python flake8 checker" (BUILD_DOCS OR BUILD_EXAMPLES) IF (NOT CMAKE_CROSSCOMPILING AND NOT APPLE_FRAMEWORK) )
|
||||
@@ -625,10 +626,15 @@ if(UNIX)
|
||||
set(HAVE_PTHREAD 1)
|
||||
endif()
|
||||
|
||||
CHECK_SYMBOL_EXISTS(posix_memalign stdlib.h HAVE_POSIX_MEMALIGN)
|
||||
CHECK_INCLUDE_FILE(malloc.h HAVE_MALLOC_H)
|
||||
if(HAVE_MALLOC_H)
|
||||
CHECK_SYMBOL_EXISTS(memalign malloc.h HAVE_MEMALIGN)
|
||||
if(OPENCV_ENABLE_MEMALIGN)
|
||||
CHECK_SYMBOL_EXISTS(posix_memalign stdlib.h HAVE_POSIX_MEMALIGN)
|
||||
CHECK_INCLUDE_FILE(malloc.h HAVE_MALLOC_H)
|
||||
if(HAVE_MALLOC_H)
|
||||
CHECK_SYMBOL_EXISTS(memalign malloc.h HAVE_MEMALIGN)
|
||||
endif()
|
||||
# TODO:
|
||||
# - _aligned_malloc() on Win32
|
||||
# - std::aligned_alloc() C++17 / C11
|
||||
endif()
|
||||
endif()
|
||||
|
||||
@@ -1519,7 +1525,7 @@ if(FLAKE8_FOUND AND FLAKE8_EXECUTABLE)
|
||||
endif()
|
||||
|
||||
# ========================== java ==========================
|
||||
if(BUILD_JAVA OR BUILD_opencv_java)
|
||||
if(BUILD_JAVA)
|
||||
status("")
|
||||
status(" Java:" BUILD_FAT_JAVA_LIB THEN "export all functions" ELSE "")
|
||||
status(" ant:" ANT_EXECUTABLE THEN "${ANT_EXECUTABLE} (ver ${ANT_VERSION})" ELSE NO)
|
||||
|
||||
@@ -5,13 +5,15 @@
|
||||
# AVX / AVX2 / AVX_512F
|
||||
# FMA3
|
||||
#
|
||||
# AVX512 details: https://en.wikipedia.org/wiki/AVX-512#CPUs_with_AVX-512
|
||||
#
|
||||
# CPU features groups:
|
||||
# AVX512_COMMON (Common instructions AVX-512F/CD for all CPUs that support AVX-512)
|
||||
# AVX512_KNL (Knights Landing with AVX-512F/CD/ER/PF)
|
||||
# AVX512_KNM (Knights Mill with AVX-512F/CD/ER/PF/4FMAPS/4VNNIW/VPOPCNTDQ)
|
||||
# AVX512_SKX (Skylake-X with AVX-512F/CD/BW/DQ/VL)
|
||||
# AVX512_CNL (Cannon Lake with AVX-512F/CD/BW/DQ/VL/IFMA/VBMI)
|
||||
# AVX512_CEL (Cascade Lake with AVX-512F/CD/BW/DQ/VL/IFMA/VBMI/VNNI)
|
||||
# AVX512_CLX (Cascade Lake with AVX-512F/CD/BW/DQ/VL/VNNI)
|
||||
# AVX512_ICL (Ice Lake with AVX-512F/CD/BW/DQ/VL/IFMA/VBMI/VNNI/VBMI2/BITALG/VPOPCNTDQ/VPCLMULQDQ*/GFNI*/VAES*)
|
||||
|
||||
# ppc64le arch:
|
||||
@@ -43,8 +45,9 @@
|
||||
# CPU_{opt}_ENABLED_DEFAULT=ON/OFF - has compiler support without additional flag (CPU_BASELINE_DETECT=ON only)
|
||||
|
||||
set(CPU_ALL_OPTIMIZATIONS "SSE;SSE2;SSE3;SSSE3;SSE4_1;SSE4_2;POPCNT;AVX;FP16;AVX2;FMA3;AVX_512F")
|
||||
list(APPEND CPU_ALL_OPTIMIZATIONS "AVX512_COMMON;AVX512_KNL;AVX512_KNM;AVX512_SKX;AVX512_CNL;AVX512_CEL;AVX512_ICL")
|
||||
list(APPEND CPU_ALL_OPTIMIZATIONS "AVX512_COMMON;AVX512_KNL;AVX512_KNM;AVX512_SKX;AVX512_CNL;AVX512_CLX;AVX512_ICL")
|
||||
list(APPEND CPU_ALL_OPTIMIZATIONS NEON VFPV3 FP16)
|
||||
list(APPEND CPU_ALL_OPTIMIZATIONS MSA)
|
||||
list(APPEND CPU_ALL_OPTIMIZATIONS VSX VSX3)
|
||||
list(REMOVE_DUPLICATES CPU_ALL_OPTIMIZATIONS)
|
||||
|
||||
@@ -162,15 +165,15 @@ elseif(" ${CMAKE_CXX_FLAGS} " MATCHES " -march=native | -xHost | /QxHost ")
|
||||
endif()
|
||||
|
||||
if(X86 OR X86_64)
|
||||
ocv_update(CPU_KNOWN_OPTIMIZATIONS "SSE;SSE2;SSE3;SSSE3;SSE4_1;POPCNT;SSE4_2;FP16;FMA3;AVX;AVX2;AVX_512F;AVX512_COMMON;AVX512_KNL;AVX512_KNM;AVX512_SKX;AVX512_CNL;AVX512_CEL;AVX512_ICL")
|
||||
ocv_update(CPU_KNOWN_OPTIMIZATIONS "SSE;SSE2;SSE3;SSSE3;SSE4_1;POPCNT;SSE4_2;FP16;FMA3;AVX;AVX2;AVX_512F;AVX512_COMMON;AVX512_KNL;AVX512_KNM;AVX512_SKX;AVX512_CNL;AVX512_CLX;AVX512_ICL")
|
||||
|
||||
ocv_update(CPU_AVX512_COMMON_GROUP "AVX_512F;AVX_512CD")
|
||||
ocv_update(CPU_AVX512_KNL_GROUP "AVX512_COMMON;AVX512_KNL_EXTRA")
|
||||
ocv_update(CPU_AVX512_KNM_GROUP "AVX512_KNL;AVX512_KNM_EXTRA;AVX_512VPOPCNTDQ")
|
||||
ocv_update(CPU_AVX512_SKX_GROUP "AVX512_COMMON;AVX_512VL;AVX_512BW;AVX_512DQ")
|
||||
ocv_update(CPU_AVX512_CNL_GROUP "AVX512_SKX;AVX_512IFMA;AVX_512VBMI")
|
||||
ocv_update(CPU_AVX512_CEL_GROUP "AVX512_CNL;AVX_512VNNI")
|
||||
ocv_update(CPU_AVX512_ICL_GROUP "AVX512_CEL;AVX_512VBMI2;AVX_512BITALG;AVX_512VPOPCNTDQ") # ? VPCLMULQDQ, GFNI, VAES
|
||||
ocv_update(CPU_AVX512_CLX_GROUP "AVX512_SKX;AVX_512VNNI")
|
||||
ocv_update(CPU_AVX512_ICL_GROUP "AVX512_SKX;AVX_512IFMA;AVX_512VBMI;AVX_512VNNI;AVX_512VBMI2;AVX_512BITALG;AVX_512VPOPCNTDQ") # ? VPCLMULQDQ, GFNI, VAES
|
||||
|
||||
ocv_update(CPU_SSE_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_sse.cpp")
|
||||
ocv_update(CPU_SSE2_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_sse2.cpp")
|
||||
@@ -188,12 +191,12 @@ if(X86 OR X86_64)
|
||||
ocv_update(CPU_AVX512_KNM_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_avx512knm.cpp")
|
||||
ocv_update(CPU_AVX512_SKX_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_avx512skx.cpp")
|
||||
ocv_update(CPU_AVX512_CNL_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_avx512cnl.cpp")
|
||||
ocv_update(CPU_AVX512_CEL_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_avx512cel.cpp")
|
||||
ocv_update(CPU_AVX512_CLX_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_avx512clx.cpp")
|
||||
ocv_update(CPU_AVX512_ICL_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_avx512icl.cpp")
|
||||
|
||||
if(NOT OPENCV_CPU_OPT_IMPLIES_IGNORE)
|
||||
ocv_update(CPU_AVX512_ICL_IMPLIES "AVX512_CEL")
|
||||
ocv_update(CPU_AVX512_CEL_IMPLIES "AVX512_CNL")
|
||||
ocv_update(CPU_AVX512_ICL_IMPLIES "AVX512_SKX")
|
||||
ocv_update(CPU_AVX512_CLX_IMPLIES "AVX512_SKX")
|
||||
ocv_update(CPU_AVX512_CNL_IMPLIES "AVX512_SKX")
|
||||
ocv_update(CPU_AVX512_SKX_IMPLIES "AVX512_COMMON")
|
||||
ocv_update(CPU_AVX512_KNM_IMPLIES "AVX512_KNL")
|
||||
@@ -250,7 +253,7 @@ if(X86 OR X86_64)
|
||||
ocv_intel_compiler_optimization_option(AVX512_KNM "-xKNM" "/Qx:KNM")
|
||||
ocv_intel_compiler_optimization_option(AVX512_SKX "-xSKYLAKE-AVX512" "/Qx:SKYLAKE-AVX512")
|
||||
ocv_intel_compiler_optimization_option(AVX512_CNL "-xCANNONLAKE" "/Qx:CANNONLAKE")
|
||||
ocv_intel_compiler_optimization_option(AVX512_CEL "-xCASCADELAKE" "/Qx:CASCADELAKE")
|
||||
ocv_intel_compiler_optimization_option(AVX512_CLX "-xCASCADELAKE" "/Qx:CASCADELAKE")
|
||||
ocv_intel_compiler_optimization_option(AVX512_ICL "-xICELAKE-CLIENT" "/Qx:ICELAKE-CLIENT")
|
||||
elseif(CV_GCC OR CV_CLANG)
|
||||
ocv_update(CPU_AVX2_FLAGS_ON "-mavx2")
|
||||
@@ -339,6 +342,11 @@ elseif(ARM OR AARCH64)
|
||||
ocv_update(CPU_FP16_IMPLIES "NEON")
|
||||
set(CPU_BASELINE "NEON;FP16" CACHE STRING "${HELP_CPU_BASELINE}")
|
||||
endif()
|
||||
elseif(MIPS)
|
||||
ocv_update(CPU_MSA_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_msa.cpp")
|
||||
ocv_update(CPU_KNOWN_OPTIMIZATIONS "MSA")
|
||||
ocv_update(CPU_MSA_FLAGS_ON "-mmsa")
|
||||
set(CPU_BASELINE "MSA" CACHE STRING "${HELP_CPU_BASELINE}")
|
||||
elseif(PPC64LE)
|
||||
ocv_update(CPU_KNOWN_OPTIMIZATIONS "VSX;VSX3")
|
||||
ocv_update(CPU_VSX_TEST_FILE "${OpenCV_SOURCE_DIR}/cmake/checks/cpu_vsx.cpp")
|
||||
|
||||
@@ -100,6 +100,8 @@ elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(powerpc|ppc)64le")
|
||||
set(PPC64LE 1)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(powerpc|ppc)64")
|
||||
set(PPC64 1)
|
||||
elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(mips.*|MIPS.*)")
|
||||
set(MIPS 1)
|
||||
endif()
|
||||
|
||||
# Workaround for 32-bit operating systems on x86_64/aarch64 processor
|
||||
|
||||
@@ -22,6 +22,9 @@ set(OPENCV_DOWNLOAD_PATH "${OpenCV_SOURCE_DIR}/.cache" CACHE PATH "${HELP_OPENCV
|
||||
set(OPENCV_DOWNLOAD_LOG "${OpenCV_BINARY_DIR}/CMakeDownloadLog.txt")
|
||||
set(OPENCV_DOWNLOAD_WITH_CURL "${OpenCV_BINARY_DIR}/download_with_curl.sh")
|
||||
set(OPENCV_DOWNLOAD_WITH_WGET "${OpenCV_BINARY_DIR}/download_with_wget.sh")
|
||||
set(OPENCV_DOWNLOAD_TRIES_LIST 1 CACHE STRING "List of download tries") # a list
|
||||
set(OPENCV_DOWNLOAD_PARAMS INACTIVITY_TIMEOUT 60 TIMEOUT 600 CACHE STRING "Download parameters to be passed to file(DOWNLAOD ...)")
|
||||
mark_as_advanced(OPENCV_DOWNLOAD_TRIES_LIST OPENCV_DOWNLOAD_PARAMS)
|
||||
|
||||
# Init download cache directory and log file and helper scripts
|
||||
if(NOT EXISTS "${OPENCV_DOWNLOAD_PATH}")
|
||||
@@ -154,11 +157,17 @@ function(ocv_download)
|
||||
# Download
|
||||
if(NOT EXISTS "${CACHE_CANDIDATE}")
|
||||
ocv_download_log("#cmake_download \"${CACHE_CANDIDATE}\" \"${DL_URL}\"")
|
||||
file(DOWNLOAD "${DL_URL}" "${CACHE_CANDIDATE}"
|
||||
INACTIVITY_TIMEOUT 60
|
||||
TIMEOUT 600
|
||||
STATUS status
|
||||
LOG __log)
|
||||
foreach(try ${OPENCV_DOWNLOAD_TRIES_LIST})
|
||||
ocv_download_log("#try ${try}")
|
||||
file(DOWNLOAD "${DL_URL}" "${CACHE_CANDIDATE}"
|
||||
STATUS status
|
||||
LOG __log
|
||||
${OPENCV_DOWNLOAD_PARAMS})
|
||||
if(status EQUAL 0)
|
||||
break()
|
||||
endif()
|
||||
message(STATUS "Try ${try} failed")
|
||||
endforeach()
|
||||
if(NOT OPENCV_SKIP_FILE_DOWNLOAD_DUMP) # workaround problem with old CMake versions: "Invalid escape sequence"
|
||||
string(LENGTH "${__log}" __log_length)
|
||||
if(__log_length LESS 65536)
|
||||
@@ -195,8 +204,8 @@ For details please refer to the download log file:
|
||||
${OPENCV_DOWNLOAD_LOG}
|
||||
")
|
||||
# write helper scripts for failed downloads
|
||||
file(APPEND "${OPENCV_DOWNLOAD_WITH_CURL}" "curl --output \"${CACHE_CANDIDATE}\" \"${DL_URL}\"\n")
|
||||
file(APPEND "${OPENCV_DOWNLOAD_WITH_WGET}" "wget -O \"${CACHE_CANDIDATE}\" \"${DL_URL}\"\n")
|
||||
file(APPEND "${OPENCV_DOWNLOAD_WITH_CURL}" "curl --create-dirs --output \"${CACHE_CANDIDATE}\" \"${DL_URL}\"\n")
|
||||
file(APPEND "${OPENCV_DOWNLOAD_WITH_WGET}" "mkdir -p $(dirname ${CACHE_CANDIDATE}) && wget -O \"${CACHE_CANDIDATE}\" \"${DL_URL}\"\n")
|
||||
return()
|
||||
endif()
|
||||
|
||||
|
||||
+23
-14
@@ -29,6 +29,16 @@ macro(ippiw_debugmsg MESSAGE)
|
||||
message(STATUS "${MESSAGE}")
|
||||
endif()
|
||||
endmacro()
|
||||
|
||||
macro(ippiw_done)
|
||||
foreach(__file ${IPP_IW_LICENSE_FILES})
|
||||
if(EXISTS "${__file}")
|
||||
ocv_install_3rdparty_licenses(ippiw "${__file}")
|
||||
endif()
|
||||
endforeach()
|
||||
return()
|
||||
endmacro()
|
||||
|
||||
file(TO_CMAKE_PATH "${IPPROOT}" IPPROOT)
|
||||
|
||||
# This function detects Intel IPP Integration Wrappers version by analyzing .h file
|
||||
@@ -81,7 +91,7 @@ macro(ippiw_setup PATH BUILD)
|
||||
if(EXISTS "${FILE}")
|
||||
set(HAVE_IPP_IW_LL 1)
|
||||
endif()
|
||||
return()
|
||||
ippiw_done()
|
||||
else()
|
||||
ippiw_debugmsg("sources\tno")
|
||||
endif()
|
||||
@@ -120,7 +130,7 @@ macro(ippiw_setup PATH BUILD)
|
||||
if(EXISTS "${FILE}")
|
||||
set(HAVE_IPP_IW_LL 1)
|
||||
endif()
|
||||
return()
|
||||
ippiw_done()
|
||||
else()
|
||||
ippiw_debugmsg("binaries\tno")
|
||||
endif()
|
||||
@@ -147,14 +157,12 @@ if(BUILD_IPP_IW)
|
||||
ippiw_setup("${OpenCV_SOURCE_DIR}/3rdparty/ippiw" 1)
|
||||
|
||||
set(IPPIW_ROOT "${IPPROOT}/../iw")
|
||||
ocv_install_3rdparty_licenses(ippiw
|
||||
"${IPPIW_ROOT}/../support.txt"
|
||||
"${IPPIW_ROOT}/../third-party-programs.txt")
|
||||
if(WIN32)
|
||||
ocv_install_3rdparty_licenses(ippiw "${IPPIW_ROOT}/../EULA.rtf")
|
||||
else()
|
||||
ocv_install_3rdparty_licenses(ippiw "${IPPIW_ROOT}/../EULA.txt")
|
||||
endif()
|
||||
set(IPP_IW_LICENSE_FILES ${IPP_IW_LICENSE_FILES_EXTRA}
|
||||
"${IPPIW_ROOT}/../support.txt"
|
||||
"${IPPIW_ROOT}/../third-party-programs.txt"
|
||||
"${IPPIW_ROOT}/../EULA.rtf"
|
||||
"${IPPIW_ROOT}/../EULA.txt"
|
||||
)
|
||||
|
||||
# Package sources
|
||||
get_filename_component(__PATH "${IPPROOT}/../iw/" ABSOLUTE)
|
||||
@@ -167,10 +175,11 @@ if(BUILD_IPP_IW)
|
||||
include("${OpenCV_SOURCE_DIR}/3rdparty/ippicv/ippicv.cmake")
|
||||
download_ippicv(TEMP_ROOT)
|
||||
set(IPPIW_ROOT "${TEMP_ROOT}/iw/")
|
||||
ocv_install_3rdparty_licenses(ippiw
|
||||
"${IPPIW_ROOT}/../EULA.txt"
|
||||
"${IPPIW_ROOT}/../support.txt"
|
||||
"${IPPIW_ROOT}/../third-party-programs.txt")
|
||||
set(IPP_IW_LICENSE_FILES ${IPP_IW_LICENSE_FILES_EXTRA}
|
||||
"${IPPIW_ROOT}/../EULA.txt"
|
||||
"${IPPIW_ROOT}/../support.txt"
|
||||
"${IPPIW_ROOT}/../third-party-programs.txt"
|
||||
)
|
||||
|
||||
ippiw_setup("${IPPIW_ROOT}" 1)
|
||||
endif()
|
||||
|
||||
@@ -123,7 +123,9 @@ endif()
|
||||
# --------------------------------------------------------------------------------------------
|
||||
if(WIN32)
|
||||
if(CMAKE_HOST_SYSTEM_NAME MATCHES Windows AND NOT OPENCV_SKIP_CMAKE_ROOT_CONFIG)
|
||||
ocv_gen_config("${CMAKE_BINARY_DIR}/win-install" "${OPENCV_LIB_INSTALL_PATH}" "OpenCVConfig.root-WIN32.cmake.in")
|
||||
ocv_gen_config("${CMAKE_BINARY_DIR}/win-install"
|
||||
"${OPENCV_INSTALL_BINARIES_PREFIX}${OPENCV_INSTALL_BINARIES_SUFFIX}"
|
||||
"OpenCVConfig.root-WIN32.cmake.in")
|
||||
else()
|
||||
ocv_gen_config("${CMAKE_BINARY_DIR}/win-install" "" "")
|
||||
endif()
|
||||
|
||||
@@ -23,15 +23,15 @@ if(ANDROID)
|
||||
elseif(WIN32 AND CMAKE_HOST_SYSTEM_NAME MATCHES Windows)
|
||||
|
||||
if(DEFINED OpenCV_RUNTIME AND DEFINED OpenCV_ARCH)
|
||||
set(_prefix "${OpenCV_ARCH}/${OpenCV_RUNTIME}/")
|
||||
ocv_update(OPENCV_INSTALL_BINARIES_PREFIX "${OpenCV_ARCH}/${OpenCV_RUNTIME}/")
|
||||
else()
|
||||
message(STATUS "Can't detect runtime and/or arch")
|
||||
set(_prefix "")
|
||||
ocv_update(OPENCV_INSTALL_BINARIES_PREFIX "")
|
||||
endif()
|
||||
if(OpenCV_STATIC)
|
||||
set(_suffix "staticlib")
|
||||
ocv_update(OPENCV_INSTALL_BINARIES_SUFFIX "staticlib")
|
||||
else()
|
||||
set(_suffix "lib")
|
||||
ocv_update(OPENCV_INSTALL_BINARIES_SUFFIX "lib")
|
||||
endif()
|
||||
if(INSTALL_CREATE_DISTRIB)
|
||||
set(_jni_suffix "/${OpenCV_ARCH}")
|
||||
@@ -39,12 +39,12 @@ elseif(WIN32 AND CMAKE_HOST_SYSTEM_NAME MATCHES Windows)
|
||||
set(_jni_suffix "")
|
||||
endif()
|
||||
|
||||
ocv_update(OPENCV_BIN_INSTALL_PATH "${_prefix}bin")
|
||||
ocv_update(OPENCV_BIN_INSTALL_PATH "${OPENCV_INSTALL_BINARIES_PREFIX}bin")
|
||||
ocv_update(OPENCV_TEST_INSTALL_PATH "${OPENCV_BIN_INSTALL_PATH}")
|
||||
ocv_update(OPENCV_SAMPLES_BIN_INSTALL_PATH "${_prefix}samples")
|
||||
ocv_update(OPENCV_LIB_INSTALL_PATH "${_prefix}${_suffix}")
|
||||
ocv_update(OPENCV_SAMPLES_BIN_INSTALL_PATH "${OPENCV_INSTALL_BINARIES_PREFIX}samples")
|
||||
ocv_update(OPENCV_LIB_INSTALL_PATH "${OPENCV_INSTALL_BINARIES_PREFIX}${OPENCV_INSTALL_BINARIES_SUFFIX}")
|
||||
ocv_update(OPENCV_LIB_ARCHIVE_INSTALL_PATH "${OPENCV_LIB_INSTALL_PATH}")
|
||||
ocv_update(OPENCV_3P_LIB_INSTALL_PATH "${OPENCV_LIB_INSTALL_PATH}")
|
||||
ocv_update(OPENCV_3P_LIB_INSTALL_PATH "${OPENCV_INSTALL_BINARIES_PREFIX}staticlib")
|
||||
ocv_update(OPENCV_CONFIG_INSTALL_PATH ".")
|
||||
ocv_update(OPENCV_INCLUDE_INSTALL_PATH "include")
|
||||
ocv_update(OPENCV_OTHER_INSTALL_PATH "etc")
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
static int test()
|
||||
{
|
||||
std::atomic<int> x;
|
||||
std::atomic<long long> x;
|
||||
return x;
|
||||
}
|
||||
|
||||
|
||||
@@ -6,5 +6,6 @@ void test()
|
||||
{
|
||||
int data[8] = {0,0,0,0, 0,0,0,0};
|
||||
__m256i a = _mm256_loadu_si256((const __m256i *)data);
|
||||
__m256i b = _mm256_bslli_epi128(a, 1); // available in GCC 4.9.3+
|
||||
}
|
||||
int main() { return 0; }
|
||||
|
||||
@@ -3,9 +3,9 @@
|
||||
void test()
|
||||
{
|
||||
__m512i a, b, c;
|
||||
a = _mm512_dpwssd_epi32(a, b, c);
|
||||
a = _mm512_dpwssd_epi32(a, b, c); // VNNI
|
||||
}
|
||||
#else
|
||||
#error "AVX512-CEL is not supported"
|
||||
#error "AVX512-CLX is not supported"
|
||||
#endif
|
||||
int main() { return 0; }
|
||||
int main() { return 0; }
|
||||
@@ -3,9 +3,10 @@
|
||||
void test()
|
||||
{
|
||||
__m512i a, b, c;
|
||||
a = _mm512_popcnt_epi8(a);
|
||||
a = _mm512_shrdv_epi64(a, b, c);
|
||||
a = _mm512_popcnt_epi64(a);
|
||||
a = _mm512_popcnt_epi8(a); // BITALG
|
||||
a = _mm512_shrdv_epi64(a, b, c); // VBMI2
|
||||
a = _mm512_popcnt_epi64(a); // VPOPCNTDQ
|
||||
a = _mm512_dpwssd_epi32(a, b, c); // VNNI
|
||||
}
|
||||
#else
|
||||
#error "AVX512-ICL is not supported"
|
||||
|
||||
Executable
+23
@@ -0,0 +1,23 @@
|
||||
#include <stdio.h>
|
||||
|
||||
#if defined(__mips_msa)
|
||||
# include <msa.h>
|
||||
# define CV_MSA 1
|
||||
#endif
|
||||
|
||||
#if defined CV_MSA
|
||||
int test()
|
||||
{
|
||||
const float src[] = { 0.0f, 0.0f, 0.0f, 0.0f };
|
||||
v4f32 val = (v4f32)__msa_ld_w((const float*)(src), 0);
|
||||
return __msa_copy_s_w(__builtin_msa_ftint_s_w (val), 0);
|
||||
}
|
||||
#else
|
||||
#error "MSA is not supported"
|
||||
#endif
|
||||
|
||||
int main()
|
||||
{
|
||||
printf("%d\n", test());
|
||||
return 0;
|
||||
}
|
||||
@@ -107,6 +107,90 @@ Building OpenCV.js from Source
|
||||
@note
|
||||
It requires `node` installed in your development environment.
|
||||
|
||||
-# [optional] To build `opencv.js` with threads optimization, append `--threads` option.
|
||||
|
||||
For example:
|
||||
@code{.bash}
|
||||
python ./platforms/js/build_js.py build_js --build_wasm --threads
|
||||
@endcode
|
||||
|
||||
The default threads number is the logic core number of your device. You can use `cv.parallel_pthreads_set_threads_num(number)` to set threads number by yourself and use `cv.parallel_pthreads_get_threads_num()` to get the current threads number.
|
||||
|
||||
@note
|
||||
You should build wasm version of `opencv.js` if you want to enable this optimization. And the threads optimization only works in browser, not in node.js. You need to enable the `WebAssembly threads support` feature first with your browser. For example, if you use Chrome, please enable this flag in chrome://flags.
|
||||
|
||||
-# [optional] To build `opencv.js` with wasm simd optimization, append `--simd` option.
|
||||
|
||||
For example:
|
||||
@code{.bash}
|
||||
python ./platforms/js/build_js.py build_js --build_wasm --simd
|
||||
@endcode
|
||||
|
||||
The simd optimization is experimental as wasm simd is still in development.
|
||||
|
||||
@note
|
||||
Now only emscripten LLVM upstream backend supports wasm simd, refering to https://emscripten.org/docs/porting/simd.html. So you need to setup upstream backend environment with the following command first:
|
||||
@code{.bash}
|
||||
./emsdk update
|
||||
./emsdk install latest-upstream
|
||||
./emsdk activate latest-upstream
|
||||
source ./emsdk_env.sh
|
||||
@endcode
|
||||
|
||||
@note
|
||||
You should build wasm version of `opencv.js` if you want to enable this optimization. For browser, you need to enable the `WebAssembly SIMD support` feature first. For example, if you use Chrome, please enable this flag in chrome://flags. For Node.js, you need to run script with flag `--experimental-wasm-simd`.
|
||||
|
||||
@note
|
||||
The simd version of `opencv.js` built by latest LLVM upstream may not work with the stable browser or old version of Node.js. Please use the latest version of unstable browser or Node.js to get new features, like `Chrome Dev`.
|
||||
|
||||
-# [optional] To build wasm intrinsics tests, append `--build_wasm_intrin_test` option.
|
||||
|
||||
For example:
|
||||
@code{.bash}
|
||||
python ./platforms/js/build_js.py build_js --build_wasm --simd --build_wasm_intrin_test
|
||||
@endcode
|
||||
|
||||
For wasm intrinsics tests, you can use the following function to test all the cases:
|
||||
@code{.js}
|
||||
cv.test_hal_intrin_all()
|
||||
@endcode
|
||||
|
||||
And the failed cases will be logged in the JavaScript debug console.
|
||||
|
||||
If you only want to test single data type of wasm intrinsics, you can use the following functions:
|
||||
@code{.js}
|
||||
cv.test_hal_intrin_uint8()
|
||||
cv.test_hal_intrin_int8()
|
||||
cv.test_hal_intrin_uint16()
|
||||
cv.test_hal_intrin_int16()
|
||||
cv.test_hal_intrin_uint32()
|
||||
cv.test_hal_intrin_int32()
|
||||
cv.test_hal_intrin_uint64()
|
||||
cv.test_hal_intrin_int64()
|
||||
cv.test_hal_intrin_float32()
|
||||
cv.test_hal_intrin_float64()
|
||||
@endcode
|
||||
|
||||
-# [optional] To build performance tests, append `--build_perf` option.
|
||||
|
||||
For example:
|
||||
@code{.bash}
|
||||
python ./platforms/js/build_js.py build_js --build_perf
|
||||
@endcode
|
||||
|
||||
To run performance tests, launch a local web server in \<build_dir\>/bin folder. For example, node http-server which serves on `localhost:8080`.
|
||||
|
||||
There are some kernels now in the performance test like `cvtColor`, `resize` and `threshold`. For example, if you want to test `threshold`, please navigate the web browser to `http://localhost:8080/perf/perf_imgproc/perf_threshold.html`. You need to input the test parameter like `(1920x1080, CV_8UC1, THRESH_BINARY)`, and then click the `Run` button to run the case. And if you don't input the parameter, it will run all the cases of this kernel.
|
||||
|
||||
You can also run tests using Node.js.
|
||||
|
||||
For example, run `threshold` with parameter `(1920x1080, CV_8UC1, THRESH_BINARY)`:
|
||||
@code{.sh}
|
||||
cd bin/perf
|
||||
npm install
|
||||
node perf_threshold.js --test_param_filter="(1920x1080, CV_8UC1, THRESH_BINARY)"
|
||||
@endcode
|
||||
|
||||
Building OpenCV.js with Docker
|
||||
---------------------------------------
|
||||
|
||||
@@ -117,14 +201,41 @@ So, make sure [docker](https://www.docker.com/) is installed in your system and
|
||||
@code{.bash}
|
||||
git clone https://github.com/opencv/opencv.git
|
||||
cd opencv
|
||||
docker run --rm --workdir /code -v "$PWD":/code "trzeci/emscripten:latest" python ./platforms/js/build_js.py build_js
|
||||
docker run --rm --workdir /code -v "$PWD":/code "trzeci/emscripten:latest" python ./platforms/js/build_js.py build
|
||||
@endcode
|
||||
|
||||
In Windows use the following PowerShell command:
|
||||
|
||||
@code{.bash}
|
||||
docker run --rm --workdir /code -v "$(get-location):/code" "trzeci/emscripten:latest" python ./platforms/js/build_js.py build_js
|
||||
docker run --rm --workdir /code -v "$(get-location):/code" "trzeci/emscripten:latest" python ./platforms/js/build_js.py build
|
||||
@endcode
|
||||
|
||||
@note
|
||||
The example uses latest version of [trzeci/emscripten](https://hub.docker.com/r/trzeci/emscripten) docker container. At this time, the latest version works fine and is `trzeci/emscripten:sdk-tag-1.38.32-64bit`
|
||||
@warning
|
||||
The example uses latest version of emscripten. If the build fails you should try a version that is known to work fine which is `1.38.32` using the following command:
|
||||
|
||||
@code{.bash}
|
||||
docker run --rm --workdir /code -v "$PWD":/code "trzeci/emscripten:sdk-tag-1.38.32-64bit" python ./platforms/js/build_js.py build
|
||||
@endcode
|
||||
|
||||
### Building the documentation with Docker
|
||||
|
||||
To build the documentation `doxygen` needs to be installed. Create a file named `Dockerfile` with the following content:
|
||||
|
||||
```
|
||||
FROM trzeci/emscripten:sdk-tag-1.38.32-64bit
|
||||
|
||||
RUN apt-get update -y
|
||||
RUN apt-get install -y doxygen
|
||||
```
|
||||
|
||||
Then we build the docker image and name it `opencv-js-doc` with the following command (that needs to be run only once):
|
||||
|
||||
@code{.bash}
|
||||
docker build . -t opencv-js-doc
|
||||
@endcode
|
||||
|
||||
Now run the build command again, this time using the new image and passing `--build_doc`:
|
||||
|
||||
@code{.bash}
|
||||
docker run --rm --workdir /code -v "$PWD":/code "opencv-js-doc" python ./platforms/js/build_js.py build --build_doc
|
||||
@endcode
|
||||
|
||||
@@ -4,7 +4,7 @@ Getting Started with Images {#tutorial_py_image_display}
|
||||
Goals
|
||||
-----
|
||||
|
||||
- Here, you will learn how to read an image, how to display it and how to save it back
|
||||
- Here, you will learn how to read an image, how to display it, and how to save it back
|
||||
- You will learn these functions : **cv.imread()**, **cv.imshow()** , **cv.imwrite()**
|
||||
- Optionally, you will learn how to display images with Matplotlib
|
||||
|
||||
@@ -30,7 +30,7 @@ See the code below:
|
||||
import numpy as np
|
||||
import cv2 as cv
|
||||
|
||||
# Load an color image in grayscale
|
||||
# Load a color image in grayscale
|
||||
img = cv.imread('messi5.jpg',0)
|
||||
@endcode
|
||||
|
||||
@@ -43,7 +43,7 @@ Even if the image path is wrong, it won't throw any error, but `print img` will
|
||||
Use the function **cv.imshow()** to display an image in a window. The window automatically fits to
|
||||
the image size.
|
||||
|
||||
First argument is a window name which is a string. second argument is our image. You can create as
|
||||
First argument is a window name which is a string. Second argument is our image. You can create as
|
||||
many windows as you wish, but with different window names.
|
||||
@code{.py}
|
||||
cv.imshow('image',img)
|
||||
@@ -66,11 +66,11 @@ MUST use it to actually display the image.
|
||||
specific window, use the function **cv.destroyWindow()** where you pass the exact window name as
|
||||
the argument.
|
||||
|
||||
@note There is a special case where you can already create a window and load image to it later. In
|
||||
that case, you can specify whether window is resizable or not. It is done with the function
|
||||
**cv.namedWindow()**. By default, the flag is cv.WINDOW_AUTOSIZE. But if you specify flag to be
|
||||
cv.WINDOW_NORMAL, you can resize window. It will be helpful when image is too large in dimension
|
||||
and adding track bar to windows.
|
||||
@note There is a special case where you can create an empty window and load an image to it later. In
|
||||
that case, you can specify whether the window is resizable or not. It is done with the function
|
||||
**cv.namedWindow()**. By default, the flag is cv.WINDOW_AUTOSIZE. But if you specify the flag to be
|
||||
cv.WINDOW_NORMAL, you can resize window. It will be helpful when an image is too large in dimension
|
||||
and when adding track bars to windows.
|
||||
|
||||
See the code below:
|
||||
@code{.py}
|
||||
@@ -91,8 +91,8 @@ This will save the image in PNG format in the working directory.
|
||||
|
||||
### Sum it up
|
||||
|
||||
Below program loads an image in grayscale, displays it, save the image if you press 's' and exit, or
|
||||
simply exit without saving if you press ESC key.
|
||||
Below program loads an image in grayscale, displays it, saves the image if you press 's' and exit, or
|
||||
simply exits without saving if you press ESC key.
|
||||
@code{.py}
|
||||
import numpy as np
|
||||
import cv2 as cv
|
||||
@@ -117,7 +117,7 @@ Using Matplotlib
|
||||
|
||||
Matplotlib is a plotting library for Python which gives you wide variety of plotting methods. You
|
||||
will see them in coming articles. Here, you will learn how to display image with Matplotlib. You can
|
||||
zoom images, save it etc using Matplotlib.
|
||||
zoom images, save them, etc, using Matplotlib.
|
||||
@code{.py}
|
||||
import numpy as np
|
||||
import cv2 as cv
|
||||
|
||||
@@ -5,7 +5,7 @@ Goals
|
||||
-----
|
||||
|
||||
In this tutorial We will learn to setup OpenCV-Python in Ubuntu System.
|
||||
Below steps are tested for Ubuntu 16.04 (64-bit) and Ubuntu 14.04 (32-bit).
|
||||
Below steps are tested for Ubuntu 16.04 and 18.04 (both 64-bit).
|
||||
|
||||
OpenCV-Python can be installed in Ubuntu in two ways:
|
||||
- Install from pre-built binaries available in Ubuntu repositories
|
||||
@@ -62,17 +62,36 @@ We need **CMake** to configure the installation, **GCC** for compilation, **Pyth
|
||||
|
||||
```
|
||||
sudo apt-get install cmake
|
||||
sudo apt-get install python-dev python-numpy
|
||||
sudo apt-get install gcc g++
|
||||
```
|
||||
to support python2:
|
||||
|
||||
```
|
||||
sudo apt-get install python-dev python-numpy
|
||||
```
|
||||
|
||||
to support python3:
|
||||
|
||||
```
|
||||
sudo apt-get install python3-dev python3-numpy
|
||||
```
|
||||
|
||||
Next we need **GTK** support for GUI features, Camera support (v4l), Media Support
|
||||
(ffmpeg, gstreamer) etc.
|
||||
|
||||
```
|
||||
sudo apt-get install gtk2-devel
|
||||
sudo apt-get install ffmpeg-devel
|
||||
sudo apt-get install gstreamer-plugins-base-devel
|
||||
sudo apt-get install libavcodec-dev libavformat-dev libswscale-dev
|
||||
sudo apt-get install libgstreamer-plugins-base1.0-dev libgstreamer1.0-dev
|
||||
```
|
||||
|
||||
to support gtk2:
|
||||
```
|
||||
sudo apt-get install libgtk2.0-dev
|
||||
```
|
||||
|
||||
to support gtk3:
|
||||
```
|
||||
sudo apt-get install libgtk-3-dev
|
||||
```
|
||||
|
||||
### Optional Dependencies
|
||||
@@ -86,14 +105,15 @@ But it may be a little old.
|
||||
If you want to get latest libraries, you can install development files for system libraries of these formats.
|
||||
|
||||
```
|
||||
sudo apt-get install libpng-devel
|
||||
sudo apt-get install libjpeg-turbo-devel
|
||||
sudo apt-get install jasper-devel
|
||||
sudo apt-get install openexr-devel
|
||||
sudo apt-get install libtiff-devel
|
||||
sudo apt-get install libwebp-devel
|
||||
sudo apt-get install libpng-dev
|
||||
sudo apt-get install libjpeg-dev
|
||||
sudo apt-get install libopenexr-dev
|
||||
sudo apt-get install libtiff-dev
|
||||
sudo apt-get install libwebp-dev
|
||||
```
|
||||
|
||||
@note If you are using Ubuntu 16.04 you can also install ```libjasper-dev``` to add a system level support for the JPEG2000 format.
|
||||
|
||||
### Downloading OpenCV
|
||||
|
||||
To download the latest source from OpenCV's [GitHub Repository](https://github.com/opencv/opencv).
|
||||
|
||||
@@ -213,7 +213,7 @@ Supported platform: Drive PX 2
|
||||
-DBUILD_JASPER=OFF \
|
||||
-DBUILD_ZLIB=OFF \
|
||||
-DBUILD_EXAMPLES=ON \
|
||||
-DBUILD_opencv_java=OFF \
|
||||
-DBUILD_JAVA=OFF \
|
||||
-DBUILD_opencv_python2=ON \
|
||||
-DBUILD_opencv_python3=OFF \
|
||||
-DENABLE_NEON=ON \
|
||||
@@ -263,7 +263,7 @@ Configuration is slightly different for the Jetson TK1 and the Jetson TX1 system
|
||||
-DBUILD_JASPER=OFF \
|
||||
-DBUILD_ZLIB=OFF \
|
||||
-DBUILD_EXAMPLES=ON \
|
||||
-DBUILD_opencv_java=OFF \
|
||||
-DBUILD_JAVA=OFF \
|
||||
-DBUILD_opencv_python2=ON \
|
||||
-DBUILD_opencv_python3=OFF \
|
||||
-DENABLE_NEON=ON \
|
||||
@@ -300,7 +300,7 @@ __Note:__ This uses CUDA 6.5, not 8.0.
|
||||
-DBUILD_JASPER=OFF \
|
||||
-DBUILD_ZLIB=OFF \
|
||||
-DBUILD_EXAMPLES=ON \
|
||||
-DBUILD_opencv_java=OFF \
|
||||
-DBUILD_JAVA=OFF \
|
||||
-DBUILD_opencv_python2=ON \
|
||||
-DBUILD_opencv_python3=OFF \
|
||||
-DENABLE_PRECOMPILED_HEADERS=OFF \
|
||||
@@ -345,7 +345,7 @@ The configuration options given to `cmake` below are targeted towards the functi
|
||||
-DBUILD_JASPER=OFF \
|
||||
-DBUILD_ZLIB=OFF \
|
||||
-DBUILD_EXAMPLES=ON \
|
||||
-DBUILD_opencv_java=OFF \
|
||||
-DBUILD_JAVA=OFF \
|
||||
-DBUILD_opencv_python2=ON \
|
||||
-DBUILD_opencv_python3=OFF \
|
||||
-DWITH_OPENCL=OFF \
|
||||
@@ -476,7 +476,7 @@ For DRIVE PX 2:
|
||||
-DBUILD_JASPER=OFF \
|
||||
-DBUILD_ZLIB=OFF \
|
||||
-DBUILD_EXAMPLES=ON \
|
||||
-DBUILD_opencv_java=OFF \
|
||||
-DBUILD_JAVA=OFF \
|
||||
-DBUILD_opencv_nonfree=OFF \
|
||||
-DBUILD_opencv_python=ON \
|
||||
-DENABLE_NEON=ON \
|
||||
@@ -513,7 +513,7 @@ For Jetson TK1:
|
||||
-DBUILD_JASPER=OFF \
|
||||
-DBUILD_ZLIB=OFF \
|
||||
-DBUILD_EXAMPLES=ON \
|
||||
-DBUILD_opencv_java=OFF \
|
||||
-DBUILD_JAVA=OFF \
|
||||
-DBUILD_opencv_nonfree=OFF \
|
||||
-DBUILD_opencv_python=ON \
|
||||
-DENABLE_NEON=ON \
|
||||
@@ -548,7 +548,7 @@ For Jetson TX1:
|
||||
-DBUILD_JASPER=OFF \
|
||||
-DBUILD_ZLIB=OFF \
|
||||
-DBUILD_EXAMPLES=ON \
|
||||
-DBUILD_opencv_java=OFF \
|
||||
-DBUILD_JAVA=OFF \
|
||||
-DBUILD_opencv_nonfree=OFF \
|
||||
-DBUILD_opencv_python=ON \
|
||||
-DENABLE_PRECOMPILED_HEADERS=OFF \
|
||||
@@ -585,7 +585,7 @@ For both 14.04 LTS and 16.04 LTS:
|
||||
-DBUILD_JASPER=OFF \
|
||||
-DBUILD_ZLIB=OFF \
|
||||
-DBUILD_EXAMPLES=ON \
|
||||
-DBUILD_opencv_java=OFF \
|
||||
-DBUILD_JAVA=OFF \
|
||||
-DBUILD_opencv_nonfree=OFF \
|
||||
-DBUILD_opencv_python=ON \
|
||||
-DWITH_OPENCL=OFF \
|
||||
@@ -626,7 +626,7 @@ The following is a table of all the parameters passed to CMake in the recommende
|
||||
|BUILD_TBB|OFF|As above, for `tbb`| |
|
||||
|BUILD_TIFF|OFF|As above, for `libtiff`| |
|
||||
|BUILD_ZLIB|OFF|As above, for `zlib`| |
|
||||
|BUILD_opencv_java|OFF|Controls the building of the Java bindings for OpenCV|Building the Java bindings requires OpenCV libraries be built for static linking only|
|
||||
|BUILD_JAVA|OFF|Controls the building of the Java bindings for OpenCV|Building the Java bindings requires OpenCV libraries be built for static linking only|
|
||||
|BUILD_opencv_nonfree|OFF|Controls the building of non-free (non-open-source) elements|Used only for building 2.4.X|
|
||||
|BUILD_opencv_python|ON|Controls the building of the Python 2 bindings in OpenCV 2.4.X|Used only for building 2.4.X|
|
||||
|BUILD_opencv_python2|ON|Controls the building of the Python 2 bindings in OpenCV 3.1.0|Not used in 2.4.X|
|
||||
|
||||
@@ -101,6 +101,7 @@ Building OpenCV from Source Using CMake
|
||||
- Add this flag when running CMake: `-DOPENCV_GENERATE_PKGCONFIG=ON`
|
||||
- Will generate the .pc file for pkg-config and install it.
|
||||
- Useful if not using CMake in projects that use OpenCV
|
||||
- Installed as `opencv4`, usage: `pkg-config --cflags --libs opencv4`
|
||||
|
||||
-# Build. From build directory execute *make*, it is recommended to do this in several threads
|
||||
|
||||
|
||||
@@ -41,20 +41,18 @@ namespace opencv_test
|
||||
using namespace perf;
|
||||
using namespace testing;
|
||||
|
||||
static void MakeArtificialExample(RNG rng, Mat& dst_left_view, Mat& dst_view);
|
||||
static void MakeArtificialExample(Mat& dst_left_view, Mat& dst_view);
|
||||
|
||||
CV_ENUM(SGBMModes, StereoSGBM::MODE_SGBM, StereoSGBM::MODE_SGBM_3WAY, StereoSGBM::MODE_HH4);
|
||||
typedef tuple<Size, int, SGBMModes> SGBMParams;
|
||||
typedef TestBaseWithParam<SGBMParams> TestStereoCorresp;
|
||||
typedef TestBaseWithParam<SGBMParams> TestStereoCorrespSGBM;
|
||||
|
||||
#ifndef _DEBUG
|
||||
PERF_TEST_P( TestStereoCorresp, SGBM, Combine(Values(Size(1280,720),Size(640,480)), Values(256,128), SGBMModes::all()) )
|
||||
PERF_TEST_P( TestStereoCorrespSGBM, SGBM, Combine(Values(Size(1280,720),Size(640,480)), Values(256,128), SGBMModes::all()) )
|
||||
#else
|
||||
PERF_TEST_P( TestStereoCorresp, DISABLED_TooLongInDebug_SGBM, Combine(Values(Size(1280,720),Size(640,480)), Values(256,128), SGBMModes::all()) )
|
||||
PERF_TEST_P( TestStereoCorrespSGBM, DISABLED_TooLongInDebug_SGBM, Combine(Values(Size(1280,720),Size(640,480)), Values(256,128), SGBMModes::all()) )
|
||||
#endif
|
||||
{
|
||||
RNG rng(0);
|
||||
|
||||
SGBMParams params = GetParam();
|
||||
|
||||
Size sz = get<0>(params);
|
||||
@@ -65,7 +63,7 @@ PERF_TEST_P( TestStereoCorresp, DISABLED_TooLongInDebug_SGBM, Combine(Values(Siz
|
||||
Mat src_right(sz, CV_8UC3);
|
||||
Mat dst(sz, CV_16S);
|
||||
|
||||
MakeArtificialExample(rng,src_left,src_right);
|
||||
MakeArtificialExample(src_left,src_right);
|
||||
|
||||
int wsize = 3;
|
||||
int P1 = 8*src_left.channels()*wsize*wsize;
|
||||
@@ -78,8 +76,34 @@ PERF_TEST_P( TestStereoCorresp, DISABLED_TooLongInDebug_SGBM, Combine(Values(Siz
|
||||
SANITY_CHECK(dst, .01, ERROR_RELATIVE);
|
||||
}
|
||||
|
||||
void MakeArtificialExample(RNG rng, Mat& dst_left_view, Mat& dst_right_view)
|
||||
typedef tuple<Size, int> BMParams;
|
||||
typedef TestBaseWithParam<BMParams> TestStereoCorrespBM;
|
||||
|
||||
PERF_TEST_P(TestStereoCorrespBM, BM, Combine(Values(Size(1280, 720), Size(640, 480)), Values(256, 128)))
|
||||
{
|
||||
BMParams params = GetParam();
|
||||
Size sz = get<0>(params);
|
||||
int num_disparities = get<1>(params);
|
||||
|
||||
Mat src_left(sz, CV_8UC1);
|
||||
Mat src_right(sz, CV_8UC1);
|
||||
Mat dst(sz, CV_16S);
|
||||
|
||||
MakeArtificialExample(src_left, src_right);
|
||||
|
||||
int wsize = 21;
|
||||
TEST_CYCLE()
|
||||
{
|
||||
Ptr<StereoBM> bm = StereoBM::create(num_disparities, wsize);
|
||||
bm->compute(src_left, src_right, dst);
|
||||
}
|
||||
|
||||
SANITY_CHECK(dst, .01, ERROR_RELATIVE);
|
||||
}
|
||||
|
||||
void MakeArtificialExample(Mat& dst_left_view, Mat& dst_right_view)
|
||||
{
|
||||
RNG rng(0);
|
||||
int w = dst_left_view.cols;
|
||||
int h = dst_left_view.rows;
|
||||
|
||||
|
||||
@@ -282,9 +282,9 @@ void FastX::rotate(float angle,const cv::Mat &img,cv::Size size,cv::Mat &out)con
|
||||
}
|
||||
else
|
||||
{
|
||||
cv::Mat_<double> m = cv::getRotationMatrix2D(cv::Point2f(float(img.cols*0.5),float(img.rows*0.5)),float(angle/CV_PI*180),1);
|
||||
m.at<double>(0,2) += 0.5*(size.width-img.cols);
|
||||
m.at<double>(1,2) += 0.5*(size.height-img.rows);
|
||||
cv::Matx23d m = cv::getRotationMatrix2D(cv::Point2f(float(img.cols*0.5),float(img.rows*0.5)),float(angle/CV_PI*180),1);
|
||||
m(0,2) += 0.5*(size.width-img.cols);
|
||||
m(1,2) += 0.5*(size.height-img.rows);
|
||||
cv::warpAffine(img,out,m,size);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -571,7 +571,8 @@ int cv::recoverPose( InputArray E, InputArray _points1, InputArray _points2,
|
||||
if (!_mask.empty())
|
||||
{
|
||||
Mat mask = _mask.getMat();
|
||||
CV_Assert(mask.size() == mask1.size());
|
||||
CV_Assert(npoints == mask.checkVector(1));
|
||||
mask = mask.reshape(1, npoints);
|
||||
bitwise_and(mask, mask1, mask1);
|
||||
bitwise_and(mask, mask2, mask2);
|
||||
bitwise_and(mask, mask3, mask3);
|
||||
|
||||
@@ -153,19 +153,24 @@ public:
|
||||
m_2 = vx_setall_f64(k5);
|
||||
m_0 /= v_muladd(v_muladd(v_muladd(m_3, r2_0, m_2), r2_0, vx_setall_f64(k4)), r2_0, v_one);
|
||||
m_1 /= v_muladd(v_muladd(v_muladd(m_3, r2_1, m_2), r2_1, vx_setall_f64(k4)), r2_1, v_one);
|
||||
|
||||
m_3 = vx_setall_f64(2.0);
|
||||
xd_0 = v_muladd(m_3, xd_0, r2_0);
|
||||
yd_0 = v_muladd(m_3, yd_0, r2_0);
|
||||
xd_1 = v_muladd(m_3, xd_1, r2_1);
|
||||
yd_1 = v_muladd(m_3, yd_1, r2_1);
|
||||
m_2 = x_0 * y_0 * m_3;
|
||||
m_3 = x_1 * y_1 * m_3;
|
||||
|
||||
x_0 *= m_0; y_0 *= m_0; x_1 *= m_1; y_1 *= m_1;
|
||||
|
||||
m_0 = vx_setall_f64(p1);
|
||||
m_1 = vx_setall_f64(p2);
|
||||
m_2 = vx_setall_f64(2.0);
|
||||
xd_0 = v_muladd(v_muladd(m_2, xd_0, r2_0), m_1, x_0);
|
||||
yd_0 = v_muladd(v_muladd(m_2, yd_0, r2_0), m_0, y_0);
|
||||
xd_1 = v_muladd(v_muladd(m_2, xd_1, r2_1), m_1, x_1);
|
||||
yd_1 = v_muladd(v_muladd(m_2, yd_1, r2_1), m_0, y_1);
|
||||
xd_0 = v_muladd(xd_0, m_1, x_0);
|
||||
yd_0 = v_muladd(yd_0, m_0, y_0);
|
||||
xd_1 = v_muladd(xd_1, m_1, x_1);
|
||||
yd_1 = v_muladd(yd_1, m_0, y_1);
|
||||
|
||||
m_0 *= m_2; m_1 *= m_2;
|
||||
m_2 = x_0 * y_0;
|
||||
m_3 = x_1 * y_1;
|
||||
xd_0 = v_muladd(m_0, m_2, xd_0);
|
||||
yd_0 = v_muladd(m_1, m_2, yd_0);
|
||||
xd_1 = v_muladd(m_0, m_3, xd_1);
|
||||
|
||||
@@ -2196,4 +2196,102 @@ TEST(Calib3d_Triangulate, accuracy)
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
TEST(CV_RecoverPoseTest, regression_15341)
|
||||
{
|
||||
// initialize test data
|
||||
const int invalid_point_count = 2;
|
||||
const float _points1_[] = {
|
||||
1537.7f, 166.8f,
|
||||
1599.1f, 179.6f,
|
||||
1288.0f, 207.5f,
|
||||
1507.1f, 193.2f,
|
||||
1742.7f, 210.0f,
|
||||
1041.6f, 271.7f,
|
||||
1591.8f, 247.2f,
|
||||
1524.0f, 261.3f,
|
||||
1330.3f, 285.0f,
|
||||
1403.1f, 284.0f,
|
||||
1506.6f, 342.9f,
|
||||
1502.8f, 347.3f,
|
||||
1344.9f, 364.9f,
|
||||
0.0f, 0.0f // last point is initial invalid
|
||||
};
|
||||
|
||||
const float _points2_[] = {
|
||||
1533.4f, 532.9f,
|
||||
1596.6f, 552.4f,
|
||||
1277.0f, 556.4f,
|
||||
1502.1f, 557.6f,
|
||||
1744.4f, 601.3f,
|
||||
1023.0f, 612.6f,
|
||||
1589.2f, 621.6f,
|
||||
1519.4f, 629.0f,
|
||||
1320.3f, 637.3f,
|
||||
1395.2f, 642.2f,
|
||||
1501.5f, 710.3f,
|
||||
1497.6f, 714.2f,
|
||||
1335.1f, 719.61f,
|
||||
1000.0f, 1000.0f // last point is initial invalid
|
||||
};
|
||||
|
||||
vector<Point2f> _points1; Mat(14, 1, CV_32FC2, (void*)_points1_).copyTo(_points1);
|
||||
vector<Point2f> _points2; Mat(14, 1, CV_32FC2, (void*)_points2_).copyTo(_points2);
|
||||
|
||||
const int point_count = (int) _points1.size();
|
||||
CV_Assert(point_count == (int) _points2.size());
|
||||
|
||||
// camera matrix with both focal lengths = 1, and principal point = (0, 0)
|
||||
const Mat cameraMatrix = Mat::eye(3, 3, CV_64F);
|
||||
|
||||
int Inliers = 0;
|
||||
|
||||
const int ntests = 3;
|
||||
for (int testcase = 1; testcase <= ntests; ++testcase)
|
||||
{
|
||||
if (testcase == 1) // testcase with vector input data
|
||||
{
|
||||
// init temporary test data
|
||||
vector<unsigned char> mask(point_count);
|
||||
vector<Point2f> points1(_points1);
|
||||
vector<Point2f> points2(_points2);
|
||||
|
||||
// Estimation of fundamental matrix using the RANSAC algorithm
|
||||
Mat E, R, t;
|
||||
E = findEssentialMat(points1, points2, cameraMatrix, RANSAC, 0.999, 1.0, mask);
|
||||
EXPECT_EQ(0, (int)mask[13]) << "Detecting outliers in function findEssentialMat failed, testcase " << testcase;
|
||||
points2[12] = Point2f(0.0f, 0.0f); // provoke another outlier detection for recover Pose
|
||||
Inliers = recoverPose(E, points1, points2, cameraMatrix, R, t, mask);
|
||||
EXPECT_EQ(0, (int)mask[12]) << "Detecting outliers in function failed, testcase " << testcase;
|
||||
}
|
||||
else // testcase with mat input data
|
||||
{
|
||||
Mat points1(_points1, true);
|
||||
Mat points2(_points2, true);
|
||||
Mat mask;
|
||||
|
||||
if (testcase == 2)
|
||||
{
|
||||
// init temporary testdata
|
||||
mask = Mat::zeros(point_count, 1, CV_8UC1);
|
||||
}
|
||||
else // testcase == 3 - with transposed mask
|
||||
{
|
||||
mask = Mat::zeros(1, point_count, CV_8UC1);
|
||||
}
|
||||
|
||||
// Estimation of fundamental matrix using the RANSAC algorithm
|
||||
Mat E, R, t;
|
||||
E = findEssentialMat(points1, points2, cameraMatrix, RANSAC, 0.999, 1.0, mask);
|
||||
EXPECT_EQ(0, (int)mask.at<unsigned char>(13)) << "Detecting outliers in function findEssentialMat failed, testcase " << testcase;
|
||||
points2.at<Point2f>(12) = Point2f(0.0f, 0.0f); // provoke an outlier detection
|
||||
Inliers = recoverPose(E, points1, points2, cameraMatrix, R, t, mask);
|
||||
EXPECT_EQ(0, (int)mask.at<unsigned char>(12)) << "Detecting outliers in function failed, testcase " << testcase;
|
||||
}
|
||||
EXPECT_EQ(Inliers, point_count - invalid_point_count) <<
|
||||
"Number of inliers differs from expected number of inliers, testcase " << testcase;
|
||||
}
|
||||
}
|
||||
|
||||
}} // namespace
|
||||
|
||||
@@ -1469,6 +1469,44 @@ TEST(Calib3d_UndistortPoints, outputShape)
|
||||
}
|
||||
}
|
||||
|
||||
TEST(Imgproc_undistort, regression_15286)
|
||||
{
|
||||
double kmat_data[9] = { 3217, 0, 1592, 0, 3217, 1201, 0, 0, 1 };
|
||||
Mat kmat(3, 3, CV_64F, kmat_data);
|
||||
double dist_coeff_data[5] = { 0.04, -0.4, -0.01, 0.04, 0.7 };
|
||||
Mat dist_coeffs(5, 1, CV_64F, dist_coeff_data);
|
||||
|
||||
Mat img = Mat::zeros(512, 512, CV_8UC1);
|
||||
img.at<uchar>(128, 128) = 255;
|
||||
img.at<uchar>(128, 384) = 255;
|
||||
img.at<uchar>(384, 384) = 255;
|
||||
img.at<uchar>(384, 128) = 255;
|
||||
|
||||
Mat ref = Mat::zeros(512, 512, CV_8UC1);
|
||||
ref.at<uchar>(Point(24, 98)) = 78;
|
||||
ref.at<uchar>(Point(24, 99)) = 114;
|
||||
ref.at<uchar>(Point(25, 98)) = 36;
|
||||
ref.at<uchar>(Point(25, 99)) = 60;
|
||||
ref.at<uchar>(Point(27, 361)) = 6;
|
||||
ref.at<uchar>(Point(28, 361)) = 188;
|
||||
ref.at<uchar>(Point(28, 362)) = 49;
|
||||
ref.at<uchar>(Point(29, 361)) = 44;
|
||||
ref.at<uchar>(Point(29, 362)) = 16;
|
||||
ref.at<uchar>(Point(317, 366)) = 134;
|
||||
ref.at<uchar>(Point(317, 367)) = 78;
|
||||
ref.at<uchar>(Point(318, 366)) = 40;
|
||||
ref.at<uchar>(Point(318, 367)) = 29;
|
||||
ref.at<uchar>(Point(310, 104)) = 106;
|
||||
ref.at<uchar>(Point(310, 105)) = 30;
|
||||
ref.at<uchar>(Point(311, 104)) = 112;
|
||||
ref.at<uchar>(Point(311, 105)) = 38;
|
||||
|
||||
Mat img_undist;
|
||||
undistort(img, img_undist, kmat, dist_coeffs);
|
||||
|
||||
ASSERT_EQ(0.0, cvtest::norm(img_undist, ref, cv::NORM_INF));
|
||||
}
|
||||
|
||||
TEST(Calib3d_initUndistortRectifyMap, regression_14467)
|
||||
{
|
||||
Size size_w_h(512 + 3, 512);
|
||||
|
||||
@@ -6,7 +6,7 @@ ocv_add_dispatched_file(arithm SSE2 SSE4_1 AVX2 VSX3)
|
||||
ocv_add_dispatched_file(convert SSE2 AVX2 VSX3)
|
||||
ocv_add_dispatched_file(convert_scale SSE2 AVX2)
|
||||
ocv_add_dispatched_file(count_non_zero SSE2 AVX2)
|
||||
ocv_add_dispatched_file(matmul SSE2 AVX2)
|
||||
ocv_add_dispatched_file(matmul SSE2 SSE4_1 AVX2 AVX512_SKX)
|
||||
ocv_add_dispatched_file(mean SSE2 AVX2)
|
||||
ocv_add_dispatched_file(merge SSE2 AVX2)
|
||||
ocv_add_dispatched_file(split SSE2 AVX2)
|
||||
|
||||
@@ -113,12 +113,18 @@
|
||||
# define CV_AVX_512IFMA 1
|
||||
# define CV_AVX_512VBMI 1
|
||||
#endif
|
||||
#ifdef CV_CPU_COMPILE_AVX512_CEL
|
||||
# define CV_AVX512_CEL 1
|
||||
#ifdef CV_CPU_COMPILE_AVX512_CLX
|
||||
# define CV_AVX512_CLX 1
|
||||
# define CV_AVX_512VNNI 1
|
||||
#endif
|
||||
#ifdef CV_CPU_COMPILE_AVX512_ICL
|
||||
# define CV_AVX512_ICL 1
|
||||
# undef CV_AVX_512IFMA
|
||||
# define CV_AVX_512IFMA 1
|
||||
# undef CV_AVX_512VBMI
|
||||
# define CV_AVX_512VBMI 1
|
||||
# undef CV_AVX_512VNNI
|
||||
# define CV_AVX_512VNNI 1
|
||||
# define CV_AVX_512VBMI2 1
|
||||
# define CV_AVX_512BITALG 1
|
||||
# define CV_AVX_512VPOPCNTDQ 1
|
||||
@@ -152,6 +158,16 @@
|
||||
# define CV_VSX3 1
|
||||
#endif
|
||||
|
||||
#ifdef CV_CPU_COMPILE_MSA
|
||||
# include "hal/msa_macros.h"
|
||||
# define CV_MSA 1
|
||||
#endif
|
||||
|
||||
#ifdef __EMSCRIPTEN__
|
||||
# define CV_WASM_SIMD 1
|
||||
# include <wasm_simd128.h>
|
||||
#endif
|
||||
|
||||
#endif // CV_ENABLE_INTRINSICS && !CV_DISABLE_OPTIMIZATION && !__CUDACC__
|
||||
|
||||
#if defined CV_CPU_COMPILE_AVX && !defined CV_CPU_BASELINE_COMPILE_AVX
|
||||
@@ -301,8 +317,8 @@ struct VZeroUpperGuard {
|
||||
#ifndef CV_AVX512_CNL
|
||||
# define CV_AVX512_CNL 0
|
||||
#endif
|
||||
#ifndef CV_AVX512_CEL
|
||||
# define CV_AVX512_CEL 0
|
||||
#ifndef CV_AVX512_CLX
|
||||
# define CV_AVX512_CLX 0
|
||||
#endif
|
||||
#ifndef CV_AVX512_ICL
|
||||
# define CV_AVX512_ICL 0
|
||||
@@ -319,3 +335,11 @@ struct VZeroUpperGuard {
|
||||
#ifndef CV_VSX3
|
||||
# define CV_VSX3 0
|
||||
#endif
|
||||
|
||||
#ifndef CV_MSA
|
||||
# define CV_MSA 0
|
||||
#endif
|
||||
|
||||
#ifndef CV_WASM_SIMD
|
||||
# define CV_WASM_SIMD 0
|
||||
#endif
|
||||
|
||||
@@ -357,26 +357,26 @@
|
||||
#endif
|
||||
#define __CV_CPU_DISPATCH_CHAIN_AVX512_CNL(fn, args, mode, ...) CV_CPU_CALL_AVX512_CNL(fn, args); __CV_EXPAND(__CV_CPU_DISPATCH_CHAIN_ ## mode(fn, args, __VA_ARGS__))
|
||||
|
||||
#if !defined CV_DISABLE_OPTIMIZATION && defined CV_ENABLE_INTRINSICS && defined CV_CPU_COMPILE_AVX512_CEL
|
||||
# define CV_TRY_AVX512_CEL 1
|
||||
# define CV_CPU_FORCE_AVX512_CEL 1
|
||||
# define CV_CPU_HAS_SUPPORT_AVX512_CEL 1
|
||||
# define CV_CPU_CALL_AVX512_CEL(fn, args) return (cpu_baseline::fn args)
|
||||
# define CV_CPU_CALL_AVX512_CEL_(fn, args) return (opt_AVX512_CEL::fn args)
|
||||
#elif !defined CV_DISABLE_OPTIMIZATION && defined CV_ENABLE_INTRINSICS && defined CV_CPU_DISPATCH_COMPILE_AVX512_CEL
|
||||
# define CV_TRY_AVX512_CEL 1
|
||||
# define CV_CPU_FORCE_AVX512_CEL 0
|
||||
# define CV_CPU_HAS_SUPPORT_AVX512_CEL (cv::checkHardwareSupport(CV_CPU_AVX512_CEL))
|
||||
# define CV_CPU_CALL_AVX512_CEL(fn, args) if (CV_CPU_HAS_SUPPORT_AVX512_CEL) return (opt_AVX512_CEL::fn args)
|
||||
# define CV_CPU_CALL_AVX512_CEL_(fn, args) if (CV_CPU_HAS_SUPPORT_AVX512_CEL) return (opt_AVX512_CEL::fn args)
|
||||
#if !defined CV_DISABLE_OPTIMIZATION && defined CV_ENABLE_INTRINSICS && defined CV_CPU_COMPILE_AVX512_CLX
|
||||
# define CV_TRY_AVX512_CLX 1
|
||||
# define CV_CPU_FORCE_AVX512_CLX 1
|
||||
# define CV_CPU_HAS_SUPPORT_AVX512_CLX 1
|
||||
# define CV_CPU_CALL_AVX512_CLX(fn, args) return (cpu_baseline::fn args)
|
||||
# define CV_CPU_CALL_AVX512_CLX_(fn, args) return (opt_AVX512_CLX::fn args)
|
||||
#elif !defined CV_DISABLE_OPTIMIZATION && defined CV_ENABLE_INTRINSICS && defined CV_CPU_DISPATCH_COMPILE_AVX512_CLX
|
||||
# define CV_TRY_AVX512_CLX 1
|
||||
# define CV_CPU_FORCE_AVX512_CLX 0
|
||||
# define CV_CPU_HAS_SUPPORT_AVX512_CLX (cv::checkHardwareSupport(CV_CPU_AVX512_CLX))
|
||||
# define CV_CPU_CALL_AVX512_CLX(fn, args) if (CV_CPU_HAS_SUPPORT_AVX512_CLX) return (opt_AVX512_CLX::fn args)
|
||||
# define CV_CPU_CALL_AVX512_CLX_(fn, args) if (CV_CPU_HAS_SUPPORT_AVX512_CLX) return (opt_AVX512_CLX::fn args)
|
||||
#else
|
||||
# define CV_TRY_AVX512_CEL 0
|
||||
# define CV_CPU_FORCE_AVX512_CEL 0
|
||||
# define CV_CPU_HAS_SUPPORT_AVX512_CEL 0
|
||||
# define CV_CPU_CALL_AVX512_CEL(fn, args)
|
||||
# define CV_CPU_CALL_AVX512_CEL_(fn, args)
|
||||
# define CV_TRY_AVX512_CLX 0
|
||||
# define CV_CPU_FORCE_AVX512_CLX 0
|
||||
# define CV_CPU_HAS_SUPPORT_AVX512_CLX 0
|
||||
# define CV_CPU_CALL_AVX512_CLX(fn, args)
|
||||
# define CV_CPU_CALL_AVX512_CLX_(fn, args)
|
||||
#endif
|
||||
#define __CV_CPU_DISPATCH_CHAIN_AVX512_CEL(fn, args, mode, ...) CV_CPU_CALL_AVX512_CEL(fn, args); __CV_EXPAND(__CV_CPU_DISPATCH_CHAIN_ ## mode(fn, args, __VA_ARGS__))
|
||||
#define __CV_CPU_DISPATCH_CHAIN_AVX512_CLX(fn, args, mode, ...) CV_CPU_CALL_AVX512_CLX(fn, args); __CV_EXPAND(__CV_CPU_DISPATCH_CHAIN_ ## mode(fn, args, __VA_ARGS__))
|
||||
|
||||
#if !defined CV_DISABLE_OPTIMIZATION && defined CV_ENABLE_INTRINSICS && defined CV_CPU_COMPILE_AVX512_ICL
|
||||
# define CV_TRY_AVX512_ICL 1
|
||||
@@ -420,6 +420,27 @@
|
||||
#endif
|
||||
#define __CV_CPU_DISPATCH_CHAIN_NEON(fn, args, mode, ...) CV_CPU_CALL_NEON(fn, args); __CV_EXPAND(__CV_CPU_DISPATCH_CHAIN_ ## mode(fn, args, __VA_ARGS__))
|
||||
|
||||
#if !defined CV_DISABLE_OPTIMIZATION && defined CV_ENABLE_INTRINSICS && defined CV_CPU_COMPILE_MSA
|
||||
# define CV_TRY_MSA 1
|
||||
# define CV_CPU_FORCE_MSA 1
|
||||
# define CV_CPU_HAS_SUPPORT_MSA 1
|
||||
# define CV_CPU_CALL_MSA(fn, args) return (cpu_baseline::fn args)
|
||||
# define CV_CPU_CALL_MSA_(fn, args) return (opt_MSA::fn args)
|
||||
#elif !defined CV_DISABLE_OPTIMIZATION && defined CV_ENABLE_INTRINSICS && defined CV_CPU_DISPATCH_COMPILE_MSA
|
||||
# define CV_TRY_MSA 1
|
||||
# define CV_CPU_FORCE_MSA 0
|
||||
# define CV_CPU_HAS_SUPPORT_MSA (cv::checkHardwareSupport(CV_CPU_MSA))
|
||||
# define CV_CPU_CALL_MSA(fn, args) if (CV_CPU_HAS_SUPPORT_MSA) return (opt_MSA::fn args)
|
||||
# define CV_CPU_CALL_MSA_(fn, args) if (CV_CPU_HAS_SUPPORT_MSA) return (opt_MSA::fn args)
|
||||
#else
|
||||
# define CV_TRY_MSA 0
|
||||
# define CV_CPU_FORCE_MSA 0
|
||||
# define CV_CPU_HAS_SUPPORT_MSA 0
|
||||
# define CV_CPU_CALL_MSA(fn, args)
|
||||
# define CV_CPU_CALL_MSA_(fn, args)
|
||||
#endif
|
||||
#define __CV_CPU_DISPATCH_CHAIN_MSA(fn, args, mode, ...) CV_CPU_CALL_MSA(fn, args); __CV_EXPAND(__CV_CPU_DISPATCH_CHAIN_ ## mode(fn, args, __VA_ARGS__))
|
||||
|
||||
#if !defined CV_DISABLE_OPTIMIZATION && defined CV_ENABLE_INTRINSICS && defined CV_CPU_COMPILE_VSX
|
||||
# define CV_TRY_VSX 1
|
||||
# define CV_CPU_FORCE_VSX 1
|
||||
|
||||
@@ -244,6 +244,8 @@ namespace cv { namespace debug_build_guard { } using namespace debug_build_guard
|
||||
|
||||
#define CV_CPU_NEON 100
|
||||
|
||||
#define CV_CPU_MSA 150
|
||||
|
||||
#define CV_CPU_VSX 200
|
||||
#define CV_CPU_VSX3 201
|
||||
|
||||
@@ -253,7 +255,7 @@ namespace cv { namespace debug_build_guard { } using namespace debug_build_guard
|
||||
#define CV_CPU_AVX512_KNL 258
|
||||
#define CV_CPU_AVX512_KNM 259
|
||||
#define CV_CPU_AVX512_CNL 260
|
||||
#define CV_CPU_AVX512_CEL 261
|
||||
#define CV_CPU_AVX512_CLX 261
|
||||
#define CV_CPU_AVX512_ICL 262
|
||||
|
||||
// when adding to this list remember to update the following enum
|
||||
@@ -294,6 +296,8 @@ enum CpuFeatures {
|
||||
|
||||
CPU_NEON = 100,
|
||||
|
||||
CPU_MSA = 150,
|
||||
|
||||
CPU_VSX = 200,
|
||||
CPU_VSX3 = 201,
|
||||
|
||||
@@ -302,7 +306,7 @@ enum CpuFeatures {
|
||||
CPU_AVX512_KNL = 258, //!< Knights Landing with AVX-512F/CD/ER/PF
|
||||
CPU_AVX512_KNM = 259, //!< Knights Mill with AVX-512F/CD/ER/PF/4FMAPS/4VNNIW/VPOPCNTDQ
|
||||
CPU_AVX512_CNL = 260, //!< Cannon Lake with AVX-512F/CD/BW/DQ/VL/IFMA/VBMI
|
||||
CPU_AVX512_CEL = 261, //!< Cascade Lake with AVX-512F/CD/BW/DQ/VL/IFMA/VBMI/VNNI
|
||||
CPU_AVX512_CLX = 261, //!< Cascade Lake with AVX-512F/CD/BW/DQ/VL/VNNI
|
||||
CPU_AVX512_ICL = 262, //!< Ice Lake with AVX-512F/CD/BW/DQ/VL/IFMA/VBMI/VNNI/VBMI2/BITALG/VPOPCNTDQ
|
||||
|
||||
CPU_MAX_FEATURE = 512 // see CV_HARDWARE_MAX_FEATURE
|
||||
@@ -691,6 +695,7 @@ __CV_ENUM_FLAGS_BITWISE_XOR_EQ (EnumType, EnumType)
|
||||
#endif
|
||||
|
||||
#define CV_CXX_MOVE_SEMANTICS 1
|
||||
#define CV_CXX_MOVE(x) std::move(x)
|
||||
#define CV_CXX_STD_ARRAY 1
|
||||
#include <array>
|
||||
#ifndef CV_OVERRIDE
|
||||
|
||||
@@ -165,9 +165,10 @@ using namespace CV_CPU_OPTIMIZATION_HAL_NAMESPACE;
|
||||
# undef CV_NEON
|
||||
# undef CV_VSX
|
||||
# undef CV_FP16
|
||||
# undef CV_MSA
|
||||
#endif
|
||||
|
||||
#if CV_SSE2 || CV_NEON || CV_VSX
|
||||
#if CV_SSE2 || CV_NEON || CV_VSX || CV_MSA || CV_WASM_SIMD
|
||||
#define CV__SIMD_FORWARD 128
|
||||
#include "opencv2/core/hal/intrin_forward.hpp"
|
||||
#endif
|
||||
@@ -185,6 +186,13 @@ using namespace CV_CPU_OPTIMIZATION_HAL_NAMESPACE;
|
||||
|
||||
#include "opencv2/core/hal/intrin_vsx.hpp"
|
||||
|
||||
#elif CV_MSA
|
||||
|
||||
#include "opencv2/core/hal/intrin_msa.hpp"
|
||||
|
||||
#elif CV_WASM_SIMD
|
||||
#include "opencv2/core/hal/intrin_wasm.hpp"
|
||||
|
||||
#else
|
||||
|
||||
#define CV_SIMD128_CPP 1
|
||||
|
||||
@@ -1241,6 +1241,11 @@ inline int v_signmask(const v_int32x8& a)
|
||||
inline int v_signmask(const v_uint32x8& a)
|
||||
{ return v_signmask(v_reinterpret_as_f32(a)); }
|
||||
|
||||
inline int v_signmask(const v_int64x4& a)
|
||||
{ return v_signmask(v_reinterpret_as_f64(a)); }
|
||||
inline int v_signmask(const v_uint64x4& a)
|
||||
{ return v_signmask(v_reinterpret_as_f64(a)); }
|
||||
|
||||
inline int v_scan_forward(const v_int8x32& a) { return trailingZeros32(v_signmask(v_reinterpret_as_s8(a))); }
|
||||
inline int v_scan_forward(const v_uint8x32& a) { return trailingZeros32(v_signmask(v_reinterpret_as_s8(a))); }
|
||||
inline int v_scan_forward(const v_int16x16& a) { return trailingZeros32(v_signmask(v_reinterpret_as_s8(a))) / 2; }
|
||||
@@ -1253,40 +1258,23 @@ inline int v_scan_forward(const v_uint64x4& a) { return trailingZeros32(v_signma
|
||||
inline int v_scan_forward(const v_float64x4& a) { return trailingZeros32(v_signmask(v_reinterpret_as_s8(a))) / 8; }
|
||||
|
||||
/** Checks **/
|
||||
#define OPENCV_HAL_IMPL_AVX_CHECK(_Tpvec, and_op, allmask) \
|
||||
inline bool v_check_all(const _Tpvec& a) \
|
||||
{ \
|
||||
int mask = v_signmask(v_reinterpret_as_s8(a)); \
|
||||
return and_op(mask, allmask) == allmask; \
|
||||
} \
|
||||
inline bool v_check_any(const _Tpvec& a) \
|
||||
{ \
|
||||
int mask = v_signmask(v_reinterpret_as_s8(a)); \
|
||||
return and_op(mask, allmask) != 0; \
|
||||
}
|
||||
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_uint8x32, OPENCV_HAL_1ST, -1)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_int8x32, OPENCV_HAL_1ST, -1)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_uint16x16, OPENCV_HAL_AND, (int)0xaaaaaaaa)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_int16x16, OPENCV_HAL_AND, (int)0xaaaaaaaa)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_uint32x8, OPENCV_HAL_AND, (int)0x88888888)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_int32x8, OPENCV_HAL_AND, (int)0x88888888)
|
||||
|
||||
#define OPENCV_HAL_IMPL_AVX_CHECK_FLT(_Tpvec, allmask) \
|
||||
inline bool v_check_all(const _Tpvec& a) \
|
||||
{ \
|
||||
int mask = v_signmask(a); \
|
||||
return mask == allmask; \
|
||||
} \
|
||||
inline bool v_check_any(const _Tpvec& a) \
|
||||
{ \
|
||||
int mask = v_signmask(a); \
|
||||
return mask != 0; \
|
||||
}
|
||||
|
||||
OPENCV_HAL_IMPL_AVX_CHECK_FLT(v_float32x8, 255)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK_FLT(v_float64x4, 15)
|
||||
#define OPENCV_HAL_IMPL_AVX_CHECK(_Tpvec, allmask) \
|
||||
inline bool v_check_all(const _Tpvec& a) { return v_signmask(a) == allmask; } \
|
||||
inline bool v_check_any(const _Tpvec& a) { return v_signmask(a) != 0; }
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_uint8x32, -1)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_int8x32, -1)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_uint32x8, 255)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_int32x8, 255)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_uint64x4, 15)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_int64x4, 15)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_float32x8, 255)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK(v_float64x4, 15)
|
||||
|
||||
#define OPENCV_HAL_IMPL_AVX_CHECK_SHORT(_Tpvec) \
|
||||
inline bool v_check_all(const _Tpvec& a) { return (v_signmask(v_reinterpret_as_s8(a)) & 0xaaaaaaaa) == 0xaaaaaaaa; } \
|
||||
inline bool v_check_any(const _Tpvec& a) { return (v_signmask(v_reinterpret_as_s8(a)) & 0xaaaaaaaa) != 0; }
|
||||
OPENCV_HAL_IMPL_AVX_CHECK_SHORT(v_uint16x16)
|
||||
OPENCV_HAL_IMPL_AVX_CHECK_SHORT(v_int16x16)
|
||||
|
||||
////////// Other math /////////
|
||||
|
||||
@@ -1443,6 +1431,28 @@ inline v_float64x4 v_cvt_f64(const v_float32x8& a)
|
||||
inline v_float64x4 v_cvt_f64_high(const v_float32x8& a)
|
||||
{ return v_float64x4(_mm256_cvtps_pd(_v256_extract_high(a.val))); }
|
||||
|
||||
// from (Mysticial and wim) https://stackoverflow.com/q/41144668
|
||||
inline v_float64x4 v_cvt_f64(const v_int64x4& v)
|
||||
{
|
||||
// constants encoded as floating-point
|
||||
__m256i magic_i_lo = _mm256_set1_epi64x(0x4330000000000000); // 2^52
|
||||
__m256i magic_i_hi32 = _mm256_set1_epi64x(0x4530000080000000); // 2^84 + 2^63
|
||||
__m256i magic_i_all = _mm256_set1_epi64x(0x4530000080100000); // 2^84 + 2^63 + 2^52
|
||||
__m256d magic_d_all = _mm256_castsi256_pd(magic_i_all);
|
||||
|
||||
// Blend the 32 lowest significant bits of v with magic_int_lo
|
||||
__m256i v_lo = _mm256_blend_epi32(magic_i_lo, v.val, 0x55);
|
||||
// Extract the 32 most significant bits of v
|
||||
__m256i v_hi = _mm256_srli_epi64(v.val, 32);
|
||||
// Flip the msb of v_hi and blend with 0x45300000
|
||||
v_hi = _mm256_xor_si256(v_hi, magic_i_hi32);
|
||||
// Compute in double precision
|
||||
__m256d v_hi_dbl = _mm256_sub_pd(_mm256_castsi256_pd(v_hi), magic_d_all);
|
||||
// (v_hi - magic_d_all) + v_lo Do not assume associativity of floating point addition
|
||||
__m256d result = _mm256_add_pd(v_hi_dbl, _mm256_castsi256_pd(v_lo));
|
||||
return v_float64x4(result);
|
||||
}
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
inline v_int8x32 v256_lut(const schar* tab, const int* idx)
|
||||
@@ -1650,12 +1660,165 @@ inline v_float32x8 v_pack_triplets(const v_float32x8& vec)
|
||||
|
||||
////////// Matrix operations /////////
|
||||
|
||||
//////// Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x8 v_dotprod(const v_int16x16& a, const v_int16x16& b)
|
||||
{ return v_int32x8(_mm256_madd_epi16(a.val, b.val)); }
|
||||
|
||||
inline v_int32x8 v_dotprod(const v_int16x16& a, const v_int16x16& b, const v_int32x8& c)
|
||||
{ return v_dotprod(a, b) + c; }
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x4 v_dotprod(const v_int32x8& a, const v_int32x8& b)
|
||||
{
|
||||
__m256i even = _mm256_mul_epi32(a.val, b.val);
|
||||
__m256i odd = _mm256_mul_epi32(_mm256_srli_epi64(a.val, 32), _mm256_srli_epi64(b.val, 32));
|
||||
return v_int64x4(_mm256_add_epi64(even, odd));
|
||||
}
|
||||
inline v_int64x4 v_dotprod(const v_int32x8& a, const v_int32x8& b, const v_int64x4& c)
|
||||
{ return v_dotprod(a, b) + c; }
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x8 v_dotprod_expand(const v_uint8x32& a, const v_uint8x32& b)
|
||||
{
|
||||
__m256i even_m = _mm256_set1_epi32(0xFF00FF00);
|
||||
__m256i even_a = _mm256_blendv_epi8(a.val, _mm256_setzero_si256(), even_m);
|
||||
__m256i odd_a = _mm256_srli_epi16(a.val, 8);
|
||||
|
||||
__m256i even_b = _mm256_blendv_epi8(b.val, _mm256_setzero_si256(), even_m);
|
||||
__m256i odd_b = _mm256_srli_epi16(b.val, 8);
|
||||
|
||||
__m256i prod0 = _mm256_madd_epi16(even_a, even_b);
|
||||
__m256i prod1 = _mm256_madd_epi16(odd_a, odd_b);
|
||||
return v_uint32x8(_mm256_add_epi32(prod0, prod1));
|
||||
}
|
||||
inline v_uint32x8 v_dotprod_expand(const v_uint8x32& a, const v_uint8x32& b, const v_uint32x8& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
inline v_int32x8 v_dotprod_expand(const v_int8x32& a, const v_int8x32& b)
|
||||
{
|
||||
__m256i even_a = _mm256_srai_epi16(_mm256_bslli_epi128(a.val, 1), 8);
|
||||
__m256i odd_a = _mm256_srai_epi16(a.val, 8);
|
||||
|
||||
__m256i even_b = _mm256_srai_epi16(_mm256_bslli_epi128(b.val, 1), 8);
|
||||
__m256i odd_b = _mm256_srai_epi16(b.val, 8);
|
||||
|
||||
__m256i prod0 = _mm256_madd_epi16(even_a, even_b);
|
||||
__m256i prod1 = _mm256_madd_epi16(odd_a, odd_b);
|
||||
return v_int32x8(_mm256_add_epi32(prod0, prod1));
|
||||
}
|
||||
inline v_int32x8 v_dotprod_expand(const v_int8x32& a, const v_int8x32& b, const v_int32x8& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x4 v_dotprod_expand(const v_uint16x16& a, const v_uint16x16& b)
|
||||
{
|
||||
__m256i mullo = _mm256_mullo_epi16(a.val, b.val);
|
||||
__m256i mulhi = _mm256_mulhi_epu16(a.val, b.val);
|
||||
__m256i mul0 = _mm256_unpacklo_epi16(mullo, mulhi);
|
||||
__m256i mul1 = _mm256_unpackhi_epi16(mullo, mulhi);
|
||||
|
||||
__m256i p02 = _mm256_blend_epi32(mul0, _mm256_setzero_si256(), 0xAA);
|
||||
__m256i p13 = _mm256_srli_epi64(mul0, 32);
|
||||
__m256i p46 = _mm256_blend_epi32(mul1, _mm256_setzero_si256(), 0xAA);
|
||||
__m256i p57 = _mm256_srli_epi64(mul1, 32);
|
||||
|
||||
__m256i p15_ = _mm256_add_epi64(p02, p13);
|
||||
__m256i p9d_ = _mm256_add_epi64(p46, p57);
|
||||
|
||||
return v_uint64x4(_mm256_add_epi64(
|
||||
_mm256_unpacklo_epi64(p15_, p9d_),
|
||||
_mm256_unpackhi_epi64(p15_, p9d_)
|
||||
));
|
||||
}
|
||||
inline v_uint64x4 v_dotprod_expand(const v_uint16x16& a, const v_uint16x16& b, const v_uint64x4& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
inline v_int64x4 v_dotprod_expand(const v_int16x16& a, const v_int16x16& b)
|
||||
{
|
||||
__m256i prod = _mm256_madd_epi16(a.val, b.val);
|
||||
__m256i sign = _mm256_srai_epi32(prod, 31);
|
||||
|
||||
__m256i lo = _mm256_unpacklo_epi32(prod, sign);
|
||||
__m256i hi = _mm256_unpackhi_epi32(prod, sign);
|
||||
|
||||
return v_int64x4(_mm256_add_epi64(
|
||||
_mm256_unpacklo_epi64(lo, hi),
|
||||
_mm256_unpackhi_epi64(lo, hi)
|
||||
));
|
||||
}
|
||||
inline v_int64x4 v_dotprod_expand(const v_int16x16& a, const v_int16x16& b, const v_int64x4& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
inline v_float64x4 v_dotprod_expand(const v_int32x8& a, const v_int32x8& b)
|
||||
{ return v_cvt_f64(v_dotprod(a, b)); }
|
||||
inline v_float64x4 v_dotprod_expand(const v_int32x8& a, const v_int32x8& b, const v_float64x4& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
//////// Fast Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x8 v_dotprod_fast(const v_int16x16& a, const v_int16x16& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
inline v_int32x8 v_dotprod_fast(const v_int16x16& a, const v_int16x16& b, const v_int32x8& c)
|
||||
{ return v_dotprod(a, b, c); }
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x4 v_dotprod_fast(const v_int32x8& a, const v_int32x8& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
inline v_int64x4 v_dotprod_fast(const v_int32x8& a, const v_int32x8& b, const v_int64x4& c)
|
||||
{ return v_dotprod(a, b, c); }
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x8 v_dotprod_expand_fast(const v_uint8x32& a, const v_uint8x32& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_uint32x8 v_dotprod_expand_fast(const v_uint8x32& a, const v_uint8x32& b, const v_uint32x8& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
inline v_int32x8 v_dotprod_expand_fast(const v_int8x32& a, const v_int8x32& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_int32x8 v_dotprod_expand_fast(const v_int8x32& a, const v_int8x32& b, const v_int32x8& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x4 v_dotprod_expand_fast(const v_uint16x16& a, const v_uint16x16& b)
|
||||
{
|
||||
__m256i mullo = _mm256_mullo_epi16(a.val, b.val);
|
||||
__m256i mulhi = _mm256_mulhi_epu16(a.val, b.val);
|
||||
__m256i mul0 = _mm256_unpacklo_epi16(mullo, mulhi);
|
||||
__m256i mul1 = _mm256_unpackhi_epi16(mullo, mulhi);
|
||||
|
||||
__m256i p02 = _mm256_blend_epi32(mul0, _mm256_setzero_si256(), 0xAA);
|
||||
__m256i p13 = _mm256_srli_epi64(mul0, 32);
|
||||
__m256i p46 = _mm256_blend_epi32(mul1, _mm256_setzero_si256(), 0xAA);
|
||||
__m256i p57 = _mm256_srli_epi64(mul1, 32);
|
||||
|
||||
__m256i p15_ = _mm256_add_epi64(p02, p13);
|
||||
__m256i p9d_ = _mm256_add_epi64(p46, p57);
|
||||
|
||||
return v_uint64x4(_mm256_add_epi64(p15_, p9d_));
|
||||
}
|
||||
inline v_uint64x4 v_dotprod_expand_fast(const v_uint16x16& a, const v_uint16x16& b, const v_uint64x4& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
inline v_int64x4 v_dotprod_expand_fast(const v_int16x16& a, const v_int16x16& b)
|
||||
{
|
||||
__m256i prod = _mm256_madd_epi16(a.val, b.val);
|
||||
__m256i sign = _mm256_srai_epi32(prod, 31);
|
||||
__m256i lo = _mm256_unpacklo_epi32(prod, sign);
|
||||
__m256i hi = _mm256_unpackhi_epi32(prod, sign);
|
||||
return v_int64x4(_mm256_add_epi64(lo, hi));
|
||||
}
|
||||
inline v_int64x4 v_dotprod_expand_fast(const v_int16x16& a, const v_int16x16& b, const v_int64x4& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
inline v_float64x4 v_dotprod_expand_fast(const v_int32x8& a, const v_int32x8& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_float64x4 v_dotprod_expand_fast(const v_int32x8& a, const v_int32x8& b, const v_float64x4& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
#define OPENCV_HAL_AVX_SPLAT2_PS(a, im) \
|
||||
v_float32x8(_mm256_permute_ps(a.val, _MM_SHUFFLE(im, im, im, im)))
|
||||
|
||||
|
||||
@@ -1473,6 +1473,32 @@ inline v_float64x8 v_cvt_f64(const v_float32x16& a)
|
||||
inline v_float64x8 v_cvt_f64_high(const v_float32x16& a)
|
||||
{ return v_float64x8(_mm512_cvtps_pd(_v512_extract_high(a.val))); }
|
||||
|
||||
// from (Mysticial and wim) https://stackoverflow.com/q/41144668
|
||||
inline v_float64x8 v_cvt_f64(const v_int64x8& v)
|
||||
{
|
||||
#if CV_AVX_512DQ
|
||||
return v_float64x8(_mm512_cvtepi64_pd(v.val));
|
||||
#else
|
||||
// constants encoded as floating-point
|
||||
__m512i magic_i_lo = _mm512_set1_epi64x(0x4330000000000000); // 2^52
|
||||
__m512i magic_i_hi32 = _mm512_set1_epi64x(0x4530000080000000); // 2^84 + 2^63
|
||||
__m512i magic_i_all = _mm512_set1_epi64x(0x4530000080100000); // 2^84 + 2^63 + 2^52
|
||||
__m512d magic_d_all = _mm512_castsi512_pd(magic_i_all);
|
||||
|
||||
// Blend the 32 lowest significant bits of v with magic_int_lo
|
||||
__m512i v_lo = _mm512_blend_epi32(magic_i_lo, v.val, 0x55);
|
||||
// Extract the 32 most significant bits of v
|
||||
__m512i v_hi = _mm512_srli_epi64(v.val, 32);
|
||||
// Flip the msb of v_hi and blend with 0x45300000
|
||||
v_hi = _mm512_xor_si512(v_hi, magic_i_hi32);
|
||||
// Compute in double precision
|
||||
__m512d v_hi_dbl = _mm512_sub_pd(_mm512_castsi512_pd(v_hi), magic_d_all);
|
||||
// (v_hi - magic_d_all) + v_lo Do not assume associativity of floating point addition
|
||||
__m512d result = _mm512_add_pd(v_hi_dbl, _mm512_castsi512_pd(v_lo));
|
||||
return v_float64x8(result);
|
||||
#endif
|
||||
}
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
inline v_int8x64 v512_lut(const schar* tab, const int* idx)
|
||||
@@ -1672,12 +1698,152 @@ inline v_float32x16 v_pack_triplets(const v_float32x16& vec)
|
||||
|
||||
////////// Matrix operations /////////
|
||||
|
||||
//////// Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x16 v_dotprod(const v_int16x32& a, const v_int16x32& b)
|
||||
{ return v_int32x16(_mm512_madd_epi16(a.val, b.val)); }
|
||||
|
||||
inline v_int32x16 v_dotprod(const v_int16x32& a, const v_int16x32& b, const v_int32x16& c)
|
||||
{ return v_dotprod(a, b) + c; }
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x8 v_dotprod(const v_int32x16& a, const v_int32x16& b)
|
||||
{
|
||||
__m512i even = _mm512_mul_epi32(a.val, b.val);
|
||||
__m512i odd = _mm512_mul_epi32(_mm512_srli_epi64(a.val, 32), _mm512_srli_epi64(b.val, 32));
|
||||
return v_int64x8(_mm512_add_epi64(even, odd));
|
||||
}
|
||||
inline v_int64x8 v_dotprod(const v_int32x16& a, const v_int32x16& b, const v_int64x8& c)
|
||||
{ return v_dotprod(a, b) + c; }
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x16 v_dotprod_expand(const v_uint8x64& a, const v_uint8x64& b)
|
||||
{
|
||||
__m512i even_a = _mm512_mask_blend_epi8(0xAAAAAAAAAAAAAAAA, a.val, _mm512_setzero_si512());
|
||||
__m512i odd_a = _mm512_srli_epi16(a.val, 8);
|
||||
|
||||
__m512i even_b = _mm512_mask_blend_epi8(0xAAAAAAAAAAAAAAAA, b.val, _mm512_setzero_si512());
|
||||
__m512i odd_b = _mm512_srli_epi16(b.val, 8);
|
||||
|
||||
__m512i prod0 = _mm512_madd_epi16(even_a, even_b);
|
||||
__m512i prod1 = _mm512_madd_epi16(odd_a, odd_b);
|
||||
return v_uint32x16(_mm512_add_epi32(prod0, prod1));
|
||||
}
|
||||
inline v_uint32x16 v_dotprod_expand(const v_uint8x64& a, const v_uint8x64& b, const v_uint32x16& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
inline v_int32x16 v_dotprod_expand(const v_int8x64& a, const v_int8x64& b)
|
||||
{
|
||||
__m512i even_a = _mm512_srai_epi16(_mm512_bslli_epi128(a.val, 1), 8);
|
||||
__m512i odd_a = _mm512_srai_epi16(a.val, 8);
|
||||
|
||||
__m512i even_b = _mm512_srai_epi16(_mm512_bslli_epi128(b.val, 1), 8);
|
||||
__m512i odd_b = _mm512_srai_epi16(b.val, 8);
|
||||
|
||||
__m512i prod0 = _mm512_madd_epi16(even_a, even_b);
|
||||
__m512i prod1 = _mm512_madd_epi16(odd_a, odd_b);
|
||||
return v_int32x16(_mm512_add_epi32(prod0, prod1));
|
||||
}
|
||||
inline v_int32x16 v_dotprod_expand(const v_int8x64& a, const v_int8x64& b, const v_int32x16& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x8 v_dotprod_expand(const v_uint16x32& a, const v_uint16x32& b)
|
||||
{
|
||||
__m512i mullo = _mm512_mullo_epi16(a.val, b.val);
|
||||
__m512i mulhi = _mm512_mulhi_epu16(a.val, b.val);
|
||||
__m512i mul0 = _mm512_unpacklo_epi16(mullo, mulhi);
|
||||
__m512i mul1 = _mm512_unpackhi_epi16(mullo, mulhi);
|
||||
|
||||
__m512i p02 = _mm512_mask_blend_epi32(0xAAAA, mul0, _mm512_setzero_si512());
|
||||
__m512i p13 = _mm512_srli_epi64(mul0, 32);
|
||||
__m512i p46 = _mm512_mask_blend_epi32(0xAAAA, mul1, _mm512_setzero_si512());
|
||||
__m512i p57 = _mm512_srli_epi64(mul1, 32);
|
||||
|
||||
__m512i p15_ = _mm512_add_epi64(p02, p13);
|
||||
__m512i p9d_ = _mm512_add_epi64(p46, p57);
|
||||
|
||||
return v_uint64x8(_mm512_add_epi64(
|
||||
_mm512_unpacklo_epi64(p15_, p9d_),
|
||||
_mm512_unpackhi_epi64(p15_, p9d_)
|
||||
));
|
||||
}
|
||||
inline v_uint64x8 v_dotprod_expand(const v_uint16x32& a, const v_uint16x32& b, const v_uint64x8& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
inline v_int64x8 v_dotprod_expand(const v_int16x32& a, const v_int16x32& b)
|
||||
{
|
||||
__m512i prod = _mm512_madd_epi16(a.val, b.val);
|
||||
__m512i even = _mm512_srai_epi64(_mm512_bslli_epi128(prod, 4), 32);
|
||||
__m512i odd = _mm512_srai_epi64(prod, 32);
|
||||
return v_int64x8(_mm512_add_epi64(even, odd));
|
||||
}
|
||||
inline v_int64x8 v_dotprod_expand(const v_int16x32& a, const v_int16x32& b, const v_int64x8& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
inline v_float64x8 v_dotprod_expand(const v_int32x16& a, const v_int32x16& b)
|
||||
{ return v_cvt_f64(v_dotprod(a, b)); }
|
||||
inline v_float64x8 v_dotprod_expand(const v_int32x16& a, const v_int32x16& b, const v_float64x8& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
//////// Fast Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x16 v_dotprod_fast(const v_int16x32& a, const v_int16x32& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
inline v_int32x16 v_dotprod_fast(const v_int16x32& a, const v_int16x32& b, const v_int32x16& c)
|
||||
{ return v_dotprod(a, b, c); }
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x8 v_dotprod_fast(const v_int32x16& a, const v_int32x16& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
inline v_int64x8 v_dotprod_fast(const v_int32x16& a, const v_int32x16& b, const v_int64x8& c)
|
||||
{ return v_dotprod(a, b, c); }
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x16 v_dotprod_expand_fast(const v_uint8x64& a, const v_uint8x64& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_uint32x16 v_dotprod_expand_fast(const v_uint8x64& a, const v_uint8x64& b, const v_uint32x16& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
inline v_int32x16 v_dotprod_expand_fast(const v_int8x64& a, const v_int8x64& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_int32x16 v_dotprod_expand_fast(const v_int8x64& a, const v_int8x64& b, const v_int32x16& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x8 v_dotprod_expand_fast(const v_uint16x32& a, const v_uint16x32& b)
|
||||
{
|
||||
__m512i mullo = _mm512_mullo_epi16(a.val, b.val);
|
||||
__m512i mulhi = _mm512_mulhi_epu16(a.val, b.val);
|
||||
__m512i mul0 = _mm512_unpacklo_epi16(mullo, mulhi);
|
||||
__m512i mul1 = _mm512_unpackhi_epi16(mullo, mulhi);
|
||||
|
||||
__m512i p02 = _mm512_mask_blend_epi32(0xAAAA, mul0, _mm512_setzero_si512());
|
||||
__m512i p13 = _mm512_srli_epi64(mul0, 32);
|
||||
__m512i p46 = _mm512_mask_blend_epi32(0xAAAA, mul1, _mm512_setzero_si512());
|
||||
__m512i p57 = _mm512_srli_epi64(mul1, 32);
|
||||
|
||||
__m512i p15_ = _mm512_add_epi64(p02, p13);
|
||||
__m512i p9d_ = _mm512_add_epi64(p46, p57);
|
||||
return v_uint64x8(_mm512_add_epi64(p15_, p9d_));
|
||||
}
|
||||
inline v_uint64x8 v_dotprod_expand_fast(const v_uint16x32& a, const v_uint16x32& b, const v_uint64x8& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
inline v_int64x8 v_dotprod_expand_fast(const v_int16x32& a, const v_int16x32& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_int64x8 v_dotprod_expand_fast(const v_int16x32& a, const v_int16x32& b, const v_int64x8& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
// 32 >> 64f
|
||||
inline v_float64x8 v_dotprod_expand_fast(const v_int32x16& a, const v_int32x16& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_float64x8 v_dotprod_expand_fast(const v_int32x16& a, const v_int32x16& b, const v_float64x8& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
|
||||
#define OPENCV_HAL_AVX512_SPLAT2_PS(a, im) \
|
||||
v_float32x16(_mm512_permute_ps(a.val, _MM_SHUFFLE(im, im, im, im)))
|
||||
|
||||
|
||||
@@ -171,7 +171,8 @@ Different type conversions and casts:
|
||||
|
||||
### Matrix operations
|
||||
|
||||
In these operations vectors represent matrix rows/columns: @ref v_dotprod, @ref v_matmul, @ref v_transpose4x4
|
||||
In these operations vectors represent matrix rows/columns: @ref v_dotprod, @ref v_dotprod_fast,
|
||||
@ref v_dotprod_expand, @ref v_dotprod_expand_fast, @ref v_matmul, @ref v_transpose4x4
|
||||
|
||||
### Usability
|
||||
|
||||
@@ -195,7 +196,10 @@ Regular integers:
|
||||
|mul_expand | x | x | x | x | x | |
|
||||
|compare | x | x | x | x | x | x |
|
||||
|shift | | | x | x | x | x |
|
||||
|dotprod | | | | x | | |
|
||||
|dotprod | | | | x | | x |
|
||||
|dotprod_fast | | | | x | | x |
|
||||
|dotprod_expand | x | x | x | x | | x |
|
||||
|dotprod_expand_fast| x | x | x | x | | x |
|
||||
|logical | x | x | x | x | x | x |
|
||||
|min, max | x | x | x | x | x | x |
|
||||
|absdiff | x | x | x | x | x | x |
|
||||
@@ -222,6 +226,7 @@ Big integers:
|
||||
|logical | x | x |
|
||||
|extract | x | x |
|
||||
|rotate (lanes) | x | x |
|
||||
|cvt_flt64 | | x |
|
||||
|
||||
Floating point:
|
||||
|
||||
@@ -853,17 +858,18 @@ inline v_reg<_Tp, n> v_muladd(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b,
|
||||
/** @brief Dot product of elements
|
||||
|
||||
Multiply values in two registers and sum adjacent result pairs.
|
||||
|
||||
Scheme:
|
||||
@code
|
||||
{A1 A2 ...} // 16-bit
|
||||
x {B1 B2 ...} // 16-bit
|
||||
-------------
|
||||
{A1B1+A2B2 ...} // 32-bit
|
||||
|
||||
@endcode
|
||||
Implemented only for 16-bit signed source type (v_int16x8).
|
||||
*/
|
||||
template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::w_type, n/2>
|
||||
v_dotprod(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b)
|
||||
v_dotprod(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b)
|
||||
{
|
||||
typedef typename V_TypeTraits<_Tp>::w_type w_type;
|
||||
v_reg<w_type, n/2> c;
|
||||
@@ -881,12 +887,11 @@ Scheme:
|
||||
x {B1 B2 ...} // 16-bit
|
||||
-------------
|
||||
{A1B1+A2B2+C1 ...} // 32-bit
|
||||
|
||||
@endcode
|
||||
Implemented only for 16-bit signed source type (v_int16x8).
|
||||
*/
|
||||
template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::w_type, n/2>
|
||||
v_dotprod(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b, const v_reg<typename V_TypeTraits<_Tp>::w_type, n / 2>& c)
|
||||
v_dotprod(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b,
|
||||
const v_reg<typename V_TypeTraits<_Tp>::w_type, n / 2>& c)
|
||||
{
|
||||
typedef typename V_TypeTraits<_Tp>::w_type w_type;
|
||||
v_reg<w_type, n/2> s;
|
||||
@@ -895,6 +900,95 @@ template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::w_type, n
|
||||
return s;
|
||||
}
|
||||
|
||||
/** @brief Fast Dot product of elements
|
||||
|
||||
Same as cv::v_dotprod, but it may perform unorder sum between result pairs in some platforms,
|
||||
this intrinsic can be used if the sum among all lanes is only matters
|
||||
and also it should be yielding better performance on the affected platforms.
|
||||
|
||||
*/
|
||||
template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::w_type, n/2>
|
||||
v_dotprod_fast(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
|
||||
/** @brief Fast Dot product of elements
|
||||
|
||||
Same as cv::v_dotprod_fast, but add a third element to the sum of adjacent pairs.
|
||||
*/
|
||||
template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::w_type, n/2>
|
||||
v_dotprod_fast(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b,
|
||||
const v_reg<typename V_TypeTraits<_Tp>::w_type, n / 2>& c)
|
||||
{ return v_dotprod(a, b, c); }
|
||||
|
||||
/** @brief Dot product of elements and expand
|
||||
|
||||
Multiply values in two registers and expand the sum of adjacent result pairs.
|
||||
|
||||
Scheme:
|
||||
@code
|
||||
{A1 A2 A3 A4 ...} // 8-bit
|
||||
x {B1 B2 B3 B4 ...} // 8-bit
|
||||
-------------
|
||||
{A1B1+A2B2+A3B3+A4B4 ...} // 32-bit
|
||||
|
||||
@endcode
|
||||
*/
|
||||
template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::q_type, n/4>
|
||||
v_dotprod_expand(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b)
|
||||
{
|
||||
typedef typename V_TypeTraits<_Tp>::q_type q_type;
|
||||
v_reg<q_type, n/4> s;
|
||||
for( int i = 0; i < (n/4); i++ )
|
||||
s.s[i] = (q_type)a.s[i*4 ]*b.s[i*4 ] + (q_type)a.s[i*4 + 1]*b.s[i*4 + 1] +
|
||||
(q_type)a.s[i*4 + 2]*b.s[i*4 + 2] + (q_type)a.s[i*4 + 3]*b.s[i*4 + 3];
|
||||
return s;
|
||||
}
|
||||
|
||||
/** @brief Dot product of elements
|
||||
|
||||
Same as cv::v_dotprod_expand, but add a third element to the sum of adjacent pairs.
|
||||
Scheme:
|
||||
@code
|
||||
{A1 A2 A3 A4 ...} // 8-bit
|
||||
x {B1 B2 B3 B4 ...} // 8-bit
|
||||
-------------
|
||||
{A1B1+A2B2+A3B3+A4B4+C1 ...} // 32-bit
|
||||
@endcode
|
||||
*/
|
||||
template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::q_type, n/4>
|
||||
v_dotprod_expand(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b,
|
||||
const v_reg<typename V_TypeTraits<_Tp>::q_type, n / 4>& c)
|
||||
{
|
||||
typedef typename V_TypeTraits<_Tp>::q_type q_type;
|
||||
v_reg<q_type, n/4> s;
|
||||
for( int i = 0; i < (n/4); i++ )
|
||||
s.s[i] = (q_type)a.s[i*4 ]*b.s[i*4 ] + (q_type)a.s[i*4 + 1]*b.s[i*4 + 1] +
|
||||
(q_type)a.s[i*4 + 2]*b.s[i*4 + 2] + (q_type)a.s[i*4 + 3]*b.s[i*4 + 3] + c.s[i];
|
||||
return s;
|
||||
}
|
||||
|
||||
/** @brief Fast Dot product of elements and expand
|
||||
|
||||
Multiply values in two registers and expand the sum of adjacent result pairs.
|
||||
|
||||
Same as cv::v_dotprod_expand, but it may perform unorder sum between result pairs in some platforms,
|
||||
this intrinsic can be used if the sum among all lanes is only matters
|
||||
and also it should be yielding better performance on the affected platforms.
|
||||
|
||||
*/
|
||||
template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::q_type, n/4>
|
||||
v_dotprod_expand_fast(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
|
||||
/** @brief Fast Dot product of elements
|
||||
|
||||
Same as cv::v_dotprod_expand_fast, but add a third element to the sum of adjacent pairs.
|
||||
*/
|
||||
template<typename _Tp, int n> inline v_reg<typename V_TypeTraits<_Tp>::q_type, n/4>
|
||||
v_dotprod_expand_fast(const v_reg<_Tp, n>& a, const v_reg<_Tp, n>& b,
|
||||
const v_reg<typename V_TypeTraits<_Tp>::q_type, n / 4>& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
/** @brief Multiply and expand
|
||||
|
||||
Multiply values two registers and store results in two registers with wider pack type.
|
||||
@@ -1080,7 +1174,7 @@ Example:
|
||||
v_int32x4 r; // set to {-1, -1, 1, 1}
|
||||
int mask = v_signmask(r); // mask = 3 <== 00000000 00000000 00000000 00000011
|
||||
@endcode
|
||||
For all types except 64-bit. */
|
||||
*/
|
||||
template<typename _Tp, int n> inline int v_signmask(const v_reg<_Tp, n>& a)
|
||||
{
|
||||
int mask = 0;
|
||||
@@ -1109,7 +1203,7 @@ template <typename _Tp, int n> inline int v_scan_forward(const v_reg<_Tp, n>& a)
|
||||
/** @brief Check if all packed values are less than zero
|
||||
|
||||
Unsigned values will be casted to signed: `uchar 254 => char -2`.
|
||||
For all types except 64-bit. */
|
||||
*/
|
||||
template<typename _Tp, int n> inline bool v_check_all(const v_reg<_Tp, n>& a)
|
||||
{
|
||||
for( int i = 0; i < n; i++ )
|
||||
@@ -1121,7 +1215,7 @@ template<typename _Tp, int n> inline bool v_check_all(const v_reg<_Tp, n>& a)
|
||||
/** @brief Check if any of packed values is less than zero
|
||||
|
||||
Unsigned values will be casted to signed: `uchar 254 => char -2`.
|
||||
For all types except 64-bit. */
|
||||
*/
|
||||
template<typename _Tp, int n> inline bool v_check_any(const v_reg<_Tp, n>& a)
|
||||
{
|
||||
for( int i = 0; i < n; i++ )
|
||||
@@ -1810,6 +1904,17 @@ template<int n> inline v_reg<double, n> v_cvt_f64(const v_reg<float, n*2>& a)
|
||||
return c;
|
||||
}
|
||||
|
||||
/** @brief Convert to double
|
||||
|
||||
Supported input type is cv::v_int64x2. */
|
||||
template<int n> inline v_reg<double, n> v_cvt_f64(const v_reg<int64, n>& a)
|
||||
{
|
||||
v_reg<double, n> c;
|
||||
for( int i = 0; i < n; i++ )
|
||||
c.s[i] = (double)a.s[i];
|
||||
return c;
|
||||
}
|
||||
|
||||
template<typename _Tp> inline v_reg<_Tp, V_TypeTraits<_Tp>::nlanes128> v_lut(const _Tp* tab, const int* idx)
|
||||
{
|
||||
v_reg<_Tp, V_TypeTraits<_Tp>::nlanes128> c;
|
||||
|
||||
@@ -160,6 +160,16 @@ void v_mul_expand(const __CV_V_UINT32&, const __CV_V_UINT32&, __CV_V_UINT64&, __
|
||||
void v_mul_expand(const __CV_V_INT32&, const __CV_V_INT32&, __CV_V_INT64&, __CV_V_INT64&);
|
||||
#endif
|
||||
|
||||
// Conversions
|
||||
__CV_V_FLOAT32 v_cvt_f32(const __CV_V_INT32& a);
|
||||
__CV_V_FLOAT32 v_cvt_f32(const __CV_V_FLOAT64& a);
|
||||
__CV_V_FLOAT32 v_cvt_f32(const __CV_V_FLOAT64& a, const __CV_V_FLOAT64& b);
|
||||
__CV_V_FLOAT64 v_cvt_f64(const __CV_V_INT32& a);
|
||||
__CV_V_FLOAT64 v_cvt_f64_high(const __CV_V_INT32& a);
|
||||
__CV_V_FLOAT64 v_cvt_f64(const __CV_V_FLOAT32& a);
|
||||
__CV_V_FLOAT64 v_cvt_f64_high(const __CV_V_FLOAT32& a);
|
||||
__CV_V_FLOAT64 v_cvt_f64(const __CV_V_INT64& a);
|
||||
|
||||
/** Cleanup **/
|
||||
#undef CV__SIMD_FORWARD
|
||||
#undef __CV_VX
|
||||
|
||||
+1783
File diff suppressed because it is too large
Load Diff
@@ -62,23 +62,63 @@ CV_CPU_OPTIMIZATION_HAL_NAMESPACE_BEGIN
|
||||
#define CV_SIMD128_64F 0
|
||||
#endif
|
||||
|
||||
// TODO
|
||||
#define CV_NEON_DOT 0
|
||||
|
||||
//////////// Utils ////////////
|
||||
|
||||
#if CV_SIMD128_64F
|
||||
#define OPENCV_HAL_IMPL_NEON_UNZIP(_Tpv, _Tpvx2, suffix) \
|
||||
inline void _v128_unzip(const _Tpv& a, const _Tpv& b, _Tpv& c, _Tpv& d) \
|
||||
{ c = vuzp1q_##suffix(a, b); d = vuzp2q_##suffix(a, b); }
|
||||
#define OPENCV_HAL_IMPL_NEON_UNZIP_L(_Tpv, _Tpvx2, suffix) \
|
||||
inline void _v128_unzip(const _Tpv&a, const _Tpv&b, _Tpv& c, _Tpv& d) \
|
||||
{ c = vuzp1_##suffix(a, b); d = vuzp2_##suffix(a, b); }
|
||||
#else
|
||||
#define OPENCV_HAL_IMPL_NEON_UNZIP(_Tpv, _Tpvx2, suffix) \
|
||||
inline void _v128_unzip(const _Tpv& a, const _Tpv& b, _Tpv& c, _Tpv& d) \
|
||||
{ _Tpvx2 ab = vuzpq_##suffix(a, b); c = ab.val[0]; d = ab.val[1]; }
|
||||
#define OPENCV_HAL_IMPL_NEON_UNZIP_L(_Tpv, _Tpvx2, suffix) \
|
||||
inline void _v128_unzip(const _Tpv& a, const _Tpv& b, _Tpv& c, _Tpv& d) \
|
||||
{ _Tpvx2 ab = vuzp_##suffix(a, b); c = ab.val[0]; d = ab.val[1]; }
|
||||
#endif
|
||||
|
||||
#if CV_SIMD128_64F
|
||||
#define OPENCV_HAL_IMPL_NEON_REINTERPRET(_Tpv, suffix) \
|
||||
template <typename T> static inline \
|
||||
_Tpv vreinterpretq_##suffix##_f64(T a) { return (_Tpv) a; } \
|
||||
template <typename T> static inline \
|
||||
float64x2_t vreinterpretq_f64_##suffix(T a) { return (float64x2_t) a; }
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(uint8x16_t, u8)
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(int8x16_t, s8)
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(uint16x8_t, u16)
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(int16x8_t, s16)
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(uint32x4_t, u32)
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(int32x4_t, s32)
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(uint64x2_t, u64)
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(int64x2_t, s64)
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(float32x4_t, f32)
|
||||
template <typename T> static inline \
|
||||
_Tpv vreinterpretq_##suffix##_f64(T a) { return (_Tpv) a; } \
|
||||
template <typename T> static inline \
|
||||
float64x2_t vreinterpretq_f64_##suffix(T a) { return (float64x2_t) a; }
|
||||
#else
|
||||
#define OPENCV_HAL_IMPL_NEON_REINTERPRET(_Tpv, suffix)
|
||||
#endif
|
||||
|
||||
#define OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX(_Tpv, _Tpvl, suffix) \
|
||||
OPENCV_HAL_IMPL_NEON_UNZIP(_Tpv##_t, _Tpv##x2_t, suffix) \
|
||||
OPENCV_HAL_IMPL_NEON_UNZIP_L(_Tpvl##_t, _Tpvl##x2_t, suffix) \
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(_Tpv##_t, suffix)
|
||||
|
||||
#define OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX_I64(_Tpv, _Tpvl, suffix) \
|
||||
OPENCV_HAL_IMPL_NEON_REINTERPRET(_Tpv##_t, suffix)
|
||||
|
||||
#define OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX_F64(_Tpv, _Tpvl, suffix) \
|
||||
OPENCV_HAL_IMPL_NEON_UNZIP(_Tpv##_t, _Tpv##x2_t, suffix)
|
||||
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX(uint8x16, uint8x8, u8)
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX(int8x16, int8x8, s8)
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX(uint16x8, uint16x4, u16)
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX(int16x8, int16x4, s16)
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX(uint32x4, uint32x2, u32)
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX(int32x4, int32x2, s32)
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX(float32x4, float32x2, f32)
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX_I64(uint64x2, uint64x1, u64)
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX_I64(int64x2, int64x1, s64)
|
||||
#if CV_SIMD128_64F
|
||||
OPENCV_HAL_IMPL_NEON_UTILS_SUFFIX_F64(float64x2, float64x1,f64)
|
||||
#endif
|
||||
|
||||
//////////// Types ////////////
|
||||
|
||||
struct v_uint8x16
|
||||
{
|
||||
typedef uchar lane_type;
|
||||
@@ -528,20 +568,272 @@ inline v_uint16x8 v_mul_hi(const v_uint16x8& a, const v_uint16x8& b)
|
||||
));
|
||||
}
|
||||
|
||||
//////// Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x4 v_dotprod(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
int32x4_t c = vmull_s16(vget_low_s16(a.val), vget_low_s16(b.val));
|
||||
int32x4_t d = vmull_s16(vget_high_s16(a.val), vget_high_s16(b.val));
|
||||
int32x4x2_t cd = vuzpq_s32(c, d);
|
||||
return v_int32x4(vaddq_s32(cd.val[0], cd.val[1]));
|
||||
int16x8_t uzp1, uzp2;
|
||||
_v128_unzip(a.val, b.val, uzp1, uzp2);
|
||||
int16x4_t a0 = vget_low_s16(uzp1);
|
||||
int16x4_t b0 = vget_high_s16(uzp1);
|
||||
int16x4_t a1 = vget_low_s16(uzp2);
|
||||
int16x4_t b1 = vget_high_s16(uzp2);
|
||||
int32x4_t p = vmull_s16(a0, b0);
|
||||
return v_int32x4(vmlal_s16(p, a1, b1));
|
||||
}
|
||||
|
||||
inline v_int32x4 v_dotprod(const v_int16x8& a, const v_int16x8& b, const v_int32x4& c)
|
||||
{
|
||||
v_int32x4 s = v_dotprod(a, b);
|
||||
return v_int32x4(vaddq_s32(s.val , c.val));
|
||||
int16x8_t uzp1, uzp2;
|
||||
_v128_unzip(a.val, b.val, uzp1, uzp2);
|
||||
int16x4_t a0 = vget_low_s16(uzp1);
|
||||
int16x4_t b0 = vget_high_s16(uzp1);
|
||||
int16x4_t a1 = vget_low_s16(uzp2);
|
||||
int16x4_t b1 = vget_high_s16(uzp2);
|
||||
int32x4_t p = vmlal_s16(c.val, a0, b0);
|
||||
return v_int32x4(vmlal_s16(p, a1, b1));
|
||||
}
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x2 v_dotprod(const v_int32x4& a, const v_int32x4& b)
|
||||
{
|
||||
int32x4_t uzp1, uzp2;
|
||||
_v128_unzip(a.val, b.val, uzp1, uzp2);
|
||||
int32x2_t a0 = vget_low_s32(uzp1);
|
||||
int32x2_t b0 = vget_high_s32(uzp1);
|
||||
int32x2_t a1 = vget_low_s32(uzp2);
|
||||
int32x2_t b1 = vget_high_s32(uzp2);
|
||||
int64x2_t p = vmull_s32(a0, b0);
|
||||
return v_int64x2(vmlal_s32(p, a1, b1));
|
||||
}
|
||||
inline v_int64x2 v_dotprod(const v_int32x4& a, const v_int32x4& b, const v_int64x2& c)
|
||||
{
|
||||
int32x4_t uzp1, uzp2;
|
||||
_v128_unzip(a.val, b.val, uzp1, uzp2);
|
||||
int32x2_t a0 = vget_low_s32(uzp1);
|
||||
int32x2_t b0 = vget_high_s32(uzp1);
|
||||
int32x2_t a1 = vget_low_s32(uzp2);
|
||||
int32x2_t b1 = vget_high_s32(uzp2);
|
||||
int64x2_t p = vmlal_s32(c.val, a0, b0);
|
||||
return v_int64x2(vmlal_s32(p, a1, b1));
|
||||
}
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x4 v_dotprod_expand(const v_uint8x16& a, const v_uint8x16& b)
|
||||
{
|
||||
#if CV_NEON_DOT
|
||||
return v_uint32x4(vdotq_u32(vdupq_n_u32(0), a.val, b.val));
|
||||
#else
|
||||
const uint8x16_t zero = vreinterpretq_u8_u32(vdupq_n_u32(0));
|
||||
const uint8x16_t mask = vreinterpretq_u8_u32(vdupq_n_u32(0x00FF00FF));
|
||||
const uint16x8_t zero32 = vreinterpretq_u16_u32(vdupq_n_u32(0));
|
||||
const uint16x8_t mask32 = vreinterpretq_u16_u32(vdupq_n_u32(0x0000FFFF));
|
||||
|
||||
uint16x8_t even = vmulq_u16(vreinterpretq_u16_u8(vbslq_u8(mask, a.val, zero)),
|
||||
vreinterpretq_u16_u8(vbslq_u8(mask, b.val, zero)));
|
||||
uint16x8_t odd = vmulq_u16(vshrq_n_u16(vreinterpretq_u16_u8(a.val), 8),
|
||||
vshrq_n_u16(vreinterpretq_u16_u8(b.val), 8));
|
||||
|
||||
uint32x4_t s0 = vaddq_u32(vreinterpretq_u32_u16(vbslq_u16(mask32, even, zero32)),
|
||||
vreinterpretq_u32_u16(vbslq_u16(mask32, odd, zero32)));
|
||||
uint32x4_t s1 = vaddq_u32(vshrq_n_u32(vreinterpretq_u32_u16(even), 16),
|
||||
vshrq_n_u32(vreinterpretq_u32_u16(odd), 16));
|
||||
return v_uint32x4(vaddq_u32(s0, s1));
|
||||
#endif
|
||||
}
|
||||
inline v_uint32x4 v_dotprod_expand(const v_uint8x16& a, const v_uint8x16& b,
|
||||
const v_uint32x4& c)
|
||||
{
|
||||
#if CV_NEON_DOT
|
||||
return v_uint32x4(vdotq_u32(c.val, a.val, b.val));
|
||||
#else
|
||||
return v_dotprod_expand(a, b) + c;
|
||||
#endif
|
||||
}
|
||||
|
||||
inline v_int32x4 v_dotprod_expand(const v_int8x16& a, const v_int8x16& b)
|
||||
{
|
||||
#if CV_NEON_DOT
|
||||
return v_int32x4(vdotq_s32(vdupq_n_s32(0), a.val, b.val));
|
||||
#else
|
||||
int16x8_t p0 = vmull_s8(vget_low_s8(a.val), vget_low_s8(b.val));
|
||||
int16x8_t p1 = vmull_s8(vget_high_s8(a.val), vget_high_s8(b.val));
|
||||
int16x8_t uzp1, uzp2;
|
||||
_v128_unzip(p0, p1, uzp1, uzp2);
|
||||
int16x8_t sum = vaddq_s16(uzp1, uzp2);
|
||||
int16x4_t uzpl1, uzpl2;
|
||||
_v128_unzip(vget_low_s16(sum), vget_high_s16(sum), uzpl1, uzpl2);
|
||||
return v_int32x4(vaddl_s16(uzpl1, uzpl2));
|
||||
#endif
|
||||
}
|
||||
inline v_int32x4 v_dotprod_expand(const v_int8x16& a, const v_int8x16& b,
|
||||
const v_int32x4& c)
|
||||
{
|
||||
#if CV_NEON_DOT
|
||||
return v_int32x4(vdotq_s32(c.val, a.val, b.val));
|
||||
#else
|
||||
return v_dotprod_expand(a, b) + c;
|
||||
#endif
|
||||
}
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x2 v_dotprod_expand(const v_uint16x8& a, const v_uint16x8& b)
|
||||
{
|
||||
const uint16x8_t zero = vreinterpretq_u16_u32(vdupq_n_u32(0));
|
||||
const uint16x8_t mask = vreinterpretq_u16_u32(vdupq_n_u32(0x0000FFFF));
|
||||
|
||||
uint32x4_t even = vmulq_u32(vreinterpretq_u32_u16(vbslq_u16(mask, a.val, zero)),
|
||||
vreinterpretq_u32_u16(vbslq_u16(mask, b.val, zero)));
|
||||
uint32x4_t odd = vmulq_u32(vshrq_n_u32(vreinterpretq_u32_u16(a.val), 16),
|
||||
vshrq_n_u32(vreinterpretq_u32_u16(b.val), 16));
|
||||
uint32x4_t uzp1, uzp2;
|
||||
_v128_unzip(even, odd, uzp1, uzp2);
|
||||
uint64x2_t s0 = vaddl_u32(vget_low_u32(uzp1), vget_high_u32(uzp1));
|
||||
uint64x2_t s1 = vaddl_u32(vget_low_u32(uzp2), vget_high_u32(uzp2));
|
||||
return v_uint64x2(vaddq_u64(s0, s1));
|
||||
}
|
||||
inline v_uint64x2 v_dotprod_expand(const v_uint16x8& a, const v_uint16x8& b, const v_uint64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
inline v_int64x2 v_dotprod_expand(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
int32x4_t p0 = vmull_s16(vget_low_s16(a.val), vget_low_s16(b.val));
|
||||
int32x4_t p1 = vmull_s16(vget_high_s16(a.val), vget_high_s16(b.val));
|
||||
|
||||
int32x4_t uzp1, uzp2;
|
||||
_v128_unzip(p0, p1, uzp1, uzp2);
|
||||
int32x4_t sum = vaddq_s32(uzp1, uzp2);
|
||||
|
||||
int32x2_t uzpl1, uzpl2;
|
||||
_v128_unzip(vget_low_s32(sum), vget_high_s32(sum), uzpl1, uzpl2);
|
||||
return v_int64x2(vaddl_s32(uzpl1, uzpl2));
|
||||
}
|
||||
inline v_int64x2 v_dotprod_expand(const v_int16x8& a, const v_int16x8& b,
|
||||
const v_int64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
#if CV_SIMD128_64F
|
||||
inline v_float64x2 v_dotprod_expand(const v_int32x4& a, const v_int32x4& b)
|
||||
{ return v_cvt_f64(v_dotprod(a, b)); }
|
||||
inline v_float64x2 v_dotprod_expand(const v_int32x4& a, const v_int32x4& b,
|
||||
const v_float64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
#endif
|
||||
|
||||
//////// Fast Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x4 v_dotprod_fast(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
int16x4_t a0 = vget_low_s16(a.val);
|
||||
int16x4_t a1 = vget_high_s16(a.val);
|
||||
int16x4_t b0 = vget_low_s16(b.val);
|
||||
int16x4_t b1 = vget_high_s16(b.val);
|
||||
int32x4_t p = vmull_s16(a0, b0);
|
||||
return v_int32x4(vmlal_s16(p, a1, b1));
|
||||
}
|
||||
inline v_int32x4 v_dotprod_fast(const v_int16x8& a, const v_int16x8& b, const v_int32x4& c)
|
||||
{
|
||||
int16x4_t a0 = vget_low_s16(a.val);
|
||||
int16x4_t a1 = vget_high_s16(a.val);
|
||||
int16x4_t b0 = vget_low_s16(b.val);
|
||||
int16x4_t b1 = vget_high_s16(b.val);
|
||||
int32x4_t p = vmlal_s16(c.val, a0, b0);
|
||||
return v_int32x4(vmlal_s16(p, a1, b1));
|
||||
}
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x2 v_dotprod_fast(const v_int32x4& a, const v_int32x4& b)
|
||||
{
|
||||
int32x2_t a0 = vget_low_s32(a.val);
|
||||
int32x2_t a1 = vget_high_s32(a.val);
|
||||
int32x2_t b0 = vget_low_s32(b.val);
|
||||
int32x2_t b1 = vget_high_s32(b.val);
|
||||
int64x2_t p = vmull_s32(a0, b0);
|
||||
return v_int64x2(vmlal_s32(p, a1, b1));
|
||||
}
|
||||
inline v_int64x2 v_dotprod_fast(const v_int32x4& a, const v_int32x4& b, const v_int64x2& c)
|
||||
{
|
||||
int32x2_t a0 = vget_low_s32(a.val);
|
||||
int32x2_t a1 = vget_high_s32(a.val);
|
||||
int32x2_t b0 = vget_low_s32(b.val);
|
||||
int32x2_t b1 = vget_high_s32(b.val);
|
||||
int64x2_t p = vmlal_s32(c.val, a0, b0);
|
||||
return v_int64x2(vmlal_s32(p, a1, b1));
|
||||
}
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x4 v_dotprod_expand_fast(const v_uint8x16& a, const v_uint8x16& b)
|
||||
{
|
||||
#if CV_NEON_DOT
|
||||
return v_uint32x4(vdotq_u32(vdupq_n_u32(0), a.val, b.val));
|
||||
#else
|
||||
uint16x8_t p0 = vmull_u8(vget_low_u8(a.val), vget_low_u8(b.val));
|
||||
uint16x8_t p1 = vmull_u8(vget_high_u8(a.val), vget_high_u8(b.val));
|
||||
uint32x4_t s0 = vaddl_u16(vget_low_u16(p0), vget_low_u16(p1));
|
||||
uint32x4_t s1 = vaddl_u16(vget_high_u16(p0), vget_high_u16(p1));
|
||||
return v_uint32x4(vaddq_u32(s0, s1));
|
||||
#endif
|
||||
}
|
||||
inline v_uint32x4 v_dotprod_expand_fast(const v_uint8x16& a, const v_uint8x16& b, const v_uint32x4& c)
|
||||
{
|
||||
#if CV_NEON_DOT
|
||||
return v_uint32x4(vdotq_u32(c.val, a.val, b.val));
|
||||
#else
|
||||
return v_dotprod_expand_fast(a, b) + c;
|
||||
#endif
|
||||
}
|
||||
|
||||
inline v_int32x4 v_dotprod_expand_fast(const v_int8x16& a, const v_int8x16& b)
|
||||
{
|
||||
#if CV_NEON_DOT
|
||||
return v_int32x4(vdotq_s32(vdupq_n_s32(0), a.val, b.val));
|
||||
#else
|
||||
int16x8_t prod = vmull_s8(vget_low_s8(a.val), vget_low_s8(b.val));
|
||||
prod = vmlal_s8(prod, vget_high_s8(a.val), vget_high_s8(b.val));
|
||||
return v_int32x4(vaddl_s16(vget_low_s16(prod), vget_high_s16(prod)));
|
||||
#endif
|
||||
}
|
||||
inline v_int32x4 v_dotprod_expand_fast(const v_int8x16& a, const v_int8x16& b, const v_int32x4& c)
|
||||
{
|
||||
#if CV_NEON_DOT
|
||||
return v_int32x4(vdotq_s32(c.val, a.val, b.val));
|
||||
#else
|
||||
return v_dotprod_expand_fast(a, b) + c;
|
||||
#endif
|
||||
}
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x2 v_dotprod_expand_fast(const v_uint16x8& a, const v_uint16x8& b)
|
||||
{
|
||||
uint32x4_t p0 = vmull_u16(vget_low_u16(a.val), vget_low_u16(b.val));
|
||||
uint32x4_t p1 = vmull_u16(vget_high_u16(a.val), vget_high_u16(b.val));
|
||||
uint64x2_t s0 = vaddl_u32(vget_low_u32(p0), vget_high_u32(p0));
|
||||
uint64x2_t s1 = vaddl_u32(vget_low_u32(p1), vget_high_u32(p1));
|
||||
return v_uint64x2(vaddq_u64(s0, s1));
|
||||
}
|
||||
inline v_uint64x2 v_dotprod_expand_fast(const v_uint16x8& a, const v_uint16x8& b, const v_uint64x2& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
inline v_int64x2 v_dotprod_expand_fast(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
int32x4_t prod = vmull_s16(vget_low_s16(a.val), vget_low_s16(b.val));
|
||||
prod = vmlal_s16(prod, vget_high_s16(a.val), vget_high_s16(b.val));
|
||||
return v_int64x2(vaddl_s32(vget_low_s32(prod), vget_high_s32(prod)));
|
||||
}
|
||||
inline v_int64x2 v_dotprod_expand_fast(const v_int16x8& a, const v_int16x8& b, const v_int64x2& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
#if CV_SIMD128_64F
|
||||
inline v_float64x2 v_dotprod_expand_fast(const v_int32x4& a, const v_int32x4& b)
|
||||
{ return v_cvt_f64(v_dotprod_fast(a, b)); }
|
||||
inline v_float64x2 v_dotprod_expand_fast(const v_int32x4& a, const v_int32x4& b, const v_float64x2& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
#endif
|
||||
|
||||
|
||||
#define OPENCV_HAL_IMPL_NEON_LOGIC_OP(_Tpvec, suffix) \
|
||||
OPENCV_HAL_IMPL_NEON_BIN_OP(&, _Tpvec, vandq_##suffix) \
|
||||
OPENCV_HAL_IMPL_NEON_BIN_OP(|, _Tpvec, vorrq_##suffix) \
|
||||
@@ -1139,9 +1431,17 @@ inline bool v_check_any(const v_##_Tpvec& a) \
|
||||
OPENCV_HAL_IMPL_NEON_CHECK_ALLANY(uint8x16, u8, 7)
|
||||
OPENCV_HAL_IMPL_NEON_CHECK_ALLANY(uint16x8, u16, 15)
|
||||
OPENCV_HAL_IMPL_NEON_CHECK_ALLANY(uint32x4, u32, 31)
|
||||
#if CV_SIMD128_64F
|
||||
OPENCV_HAL_IMPL_NEON_CHECK_ALLANY(uint64x2, u64, 63)
|
||||
#endif
|
||||
|
||||
inline bool v_check_all(const v_uint64x2& a)
|
||||
{
|
||||
uint64x2_t v0 = vshrq_n_u64(a.val, 63);
|
||||
return (vgetq_lane_u64(v0, 0) & vgetq_lane_u64(v0, 1)) == 1;
|
||||
}
|
||||
inline bool v_check_any(const v_uint64x2& a)
|
||||
{
|
||||
uint64x2_t v0 = vshrq_n_u64(a.val, 63);
|
||||
return (vgetq_lane_u64(v0, 0) | vgetq_lane_u64(v0, 1)) != 0;
|
||||
}
|
||||
|
||||
inline bool v_check_all(const v_int8x16& a)
|
||||
{ return v_check_all(v_reinterpret_as_u8(a)); }
|
||||
@@ -1161,13 +1461,13 @@ inline bool v_check_any(const v_int32x4& a)
|
||||
inline bool v_check_any(const v_float32x4& a)
|
||||
{ return v_check_any(v_reinterpret_as_u32(a)); }
|
||||
|
||||
#if CV_SIMD128_64F
|
||||
inline bool v_check_all(const v_int64x2& a)
|
||||
{ return v_check_all(v_reinterpret_as_u64(a)); }
|
||||
inline bool v_check_all(const v_float64x2& a)
|
||||
{ return v_check_all(v_reinterpret_as_u64(a)); }
|
||||
inline bool v_check_any(const v_int64x2& a)
|
||||
{ return v_check_any(v_reinterpret_as_u64(a)); }
|
||||
#if CV_SIMD128_64F
|
||||
inline bool v_check_all(const v_float64x2& a)
|
||||
{ return v_check_all(v_reinterpret_as_u64(a)); }
|
||||
inline bool v_check_any(const v_float64x2& a)
|
||||
{ return v_check_any(v_reinterpret_as_u64(a)); }
|
||||
#endif
|
||||
@@ -1585,6 +1885,10 @@ inline v_float64x2 v_cvt_f64_high(const v_float32x4& a)
|
||||
{
|
||||
return v_float64x2(vcvt_f64_f32(vget_high_f32(a.val)));
|
||||
}
|
||||
|
||||
inline v_float64x2 v_cvt_f64(const v_int64x2& a)
|
||||
{ return v_float64x2(vcvtq_f64_s64(a.val)); }
|
||||
|
||||
#endif
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
@@ -225,9 +225,13 @@ struct v_uint64x2
|
||||
}
|
||||
uint64 get0() const
|
||||
{
|
||||
#if !defined(__x86_64__) && !defined(_M_X64)
|
||||
int a = _mm_cvtsi128_si32(val);
|
||||
int b = _mm_cvtsi128_si32(_mm_srli_epi64(val, 32));
|
||||
return (unsigned)a | ((uint64)(unsigned)b << 32);
|
||||
#else
|
||||
return (uint64)_mm_cvtsi128_si64(val);
|
||||
#endif
|
||||
}
|
||||
|
||||
__m128i val;
|
||||
@@ -247,9 +251,13 @@ struct v_int64x2
|
||||
}
|
||||
int64 get0() const
|
||||
{
|
||||
#if !defined(__x86_64__) && !defined(_M_X64)
|
||||
int a = _mm_cvtsi128_si32(val);
|
||||
int b = _mm_cvtsi128_si32(_mm_srli_epi64(val, 32));
|
||||
return (int64)((unsigned)a | ((uint64)(unsigned)b << 32));
|
||||
#else
|
||||
return _mm_cvtsi128_si64(val);
|
||||
#endif
|
||||
}
|
||||
|
||||
__m128i val;
|
||||
@@ -791,15 +799,195 @@ inline void v_mul_expand(const v_uint32x4& a, const v_uint32x4& b,
|
||||
inline v_int16x8 v_mul_hi(const v_int16x8& a, const v_int16x8& b) { return v_int16x8(_mm_mulhi_epi16(a.val, b.val)); }
|
||||
inline v_uint16x8 v_mul_hi(const v_uint16x8& a, const v_uint16x8& b) { return v_uint16x8(_mm_mulhi_epu16(a.val, b.val)); }
|
||||
|
||||
inline v_int32x4 v_dotprod(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
return v_int32x4(_mm_madd_epi16(a.val, b.val));
|
||||
}
|
||||
//////// Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x4 v_dotprod(const v_int16x8& a, const v_int16x8& b)
|
||||
{ return v_int32x4(_mm_madd_epi16(a.val, b.val)); }
|
||||
inline v_int32x4 v_dotprod(const v_int16x8& a, const v_int16x8& b, const v_int32x4& c)
|
||||
{ return v_dotprod(a, b) + c; }
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x2 v_dotprod(const v_int32x4& a, const v_int32x4& b)
|
||||
{
|
||||
return v_int32x4(_mm_add_epi32(_mm_madd_epi16(a.val, b.val), c.val));
|
||||
#if CV_SSE4_1
|
||||
__m128i even = _mm_mul_epi32(a.val, b.val);
|
||||
__m128i odd = _mm_mul_epi32(_mm_srli_epi64(a.val, 32), _mm_srli_epi64(b.val, 32));
|
||||
return v_int64x2(_mm_add_epi64(even, odd));
|
||||
#else
|
||||
__m128i even_u = _mm_mul_epu32(a.val, b.val);
|
||||
__m128i odd_u = _mm_mul_epu32(_mm_srli_epi64(a.val, 32), _mm_srli_epi64(b.val, 32));
|
||||
// convert unsigned to signed high multiplication (from: Agner Fog(veclib) and H S Warren: Hacker's delight, 2003, p. 132)
|
||||
__m128i a_sign = _mm_srai_epi32(a.val, 31);
|
||||
__m128i b_sign = _mm_srai_epi32(b.val, 31);
|
||||
// |x * sign of x
|
||||
__m128i axb = _mm_and_si128(a.val, b_sign);
|
||||
__m128i bxa = _mm_and_si128(b.val, a_sign);
|
||||
// sum of sign corrections
|
||||
__m128i ssum = _mm_add_epi32(bxa, axb);
|
||||
__m128i even_ssum = _mm_slli_epi64(ssum, 32);
|
||||
__m128i odd_ssum = _mm_and_si128(ssum, _mm_set_epi32(-1, 0, -1, 0));
|
||||
// convert to signed and prod
|
||||
return v_int64x2(_mm_add_epi64(_mm_sub_epi64(even_u, even_ssum), _mm_sub_epi64(odd_u, odd_ssum)));
|
||||
#endif
|
||||
}
|
||||
inline v_int64x2 v_dotprod(const v_int32x4& a, const v_int32x4& b, const v_int64x2& c)
|
||||
{ return v_dotprod(a, b) + c; }
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x4 v_dotprod_expand(const v_uint8x16& a, const v_uint8x16& b)
|
||||
{
|
||||
__m128i a0 = _mm_srli_epi16(_mm_slli_si128(a.val, 1), 8); // even
|
||||
__m128i a1 = _mm_srli_epi16(a.val, 8); // odd
|
||||
__m128i b0 = _mm_srli_epi16(_mm_slli_si128(b.val, 1), 8);
|
||||
__m128i b1 = _mm_srli_epi16(b.val, 8);
|
||||
__m128i p0 = _mm_madd_epi16(a0, b0);
|
||||
__m128i p1 = _mm_madd_epi16(a1, b1);
|
||||
return v_uint32x4(_mm_add_epi32(p0, p1));
|
||||
}
|
||||
inline v_uint32x4 v_dotprod_expand(const v_uint8x16& a, const v_uint8x16& b, const v_uint32x4& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
inline v_int32x4 v_dotprod_expand(const v_int8x16& a, const v_int8x16& b)
|
||||
{
|
||||
__m128i a0 = _mm_srai_epi16(_mm_slli_si128(a.val, 1), 8); // even
|
||||
__m128i a1 = _mm_srai_epi16(a.val, 8); // odd
|
||||
__m128i b0 = _mm_srai_epi16(_mm_slli_si128(b.val, 1), 8);
|
||||
__m128i b1 = _mm_srai_epi16(b.val, 8);
|
||||
__m128i p0 = _mm_madd_epi16(a0, b0);
|
||||
__m128i p1 = _mm_madd_epi16(a1, b1);
|
||||
return v_int32x4(_mm_add_epi32(p0, p1));
|
||||
}
|
||||
inline v_int32x4 v_dotprod_expand(const v_int8x16& a, const v_int8x16& b, const v_int32x4& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x2 v_dotprod_expand(const v_uint16x8& a, const v_uint16x8& b)
|
||||
{
|
||||
v_uint32x4 c, d;
|
||||
v_mul_expand(a, b, c, d);
|
||||
|
||||
v_uint64x2 c0, c1, d0, d1;
|
||||
v_expand(c, c0, c1);
|
||||
v_expand(d, d0, d1);
|
||||
|
||||
c0 += c1; d0 += d1;
|
||||
return v_uint64x2(_mm_add_epi64(
|
||||
_mm_unpacklo_epi64(c0.val, d0.val),
|
||||
_mm_unpackhi_epi64(c0.val, d0.val)
|
||||
));
|
||||
}
|
||||
inline v_uint64x2 v_dotprod_expand(const v_uint16x8& a, const v_uint16x8& b, const v_uint64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
inline v_int64x2 v_dotprod_expand(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
v_int32x4 prod = v_dotprod(a, b);
|
||||
v_int64x2 c, d;
|
||||
v_expand(prod, c, d);
|
||||
return v_int64x2(_mm_add_epi64(
|
||||
_mm_unpacklo_epi64(c.val, d.val),
|
||||
_mm_unpackhi_epi64(c.val, d.val)
|
||||
));
|
||||
}
|
||||
inline v_int64x2 v_dotprod_expand(const v_int16x8& a, const v_int16x8& b, const v_int64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
inline v_float64x2 v_dotprod_expand(const v_int32x4& a, const v_int32x4& b)
|
||||
{
|
||||
#if CV_SSE4_1
|
||||
return v_cvt_f64(v_dotprod(a, b));
|
||||
#else
|
||||
v_float64x2 c = v_cvt_f64(a) * v_cvt_f64(b);
|
||||
v_float64x2 d = v_cvt_f64_high(a) * v_cvt_f64_high(b);
|
||||
|
||||
return v_float64x2(_mm_add_pd(
|
||||
_mm_unpacklo_pd(c.val, d.val),
|
||||
_mm_unpackhi_pd(c.val, d.val)
|
||||
));
|
||||
#endif
|
||||
}
|
||||
inline v_float64x2 v_dotprod_expand(const v_int32x4& a, const v_int32x4& b, const v_float64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
//////// Fast Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x4 v_dotprod_fast(const v_int16x8& a, const v_int16x8& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
inline v_int32x4 v_dotprod_fast(const v_int16x8& a, const v_int16x8& b, const v_int32x4& c)
|
||||
{ return v_dotprod(a, b) + c; }
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x2 v_dotprod_fast(const v_int32x4& a, const v_int32x4& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
inline v_int64x2 v_dotprod_fast(const v_int32x4& a, const v_int32x4& b, const v_int64x2& c)
|
||||
{ return v_dotprod_fast(a, b) + c; }
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x4 v_dotprod_expand_fast(const v_uint8x16& a, const v_uint8x16& b)
|
||||
{
|
||||
__m128i a0 = v_expand_low(a).val;
|
||||
__m128i a1 = v_expand_high(a).val;
|
||||
__m128i b0 = v_expand_low(b).val;
|
||||
__m128i b1 = v_expand_high(b).val;
|
||||
__m128i p0 = _mm_madd_epi16(a0, b0);
|
||||
__m128i p1 = _mm_madd_epi16(a1, b1);
|
||||
return v_uint32x4(_mm_add_epi32(p0, p1));
|
||||
}
|
||||
inline v_uint32x4 v_dotprod_expand_fast(const v_uint8x16& a, const v_uint8x16& b, const v_uint32x4& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
inline v_int32x4 v_dotprod_expand_fast(const v_int8x16& a, const v_int8x16& b)
|
||||
{
|
||||
#if CV_SSE4_1
|
||||
__m128i a0 = _mm_cvtepi8_epi16(a.val);
|
||||
__m128i a1 = v_expand_high(a).val;
|
||||
__m128i b0 = _mm_cvtepi8_epi16(b.val);
|
||||
__m128i b1 = v_expand_high(b).val;
|
||||
__m128i p0 = _mm_madd_epi16(a0, b0);
|
||||
__m128i p1 = _mm_madd_epi16(a1, b1);
|
||||
return v_int32x4(_mm_add_epi32(p0, p1));
|
||||
#else
|
||||
return v_dotprod_expand(a, b);
|
||||
#endif
|
||||
}
|
||||
inline v_int32x4 v_dotprod_expand_fast(const v_int8x16& a, const v_int8x16& b, const v_int32x4& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x2 v_dotprod_expand_fast(const v_uint16x8& a, const v_uint16x8& b)
|
||||
{
|
||||
v_uint32x4 c, d;
|
||||
v_mul_expand(a, b, c, d);
|
||||
|
||||
v_uint64x2 c0, c1, d0, d1;
|
||||
v_expand(c, c0, c1);
|
||||
v_expand(d, d0, d1);
|
||||
|
||||
c0 += c1; d0 += d1;
|
||||
return c0 + d0;
|
||||
}
|
||||
inline v_uint64x2 v_dotprod_expand_fast(const v_uint16x8& a, const v_uint16x8& b, const v_uint64x2& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
inline v_int64x2 v_dotprod_expand_fast(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
v_int32x4 prod = v_dotprod(a, b);
|
||||
v_int64x2 c, d;
|
||||
v_expand(prod, c, d);
|
||||
return c + d;
|
||||
}
|
||||
inline v_int64x2 v_dotprod_expand_fast(const v_int16x8& a, const v_int16x8& b, const v_int64x2& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
v_float64x2 v_fma(const v_float64x2& a, const v_float64x2& b, const v_float64x2& c);
|
||||
inline v_float64x2 v_dotprod_expand_fast(const v_int32x4& a, const v_int32x4& b)
|
||||
{ return v_fma(v_cvt_f64(a), v_cvt_f64(b), v_cvt_f64_high(a) * v_cvt_f64_high(b)); }
|
||||
inline v_float64x2 v_dotprod_expand_fast(const v_int32x4& a, const v_int32x4& b, const v_float64x2& c)
|
||||
{ return v_fma(v_cvt_f64(a), v_cvt_f64(b), v_fma(v_cvt_f64_high(a), v_cvt_f64_high(b), c)); }
|
||||
|
||||
#define OPENCV_HAL_IMPL_SSE_LOGIC_OP(_Tpvec, suffix, not_const) \
|
||||
OPENCV_HAL_IMPL_SSE_BIN_OP(&, _Tpvec, _mm_and_##suffix) \
|
||||
@@ -1591,31 +1779,25 @@ inline v_uint32x4 v_popcount(const v_int32x4& a)
|
||||
inline v_uint64x2 v_popcount(const v_int64x2& a)
|
||||
{ return v_popcount(v_reinterpret_as_u64(a)); }
|
||||
|
||||
#define OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(_Tpvec, suffix, pack_op, and_op, signmask, allmask) \
|
||||
inline int v_signmask(const _Tpvec& a) \
|
||||
{ \
|
||||
return and_op(_mm_movemask_##suffix(pack_op(a.val)), signmask); \
|
||||
} \
|
||||
inline bool v_check_all(const _Tpvec& a) \
|
||||
{ return and_op(_mm_movemask_##suffix(a.val), allmask) == allmask; } \
|
||||
inline bool v_check_any(const _Tpvec& a) \
|
||||
{ return and_op(_mm_movemask_##suffix(a.val), allmask) != 0; }
|
||||
#define OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(_Tpvec, suffix, cast_op, allmask) \
|
||||
inline int v_signmask(const _Tpvec& a) { return _mm_movemask_##suffix(cast_op(a.val)); } \
|
||||
inline bool v_check_all(const _Tpvec& a) { return _mm_movemask_##suffix(cast_op(a.val)) == allmask; } \
|
||||
inline bool v_check_any(const _Tpvec& a) { return _mm_movemask_##suffix(cast_op(a.val)) != 0; }
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_uint8x16, epi8, OPENCV_HAL_NOP, 65535)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_int8x16, epi8, OPENCV_HAL_NOP, 65535)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_uint32x4, ps, _mm_castsi128_ps, 15)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_int32x4, ps, _mm_castsi128_ps, 15)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_uint64x2, pd, _mm_castsi128_pd, 3)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_int64x2, pd, _mm_castsi128_pd, 3)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_float32x4, ps, OPENCV_HAL_NOP, 15)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_float64x2, pd, OPENCV_HAL_NOP, 3)
|
||||
|
||||
#define OPENCV_HAL_PACKS(a) _mm_packs_epi16(a, a)
|
||||
inline __m128i v_packq_epi32(__m128i a)
|
||||
{
|
||||
__m128i b = _mm_packs_epi32(a, a);
|
||||
return _mm_packs_epi16(b, b);
|
||||
}
|
||||
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_uint8x16, epi8, OPENCV_HAL_NOP, OPENCV_HAL_1ST, 65535, 65535)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_int8x16, epi8, OPENCV_HAL_NOP, OPENCV_HAL_1ST, 65535, 65535)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_uint16x8, epi8, OPENCV_HAL_PACKS, OPENCV_HAL_AND, 255, (int)0xaaaa)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_int16x8, epi8, OPENCV_HAL_PACKS, OPENCV_HAL_AND, 255, (int)0xaaaa)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_uint32x4, epi8, v_packq_epi32, OPENCV_HAL_AND, 15, (int)0x8888)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_int32x4, epi8, v_packq_epi32, OPENCV_HAL_AND, 15, (int)0x8888)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_float32x4, ps, OPENCV_HAL_NOP, OPENCV_HAL_1ST, 15, 15)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS(v_float64x2, pd, OPENCV_HAL_NOP, OPENCV_HAL_1ST, 3, 3)
|
||||
#define OPENCV_HAL_IMPL_SSE_CHECK_SIGNS_SHORT(_Tpvec) \
|
||||
inline int v_signmask(const _Tpvec& a) { return _mm_movemask_epi8(_mm_packs_epi16(a.val, a.val)) & 255; } \
|
||||
inline bool v_check_all(const _Tpvec& a) { return (_mm_movemask_epi8(a.val) & 0xaaaa) == 0xaaaa; } \
|
||||
inline bool v_check_any(const _Tpvec& a) { return (_mm_movemask_epi8(a.val) & 0xaaaa) != 0; }
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS_SHORT(v_uint16x8)
|
||||
OPENCV_HAL_IMPL_SSE_CHECK_SIGNS_SHORT(v_int16x8)
|
||||
|
||||
inline int v_scan_forward(const v_int8x16& a) { return trailingZeros32(v_signmask(v_reinterpret_as_s8(a))); }
|
||||
inline int v_scan_forward(const v_uint8x16& a) { return trailingZeros32(v_signmask(v_reinterpret_as_s8(a))); }
|
||||
@@ -2745,6 +2927,32 @@ inline v_float64x2 v_cvt_f64_high(const v_float32x4& a)
|
||||
return v_float64x2(_mm_cvtps_pd(_mm_movehl_ps(a.val, a.val)));
|
||||
}
|
||||
|
||||
// from (Mysticial and wim) https://stackoverflow.com/q/41144668
|
||||
inline v_float64x2 v_cvt_f64(const v_int64x2& v)
|
||||
{
|
||||
// constants encoded as floating-point
|
||||
__m128i magic_i_hi32 = _mm_set1_epi64x(0x4530000080000000); // 2^84 + 2^63
|
||||
__m128i magic_i_all = _mm_set1_epi64x(0x4530000080100000); // 2^84 + 2^63 + 2^52
|
||||
__m128d magic_d_all = _mm_castsi128_pd(magic_i_all);
|
||||
// Blend the 32 lowest significant bits of v with magic_int_lo
|
||||
#if CV_SSE4_1
|
||||
__m128i magic_i_lo = _mm_set1_epi64x(0x4330000000000000); // 2^52
|
||||
__m128i v_lo = _mm_blend_epi16(v.val, magic_i_lo, 0xcc);
|
||||
#else
|
||||
__m128i magic_i_lo = _mm_set1_epi32(0x43300000); // 2^52
|
||||
__m128i v_lo = _mm_unpacklo_epi32(_mm_shuffle_epi32(v.val, _MM_SHUFFLE(0, 0, 2, 0)), magic_i_lo);
|
||||
#endif
|
||||
// Extract the 32 most significant bits of v
|
||||
__m128i v_hi = _mm_srli_epi64(v.val, 32);
|
||||
// Flip the msb of v_hi and blend with 0x45300000
|
||||
v_hi = _mm_xor_si128(v_hi, magic_i_hi32);
|
||||
// Compute in double precision
|
||||
__m128d v_hi_dbl = _mm_sub_pd(_mm_castsi128_pd(v_hi), magic_d_all);
|
||||
// (v_hi - magic_d_all) + v_lo Do not assume associativity of floating point addition
|
||||
__m128d result = _mm_add_pd(v_hi_dbl, _mm_castsi128_pd(v_lo));
|
||||
return v_float64x2(result);
|
||||
}
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
inline v_int8x16 v_lut(const schar* tab, const int* idx)
|
||||
|
||||
@@ -499,12 +499,6 @@ inline void v_mul_expand(const Tvec& a, const Tvec& b, Twvec& c, Twvec& d)
|
||||
v_zip(p0, p1, c, d);
|
||||
}
|
||||
|
||||
inline void v_mul_expand(const v_uint32x4& a, const v_uint32x4& b, v_uint64x2& c, v_uint64x2& d)
|
||||
{
|
||||
c.val = vec_mul(vec_unpackhu(a.val), vec_unpackhu(b.val));
|
||||
d.val = vec_mul(vec_unpacklu(a.val), vec_unpacklu(b.val));
|
||||
}
|
||||
|
||||
inline v_int16x8 v_mul_hi(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
vec_int4 p0 = vec_mule(a.val, b.val);
|
||||
@@ -899,6 +893,8 @@ inline bool v_check_all(const v_uint16x8& a)
|
||||
{ return v_check_all(v_reinterpret_as_s16(a)); }
|
||||
inline bool v_check_all(const v_uint32x4& a)
|
||||
{ return v_check_all(v_reinterpret_as_s32(a)); }
|
||||
inline bool v_check_all(const v_uint64x2& a)
|
||||
{ return v_check_all(v_reinterpret_as_s64(a)); }
|
||||
inline bool v_check_all(const v_float32x4& a)
|
||||
{ return v_check_all(v_reinterpret_as_s32(a)); }
|
||||
inline bool v_check_all(const v_float64x2& a)
|
||||
@@ -913,6 +909,8 @@ inline bool v_check_any(const v_uint16x8& a)
|
||||
{ return v_check_any(v_reinterpret_as_s16(a)); }
|
||||
inline bool v_check_any(const v_uint32x4& a)
|
||||
{ return v_check_any(v_reinterpret_as_s32(a)); }
|
||||
inline bool v_check_any(const v_uint64x2& a)
|
||||
{ return v_check_any(v_reinterpret_as_s64(a)); }
|
||||
inline bool v_check_any(const v_float32x4& a)
|
||||
{ return v_check_any(v_reinterpret_as_s32(a)); }
|
||||
inline bool v_check_any(const v_float64x2& a)
|
||||
@@ -1039,14 +1037,8 @@ inline v_float64x2 v_cvt_f64(const v_float32x4& a)
|
||||
inline v_float64x2 v_cvt_f64_high(const v_float32x4& a)
|
||||
{ return v_float64x2(vec_cvfo(vec_mergel(a.val, a.val))); }
|
||||
|
||||
// The altivec intrinsic is missing for this 2.06 insn
|
||||
inline v_float64x2 v_cvt_f64(const v_int64x2& a)
|
||||
{
|
||||
vec_double2 out;
|
||||
|
||||
__asm__ ("xvcvsxddp %x0,%x1" : "=wa"(out) : "wa"(a.val));
|
||||
return v_float64x2(out);
|
||||
}
|
||||
{ return v_float64x2(vec_ctd(a.val)); }
|
||||
|
||||
////////////// Lookup table access ////////////////////
|
||||
|
||||
@@ -1318,12 +1310,134 @@ inline void v_cleanup() {}
|
||||
|
||||
////////// Matrix operations /////////
|
||||
|
||||
//////// Dot Product ////////
|
||||
// 16 >> 32
|
||||
inline v_int32x4 v_dotprod(const v_int16x8& a, const v_int16x8& b)
|
||||
{ return v_int32x4(vec_msum(a.val, b.val, vec_int4_z)); }
|
||||
|
||||
inline v_int32x4 v_dotprod(const v_int16x8& a, const v_int16x8& b, const v_int32x4& c)
|
||||
{ return v_int32x4(vec_msum(a.val, b.val, c.val)); }
|
||||
|
||||
// 32 >> 64
|
||||
inline v_int64x2 v_dotprod(const v_int32x4& a, const v_int32x4& b)
|
||||
{
|
||||
vec_dword2 even = vec_mule(a.val, b.val);
|
||||
vec_dword2 odd = vec_mulo(a.val, b.val);
|
||||
return v_int64x2(vec_add(even, odd));
|
||||
}
|
||||
inline v_int64x2 v_dotprod(const v_int32x4& a, const v_int32x4& b, const v_int64x2& c)
|
||||
{ return v_dotprod(a, b) + c; }
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x4 v_dotprod_expand(const v_uint8x16& a, const v_uint8x16& b, const v_uint32x4& c)
|
||||
{ return v_uint32x4(vec_msum(a.val, b.val, c.val)); }
|
||||
inline v_uint32x4 v_dotprod_expand(const v_uint8x16& a, const v_uint8x16& b)
|
||||
{ return v_uint32x4(vec_msum(a.val, b.val, vec_uint4_z)); }
|
||||
|
||||
inline v_int32x4 v_dotprod_expand(const v_int8x16& a, const v_int8x16& b)
|
||||
{
|
||||
const vec_ushort8 eight = vec_ushort8_sp(8);
|
||||
vec_short8 a0 = vec_sra((vec_short8)vec_sld(a.val, a.val, 1), eight); // even
|
||||
vec_short8 a1 = vec_sra((vec_short8)a.val, eight); // odd
|
||||
vec_short8 b0 = vec_sra((vec_short8)vec_sld(b.val, b.val, 1), eight);
|
||||
vec_short8 b1 = vec_sra((vec_short8)b.val, eight);
|
||||
return v_int32x4(vec_msum(a0, b0, vec_msum(a1, b1, vec_int4_z)));
|
||||
}
|
||||
|
||||
inline v_int32x4 v_dotprod_expand(const v_int8x16& a, const v_int8x16& b, const v_int32x4& c)
|
||||
{
|
||||
const vec_ushort8 eight = vec_ushort8_sp(8);
|
||||
vec_short8 a0 = vec_sra((vec_short8)vec_sld(a.val, a.val, 1), eight); // even
|
||||
vec_short8 a1 = vec_sra((vec_short8)a.val, eight); // odd
|
||||
vec_short8 b0 = vec_sra((vec_short8)vec_sld(b.val, b.val, 1), eight);
|
||||
vec_short8 b1 = vec_sra((vec_short8)b.val, eight);
|
||||
return v_int32x4(vec_msum(a0, b0, vec_msum(a1, b1, c.val)));
|
||||
}
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x2 v_dotprod_expand(const v_uint16x8& a, const v_uint16x8& b)
|
||||
{
|
||||
const vec_uint4 zero = vec_uint4_z;
|
||||
vec_uint4 even = vec_mule(a.val, b.val);
|
||||
vec_uint4 odd = vec_mulo(a.val, b.val);
|
||||
vec_udword2 e0 = (vec_udword2)vec_mergee(even, zero);
|
||||
vec_udword2 e1 = (vec_udword2)vec_mergeo(even, zero);
|
||||
vec_udword2 o0 = (vec_udword2)vec_mergee(odd, zero);
|
||||
vec_udword2 o1 = (vec_udword2)vec_mergeo(odd, zero);
|
||||
vec_udword2 s0 = vec_add(e0, o0);
|
||||
vec_udword2 s1 = vec_add(e1, o1);
|
||||
return v_uint64x2(vec_add(s0, s1));
|
||||
}
|
||||
inline v_uint64x2 v_dotprod_expand(const v_uint16x8& a, const v_uint16x8& b, const v_uint64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
inline v_int64x2 v_dotprod_expand(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
v_int32x4 prod = v_dotprod(a, b);
|
||||
v_int64x2 c, d;
|
||||
v_expand(prod, c, d);
|
||||
return v_int64x2(vec_add(vec_mergeh(c.val, d.val), vec_mergel(c.val, d.val)));
|
||||
}
|
||||
inline v_int64x2 v_dotprod_expand(const v_int16x8& a, const v_int16x8& b, const v_int64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
inline v_float64x2 v_dotprod_expand(const v_int32x4& a, const v_int32x4& b)
|
||||
{ return v_cvt_f64(v_dotprod(a, b)); }
|
||||
inline v_float64x2 v_dotprod_expand(const v_int32x4& a, const v_int32x4& b, const v_float64x2& c)
|
||||
{ return v_dotprod_expand(a, b) + c; }
|
||||
|
||||
//////// Fast Dot Product ////////
|
||||
|
||||
// 16 >> 32
|
||||
inline v_int32x4 v_dotprod_fast(const v_int16x8& a, const v_int16x8& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
inline v_int32x4 v_dotprod_fast(const v_int16x8& a, const v_int16x8& b, const v_int32x4& c)
|
||||
{ return v_int32x4(vec_msum(a.val, b.val, vec_int4_z)) + c; }
|
||||
// 32 >> 64
|
||||
inline v_int64x2 v_dotprod_fast(const v_int32x4& a, const v_int32x4& b)
|
||||
{ return v_dotprod(a, b); }
|
||||
inline v_int64x2 v_dotprod_fast(const v_int32x4& a, const v_int32x4& b, const v_int64x2& c)
|
||||
{ return v_dotprod(a, b, c); }
|
||||
|
||||
// 8 >> 32
|
||||
inline v_uint32x4 v_dotprod_expand_fast(const v_uint8x16& a, const v_uint8x16& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_uint32x4 v_dotprod_expand_fast(const v_uint8x16& a, const v_uint8x16& b, const v_uint32x4& c)
|
||||
{ return v_uint32x4(vec_msum(a.val, b.val, vec_uint4_z)) + c; }
|
||||
|
||||
inline v_int32x4 v_dotprod_expand_fast(const v_int8x16& a, const v_int8x16& b)
|
||||
{
|
||||
vec_short8 a0 = vec_unpackh(a.val);
|
||||
vec_short8 a1 = vec_unpackl(a.val);
|
||||
vec_short8 b0 = vec_unpackh(b.val);
|
||||
vec_short8 b1 = vec_unpackl(b.val);
|
||||
return v_int32x4(vec_msum(a0, b0, vec_msum(a1, b1, vec_int4_z)));
|
||||
}
|
||||
inline v_int32x4 v_dotprod_expand_fast(const v_int8x16& a, const v_int8x16& b, const v_int32x4& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
// 16 >> 64
|
||||
inline v_uint64x2 v_dotprod_expand_fast(const v_uint16x8& a, const v_uint16x8& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_uint64x2 v_dotprod_expand_fast(const v_uint16x8& a, const v_uint16x8& b, const v_uint64x2& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
inline v_int64x2 v_dotprod_expand_fast(const v_int16x8& a, const v_int16x8& b)
|
||||
{
|
||||
v_int32x4 prod = v_dotprod(a, b);
|
||||
v_int64x2 c, d;
|
||||
v_expand(prod, c, d);
|
||||
return c + d;
|
||||
}
|
||||
inline v_int64x2 v_dotprod_expand_fast(const v_int16x8& a, const v_int16x8& b, const v_int64x2& c)
|
||||
{ return v_dotprod_expand_fast(a, b) + c; }
|
||||
|
||||
// 32 >> 64f
|
||||
inline v_float64x2 v_dotprod_expand_fast(const v_int32x4& a, const v_int32x4& b)
|
||||
{ return v_dotprod_expand(a, b); }
|
||||
inline v_float64x2 v_dotprod_expand_fast(const v_int32x4& a, const v_int32x4& b, const v_float64x2& c)
|
||||
{ return v_dotprod_expand(a, b, c); }
|
||||
|
||||
inline v_float32x4 v_matmul(const v_float32x4& v, const v_float32x4& m0,
|
||||
const v_float32x4& m1, const v_float32x4& m2,
|
||||
const v_float32x4& m3)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+1558
File diff suppressed because it is too large
Load Diff
@@ -72,7 +72,9 @@
|
||||
# include "opencv2/core/cuda_stream_accessor.hpp"
|
||||
# include "opencv2/core/cuda/common.hpp"
|
||||
|
||||
# ifndef NPP_VERSION
|
||||
# define NPP_VERSION (NPP_VERSION_MAJOR * 1000 + NPP_VERSION_MINOR * 100 + NPP_VERSION_BUILD)
|
||||
# endif
|
||||
|
||||
# define CUDART_MINIMUM_REQUIRED_VERSION 6050
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
|
||||
/**
|
||||
Helper header to support SIMD intrinsics (universal intrinsics) in user code.
|
||||
Intrinsics documentation: https://docs.opencv.org/3.4/df/d91/group__core__hal__intrin.html
|
||||
Intrinsics documentation: https://docs.opencv.org/master/df/d91/group__core__hal__intrin.html
|
||||
|
||||
|
||||
Checks of target CPU instruction set based on compiler definitions don't work well enough.
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
#define CV_VERSION_MAJOR 4
|
||||
#define CV_VERSION_MINOR 1
|
||||
#define CV_VERSION_REVISION 2
|
||||
#define CV_VERSION_STATUS "-openvino"
|
||||
#define CV_VERSION_STATUS ""
|
||||
|
||||
#define CVAUX_STR_EXP(__A) #__A
|
||||
#define CVAUX_STR(__A) CVAUX_STR_EXP(__A)
|
||||
|
||||
@@ -124,6 +124,33 @@ VSX_FINLINE(rt) fnm(const rg& a, const rg& b) \
|
||||
|
||||
#define VSX_IMPL_2VRG(rt, rg, opc, fnm) VSX_IMPL_2VRG_F(rt, rg, #opc" %0,%1,%2", fnm)
|
||||
|
||||
#if __GNUG__ < 8
|
||||
|
||||
// Support for int4 -> dword2 expanding multiply was added in GCC 8.
|
||||
#ifdef vec_mule
|
||||
#undef vec_mule
|
||||
#endif
|
||||
#ifdef vec_mulo
|
||||
#undef vec_mulo
|
||||
#endif
|
||||
|
||||
VSX_REDIRECT_2RG(vec_ushort8, vec_uchar16, vec_mule, __builtin_vec_mule)
|
||||
VSX_REDIRECT_2RG(vec_short8, vec_char16, vec_mule, __builtin_vec_mule)
|
||||
VSX_REDIRECT_2RG(vec_int4, vec_short8, vec_mule, __builtin_vec_mule)
|
||||
VSX_REDIRECT_2RG(vec_uint4, vec_ushort8, vec_mule, __builtin_vec_mule)
|
||||
VSX_REDIRECT_2RG(vec_ushort8, vec_uchar16, vec_mulo, __builtin_vec_mulo)
|
||||
VSX_REDIRECT_2RG(vec_short8, vec_char16, vec_mulo, __builtin_vec_mulo)
|
||||
VSX_REDIRECT_2RG(vec_int4, vec_short8, vec_mulo, __builtin_vec_mulo)
|
||||
VSX_REDIRECT_2RG(vec_uint4, vec_ushort8, vec_mulo, __builtin_vec_mulo)
|
||||
|
||||
// dword2 support arrived in ISA 2.07 and GCC 8+
|
||||
VSX_IMPL_2VRG(vec_dword2, vec_int4, vmulosw, vec_mule)
|
||||
VSX_IMPL_2VRG(vec_udword2, vec_uint4, vmulouw, vec_mule)
|
||||
VSX_IMPL_2VRG(vec_dword2, vec_int4, vmulesw, vec_mulo)
|
||||
VSX_IMPL_2VRG(vec_udword2, vec_uint4, vmuleuw, vec_mulo)
|
||||
|
||||
#endif
|
||||
|
||||
#if __GNUG__ < 7
|
||||
// up to GCC 6 vec_mul only supports precisions and llong
|
||||
# ifdef vec_mul
|
||||
|
||||
@@ -9,7 +9,7 @@ typedef TestBaseWithParam<MatType_Length_t> MatType_Length;
|
||||
|
||||
PERF_TEST_P( MatType_Length, dot,
|
||||
testing::Combine(
|
||||
testing::Values( CV_8UC1, CV_32SC1, CV_32FC1 ),
|
||||
testing::Values( CV_8UC1, CV_8SC1, CV_16SC1, CV_16UC1, CV_32SC1, CV_32FC1 ),
|
||||
testing::Values( 32, 64, 128, 256, 512, 1024 )
|
||||
))
|
||||
{
|
||||
|
||||
+56
-15
@@ -46,6 +46,7 @@
|
||||
#undef CV_LOG_STRIP_LEVEL
|
||||
#define CV_LOG_STRIP_LEVEL CV_LOG_LEVEL_VERBOSE + 1
|
||||
#include <opencv2/core/utils/logger.hpp>
|
||||
#include <opencv2/core/utils/configuration.private.hpp>
|
||||
|
||||
#define CV__ALLOCATOR_STATS_LOG(...) CV_LOG_VERBOSE(NULL, 0, "alloc.cpp: " << __VA_ARGS__)
|
||||
#include "opencv2/core/utils/allocator_stats.impl.hpp"
|
||||
@@ -81,6 +82,38 @@ cv::utils::AllocatorStatisticsInterface& getAllocatorStatistics()
|
||||
return allocator_stats;
|
||||
}
|
||||
|
||||
#if defined HAVE_POSIX_MEMALIGN || defined HAVE_MEMALIGN
|
||||
static bool readMemoryAlignmentParameter()
|
||||
{
|
||||
bool value = true;
|
||||
#if defined(__GLIBC__) && defined(__linux__) \
|
||||
&& !defined(CV_STATIC_ANALYSIS) \
|
||||
&& !defined(OPENCV_ENABLE_MEMORY_SANITIZER) \
|
||||
&& !defined(FUZZING_BUILD_MODE_UNSAFE_FOR_PRODUCTION) /* oss-fuzz */ \
|
||||
&& !defined(_WIN32) /* MinGW? */
|
||||
{
|
||||
// https://github.com/opencv/opencv/issues/15526
|
||||
value = false;
|
||||
}
|
||||
#endif
|
||||
value = cv::utils::getConfigurationParameterBool("OPENCV_ENABLE_MEMALIGN", value); // should not call fastMalloc() internally
|
||||
// TODO add checks for valgrind, ASAN if value == false
|
||||
return value;
|
||||
}
|
||||
static inline
|
||||
bool isAlignedAllocationEnabled()
|
||||
{
|
||||
static bool initialized = false;
|
||||
static bool useMemalign = true;
|
||||
if (!initialized)
|
||||
{
|
||||
initialized = true; // trick to avoid stuck in acquire (works only if allocations are scope based)
|
||||
useMemalign = readMemoryAlignmentParameter();
|
||||
}
|
||||
return useMemalign;
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifdef OPENCV_ALLOC_ENABLE_STATISTICS
|
||||
static inline
|
||||
void* fastMalloc_(size_t size)
|
||||
@@ -89,25 +122,30 @@ void* fastMalloc(size_t size)
|
||||
#endif
|
||||
{
|
||||
#ifdef HAVE_POSIX_MEMALIGN
|
||||
void* ptr = NULL;
|
||||
if(posix_memalign(&ptr, CV_MALLOC_ALIGN, size))
|
||||
ptr = NULL;
|
||||
if(!ptr)
|
||||
return OutOfMemoryError(size);
|
||||
return ptr;
|
||||
if (isAlignedAllocationEnabled())
|
||||
{
|
||||
void* ptr = NULL;
|
||||
if(posix_memalign(&ptr, CV_MALLOC_ALIGN, size))
|
||||
ptr = NULL;
|
||||
if(!ptr)
|
||||
return OutOfMemoryError(size);
|
||||
return ptr;
|
||||
}
|
||||
#elif defined HAVE_MEMALIGN
|
||||
void* ptr = memalign(CV_MALLOC_ALIGN, size);
|
||||
if(!ptr)
|
||||
return OutOfMemoryError(size);
|
||||
return ptr;
|
||||
#else
|
||||
if (isAlignedAllocationEnabled())
|
||||
{
|
||||
void* ptr = memalign(CV_MALLOC_ALIGN, size);
|
||||
if(!ptr)
|
||||
return OutOfMemoryError(size);
|
||||
return ptr;
|
||||
}
|
||||
#endif
|
||||
uchar* udata = (uchar*)malloc(size + sizeof(void*) + CV_MALLOC_ALIGN);
|
||||
if(!udata)
|
||||
return OutOfMemoryError(size);
|
||||
uchar** adata = alignPtr((uchar**)udata + 1, CV_MALLOC_ALIGN);
|
||||
adata[-1] = udata;
|
||||
return adata;
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifdef OPENCV_ALLOC_ENABLE_STATISTICS
|
||||
@@ -118,8 +156,12 @@ void fastFree(void* ptr)
|
||||
#endif
|
||||
{
|
||||
#if defined HAVE_POSIX_MEMALIGN || defined HAVE_MEMALIGN
|
||||
free(ptr);
|
||||
#else
|
||||
if (isAlignedAllocationEnabled())
|
||||
{
|
||||
free(ptr);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
if(ptr)
|
||||
{
|
||||
uchar* udata = ((uchar**)ptr)[-1];
|
||||
@@ -127,7 +169,6 @@ void fastFree(void* ptr)
|
||||
((uchar*)ptr - udata) <= (ptrdiff_t)(sizeof(void*)+CV_MALLOC_ALIGN));
|
||||
free(udata);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
#ifdef OPENCV_ALLOC_ENABLE_STATISTICS
|
||||
|
||||
@@ -409,13 +409,13 @@ static void bin_loop(const T1* src1, size_t step1, const T1* src2, size_t step2,
|
||||
int x = 0;
|
||||
|
||||
#if CV_SIMD
|
||||
#if !CV_NEON
|
||||
#if !CV_NEON && !CV_MSA
|
||||
if (is_aligned(src1, src2, dst))
|
||||
{
|
||||
for (; x <= width - wide_step_l; x += wide_step_l)
|
||||
{
|
||||
ldr::la(src1 + x, src2 + x, dst + x);
|
||||
#if !CV_NEON && CV_SIMD_WIDTH == 16
|
||||
#if CV_SIMD_WIDTH == 16
|
||||
ldr::la(src1 + x + wide_step, src2 + x + wide_step, dst + x + wide_step);
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -782,36 +782,10 @@ void flip( InputArray _src, OutputArray _dst, int flip_mode )
|
||||
flipHoriz( dst.ptr(), dst.step, dst.ptr(), dst.step, dst.size(), esz );
|
||||
}
|
||||
|
||||
#ifdef HAVE_OPENCL
|
||||
|
||||
static bool ocl_rotate(InputArray _src, OutputArray _dst, int rotateMode)
|
||||
{
|
||||
switch (rotateMode)
|
||||
{
|
||||
case ROTATE_90_CLOCKWISE:
|
||||
transpose(_src, _dst);
|
||||
flip(_dst, _dst, 1);
|
||||
break;
|
||||
case ROTATE_180:
|
||||
flip(_src, _dst, -1);
|
||||
break;
|
||||
case ROTATE_90_COUNTERCLOCKWISE:
|
||||
transpose(_src, _dst);
|
||||
flip(_dst, _dst, 0);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
void rotate(InputArray _src, OutputArray _dst, int rotateMode)
|
||||
{
|
||||
CV_Assert(_src.dims() <= 2);
|
||||
|
||||
CV_OCL_RUN(_dst.isUMat(), ocl_rotate(_src, _dst, rotateMode))
|
||||
|
||||
switch (rotateMode)
|
||||
{
|
||||
case ROTATE_90_CLOCKWISE:
|
||||
|
||||
@@ -31,6 +31,11 @@ using namespace cv;
|
||||
|
||||
namespace {
|
||||
|
||||
static const float atan2_p1 = 0.9997878412794807f*(float)(180/CV_PI);
|
||||
static const float atan2_p3 = -0.3258083974640975f*(float)(180/CV_PI);
|
||||
static const float atan2_p5 = 0.1555786518463281f*(float)(180/CV_PI);
|
||||
static const float atan2_p7 = -0.04432655554792128f*(float)(180/CV_PI);
|
||||
|
||||
#ifdef __EMSCRIPTEN__
|
||||
static inline float atan_f32(float y, float x)
|
||||
{
|
||||
@@ -42,11 +47,6 @@ static inline float atan_f32(float y, float x)
|
||||
return a; // range [0; 360)
|
||||
}
|
||||
#else
|
||||
static const float atan2_p1 = 0.9997878412794807f*(float)(180/CV_PI);
|
||||
static const float atan2_p3 = -0.3258083974640975f*(float)(180/CV_PI);
|
||||
static const float atan2_p5 = 0.1555786518463281f*(float)(180/CV_PI);
|
||||
static const float atan2_p7 = -0.04432655554792128f*(float)(180/CV_PI);
|
||||
|
||||
static inline float atan_f32(float y, float x)
|
||||
{
|
||||
float ax = std::abs(x), ay = std::abs(y);
|
||||
|
||||
@@ -442,6 +442,12 @@ void transform(InputArray _src, OutputArray _dst, InputArray _mtx)
|
||||
_dst.create( src.size(), CV_MAKETYPE(depth, dcn) );
|
||||
Mat dst = _dst.getMat();
|
||||
|
||||
if (src.data == dst.data) // inplace case
|
||||
{
|
||||
CV_Assert(scn == dcn);
|
||||
src = src.clone(); // TODO Add performance warning
|
||||
}
|
||||
|
||||
int mtype = depth == CV_32S || depth == CV_64F ? CV_64F : CV_32F;
|
||||
AutoBuffer<double> _mbuf;
|
||||
double* mbuf;
|
||||
|
||||
@@ -2320,26 +2320,22 @@ double dotProd_8u(const uchar* src1, const uchar* src2, int len)
|
||||
while (i < len0)
|
||||
{
|
||||
blockSize = std::min(len0 - i, blockSize0);
|
||||
v_int32 v_sum = vx_setzero_s32();
|
||||
v_uint32 v_sum = vx_setzero_u32();
|
||||
const int cWidth = v_uint16::nlanes;
|
||||
|
||||
int j = 0;
|
||||
for (; j <= blockSize - cWidth * 2; j += cWidth * 2)
|
||||
{
|
||||
v_uint16 v_src10, v_src20, v_src11, v_src21;
|
||||
v_expand(vx_load(src1 + j), v_src10, v_src11);
|
||||
v_expand(vx_load(src2 + j), v_src20, v_src21);
|
||||
|
||||
v_sum += v_dotprod(v_reinterpret_as_s16(v_src10), v_reinterpret_as_s16(v_src20));
|
||||
v_sum += v_dotprod(v_reinterpret_as_s16(v_src11), v_reinterpret_as_s16(v_src21));
|
||||
v_uint8 v_src1 = vx_load(src1 + j);
|
||||
v_uint8 v_src2 = vx_load(src2 + j);
|
||||
v_sum = v_dotprod_expand_fast(v_src1, v_src2, v_sum);
|
||||
}
|
||||
|
||||
for (; j <= blockSize - cWidth; j += cWidth)
|
||||
{
|
||||
v_int16 v_src10 = v_reinterpret_as_s16(vx_load_expand(src1 + j));
|
||||
v_int16 v_src20 = v_reinterpret_as_s16(vx_load_expand(src2 + j));
|
||||
|
||||
v_sum += v_dotprod(v_src10, v_src20);
|
||||
v_sum += v_reinterpret_as_u32(v_dotprod_fast(v_src10, v_src20));
|
||||
}
|
||||
r += (double)v_reduce_sum(v_sum);
|
||||
|
||||
@@ -2348,48 +2344,6 @@ double dotProd_8u(const uchar* src1, const uchar* src2, int len)
|
||||
i += blockSize;
|
||||
}
|
||||
vx_cleanup();
|
||||
#elif CV_NEON
|
||||
if( cv::checkHardwareSupport(CV_CPU_NEON) )
|
||||
{
|
||||
int len0 = len & -8, blockSize0 = (1 << 15), blockSize;
|
||||
uint32x4_t v_zero = vdupq_n_u32(0u);
|
||||
CV_DECL_ALIGNED(16) uint buf[4];
|
||||
|
||||
while( i < len0 )
|
||||
{
|
||||
blockSize = std::min(len0 - i, blockSize0);
|
||||
uint32x4_t v_sum = v_zero;
|
||||
|
||||
int j = 0;
|
||||
for( ; j <= blockSize - 16; j += 16 )
|
||||
{
|
||||
uint8x16_t v_src1 = vld1q_u8(src1 + j), v_src2 = vld1q_u8(src2 + j);
|
||||
|
||||
uint16x8_t v_src10 = vmovl_u8(vget_low_u8(v_src1)), v_src20 = vmovl_u8(vget_low_u8(v_src2));
|
||||
v_sum = vmlal_u16(v_sum, vget_low_u16(v_src10), vget_low_u16(v_src20));
|
||||
v_sum = vmlal_u16(v_sum, vget_high_u16(v_src10), vget_high_u16(v_src20));
|
||||
|
||||
v_src10 = vmovl_u8(vget_high_u8(v_src1));
|
||||
v_src20 = vmovl_u8(vget_high_u8(v_src2));
|
||||
v_sum = vmlal_u16(v_sum, vget_low_u16(v_src10), vget_low_u16(v_src20));
|
||||
v_sum = vmlal_u16(v_sum, vget_high_u16(v_src10), vget_high_u16(v_src20));
|
||||
}
|
||||
|
||||
for( ; j <= blockSize - 8; j += 8 )
|
||||
{
|
||||
uint16x8_t v_src1 = vmovl_u8(vld1_u8(src1 + j)), v_src2 = vmovl_u8(vld1_u8(src2 + j));
|
||||
v_sum = vmlal_u16(v_sum, vget_low_u16(v_src1), vget_low_u16(v_src2));
|
||||
v_sum = vmlal_u16(v_sum, vget_high_u16(v_src1), vget_high_u16(v_src2));
|
||||
}
|
||||
|
||||
vst1q_u32(buf, v_sum);
|
||||
r += buf[0] + buf[1] + buf[2] + buf[3];
|
||||
|
||||
src1 += blockSize;
|
||||
src2 += blockSize;
|
||||
i += blockSize;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
return r + dotProd_(src1, src2, len - i);
|
||||
}
|
||||
@@ -2412,20 +2366,16 @@ double dotProd_8s(const schar* src1, const schar* src2, int len)
|
||||
int j = 0;
|
||||
for (; j <= blockSize - cWidth * 2; j += cWidth * 2)
|
||||
{
|
||||
v_int16 v_src10, v_src20, v_src11, v_src21;
|
||||
v_expand(vx_load(src1 + j), v_src10, v_src11);
|
||||
v_expand(vx_load(src2 + j), v_src20, v_src21);
|
||||
|
||||
v_sum += v_dotprod(v_src10, v_src20);
|
||||
v_sum += v_dotprod(v_src11, v_src21);
|
||||
v_int8 v_src1 = vx_load(src1 + j);
|
||||
v_int8 v_src2 = vx_load(src2 + j);
|
||||
v_sum = v_dotprod_expand_fast(v_src1, v_src2, v_sum);
|
||||
}
|
||||
|
||||
for (; j <= blockSize - cWidth; j += cWidth)
|
||||
{
|
||||
v_int16 v_src10 = vx_load_expand(src1 + j);
|
||||
v_int16 v_src20 = vx_load_expand(src2 + j);
|
||||
|
||||
v_sum += v_dotprod(v_src10, v_src20);
|
||||
v_int16 v_src1 = vx_load_expand(src1 + j);
|
||||
v_int16 v_src2 = vx_load_expand(src2 + j);
|
||||
v_sum = v_dotprod_fast(v_src1, v_src2, v_sum);
|
||||
}
|
||||
r += (double)v_reduce_sum(v_sum);
|
||||
|
||||
@@ -2434,48 +2384,6 @@ double dotProd_8s(const schar* src1, const schar* src2, int len)
|
||||
i += blockSize;
|
||||
}
|
||||
vx_cleanup();
|
||||
#elif CV_NEON
|
||||
if( cv::checkHardwareSupport(CV_CPU_NEON) )
|
||||
{
|
||||
int len0 = len & -8, blockSize0 = (1 << 14), blockSize;
|
||||
int32x4_t v_zero = vdupq_n_s32(0);
|
||||
CV_DECL_ALIGNED(16) int buf[4];
|
||||
|
||||
while( i < len0 )
|
||||
{
|
||||
blockSize = std::min(len0 - i, blockSize0);
|
||||
int32x4_t v_sum = v_zero;
|
||||
|
||||
int j = 0;
|
||||
for( ; j <= blockSize - 16; j += 16 )
|
||||
{
|
||||
int8x16_t v_src1 = vld1q_s8(src1 + j), v_src2 = vld1q_s8(src2 + j);
|
||||
|
||||
int16x8_t v_src10 = vmovl_s8(vget_low_s8(v_src1)), v_src20 = vmovl_s8(vget_low_s8(v_src2));
|
||||
v_sum = vmlal_s16(v_sum, vget_low_s16(v_src10), vget_low_s16(v_src20));
|
||||
v_sum = vmlal_s16(v_sum, vget_high_s16(v_src10), vget_high_s16(v_src20));
|
||||
|
||||
v_src10 = vmovl_s8(vget_high_s8(v_src1));
|
||||
v_src20 = vmovl_s8(vget_high_s8(v_src2));
|
||||
v_sum = vmlal_s16(v_sum, vget_low_s16(v_src10), vget_low_s16(v_src20));
|
||||
v_sum = vmlal_s16(v_sum, vget_high_s16(v_src10), vget_high_s16(v_src20));
|
||||
}
|
||||
|
||||
for( ; j <= blockSize - 8; j += 8 )
|
||||
{
|
||||
int16x8_t v_src1 = vmovl_s8(vld1_s8(src1 + j)), v_src2 = vmovl_s8(vld1_s8(src2 + j));
|
||||
v_sum = vmlal_s16(v_sum, vget_low_s16(v_src1), vget_low_s16(v_src2));
|
||||
v_sum = vmlal_s16(v_sum, vget_high_s16(v_src1), vget_high_s16(v_src2));
|
||||
}
|
||||
|
||||
vst1q_s32(buf, v_sum);
|
||||
r += buf[0] + buf[1] + buf[2] + buf[3];
|
||||
|
||||
src1 += blockSize;
|
||||
src2 += blockSize;
|
||||
i += blockSize;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
return r + dotProd_(src1, src2, len - i);
|
||||
@@ -2483,42 +2391,97 @@ double dotProd_8s(const schar* src1, const schar* src2, int len)
|
||||
|
||||
double dotProd_16u(const ushort* src1, const ushort* src2, int len)
|
||||
{
|
||||
return dotProd_(src1, src2, len);
|
||||
double r = 0.0;
|
||||
int i = 0;
|
||||
|
||||
#if CV_SIMD
|
||||
int len0 = len & -v_uint16::nlanes, blockSize0 = (1 << 24), blockSize;
|
||||
|
||||
while (i < len0)
|
||||
{
|
||||
blockSize = std::min(len0 - i, blockSize0);
|
||||
v_uint64 v_sum = vx_setzero_u64();
|
||||
const int cWidth = v_uint16::nlanes;
|
||||
|
||||
int j = 0;
|
||||
for (; j <= blockSize - cWidth; j += cWidth)
|
||||
{
|
||||
v_uint16 v_src1 = vx_load(src1 + j);
|
||||
v_uint16 v_src2 = vx_load(src2 + j);
|
||||
v_sum = v_dotprod_expand_fast(v_src1, v_src2, v_sum);
|
||||
}
|
||||
r += (double)v_reduce_sum(v_sum);
|
||||
|
||||
src1 += blockSize;
|
||||
src2 += blockSize;
|
||||
i += blockSize;
|
||||
}
|
||||
vx_cleanup();
|
||||
#endif
|
||||
return r + dotProd_(src1, src2, len - i);
|
||||
}
|
||||
|
||||
double dotProd_16s(const short* src1, const short* src2, int len)
|
||||
{
|
||||
return dotProd_(src1, src2, len);
|
||||
double r = 0.0;
|
||||
int i = 0;
|
||||
|
||||
#if CV_SIMD
|
||||
int len0 = len & -v_int16::nlanes, blockSize0 = (1 << 24), blockSize;
|
||||
|
||||
while (i < len0)
|
||||
{
|
||||
blockSize = std::min(len0 - i, blockSize0);
|
||||
v_int64 v_sum = vx_setzero_s64();
|
||||
const int cWidth = v_int16::nlanes;
|
||||
|
||||
int j = 0;
|
||||
for (; j <= blockSize - cWidth; j += cWidth)
|
||||
{
|
||||
v_int16 v_src1 = vx_load(src1 + j);
|
||||
v_int16 v_src2 = vx_load(src2 + j);
|
||||
v_sum = v_dotprod_expand_fast(v_src1, v_src2, v_sum);
|
||||
}
|
||||
r += (double)v_reduce_sum(v_sum);
|
||||
|
||||
src1 += blockSize;
|
||||
src2 += blockSize;
|
||||
i += blockSize;
|
||||
}
|
||||
vx_cleanup();
|
||||
#endif
|
||||
return r + dotProd_(src1, src2, len - i);
|
||||
}
|
||||
|
||||
double dotProd_32s(const int* src1, const int* src2, int len)
|
||||
{
|
||||
#if CV_SIMD128_64F
|
||||
double r = 0.0;
|
||||
#if CV_SIMD_64F
|
||||
double r = .0;
|
||||
int i = 0;
|
||||
int lenAligned = len & -v_int32x4::nlanes;
|
||||
v_float64x2 a(0.0, 0.0);
|
||||
v_float64x2 b(0.0, 0.0);
|
||||
|
||||
for( i = 0; i < lenAligned; i += v_int32x4::nlanes )
|
||||
{
|
||||
v_int32x4 s1 = v_load(src1);
|
||||
v_int32x4 s2 = v_load(src2);
|
||||
|
||||
#if CV_VSX
|
||||
// Do 32x32->64 multiplies, convert/round to double, accumulate
|
||||
// Potentially less precise than FMA, but 1.5x faster than fma below.
|
||||
a += v_cvt_f64(v_int64(vec_mule(s1.val, s2.val)));
|
||||
b += v_cvt_f64(v_int64(vec_mulo(s1.val, s2.val)));
|
||||
#else
|
||||
a = v_fma(v_cvt_f64(s1), v_cvt_f64(s2), a);
|
||||
b = v_fma(v_cvt_f64_high(s1), v_cvt_f64_high(s2), b);
|
||||
const int step = v_int32::nlanes;
|
||||
v_float64 v_sum0 = vx_setzero_f64();
|
||||
#if CV_SIMD_WIDTH == 16
|
||||
const int wstep = step * 2;
|
||||
v_float64 v_sum1 = vx_setzero_f64();
|
||||
for (; i < len - wstep; i += wstep, src1 += wstep, src2 += wstep)
|
||||
{
|
||||
v_int32 v_src10 = vx_load(src1);
|
||||
v_int32 v_src20 = vx_load(src2);
|
||||
v_int32 v_src11 = vx_load(src1 + step);
|
||||
v_int32 v_src21 = vx_load(src2 + step);
|
||||
v_sum0 = v_dotprod_expand_fast(v_src10, v_src20, v_sum0);
|
||||
v_sum1 = v_dotprod_expand_fast(v_src11, v_src21, v_sum1);
|
||||
}
|
||||
v_sum0 += v_sum1;
|
||||
#endif
|
||||
src1 += v_int32x4::nlanes;
|
||||
src2 += v_int32x4::nlanes;
|
||||
}
|
||||
a += b;
|
||||
r = v_reduce_sum(a);
|
||||
for (; i < len - step; i += step, src1 += step, src2 += step)
|
||||
{
|
||||
v_int32 v_src1 = vx_load(src1);
|
||||
v_int32 v_src2 = vx_load(src2);
|
||||
v_sum0 = v_dotprod_expand_fast(v_src1, v_src2, v_sum0);
|
||||
}
|
||||
r = v_reduce_sum(v_sum0);
|
||||
vx_cleanup();
|
||||
return r + dotProd_(src1, src2, len - i);
|
||||
#else
|
||||
return dotProd_(src1, src2, len);
|
||||
|
||||
@@ -267,6 +267,9 @@ static const String getBuildExtraOptions()
|
||||
return param_buildExtraOptions;
|
||||
}
|
||||
|
||||
static const bool CV_OPENCL_ENABLE_MEM_USE_HOST_PTR = utils::getConfigurationParameterBool("OPENCV_OPENCL_ENABLE_MEM_USE_HOST_PTR", true);
|
||||
static const size_t CV_OPENCL_ALIGNMENT_MEM_USE_HOST_PTR = utils::getConfigurationParameterSizeT("OPENCV_OPENCL_ALIGNMENT_MEM_USE_HOST_PTR", 4);
|
||||
|
||||
#endif // HAVE_OPENCL
|
||||
|
||||
struct UMat2D
|
||||
@@ -4671,6 +4674,9 @@ public:
|
||||
|
||||
bool allocate(UMatData* u, AccessFlag accessFlags, UMatUsageFlags usageFlags) const CV_OVERRIDE
|
||||
{
|
||||
#ifndef HAVE_OPENCL
|
||||
return false;
|
||||
#else
|
||||
if(!u)
|
||||
return false;
|
||||
|
||||
@@ -4746,8 +4752,12 @@ public:
|
||||
#endif
|
||||
{
|
||||
tempUMatFlags = UMatData::TEMP_UMAT;
|
||||
if (u->origdata == cv::alignPtr(u->origdata, 4) // There are OpenCL runtime issues for less aligned data
|
||||
&& !(u->originalUMatData && u->originalUMatData->handle) // Avoid sharing of host memory between OpenCL buffers
|
||||
if (CV_OPENCL_ENABLE_MEM_USE_HOST_PTR
|
||||
// There are OpenCL runtime issues for less aligned data
|
||||
&& (CV_OPENCL_ALIGNMENT_MEM_USE_HOST_PTR != 0
|
||||
&& u->origdata == cv::alignPtr(u->origdata, (int)CV_OPENCL_ALIGNMENT_MEM_USE_HOST_PTR))
|
||||
// Avoid sharing of host memory between OpenCL buffers
|
||||
&& !(u->originalUMatData && u->originalUMatData->handle)
|
||||
)
|
||||
{
|
||||
handle = clCreateBuffer(ctx_handle, CL_MEM_USE_HOST_PTR|createFlags,
|
||||
@@ -4777,6 +4787,7 @@ public:
|
||||
u->markHostCopyObsolete(true);
|
||||
opencl_allocator_stats.onAllocate(u->size);
|
||||
return true;
|
||||
#endif // HAVE_OPENCL
|
||||
}
|
||||
|
||||
/*void sync(UMatData* u) const
|
||||
@@ -4905,7 +4916,7 @@ public:
|
||||
(CL_MAP_READ | CL_MAP_WRITE),
|
||||
0, u->size, 0, 0, 0, &retval);
|
||||
CV_OCL_CHECK_RESULT(retval, cv::format("clEnqueueMapBuffer(handle=%p, sz=%lld) => %p", (void*)u->handle, (long long int)u->size, data).c_str());
|
||||
CV_Assert(u->origdata == data);
|
||||
CV_Assert(u->origdata == data && "Details: https://github.com/opencv/opencv/issues/6293");
|
||||
if (u->originalUMatData)
|
||||
{
|
||||
CV_Assert(u->originalUMatData->data == data);
|
||||
|
||||
@@ -45,6 +45,13 @@
|
||||
#ifdef HAVE_OPENGL
|
||||
# include "gl_core_3_1.hpp"
|
||||
# ifdef HAVE_CUDA
|
||||
# if (defined(__arm__) || defined(__aarch64__)) \
|
||||
&& !defined(OPENCV_SKIP_CUDA_OPENGL_ARM_WORKAROUND)
|
||||
# include <GL/gl.h>
|
||||
# ifndef GL_VERSION
|
||||
# define GL_VERSION 0x1F02
|
||||
# endif
|
||||
# endif
|
||||
# include <cuda_gl_interop.h>
|
||||
# endif
|
||||
#else // HAVE_OPENGL
|
||||
|
||||
@@ -54,7 +54,7 @@
|
||||
#endif
|
||||
|
||||
#if defined __linux__ || defined __APPLE__ || defined __GLIBC__ \
|
||||
|| defined __HAIKU__
|
||||
|| defined __HAIKU__ || defined __EMSCRIPTEN__
|
||||
#include <unistd.h>
|
||||
#include <stdio.h>
|
||||
#include <sys/types.h>
|
||||
@@ -808,7 +808,7 @@ int cv::getNumberOfCPUs(void)
|
||||
#elif defined __ANDROID__
|
||||
static int ncpus = getNumberOfCPUsImpl();
|
||||
return ncpus;
|
||||
#elif defined __linux__ || defined __GLIBC__ || defined __HAIKU__
|
||||
#elif defined __linux__ || defined __GLIBC__ || defined __HAIKU__ || defined __EMSCRIPTEN__
|
||||
return (int)sysconf( _SC_NPROCESSORS_ONLN );
|
||||
#elif defined __APPLE__
|
||||
int numCPU=0;
|
||||
|
||||
@@ -47,6 +47,8 @@ DECLARE_CV_PAUSE
|
||||
# define CV_PAUSE(v) do { for (int __delay = (v); __delay > 0; --__delay) { asm volatile("yield" ::: "memory"); } } while (0)
|
||||
# elif defined __GNUC__ && defined __arm__
|
||||
# define CV_PAUSE(v) do { for (int __delay = (v); __delay > 0; --__delay) { asm volatile("" ::: "memory"); } } while (0)
|
||||
# elif defined __GNUC__ && defined __mips__ && __mips_isa_rev >= 2
|
||||
# define CV_PAUSE(v) do { for (int __delay = (v); __delay > 0; --__delay) { asm volatile("pause" ::: "memory"); } } while (0)
|
||||
# elif defined __GNUC__ && defined __PPC64__
|
||||
# define CV_PAUSE(v) do { for (int __delay = (v); __delay > 0; --__delay) { asm volatile("or 27,27,27" ::: "memory"); } } while (0)
|
||||
# else
|
||||
|
||||
@@ -1307,6 +1307,9 @@ public:
|
||||
// In the case (b) the existing tag and the name are copied automatically.
|
||||
uchar* reserveNodeSpace(FileNode& node, size_t sz)
|
||||
{
|
||||
bool shrinkBlock = false;
|
||||
size_t shrinkBlockIdx = 0, shrinkSize = 0;
|
||||
|
||||
uchar *ptr = 0, *blockEnd = 0;
|
||||
|
||||
if( !fs_data_ptrs.empty() )
|
||||
@@ -1315,19 +1318,32 @@ public:
|
||||
size_t ofs = node.ofs;
|
||||
CV_Assert( blockIdx == fs_data_ptrs.size()-1 );
|
||||
CV_Assert( ofs <= fs_data_blksz[blockIdx] );
|
||||
CV_Assert( freeSpaceOfs <= fs_data_blksz[blockIdx] );
|
||||
//CV_Assert( freeSpaceOfs <= ofs + sz );
|
||||
|
||||
ptr = fs_data_ptrs[blockIdx] + ofs;
|
||||
blockEnd = fs_data_ptrs[blockIdx] + fs_data_blksz[blockIdx];
|
||||
|
||||
CV_Assert(ptr >= fs_data_ptrs[blockIdx] && ptr <= blockEnd);
|
||||
if( ptr + sz <= blockEnd )
|
||||
{
|
||||
freeSpaceOfs = ofs + sz;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
fs_data[blockIdx]->resize(ofs);
|
||||
fs_data_blksz[blockIdx] = ofs;
|
||||
if (ofs == 0) // FileNode is a first component of this block. Resize current block instead of allocation of new one.
|
||||
{
|
||||
fs_data[blockIdx]->resize(sz);
|
||||
ptr = &fs_data[blockIdx]->at(0);
|
||||
fs_data_ptrs[blockIdx] = ptr;
|
||||
fs_data_blksz[blockIdx] = sz;
|
||||
freeSpaceOfs = sz;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
shrinkBlock = true;
|
||||
shrinkBlockIdx = blockIdx;
|
||||
shrinkSize = ofs;
|
||||
}
|
||||
|
||||
size_t blockSize = std::max((size_t)CV_FS_MAX_LEN*4 - 256, sz) + 256;
|
||||
@@ -1352,6 +1368,12 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
if (shrinkBlock)
|
||||
{
|
||||
fs_data[shrinkBlockIdx]->resize(shrinkSize);
|
||||
fs_data_blksz[shrinkBlockIdx] = shrinkSize;
|
||||
}
|
||||
|
||||
return new_ptr;
|
||||
}
|
||||
|
||||
|
||||
@@ -2065,13 +2065,17 @@ static int_fast64_t f64_to_i64(float64_t a, uint_fast8_t roundingMode, bool exac
|
||||
if (exp) sig |= UINT64_C(0x0010000000000000);
|
||||
shiftDist = 0x433 - exp;
|
||||
if (shiftDist <= 0) {
|
||||
uint_fast64_t z = sig << -shiftDist;
|
||||
if ((shiftDist < -11) || (z & UINT64_C(0x8000000000000000)))
|
||||
bool isValid = shiftDist >= -11;
|
||||
if (isValid)
|
||||
{
|
||||
raiseFlags(flag_invalid);
|
||||
return sign ? i64_fromNegOverflow : i64_fromPosOverflow;
|
||||
uint_fast64_t z = sig << -shiftDist;
|
||||
if (0 == (z & UINT64_C(0x8000000000000000)))
|
||||
{
|
||||
return sign ? -(int_fast64_t)z : (int_fast64_t)z;
|
||||
}
|
||||
}
|
||||
return sign ? -(int_fast64_t)z : (int_fast64_t)z;
|
||||
raiseFlags(flag_invalid);
|
||||
return sign ? i64_fromNegOverflow : i64_fromPosOverflow;
|
||||
}
|
||||
else {
|
||||
if (shiftDist < 64)
|
||||
|
||||
@@ -368,11 +368,14 @@ struct HWFeatures
|
||||
g_hwFeatureNames[CPU_VSX] = "VSX";
|
||||
g_hwFeatureNames[CPU_VSX3] = "VSX3";
|
||||
|
||||
g_hwFeatureNames[CPU_MSA] = "CPU_MSA";
|
||||
|
||||
g_hwFeatureNames[CPU_AVX512_COMMON] = "AVX512-COMMON";
|
||||
g_hwFeatureNames[CPU_AVX512_SKX] = "AVX512-SKX";
|
||||
g_hwFeatureNames[CPU_AVX512_KNL] = "AVX512-KNL";
|
||||
g_hwFeatureNames[CPU_AVX512_KNM] = "AVX512-KNM";
|
||||
g_hwFeatureNames[CPU_AVX512_CNL] = "AVX512-CNL";
|
||||
g_hwFeatureNames[CPU_AVX512_CEL] = "AVX512-CEL";
|
||||
g_hwFeatureNames[CPU_AVX512_CLX] = "AVX512-CLX";
|
||||
g_hwFeatureNames[CPU_AVX512_ICL] = "AVX512-ICL";
|
||||
}
|
||||
|
||||
@@ -483,9 +486,11 @@ struct HWFeatures
|
||||
have[CV_CPU_AVX_5124VNNIW] && have[CV_CPU_AVX_512VPOPCNTDQ];
|
||||
have[CV_CPU_AVX512_SKX] = have[CV_CPU_AVX_512BW] && have[CV_CPU_AVX_512DQ] && have[CV_CPU_AVX_512VL];
|
||||
have[CV_CPU_AVX512_CNL] = have[CV_CPU_AVX512_SKX] && have[CV_CPU_AVX_512IFMA] && have[CV_CPU_AVX_512VBMI];
|
||||
have[CV_CPU_AVX512_CEL] = have[CV_CPU_AVX512_CNL] && have[CV_CPU_AVX_512VNNI];
|
||||
have[CV_CPU_AVX512_ICL] = have[CV_CPU_AVX512_CEL] && have[CV_CPU_AVX_512VBMI2] &&
|
||||
have[CV_CPU_AVX_512BITALG] && have[CV_CPU_AVX_512VPOPCNTDQ];
|
||||
have[CV_CPU_AVX512_CLX] = have[CV_CPU_AVX512_SKX] && have[CV_CPU_AVX_512VNNI];
|
||||
have[CV_CPU_AVX512_ICL] = have[CV_CPU_AVX512_SKX] &&
|
||||
have[CV_CPU_AVX_512IFMA] && have[CV_CPU_AVX_512VBMI] &&
|
||||
have[CV_CPU_AVX_512VNNI] &&
|
||||
have[CV_CPU_AVX_512VBMI2] && have[CV_CPU_AVX_512BITALG] && have[CV_CPU_AVX_512VPOPCNTDQ];
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -493,7 +498,7 @@ struct HWFeatures
|
||||
have[CV_CPU_AVX512_KNM] = false;
|
||||
have[CV_CPU_AVX512_SKX] = false;
|
||||
have[CV_CPU_AVX512_CNL] = false;
|
||||
have[CV_CPU_AVX512_CEL] = false;
|
||||
have[CV_CPU_AVX512_CLX] = false;
|
||||
have[CV_CPU_AVX512_ICL] = false;
|
||||
}
|
||||
}
|
||||
@@ -557,6 +562,9 @@ struct HWFeatures
|
||||
#if defined _ARM_ && (defined(_WIN32_WCE) && _WIN32_WCE >= 0x800)
|
||||
have[CV_CPU_NEON] = true;
|
||||
#endif
|
||||
#ifdef __mips_msa
|
||||
have[CV_CPU_MSA] = true;
|
||||
#endif
|
||||
// there's no need to check VSX availability in runtime since it's always available on ppc64le CPUs
|
||||
have[CV_CPU_VSX] = (CV_VSX);
|
||||
// TODO: Check VSX3 availability in runtime for other platforms
|
||||
@@ -567,8 +575,16 @@ struct HWFeatures
|
||||
have[CV_CPU_VSX3] = (CV_VSX3);
|
||||
#endif
|
||||
|
||||
bool skip_baseline_check = false;
|
||||
#ifndef NO_GETENV
|
||||
if (getenv("OPENCV_SKIP_CPU_BASELINE_CHECK"))
|
||||
{
|
||||
skip_baseline_check = true;
|
||||
}
|
||||
#endif
|
||||
int baseline_features[] = { CV_CPU_BASELINE_FEATURES };
|
||||
if (!checkFeatures(baseline_features, sizeof(baseline_features) / sizeof(baseline_features[0])))
|
||||
if (!checkFeatures(baseline_features, sizeof(baseline_features) / sizeof(baseline_features[0]))
|
||||
&& !skip_baseline_check)
|
||||
{
|
||||
fprintf(stderr, "\n"
|
||||
"******************************************************************\n"
|
||||
@@ -595,12 +611,12 @@ struct HWFeatures
|
||||
{
|
||||
if (have[feature])
|
||||
{
|
||||
if (dump) fprintf(stderr, "%s - OK\n", getHWFeatureNameSafe(feature));
|
||||
if (dump) fprintf(stderr, " ID=%3d (%s) - OK\n", feature, getHWFeatureNameSafe(feature));
|
||||
}
|
||||
else
|
||||
{
|
||||
result = false;
|
||||
if (dump) fprintf(stderr, "%s - NOT AVAILABLE\n", getHWFeatureNameSafe(feature));
|
||||
if (dump) fprintf(stderr, " ID=%3d (%s) - NOT AVAILABLE\n", feature, getHWFeatureNameSafe(feature));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -603,12 +603,14 @@ template<typename R> struct TheTest
|
||||
return *this;
|
||||
}
|
||||
|
||||
TheTest & test_dot_prod()
|
||||
TheTest & test_dotprod()
|
||||
{
|
||||
typedef typename V_RegTraits<R>::w_reg Rx2;
|
||||
typedef typename Rx2::lane_type w_type;
|
||||
|
||||
Data<R> dataA, dataB(2);
|
||||
Data<R> dataA, dataB;
|
||||
dataA += std::numeric_limits<LaneType>::max() - R::nlanes;
|
||||
dataB += std::numeric_limits<LaneType>::min() + R::nlanes;
|
||||
R a = dataA, b = dataB;
|
||||
|
||||
Data<Rx2> dataC;
|
||||
@@ -621,12 +623,95 @@ template<typename R> struct TheTest
|
||||
resE = v_dotprod(a, b, c);
|
||||
|
||||
const int n = R::nlanes / 2;
|
||||
w_type sumAB = 0, sumABC = 0, tmp_sum;
|
||||
for (int i = 0; i < n; ++i)
|
||||
{
|
||||
SCOPED_TRACE(cv::format("i=%d", i));
|
||||
EXPECT_EQ(dataA[i*2] * dataB[i*2] + dataA[i*2 + 1] * dataB[i*2 + 1], resD[i]);
|
||||
EXPECT_EQ(dataA[i*2] * dataB[i*2] + dataA[i*2 + 1] * dataB[i*2 + 1] + dataC[i], resE[i]);
|
||||
|
||||
tmp_sum = (w_type)dataA[i*2] * (w_type)dataB[i*2] +
|
||||
(w_type)dataA[i*2 + 1] * (w_type)dataB[i*2 + 1];
|
||||
sumAB += tmp_sum;
|
||||
EXPECT_EQ(tmp_sum, resD[i]);
|
||||
|
||||
tmp_sum = tmp_sum + dataC[i];
|
||||
sumABC += tmp_sum;
|
||||
EXPECT_EQ(tmp_sum, resE[i]);
|
||||
}
|
||||
|
||||
w_type resF = v_reduce_sum(v_dotprod_fast(a, b)),
|
||||
resG = v_reduce_sum(v_dotprod_fast(a, b, c));
|
||||
EXPECT_EQ(sumAB, resF);
|
||||
EXPECT_EQ(sumABC, resG);
|
||||
return *this;
|
||||
}
|
||||
|
||||
TheTest & test_dotprod_expand()
|
||||
{
|
||||
typedef typename V_RegTraits<R>::q_reg Rx4;
|
||||
typedef typename Rx4::lane_type l4_type;
|
||||
|
||||
Data<R> dataA, dataB;
|
||||
dataA += std::numeric_limits<LaneType>::max() - R::nlanes;
|
||||
dataB += std::numeric_limits<LaneType>::min() + R::nlanes;
|
||||
R a = dataA, b = dataB;
|
||||
|
||||
Data<Rx4> dataC;
|
||||
Rx4 c = dataC;
|
||||
|
||||
Data<Rx4> resD = v_dotprod_expand(a, b),
|
||||
resE = v_dotprod_expand(a, b, c);
|
||||
|
||||
l4_type sumAB = 0, sumABC = 0, tmp_sum;
|
||||
for (int i = 0; i < Rx4::nlanes; ++i)
|
||||
{
|
||||
SCOPED_TRACE(cv::format("i=%d", i));
|
||||
tmp_sum = (l4_type)dataA[i*4] * (l4_type)dataB[i*4] +
|
||||
(l4_type)dataA[i*4 + 1] * (l4_type)dataB[i*4 + 1] +
|
||||
(l4_type)dataA[i*4 + 2] * (l4_type)dataB[i*4 + 2] +
|
||||
(l4_type)dataA[i*4 + 3] * (l4_type)dataB[i*4 + 3];
|
||||
sumAB += tmp_sum;
|
||||
EXPECT_EQ(tmp_sum, resD[i]);
|
||||
|
||||
tmp_sum = tmp_sum + dataC[i];
|
||||
sumABC += tmp_sum;
|
||||
EXPECT_EQ(tmp_sum, resE[i]);
|
||||
}
|
||||
|
||||
l4_type resF = v_reduce_sum(v_dotprod_expand_fast(a, b)),
|
||||
resG = v_reduce_sum(v_dotprod_expand_fast(a, b, c));
|
||||
EXPECT_EQ(sumAB, resF);
|
||||
EXPECT_EQ(sumABC, resG);
|
||||
|
||||
return *this;
|
||||
}
|
||||
|
||||
TheTest & test_dotprod_expand_f64()
|
||||
{
|
||||
#if CV_SIMD_64F
|
||||
Data<R> dataA, dataB;
|
||||
dataA += std::numeric_limits<LaneType>::max() - R::nlanes;
|
||||
dataB += std::numeric_limits<LaneType>::min();
|
||||
R a = dataA, b = dataB;
|
||||
|
||||
Data<v_float64> dataC;
|
||||
v_float64 c = dataC;
|
||||
|
||||
Data<v_float64> resA = v_dotprod_expand(a, a),
|
||||
resB = v_dotprod_expand(b, b),
|
||||
resC = v_dotprod_expand(a, b, c);
|
||||
|
||||
const int n = R::nlanes / 2;
|
||||
for (int i = 0; i < n; ++i)
|
||||
{
|
||||
SCOPED_TRACE(cv::format("i=%d", i));
|
||||
EXPECT_EQ((double)dataA[i*2] * (double)dataA[i*2] +
|
||||
(double)dataA[i*2 + 1] * (double)dataA[i*2 + 1], resA[i]);
|
||||
EXPECT_EQ((double)dataB[i*2] * (double)dataB[i*2] +
|
||||
(double)dataB[i*2 + 1] * (double)dataB[i*2 + 1], resB[i]);
|
||||
EXPECT_EQ((double)dataA[i*2] * (double)dataB[i*2] +
|
||||
(double)dataA[i*2 + 1] * (double)dataB[i*2 + 1] + dataC[i], resC[i]);
|
||||
}
|
||||
#endif
|
||||
return *this;
|
||||
}
|
||||
|
||||
@@ -1165,6 +1250,29 @@ template<typename R> struct TheTest
|
||||
return *this;
|
||||
}
|
||||
|
||||
TheTest & test_cvt64_double()
|
||||
{
|
||||
#if CV_SIMD_64F
|
||||
Data<R> dataA(std::numeric_limits<LaneType>::max()),
|
||||
dataB(std::numeric_limits<LaneType>::min());
|
||||
dataB += R::nlanes;
|
||||
|
||||
R a = dataA, b = dataB;
|
||||
v_float64 c = v_cvt_f64(a), d = v_cvt_f64(b);
|
||||
|
||||
Data<v_float64> resC = c;
|
||||
Data<v_float64> resD = d;
|
||||
|
||||
for (int i = 0; i < R::nlanes; ++i)
|
||||
{
|
||||
SCOPED_TRACE(cv::format("i=%d", i));
|
||||
EXPECT_EQ((double)dataA[i], resC[i]);
|
||||
EXPECT_EQ((double)dataB[i], resD[i]);
|
||||
}
|
||||
#endif
|
||||
return *this;
|
||||
}
|
||||
|
||||
TheTest & test_matmul()
|
||||
{
|
||||
Data<R> dataV, dataA, dataB, dataC, dataD;
|
||||
@@ -1341,6 +1449,7 @@ void test_hal_intrin_uint8()
|
||||
.test_mul_expand()
|
||||
.test_cmp()
|
||||
.test_logic()
|
||||
.test_dotprod_expand()
|
||||
.test_min_max()
|
||||
.test_absdiff()
|
||||
.test_reduce_sad()
|
||||
@@ -1378,6 +1487,7 @@ void test_hal_intrin_int8()
|
||||
.test_mul_expand()
|
||||
.test_cmp()
|
||||
.test_logic()
|
||||
.test_dotprod_expand()
|
||||
.test_min_max()
|
||||
.test_absdiff()
|
||||
.test_absdiffs()
|
||||
@@ -1408,6 +1518,7 @@ void test_hal_intrin_uint16()
|
||||
.test_cmp()
|
||||
.test_shift<1>()
|
||||
.test_shift<8>()
|
||||
.test_dotprod_expand()
|
||||
.test_logic()
|
||||
.test_min_max()
|
||||
.test_absdiff()
|
||||
@@ -1437,7 +1548,8 @@ void test_hal_intrin_int16()
|
||||
.test_cmp()
|
||||
.test_shift<1>()
|
||||
.test_shift<8>()
|
||||
.test_dot_prod()
|
||||
.test_dotprod()
|
||||
.test_dotprod_expand()
|
||||
.test_logic()
|
||||
.test_min_max()
|
||||
.test_absdiff()
|
||||
@@ -1497,6 +1609,8 @@ void test_hal_intrin_int32()
|
||||
.test_cmp()
|
||||
.test_popcount()
|
||||
.test_shift<1>().test_shift<8>()
|
||||
.test_dotprod()
|
||||
.test_dotprod_expand_f64()
|
||||
.test_logic()
|
||||
.test_min_max()
|
||||
.test_absdiff()
|
||||
@@ -1538,6 +1652,7 @@ void test_hal_intrin_int64()
|
||||
.test_logic()
|
||||
.test_extract<0>().test_extract<1>()
|
||||
.test_rotate<0>().test_rotate<1>()
|
||||
.test_cvt64_double()
|
||||
;
|
||||
}
|
||||
|
||||
|
||||
@@ -707,10 +707,10 @@ static void test_filestorage_basic(int write_flags, const char* suffix_name, boo
|
||||
EXPECT_EQ(_em_in.depth(), _em_out.depth());
|
||||
EXPECT_TRUE(_em_in.empty());
|
||||
|
||||
EXPECT_EQ(_2d_in.rows , _2d_out.rows);
|
||||
EXPECT_EQ(_2d_in.cols , _2d_out.cols);
|
||||
EXPECT_EQ(_2d_in.dims , _2d_out.dims);
|
||||
EXPECT_EQ(_2d_in.depth(), _2d_out.depth());
|
||||
ASSERT_EQ(_2d_in.rows , _2d_out.rows);
|
||||
ASSERT_EQ(_2d_in.cols , _2d_out.cols);
|
||||
ASSERT_EQ(_2d_in.dims , _2d_out.dims);
|
||||
ASSERT_EQ(_2d_in.depth(), _2d_out.depth());
|
||||
|
||||
errors = 0;
|
||||
for(int i = 0; i < _2d_out.rows; ++i)
|
||||
@@ -731,16 +731,16 @@ static void test_filestorage_basic(int write_flags, const char* suffix_name, boo
|
||||
}
|
||||
}
|
||||
|
||||
EXPECT_EQ(_nd_in.rows , _nd_out.rows);
|
||||
EXPECT_EQ(_nd_in.cols , _nd_out.cols);
|
||||
EXPECT_EQ(_nd_in.dims , _nd_out.dims);
|
||||
EXPECT_EQ(_nd_in.depth(), _nd_out.depth());
|
||||
ASSERT_EQ(_nd_in.rows , _nd_out.rows);
|
||||
ASSERT_EQ(_nd_in.cols , _nd_out.cols);
|
||||
ASSERT_EQ(_nd_in.dims , _nd_out.dims);
|
||||
ASSERT_EQ(_nd_in.depth(), _nd_out.depth());
|
||||
EXPECT_EQ(0, cv::norm(_nd_in, _nd_out, NORM_INF));
|
||||
|
||||
EXPECT_EQ(_rd_in.rows , _rd_out.rows);
|
||||
EXPECT_EQ(_rd_in.cols , _rd_out.cols);
|
||||
EXPECT_EQ(_rd_in.dims , _rd_out.dims);
|
||||
EXPECT_EQ(_rd_in.depth(), _rd_out.depth());
|
||||
ASSERT_EQ(_rd_in.rows , _rd_out.rows);
|
||||
ASSERT_EQ(_rd_in.cols , _rd_out.cols);
|
||||
ASSERT_EQ(_rd_in.dims , _rd_out.dims);
|
||||
ASSERT_EQ(_rd_in.depth(), _rd_out.depth());
|
||||
EXPECT_EQ(0, cv::norm(_rd_in, _rd_out, NORM_INF));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -62,12 +62,21 @@ def printParams(backend, target):
|
||||
}
|
||||
print('%s/%s' % (backendNames[backend], targetNames[target]))
|
||||
|
||||
testdata_required = bool(os.environ.get('OPENCV_DNN_TEST_REQUIRE_TESTDATA', False))
|
||||
|
||||
g_dnnBackendsAndTargets = None
|
||||
|
||||
class dnn_test(NewOpenCVTests):
|
||||
|
||||
def setUp(self):
|
||||
super(dnn_test, self).setUp()
|
||||
|
||||
global g_dnnBackendsAndTargets
|
||||
if g_dnnBackendsAndTargets is None:
|
||||
g_dnnBackendsAndTargets = self.initBackendsAndTargets()
|
||||
self.dnnBackendsAndTargets = g_dnnBackendsAndTargets
|
||||
|
||||
def initBackendsAndTargets(self):
|
||||
self.dnnBackendsAndTargets = [
|
||||
[cv.dnn.DNN_BACKEND_OPENCV, cv.dnn.DNN_TARGET_CPU],
|
||||
]
|
||||
@@ -85,15 +94,18 @@ class dnn_test(NewOpenCVTests):
|
||||
self.dnnBackendsAndTargets.append([cv.dnn.DNN_BACKEND_INFERENCE_ENGINE, cv.dnn.DNN_TARGET_OPENCL])
|
||||
if self.checkIETarget(cv.dnn.DNN_BACKEND_INFERENCE_ENGINE, cv.dnn.DNN_TARGET_OPENCL_FP16):
|
||||
self.dnnBackendsAndTargets.append([cv.dnn.DNN_BACKEND_INFERENCE_ENGINE, cv.dnn.DNN_TARGET_OPENCL_FP16])
|
||||
return self.dnnBackendsAndTargets
|
||||
|
||||
def find_dnn_file(self, filename, required=True):
|
||||
if not required:
|
||||
required = testdata_required
|
||||
return self.find_file(filename, [os.environ.get('OPENCV_DNN_TEST_DATA_PATH', os.getcwd()),
|
||||
os.environ['OPENCV_TEST_DATA_PATH']],
|
||||
required=required)
|
||||
|
||||
def checkIETarget(self, backend, target):
|
||||
proto = self.find_dnn_file('dnn/layers/layer_convolution.prototxt', required=True)
|
||||
model = self.find_dnn_file('dnn/layers/layer_convolution.caffemodel', required=True)
|
||||
proto = self.find_dnn_file('dnn/layers/layer_convolution.prototxt')
|
||||
model = self.find_dnn_file('dnn/layers/layer_convolution.caffemodel')
|
||||
net = cv.dnn.readNet(proto, model)
|
||||
net.setPreferableBackend(backend)
|
||||
net.setPreferableTarget(target)
|
||||
@@ -134,8 +146,11 @@ class dnn_test(NewOpenCVTests):
|
||||
|
||||
def test_model(self):
|
||||
img_path = self.find_dnn_file("dnn/street.png")
|
||||
weights = self.find_dnn_file("dnn/MobileNetSSD_deploy.caffemodel")
|
||||
config = self.find_dnn_file("dnn/MobileNetSSD_deploy.prototxt")
|
||||
weights = self.find_dnn_file("dnn/MobileNetSSD_deploy.caffemodel", required=False)
|
||||
config = self.find_dnn_file("dnn/MobileNetSSD_deploy.prototxt", required=False)
|
||||
if weights is None or config is None:
|
||||
raise unittest.SkipTest("Missing DNN test files (dnn/MobileNetSSD_deploy.{prototxt/caffemodel}). Verify OPENCV_DNN_TEST_DATA_PATH configuration parameter.")
|
||||
|
||||
frame = cv.imread(img_path)
|
||||
model = cv.dnn_DetectionModel(weights, config)
|
||||
model.setInputParams(size=(300, 300), mean=(127.5, 127.5, 127.5), scale=1.0/127.5)
|
||||
@@ -163,9 +178,11 @@ class dnn_test(NewOpenCVTests):
|
||||
|
||||
def test_classification_model(self):
|
||||
img_path = self.find_dnn_file("dnn/googlenet_0.png")
|
||||
weights = self.find_dnn_file("dnn/squeezenet_v1.1.caffemodel")
|
||||
weights = self.find_dnn_file("dnn/squeezenet_v1.1.caffemodel", required=False)
|
||||
config = self.find_dnn_file("dnn/squeezenet_v1.1.prototxt")
|
||||
ref = np.load(self.find_dnn_file("dnn/squeezenet_v1.1_prob.npy"))
|
||||
if weights is None or config is None:
|
||||
raise unittest.SkipTest("Missing DNN test files (dnn/squeezenet_v1.1.{prototxt/caffemodel}). Verify OPENCV_DNN_TEST_DATA_PATH configuration parameter.")
|
||||
|
||||
frame = cv.imread(img_path)
|
||||
model = cv.dnn_ClassificationModel(config, weights)
|
||||
@@ -177,9 +194,8 @@ class dnn_test(NewOpenCVTests):
|
||||
|
||||
|
||||
def test_face_detection(self):
|
||||
testdata_required = bool(os.environ.get('OPENCV_DNN_TEST_REQUIRE_TESTDATA', False))
|
||||
proto = self.find_dnn_file('dnn/opencv_face_detector.prototxt', required=testdata_required)
|
||||
model = self.find_dnn_file('dnn/opencv_face_detector.caffemodel', required=testdata_required)
|
||||
proto = self.find_dnn_file('dnn/opencv_face_detector.prototxt')
|
||||
model = self.find_dnn_file('dnn/opencv_face_detector.caffemodel', required=False)
|
||||
if proto is None or model is None:
|
||||
raise unittest.SkipTest("Missing DNN test files (dnn/opencv_face_detector.{prototxt/caffemodel}). Verify OPENCV_DNN_TEST_DATA_PATH configuration parameter.")
|
||||
|
||||
@@ -215,10 +231,9 @@ class dnn_test(NewOpenCVTests):
|
||||
testScores, testBoxes, 0.5, scoresDiff, iouDiff)
|
||||
|
||||
def test_async(self):
|
||||
timeout = 500*10**6 # in nanoseconds (500ms)
|
||||
testdata_required = bool(os.environ.get('OPENCV_DNN_TEST_REQUIRE_TESTDATA', False))
|
||||
proto = self.find_dnn_file('dnn/layers/layer_convolution.prototxt', required=testdata_required)
|
||||
model = self.find_dnn_file('dnn/layers/layer_convolution.caffemodel', required=testdata_required)
|
||||
timeout = 10*1000*10**6 # in nanoseconds (10 sec)
|
||||
proto = self.find_dnn_file('dnn/layers/layer_convolution.prototxt')
|
||||
model = self.find_dnn_file('dnn/layers/layer_convolution.caffemodel')
|
||||
if proto is None or model is None:
|
||||
raise unittest.SkipTest("Missing DNN test files (dnn/layers/layer_convolution.{prototxt/caffemodel}). Verify OPENCV_DNN_TEST_DATA_PATH configuration parameter.")
|
||||
|
||||
|
||||
+22
-8
@@ -2506,17 +2506,25 @@ struct Net::Impl
|
||||
{
|
||||
std::vector<LayerPin>& inputLayerIds = layers[id].inputBlobsId;
|
||||
|
||||
if (inOutShapes[0].in[0].empty() && !layers[0].outputBlobs.empty())
|
||||
if (id == 0 && inOutShapes[id].in[0].empty())
|
||||
{
|
||||
ShapesVec shapes;
|
||||
for (int i = 0; i < layers[0].outputBlobs.size(); i++)
|
||||
if (!layers[0].outputBlobs.empty())
|
||||
{
|
||||
Mat& inp = layers[0].outputBlobs[i];
|
||||
CV_Assert(inp.total());
|
||||
shapes.push_back(shape(inp));
|
||||
ShapesVec shapes;
|
||||
for (int i = 0; i < layers[0].outputBlobs.size(); i++)
|
||||
{
|
||||
Mat& inp = layers[0].outputBlobs[i];
|
||||
CV_Assert(inp.total());
|
||||
shapes.push_back(shape(inp));
|
||||
}
|
||||
inOutShapes[0].in = shapes;
|
||||
}
|
||||
inOutShapes[0].in = shapes;
|
||||
}
|
||||
else
|
||||
{
|
||||
inOutShapes[0].out.clear();
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (inOutShapes[id].in.empty())
|
||||
{
|
||||
@@ -2540,6 +2548,12 @@ struct Net::Impl
|
||||
int requiredOutputs = layers[id].requiredOutputs.size();
|
||||
inOutShapes[id].supportInPlace =
|
||||
layers[id].getLayerInstance()->getMemoryShapes(is, requiredOutputs, os, ints);
|
||||
|
||||
for (int i = 0; i < ints.size(); i++)
|
||||
CV_Assert(total(ints[i]) > 0);
|
||||
|
||||
for (int i = 0; i < os.size(); i++)
|
||||
CV_Assert(total(os[i]) > 0);
|
||||
}
|
||||
|
||||
void getLayersShapes(const ShapesVec& netInputShapes,
|
||||
|
||||
@@ -241,10 +241,14 @@ public:
|
||||
|
||||
MatShape computeColRowShape(const MatShape &inpShape, const MatShape &outShape) const CV_OVERRIDE
|
||||
{
|
||||
Size out(outShape[3], outShape[2]);
|
||||
int dims = inpShape.size();
|
||||
int inpD = dims == 5 ? inpShape[2] : 1;
|
||||
int inpH = inpShape[dims - 2];
|
||||
int inpW = inpShape.back();
|
||||
int inpGroupCn = blobs[0].size[1];
|
||||
int ksize = inpGroupCn * kernel.height * kernel.width;
|
||||
return shape(out.area(), ksize);
|
||||
int ksize = inpGroupCn * std::accumulate(kernel_size.begin(), kernel_size.end(),
|
||||
1, std::multiplies<size_t>());
|
||||
return shape(inpD * inpH * inpW, ksize);
|
||||
}
|
||||
|
||||
virtual bool supportBackend(int backendId) CV_OVERRIDE
|
||||
@@ -1304,14 +1308,17 @@ public:
|
||||
|
||||
MatShape computeColRowShape(const MatShape &inpShape, const MatShape &outShape) const CV_OVERRIDE
|
||||
{
|
||||
int dims = inpShape.size();
|
||||
int inpCn = inpShape[1];
|
||||
int inpH = inpShape[2];
|
||||
int inpW = inpShape[3];
|
||||
int inpD = dims == 5 ? inpShape[2] : 1;
|
||||
int inpH = inpShape[dims - 2];
|
||||
int inpW = inpShape.back();
|
||||
int outCn = outShape[1];
|
||||
int ngroups = inpCn / blobs[0].size[0];
|
||||
int outGroupCn = outCn / ngroups;
|
||||
int ksize = outGroupCn * kernel.height * kernel.width;
|
||||
return shape(ksize, inpH * inpW);
|
||||
int ksize = outGroupCn * std::accumulate(kernel_size.begin(), kernel_size.end(),
|
||||
1, std::multiplies<size_t>());
|
||||
return shape(ksize, inpD * inpH * inpW);
|
||||
}
|
||||
|
||||
virtual bool supportBackend(int backendId) CV_OVERRIDE
|
||||
|
||||
@@ -163,6 +163,8 @@ public:
|
||||
{
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE)
|
||||
{
|
||||
if (computeMaxIdx)
|
||||
return false;
|
||||
#ifdef HAVE_INF_ENGINE
|
||||
if (kernel_size.size() == 3)
|
||||
return preferableTarget == DNN_TARGET_CPU;
|
||||
|
||||
@@ -92,6 +92,7 @@ class LSTMLayerImpl CV_FINAL : public LSTMLayer
|
||||
bool produceCellOutput;
|
||||
float forgetBias, cellClip;
|
||||
bool useCellClip, usePeephole;
|
||||
bool reverse; // If true, go in negative direction along the time axis
|
||||
|
||||
public:
|
||||
|
||||
@@ -133,6 +134,7 @@ public:
|
||||
cellClip = params.get<float>("cell_clip", 0.0f);
|
||||
useCellClip = params.get<bool>("use_cell_clip", false);
|
||||
usePeephole = params.get<bool>("use_peephole", false);
|
||||
reverse = params.get<bool>("reverse", false);
|
||||
|
||||
allocated = false;
|
||||
outTailShape.clear();
|
||||
@@ -288,7 +290,18 @@ public:
|
||||
Mat hOutTs = output[0].reshape(1, numSamplesTotal);
|
||||
Mat cOutTs = produceCellOutput ? output[1].reshape(1, numSamplesTotal) : Mat();
|
||||
|
||||
for (int ts = 0; ts < numTimeStamps; ts++)
|
||||
int tsStart, tsEnd, tsInc;
|
||||
if (reverse) {
|
||||
tsStart = numTimeStamps - 1;
|
||||
tsEnd = -1;
|
||||
tsInc = -1;
|
||||
}
|
||||
else {
|
||||
tsStart = 0;
|
||||
tsEnd = numTimeStamps;
|
||||
tsInc = 1;
|
||||
}
|
||||
for (int ts = tsStart; ts != tsEnd; ts += tsInc)
|
||||
{
|
||||
Range curRowRange(ts*numSamples, (ts + 1)*numSamples);
|
||||
Mat xCurr = xTs.rowRange(curRowRange);
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
#define CV_TEST_TAG_DNN_SKIP_IE_2019R1 "dnn_skip_ie_2019r1"
|
||||
#define CV_TEST_TAG_DNN_SKIP_IE_2019R1_1 "dnn_skip_ie_2019r1_1"
|
||||
#define CV_TEST_TAG_DNN_SKIP_IE_2019R2 "dnn_skip_ie_2019r2"
|
||||
#define CV_TEST_TAG_DNN_SKIP_IE_2019R3 "dnn_skip_ie_2019r3"
|
||||
#define CV_TEST_TAG_DNN_SKIP_IE_OPENCL "dnn_skip_ie_ocl"
|
||||
#define CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16 "dnn_skip_ie_ocl_fp16"
|
||||
#define CV_TEST_TAG_DNN_SKIP_IE_MYRIAD_2 "dnn_skip_ie_myriad2"
|
||||
|
||||
@@ -324,6 +324,8 @@ void initDNNTests()
|
||||
# endif
|
||||
#elif INF_ENGINE_VER_MAJOR_EQ(2019020000)
|
||||
CV_TEST_TAG_DNN_SKIP_IE_2019R2,
|
||||
#elif INF_ENGINE_VER_MAJOR_EQ(2019030000)
|
||||
CV_TEST_TAG_DNN_SKIP_IE_2019R3,
|
||||
#endif
|
||||
CV_TEST_TAG_DNN_SKIP_IE
|
||||
);
|
||||
|
||||
@@ -329,7 +329,7 @@ TEST_P(Test_Darknet_nets, TinyYoloVoc)
|
||||
}
|
||||
|
||||
#ifdef HAVE_INF_ENGINE
|
||||
static const std::chrono::milliseconds async_timeout(500);
|
||||
static const std::chrono::milliseconds async_timeout(10000);
|
||||
|
||||
typedef testing::TestWithParam<tuple<std::string, Target> > Test_Darknet_nets_async;
|
||||
TEST_P(Test_Darknet_nets_async, Accuracy)
|
||||
|
||||
@@ -554,9 +554,9 @@ TEST_P(ReLU, Accuracy)
|
||||
Backend backendId = get<0>(get<1>(GetParam()));
|
||||
Target targetId = get<1>(get<1>(GetParam()));
|
||||
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2019020000)
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_GE(2019020000)
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE && targetId == DNN_TARGET_MYRIAD && negativeSlope < 0)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE, CV_TEST_TAG_DNN_SKIP_IE_2019R2);
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_2019R3, CV_TEST_TAG_DNN_SKIP_IE_2019R2, CV_TEST_TAG_DNN_SKIP_IE);
|
||||
#endif
|
||||
|
||||
LayerParams lp;
|
||||
@@ -758,6 +758,12 @@ TEST_P(Eltwise, Accuracy)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE, CV_TEST_TAG_DNN_SKIP_IE_2019R1, CV_TEST_TAG_DNN_SKIP_IE_2019R1_1);
|
||||
#endif
|
||||
|
||||
#if defined(INF_ENGINE_RELEASE)
|
||||
if (backendId == DNN_BACKEND_INFERENCE_ENGINE && targetId == DNN_TARGET_OPENCL &&
|
||||
op == "sum" && numConv == 1 && !weighted)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL, CV_TEST_TAG_DNN_SKIP_IE);
|
||||
#endif
|
||||
|
||||
Net net;
|
||||
|
||||
std::vector<int> convLayerIds(numConv);
|
||||
|
||||
@@ -461,6 +461,55 @@ TEST(Layer_RNN_Test_Accuracy_with_, CaffeRecurrent)
|
||||
normAssert(h_ref, output[0]);
|
||||
}
|
||||
|
||||
TEST(Layer_LSTM_Test_Accuracy_, Reverse)
|
||||
{
|
||||
// This handcrafted setup calculates (approximately) the prefix sum of the
|
||||
// input, assuming the inputs are suitably small.
|
||||
cv::Mat input(2, 1, CV_32FC1);
|
||||
input.at<float>(0, 0) = 1e-5f;
|
||||
input.at<float>(1, 0) = 2e-5f;
|
||||
|
||||
cv::Mat Wx(4, 1, CV_32FC1);
|
||||
Wx.at<float>(0, 0) = 0.f; // Input gate
|
||||
Wx.at<float>(1, 0) = 0.f; // Forget gate
|
||||
Wx.at<float>(2, 0) = 0.f; // Output gate
|
||||
Wx.at<float>(3, 0) = 1.f; // Update signal
|
||||
|
||||
cv::Mat Wh(4, 1, CV_32FC1);
|
||||
Wh.at<float>(0, 0) = 0.f; // Input gate
|
||||
Wh.at<float>(1, 0) = 0.f; // Forget gate
|
||||
Wh.at<float>(2, 0) = 0.f; // Output gate
|
||||
Wh.at<float>(3, 0) = 0.f; // Update signal
|
||||
|
||||
cv::Mat bias(4, 1, CV_32FC1);
|
||||
bias.at<float>(0, 0) = 1e10f; // Input gate - always allows input to c
|
||||
bias.at<float>(1, 0) = 1e10f; // Forget gate - never forget anything on c
|
||||
bias.at<float>(2, 0) = 1e10f; // Output gate - always output everything
|
||||
bias.at<float>(3, 0) = 0.f; // Update signal
|
||||
|
||||
LayerParams lp;
|
||||
lp.set("reverse", true);
|
||||
lp.set("use_timestamp_dim", true);
|
||||
lp.blobs.clear();
|
||||
lp.blobs.push_back(Wh);
|
||||
lp.blobs.push_back(Wx);
|
||||
lp.blobs.push_back(bias);
|
||||
|
||||
cv::Ptr<cv::dnn::LSTMLayer> layer = LSTMLayer::create(lp);
|
||||
std::vector<cv::Mat> outputs;
|
||||
std::vector<cv::Mat> inputs;
|
||||
inputs.push_back(input);
|
||||
runLayer(layer, inputs, outputs);
|
||||
|
||||
ASSERT_EQ(1, outputs.size());
|
||||
cv::Mat out = outputs[0];
|
||||
ASSERT_EQ(3, out.dims);
|
||||
ASSERT_EQ(shape(2, 1, 1), shape(out));
|
||||
float* data = reinterpret_cast<float*>(out.data);
|
||||
EXPECT_NEAR(std::tanh(1e-5f) + std::tanh(2e-5f), data[0], 1e-10);
|
||||
EXPECT_NEAR(std::tanh(2e-5f), data[1], 1e-10);
|
||||
}
|
||||
|
||||
|
||||
class Layer_RNN_Test : public ::testing::Test
|
||||
{
|
||||
|
||||
@@ -363,7 +363,7 @@ TEST(Net, forwardAndRetrieve)
|
||||
}
|
||||
|
||||
#ifdef HAVE_INF_ENGINE
|
||||
static const std::chrono::milliseconds async_timeout(500);
|
||||
static const std::chrono::milliseconds async_timeout(10000);
|
||||
|
||||
// This test runs network in synchronous mode for different inputs and then
|
||||
// runs the same model asynchronously for the same inputs.
|
||||
|
||||
@@ -8,10 +8,10 @@
|
||||
namespace opencv_test { namespace {
|
||||
|
||||
template<typename TString>
|
||||
static std::string _tf(TString filename)
|
||||
static std::string _tf(TString filename, bool required = true)
|
||||
{
|
||||
String rootFolder = "dnn/";
|
||||
return findDataFile(rootFolder + filename);
|
||||
return findDataFile(rootFolder + filename, required);
|
||||
}
|
||||
|
||||
|
||||
@@ -96,7 +96,7 @@ TEST_P(Test_Model, Classify)
|
||||
|
||||
std::string img_path = _tf("grace_hopper_227.png");
|
||||
std::string config_file = _tf("bvlc_alexnet.prototxt");
|
||||
std::string weights_file = _tf("bvlc_alexnet.caffemodel");
|
||||
std::string weights_file = _tf("bvlc_alexnet.caffemodel", false);
|
||||
|
||||
Size size{227, 227};
|
||||
float norm = 1e-4;
|
||||
@@ -127,7 +127,7 @@ TEST_P(Test_Model, DetectRegion)
|
||||
Rect2d(58, 141, 117, 249)};
|
||||
|
||||
std::string img_path = _tf("dog416.png");
|
||||
std::string weights_file = _tf("yolo-voc.weights");
|
||||
std::string weights_file = _tf("yolo-voc.weights", false);
|
||||
std::string config_file = _tf("yolo-voc.cfg");
|
||||
|
||||
double scale = 1.0 / 255.0;
|
||||
@@ -160,7 +160,7 @@ TEST_P(Test_Model, DetectionOutput)
|
||||
Rect2d(132, 223, 207, 344)};
|
||||
|
||||
std::string img_path = _tf("dog416.png");
|
||||
std::string weights_file = _tf("resnet50_rfcn_final.caffemodel");
|
||||
std::string weights_file = _tf("resnet50_rfcn_final.caffemodel", false);
|
||||
std::string config_file = _tf("rfcn_pascal_voc_resnet50.prototxt");
|
||||
|
||||
Scalar mean = Scalar(102.9801, 115.9465, 122.7717);
|
||||
@@ -203,7 +203,7 @@ TEST_P(Test_Model, DetectionMobilenetSSD)
|
||||
refBoxes.emplace_back(left, top, width, height);
|
||||
}
|
||||
|
||||
std::string weights_file = _tf("MobileNetSSD_deploy.caffemodel");
|
||||
std::string weights_file = _tf("MobileNetSSD_deploy.caffemodel", false);
|
||||
std::string config_file = _tf("MobileNetSSD_deploy.prototxt");
|
||||
|
||||
Scalar mean = Scalar(127.5, 127.5, 127.5);
|
||||
@@ -228,7 +228,7 @@ TEST_P(Test_Model, Detection_normalized)
|
||||
std::vector<float> refConfidences = {0.999222f};
|
||||
std::vector<Rect2d> refBoxes = {Rect2d(0, 4, 227, 222)};
|
||||
|
||||
std::string weights_file = _tf("MobileNetSSD_deploy.caffemodel");
|
||||
std::string weights_file = _tf("MobileNetSSD_deploy.caffemodel", false);
|
||||
std::string config_file = _tf("MobileNetSSD_deploy.prototxt");
|
||||
|
||||
Scalar mean = Scalar(127.5, 127.5, 127.5);
|
||||
@@ -247,7 +247,7 @@ TEST_P(Test_Model, Segmentation)
|
||||
{
|
||||
std::string inp = _tf("dog416.png");
|
||||
std::string weights_file = _tf("fcn8s-heavy-pascal.prototxt");
|
||||
std::string config_file = _tf("fcn8s-heavy-pascal.caffemodel");
|
||||
std::string config_file = _tf("fcn8s-heavy-pascal.caffemodel", false);
|
||||
std::string exp = _tf("segmentation_exp.png");
|
||||
|
||||
Size size{128, 128};
|
||||
|
||||
@@ -86,8 +86,8 @@ TEST_P(Test_ONNX_layers, InstanceNorm)
|
||||
|
||||
TEST_P(Test_ONNX_layers, MaxPooling)
|
||||
{
|
||||
testONNXModels("maxpooling");
|
||||
testONNXModels("two_maxpooling");
|
||||
testONNXModels("maxpooling", npy, 0, 0, false, false);
|
||||
testONNXModels("two_maxpooling", npy, 0, 0, false, false);
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, Convolution)
|
||||
@@ -212,7 +212,7 @@ TEST_P(Test_ONNX_layers, MaxPooling3D)
|
||||
#endif
|
||||
if (target != DNN_TARGET_CPU)
|
||||
throw SkipTestException("Only CPU is supported");
|
||||
testONNXModels("max_pool3d");
|
||||
testONNXModels("max_pool3d", npy, 0, 0, false, false);
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_layers, AvePooling3D)
|
||||
@@ -422,13 +422,22 @@ TEST_P(Test_ONNX_nets, Googlenet)
|
||||
TEST_P(Test_ONNX_nets, CaffeNet)
|
||||
{
|
||||
applyTestTag(target == DNN_TARGET_CPU ? CV_TEST_TAG_MEMORY_512MB : CV_TEST_TAG_MEMORY_1GB);
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2019030000)
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD
|
||||
&& getInferenceEngineVPUType() == CV_DNN_INFERENCE_ENGINE_VPU_TYPE_MYRIAD_X)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD_X, CV_TEST_TAG_DNN_SKIP_IE_2019R3);
|
||||
#endif
|
||||
testONNXModels("caffenet", pb);
|
||||
}
|
||||
|
||||
TEST_P(Test_ONNX_nets, RCNN_ILSVRC13)
|
||||
{
|
||||
applyTestTag(target == DNN_TARGET_CPU ? CV_TEST_TAG_MEMORY_512MB : CV_TEST_TAG_MEMORY_1GB);
|
||||
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2019030000)
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target == DNN_TARGET_MYRIAD
|
||||
&& getInferenceEngineVPUType() == CV_DNN_INFERENCE_ENGINE_VPU_TYPE_MYRIAD_X)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD_X, CV_TEST_TAG_DNN_SKIP_IE_2019R3);
|
||||
#endif
|
||||
// Reference output values are in range [-4.992, -1.161]
|
||||
testONNXModels("rcnn_ilsvrc13", pb, 0.0045);
|
||||
}
|
||||
|
||||
@@ -146,13 +146,13 @@ TEST_P(Test_TensorFlow_layers, padding)
|
||||
runTensorFlowNet("padding_valid");
|
||||
runTensorFlowNet("spatial_padding");
|
||||
runTensorFlowNet("mirror_pad");
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_EQ(2019020000)
|
||||
#if defined(INF_ENGINE_RELEASE) && INF_ENGINE_VER_MAJOR_GE(2019020000)
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE)
|
||||
{
|
||||
if (target == DNN_TARGET_MYRIAD)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_2019R2);
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD, CV_TEST_TAG_DNN_SKIP_IE_2019R3, CV_TEST_TAG_DNN_SKIP_IE_2019R2);
|
||||
if (target == DNN_TARGET_OPENCL_FP16)
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16, CV_TEST_TAG_DNN_SKIP_IE_2019R2);
|
||||
applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16, CV_TEST_TAG_DNN_SKIP_IE_2019R3, CV_TEST_TAG_DNN_SKIP_IE_2019R2);
|
||||
}
|
||||
#endif
|
||||
runTensorFlowNet("keras_pad_concat");
|
||||
|
||||
@@ -337,9 +337,15 @@ TEST_P(Test_Torch_nets, ENet_accuracy)
|
||||
{
|
||||
applyTestTag(target == DNN_TARGET_CPU ? "" : CV_TEST_TAG_MEMORY_512MB);
|
||||
checkBackend();
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE ||
|
||||
(backend == DNN_BACKEND_OPENCV && target == DNN_TARGET_OPENCL_FP16))
|
||||
applyTestTag(target == DNN_TARGET_OPENCL ? CV_TEST_TAG_DNN_SKIP_IE_OPENCL : CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16);
|
||||
if (backend == DNN_BACKEND_OPENCV && target == DNN_TARGET_OPENCL_FP16)
|
||||
throw SkipTestException("");
|
||||
if (backend == DNN_BACKEND_INFERENCE_ENGINE && target != DNN_TARGET_CPU)
|
||||
{
|
||||
if (target == DNN_TARGET_OPENCL_FP16) applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL_FP16);
|
||||
if (target == DNN_TARGET_OPENCL) applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_OPENCL);
|
||||
if (target == DNN_TARGET_MYRIAD) applyTestTag(CV_TEST_TAG_DNN_SKIP_IE_MYRIAD);
|
||||
throw SkipTestException("");
|
||||
}
|
||||
|
||||
Net net;
|
||||
{
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
# (Restructure directories, add common pass, etc)
|
||||
if (NOT DEFINED OPENCV_INITIAL_PASS)
|
||||
cmake_minimum_required(VERSION 3.3)
|
||||
project(gapi_standalone)
|
||||
include("cmake/standalone.cmake")
|
||||
return()
|
||||
endif()
|
||||
@@ -47,6 +48,7 @@ set(gapi_srcs
|
||||
src/api/kernels_core.cpp
|
||||
src/api/kernels_imgproc.cpp
|
||||
src/api/render.cpp
|
||||
src/api/render_ocv.cpp
|
||||
src/api/ginfer.cpp
|
||||
|
||||
# Compiler part
|
||||
@@ -61,7 +63,9 @@ set(gapi_srcs
|
||||
src/compiler/passes/meta.cpp
|
||||
src/compiler/passes/kernels.cpp
|
||||
src/compiler/passes/exec.cpp
|
||||
src/compiler/passes/transformations.cpp
|
||||
src/compiler/passes/pattern_matching.cpp
|
||||
src/compiler/passes/perform_substitution.cpp
|
||||
|
||||
# Executor
|
||||
src/executor/gexecutor.cpp
|
||||
|
||||
@@ -1275,8 +1275,8 @@ GAPI_EXPORTS std::tuple<GMat, GMat> integral(const GMat& src, int sdepth = -1, i
|
||||
The function applies fixed-level thresholding to a single- or multiple-channel matrix.
|
||||
The function is typically used to get a bi-level (binary) image out of a grayscale image ( cmp functions could be also used for
|
||||
this purpose) or for removing a noise, that is, filtering out pixels with too small or too large
|
||||
values. There are several depths of thresholding supported by the function. They are determined by
|
||||
depth parameter.
|
||||
values. There are several types of thresholding supported by the function. They are determined by
|
||||
type parameter.
|
||||
|
||||
Also, the special values cv::THRESH_OTSU or cv::THRESH_TRIANGLE may be combined with one of the
|
||||
above values. In these cases, the function determines the optimal threshold value using the Otsu's
|
||||
@@ -1292,17 +1292,17 @@ Output matrix must be of the same size and depth as src.
|
||||
@param src input matrix (@ref CV_8UC1, @ref CV_8UC3, or @ref CV_32FC1).
|
||||
@param thresh threshold value.
|
||||
@param maxval maximum value to use with the cv::THRESH_BINARY and cv::THRESH_BINARY_INV thresholding
|
||||
depths.
|
||||
@param depth thresholding depth (see the cv::ThresholdTypes).
|
||||
types.
|
||||
@param type thresholding type (see the cv::ThresholdTypes).
|
||||
|
||||
@sa min, max, cmpGT, cmpLE, cmpGE, cmpLS
|
||||
*/
|
||||
GAPI_EXPORTS GMat threshold(const GMat& src, const GScalar& thresh, const GScalar& maxval, int depth);
|
||||
GAPI_EXPORTS GMat threshold(const GMat& src, const GScalar& thresh, const GScalar& maxval, int type);
|
||||
/** @overload
|
||||
This function applicable for all threshold depths except CV_THRESH_OTSU and CV_THRESH_TRIANGLE
|
||||
This function applicable for all threshold types except CV_THRESH_OTSU and CV_THRESH_TRIANGLE
|
||||
@note Function textual ID is "org.opencv.core.matrixop.thresholdOT"
|
||||
*/
|
||||
GAPI_EXPORTS std::tuple<GMat, GScalar> threshold(const GMat& src, const GScalar& maxval, int depth);
|
||||
GAPI_EXPORTS std::tuple<GMat, GScalar> threshold(const GMat& src, const GScalar& maxval, int type);
|
||||
|
||||
/** @brief Applies a range-level threshold to each matrix element.
|
||||
|
||||
|
||||
@@ -106,7 +106,10 @@ struct GFluidParallelOutputRois
|
||||
|
||||
struct GFluidParallelFor
|
||||
{
|
||||
std::function<void(std::size_t, std::function<void(std::size_t)>)> parallel_for;
|
||||
//this function accepts:
|
||||
// - size of the "parallel" range as the first argument
|
||||
// - and a function to be called on the range items, designated by item index
|
||||
std::function<void(std::size_t size, std::function<void(std::size_t index)>)> parallel_for;
|
||||
};
|
||||
|
||||
namespace detail
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
//
|
||||
// Copyright (C) 2018 Intel Corporation
|
||||
// Copyright (C) 2018-2019 Intel Corporation
|
||||
|
||||
|
||||
#ifndef OPENCV_GAPI_RENDER_HPP
|
||||
@@ -11,6 +11,8 @@
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include <opencv2/gapi.hpp>
|
||||
|
||||
#include <opencv2/gapi/opencv_includes.hpp>
|
||||
#include <opencv2/gapi/util/variant.hpp>
|
||||
#include <opencv2/gapi/own/exports.hpp>
|
||||
@@ -20,6 +22,9 @@ namespace cv
|
||||
{
|
||||
namespace gapi
|
||||
{
|
||||
|
||||
namespace ocv { GAPI_EXPORTS cv::gapi::GKernelPackage kernels(); }
|
||||
|
||||
namespace wip
|
||||
{
|
||||
namespace draw
|
||||
@@ -78,7 +83,38 @@ struct Line
|
||||
int thick; //!< The thickness of line
|
||||
int lt; //!< The Type of the line. See #LineTypes
|
||||
int shift; //!< The number of fractional bits in the point coordinates
|
||||
};
|
||||
|
||||
/**
|
||||
* A structure to represent parameters for drawing a mosaic
|
||||
*/
|
||||
struct Mosaic
|
||||
{
|
||||
cv::Rect mos; //!< Coordinates of the mosaic
|
||||
int cellSz; //!< Cell size (same for X, Y). Note: mos size must be multiple of cell size
|
||||
int decim; //!< Decimation (0 stands for no decimation)
|
||||
};
|
||||
|
||||
/**
|
||||
* A structure to represent parameters for drawing an image
|
||||
*/
|
||||
struct Image
|
||||
{
|
||||
cv::Point org; //!< The bottom-left corner of the image
|
||||
cv::Mat img; //!< Image to draw
|
||||
cv::Mat alpha; //!< Alpha channel for image to draw (same size and number of channels)
|
||||
};
|
||||
|
||||
/**
|
||||
* A structure to represent parameters for drawing a polygon
|
||||
*/
|
||||
struct Poly
|
||||
{
|
||||
std::vector<cv::Point> points; //!< Points to connect
|
||||
cv::Scalar color; //!< The line color
|
||||
int thick; //!< The thickness of line
|
||||
int lt; //!< The Type of the line. See #LineTypes
|
||||
int shift; //!< The number of fractional bits in the point coordinates
|
||||
};
|
||||
|
||||
using Prim = util::variant
|
||||
@@ -86,27 +122,56 @@ using Prim = util::variant
|
||||
, Rect
|
||||
, Circle
|
||||
, Line
|
||||
, Mosaic
|
||||
, Image
|
||||
, Poly
|
||||
>;
|
||||
|
||||
using Prims = std::vector<Prim>;
|
||||
using Prims = std::vector<Prim>;
|
||||
using GMat2 = std::tuple<cv::GMat,cv::GMat>;
|
||||
using GMatDesc2 = std::tuple<cv::GMatDesc,cv::GMatDesc>;
|
||||
|
||||
G_TYPED_KERNEL_M(GRenderNV12, <GMat2(cv::GMat,cv::GMat,cv::GArray<wip::draw::Prim>)>, "org.opencv.render.nv12")
|
||||
{
|
||||
static GMatDesc2 outMeta(GMatDesc y_plane, GMatDesc uv_plane, GArrayDesc)
|
||||
{
|
||||
return std::make_tuple(y_plane, uv_plane);
|
||||
}
|
||||
};
|
||||
|
||||
G_TYPED_KERNEL(GRenderBGR, <cv::GMat(cv::GMat,cv::GArray<wip::draw::Prim>)>, "org.opencv.render.bgr")
|
||||
{
|
||||
static GMatDesc outMeta(GMatDesc bgr, GArrayDesc)
|
||||
{
|
||||
return bgr;
|
||||
}
|
||||
};
|
||||
|
||||
/** @brief The function renders on the input image passed drawing primitivies
|
||||
|
||||
@param bgr input image: 8-bit unsigned 3-channel image @ref CV_8UC3.
|
||||
@param prims vector of drawing primitivies
|
||||
@param pkg contains render kernel implementation
|
||||
*/
|
||||
GAPI_EXPORTS void render(cv::Mat& bgr, const Prims& prims);
|
||||
GAPI_EXPORTS void render(cv::Mat& bgr,
|
||||
const Prims& prims,
|
||||
const cv::gapi::GKernelPackage& pkg = ocv::kernels());
|
||||
|
||||
/** @brief The function renders on two NV12 planes passed drawing primitivies
|
||||
|
||||
@param y_plane input image: 8-bit unsigned 1-channel image @ref CV_8UC1.
|
||||
@param uv_plane input image: 8-bit unsigned 2-channel image @ref CV_8UC2.
|
||||
@param prims vector of drawing primitivies
|
||||
@param pkg contains render kernel implementation
|
||||
*/
|
||||
GAPI_EXPORTS void render(cv::Mat& y_plane, cv::Mat& uv_plane , const Prims& prims);
|
||||
GAPI_EXPORTS void render(cv::Mat& y_plane,
|
||||
cv::Mat& uv_plane,
|
||||
const Prims& prims,
|
||||
const cv::gapi::GKernelPackage& pkg = ocv::kernels());
|
||||
|
||||
} // namespace draw
|
||||
} // namespace wip
|
||||
|
||||
} // namespace gapi
|
||||
} // namespace cv
|
||||
|
||||
|
||||
@@ -1,91 +1,50 @@
|
||||
#include <opencv2/imgproc.hpp>
|
||||
|
||||
#include "opencv2/gapi/render.hpp"
|
||||
#include "opencv2/gapi/own/assert.hpp"
|
||||
#include <opencv2/gapi/render.hpp>
|
||||
#include <opencv2/gapi/own/assert.hpp>
|
||||
|
||||
#include "api/render_priv.hpp"
|
||||
|
||||
using namespace cv::gapi::wip::draw;
|
||||
// FXIME util::visitor ?
|
||||
void cv::gapi::wip::draw::render(cv::Mat& bgr, const Prims& prims)
|
||||
void cv::gapi::wip::draw::render(cv::Mat &bgr,
|
||||
const cv::gapi::wip::draw::Prims &prims,
|
||||
const cv::gapi::GKernelPackage& pkg)
|
||||
{
|
||||
for (const auto& p : prims)
|
||||
{
|
||||
switch (p.index())
|
||||
{
|
||||
case Prim::index_of<Rect>():
|
||||
{
|
||||
const auto& t_p = cv::util::get<Rect>(p);
|
||||
cv::rectangle(bgr, t_p.rect, t_p.color , t_p.thick, t_p.lt, t_p.shift);
|
||||
break;
|
||||
}
|
||||
cv::GMat in;
|
||||
cv::GArray<Prim> arr;
|
||||
|
||||
case Prim::index_of<Text>():
|
||||
{
|
||||
const auto& t_p = cv::util::get<Text>(p);
|
||||
cv::putText(bgr, t_p.text, t_p.org, t_p.ff, t_p.fs,
|
||||
t_p.color, t_p.thick, t_p.lt, t_p.bottom_left_origin);
|
||||
break;
|
||||
}
|
||||
|
||||
case Prim::index_of<Circle>():
|
||||
{
|
||||
const auto& c_p = cv::util::get<Circle>(p);
|
||||
cv::circle(bgr, c_p.center, c_p.radius, c_p.color, c_p.thick, c_p.lt, c_p.shift);
|
||||
break;
|
||||
}
|
||||
|
||||
case Prim::index_of<Line>():
|
||||
{
|
||||
const auto& l_p = cv::util::get<Line>(p);
|
||||
cv::line(bgr, l_p.pt1, l_p.pt2, l_p.color, l_p.thick, l_p.lt, l_p.shift);
|
||||
break;
|
||||
}
|
||||
|
||||
default: util::throw_error(std::logic_error("Unsupported draw operation"));
|
||||
}
|
||||
}
|
||||
cv::GComputation comp(cv::GIn(in, arr),
|
||||
cv::GOut(cv::gapi::wip::draw::GRenderBGR::on(in, arr)));
|
||||
comp.apply(cv::gin(bgr, prims), cv::gout(bgr), cv::compile_args(pkg));
|
||||
}
|
||||
|
||||
void cv::gapi::wip::draw::render(cv::Mat& y_plane, cv::Mat& uv_plane , const Prims& prims)
|
||||
void cv::gapi::wip::draw::render(cv::Mat &y_plane,
|
||||
cv::Mat &uv_plane,
|
||||
const Prims &prims,
|
||||
const GKernelPackage& pkg)
|
||||
{
|
||||
cv::Mat bgr;
|
||||
cv::cvtColorTwoPlane(y_plane, uv_plane, bgr, cv::COLOR_YUV2BGR_NV12);
|
||||
render(bgr, prims);
|
||||
BGR2NV12(bgr, y_plane, uv_plane);
|
||||
cv::GMat y_in, uv_in, y_out, uv_out;
|
||||
cv::GArray<Prim> arr;
|
||||
std::tie(y_out, uv_out) = cv::gapi::wip::draw::GRenderNV12::on(y_in, uv_in, arr);
|
||||
|
||||
cv::GComputation comp(cv::GIn(y_in, uv_in, arr), cv::GOut(y_out, uv_out));
|
||||
comp.apply(cv::gin(y_plane, uv_plane, prims),
|
||||
cv::gout(y_plane, uv_plane),
|
||||
cv::compile_args(pkg));
|
||||
}
|
||||
|
||||
void cv::gapi::wip::draw::splitNV12TwoPlane(const cv::Mat& yuv, cv::Mat& y_plane, cv::Mat& uv_plane) {
|
||||
y_plane.create(yuv.size(), CV_8UC1);
|
||||
uv_plane.create(yuv.size() / 2, CV_8UC2);
|
||||
|
||||
// Fill Y plane
|
||||
for (int i = 0; i < yuv.rows; ++i)
|
||||
{
|
||||
const uchar* in = yuv.ptr<uchar>(i);
|
||||
uchar* out = y_plane.ptr<uchar>(i);
|
||||
for (int j = 0; j < yuv.cols; j++) {
|
||||
out[j] = in[3 * j];
|
||||
}
|
||||
}
|
||||
|
||||
// Fill UV plane
|
||||
for (int i = 0; i < uv_plane.rows; i++)
|
||||
{
|
||||
const uchar* in = yuv.ptr<uchar>(2 * i);
|
||||
uchar* out = uv_plane.ptr<uchar>(i);
|
||||
for (int j = 0; j < uv_plane.cols; j++) {
|
||||
out[j * 2 ] = in[6 * j + 1];
|
||||
out[j * 2 + 1] = in[6 * j + 2];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void cv::gapi::wip::draw::BGR2NV12(const cv::Mat& bgr, cv::Mat& y_plane, cv::Mat& uv_plane)
|
||||
void cv::gapi::wip::draw::BGR2NV12(const cv::Mat &bgr,
|
||||
cv::Mat &y_plane,
|
||||
cv::Mat &uv_plane)
|
||||
{
|
||||
GAPI_Assert(bgr.size().width % 2 == 0);
|
||||
GAPI_Assert(bgr.size().height % 2 == 0);
|
||||
|
||||
cvtColor(bgr, bgr, cv::COLOR_BGR2YUV);
|
||||
splitNV12TwoPlane(bgr, y_plane, uv_plane);
|
||||
cv::Mat yuv;
|
||||
cvtColor(bgr, yuv, cv::COLOR_BGR2YUV);
|
||||
|
||||
std::vector<cv::Mat> chs(3);
|
||||
cv::split(yuv, chs);
|
||||
y_plane = chs[0];
|
||||
|
||||
cv::merge(std::vector<cv::Mat>{chs[1], chs[2]}, uv_plane);
|
||||
cv::resize(uv_plane, uv_plane, uv_plane.size() / 2, cv::INTER_LINEAR);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
#include <opencv2/gapi/cpu/gcpukernel.hpp>
|
||||
#include <opencv2/imgproc.hpp>
|
||||
#include <opencv2/gapi/render.hpp> // Kernel API's
|
||||
|
||||
#include "api/render_ocv.hpp"
|
||||
|
||||
namespace cv
|
||||
{
|
||||
namespace gapi
|
||||
{
|
||||
|
||||
namespace ocv
|
||||
{
|
||||
|
||||
GAPI_OCV_KERNEL(GOCVRenderNV12, cv::gapi::wip::draw::GRenderNV12)
|
||||
{
|
||||
static void run(const cv::Mat& y, const cv::Mat& uv, const cv::gapi::wip::draw::Prims& prims,
|
||||
cv::Mat& out_y, cv::Mat& out_uv)
|
||||
{
|
||||
/* FIXME How to render correctly on NV12 format ?
|
||||
*
|
||||
* Rendering on NV12 via OpenCV looks like this:
|
||||
*
|
||||
* y --------> 1)(NV12 -> YUV) -> yuv -> 2)draw -> yuv -> 3)split -------> out_y
|
||||
* ^ |
|
||||
* | |
|
||||
* uv -------------- `----------> out_uv
|
||||
*
|
||||
*
|
||||
* 1) Collect yuv mat from two planes, uv plain in two times less than y plane
|
||||
* so, upsample uv in tow times, with bilinear interpolation
|
||||
*
|
||||
* 2) Render primitives on YUV
|
||||
*
|
||||
* 3) Convert yuv to NV12 (using bilinear interpolation)
|
||||
*
|
||||
*/
|
||||
|
||||
// NV12 -> YUV
|
||||
cv::Mat upsample_uv, yuv;
|
||||
cv::resize(uv, upsample_uv, uv.size() * 2, cv::INTER_LINEAR);
|
||||
cv::merge(std::vector<cv::Mat>{y, upsample_uv}, yuv);
|
||||
|
||||
cv::gapi::wip::draw::drawPrimitivesOCVYUV(yuv, prims);
|
||||
|
||||
// YUV -> NV12
|
||||
cv::Mat out_u, out_v, uv_plane;
|
||||
std::vector<cv::Mat> chs = {out_y, out_u, out_v};
|
||||
cv::split(yuv, chs);
|
||||
cv::merge(std::vector<cv::Mat>{chs[1], chs[2]}, uv_plane);
|
||||
cv::resize(uv_plane, out_uv, uv_plane.size() / 2, cv::INTER_LINEAR);
|
||||
}
|
||||
};
|
||||
|
||||
GAPI_OCV_KERNEL(GOCVRenderBGR, cv::gapi::wip::draw::GRenderBGR)
|
||||
{
|
||||
static void run(const cv::Mat&, const cv::gapi::wip::draw::Prims& prims, cv::Mat& out)
|
||||
{
|
||||
cv::gapi::wip::draw::drawPrimitivesOCVBGR(out, prims);
|
||||
}
|
||||
};
|
||||
|
||||
cv::gapi::GKernelPackage kernels()
|
||||
{
|
||||
static const auto pkg = cv::gapi::kernels<GOCVRenderNV12, GOCVRenderBGR>();
|
||||
return pkg;
|
||||
}
|
||||
|
||||
} // namespace ocv
|
||||
|
||||
namespace wip
|
||||
{
|
||||
namespace draw
|
||||
{
|
||||
|
||||
void mosaic(cv::Mat mat, const cv::Rect &rect, int cellSz);
|
||||
void image(cv::Mat mat, cv::Point org, cv::Mat img, cv::Mat alpha);
|
||||
void poly(cv::Mat mat, std::vector<cv::Point>, cv::Scalar color, int lt, int shift);
|
||||
|
||||
void mosaic(cv::Mat mat, const cv::Rect &rect, int cellSz)
|
||||
{
|
||||
cv::Mat msc_roi = mat(rect);
|
||||
int crop_x = msc_roi.cols - msc_roi.cols % cellSz;
|
||||
int crop_y = msc_roi.rows - msc_roi.rows % cellSz;
|
||||
|
||||
for(int i = 0; i < crop_y; i += cellSz )
|
||||
for(int j = 0; j < crop_x; j += cellSz) {
|
||||
auto cell_roi = msc_roi(cv::Rect(j, i, cellSz, cellSz));
|
||||
cell_roi = cv::mean(cell_roi);
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
void image(cv::Mat mat, cv::Point org, cv::Mat img, cv::Mat alpha)
|
||||
{
|
||||
auto roi = mat(cv::Rect(org.x, org.y, img.size().width, img.size().height));
|
||||
cv::Mat img32f_w;
|
||||
cv::merge(std::vector<cv::Mat>(3, alpha), img32f_w);
|
||||
|
||||
cv::Mat roi32f_w(roi.size(), CV_32FC3, cv::Scalar::all(1.0));
|
||||
roi32f_w -= img32f_w;
|
||||
|
||||
cv::Mat img32f, roi32f;
|
||||
img.convertTo(img32f, CV_32F, 1.0/255);
|
||||
roi.convertTo(roi32f, CV_32F, 1.0/255);
|
||||
|
||||
cv::multiply(img32f, img32f_w, img32f);
|
||||
cv::multiply(roi32f, roi32f_w, roi32f);
|
||||
roi32f += img32f;
|
||||
|
||||
roi32f.convertTo(roi, CV_8U, 255.0);
|
||||
};
|
||||
|
||||
void poly(cv::Mat mat, std::vector<cv::Point> points, cv::Scalar color, int lt, int shift)
|
||||
{
|
||||
std::vector<std::vector<cv::Point>> pp{points};
|
||||
cv::fillPoly(mat, pp, color, lt, shift);
|
||||
};
|
||||
|
||||
struct BGR2YUVConverter
|
||||
{
|
||||
cv::Scalar cvtColor(const cv::Scalar& bgr) const
|
||||
{
|
||||
double y = bgr[2] * 0.299000 + bgr[1] * 0.587000 + bgr[0] * 0.114000;
|
||||
double u = bgr[2] * -0.168736 + bgr[1] * -0.331264 + bgr[0] * 0.500000 + 128;
|
||||
double v = bgr[2] * 0.500000 + bgr[1] * -0.418688 + bgr[0] * -0.081312 + 128;
|
||||
|
||||
return {y, u, v};
|
||||
}
|
||||
|
||||
void cvtImg(const cv::Mat& in, cv::Mat& out) { cv::cvtColor(in, out, cv::COLOR_BGR2YUV); };
|
||||
};
|
||||
|
||||
struct EmptyConverter
|
||||
{
|
||||
cv::Scalar cvtColor(const cv::Scalar& bgr) const { return bgr; };
|
||||
void cvtImg(const cv::Mat& in, cv::Mat& out) const { out = in; };
|
||||
};
|
||||
|
||||
// FIXME util::visitor ?
|
||||
template <typename ColorConverter>
|
||||
void drawPrimitivesOCV(cv::Mat &in, const Prims &prims)
|
||||
{
|
||||
ColorConverter converter;
|
||||
for (const auto &p : prims)
|
||||
{
|
||||
switch (p.index())
|
||||
{
|
||||
case Prim::index_of<Rect>():
|
||||
{
|
||||
const auto& t_p = cv::util::get<Rect>(p);
|
||||
const auto color = converter.cvtColor(t_p.color);
|
||||
cv::rectangle(in, t_p.rect, color , t_p.thick, t_p.lt, t_p.shift);
|
||||
break;
|
||||
}
|
||||
|
||||
case Prim::index_of<Text>():
|
||||
{
|
||||
const auto& t_p = cv::util::get<Text>(p);
|
||||
const auto color = converter.cvtColor(t_p.color);
|
||||
cv::putText(in, t_p.text, t_p.org, t_p.ff, t_p.fs,
|
||||
color, t_p.thick, t_p.lt, t_p.bottom_left_origin);
|
||||
break;
|
||||
}
|
||||
|
||||
case Prim::index_of<Circle>():
|
||||
{
|
||||
const auto& c_p = cv::util::get<Circle>(p);
|
||||
const auto color = converter.cvtColor(c_p.color);
|
||||
cv::circle(in, c_p.center, c_p.radius, color, c_p.thick, c_p.lt, c_p.shift);
|
||||
break;
|
||||
}
|
||||
|
||||
case Prim::index_of<Line>():
|
||||
{
|
||||
const auto& l_p = cv::util::get<Line>(p);
|
||||
const auto color = converter.cvtColor(l_p.color);
|
||||
cv::line(in, l_p.pt1, l_p.pt2, color, l_p.thick, l_p.lt, l_p.shift);
|
||||
break;
|
||||
}
|
||||
|
||||
case Prim::index_of<Mosaic>():
|
||||
{
|
||||
const auto& l_p = cv::util::get<Mosaic>(p);
|
||||
mosaic(in, l_p.mos, l_p.cellSz);
|
||||
break;
|
||||
}
|
||||
|
||||
case Prim::index_of<Image>():
|
||||
{
|
||||
const auto& i_p = cv::util::get<Image>(p);
|
||||
|
||||
cv::Mat img;
|
||||
converter.cvtImg(i_p.img, img);
|
||||
|
||||
image(in, i_p.org, img, i_p.alpha);
|
||||
break;
|
||||
}
|
||||
|
||||
case Prim::index_of<Poly>():
|
||||
{
|
||||
const auto& p_p = cv::util::get<Poly>(p);
|
||||
const auto color = converter.cvtColor(p_p.color);
|
||||
poly(in, p_p.points, color, p_p.lt, p_p.shift);
|
||||
break;
|
||||
}
|
||||
|
||||
default: cv::util::throw_error(std::logic_error("Unsupported draw operation"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void drawPrimitivesOCVBGR(cv::Mat &in, const Prims &prims)
|
||||
{
|
||||
drawPrimitivesOCV<EmptyConverter>(in, prims);
|
||||
}
|
||||
|
||||
void drawPrimitivesOCVYUV(cv::Mat &in, const Prims &prims)
|
||||
{
|
||||
drawPrimitivesOCV<BGR2YUVConverter>(in, prims);
|
||||
}
|
||||
|
||||
} // namespace draw
|
||||
} // namespace wip
|
||||
} // namespace gapi
|
||||
} // namespace cv
|
||||
@@ -0,0 +1,25 @@
|
||||
#include <vector>
|
||||
#include "render_priv.hpp"
|
||||
|
||||
#ifndef OPENCV_RENDER_OCV_HPP
|
||||
#define OPENCV_RENDER_OCV_HPP
|
||||
|
||||
namespace cv
|
||||
{
|
||||
namespace gapi
|
||||
{
|
||||
namespace wip
|
||||
{
|
||||
namespace draw
|
||||
{
|
||||
|
||||
// FIXME only for tests
|
||||
void GAPI_EXPORTS drawPrimitivesOCVYUV(cv::Mat &yuv, const Prims &prims);
|
||||
void GAPI_EXPORTS drawPrimitivesOCVBGR(cv::Mat &bgr, const Prims &prims);
|
||||
|
||||
} // namespace draw
|
||||
} // namespace wip
|
||||
} // namespace gapi
|
||||
} // namespace cv
|
||||
|
||||
#endif // OPENCV_RENDER_OCV_HPP
|
||||
@@ -18,9 +18,9 @@ namespace wip
|
||||
{
|
||||
namespace draw
|
||||
{
|
||||
|
||||
// FIXME only for tests
|
||||
GAPI_EXPORTS void BGR2NV12(const cv::Mat& bgr, cv::Mat& y_plane, cv::Mat& uv_plane);
|
||||
void splitNV12TwoPlane(const cv::Mat& yuv, cv::Mat& y_plane, cv::Mat& uv_plane);
|
||||
|
||||
} // namespace draw
|
||||
} // namespace wip
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
|
||||
#include <ade/util/algorithm.hpp>
|
||||
#include <ade/util/chain_range.hpp>
|
||||
#include <ade/util/iota_range.hpp>
|
||||
#include <ade/util/range.hpp>
|
||||
#include <ade/util/zip_range.hpp>
|
||||
|
||||
@@ -96,12 +97,31 @@ namespace
|
||||
const auto parallel_out_rois = cv::gimpl::getCompileArg<cv::GFluidParallelOutputRois>(args);
|
||||
const auto gpfor = cv::gimpl::getCompileArg<cv::GFluidParallelFor>(args);
|
||||
|
||||
auto serial_for = [](std::size_t count, std::function<void(std::size_t)> f){
|
||||
for (std::size_t i = 0; i < count; ++i){
|
||||
#if !defined(GAPI_STANDALONE)
|
||||
auto default_pfor = [](std::size_t count, std::function<void(std::size_t)> f){
|
||||
struct Body : cv::ParallelLoopBody {
|
||||
decltype(f) func;
|
||||
Body( decltype(f) && _f) : func(_f){}
|
||||
virtual void operator() (const cv::Range& r) const CV_OVERRIDE
|
||||
{
|
||||
for (std::size_t i : ade::util::iota(r.start, r.end))
|
||||
{
|
||||
func(i);
|
||||
}
|
||||
}
|
||||
};
|
||||
cv::parallel_for_(cv::Range{0,static_cast<int>(count)}, Body{std::move(f)});
|
||||
};
|
||||
#else
|
||||
auto default_pfor = [](std::size_t count, std::function<void(std::size_t)> f){
|
||||
for (auto i : ade::util::iota(count)){
|
||||
f(i);
|
||||
}
|
||||
};
|
||||
auto pfor = gpfor.has_value() ? gpfor.value().parallel_for : serial_for;
|
||||
#endif
|
||||
|
||||
auto pfor = gpfor.has_value() ? gpfor.value().parallel_for : default_pfor;
|
||||
|
||||
return parallel_out_rois.has_value() ?
|
||||
EPtr{new cv::gimpl::GParallelFluidExecutable (graph, graph_data, std::move(parallel_out_rois.value().parallel_rois), pfor)}
|
||||
: EPtr{new cv::gimpl::GFluidExecutable (graph, graph_data, std::move(rois.rois))}
|
||||
|
||||
@@ -29,6 +29,7 @@
|
||||
#include "compiler/gcompiler.hpp"
|
||||
#include "compiler/gcompiled_priv.hpp"
|
||||
#include "compiler/passes/passes.hpp"
|
||||
#include "compiler/passes/pattern_matching.hpp"
|
||||
|
||||
#include "executor/gexecutor.hpp"
|
||||
#include "backends/common/gbackend.hpp"
|
||||
@@ -103,6 +104,84 @@ namespace
|
||||
return result;
|
||||
}
|
||||
|
||||
// Creates ADE graph from input/output proto args
|
||||
std::unique_ptr<ade::Graph> makeGraph(const cv::GProtoArgs &ins, const cv::GProtoArgs &outs) {
|
||||
std::unique_ptr<ade::Graph> pG(new ade::Graph);
|
||||
ade::Graph& g = *pG;
|
||||
|
||||
cv::gimpl::GModel::Graph gm(g);
|
||||
cv::gimpl::GModel::init(gm);
|
||||
cv::gimpl::GModelBuilder builder(g);
|
||||
auto proto_slots = builder.put(ins, outs);
|
||||
|
||||
// Store Computation's protocol in metadata
|
||||
cv::gimpl::Protocol p;
|
||||
std::tie(p.inputs, p.outputs, p.in_nhs, p.out_nhs) = proto_slots;
|
||||
gm.metadata().set(p);
|
||||
|
||||
return pG;
|
||||
}
|
||||
|
||||
using adeGraphs = std::vector<std::unique_ptr<ade::Graph>>;
|
||||
|
||||
// Creates ADE graphs (patterns and substitutes) from pkg's transformations
|
||||
void makeTransformationGraphs(const cv::gapi::GKernelPackage& pkg,
|
||||
adeGraphs& patterns,
|
||||
adeGraphs& substitutes) {
|
||||
const auto& transforms = pkg.get_transformations();
|
||||
const auto size = transforms.size();
|
||||
if (0u == size) return;
|
||||
|
||||
// pre-generate all required graphs
|
||||
patterns.resize(size);
|
||||
substitutes.resize(size);
|
||||
for (auto it : ade::util::zip(ade::util::toRange(transforms),
|
||||
ade::util::toRange(patterns),
|
||||
ade::util::toRange(substitutes))) {
|
||||
const auto& t = std::get<0>(it);
|
||||
auto& p = std::get<1>(it);
|
||||
auto& s = std::get<2>(it);
|
||||
|
||||
auto pattern_comp = t.pattern();
|
||||
p = makeGraph(pattern_comp.priv().m_ins, pattern_comp.priv().m_outs);
|
||||
|
||||
auto substitute_comp = t.substitute();
|
||||
s = makeGraph(substitute_comp.priv().m_ins, substitute_comp.priv().m_outs);
|
||||
}
|
||||
}
|
||||
|
||||
void checkTransformations(const cv::gapi::GKernelPackage& pkg,
|
||||
const adeGraphs& patterns,
|
||||
const adeGraphs& substitutes) {
|
||||
const auto& transforms = pkg.get_transformations();
|
||||
const auto size = transforms.size();
|
||||
if (0u == size) return;
|
||||
|
||||
GAPI_Assert(size == patterns.size());
|
||||
GAPI_Assert(size == substitutes.size());
|
||||
|
||||
const auto empty = [] (const cv::gimpl::SubgraphMatch& m) {
|
||||
return m.inputDataNodes.empty() && m.startOpNodes.empty()
|
||||
&& m.finishOpNodes.empty() && m.outputDataNodes.empty()
|
||||
&& m.inputTestDataNodes.empty() && m.outputTestDataNodes.empty();
|
||||
};
|
||||
|
||||
// **FIXME**: verify other types of endless loops. now, only checking if pattern exists in
|
||||
// substitute within __the same__ transformation
|
||||
for (size_t i = 0; i < size; ++i) {
|
||||
const auto& p = patterns[i];
|
||||
const auto& s = substitutes[i];
|
||||
|
||||
auto matchInSubstitute = cv::gimpl::findMatches(*p, *s);
|
||||
if (!empty(matchInSubstitute)) {
|
||||
std::stringstream ss;
|
||||
ss << "Error: (in transformation with description: '"
|
||||
<< transforms[i].description
|
||||
<< "') pattern is detected inside substitute";
|
||||
throw std::runtime_error(ss.str());
|
||||
}
|
||||
}
|
||||
}
|
||||
} // anonymous namespace
|
||||
|
||||
|
||||
@@ -129,10 +208,26 @@ cv::gimpl::GCompiler::GCompiler(const cv::GComputation &c,
|
||||
// NN backends (present here via network package) always add their
|
||||
// inference kernels via auxiliary...()
|
||||
|
||||
// sanity check transformations
|
||||
{
|
||||
adeGraphs patterns, substitutes;
|
||||
// FIXME: returning vectors of unique_ptrs from makeTransformationGraphs results in
|
||||
// compile error (at least) on GCC 9 with usage of copy-ctor of std::unique_ptr, so
|
||||
// using initialization by lvalue reference instead
|
||||
makeTransformationGraphs(m_all_kernels, patterns, substitutes);
|
||||
checkTransformations(m_all_kernels, patterns, substitutes);
|
||||
|
||||
// NB: saving generated patterns to m_all_patterns to be used later in passes
|
||||
m_all_patterns = std::move(patterns);
|
||||
}
|
||||
|
||||
auto dump_path = getGraphDumpDirectory(m_args);
|
||||
|
||||
m_e.addPassStage("init");
|
||||
m_e.addPass("init", "check_cycles", ade::passes::CheckCycles());
|
||||
m_e.addPass("init", "apply_transformations",
|
||||
std::bind(passes::applyTransformations, _1, std::cref(m_all_kernels),
|
||||
std::cref(m_all_patterns))); // Note: and re-using patterns here
|
||||
m_e.addPass("init", "expand_kernels",
|
||||
std::bind(passes::expandKernels, _1,
|
||||
m_all_kernels)); // NB: package is copied
|
||||
@@ -255,23 +350,7 @@ cv::gimpl::GCompiler::GPtr cv::gimpl::GCompiler::generateGraph()
|
||||
{
|
||||
validateInputMeta();
|
||||
validateOutProtoArgs();
|
||||
|
||||
// Generate ADE graph from expression-based computation
|
||||
std::unique_ptr<ade::Graph> pG(new ade::Graph);
|
||||
ade::Graph& g = *pG;
|
||||
|
||||
GModel::Graph gm(g);
|
||||
cv::gimpl::GModel::init(gm);
|
||||
cv::gimpl::GModelBuilder builder(g);
|
||||
auto proto_slots = builder.put(m_c.priv().m_ins, m_c.priv().m_outs);
|
||||
GAPI_LOG_INFO(NULL, "Generated graph: " << g.nodes().size() << " nodes" << std::endl);
|
||||
|
||||
// Store Computation's protocol in metadata
|
||||
Protocol p;
|
||||
std::tie(p.inputs, p.outputs, p.in_nhs, p.out_nhs) = proto_slots;
|
||||
gm.metadata().set(p);
|
||||
|
||||
return pG;
|
||||
return makeGraph(m_c.priv().m_ins, m_c.priv().m_outs);
|
||||
}
|
||||
|
||||
void cv::gimpl::GCompiler::runPasses(ade::Graph &g)
|
||||
|
||||
@@ -29,6 +29,8 @@ class GAPI_EXPORTS GCompiler
|
||||
cv::gapi::GKernelPackage m_all_kernels;
|
||||
cv::gapi::GNetPackage m_all_networks;
|
||||
|
||||
std::vector<std::unique_ptr<ade::Graph>> m_all_patterns; // built patterns from transformations
|
||||
|
||||
void validateInputMeta();
|
||||
void validateOutProtoArgs();
|
||||
|
||||
|
||||
@@ -59,6 +59,10 @@ void resolveKernels(ade::passes::PassContext &ctx,
|
||||
void fuseIslands(ade::passes::PassContext &ctx);
|
||||
void syncIslandTags(ade::passes::PassContext &ctx);
|
||||
|
||||
void applyTransformations(ade::passes::PassContext &ctx,
|
||||
const gapi::GKernelPackage &pkg,
|
||||
const std::vector<std::unique_ptr<ade::Graph>> &preGeneratedPatterns);
|
||||
|
||||
}} // namespace gimpl::passes
|
||||
|
||||
} // namespace cv
|
||||
|
||||
@@ -54,8 +54,7 @@ bool compareDataNodes(const ade::NodeHandle& first, const std::vector<std::size_
|
||||
"shall be NodeType::DATA!");
|
||||
}
|
||||
|
||||
if (firstMeta.get<cv::gimpl::Data>().shape !=
|
||||
secondMeta.get<cv::gimpl::Data>().shape) {
|
||||
if (firstMeta.get<cv::gimpl::Data>().shape != secondMeta.get<cv::gimpl::Data>().shape) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -180,7 +179,7 @@ inline bool IS_ENDPOINT(const ade::NodeHandle& nh){
|
||||
// Try to rely on the nh Data::Storage::OUTPUT
|
||||
return nh->outEdges().empty();
|
||||
}
|
||||
}
|
||||
} // anonymous namespace
|
||||
|
||||
// Routine relies on the logic that 1 DATA node may have only 1 input edge.
|
||||
cv::gimpl::SubgraphMatch
|
||||
@@ -509,7 +508,7 @@ cv::gimpl::findMatches(const cv::gimpl::GModel::Graph& patternGraph,
|
||||
// Create vector with the correctly ordered IN data nodes in the test subgraph
|
||||
std::vector<ade::NodeHandle> inputTestDataNodes;
|
||||
for (const auto& patternInNode : patternInputDataNodes) {
|
||||
inputTestDataNodes.push_back(inputApiMatch[patternInNode]);
|
||||
inputTestDataNodes.push_back(inputApiMatch.at(patternInNode));
|
||||
}
|
||||
|
||||
// Traversing current result for ending OPs
|
||||
@@ -560,7 +559,7 @@ cv::gimpl::findMatches(const cv::gimpl::GModel::Graph& patternGraph,
|
||||
// Create vector with the correctly ordered OUT data nodes in the test subgraph
|
||||
std::vector<ade::NodeHandle> outputTestDataNodes;
|
||||
for (const auto& patternOutNode : patternOutputDataNodes) {
|
||||
outputTestDataNodes.push_back(outputApiMatch[patternOutNode]);
|
||||
outputTestDataNodes.push_back(outputApiMatch.at(patternOutNode));
|
||||
}
|
||||
|
||||
SubgraphMatch subgraph;
|
||||
|
||||
@@ -41,7 +41,6 @@ namespace gimpl {
|
||||
return !inputDataNodes.empty() && !startOpNodes.empty()
|
||||
&& !finishOpNodes.empty() && !outputDataNodes.empty()
|
||||
&& !inputTestDataNodes.empty() && !outputTestDataNodes.empty();
|
||||
|
||||
}
|
||||
|
||||
S nodes() const {
|
||||
@@ -93,6 +92,11 @@ namespace gimpl {
|
||||
GAPI_EXPORTS SubgraphMatch findMatches(const cv::gimpl::GModel::Graph& patternGraph,
|
||||
const cv::gimpl::GModel::Graph& compGraph);
|
||||
|
||||
GAPI_EXPORTS void performSubstitution(cv::gimpl::GModel::Graph& graph,
|
||||
const cv::gimpl::Protocol& patternP,
|
||||
const cv::gimpl::Protocol& substituteP,
|
||||
const cv::gimpl::SubgraphMatch& patternToGraphMatch);
|
||||
|
||||
} //namespace gimpl
|
||||
} //namespace cv
|
||||
#endif // OPENCV_GAPI_PATTERN_MATCHING_HPP
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
//
|
||||
// Copyright (C) 2019 Intel Corporation
|
||||
|
||||
#include "pattern_matching.hpp"
|
||||
|
||||
#include "ade/util/zip_range.hpp"
|
||||
|
||||
namespace cv { namespace gimpl {
|
||||
namespace {
|
||||
using Graph = GModel::Graph;
|
||||
|
||||
template<typename Iterator>
|
||||
ade::NodeHandle getNh(Iterator it) { return *it; }
|
||||
|
||||
template<>
|
||||
ade::NodeHandle getNh(SubgraphMatch::M::const_iterator it) { return it->second; }
|
||||
|
||||
template<typename Container>
|
||||
void erase(Graph& g, const Container& c)
|
||||
{
|
||||
for (auto first = c.begin(); first != c.end(); ++first) {
|
||||
ade::NodeHandle node = getNh(first);
|
||||
if (node == nullptr) continue; // some nodes might already be erased
|
||||
g.erase(node);
|
||||
}
|
||||
}
|
||||
} // anonymous namespace
|
||||
|
||||
void performSubstitution(GModel::Graph& graph,
|
||||
const Protocol& patternP,
|
||||
const Protocol& substituteP,
|
||||
const SubgraphMatch& patternToGraphMatch)
|
||||
{
|
||||
// 1. substitute input nodes
|
||||
const auto& patternIns = patternP.in_nhs;
|
||||
const auto& substituteIns = substituteP.in_nhs;
|
||||
|
||||
for (auto it : ade::util::zip(ade::util::toRange(patternIns),
|
||||
ade::util::toRange(substituteIns))) {
|
||||
// Note: we don't replace input DATA nodes here, only redirect their output edges
|
||||
const auto& patternDataNode = std::get<0>(it);
|
||||
const auto& substituteDataNode = std::get<1>(it);
|
||||
const auto& graphDataNode = patternToGraphMatch.inputDataNodes.at(patternDataNode);
|
||||
GModel::redirectReaders(graph, substituteDataNode, graphDataNode);
|
||||
}
|
||||
|
||||
// 2. substitute output nodes
|
||||
const auto& patternOuts = patternP.out_nhs;
|
||||
const auto& substituteOuts = substituteP.out_nhs;
|
||||
|
||||
for (auto it : ade::util::zip(ade::util::toRange(patternOuts),
|
||||
ade::util::toRange(substituteOuts))) {
|
||||
// Note: we don't replace output DATA nodes here, only redirect their input edges
|
||||
const auto& patternDataNode = std::get<0>(it);
|
||||
const auto& substituteDataNode = std::get<1>(it);
|
||||
const auto& graphDataNode = patternToGraphMatch.outputDataNodes.at(patternDataNode);
|
||||
|
||||
// delete existing edges (otherwise we cannot redirect)
|
||||
auto existingEdges = graphDataNode->inEdges();
|
||||
// NB: we cannot iterate over node->inEdges() here directly because it gets modified when
|
||||
// edges are erased. Erasing an edge supposes that src/dst nodes will remove
|
||||
// (correspondingly) out/in edge (which is _our edge_). Now, this deleting means
|
||||
// node->inEdges() will also get updated in the process: so, we'd iterate over a
|
||||
// container which changes in this case. Using supplementary std::vector instead:
|
||||
std::vector<ade::EdgeHandle> handles(existingEdges.begin(), existingEdges.end());
|
||||
for (const auto& e : handles) {
|
||||
graph.erase(e);
|
||||
}
|
||||
|
||||
GModel::redirectWriter(graph, substituteDataNode, graphDataNode);
|
||||
}
|
||||
|
||||
// 3. erase redundant nodes:
|
||||
// erase input data nodes of __substitute__
|
||||
erase(graph, substituteIns);
|
||||
|
||||
// erase old start OP nodes of __main graph__
|
||||
erase(graph, patternToGraphMatch.startOpNodes);
|
||||
|
||||
// erase old internal nodes of __main graph__
|
||||
erase(graph, patternToGraphMatch.internalLayers);
|
||||
|
||||
// erase old finish OP nodes of __main graph__
|
||||
erase(graph, patternToGraphMatch.finishOpNodes);
|
||||
|
||||
// erase output data nodes of __substitute__
|
||||
erase(graph, substituteOuts);
|
||||
}
|
||||
|
||||
} // namespace gimpl
|
||||
} // namespace cv
|
||||
@@ -0,0 +1,139 @@
|
||||
// This file is part of OpenCV project.
|
||||
// It is subject to the license terms in the LICENSE file found in the top-level directory
|
||||
// of this distribution and at http://opencv.org/license.html.
|
||||
//
|
||||
// Copyright (C) 2019 Intel Corporation
|
||||
|
||||
#include "precomp.hpp"
|
||||
|
||||
#include <ade/util/zip_range.hpp>
|
||||
#include <ade/graph.hpp>
|
||||
|
||||
#include "api/gcomputation_priv.hpp"
|
||||
|
||||
#include "compiler/gmodel.hpp"
|
||||
#include "compiler/gmodelbuilder.hpp"
|
||||
#include "compiler/passes/passes.hpp"
|
||||
#include "compiler/passes/pattern_matching.hpp"
|
||||
|
||||
#include <sstream>
|
||||
|
||||
namespace cv { namespace gimpl { namespace passes {
|
||||
namespace
|
||||
{
|
||||
using Graph = GModel::Graph;
|
||||
using Metadata = typename Graph::CMetadataT;
|
||||
|
||||
// Checks pairs of {pattern node, substitute node} and asserts if there are any incompatibilities
|
||||
void checkDataNodes(const Graph& pattern,
|
||||
const Graph& substitute,
|
||||
const std::vector<ade::NodeHandle>& patternNodes,
|
||||
const std::vector<ade::NodeHandle>& substituteNodes)
|
||||
{
|
||||
for (auto it : ade::util::zip(patternNodes, substituteNodes)) {
|
||||
auto pNodeMeta = pattern.metadata(std::get<0>(it));
|
||||
auto sNodeMeta = substitute.metadata(std::get<1>(it));
|
||||
GAPI_Assert(pNodeMeta.get<NodeType>().t == NodeType::DATA);
|
||||
GAPI_Assert(pNodeMeta.get<NodeType>().t == sNodeMeta.get<NodeType>().t);
|
||||
GAPI_Assert(pNodeMeta.get<Data>().shape == sNodeMeta.get<Data>().shape);
|
||||
}
|
||||
}
|
||||
|
||||
// Checks compatibility of pattern and substitute graphs based on in/out nodes
|
||||
void checkCompatibility(const Graph& pattern,
|
||||
const Graph& substitute,
|
||||
const Protocol& patternP,
|
||||
const Protocol& substituteP)
|
||||
{
|
||||
const auto& patternDataInputs = patternP.in_nhs;
|
||||
const auto& patternDataOutputs = patternP.out_nhs;
|
||||
|
||||
const auto& substituteDataInputs = substituteP.in_nhs;
|
||||
const auto& substituteDataOutputs = substituteP.out_nhs;
|
||||
|
||||
// number of data nodes must be the same
|
||||
GAPI_Assert(patternDataInputs.size() == substituteDataInputs.size());
|
||||
GAPI_Assert(patternDataOutputs.size() == substituteDataOutputs.size());
|
||||
|
||||
// for each pattern input node, verify a corresponding substitute input node
|
||||
checkDataNodes(pattern, substitute, patternDataInputs, substituteDataInputs);
|
||||
|
||||
// for each pattern output node, verify a corresponding substitute output node
|
||||
checkDataNodes(pattern, substitute, patternDataOutputs, substituteDataOutputs);
|
||||
}
|
||||
|
||||
// Tries to substitute __single__ pattern with substitute in the given graph
|
||||
bool tryToSubstitute(ade::Graph& main,
|
||||
const std::unique_ptr<ade::Graph>& patternG,
|
||||
const cv::GComputation& substitute)
|
||||
{
|
||||
GModel::Graph gm(main);
|
||||
|
||||
// 1. find a pattern in main graph
|
||||
auto match1 = findMatches(*patternG, gm);
|
||||
if (!match1.ok()) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// 2. build substitute graph inside the main graph
|
||||
cv::gimpl::GModelBuilder builder(main);
|
||||
const auto& proto_slots = builder.put(substitute.priv().m_ins, substitute.priv().m_outs);
|
||||
Protocol substituteP;
|
||||
std::tie(substituteP.inputs, substituteP.outputs, substituteP.in_nhs, substituteP.out_nhs) =
|
||||
proto_slots;
|
||||
|
||||
const Protocol& patternP = GModel::Graph(*patternG).metadata().get<Protocol>();
|
||||
|
||||
// 3. check that pattern and substitute are compatible
|
||||
// FIXME: in theory, we should always have compatible pattern/substitute. if not, we're in
|
||||
// half-completed state where some transformations are already applied - what can we do
|
||||
// to handle the situation better? -- use transactional API as in fuse_islands pass?
|
||||
checkCompatibility(*patternG, gm, patternP, substituteP);
|
||||
|
||||
// 4. make substitution
|
||||
performSubstitution(gm, patternP, substituteP, match1);
|
||||
|
||||
return true;
|
||||
}
|
||||
} // anonymous namespace
|
||||
|
||||
void applyTransformations(ade::passes::PassContext& ctx,
|
||||
const gapi::GKernelPackage& pkg,
|
||||
const std::vector<std::unique_ptr<ade::Graph>>& patterns)
|
||||
{
|
||||
const auto& transforms = pkg.get_transformations();
|
||||
const auto size = transforms.size();
|
||||
if (0u == size) return;
|
||||
// Note: patterns are already generated at this point
|
||||
GAPI_Assert(patterns.size() == transforms.size());
|
||||
|
||||
// transform as long as it is possible
|
||||
bool canTransform = true;
|
||||
while (canTransform)
|
||||
{
|
||||
canTransform = false;
|
||||
|
||||
// iterate through every transformation and try to transform graph parts
|
||||
for (auto it : ade::util::zip(ade::util::toRange(transforms), ade::util::toRange(patterns)))
|
||||
{
|
||||
const auto& t = std::get<0>(it);
|
||||
auto& pattern = std::get<1>(it); // Note: using pre-created graphs
|
||||
GAPI_Assert(nullptr != pattern);
|
||||
|
||||
// if transformation is successful (pattern found and substituted), it is possible that
|
||||
// other transformations will also be successful, so set canTransform to the returned
|
||||
// value from tryToSubstitute
|
||||
canTransform = tryToSubstitute(ctx.graph, pattern, t.substitute());
|
||||
|
||||
// Note: apply the __same__ substitution as many times as possible and only after go to
|
||||
// the next one. BUT it can happen that after applying some substitution, some
|
||||
// _previous_ patterns will also be found and these will be applied first
|
||||
if (canTransform) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace passes
|
||||
} // namespace gimpl
|
||||
} // namespace cv
|
||||
@@ -100,7 +100,7 @@ GAPI_TEST_FIXTURE(AddWeightedTest, initMatsRandU, FIXTURE_API(CompareMats), 1, c
|
||||
GAPI_TEST_FIXTURE(NormTest, initMatrixRandU, FIXTURE_API(CompareScalars,NormTypes), 2,
|
||||
cmpF, opType)
|
||||
GAPI_TEST_FIXTURE(IntegralTest, initNothing, <>, 0)
|
||||
GAPI_TEST_FIXTURE(ThresholdTest, initMatrixRandU, FIXTURE_API(int), 1, tt)
|
||||
GAPI_TEST_FIXTURE(ThresholdTest, initMatrixRandU, FIXTURE_API(int, cv::Scalar), 2, tt, maxval)
|
||||
GAPI_TEST_FIXTURE(ThresholdOTTest, initMatrixRandU, FIXTURE_API(int), 1, tt)
|
||||
GAPI_TEST_FIXTURE(InRangeTest, initMatrixRandU, <>, 0)
|
||||
GAPI_TEST_FIXTURE(Split3Test, initMatrixRandU, <>, 0)
|
||||
|
||||
@@ -682,7 +682,6 @@ TEST_P(IntegralTest, AccuracyTest)
|
||||
TEST_P(ThresholdTest, AccuracyTestBinary)
|
||||
{
|
||||
cv::Scalar thr = initScalarRandU(50);
|
||||
cv::Scalar maxval = initScalarRandU(50) + cv::Scalar(50, 50, 50, 50);
|
||||
cv::Scalar out_scalar;
|
||||
|
||||
// G-API code //////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -29,8 +29,8 @@ namespace opencv_test
|
||||
// - created (and initialized) automatically
|
||||
// - available in test body
|
||||
// Note: all parameter _values_ (e.g. type CV_8UC3) are set via INSTANTIATE_TEST_CASE_P macro
|
||||
GAPI_TEST_FIXTURE(Filter2DTest, initMatrixRandN, FIXTURE_API(CompareMats,int,int), 3,
|
||||
cmpF, kernSize, borderType)
|
||||
GAPI_TEST_FIXTURE(Filter2DTest, initMatrixRandN, FIXTURE_API(CompareMats,cv::Size,int), 3,
|
||||
cmpF, filterSize, borderType)
|
||||
GAPI_TEST_FIXTURE(BoxFilterTest, initMatrixRandN, FIXTURE_API(CompareMats,int,int), 3,
|
||||
cmpF, filterSize, borderType)
|
||||
GAPI_TEST_FIXTURE(SepFilterTest, initMatrixRandN, FIXTURE_API(CompareMats,int), 2, cmpF, kernSize)
|
||||
|
||||
@@ -57,9 +57,23 @@ TEST_P(Filter2DTest, AccuracyTest)
|
||||
cv::Point anchor = {-1, -1};
|
||||
double delta = 0;
|
||||
|
||||
cv::Mat kernel = cv::Mat(kernSize, kernSize, CV_32FC1);
|
||||
cv::Scalar kernMean = cv::Scalar(1.0);
|
||||
cv::Scalar kernStddev = cv::Scalar(2.0/3);
|
||||
cv::Mat kernel = cv::Mat(filterSize, CV_32FC1);
|
||||
cv::Scalar kernMean, kernStddev;
|
||||
|
||||
const auto kernSize = filterSize.width * filterSize.height;
|
||||
const auto bigKernSize = 49;
|
||||
|
||||
if (kernSize < bigKernSize)
|
||||
{
|
||||
kernMean = cv::Scalar(0.3);
|
||||
kernStddev = cv::Scalar(0.5);
|
||||
}
|
||||
else
|
||||
{
|
||||
kernMean = cv::Scalar(0.008);
|
||||
kernStddev = cv::Scalar(0.008);
|
||||
}
|
||||
|
||||
randn(kernel, kernMean, kernStddev);
|
||||
|
||||
// G-API code //////////////////////////////////////////////////////////////
|
||||
@@ -67,6 +81,7 @@ TEST_P(Filter2DTest, AccuracyTest)
|
||||
auto out = cv::gapi::filter2D(in, dtype, kernel, anchor, delta, borderType);
|
||||
|
||||
cv::GComputation c(in, out);
|
||||
|
||||
c.apply(in_mat1, out_mat_gapi, getCompileArgs());
|
||||
// OpenCV code /////////////////////////////////////////////////////////////
|
||||
{
|
||||
|
||||
@@ -10,14 +10,53 @@
|
||||
|
||||
#include "gapi_tests_common.hpp"
|
||||
#include "api/render_priv.hpp"
|
||||
#include "api/render_ocv.hpp"
|
||||
|
||||
#define rect1 Prim{cv::gapi::wip::draw::Rect{cv::Rect{101, 101, 199, 199}, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define rect2 Prim{cv::gapi::wip::draw::Rect{cv::Rect{100, 100, 199, 199}, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define rect3 Prim{cv::gapi::wip::draw::Rect{cv::Rect{0 , 0 , 199, 199}, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define rect4 Prim{cv::gapi::wip::draw::Rect{cv::Rect{100, 100, 0, 199 }, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define rect5 Prim{cv::gapi::wip::draw::Rect{cv::Rect{0 , -1 , 199, 199}, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define rect6 Prim{cv::gapi::wip::draw::Rect{cv::Rect{100, 100, 199, 199}, cv::Scalar{153, 172, 58}, 10, LINE_8, 0}}
|
||||
#define rect7 Prim{cv::gapi::wip::draw::Rect{cv::Rect{100, 100, 200, 200}, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define box1 Prim{cv::gapi::wip::draw::Rect{cv::Rect{101, 101, 200, 200}, cv::Scalar{153, 172, 58}, -1, LINE_8, 0}}
|
||||
#define box2 Prim{cv::gapi::wip::draw::Rect{cv::Rect{100, 100, 199, 199}, cv::Scalar{153, 172, 58}, -1, LINE_8, 0}}
|
||||
#define rects Prims{rect1, rect2, rect3, rect4, rect5, rect6, rect7, box1, box2}
|
||||
|
||||
#define circle1 Prim{cv::gapi::wip::draw::Circle{cv::Point{200, 200}, 100, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define circle2 Prim{cv::gapi::wip::draw::Circle{cv::Point{10, 30} , 2 , cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define circle3 Prim{cv::gapi::wip::draw::Circle{cv::Point{75, 100} , 50 , cv::Scalar{153, 172, 58}, 5, LINE_8, 0}}
|
||||
#define circles Prims{circle1, circle2, circle3}
|
||||
|
||||
#define line1 Prim{cv::gapi::wip::draw::Line{cv::Point{50, 50}, cv::Point{250, 200}, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define line2 Prim{cv::gapi::wip::draw::Line{cv::Point{51, 51}, cv::Point{51, 100}, cv::Scalar{153, 172, 58}, 1, LINE_8, 0}}
|
||||
#define lines Prims{line1, line2}
|
||||
|
||||
#define mosaic1 Prim{cv::gapi::wip::draw::Mosaic{cv::Rect{100, 100, 200, 200}, 5, 0}}
|
||||
#define mosaics Prims{mosaic1}
|
||||
|
||||
#define image1 Prim{cv::gapi::wip::draw::Image{cv::Point(100, 100), cv::Mat(cv::Size(200, 200), CV_8UC3, cv::Scalar::all(255)),\
|
||||
cv::Mat(cv::Size(200, 200), CV_32FC1, cv::Scalar::all(1))}}
|
||||
|
||||
#define image2 Prim{cv::gapi::wip::draw::Image{cv::Point(100, 100), cv::Mat(cv::Size(200, 200), CV_8UC3, cv::Scalar::all(255)),\
|
||||
cv::Mat(cv::Size(200, 200), CV_32FC1, cv::Scalar::all(0.5))}}
|
||||
|
||||
#define image3 Prim{cv::gapi::wip::draw::Image{cv::Point(100, 100), cv::Mat(cv::Size(200, 200), CV_8UC3, cv::Scalar::all(255)),\
|
||||
cv::Mat(cv::Size(200, 200), CV_32FC1, cv::Scalar::all(0.0))}}
|
||||
|
||||
#define images Prims{image1, image2, image3}
|
||||
|
||||
#define polygon1 Prim{cv::gapi::wip::draw::Poly{ {cv::Point{100, 100}, cv::Point{50, 200}, cv::Point{200, 30}, cv::Point{150, 50} }, cv::Scalar{153, 172, 58}, 1, LINE_8, 0} }
|
||||
#define polygons Prims{polygon1}
|
||||
|
||||
#define text1 Prim{cv::gapi::wip::draw::Text{"TheBrownFoxJump", cv::Point{100, 100}, FONT_HERSHEY_SIMPLEX, 2, cv::Scalar{102, 178, 240}, 1, LINE_8, false} }
|
||||
#define texts Prims{text1}
|
||||
|
||||
namespace opencv_test
|
||||
{
|
||||
|
||||
using Points = std::vector<cv::Point>;
|
||||
using Rects = std::vector<cv::Rect>;
|
||||
using PairOfPoints = std::pair<cv::Point, cv::Point>;
|
||||
using VecOfPairOfPoints = std::vector<PairOfPoints>;
|
||||
using Prims = cv::gapi::wip::draw::Prims;
|
||||
using Prim = cv::gapi::wip::draw::Prim;
|
||||
|
||||
template<class T>
|
||||
class RenderWithParam : public TestWithParam<T>
|
||||
@@ -26,47 +65,50 @@ protected:
|
||||
void Init()
|
||||
{
|
||||
MatType type = CV_8UC3;
|
||||
out_mat_ocv = cv::Mat(sz, type, cv::Scalar(255));
|
||||
out_mat_gapi = cv::Mat(sz, type, cv::Scalar(255));
|
||||
|
||||
if (isNV12Format) {
|
||||
/* NB: When converting data from BGR to NV12, data loss occurs,
|
||||
* so the reference data is subjected to the same transformation
|
||||
* for correct comparison of the test results */
|
||||
cv::gapi::wip::draw::BGR2NV12(out_mat_ocv, y, uv);
|
||||
cv::cvtColorTwoPlane(y, uv, out_mat_ocv, cv::COLOR_YUV2BGR_NV12);
|
||||
}
|
||||
}
|
||||
|
||||
void Run()
|
||||
{
|
||||
if (isNV12Format) {
|
||||
cv::gapi::wip::draw::BGR2NV12(out_mat_gapi, y, uv);
|
||||
cv::gapi::wip::draw::render(y, uv, prims);
|
||||
cv::cvtColorTwoPlane(y, uv, out_mat_gapi, cv::COLOR_YUV2BGR_NV12);
|
||||
|
||||
// NB: Also due to data loss
|
||||
cv::gapi::wip::draw::BGR2NV12(out_mat_ocv, y, uv);
|
||||
cv::cvtColorTwoPlane(y, uv, out_mat_ocv, cv::COLOR_YUV2BGR_NV12);
|
||||
} else {
|
||||
cv::gapi::wip::draw::render(out_mat_gapi, prims);
|
||||
}
|
||||
mat_ocv.create(sz, type);
|
||||
mat_gapi.create(sz, type);
|
||||
cv::randu(mat_ocv, cv::Scalar::all(0), cv::Scalar::all(255));
|
||||
mat_ocv.copyTo(mat_gapi);
|
||||
}
|
||||
|
||||
cv::Size sz;
|
||||
cv::Scalar color;
|
||||
int thick;
|
||||
int lt;
|
||||
bool isNV12Format;
|
||||
std::vector<cv::gapi::wip::draw::Prim> prims;
|
||||
cv::Mat y, uv;
|
||||
cv::Mat out_mat_ocv, out_mat_gapi;
|
||||
cv::gapi::GKernelPackage pkg;
|
||||
|
||||
cv::Mat y_mat_ocv, uv_mat_ocv, y_mat_gapi, uv_mat_gapi, mat_ocv, mat_gapi;
|
||||
};
|
||||
|
||||
struct RenderTextTest : public RenderWithParam <std::tuple<cv::Size,std::string,Points,int,double,cv::Scalar,int,int,bool,bool>> {};
|
||||
struct RenderRectTest : public RenderWithParam <std::tuple<cv::Size,Rects,cv::Scalar,int,int,int,bool>> {};
|
||||
struct RenderCircleTest : public RenderWithParam <std::tuple<cv::Size,Points,int,cv::Scalar,int,int,int,bool>> {};
|
||||
struct RenderLineTest : public RenderWithParam <std::tuple<cv::Size,VecOfPairOfPoints,cv::Scalar,int,int,int,bool>> {};
|
||||
using TestArgs = std::tuple<cv::Size,cv::gapi::wip::draw::Prims,cv::gapi::GKernelPackage>;
|
||||
|
||||
struct RenderNV12 : public RenderWithParam<TestArgs>
|
||||
{
|
||||
void ComputeRef()
|
||||
{
|
||||
cv::gapi::wip::draw::BGR2NV12(mat_ocv, y_mat_ocv, uv_mat_ocv);
|
||||
|
||||
// NV12 -> YUV
|
||||
cv::Mat upsample_uv, yuv;
|
||||
cv::resize(uv_mat_ocv, upsample_uv, uv_mat_ocv.size() * 2, cv::INTER_LINEAR);
|
||||
cv::merge(std::vector<cv::Mat>{y_mat_ocv, upsample_uv}, yuv);
|
||||
|
||||
cv::gapi::wip::draw::drawPrimitivesOCVYUV(yuv, prims);
|
||||
|
||||
// YUV -> NV12
|
||||
std::vector<cv::Mat> chs(3);
|
||||
cv::split(yuv, chs);
|
||||
cv::merge(std::vector<cv::Mat>{chs[1], chs[2]}, uv_mat_ocv);
|
||||
y_mat_ocv = chs[0];
|
||||
cv::resize(uv_mat_ocv, uv_mat_ocv, uv_mat_ocv.size() / 2, cv::INTER_LINEAR);
|
||||
}
|
||||
};
|
||||
|
||||
struct RenderBGR : public RenderWithParam<TestArgs>
|
||||
{
|
||||
void ComputeRef()
|
||||
{
|
||||
cv::gapi::wip::draw::drawPrimitivesOCVBGR(mat_ocv, prims);
|
||||
}
|
||||
};
|
||||
|
||||
} // opencv_test
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user