mirror of
https://github.com/opencv/opencv.git
synced 2026-07-30 15:53:03 +04:00
Merge pull request #25881 from fengyuentau:dnn/cpu/optimize_activations_with_v_exp
dnn: optimize activations with v_exp #25881 Merge with https://github.com/opencv/opencv_extra/pull/1191. This PR optimizes the following activations: - [x] Swish - [x] Mish - [x] Elu - [x] Celu - [x] Selu - [x] HardSwish ### Performance (Updated on 2024-07-18) #### AmLogic A311D2 (ARM Cortex A73 + A53) ``` Geometric mean (ms) Name of Test activations activations.patch activations.patch vs activations (x-factor) Celu::Layer_Elementwise::OCV/CPU 115.859 27.930 4.15 Elu::Layer_Elementwise::OCV/CPU 27.846 27.003 1.03 Gelu::Layer_Elementwise::OCV/CPU 0.657 0.602 1.09 HardSwish::Layer_Elementwise::OCV/CPU 31.885 6.781 4.70 Mish::Layer_Elementwise::OCV/CPU 35.729 32.089 1.11 Selu::Layer_Elementwise::OCV/CPU 61.955 27.850 2.22 Swish::Layer_Elementwise::OCV/CPU 30.819 26.688 1.15 ``` #### Apple M1 ``` Geometric mean (ms) Name of Test activations activations.patch activations.patch vs activations (x-factor) Celu::Layer_Elementwise::OCV/CPU 16.184 2.118 7.64 Celu::Layer_Elementwise::OCV/CPU_FP16 16.280 2.123 7.67 Elu::Layer_Elementwise::OCV/CPU 9.123 1.878 4.86 Elu::Layer_Elementwise::OCV/CPU_FP16 9.085 1.897 4.79 Gelu::Layer_Elementwise::OCV/CPU 0.089 0.081 1.11 Gelu::Layer_Elementwise::OCV/CPU_FP16 0.086 0.074 1.17 HardSwish::Layer_Elementwise::OCV/CPU 1.560 1.555 1.00 HardSwish::Layer_Elementwise::OCV/CPU_FP16 1.536 1.523 1.01 Mish::Layer_Elementwise::OCV/CPU 6.077 2.476 2.45 Mish::Layer_Elementwise::OCV/CPU_FP16 5.990 2.496 2.40 Selu::Layer_Elementwise::OCV/CPU 11.351 1.976 5.74 Selu::Layer_Elementwise::OCV/CPU_FP16 11.533 1.985 5.81 Swish::Layer_Elementwise::OCV/CPU 4.687 1.890 2.48 Swish::Layer_Elementwise::OCV/CPU_FP16 4.715 1.873 2.52 ``` #### Intel i7-12700K ``` Geometric mean (ms) Name of Test activations activations.patch activations.patch vs activations (x-factor) Celu::Layer_Elementwise::OCV/CPU 17.106 3.560 4.81 Elu::Layer_Elementwise::OCV/CPU 5.064 3.478 1.46 Gelu::Layer_Elementwise::OCV/CPU 0.036 0.035 1.04 HardSwish::Layer_Elementwise::OCV/CPU 2.914 2.893 1.01 Mish::Layer_Elementwise::OCV/CPU 3.820 3.529 1.08 Selu::Layer_Elementwise::OCV/CPU 10.799 3.593 3.01 Swish::Layer_Elementwise::OCV/CPU 3.651 3.473 1.05 ``` ### Pull Request Readiness Checklist See details at https://github.com/opencv/opencv/wiki/How_to_contribute#making-a-good-pull-request - [x] I agree to contribute to the project under Apache 2 License. - [x] To the best of my knowledge, the proposed patch is not based on a code under GPL or another license that is incompatible with OpenCV - [x] The PR is proposed to the proper branch - [x] There is a reference to the original bug report and related work - [x] There is accuracy test, performance test and test data in opencv_extra repository, if applicable Patch to opencv_extra has the same branch name. - [x] The feature is well documented and sample code can be built with the project CMake
This commit is contained in:
@@ -975,49 +975,72 @@ INSTANTIATE_TEST_CASE_P(/**/, Layer_Softmax, Combine(
|
||||
/* withCann= */ false) // only test on CPU
|
||||
));
|
||||
|
||||
using Layer_Elementwise = TestBaseWithParam<tuple<std::vector<int>, std::string, tuple<Backend, Target>>>;
|
||||
PERF_TEST_P_(Layer_Elementwise, elementwise) {
|
||||
std::vector<int> input_shape = get<0>(GetParam());
|
||||
std::string op = get<1>(GetParam());
|
||||
int backend_id = get<0>(get<2>(GetParam()));
|
||||
int target_id = get<1>(get<2>(GetParam()));
|
||||
struct Layer_Elementwise : public TestBaseWithParam<tuple<Backend, Target>> {
|
||||
void test_layer(const std::string &op_type, const std::vector<int> &input_shape) {
|
||||
int backend_id = get<0>(GetParam());
|
||||
int target_id = get<1>(GetParam());
|
||||
|
||||
Mat input(input_shape, CV_32F);
|
||||
randn(input, 0.f, 1.f);
|
||||
Mat input(input_shape, CV_32F);
|
||||
randu(input, -10.0f, 10.f);
|
||||
|
||||
LayerParams lp;
|
||||
lp.type = op;
|
||||
lp.name = "TestLayer";
|
||||
LayerParams lp;
|
||||
lp.type = op_type;
|
||||
lp.name = cv::format("PerfLayer/%s", op_type.c_str());
|
||||
|
||||
Net net;
|
||||
net.addLayerToPrev(lp.name, lp.type, lp);
|
||||
Net net;
|
||||
net.addLayerToPrev(lp.name, lp.type, lp);
|
||||
|
||||
// Warmup
|
||||
{
|
||||
net.setInput(input);
|
||||
net.setPreferableBackend(backend_id);
|
||||
net.setPreferableTarget(target_id);
|
||||
Mat out = net.forward();
|
||||
// Warmup
|
||||
{
|
||||
net.setInput(input);
|
||||
net.setPreferableBackend(backend_id);
|
||||
net.setPreferableTarget(target_id);
|
||||
net.forward();
|
||||
}
|
||||
|
||||
TEST_CYCLE() {
|
||||
net.forward();
|
||||
}
|
||||
|
||||
SANITY_CHECK_NOTHING();
|
||||
}
|
||||
|
||||
TEST_CYCLE() {
|
||||
net.forward();
|
||||
}
|
||||
int N = 2;
|
||||
int C = 32;
|
||||
int H = 416;
|
||||
int W = 416;
|
||||
};
|
||||
|
||||
SANITY_CHECK_NOTHING();
|
||||
PERF_TEST_P_(Layer_Elementwise, Gelu) {
|
||||
test_layer("Gelu", std::vector<int>{1, 50, 3072});
|
||||
}
|
||||
PERF_TEST_P_(Layer_Elementwise, Swish) {
|
||||
test_layer("Swish", std::vector<int>{N, C, H, W});
|
||||
}
|
||||
PERF_TEST_P_(Layer_Elementwise, Mish) {
|
||||
test_layer("Mish", std::vector<int>{N, C, H, W});
|
||||
}
|
||||
PERF_TEST_P_(Layer_Elementwise, Elu) {
|
||||
test_layer("ELU", std::vector<int>{N, C, H, W});
|
||||
}
|
||||
PERF_TEST_P_(Layer_Elementwise, Celu) {
|
||||
test_layer("Celu", std::vector<int>{N, C, H, W});
|
||||
}
|
||||
PERF_TEST_P_(Layer_Elementwise, Selu) {
|
||||
test_layer("Selu", std::vector<int>{N, C, H, W});
|
||||
}
|
||||
PERF_TEST_P_(Layer_Elementwise, HardSwish) {
|
||||
test_layer("HardSwish", std::vector<int>{N, C, H, W});
|
||||
}
|
||||
|
||||
INSTANTIATE_TEST_CASE_P(/**/, Layer_Elementwise, testing::Combine(
|
||||
testing::Values(std::vector<int>{1, 50, 3072}),
|
||||
testing::Values(std::string{"Gelu"}),
|
||||
dnnBackendsAndTargets(/* withInferenceEngine= */ true,
|
||||
/* withHalide= */ false,
|
||||
/* withCpuOCV= */ true,
|
||||
/* withVkCom= */ false,
|
||||
/* withCUDA= */ true,
|
||||
/* withNgraph= */ true,
|
||||
/* withWebnn= */ false,
|
||||
/* withCann= */ false) // only test on CPU
|
||||
));
|
||||
INSTANTIATE_TEST_CASE_P(/**/, Layer_Elementwise,
|
||||
dnnBackendsAndTargets(/* withInferenceEngine= */ true,
|
||||
/* withHalide= */ false,
|
||||
/* withCpuOCV= */ true,
|
||||
/* withVkCom= */ false,
|
||||
/* withCUDA= */ true,
|
||||
/* withNgraph= */ true,
|
||||
/* withWebnn= */ false,
|
||||
/* withCann= */ false));
|
||||
|
||||
} // namespace
|
||||
|
||||
Reference in New Issue
Block a user