From d985378af0c9a4db6a483634dd40526cd4031dee Mon Sep 17 00:00:00 2001
From: Giuseppe Rossini <giuseppe.rossini@arm.com>
Date: Fri, 25 Oct 2019 11:11:44 +0100
Subject: COMPMID-2588: Optimize the output detection kernel required by
 MobileNet-SSD (~27% improvement)

Change-Id: Ic6ce570af3878a0666ec680e0efabba3fcfd1222
Signed-off-by: Giuseppe Rossini <giuseppe.rossini@arm.com>
Reviewed-on: https://review.mlplatform.org/c/2160
Comments-Addressed: Arm Jenkins <bsgcomp@arm.com>
Reviewed-by: Georgios Pinitas <georgios.pinitas@arm.com>
Reviewed-by: Gian Marco Iodice <gianmarco.iodice@arm.com>
Tested-by: Arm Jenkins <bsgcomp@arm.com>
---
 arm_compute/core/Types.h                           |  19 ++--
 .../CPP/functions/CPPDetectionPostProcessLayer.h   |   1 +
 arm_compute/runtime/NEON/NEFunctions.h             |   1 +
 .../NEON/functions/NEDetectionPostProcessLayer.h   | 100 +++++++++++++++++++++
 4 files changed, 116 insertions(+), 5 deletions(-)
 create mode 100644 arm_compute/runtime/NEON/functions/NEDetectionPostProcessLayer.h

(limited to 'arm_compute')

diff --git a/arm_compute/core/Types.h b/arm_compute/core/Types.h
index d7b47ac512..0a25277b57 100644
--- a/arm_compute/core/Types.h
+++ b/arm_compute/core/Types.h
@@ -1099,7 +1099,8 @@ public:
           _num_classes(),
           _scales_values(),
           _use_regular_nms(),
-          _detection_per_class()
+          _detection_per_class(),
+          _dequantize_scores()
     {
     }
     /** Constructor
@@ -1110,11 +1111,12 @@ public:
      * @param[in] iou_threshold             Threshold to be used during the intersection over union.
      * @param[in] num_classes               Number of classes.
      * @param[in] scales_values             Scales values used for decode center size boxes.
-     * @param[in] use_regular_nms           (Optional) Boolean to determinate if use regular or fast nms.
-     * @param[in] detection_per_class       (Optional) Number of detection per class. Used in the Regular Non-Max-Suppression
+     * @param[in] use_regular_nms           (Optional) Boolean to determinate if use regular or fast nms. Defaults to false.
+     * @param[in] detection_per_class       (Optional) Number of detection per class. Used in the Regular Non-Max-Suppression. Defaults to 100.
+     * @param[in] dequantize_scores         (Optional) If the scores need to be dequantized. Defaults to true.
      */
     DetectionPostProcessLayerInfo(unsigned int max_detections, unsigned int max_classes_per_detection, float nms_score_threshold, float iou_threshold, unsigned int num_classes,
-                                  std::array<float, 4> scales_values, bool use_regular_nms = false, unsigned int detection_per_class = 100)
+                                  std::array<float, 4> scales_values, bool use_regular_nms = false, unsigned int detection_per_class = 100, bool dequantize_scores = true)
         : _max_detections(max_detections),
           _max_classes_per_detection(max_classes_per_detection),
           _nms_score_threshold(nms_score_threshold),
@@ -1122,7 +1124,8 @@ public:
           _num_classes(num_classes),
           _scales_values(scales_values),
           _use_regular_nms(use_regular_nms),
-          _detection_per_class(detection_per_class)
+          _detection_per_class(detection_per_class),
+          _dequantize_scores(dequantize_scores)
     {
     }
     /** Get max detections. */
@@ -1184,6 +1187,11 @@ public:
         // Saved as [y,x,h,w]
         return _scales_values[3];
     }
+    /** Get dequantize_scores value. */
+    bool dequantize_scores() const
+    {
+        return _dequantize_scores;
+    }
 
 private:
     unsigned int _max_detections;
@@ -1194,6 +1202,7 @@ private:
     std::array<float, 4> _scales_values;
     bool         _use_regular_nms;
     unsigned int _detection_per_class;
+    bool         _dequantize_scores;
 };
 
 /** Pooling Layer Information class */
diff --git a/arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h b/arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h
index 1c918d220c..64568e8b96 100644
--- a/arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h
+++ b/arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h
@@ -103,6 +103,7 @@ private:
     unsigned int       _num_boxes;
     unsigned int       _num_classes_with_background;
     unsigned int       _num_max_detected_boxes;
+    bool               _dequantize_scores;
 
     Tensor         _decoded_boxes;
     Tensor         _decoded_scores;
diff --git a/arm_compute/runtime/NEON/NEFunctions.h b/arm_compute/runtime/NEON/NEFunctions.h
index 28fd7f37b9..95369cebf1 100644
--- a/arm_compute/runtime/NEON/NEFunctions.h
+++ b/arm_compute/runtime/NEON/NEFunctions.h
@@ -59,6 +59,7 @@
 #include "arm_compute/runtime/NEON/functions/NEDepthwiseConvolutionLayer.h"
 #include "arm_compute/runtime/NEON/functions/NEDequantizationLayer.h"
 #include "arm_compute/runtime/NEON/functions/NEDerivative.h"
+#include "arm_compute/runtime/NEON/functions/NEDetectionPostProcessLayer.h"
 #include "arm_compute/runtime/NEON/functions/NEDilate.h"
 #include "arm_compute/runtime/NEON/functions/NEDirectConvolutionLayer.h"
 #include "arm_compute/runtime/NEON/functions/NEElementwiseOperations.h"
diff --git a/arm_compute/runtime/NEON/functions/NEDetectionPostProcessLayer.h b/arm_compute/runtime/NEON/functions/NEDetectionPostProcessLayer.h
new file mode 100644
index 0000000000..58ba98a376
--- /dev/null
+++ b/arm_compute/runtime/NEON/functions/NEDetectionPostProcessLayer.h
@@ -0,0 +1,100 @@
+/*
+ * Copyright (c) 2019 ARM Limited.
+ *
+ * SPDX-License-Identifier: MIT
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining a copy
+ * of this software and associated documentation files (the "Software"), to
+ * deal in the Software without restriction, including without limitation the
+ * rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
+ * sell copies of the Software, and to permit persons to whom the Software is
+ * furnished to do so, subject to the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be included in all
+ * copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+ * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+ * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+ * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+ * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+ * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+ * SOFTWARE.
+ */
+#ifndef __ARM_COMPUTE_NE_DETECTION_POSTPROCESS_H__
+#define __ARM_COMPUTE_NE_DETECTION_POSTPROCESS_H__
+
+#include "arm_compute/runtime/NEON/INESimpleFunction.h"
+
+#include "arm_compute/core/Types.h"
+#include "arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h"
+#include "arm_compute/runtime/IMemoryManager.h"
+#include "arm_compute/runtime/MemoryGroup.h"
+#include "arm_compute/runtime/NEON/functions/NEDequantizationLayer.h"
+#include "arm_compute/runtime/Tensor.h"
+
+#include <map>
+
+namespace arm_compute
+{
+class ITensor;
+
+/** NE Function to generate the detection output based on center size encoded boxes, class prediction and anchors
+ *  by doing non maximum suppression.
+ *
+ * @note Intended for use with MultiBox detection method.
+ */
+class NEDetectionPostProcessLayer : public IFunction
+{
+public:
+    /** Constructor */
+    NEDetectionPostProcessLayer(std::shared_ptr<IMemoryManager> memory_manager = nullptr);
+    /** Prevent instances of this class from being copied (As this class contains pointers) */
+    NEDetectionPostProcessLayer(const NEDetectionPostProcessLayer &) = delete;
+    /** Prevent instances of this class from being copied (As this class contains pointers) */
+    NEDetectionPostProcessLayer &operator=(const NEDetectionPostProcessLayer &) = delete;
+    /** Configure the detection output layer NE function
+     *
+     * @param[in]  input_box_encoding The bounding box input tensor. Data types supported: F32, QASYMM8.
+     * @param[in]  input_score        The class prediction input tensor. Data types supported: Same as @p input_box_encoding.
+     * @param[in]  input_anchors      The anchors input tensor. Data types supported: Same as @p input_box_encoding.
+     * @param[out] output_boxes       The boxes output tensor. Data types supported: F32.
+     * @param[out] output_classes     The classes output tensor. Data types supported: Same as @p output_boxes.
+     * @param[out] output_scores      The scores output tensor. Data types supported: Same as @p output_boxes.
+     * @param[out] num_detection      The number of output detection. Data types supported: Same as @p output_boxes.
+     * @param[in]  info               (Optional) DetectionPostProcessLayerInfo information.
+     *
+     * @note Output contains all the detections. Of those, only the ones selected by the valid region are valid.
+     */
+    void configure(const ITensor *input_box_encoding, const ITensor *input_score, const ITensor *input_anchors,
+                   ITensor *output_boxes, ITensor *output_classes, ITensor *output_scores, ITensor *num_detection, DetectionPostProcessLayerInfo info = DetectionPostProcessLayerInfo());
+    /** Static function to check if given info will lead to a valid configuration of @ref NEDetectionPostProcessLayer
+     *
+     * @param[in]  input_box_encoding The bounding box input tensor info. Data types supported: F32, QASYMM8.
+     * @param[in]  input_class_score  The class prediction input tensor info. Data types supported: F32, QASYMM8.
+     * @param[in]  input_anchors      The anchors input tensor. Data types supported: F32, QASYMM8.
+     * @param[out] output_boxes       The output tensor. Data types supported: F32.
+     * @param[out] output_classes     The output tensor. Data types supported: Same as @p output_boxes.
+     * @param[out] output_scores      The output tensor. Data types supported: Same as @p output_boxes.
+     * @param[out] num_detection      The number of output detection. Data types supported: Same as @p output_boxes.
+     * @param[in]  info               (Optional) DetectionPostProcessLayerInfo information.
+     *
+     * @return a status
+     */
+    static Status validate(const ITensorInfo *input_box_encoding, const ITensorInfo *input_class_score, const ITensorInfo *input_anchors,
+                           ITensorInfo *output_boxes, ITensorInfo *output_classes, ITensorInfo *output_scores, ITensorInfo *num_detection,
+                           DetectionPostProcessLayerInfo info = DetectionPostProcessLayerInfo());
+    // Inherited methods overridden:
+    void run() override;
+
+private:
+    MemoryGroup _memory_group;
+
+    NEDequantizationLayer        _dequantize;
+    CPPDetectionPostProcessLayer _detection_post_process;
+
+    Tensor _decoded_scores;
+    bool   _run_dequantize;
+};
+} // namespace arm_compute
+#endif /* __ARM_COMPUTE_NE_DETECTION_POSTPROCESS_H__ */
-- 
cgit v1.2.1