COMPMID-1849: Implement CPPDetectionPostProcessLayer

* Add DetectionPostProcessLayer
* Add DetectionPostProcessLayer at the graph

Change-Id: I7e56f6cffc26f112d26dfe74853085bb8ec7d849
Signed-off-by: Isabella Gottardi <isabella.gottardi@arm.com>
Reviewed-on: https://review.mlplatform.org/c/1639
Reviewed-by: Giuseppe Rossini <giuseppe.rossini@arm.com>
Tested-by: Arm Jenkins <bsgcomp@arm.com>
diff --git a/arm_compute/runtime/CPP/CPPFunctions.h b/arm_compute/runtime/CPP/CPPFunctions.h
index 1dff03f..743929f 100644
--- a/arm_compute/runtime/CPP/CPPFunctions.h
+++ b/arm_compute/runtime/CPP/CPPFunctions.h
@@ -27,6 +27,7 @@
 /* Header regrouping all the CPP functions */
 #include "arm_compute/runtime/CPP/functions/CPPBoxWithNonMaximaSuppressionLimit.h"
 #include "arm_compute/runtime/CPP/functions/CPPDetectionOutputLayer.h"
+#include "arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h"
 #include "arm_compute/runtime/CPP/functions/CPPNonMaximumSuppression.h"
 #include "arm_compute/runtime/CPP/functions/CPPPermute.h"
 #include "arm_compute/runtime/CPP/functions/CPPTopKV.h"
diff --git a/arm_compute/runtime/CPP/functions/CPPDetectionOutputLayer.h b/arm_compute/runtime/CPP/functions/CPPDetectionOutputLayer.h
index 71be8a0..4e1b8f2 100644
--- a/arm_compute/runtime/CPP/functions/CPPDetectionOutputLayer.h
+++ b/arm_compute/runtime/CPP/functions/CPPDetectionOutputLayer.h
@@ -28,17 +28,10 @@
 
 #include "arm_compute/core/Types.h"
 
-#include <map>
-
 namespace arm_compute
 {
 class ITensor;
 
-// Normalized Bounding Box [xmin, ymin, xmax, ymax]
-using NormalizedBBox = std::array<float, 4>;
-// LabelBBox used for map label and bounding box
-using LabelBBox = std::map<int, std::vector<NormalizedBBox>>;
-
 /** CPP Function to generate the detection output based on location and confidence
  * predictions by doing non maximum suppression.
  *
@@ -91,7 +84,7 @@
 
     std::vector<LabelBBox> _all_location_predictions;
     std::vector<std::map<int, std::vector<float>>> _all_confidence_scores;
-    std::vector<NormalizedBBox> _all_prior_bboxes;
+    std::vector<BBox> _all_prior_bboxes;
     std::vector<std::array<float, 4>> _all_prior_variances;
     std::vector<LabelBBox> _all_decode_bboxes;
     std::vector<std::map<int, std::vector<int>>> _all_indices;
diff --git a/arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h b/arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h
new file mode 100644
index 0000000..c13def6
--- /dev/null
+++ b/arm_compute/runtime/CPP/functions/CPPDetectionPostProcessLayer.h
@@ -0,0 +1,123 @@
+/*
+ * Copyright (c) 2019 ARM Limited.
+ *
+ * SPDX-License-Identifier: MIT
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining a copy
+ * of this software and associated documentation files (the "Software"), to
+ * deal in the Software without restriction, including without limitation the
+ * rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
+ * sell copies of the Software, and to permit persons to whom the Software is
+ * furnished to do so, subject to the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be included in all
+ * copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+ * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+ * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+ * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+ * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+ * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+ * SOFTWARE.
+ */
+#ifndef __ARM_COMPUTE_CPP_DETECTION_POSTPROCESS_H__
+#define __ARM_COMPUTE_CPP_DETECTION_POSTPROCESS_H__
+
+#include "arm_compute/runtime/CPP/ICPPSimpleFunction.h"
+
+#include "arm_compute/core/Types.h"
+#include "arm_compute/runtime/CPP/functions/CPPNonMaximumSuppression.h"
+#include "arm_compute/runtime/IMemoryManager.h"
+#include "arm_compute/runtime/MemoryGroup.h"
+#include "arm_compute/runtime/Tensor.h"
+
+#include <map>
+
+namespace arm_compute
+{
+class ITensor;
+
+/** CPP Function to generate the detection output based on center size encoded boxes, class prediction and anchors
+ *  by doing non maximum suppression.
+ *
+ * @note Intended for use with MultiBox detection method.
+ */
+class CPPDetectionPostProcessLayer : public IFunction
+{
+public:
+    /** Constructor */
+    CPPDetectionPostProcessLayer(std::shared_ptr<IMemoryManager> memory_manager = nullptr);
+    /** Prevent instances of this class from being copied (As this class contains pointers) */
+    CPPDetectionPostProcessLayer(const CPPDetectionPostProcessLayer &) = delete;
+    /** Prevent instances of this class from being copied (As this class contains pointers) */
+    CPPDetectionPostProcessLayer &operator=(const CPPDetectionPostProcessLayer &) = delete;
+    /** Configure the detection output layer CPP function
+     *
+     * @param[in]  input_box_encoding The bounding box input tensor. Data types supported: F32, QASYMM8.
+     * @param[in]  input_score        The class prediction input tensor. Data types supported: Same as @p input_box_encoding.
+     * @param[in]  input_anchors      The anchors input tensor. Data types supported: Same as @p input_box_encoding.
+     * @param[out] output_boxes       The boxes output tensor. Data types supported: F32.
+     * @param[out] output_classes     The classes output tensor. Data types supported: Same as @p output_boxes.
+     * @param[out] output_scores      The scores output tensor. Data types supported: Same as @p output_boxes.
+     * @param[out] num_detection      The number of output detection. Data types supported: Same as @p output_boxes.
+     * @param[in]  info               (Optional) DetectionPostProcessLayerInfo information.
+     *
+     * @note Output contains all the detections. Of those, only the ones selected by the valid region are valid.
+     */
+    void configure(const ITensor *input_box_encoding, const ITensor *input_score, const ITensor *input_anchors,
+                   ITensor *output_boxes, ITensor *output_classes, ITensor *output_scores, ITensor *num_detection, DetectionPostProcessLayerInfo info = DetectionPostProcessLayerInfo());
+    /** Static function to check if given info will lead to a valid configuration of @ref CPPDetectionPostProcessLayer
+     *
+     * @param[in]  input_box_encoding The bounding box input tensor info. Data types supported: F32, QASYMM8.
+     * @param[in]  input_class_score  The class prediction input tensor info. Data types supported: F32, QASYMM8.
+     * @param[in]  input_anchors      The anchors input tensor. Data types supported: F32, QASYMM8.
+     * @param[out] output_boxes       The output tensor. Data types supported: F32.
+     * @param[out] output_classes     The output tensor. Data types supported: Same as @p output_boxes.
+     * @param[out] output_scores      The output tensor. Data types supported: Same as @p output_boxes.
+     * @param[out] num_detection      The number of output detection. Data types supported: Same as @p output_boxes.
+     * @param[in]  info               (Optional) DetectionPostProcessLayerInfo information.
+     *
+     * @return a status
+     */
+    static Status validate(const ITensorInfo *input_box_encoding, const ITensorInfo *input_class_score, const ITensorInfo *input_anchors,
+                           ITensorInfo *output_boxes, ITensorInfo *output_classes, ITensorInfo *output_scores, ITensorInfo *num_detection,
+                           DetectionPostProcessLayerInfo info = DetectionPostProcessLayerInfo());
+    // Inherited methods overridden:
+    void run() override;
+
+private:
+    MemoryGroup                   _memory_group;
+    CPPNonMaximumSuppression      _nms;
+    const ITensor                *_input_box_encoding;
+    const ITensor                *_input_scores;
+    const ITensor                *_input_anchors;
+    ITensor                      *_output_boxes;
+    ITensor                      *_output_classes;
+    ITensor                      *_output_scores;
+    ITensor                      *_num_detection;
+    DetectionPostProcessLayerInfo _info;
+
+    const unsigned int _kBatchSize   = 1;
+    const unsigned int _kNumCoordBox = 4;
+    unsigned int       _num_boxes;
+    unsigned int       _num_classes_with_background;
+    unsigned int       _num_max_detected_boxes;
+
+    Tensor         _decoded_boxes;
+    Tensor         _decoded_scores;
+    Tensor         _selected_indices;
+    Tensor         _class_scores;
+    const ITensor *_input_scores_to_use;
+
+    // Intermediate results
+    std::vector<int>          _result_idx_boxes_after_nms;
+    std::vector<int>          _result_classes_after_nms;
+    std::vector<float>        _result_scores_after_nms;
+    std::vector<unsigned int> _sorted_indices;
+
+    // Temporary values
+    std::vector<float> _box_scores;
+};
+} // namespace arm_compute
+#endif /* __ARM_COMPUTE_CPP_DETECTION_POSTPROCESS_H__ */