From 215b4ea6c9dee480a22070d5873b0b8cb52531a0 Mon Sep 17 00:00:00 2001 From: Gian Marco Iodice Date: Thu, 28 Jun 2018 16:29:29 +0100 Subject: COMPMID-1277 - Optimizing CLIm2ColKernel for NHWC. This patch includes: - Im2Col optimizations for NHWC using a new data layout - Refactoring of CLIm2ColKernel adding validation method and auto-init - Removed im2col_reduced from CLIm2ColKernel and created a new kernel CLFlattenLayerKernel Change-Id: I1620640b6796baa268324b33ae92cdd8de53e27c Reviewed-on: https://eu-gerrit-1.euhpc.arm.com/141241 Tested-by: Jenkins Reviewed-by: Giorgio Arena --- arm_compute/core/CL/kernels/CLIm2ColKernel.h | 41 ++-------------------------- 1 file changed, 3 insertions(+), 38 deletions(-) (limited to 'arm_compute/core/CL/kernels/CLIm2ColKernel.h') diff --git a/arm_compute/core/CL/kernels/CLIm2ColKernel.h b/arm_compute/core/CL/kernels/CLIm2ColKernel.h index fc930abcbe..ae19319047 100644 --- a/arm_compute/core/CL/kernels/CLIm2ColKernel.h +++ b/arm_compute/core/CL/kernels/CLIm2ColKernel.h @@ -96,48 +96,13 @@ public: // Inherited methods overridden: void run(const Window &window, cl::CommandQueue &queue) override; -private: - /** Run the reshape kernel optimised for the special case (stride is 1, padding is 0 and kernel's low 3 dimensions are same as input) - * - * @param[in] window Region on which to execute the kernel. (Must be a valid region of the window returned by window()). - * @param[in,out] queue Command queue on which to enqueue the kernel. - */ - void run_reduced(const Window &window, cl::CommandQueue &queue); - /** run the generic convolution layer input reshape kernel - * - * @param[in] window Region on which to execute the kernel. (Must be a valid region of the window returned by window()). - * @param[in,out] queue Command queue on which to enqueue the kernel. - */ - void run_generic(const Window &window, cl::CommandQueue &queue); - - /** Chooses and configure the right kernel for the given input arguments. - * - * @param[in] input The input tensor to convert. 3 lower dimensions represent a single input [width, height, IFM], - * while every optional dimension from 4 and above represent a batch of inputs. Data types supported: QASYMM8/F16/F32 - * @param[in] output The output tensor. First 2 lower dimensions represent a transform of each 3D input, - * while every dimension above represents a batch. Data types supported: Same as @p input - * @param[in] kernel_dims The kernel dimensions (width and height). - * @param[in] dilation Dilation, in elements, across x and y. Defaults to (1, 1). - * @param[in] conv_info Contains padding and stride information described in @ref PadStrideInfo. - * @param[in] has_bias In case biases are provided expands the matrix with 1. - * @param[out] build_opts OpenCL buil program options. - * - * @return the name of the kernel chosen - */ - std::string configure_window(const ICLTensor *input, ICLTensor *output, const Size2D &kernel_dims, - const Size2D &dilation, const PadStrideInfo &conv_info, CLBuildOptions &build_opts); - - /** Common signature for the kernel to run */ - using Im2ColFunction = void (CLIm2ColKernel::*)(const Window &, cl::CommandQueue &); - public: const ICLTensor *_input; ICLTensor *_output; - PadStrideInfo _conv_info; std::pair _convolved_dims; - unsigned int _num_elems_processed_per_iteration; - Im2ColFunction _run_func; - Size2D _kernel_dims; + unsigned int _num_elems_processed_per_iteration; + Size2D _kernel_dims; + PadStrideInfo _conv_info; }; } // namespace arm_compute #endif /*__ARM_COMPUTE_CLIM2COLKERNEL_H__ */ -- cgit v1.2.1