blob: dacf3cd4350d6f5b47518b0c1f51f01caa97d91a [file] [log] [blame]
Gunes Bayird5f9a1c2023-08-17 11:04:02 +01001/*
2 * Copyright (c) 2023 Arm Limited.
3 *
4 * SPDX-License-Identifier: MIT
5 *
6 * Permission is hereby granted, free of charge, to any person obtaining a copy
7 * of this software and associated documentation files (the "Software"), to
8 * deal in the Software without restriction, including without limitation the
9 * rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
10 * sell copies of the Software, and to permit persons to whom the Software is
11 * furnished to do so, subject to the following conditions:
12 *
13 * The above copyright notice and this permission notice shall be included in all
14 * copies or substantial portions of the Software.
15 *
16 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
17 * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
18 * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
19 * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
20 * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
21 * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
22 * SOFTWARE.
23 */
24
25#ifndef CKW_VALIDATION_TESTS_CLKERNELWRITEROPLOADINDIRECTTEST_H
26#define CKW_VALIDATION_TESTS_CLKERNELWRITEROPLOADINDIRECTTEST_H
27
28#include "ckw/TileInfo.h"
29#include "ckw/types/DataType.h"
30#include "ckw/TensorSampler.h"
31#include "ckw/types/MemoryOperation.h"
32#include "ckw/types/TensorSamplerTypes.h"
33#include "src/cl/CLKernelWriter.h"
34#include "validation/tests/common/KernelWriterInterceptor.h"
35#include "validation/tests/common/Common.h"
36
37#include <vector>
38
39namespace ckw
40{
41
42class CLKernelWriterOpLoadIndirectTest : public ITest
43{
44private:
45 using AddressModeX = TensorSamplerAddressModeX;
46 using AddressModeY = TensorSamplerAddressModeY;
47 using AddressModeZ = TensorSamplerAddressModeZ;
48 using Format = TensorSamplerFormat;
49 using Storage = TensorStorageType;
50
51 struct Coordinates
52 {
53 Coordinates(std::string x, std::string y, std::string z, std::string batch)
54 : x(x), y(y), z(z), batch(batch)
55 {
56 }
57
58 std::string x;
59 std::string y;
60 std::string z;
61 std::string batch;
62 };
63
64 struct SamplerData
65 {
66 SamplerData(Format format, AddressModeX mode_x, AddressModeY mode_y, AddressModeZ mode_z)
67 : format(format), mode_x(mode_x), mode_y(mode_y), mode_z(mode_z)
68 {
69 }
70
71 Format format;
72 AddressModeX mode_x;
73 AddressModeY mode_y;
74 AddressModeZ mode_z;
75 };
76
77 using CLKernelWriterOpLoadIndirectConfig = std::tuple<TileInfo, TensorStorageType, SamplerData, Coordinates, std::string>;
78
79public:
80 CLKernelWriterOpLoadIndirectTest()
81 {
82 const std::string fp_2x3_tile = R"_(
83G0__tile__0 = vload3(0, (__global float*)(G0__tensor_ptr + (G0__x) * sizeof(float) + (G0__indirect_addr__0) * G0__tensor_stride1 + (G0__z) * G0__tensor_stride2 + (G0__b) * G0__tensor_stride3));
84G0__tile__1 = vload3(0, (__global float*)(G0__tensor_ptr + (G0__x) * sizeof(float) + (G0__indirect_addr__1) * G0__tensor_stride1 + (G0__z) * G0__tensor_stride2 + (G0__b) * G0__tensor_stride3));
85)_";
86
87 const std::string half_2x4_yz_collapsed_y_clamped_to_border_max_only_image = R"_(
88G0__tile__0 = read_imageh(G0__tensor_img2d, CLK_NORMALIZED_COORDS_FALSE | CLK_ADDRESS_CLAMP | CLK_FILTER_NEAREST, (int2)((G0__x) >> 2, (G0__indirect_addr__0 + (G0__b) * G0__tensor_dim1xdim2 * 1)));
89G0__tile__1 = read_imageh(G0__tensor_img2d, CLK_NORMALIZED_COORDS_FALSE | CLK_ADDRESS_CLAMP | CLK_FILTER_NEAREST, (int2)((G0__x) >> 2, (G0__indirect_addr__1 + (G0__b) * G0__tensor_dim1xdim2 * 1)));
90)_";
91
92 const std::string int_2x4_y_skip_less_than_zero = R"_(
93if(G0__indirect_addr__0 >= 0)
94{
95G0__tile__0 = vload4(0, (__global int*)(G0__tensor_ptr + (G0__x) * sizeof(int) + (G0__indirect_addr__0) * G0__tensor_stride1 + (G0__z) * G0__tensor_stride2 + (G0__b) * G0__tensor_stride3));
96}
97if(G0__indirect_addr__1 >= 0)
98{
99G0__tile__1 = vload4(0, (__global int*)(G0__tensor_ptr + (G0__x) * sizeof(int) + (G0__indirect_addr__1) * G0__tensor_stride1 + (G0__z) * G0__tensor_stride2 + (G0__b) * G0__tensor_stride3));
100}
101)_";
102
103 // tensor shape in x-dim is 10 (thus the 8, 2 vloads in if, else blocks respectively)
104 const std::string uint16_3x8_yz_collapsed_b_eq_0_x_overlapping_min_y_skip_less_than_zero = R"_(
105if(G0__x > 0)
106{
107if(G0__indirect_addr__0 >= 0)
108{
109G0__tile__0 = vload8(0, (__global ushort*)(G0__tensor_ptr + (G0__x) * sizeof(ushort) + (G0__indirect_addr__0) * G0__tensor_stride1 + (G0__0) * G0__tensor_stride3));
110}
111if(G0__indirect_addr__1 >= 0)
112{
113G0__tile__1 = vload8(0, (__global ushort*)(G0__tensor_ptr + (G0__x) * sizeof(ushort) + (G0__indirect_addr__1) * G0__tensor_stride1 + (G0__0) * G0__tensor_stride3));
114}
115if(G0__indirect_addr__2 >= 0)
116{
117G0__tile__2 = vload8(0, (__global ushort*)(G0__tensor_ptr + (G0__x) * sizeof(ushort) + (G0__indirect_addr__2) * G0__tensor_stride1 + (G0__0) * G0__tensor_stride3));
118}
119}
120else
121{
122if(G0__indirect_addr__0 >= 0)
123{
124G0__tile__0.s01 = vload2(0, (__global ushort*)(G0__tensor_ptr + (G0__x + 0) * sizeof(ushort) + (G0__indirect_addr__0) * G0__tensor_stride1 + (G0__0) * G0__tensor_stride3));
125}
126if(G0__indirect_addr__1 >= 0)
127{
128G0__tile__1.s01 = vload2(0, (__global ushort*)(G0__tensor_ptr + (G0__x + 0) * sizeof(ushort) + (G0__indirect_addr__1) * G0__tensor_stride1 + (G0__0) * G0__tensor_stride3));
129}
130if(G0__indirect_addr__2 >= 0)
131{
132G0__tile__2.s01 = vload2(0, (__global ushort*)(G0__tensor_ptr + (G0__x + 0) * sizeof(ushort) + (G0__indirect_addr__2) * G0__tensor_stride1 + (G0__0) * G0__tensor_stride3));
133}
134}
135)_";
136
137 // Configs Bundled
138 _configs = {
139 {
140 TileInfo(DataType::Fp32, 2, 3),
141 TensorStorageType::BufferUint8Ptr,
142 SamplerData(Format::Dim0_Dim1_Dim2, AddressModeX::None, AddressModeY::None, AddressModeZ::None),
143 Coordinates("x", "y", "z", "b"),
144 fp_2x3_tile
145 },
146 {
147 TileInfo(DataType::Fp16, 2, 4),
148 TensorStorageType::Texture2dReadOnly,
149 SamplerData(Format::Dim0_Dim1xDim2_1, AddressModeX::None, AddressModeY::ClampToBorderMaxOnly, AddressModeZ::None),
150 Coordinates("x", "y", "z", "b"),
151 half_2x4_yz_collapsed_y_clamped_to_border_max_only_image
152 },
153 {
154 TileInfo(DataType::Int32, 2, 4),
155 TensorStorageType::BufferUint8Ptr,
156 SamplerData(Format::Dim0_Dim1_Dim2, AddressModeX::None, AddressModeY::SkipLessThanZero, AddressModeZ::None),
157 Coordinates("x", "y", "z", "b"),
158 int_2x4_y_skip_less_than_zero
159 },
160 {
161 TileInfo(DataType::Uint16, 3, 8),
162 TensorStorageType::BufferUint8Ptr,
163 SamplerData(Format::Dim0_Dim1xDim2_1, AddressModeX::OverlappingMin, AddressModeY::SkipLessThanZero, AddressModeZ::None),
164 Coordinates("x", "y", "z", "0"),
165 uint16_3x8_yz_collapsed_b_eq_0_x_overlapping_min_y_skip_less_than_zero
166 }
167 };
168 }
169
170 bool run() override
171 {
172 bool all_tests_passed = true;
173 int32_t test_idx = 0;
174
175 for(auto _config: _configs)
176 {
177 KernelWriterInterceptor<CLKernelWriter> writer;
178
179 const TileInfo tile_info = std::get<0>(_config);
180 const Storage storage = std::get<1>(_config);
181 const SamplerData sampler_data = std::get<2>(_config);
182 const Coordinates coord = std::get<3>(_config);
183 const std::string expected_code = std::get<4>(_config).substr(1); // ignore initial newline, which was added for convenience
184
185 TileOperand tile_op = writer.declare_tile("tile", TileInfo(tile_info.data_type(), tile_info.height(), tile_info.width()));
186 TileOperand indirect_addr_op = writer.declare_tile("indirect_addr", TileInfo(DataType::Int32, tile_info.height(), 1)); // (M0, 1)
187 TileOperand x_op = writer.declare_tile(coord.x, TileInfo(DataType::Int32));
188 TileOperand z_op = writer.declare_tile(coord.z, TileInfo(DataType::Int32));
189 TileOperand batch_op = writer.declare_tile(coord.batch, TileInfo(DataType::Int32));
190
191 TensorShape tensor_shape {10, 10, 10, 10};
192 TensorInfo tensor_info(tile_info.data_type(), tensor_shape, TensorDataLayout::Nhwc, 0 /* id */);
193 TensorOperand tensor_op = writer.declare_tensor_argument("tensor", tensor_info);
194 TensorSampler sampler(storage, sampler_data.format, sampler_data.mode_x, sampler_data.mode_y, sampler_data.mode_z);
195
196 writer.start_capture_code();
197 writer.op_load_indirect(tile_op, tensor_op, sampler, x_op, indirect_addr_op, z_op, batch_op);
198
199 VALIDATE_TEST(writer.check_added_code(expected_code), all_tests_passed, test_idx++);
200 }
201
202 return all_tests_passed;
203 }
204
205 std::string name() override
206 {
207 return "CLKernelWriterOpLoadIndirectTest";
208 }
209
210private:
211 std::vector<CLKernelWriterOpLoadIndirectConfig> _configs {};
212};
213
214} // namespace ckw
215
216#endif // CKW_VALIDATION_TESTS_CLKERNELWRITEROPLOADINDIRECTTEST_H