xref: /aosp_15_r20/external/ComputeLibrary/cl_kernels/nhwc/indirect_convolution.clembed (revision c217d954acce2dbc11938adb493fc0abd69584f3)
1*c217d954SCole FaustR"(
2*c217d954SCole Faust
3*c217d954SCole Faust
4*c217d954SCole Faust
5*c217d954SCole Faust#ifndef ARM_COMPUTE_HELPER_H
6*c217d954SCole Faust#define ARM_COMPUTE_HELPER_H
7*c217d954SCole Faust
8*c217d954SCole Faust
9*c217d954SCole Faust
10*c217d954SCole Faust
11*c217d954SCole Faust#define STORE_ROW_1(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
12*c217d954SCole Faust    VSTORE(N0)                                                 \
13*c217d954SCole Faust    (BASENAME##0, 0, (__global DATA_TYPE *)(PTR + 0 * STRIDE_Y + Z##0));
14*c217d954SCole Faust
15*c217d954SCole Faust#define STORE_ROW_2(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
16*c217d954SCole Faust    STORE_ROW_1(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
17*c217d954SCole Faust    VSTORE(N0)                                                 \
18*c217d954SCole Faust    (BASENAME##1, 0, (__global DATA_TYPE *)(PTR + 1 * STRIDE_Y + Z##1));
19*c217d954SCole Faust
20*c217d954SCole Faust#define STORE_ROW_3(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
21*c217d954SCole Faust    STORE_ROW_2(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
22*c217d954SCole Faust    VSTORE(N0)                                                 \
23*c217d954SCole Faust    (BASENAME##2, 0, (__global DATA_TYPE *)(PTR + 2 * STRIDE_Y + Z##2));
24*c217d954SCole Faust
25*c217d954SCole Faust#define STORE_ROW_4(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
26*c217d954SCole Faust    STORE_ROW_3(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
27*c217d954SCole Faust    VSTORE(N0)                                                 \
28*c217d954SCole Faust    (BASENAME##3, 0, (__global DATA_TYPE *)(PTR + 3 * STRIDE_Y + Z##3));
29*c217d954SCole Faust
30*c217d954SCole Faust#define STORE_ROW_5(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
31*c217d954SCole Faust    STORE_ROW_4(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
32*c217d954SCole Faust    VSTORE(N0)                                                 \
33*c217d954SCole Faust    (BASENAME##4, 0, (__global DATA_TYPE *)(PTR + 4 * STRIDE_Y + Z##4));
34*c217d954SCole Faust
35*c217d954SCole Faust#define STORE_ROW_6(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
36*c217d954SCole Faust    STORE_ROW_5(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
37*c217d954SCole Faust    VSTORE(N0)                                                 \
38*c217d954SCole Faust    (BASENAME##5, 0, (__global DATA_TYPE *)(PTR + 5 * STRIDE_Y + Z##5));
39*c217d954SCole Faust
40*c217d954SCole Faust#define STORE_ROW_7(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
41*c217d954SCole Faust    STORE_ROW_6(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
42*c217d954SCole Faust    VSTORE(N0)                                                 \
43*c217d954SCole Faust    (BASENAME##6, 0, (__global DATA_TYPE *)(PTR + 6 * STRIDE_Y + Z##6));
44*c217d954SCole Faust
45*c217d954SCole Faust#define STORE_ROW_8(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
46*c217d954SCole Faust    STORE_ROW_7(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
47*c217d954SCole Faust    VSTORE(N0)                                                 \
48*c217d954SCole Faust    (BASENAME##7, 0, (__global DATA_TYPE *)(PTR + 7 * STRIDE_Y + Z##7));
49*c217d954SCole Faust
50*c217d954SCole Faust#define STORE_ROW_9(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
51*c217d954SCole Faust    STORE_ROW_8(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
52*c217d954SCole Faust    VSTORE(N0)                                                 \
53*c217d954SCole Faust    (BASENAME##8, 0, (__global DATA_TYPE *)(PTR + 8 * STRIDE_Y + Z##8));
54*c217d954SCole Faust
55*c217d954SCole Faust#define STORE_ROW_10(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
56*c217d954SCole Faust    STORE_ROW_9(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)      \
57*c217d954SCole Faust    VSTORE(N0)                                                  \
58*c217d954SCole Faust    (BASENAME##9, 0, (__global DATA_TYPE *)(PTR + 9 * STRIDE_Y + Z##9));
59*c217d954SCole Faust
60*c217d954SCole Faust#define STORE_ROW_11(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
61*c217d954SCole Faust    STORE_ROW_10(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
62*c217d954SCole Faust    VSTORE(N0)                                                  \
63*c217d954SCole Faust    (BASENAME##A, 0, (__global DATA_TYPE *)(PTR + 10 * STRIDE_Y + Z##A));
64*c217d954SCole Faust
65*c217d954SCole Faust#define STORE_ROW_12(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
66*c217d954SCole Faust    STORE_ROW_11(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
67*c217d954SCole Faust    VSTORE(N0)                                                  \
68*c217d954SCole Faust    (BASENAME##B, 0, (__global DATA_TYPE *)(PTR + 11 * STRIDE_Y + Z##B));
69*c217d954SCole Faust
70*c217d954SCole Faust#define STORE_ROW_13(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
71*c217d954SCole Faust    STORE_ROW_12(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
72*c217d954SCole Faust    VSTORE(N0)                                                  \
73*c217d954SCole Faust    (BASENAME##C, 0, (__global DATA_TYPE *)(PTR + 12 * STRIDE_Y + Z##C));
74*c217d954SCole Faust
75*c217d954SCole Faust#define STORE_ROW_14(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
76*c217d954SCole Faust    STORE_ROW_13(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
77*c217d954SCole Faust    VSTORE(N0)                                                  \
78*c217d954SCole Faust    (BASENAME##D, 0, (__global DATA_TYPE *)(PTR + 13 * STRIDE_Y + Z##D));
79*c217d954SCole Faust
80*c217d954SCole Faust#define STORE_ROW_15(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
81*c217d954SCole Faust    STORE_ROW_14(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
82*c217d954SCole Faust    VSTORE(N0)                                                  \
83*c217d954SCole Faust    (BASENAME##E, 0, (__global DATA_TYPE *)(PTR + 14 * STRIDE_Y + Z##E));
84*c217d954SCole Faust
85*c217d954SCole Faust#define STORE_ROW_16(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
86*c217d954SCole Faust    STORE_ROW_15(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
87*c217d954SCole Faust    VSTORE(N0)                                                  \
88*c217d954SCole Faust    (BASENAME##F, 0, (__global DATA_TYPE *)(PTR + 15 * STRIDE_Y + Z##F));
89*c217d954SCole Faust
90*c217d954SCole Faust
91*c217d954SCole Faust
92*c217d954SCole Faust#define CONVERT_STORE_ROW_1(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
93*c217d954SCole Faust    VSTORE(N0)                                                         \
94*c217d954SCole Faust    (CONVERT_SAT((BASENAME##0), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 0 * STRIDE_Y + Z##0));
95*c217d954SCole Faust
96*c217d954SCole Faust#define CONVERT_STORE_ROW_2(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
97*c217d954SCole Faust    CONVERT_STORE_ROW_1(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
98*c217d954SCole Faust    VSTORE(N0)                                                         \
99*c217d954SCole Faust    (CONVERT_SAT((BASENAME##1), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 1 * STRIDE_Y + Z##1));
100*c217d954SCole Faust
101*c217d954SCole Faust#define CONVERT_STORE_ROW_3(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
102*c217d954SCole Faust    CONVERT_STORE_ROW_2(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
103*c217d954SCole Faust    VSTORE(N0)                                                         \
104*c217d954SCole Faust    (CONVERT_SAT((BASENAME##2), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 2 * STRIDE_Y + Z##2));
105*c217d954SCole Faust
106*c217d954SCole Faust#define CONVERT_STORE_ROW_4(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
107*c217d954SCole Faust    CONVERT_STORE_ROW_3(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
108*c217d954SCole Faust    VSTORE(N0)                                                         \
109*c217d954SCole Faust    (CONVERT_SAT((BASENAME##3), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 3 * STRIDE_Y + Z##3));
110*c217d954SCole Faust
111*c217d954SCole Faust#define CONVERT_STORE_ROW_5(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
112*c217d954SCole Faust    CONVERT_STORE_ROW_4(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
113*c217d954SCole Faust    VSTORE(N0)                                                         \
114*c217d954SCole Faust    (CONVERT_SAT((BASENAME##4), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 4 * STRIDE_Y + Z##4));
115*c217d954SCole Faust
116*c217d954SCole Faust#define CONVERT_STORE_ROW_6(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
117*c217d954SCole Faust    CONVERT_STORE_ROW_5(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
118*c217d954SCole Faust    VSTORE(N0)                                                         \
119*c217d954SCole Faust    (CONVERT_SAT((BASENAME##5), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 5 * STRIDE_Y + Z##5));
120*c217d954SCole Faust
121*c217d954SCole Faust#define CONVERT_STORE_ROW_7(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
122*c217d954SCole Faust    CONVERT_STORE_ROW_6(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
123*c217d954SCole Faust    VSTORE(N0)                                                         \
124*c217d954SCole Faust    (CONVERT_SAT((BASENAME##6), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 6 * STRIDE_Y + Z##6));
125*c217d954SCole Faust
126*c217d954SCole Faust#define CONVERT_STORE_ROW_8(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
127*c217d954SCole Faust    CONVERT_STORE_ROW_7(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
128*c217d954SCole Faust    VSTORE(N0)                                                         \
129*c217d954SCole Faust    (CONVERT_SAT((BASENAME##7), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 7 * STRIDE_Y + Z##7));
130*c217d954SCole Faust
131*c217d954SCole Faust#define CONVERT_STORE_ROW_9(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
132*c217d954SCole Faust    CONVERT_STORE_ROW_8(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
133*c217d954SCole Faust    VSTORE(N0)                                                         \
134*c217d954SCole Faust    (CONVERT_SAT((BASENAME##8), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 8 * STRIDE_Y + Z##8));
135*c217d954SCole Faust
136*c217d954SCole Faust#define CONVERT_STORE_ROW_10(N0, DATA, BASENAME, PTR, STRIDE_Y, Z) \
137*c217d954SCole Faust    CONVERT_STORE_ROW_9(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
138*c217d954SCole Faust    VSTORE(N0)                                                     \
139*c217d954SCole Faust    (CONVERT_SAT((BASENAME##9), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 9 * STRIDE_Y + Z##9));
140*c217d954SCole Faust
141*c217d954SCole Faust#define CONVERT_STORE_ROW_11(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
142*c217d954SCole Faust    CONVERT_STORE_ROW_10(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
143*c217d954SCole Faust    VSTORE(N0)                                                          \
144*c217d954SCole Faust    (CONVERT_SAT((BASENAME##A), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 10 * STRIDE_Y + Z##A));
145*c217d954SCole Faust
146*c217d954SCole Faust#define CONVERT_STORE_ROW_12(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
147*c217d954SCole Faust    CONVERT_STORE_ROW_11(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
148*c217d954SCole Faust    VSTORE(N0)                                                          \
149*c217d954SCole Faust    (CONVERT_SAT((BASENAME##B), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 11 * STRIDE_Y + Z##B));
150*c217d954SCole Faust
151*c217d954SCole Faust#define CONVERT_STORE_ROW_13(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
152*c217d954SCole Faust    CONVERT_STORE_ROW_12(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
153*c217d954SCole Faust    VSTORE(N0)                                                          \
154*c217d954SCole Faust    (CONVERT_SAT((BASENAME##C), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 12 * STRIDE_Y + Z##C));
155*c217d954SCole Faust
156*c217d954SCole Faust#define CONVERT_STORE_ROW_14(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
157*c217d954SCole Faust    CONVERT_STORE_ROW_13(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
158*c217d954SCole Faust    VSTORE(N0)                                                          \
159*c217d954SCole Faust    (CONVERT_SAT((BASENAME##D), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 13 * STRIDE_Y + Z##D));
160*c217d954SCole Faust
161*c217d954SCole Faust#define CONVERT_STORE_ROW_15(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
162*c217d954SCole Faust    CONVERT_STORE_ROW_14(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
163*c217d954SCole Faust    VSTORE(N0)                                                          \
164*c217d954SCole Faust    (CONVERT_SAT((BASENAME##E), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 14 * STRIDE_Y + Z##E));
165*c217d954SCole Faust
166*c217d954SCole Faust#define CONVERT_STORE_ROW_16(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
167*c217d954SCole Faust    CONVERT_STORE_ROW_15(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
168*c217d954SCole Faust    VSTORE(N0)                                                          \
169*c217d954SCole Faust    (CONVERT_SAT((BASENAME##F), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 15 * STRIDE_Y + Z##F));
170*c217d954SCole Faust
171*c217d954SCole Faust
172*c217d954SCole Faust
173*c217d954SCole Faust
174*c217d954SCole Faust#define STORE_BLOCK_STR(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) STORE_ROW_##M0(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
175*c217d954SCole Faust#define STORE_BLOCK(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) STORE_BLOCK_STR(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
176*c217d954SCole Faust
177*c217d954SCole Faust
178*c217d954SCole Faust
179*c217d954SCole Faust#define CONVERT_STORE_BLOCK_STR(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) CONVERT_STORE_ROW_##M0(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
180*c217d954SCole Faust#define CONVERT_STORE_BLOCK(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) CONVERT_STORE_BLOCK_STR(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
181*c217d954SCole Faust
182*c217d954SCole Faust
183*c217d954SCole Faust
184*c217d954SCole Faust#define STORE_ROW_PARTIAL_1(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
185*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
186*c217d954SCole Faust    (BASENAME##0, 0, (__global DATA_TYPE *)(PTR + 0 * STRIDE_Y + Z##0));
187*c217d954SCole Faust
188*c217d954SCole Faust#define STORE_ROW_PARTIAL_2(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
189*c217d954SCole Faust    STORE_ROW_PARTIAL_1(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
190*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
191*c217d954SCole Faust    (BASENAME##1, 0, (__global DATA_TYPE *)(PTR + 1 * STRIDE_Y + Z##1));
192*c217d954SCole Faust
193*c217d954SCole Faust#define STORE_ROW_PARTIAL_3(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
194*c217d954SCole Faust    STORE_ROW_PARTIAL_2(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
195*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
196*c217d954SCole Faust    (BASENAME##2, 0, (__global DATA_TYPE *)(PTR + 2 * STRIDE_Y + Z##2));
197*c217d954SCole Faust
198*c217d954SCole Faust#define STORE_ROW_PARTIAL_4(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
199*c217d954SCole Faust    STORE_ROW_PARTIAL_3(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
200*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
201*c217d954SCole Faust    (BASENAME##3, 0, (__global DATA_TYPE *)(PTR + 3 * STRIDE_Y + Z##3));
202*c217d954SCole Faust
203*c217d954SCole Faust#define STORE_ROW_PARTIAL_5(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
204*c217d954SCole Faust    STORE_ROW_PARTIAL_4(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
205*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
206*c217d954SCole Faust    (BASENAME##4, 0, (__global DATA_TYPE *)(PTR + 4 * STRIDE_Y + Z##4));
207*c217d954SCole Faust
208*c217d954SCole Faust#define STORE_ROW_PARTIAL_6(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
209*c217d954SCole Faust    STORE_ROW_PARTIAL_5(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
210*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
211*c217d954SCole Faust    (BASENAME##5, 0, (__global DATA_TYPE *)(PTR + 5 * STRIDE_Y + Z##5));
212*c217d954SCole Faust
213*c217d954SCole Faust#define STORE_ROW_PARTIAL_7(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
214*c217d954SCole Faust    STORE_ROW_PARTIAL_6(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
215*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
216*c217d954SCole Faust    (BASENAME##6, 0, (__global DATA_TYPE *)(PTR + 6 * STRIDE_Y + Z##6));
217*c217d954SCole Faust
218*c217d954SCole Faust#define STORE_ROW_PARTIAL_8(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
219*c217d954SCole Faust    STORE_ROW_PARTIAL_7(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
220*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
221*c217d954SCole Faust    (BASENAME##7, 0, (__global DATA_TYPE *)(PTR + 7 * STRIDE_Y + Z##7));
222*c217d954SCole Faust
223*c217d954SCole Faust#define STORE_ROW_PARTIAL_9(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
224*c217d954SCole Faust    STORE_ROW_PARTIAL_8(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
225*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
226*c217d954SCole Faust    (BASENAME##8, 0, (__global DATA_TYPE *)(PTR + 8 * STRIDE_Y + Z##8));
227*c217d954SCole Faust
228*c217d954SCole Faust#define STORE_ROW_PARTIAL_10(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
229*c217d954SCole Faust    STORE_ROW_PARTIAL_9(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)      \
230*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
231*c217d954SCole Faust    (BASENAME##9, 0, (__global DATA_TYPE *)(PTR + 9 * STRIDE_Y + Z##9));
232*c217d954SCole Faust
233*c217d954SCole Faust#define STORE_ROW_PARTIAL_11(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
234*c217d954SCole Faust    STORE_ROW_PARTIAL_10(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
235*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
236*c217d954SCole Faust    (BASENAME##A, 0, (__global DATA_TYPE *)(PTR + 10 * STRIDE_Y + Z##A));
237*c217d954SCole Faust
238*c217d954SCole Faust#define STORE_ROW_PARTIAL_12(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
239*c217d954SCole Faust    STORE_ROW_PARTIAL_11(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
240*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
241*c217d954SCole Faust    (BASENAME##B, 0, (__global DATA_TYPE *)(PTR + 11 * STRIDE_Y + Z##B));
242*c217d954SCole Faust
243*c217d954SCole Faust#define STORE_ROW_PARTIAL_13(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
244*c217d954SCole Faust    STORE_ROW_PARTIAL_12(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
245*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
246*c217d954SCole Faust    (BASENAME##C, 0, (__global DATA_TYPE *)(PTR + 12 * STRIDE_Y + Z##C));
247*c217d954SCole Faust
248*c217d954SCole Faust#define STORE_ROW_PARTIAL_14(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
249*c217d954SCole Faust    STORE_ROW_PARTIAL_13(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
250*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
251*c217d954SCole Faust    (BASENAME##D, 0, (__global DATA_TYPE *)(PTR + 13 * STRIDE_Y + Z##D));
252*c217d954SCole Faust
253*c217d954SCole Faust#define STORE_ROW_PARTIAL_15(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
254*c217d954SCole Faust    STORE_ROW_PARTIAL_14(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
255*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
256*c217d954SCole Faust    (BASENAME##E, 0, (__global DATA_TYPE *)(PTR + 14 * STRIDE_Y + Z##E));
257*c217d954SCole Faust
258*c217d954SCole Faust#define STORE_ROW_PARTIAL_16(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
259*c217d954SCole Faust    STORE_ROW_PARTIAL_15(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
260*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
261*c217d954SCole Faust    (BASENAME##F, 0, (__global DATA_TYPE *)(PTR + 15 * STRIDE_Y + Z##F));
262*c217d954SCole Faust
263*c217d954SCole Faust
264*c217d954SCole Faust
265*c217d954SCole Faust#define STORE_BLOCK_PARTIAL_STR(STORE_M0, STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) STORE_ROW_PARTIAL_##STORE_M0(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
266*c217d954SCole Faust#define STORE_BLOCK_PARTIAL(STORE_M0, STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) STORE_BLOCK_PARTIAL_STR(STORE_M0, STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
267*c217d954SCole Faust
268*c217d954SCole Faust#define STORE_BLOCK_PARTIAL_IN_X_AND_Y(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
269*c217d954SCole Faust    if(!(PARTIAL_COND_X) && !(PARTIAL_COND_Y))                                                                                                            \
270*c217d954SCole Faust    {                                                                                                                                                     \
271*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                                                           \
272*c217d954SCole Faust    }                                                                                                                                                     \
273*c217d954SCole Faust    else if((PARTIAL_COND_Y) && !(PARTIAL_COND_X))                                                                                                        \
274*c217d954SCole Faust    {                                                                                                                                                     \
275*c217d954SCole Faust        STORE_BLOCK_PARTIAL(PARTIAL_STORE_M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                                             \
276*c217d954SCole Faust    }                                                                                                                                                     \
277*c217d954SCole Faust    else if(!(PARTIAL_COND_Y) && (PARTIAL_COND_X))                                                                                                        \
278*c217d954SCole Faust    {                                                                                                                                                     \
279*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, PARTIAL_STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                                             \
280*c217d954SCole Faust    }                                                                                                                                                     \
281*c217d954SCole Faust    else                                                                                                                                                  \
282*c217d954SCole Faust    {                                                                                                                                                     \
283*c217d954SCole Faust        STORE_BLOCK_PARTIAL(PARTIAL_STORE_M0, PARTIAL_STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                               \
284*c217d954SCole Faust    }
285*c217d954SCole Faust
286*c217d954SCole Faust#define STORE_BLOCK_PARTIAL_IN_X(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_N0, PARTIAL_COND_X) \
287*c217d954SCole Faust    if(!(PARTIAL_COND_X))                                                                                         \
288*c217d954SCole Faust    {                                                                                                             \
289*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                   \
290*c217d954SCole Faust    }                                                                                                             \
291*c217d954SCole Faust    else                                                                                                          \
292*c217d954SCole Faust    {                                                                                                             \
293*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, PARTIAL_STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                     \
294*c217d954SCole Faust    }
295*c217d954SCole Faust
296*c217d954SCole Faust#define STORE_BLOCK_PARTIAL_IN_Y(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_COND_Y) \
297*c217d954SCole Faust    if(!(PARTIAL_COND_Y))                                                                                         \
298*c217d954SCole Faust    {                                                                                                             \
299*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                   \
300*c217d954SCole Faust    }                                                                                                             \
301*c217d954SCole Faust    else                                                                                                          \
302*c217d954SCole Faust    {                                                                                                             \
303*c217d954SCole Faust        STORE_BLOCK_PARTIAL(PARTIAL_STORE_M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                     \
304*c217d954SCole Faust    }
305*c217d954SCole Faust
306*c217d954SCole Faust
307*c217d954SCole Faust#if defined(PARTIAL_STORE_M0) && defined(PARTIAL_STORE_N0)
308*c217d954SCole Faust
309*c217d954SCole Faust
310*c217d954SCole Faust#if PARTIAL_STORE_M0 == 0 && PARTIAL_STORE_N0 == 0
311*c217d954SCole Faust
312*c217d954SCole Faust#define STORE_BLOCK_BOUNDARY_AWARE(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
313*c217d954SCole Faust    STORE_BLOCK(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
314*c217d954SCole Faust
315*c217d954SCole Faust#elif PARTIAL_STORE_M0 > 0 && PARTIAL_STORE_N0 == 0
316*c217d954SCole Faust
317*c217d954SCole Faust#define STORE_BLOCK_BOUNDARY_AWARE(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
318*c217d954SCole Faust    STORE_BLOCK_PARTIAL_IN_Y(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_COND_Y)
319*c217d954SCole Faust
320*c217d954SCole Faust#elif PARTIAL_STORE_M0 == 0 && PARTIAL_STORE_N0 > 0
321*c217d954SCole Faust
322*c217d954SCole Faust#define STORE_BLOCK_BOUNDARY_AWARE(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
323*c217d954SCole Faust    STORE_BLOCK_PARTIAL_IN_X(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_N0, PARTIAL_COND_X)
324*c217d954SCole Faust
325*c217d954SCole Faust#else
326*c217d954SCole Faust
327*c217d954SCole Faust#define STORE_BLOCK_BOUNDARY_AWARE(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
328*c217d954SCole Faust    STORE_BLOCK_PARTIAL_IN_X_AND_Y(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X)
329*c217d954SCole Faust
330*c217d954SCole Faust#endif
331*c217d954SCole Faust
332*c217d954SCole Faust#endif
333*c217d954SCole Faust
334*c217d954SCole Faust
335*c217d954SCole Faust#if defined(PARTIAL_STORE_M0)
336*c217d954SCole Faust
337*c217d954SCole Faust#define COMPUTE_M0_START_ROW(y, M0, PARTIAL_STORE_M0) \
338*c217d954SCole Faust    ((uint)(max(0, (int)(y * M0) - (int)((M0 - PARTIAL_STORE_M0) % M0))))
339*c217d954SCole Faust#else
340*c217d954SCole Faust#define COMPUTE_M0_START_ROW(y, M0, PARTIAL_STORE_M0) \
341*c217d954SCole Faust    ((uint)(y * M0))
342*c217d954SCole Faust#endif
343*c217d954SCole Faust
344*c217d954SCole Faust
345*c217d954SCole Faust
346*c217d954SCole Faust#define STORE_VECTOR_SELECT(basename, data_type, ptr, vec_size, leftover, cond) \
347*c217d954SCole Faust    STORE_BLOCK_PARTIAL_IN_X(1, vec_size, data_type, basename, ptr, 0, 0, leftover, cond)
348*c217d954SCole Faust
349*c217d954SCole Faust
350*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_FP16_ENABLED) && defined(cl_khr_fp16)
351*c217d954SCole Faust#pragma OPENCL EXTENSION cl_khr_fp16 : enable
352*c217d954SCole Faust#endif
353*c217d954SCole Faust
354*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_DOT8_ENABLED) && defined(cl_arm_integer_dot_product_int8)
355*c217d954SCole Faust#pragma OPENCL EXTENSION cl_arm_integer_dot_product_int8 : enable
356*c217d954SCole Faust#endif
357*c217d954SCole Faust
358*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_DOT8_ACC_ENABLED) && defined(cl_arm_integer_dot_product_accumulate_int8)
359*c217d954SCole Faust#pragma OPENCL EXTENSION cl_arm_integer_dot_product_accumulate_int8 : enable
360*c217d954SCole Faust#endif
361*c217d954SCole Faust
362*c217d954SCole Faust#if defined(ARM_COMPUTE_DEBUG_ENABLED) && defined(cl_arm_printf)
363*c217d954SCole Faust#pragma OPENCL EXTENSION cl_arm_printf : enable
364*c217d954SCole Faust#endif
365*c217d954SCole Faust
366*c217d954SCole Faust#define GPU_ARCH_MIDGARD 0x100
367*c217d954SCole Faust#define GPU_ARCH_BIFROST 0x200
368*c217d954SCole Faust#define GPU_ARCH_VALHALL 0x300
369*c217d954SCole Faust
370*c217d954SCole Faust
371*c217d954SCole Faust#define CONCAT(a, b) a##b
372*c217d954SCole Faust
373*c217d954SCole Faust
374*c217d954SCole Faust#define EXPAND(x) x
375*c217d954SCole Faust
376*c217d954SCole Faust
377*c217d954SCole Faust#define CLAMP(x, min_val, max_val) min(max(x, min_val), max_val)
378*c217d954SCole Faust
379*c217d954SCole Faust
380*c217d954SCole Faust#define REV1(x) ((x))
381*c217d954SCole Faust#define REV2(x) ((x).s10)
382*c217d954SCole Faust#define REV3(x) ((x).s210)
383*c217d954SCole Faust#define REV4(x) ((x).s3210)
384*c217d954SCole Faust#define REV8(x) ((x).s76543210)
385*c217d954SCole Faust#define REV16(x) ((x).sFEDCBA9876543210)
386*c217d954SCole Faust
387*c217d954SCole Faust
388*c217d954SCole Faust
389*c217d954SCole Faust#define REVERSE_STR(x, s) REV##s((x))
390*c217d954SCole Faust#define REVERSE(x, s) REVERSE_STR(x, s)
391*c217d954SCole Faust
392*c217d954SCole Faust
393*c217d954SCole Faust
394*c217d954SCole Faust#define ROT1_0(x) ((x))
395*c217d954SCole Faust#define ROT1_1(x) ((x))
396*c217d954SCole Faust
397*c217d954SCole Faust#define ROT2_0(x) ((x))
398*c217d954SCole Faust#define ROT2_1(x) ((x).s10)
399*c217d954SCole Faust#define ROT2_2(x) ((x))
400*c217d954SCole Faust
401*c217d954SCole Faust#define ROT3_0(x) ((x))
402*c217d954SCole Faust#define ROT3_1(x) ((x).s201)
403*c217d954SCole Faust#define ROT3_2(x) ((x).s120)
404*c217d954SCole Faust#define ROT3_3(x) ((x))
405*c217d954SCole Faust
406*c217d954SCole Faust#define ROT4_0(x) ((x))
407*c217d954SCole Faust#define ROT4_1(x) ((x).s3012)
408*c217d954SCole Faust#define ROT4_2(x) ((x).s2301)
409*c217d954SCole Faust#define ROT4_3(x) ((x).s1230)
410*c217d954SCole Faust#define ROT4_4(x) ((x))
411*c217d954SCole Faust
412*c217d954SCole Faust#define ROT8_0(x) ((x))
413*c217d954SCole Faust#define ROT8_1(x) ((x).s70123456)
414*c217d954SCole Faust#define ROT8_2(x) ((x).s67012345)
415*c217d954SCole Faust#define ROT8_3(x) ((x).s56701234)
416*c217d954SCole Faust#define ROT8_4(x) ((x).s45670123)
417*c217d954SCole Faust#define ROT8_5(x) ((x).s34567012)
418*c217d954SCole Faust#define ROT8_6(x) ((x).s23456701)
419*c217d954SCole Faust#define ROT8_7(x) ((x).s12345670)
420*c217d954SCole Faust#define ROT8_8(x) ((x))
421*c217d954SCole Faust
422*c217d954SCole Faust#define ROT16_0(x) ((x))
423*c217d954SCole Faust#define ROT16_1(x) ((x).sF0123456789ABCDE)
424*c217d954SCole Faust#define ROT16_2(x) ((x).sEF0123456789ABCD)
425*c217d954SCole Faust#define ROT16_3(x) ((x).sDEF0123456789ABC)
426*c217d954SCole Faust#define ROT16_4(x) ((x).sCDEF0123456789AB)
427*c217d954SCole Faust#define ROT16_5(x) ((x).sBCDEF0123456789A)
428*c217d954SCole Faust#define ROT16_6(x) ((x).sABCDEF0123456789)
429*c217d954SCole Faust#define ROT16_7(x) ((x).s9ABCDEF012345678)
430*c217d954SCole Faust#define ROT16_8(x) ((x).s89ABCDEF01234567)
431*c217d954SCole Faust#define ROT16_9(x) ((x).s789ABCDEF0123456)
432*c217d954SCole Faust#define ROT16_10(x) ((x).s6789ABCDEF012345)
433*c217d954SCole Faust#define ROT16_11(x) ((x).s56789ABCDEF01234)
434*c217d954SCole Faust#define ROT16_12(x) ((x).s456789ABCDEF0123)
435*c217d954SCole Faust#define ROT16_13(x) ((x).s3456789ABCDEF012)
436*c217d954SCole Faust#define ROT16_14(x) ((x).s23456789ABCDEF01)
437*c217d954SCole Faust#define ROT16_15(x) ((x).s123456789ABCDEF0)
438*c217d954SCole Faust#define ROT16_16(x) ((x))
439*c217d954SCole Faust
440*c217d954SCole Faust
441*c217d954SCole Faust
442*c217d954SCole Faust#define ROTATE_STR(x, s, n) ROT##s##_##n(x)
443*c217d954SCole Faust#define ROTATE(x, s, n) ROTATE_STR(x, s, n)
444*c217d954SCole Faust
445*c217d954SCole Faust
446*c217d954SCole Faust
447*c217d954SCole Faust#define V_OFFS1(dt) (dt##1)(0)
448*c217d954SCole Faust#define V_OFFS2(dt) (dt##2)(0, 1)
449*c217d954SCole Faust#define V_OFFS3(dt) (dt##3)(0, 1, 2)
450*c217d954SCole Faust#define V_OFFS4(dt) (dt##4)(0, 1, 2, 3)
451*c217d954SCole Faust#define V_OFFS8(dt) (dt##8)(0, 1, 2, 3, 4, 5, 6, 7)
452*c217d954SCole Faust#define V_OFFS16(dt) (dt##16)(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15)
453*c217d954SCole Faust
454*c217d954SCole Faust
455*c217d954SCole Faust
456*c217d954SCole Faust#define VEC_OFFS_STR(dt, s) V_OFFS##s(dt)
457*c217d954SCole Faust#define VEC_OFFS(dt, s) VEC_OFFS_STR(dt, s)
458*c217d954SCole Faust
459*c217d954SCole Faust
460*c217d954SCole Faust#define VLOAD_STR(size) vload##size
461*c217d954SCole Faust#define VLOAD(size) VLOAD_STR(size)
462*c217d954SCole Faust
463*c217d954SCole Faust
464*c217d954SCole Faust#define VLOAD_PARTIAL_STR(size, load_size) vload_partial_##size##_##load_size
465*c217d954SCole Faust#define VLOAD_PARTIAL(size, load_size) VLOAD_PARTIAL_STR(size, load_size)
466*c217d954SCole Faust
467*c217d954SCole Faust#define NO_LOAD(data, offs, ptr) \
468*c217d954SCole Faust    {                            \
469*c217d954SCole Faust    }
470*c217d954SCole Faust
471*c217d954SCole Faust
472*c217d954SCole Faust#define vload_partial_1_0 NO_LOAD
473*c217d954SCole Faust#define vload_partial_1_1 vload1
474*c217d954SCole Faust#define vload_partial_1_2 NO_LOAD
475*c217d954SCole Faust#define vload_partial_1_3 NO_LOAD
476*c217d954SCole Faust#define vload_partial_1_4 NO_LOAD
477*c217d954SCole Faust#define vload_partial_1_5 NO_LOAD
478*c217d954SCole Faust#define vload_partial_1_6 NO_LOAD
479*c217d954SCole Faust#define vload_partial_1_7 NO_LOAD
480*c217d954SCole Faust#define vload_partial_1_8 NO_LOAD
481*c217d954SCole Faust#define vload_partial_1_9 NO_LOAD
482*c217d954SCole Faust#define vload_partial_1_10 NO_LOAD
483*c217d954SCole Faust#define vload_partial_1_11 NO_LOAD
484*c217d954SCole Faust#define vload_partial_1_12 NO_LOAD
485*c217d954SCole Faust#define vload_partial_1_13 NO_LOAD
486*c217d954SCole Faust#define vload_partial_1_14 NO_LOAD
487*c217d954SCole Faust#define vload_partial_1_15 NO_LOAD
488*c217d954SCole Faust#define vload_partial_1_16 NO_LOAD
489*c217d954SCole Faust
490*c217d954SCole Faust#define vload_partial_2_0 NO_LOAD
491*c217d954SCole Faust#define vload_partial_2_1 vload_partial_1
492*c217d954SCole Faust#define vload_partial_2_2 vload_partial_2
493*c217d954SCole Faust#define vload_partial_2_3 NO_LOAD
494*c217d954SCole Faust#define vload_partial_2_4 NO_LOAD
495*c217d954SCole Faust#define vload_partial_2_5 NO_LOAD
496*c217d954SCole Faust#define vload_partial_2_6 NO_LOAD
497*c217d954SCole Faust#define vload_partial_2_7 NO_LOAD
498*c217d954SCole Faust#define vload_partial_2_8 NO_LOAD
499*c217d954SCole Faust#define vload_partial_2_9 NO_LOAD
500*c217d954SCole Faust#define vload_partial_2_10 NO_LOAD
501*c217d954SCole Faust#define vload_partial_2_11 NO_LOAD
502*c217d954SCole Faust#define vload_partial_2_12 NO_LOAD
503*c217d954SCole Faust#define vload_partial_2_13 NO_LOAD
504*c217d954SCole Faust#define vload_partial_2_14 NO_LOAD
505*c217d954SCole Faust#define vload_partial_2_15 NO_LOAD
506*c217d954SCole Faust#define vload_partial_2_16 NO_LOAD
507*c217d954SCole Faust
508*c217d954SCole Faust#define vload_partial_3_0 NO_LOAD
509*c217d954SCole Faust#define vload_partial_3_1 vload_partial_1
510*c217d954SCole Faust#define vload_partial_3_2 vload_partial_2
511*c217d954SCole Faust#define vload_partial_3_3 vload_partial_3
512*c217d954SCole Faust#define vload_partial_3_4 NO_LOAD
513*c217d954SCole Faust#define vload_partial_3_5 NO_LOAD
514*c217d954SCole Faust#define vload_partial_3_6 NO_LOAD
515*c217d954SCole Faust#define vload_partial_3_7 NO_LOAD
516*c217d954SCole Faust#define vload_partial_3_8 NO_LOAD
517*c217d954SCole Faust#define vload_partial_3_9 NO_LOAD
518*c217d954SCole Faust#define vload_partial_3_10 NO_LOAD
519*c217d954SCole Faust#define vload_partial_3_11 NO_LOAD
520*c217d954SCole Faust#define vload_partial_3_12 NO_LOAD
521*c217d954SCole Faust#define vload_partial_3_13 NO_LOAD
522*c217d954SCole Faust#define vload_partial_3_14 NO_LOAD
523*c217d954SCole Faust#define vload_partial_3_15 NO_LOAD
524*c217d954SCole Faust#define vload_partial_3_16 NO_LOAD
525*c217d954SCole Faust
526*c217d954SCole Faust#define vload_partial_4_0 NO_LOAD
527*c217d954SCole Faust#define vload_partial_4_1 vload_partial_1
528*c217d954SCole Faust#define vload_partial_4_2 vload_partial_2
529*c217d954SCole Faust#define vload_partial_4_3 vload_partial_3
530*c217d954SCole Faust#define vload_partial_4_4 vload_partial_4
531*c217d954SCole Faust#define vload_partial_4_5 NO_LOAD
532*c217d954SCole Faust#define vload_partial_4_6 NO_LOAD
533*c217d954SCole Faust#define vload_partial_4_7 NO_LOAD
534*c217d954SCole Faust#define vload_partial_4_8 NO_LOAD
535*c217d954SCole Faust#define vload_partial_4_9 NO_LOAD
536*c217d954SCole Faust#define vload_partial_4_10 NO_LOAD
537*c217d954SCole Faust#define vload_partial_4_11 NO_LOAD
538*c217d954SCole Faust#define vload_partial_4_12 NO_LOAD
539*c217d954SCole Faust#define vload_partial_4_13 NO_LOAD
540*c217d954SCole Faust#define vload_partial_4_14 NO_LOAD
541*c217d954SCole Faust#define vload_partial_4_15 NO_LOAD
542*c217d954SCole Faust#define vload_partial_4_16 NO_LOAD
543*c217d954SCole Faust
544*c217d954SCole Faust#define vload_partial_8_0 NO_LOAD
545*c217d954SCole Faust#define vload_partial_8_1 vload_partial_1
546*c217d954SCole Faust#define vload_partial_8_2 vload_partial_2
547*c217d954SCole Faust#define vload_partial_8_3 vload_partial_3
548*c217d954SCole Faust#define vload_partial_8_4 vload_partial_4
549*c217d954SCole Faust#define vload_partial_8_5 vload_partial_5
550*c217d954SCole Faust#define vload_partial_8_6 vload_partial_6
551*c217d954SCole Faust#define vload_partial_8_7 vload_partial_7
552*c217d954SCole Faust#define vload_partial_8_8 vload_partial_8
553*c217d954SCole Faust#define vload_partial_8_9 NO_LOAD
554*c217d954SCole Faust#define vload_partial_8_10 NO_LOAD
555*c217d954SCole Faust#define vload_partial_8_11 NO_LOAD
556*c217d954SCole Faust#define vload_partial_8_12 NO_LOAD
557*c217d954SCole Faust#define vload_partial_8_13 NO_LOAD
558*c217d954SCole Faust#define vload_partial_8_14 NO_LOAD
559*c217d954SCole Faust#define vload_partial_8_15 NO_LOAD
560*c217d954SCole Faust#define vload_partial_8_16 NO_LOAD
561*c217d954SCole Faust
562*c217d954SCole Faust#define vload_partial_16_0 NO_LOAD
563*c217d954SCole Faust#define vload_partial_16_1 vload_partial_1
564*c217d954SCole Faust#define vload_partial_16_2 vload_partial_2
565*c217d954SCole Faust#define vload_partial_16_3 vload_partial_3
566*c217d954SCole Faust#define vload_partial_16_4 vload_partial_4
567*c217d954SCole Faust#define vload_partial_16_5 vload_partial_5
568*c217d954SCole Faust#define vload_partial_16_6 vload_partial_6
569*c217d954SCole Faust#define vload_partial_16_7 vload_partial_7
570*c217d954SCole Faust#define vload_partial_16_8 vload_partial_8
571*c217d954SCole Faust#define vload_partial_16_9 vload_partial_9
572*c217d954SCole Faust#define vload_partial_16_10 vload_partial_10
573*c217d954SCole Faust#define vload_partial_16_11 vload_partial_11
574*c217d954SCole Faust#define vload_partial_16_12 vload_partial_12
575*c217d954SCole Faust#define vload_partial_16_13 vload_partial_13
576*c217d954SCole Faust#define vload_partial_16_14 vload_partial_14
577*c217d954SCole Faust#define vload_partial_16_15 vload_partial_15
578*c217d954SCole Faust#define vload_partial_16_16 vload_partial_16
579*c217d954SCole Faust
580*c217d954SCole Faust
581*c217d954SCole Faust#define vload_partial_1(DATA, OFFSET, PTR) \
582*c217d954SCole Faust    DATA.s0 = vload1(OFFSET, PTR);
583*c217d954SCole Faust
584*c217d954SCole Faust#define vload_partial_2(DATA, OFFSET, PTR) \
585*c217d954SCole Faust    DATA.s01 = vload2(OFFSET, PTR);
586*c217d954SCole Faust
587*c217d954SCole Faust#define vload_partial_3(DATA, OFFSET, PTR) \
588*c217d954SCole Faust    DATA.s012 = vload3(OFFSET, PTR);
589*c217d954SCole Faust
590*c217d954SCole Faust#define vload_partial_4(DATA, OFFSET, PTR) \
591*c217d954SCole Faust    DATA.s0123 = vload4(OFFSET, PTR);
592*c217d954SCole Faust
593*c217d954SCole Faust#define vload_partial_5(DATA, OFFSET, PTR)    \
594*c217d954SCole Faust    vload_partial_4(DATA.s0123, OFFSET, PTR); \
595*c217d954SCole Faust    DATA.s4 = vload1(OFFSET, PTR + 4);
596*c217d954SCole Faust
597*c217d954SCole Faust#define vload_partial_6(DATA, OFFSET, PTR)    \
598*c217d954SCole Faust    vload_partial_4(DATA.s0123, OFFSET, PTR); \
599*c217d954SCole Faust    vload_partial_2(DATA.s45, OFFSET, PTR + 4);
600*c217d954SCole Faust
601*c217d954SCole Faust#define vload_partial_7(DATA, OFFSET, PTR)    \
602*c217d954SCole Faust    vload_partial_4(DATA.s0123, OFFSET, PTR); \
603*c217d954SCole Faust    vload_partial_3(DATA.s456, OFFSET, PTR + 4);
604*c217d954SCole Faust
605*c217d954SCole Faust#define vload_partial_8(DATA, OFFSET, PTR) \
606*c217d954SCole Faust    DATA.s01234567 = vload8(OFFSET, PTR);
607*c217d954SCole Faust
608*c217d954SCole Faust#define vload_partial_9(DATA, OFFSET, PTR)        \
609*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
610*c217d954SCole Faust    DATA.s8 = vload1(OFFSET, PTR + 8);
611*c217d954SCole Faust
612*c217d954SCole Faust#define vload_partial_10(DATA, OFFSET, PTR)       \
613*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
614*c217d954SCole Faust    vload_partial_2(DATA.s89, OFFSET, PTR + 8);
615*c217d954SCole Faust
616*c217d954SCole Faust#define vload_partial_11(DATA, OFFSET, PTR)       \
617*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
618*c217d954SCole Faust    vload_partial_3(DATA.s89A, OFFSET, PTR + 8);
619*c217d954SCole Faust
620*c217d954SCole Faust#define vload_partial_12(DATA, OFFSET, PTR)       \
621*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
622*c217d954SCole Faust    vload_partial_4(DATA.s89AB, OFFSET, PTR + 8);
623*c217d954SCole Faust
624*c217d954SCole Faust#define vload_partial_13(DATA, OFFSET, PTR)       \
625*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
626*c217d954SCole Faust    vload_partial_5(DATA.s89ABCDEF, OFFSET, PTR + 8);
627*c217d954SCole Faust
628*c217d954SCole Faust#define vload_partial_14(DATA, OFFSET, PTR)       \
629*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
630*c217d954SCole Faust    vload_partial_6(DATA.s89ABCDEF, OFFSET, PTR + 8);
631*c217d954SCole Faust
632*c217d954SCole Faust#define vload_partial_15(DATA, OFFSET, PTR)       \
633*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
634*c217d954SCole Faust    vload_partial_7(DATA.s89ABCDEF, OFFSET, PTR + 8);
635*c217d954SCole Faust
636*c217d954SCole Faust#define vload_partial_16(DATA, OFFSET, PTR) \
637*c217d954SCole Faust    DATA = vload16(OFFSET, PTR);
638*c217d954SCole Faust
639*c217d954SCole Faust
640*c217d954SCole Faust
641*c217d954SCole Faust#define PIXEL_UNIT4 1
642*c217d954SCole Faust#define PIXEL_UNIT8 2
643*c217d954SCole Faust#define PIXEL_UNIT16 4
644*c217d954SCole Faust
645*c217d954SCole Faust
646*c217d954SCole Faust#define CONVERT_VECTOR_SIZE_TO_PIXEL_UNIT_STR(vec_size) PIXEL_UNIT##vec_size
647*c217d954SCole Faust#define CONVERT_VECTOR_SIZE_TO_PIXEL_UNIT(vec_size) CONVERT_VECTOR_SIZE_TO_PIXEL_UNIT_STR(vec_size)
648*c217d954SCole Faust
649*c217d954SCole Faust
650*c217d954SCole Faust#define read_image2d_floatx1(img, x_coord, y_coord) (float4)(read_imagef(img, (int2)(x_coord, y_coord)));
651*c217d954SCole Faust#define read_image2d_floatx2(img, x_coord, y_coord) (float8)(read_imagef(img, (int2)(x_coord, y_coord)), read_imagef(img, (int2)(x_coord + 1, y_coord)));
652*c217d954SCole Faust#define read_image2d_floatx4(img, x_coord, y_coord) (float16)(read_imagef(img, (int2)(x_coord, y_coord)), read_imagef(img, (int2)(x_coord + 1, y_coord)), read_imagef(img, (int2)(x_coord + 2, y_coord)), read_imagef(img, (int2)(x_coord + 3, y_coord)));
653*c217d954SCole Faust
654*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_FP16_ENABLED) && defined(cl_khr_fp16)
655*c217d954SCole Faust#define read_image2d_halfx1(img, x_coord, y_coord) (half4)(read_imageh(img, (int2)(x_coord, y_coord)));
656*c217d954SCole Faust#define read_image2d_halfx2(img, x_coord, y_coord) (half8)(read_imageh(img, (int2)(x_coord, y_coord)), read_imageh(img, (int2)(x_coord + 1, y_coord)));
657*c217d954SCole Faust#define read_image2d_halfx4(img, x_coord, y_coord) (half16)(read_imageh(img, (int2)(x_coord, y_coord)), read_imageh(img, (int2)(x_coord + 1, y_coord)), read_imageh(img, (int2)(x_coord + 2, y_coord)), read_imageh(img, (int2)(x_coord + 3, y_coord)));
658*c217d954SCole Faust#endif
659*c217d954SCole Faust
660*c217d954SCole Faust#define write_image2d_floatx1(img, x_coord, y_coord, values) (write_imagef(img, (int2)(x_coord, y_coord), values));
661*c217d954SCole Faust#define write_image2d_floatx2(img, x_coord, y_coord, values) (write_imagef(img, (int2)(x_coord, y_coord), values.s0123), write_imagef(img, (int2)(x_coord + 1, y_coord), values.s4567));
662*c217d954SCole Faust#define write_image2d_floatx4(img, x_coord, y_coord, values) (write_imagef(img, (int2)(x_coord, y_coord), values.s0123), write_imagef(img, (int2)(x_coord + 1, y_coord), values.s4567), write_imagef(img, (int2)(x_coord + 2, y_coord), values.s89AB), write_imagef(img, (int2)(x_coord + 3, y_coord), values.sCDEF));
663*c217d954SCole Faust
664*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_FP16_ENABLED) && defined(cl_khr_fp16)
665*c217d954SCole Faust#define write_image2d_halfx1(img, x_coord, y_coord, values) (write_imageh(img, (int2)(x_coord, y_coord), values));
666*c217d954SCole Faust#define write_image2d_halfx2(img, x_coord, y_coord, values) (write_imageh(img, (int2)(x_coord, y_coord), values.s0123), write_imageh(img, (int2)(x_coord + 1, y_coord), values.s4567));
667*c217d954SCole Faust#define write_image2d_halfx4(img, x_coord, y_coord, values) (write_imageh(img, (int2)(x_coord, y_coord), values.s0123), write_imageh(img, (int2)(x_coord + 1, y_coord), values.s4567), write_imageh(img, (int2)(x_coord + 2, y_coord), values.s89AB), write_imageh(img, (int2)(x_coord + 3, y_coord), values.sCDEF));
668*c217d954SCole Faust#endif
669*c217d954SCole Faust
670*c217d954SCole Faust
671*c217d954SCole Faust#define READ_IMAGE2D_STR(data_type, n0, img, x_coord, y_coord) read_image2d_##data_type##x##n0(img, x_coord, y_coord)
672*c217d954SCole Faust#define READ_IMAGE2D(data_type, n0, img, x_coord, y_coord) READ_IMAGE2D_STR(data_type, n0, img, x_coord, y_coord)
673*c217d954SCole Faust
674*c217d954SCole Faust
675*c217d954SCole Faust#define WRITE_IMAGE2D_STR(data_type, n0, img, x_coord, y_coord, values) write_image2d_##data_type##x##n0(img, x_coord, y_coord, values)
676*c217d954SCole Faust#define WRITE_IMAGE2D(data_type, n0, img, x_coord, y_coord, values) WRITE_IMAGE2D_STR(data_type, n0, img, x_coord, y_coord, values)
677*c217d954SCole Faust
678*c217d954SCole Faust#define VSTORE_STR(size) vstore##size
679*c217d954SCole Faust#define VSTORE(size) VSTORE_STR(size)
680*c217d954SCole Faust
681*c217d954SCole Faust#define float1 float
682*c217d954SCole Faust#define half1 half
683*c217d954SCole Faust#define char1 char
684*c217d954SCole Faust#define uchar1 uchar
685*c217d954SCole Faust#define short1 short
686*c217d954SCole Faust#define ushort1 ushort
687*c217d954SCole Faust#define int1 int
688*c217d954SCole Faust#define uint1 uint
689*c217d954SCole Faust#define long1 long
690*c217d954SCole Faust#define ulong1 ulong
691*c217d954SCole Faust#define double1 double
692*c217d954SCole Faust
693*c217d954SCole Faust#define vload1(OFFSET, PTR) *(OFFSET + PTR)
694*c217d954SCole Faust#define vstore1(DATA, OFFSET, PTR) *(OFFSET + PTR) = DATA
695*c217d954SCole Faust
696*c217d954SCole Faust
697*c217d954SCole Faust#define VSTORE_PARTIAL_STR(size, store_size) vstore_partial_##size##_##store_size
698*c217d954SCole Faust#define VSTORE_PARTIAL(size, store_size) VSTORE_PARTIAL_STR(size, store_size)
699*c217d954SCole Faust
700*c217d954SCole Faust#define NO_STORE(data, offs, ptr) \
701*c217d954SCole Faust    {                             \
702*c217d954SCole Faust    }
703*c217d954SCole Faust
704*c217d954SCole Faust
705*c217d954SCole Faust#define vstore_partial_1_0 NO_STORE
706*c217d954SCole Faust#define vstore_partial_1_1 vstore1
707*c217d954SCole Faust#define vstore_partial_1_2 NO_STORE
708*c217d954SCole Faust#define vstore_partial_1_3 NO_STORE
709*c217d954SCole Faust#define vstore_partial_1_4 NO_STORE
710*c217d954SCole Faust#define vstore_partial_1_5 NO_STORE
711*c217d954SCole Faust#define vstore_partial_1_6 NO_STORE
712*c217d954SCole Faust#define vstore_partial_1_7 NO_STORE
713*c217d954SCole Faust#define vstore_partial_1_8 NO_STORE
714*c217d954SCole Faust#define vstore_partial_1_9 NO_STORE
715*c217d954SCole Faust#define vstore_partial_1_10 NO_STORE
716*c217d954SCole Faust#define vstore_partial_1_11 NO_STORE
717*c217d954SCole Faust#define vstore_partial_1_12 NO_STORE
718*c217d954SCole Faust#define vstore_partial_1_13 NO_STORE
719*c217d954SCole Faust#define vstore_partial_1_14 NO_STORE
720*c217d954SCole Faust#define vstore_partial_1_15 NO_STORE
721*c217d954SCole Faust#define vstore_partial_1_16 NO_STORE
722*c217d954SCole Faust
723*c217d954SCole Faust#define vstore_partial_2_0 NO_STORE
724*c217d954SCole Faust#define vstore_partial_2_1 vstore_partial_1
725*c217d954SCole Faust#define vstore_partial_2_2 vstore_partial_2
726*c217d954SCole Faust#define vstore_partial_2_3 NO_STORE
727*c217d954SCole Faust#define vstore_partial_2_4 NO_STORE
728*c217d954SCole Faust#define vstore_partial_2_5 NO_STORE
729*c217d954SCole Faust#define vstore_partial_2_6 NO_STORE
730*c217d954SCole Faust#define vstore_partial_2_7 NO_STORE
731*c217d954SCole Faust#define vstore_partial_2_8 NO_STORE
732*c217d954SCole Faust#define vstore_partial_2_9 NO_STORE
733*c217d954SCole Faust#define vstore_partial_2_10 NO_STORE
734*c217d954SCole Faust#define vstore_partial_2_11 NO_STORE
735*c217d954SCole Faust#define vstore_partial_2_12 NO_STORE
736*c217d954SCole Faust#define vstore_partial_2_13 NO_STORE
737*c217d954SCole Faust#define vstore_partial_2_14 NO_STORE
738*c217d954SCole Faust#define vstore_partial_2_15 NO_STORE
739*c217d954SCole Faust#define vstore_partial_2_16 NO_STORE
740*c217d954SCole Faust
741*c217d954SCole Faust#define vstore_partial_3_0 NO_STORE
742*c217d954SCole Faust#define vstore_partial_3_1 vstore_partial_1
743*c217d954SCole Faust#define vstore_partial_3_2 vstore_partial_2
744*c217d954SCole Faust#define vstore_partial_3_3 vstore_partial_3
745*c217d954SCole Faust#define vstore_partial_3_4 NO_STORE
746*c217d954SCole Faust#define vstore_partial_3_5 NO_STORE
747*c217d954SCole Faust#define vstore_partial_3_6 NO_STORE
748*c217d954SCole Faust#define vstore_partial_3_7 NO_STORE
749*c217d954SCole Faust#define vstore_partial_3_8 NO_STORE
750*c217d954SCole Faust#define vstore_partial_3_9 NO_STORE
751*c217d954SCole Faust#define vstore_partial_3_10 NO_STORE
752*c217d954SCole Faust#define vstore_partial_3_11 NO_STORE
753*c217d954SCole Faust#define vstore_partial_3_12 NO_STORE
754*c217d954SCole Faust#define vstore_partial_3_13 NO_STORE
755*c217d954SCole Faust#define vstore_partial_3_14 NO_STORE
756*c217d954SCole Faust#define vstore_partial_3_15 NO_STORE
757*c217d954SCole Faust#define vstore_partial_3_16 NO_STORE
758*c217d954SCole Faust
759*c217d954SCole Faust#define vstore_partial_4_0 NO_STORE
760*c217d954SCole Faust#define vstore_partial_4_1 vstore_partial_1
761*c217d954SCole Faust#define vstore_partial_4_2 vstore_partial_2
762*c217d954SCole Faust#define vstore_partial_4_3 vstore_partial_3
763*c217d954SCole Faust#define vstore_partial_4_4 vstore_partial_4
764*c217d954SCole Faust#define vstore_partial_4_5 NO_STORE
765*c217d954SCole Faust#define vstore_partial_4_6 NO_STORE
766*c217d954SCole Faust#define vstore_partial_4_7 NO_STORE
767*c217d954SCole Faust#define vstore_partial_4_8 NO_STORE
768*c217d954SCole Faust#define vstore_partial_4_9 NO_STORE
769*c217d954SCole Faust#define vstore_partial_4_10 NO_STORE
770*c217d954SCole Faust#define vstore_partial_4_11 NO_STORE
771*c217d954SCole Faust#define vstore_partial_4_12 NO_STORE
772*c217d954SCole Faust#define vstore_partial_4_13 NO_STORE
773*c217d954SCole Faust#define vstore_partial_4_14 NO_STORE
774*c217d954SCole Faust#define vstore_partial_4_15 NO_STORE
775*c217d954SCole Faust#define vstore_partial_4_16 NO_STORE
776*c217d954SCole Faust
777*c217d954SCole Faust#define vstore_partial_8_0 NO_STORE
778*c217d954SCole Faust#define vstore_partial_8_1 vstore_partial_1
779*c217d954SCole Faust#define vstore_partial_8_2 vstore_partial_2
780*c217d954SCole Faust#define vstore_partial_8_3 vstore_partial_3
781*c217d954SCole Faust#define vstore_partial_8_4 vstore_partial_4
782*c217d954SCole Faust#define vstore_partial_8_5 vstore_partial_5
783*c217d954SCole Faust#define vstore_partial_8_6 vstore_partial_6
784*c217d954SCole Faust#define vstore_partial_8_7 vstore_partial_7
785*c217d954SCole Faust#define vstore_partial_8_8 vstore_partial_8
786*c217d954SCole Faust#define vstore_partial_8_9 NO_STORE
787*c217d954SCole Faust#define vstore_partial_8_10 NO_STORE
788*c217d954SCole Faust#define vstore_partial_8_11 NO_STORE
789*c217d954SCole Faust#define vstore_partial_8_12 NO_STORE
790*c217d954SCole Faust#define vstore_partial_8_13 NO_STORE
791*c217d954SCole Faust#define vstore_partial_8_14 NO_STORE
792*c217d954SCole Faust#define vstore_partial_8_15 NO_STORE
793*c217d954SCole Faust#define vstore_partial_8_16 NO_STORE
794*c217d954SCole Faust
795*c217d954SCole Faust#define vstore_partial_16_0 NO_STORE
796*c217d954SCole Faust#define vstore_partial_16_1 vstore_partial_1
797*c217d954SCole Faust#define vstore_partial_16_2 vstore_partial_2
798*c217d954SCole Faust#define vstore_partial_16_3 vstore_partial_3
799*c217d954SCole Faust#define vstore_partial_16_4 vstore_partial_4
800*c217d954SCole Faust#define vstore_partial_16_5 vstore_partial_5
801*c217d954SCole Faust#define vstore_partial_16_6 vstore_partial_6
802*c217d954SCole Faust#define vstore_partial_16_7 vstore_partial_7
803*c217d954SCole Faust#define vstore_partial_16_8 vstore_partial_8
804*c217d954SCole Faust#define vstore_partial_16_9 vstore_partial_9
805*c217d954SCole Faust#define vstore_partial_16_10 vstore_partial_10
806*c217d954SCole Faust#define vstore_partial_16_11 vstore_partial_11
807*c217d954SCole Faust#define vstore_partial_16_12 vstore_partial_12
808*c217d954SCole Faust#define vstore_partial_16_13 vstore_partial_13
809*c217d954SCole Faust#define vstore_partial_16_14 vstore_partial_14
810*c217d954SCole Faust#define vstore_partial_16_15 vstore_partial_15
811*c217d954SCole Faust#define vstore_partial_16_16 vstore_partial_16
812*c217d954SCole Faust
813*c217d954SCole Faust
814*c217d954SCole Faust#define vstore_partial_1(DATA, OFFSET, PTR) \
815*c217d954SCole Faust    vstore1(DATA.s0, OFFSET, PTR);
816*c217d954SCole Faust
817*c217d954SCole Faust#define vstore_partial_2(DATA, OFFSET, PTR) \
818*c217d954SCole Faust    vstore2(DATA.s01, OFFSET, PTR);
819*c217d954SCole Faust
820*c217d954SCole Faust#define vstore_partial_3(DATA, OFFSET, PTR) \
821*c217d954SCole Faust    vstore3(DATA.s012, OFFSET, PTR);
822*c217d954SCole Faust
823*c217d954SCole Faust#define vstore_partial_4(DATA, OFFSET, PTR) \
824*c217d954SCole Faust    vstore4(DATA.s0123, OFFSET, PTR);
825*c217d954SCole Faust
826*c217d954SCole Faust#define vstore_partial_5(DATA, OFFSET, PTR)    \
827*c217d954SCole Faust    vstore_partial_4(DATA.s0123, OFFSET, PTR); \
828*c217d954SCole Faust    vstore1(DATA.s4, OFFSET, PTR + 4);
829*c217d954SCole Faust
830*c217d954SCole Faust#define vstore_partial_6(DATA, OFFSET, PTR)    \
831*c217d954SCole Faust    vstore_partial_4(DATA.s0123, OFFSET, PTR); \
832*c217d954SCole Faust    vstore_partial_2(DATA.s45, OFFSET, PTR + 4);
833*c217d954SCole Faust
834*c217d954SCole Faust#define vstore_partial_7(DATA, OFFSET, PTR)    \
835*c217d954SCole Faust    vstore_partial_4(DATA.s0123, OFFSET, PTR); \
836*c217d954SCole Faust    vstore_partial_3(DATA.s456, OFFSET, PTR + 4);
837*c217d954SCole Faust
838*c217d954SCole Faust#define vstore_partial_8(DATA, OFFSET, PTR) \
839*c217d954SCole Faust    vstore8(DATA.s01234567, OFFSET, PTR);
840*c217d954SCole Faust
841*c217d954SCole Faust#define vstore_partial_9(DATA, OFFSET, PTR)        \
842*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
843*c217d954SCole Faust    vstore1(DATA.s8, OFFSET, PTR + 8);
844*c217d954SCole Faust
845*c217d954SCole Faust#define vstore_partial_10(DATA, OFFSET, PTR)       \
846*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
847*c217d954SCole Faust    vstore_partial_2(DATA.s89, OFFSET, PTR + 8);
848*c217d954SCole Faust
849*c217d954SCole Faust#define vstore_partial_11(DATA, OFFSET, PTR)       \
850*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
851*c217d954SCole Faust    vstore_partial_3(DATA.s89a, OFFSET, PTR + 8);
852*c217d954SCole Faust
853*c217d954SCole Faust#define vstore_partial_12(DATA, OFFSET, PTR)       \
854*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
855*c217d954SCole Faust    vstore_partial_4(DATA.s89ab, OFFSET, PTR + 8);
856*c217d954SCole Faust
857*c217d954SCole Faust#define vstore_partial_13(DATA, OFFSET, PTR)       \
858*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
859*c217d954SCole Faust    vstore_partial_5(DATA.s89abcdef, OFFSET, PTR + 8);
860*c217d954SCole Faust
861*c217d954SCole Faust#define vstore_partial_14(DATA, OFFSET, PTR)       \
862*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
863*c217d954SCole Faust    vstore_partial_6(DATA.s89abcdef, OFFSET, PTR + 8);
864*c217d954SCole Faust
865*c217d954SCole Faust#define vstore_partial_15(DATA, OFFSET, PTR)       \
866*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
867*c217d954SCole Faust    vstore_partial_7(DATA.s89abcdef, OFFSET, PTR + 8);
868*c217d954SCole Faust
869*c217d954SCole Faust#define vstore_partial_16(DATA, OFFSET, PTR) \
870*c217d954SCole Faust    vstore16(DATA, OFFSET, PTR);
871*c217d954SCole Faust
872*c217d954SCole Faust
873*c217d954SCole Faust
874*c217d954SCole Faust
875*c217d954SCole Faust
876*c217d954SCole Faust#define convert_float_sat convert_float
877*c217d954SCole Faust#define convert_float1_sat convert_float
878*c217d954SCole Faust#define convert_float2_sat convert_float2
879*c217d954SCole Faust#define convert_float3_sat convert_float3
880*c217d954SCole Faust#define convert_float4_sat convert_float4
881*c217d954SCole Faust#define convert_float8_sat convert_float8
882*c217d954SCole Faust#define convert_float16_sat convert_float16
883*c217d954SCole Faust#define convert_half_sat convert_float
884*c217d954SCole Faust#define convert_half1_sat convert_half
885*c217d954SCole Faust#define convert_half2_sat convert_half2
886*c217d954SCole Faust#define convert_half3_sat convert_half3
887*c217d954SCole Faust#define convert_half4_sat convert_half4
888*c217d954SCole Faust#define convert_half8_sat convert_half8
889*c217d954SCole Faust#define convert_half16_sat convert_half16
890*c217d954SCole Faust
891*c217d954SCole Faust#define convert_float1 convert_float
892*c217d954SCole Faust#define convert_half1 convert_half
893*c217d954SCole Faust#define convert_char1 convert_char
894*c217d954SCole Faust#define convert_uchar1 convert_uchar
895*c217d954SCole Faust#define convert_short1 convert_short
896*c217d954SCole Faust#define convert_ushort1 convert_ushort
897*c217d954SCole Faust#define convert_int1 convert_int
898*c217d954SCole Faust#define convert_uint1 convert_uint
899*c217d954SCole Faust#define convert_long1 convert_long
900*c217d954SCole Faust#define convert_ulong1 convert_ulong
901*c217d954SCole Faust#define convert_double1 convert_double
902*c217d954SCole Faust
903*c217d954SCole Faust#define convert_char1_sat convert_char_sat
904*c217d954SCole Faust#define convert_uchar1_sat convert_uchar_sat
905*c217d954SCole Faust#define convert_uchar2_sat convert_uchar2_sat
906*c217d954SCole Faust#define convert_uchar3_sat convert_uchar3_sat
907*c217d954SCole Faust#define convert_uchar4_sat convert_uchar4_sat
908*c217d954SCole Faust#define convert_uchar8_sat convert_uchar8_sat
909*c217d954SCole Faust#define convert_uchar16_sat convert_uchar16_sat
910*c217d954SCole Faust#define convert_short1_sat convert_short_sat
911*c217d954SCole Faust#define convert_ushort1_sat convert_ushort_sat
912*c217d954SCole Faust#define convert_int1_sat convert_int_sat
913*c217d954SCole Faust#define convert_uint1_sat convert_uint_sat
914*c217d954SCole Faust#define convert_long1_sat convert_long_sat
915*c217d954SCole Faust#define convert_ulong1_sat convert_ulong_sat
916*c217d954SCole Faust#define convert_double1_sat convert_double_sat
917*c217d954SCole Faust
918*c217d954SCole Faust#define VEC_DATA_TYPE_STR(type, size) type##size
919*c217d954SCole Faust#define VEC_DATA_TYPE(type, size) VEC_DATA_TYPE_STR(type, size)
920*c217d954SCole Faust
921*c217d954SCole Faust#define CONVERT_STR(x, type) (convert_##type((x)))
922*c217d954SCole Faust#define CONVERT(x, type) CONVERT_STR(x, type)
923*c217d954SCole Faust
924*c217d954SCole Faust#define CONVERT_SAT_STR(x, type) (convert_##type##_sat((x)))
925*c217d954SCole Faust#define CONVERT_SAT(x, type) CONVERT_SAT_STR(x, type)
926*c217d954SCole Faust
927*c217d954SCole Faust#define CONVERT_SAT_ROUND_STR(x, type, round) (convert_##type##_sat_##round((x)))
928*c217d954SCole Faust#define CONVERT_SAT_ROUND(x, type, round) CONVERT_SAT_ROUND_STR(x, type, round)
929*c217d954SCole Faust
930*c217d954SCole Faust#define select_vec_dt_uchar(size) uchar##size
931*c217d954SCole Faust#define select_vec_dt_char(size) char##size
932*c217d954SCole Faust#define select_vec_dt_ushort(size) ushort##size
933*c217d954SCole Faust#define select_vec_dt_short(size) short##size
934*c217d954SCole Faust#define select_vec_dt_half(size) short##size
935*c217d954SCole Faust#define select_vec_dt_uint(size) uint##size
936*c217d954SCole Faust#define select_vec_dt_int(size) int##size
937*c217d954SCole Faust#define select_vec_dt_float(size) int##size
938*c217d954SCole Faust#define select_vec_dt_ulong(size) ulong##size
939*c217d954SCole Faust#define select_vec_dt_long(size) long##size
940*c217d954SCole Faust
941*c217d954SCole Faust#define SELECT_VEC_DATA_TYPE_STR(type, size) select_vec_dt_##type(size)
942*c217d954SCole Faust#define SELECT_VEC_DATA_TYPE(type, size) SELECT_VEC_DATA_TYPE_STR(type, size)
943*c217d954SCole Faust#define SELECT_DATA_TYPE(type) SELECT_VEC_DATA_TYPE_STR(type, 1)
944*c217d954SCole Faust
945*c217d954SCole Faust#define signed_int_vec_dt_uchar(size) char##size
946*c217d954SCole Faust#define signed_int_vec_dt_char(size) char##size
947*c217d954SCole Faust#define signed_int_vec_dt_ushort(size) short##size
948*c217d954SCole Faust#define signed_int_vec_dt_short(size) short##size
949*c217d954SCole Faust#define signed_int_vec_dt_half(size) short##size
950*c217d954SCole Faust#define signed_int_vec_dt_uint(size) int##size
951*c217d954SCole Faust#define signed_int_vec_dt_int(size) int##size
952*c217d954SCole Faust#define signed_int_vec_dt_float(size) int##size
953*c217d954SCole Faust#define signed_int_vec_dt_ulong(size) long##size
954*c217d954SCole Faust#define signed_int_vec_dt_long(size) long##size
955*c217d954SCole Faust
956*c217d954SCole Faust#define SIGNED_INT_VEC_DATA_TYPE_STR(type, size) signed_int_vec_dt_##type(size)
957*c217d954SCole Faust#define SIGNED_INT_VEC_DATA_TYPE(type, size) SIGNED_INT_VEC_DATA_TYPE_STR(type, size)
958*c217d954SCole Faust#define SIGNED_INT_DATA_TYPE(type) SIGNED_INT_VEC_DATA_TYPE_STR(type, 1)
959*c217d954SCole Faust
960*c217d954SCole Faust#define sum_reduce_1(x) (x)
961*c217d954SCole Faust#define sum_reduce_2(x) ((x).s0) + ((x).s1)
962*c217d954SCole Faust#define sum_reduce_3(x) sum_reduce_2((x).s01) + ((x).s2)
963*c217d954SCole Faust#define sum_reduce_4(x) sum_reduce_2((x).s01) + sum_reduce_2((x).s23)
964*c217d954SCole Faust#define sum_reduce_8(x) sum_reduce_4((x).s0123) + sum_reduce_4((x).s4567)
965*c217d954SCole Faust#define sum_reduce_16(x) sum_reduce_8((x).s01234567) + sum_reduce_8((x).s89ABCDEF)
966*c217d954SCole Faust
967*c217d954SCole Faust#define SUM_REDUCE_STR(x, size) sum_reduce_##size(x)
968*c217d954SCole Faust#define SUM_REDUCE(x, size) SUM_REDUCE_STR(x, size)
969*c217d954SCole Faust
970*c217d954SCole Faust#define prod_reduce_1(x) (x)
971*c217d954SCole Faust#define prod_reduce_2(x) ((x).s0) * ((x).s1)
972*c217d954SCole Faust#define prod_reduce_3(x) prod_reduce_2((x).s01) * ((x).s2)
973*c217d954SCole Faust#define prod_reduce_4(x) prod_reduce_2((x).s01) * prod_reduce_2((x).s23)
974*c217d954SCole Faust#define prod_reduce_8(x) prod_reduce_4((x).s0123) * prod_reduce_4((x).s4567)
975*c217d954SCole Faust#define prod_reduce_16(x) prod_reduce_8((x).s01234567) * prod_reduce_8((x).s89ABCDEF)
976*c217d954SCole Faust
977*c217d954SCole Faust#define PROD_REDUCE_STR(x, size) prod_reduce_##size(x)
978*c217d954SCole Faust#define PROD_REDUCE(x, size) PROD_REDUCE_STR(x, size)
979*c217d954SCole Faust
980*c217d954SCole Faust#define max_reduce_1(x) (x)
981*c217d954SCole Faust#define max_reduce_2(x) max(((x).s0), ((x).s1))
982*c217d954SCole Faust#define max_reduce_3(x) max(max_reduce_2((x).s01), ((x).s2))
983*c217d954SCole Faust#define max_reduce_4(x) max(max_reduce_2((x).s01), max_reduce_2((x).s23))
984*c217d954SCole Faust#define max_reduce_8(x) max(max_reduce_4((x).s0123), max_reduce_4((x).s4567))
985*c217d954SCole Faust#define max_reduce_16(x) max(max_reduce_8((x).s01234567), max_reduce_8((x).s89ABCDEF))
986*c217d954SCole Faust
987*c217d954SCole Faust#define MAX_REDUCE_STR(x, size) max_reduce_##size(x)
988*c217d954SCole Faust#define MAX_REDUCE(x, size) MAX_REDUCE_STR(x, size)
989*c217d954SCole Faust
990*c217d954SCole Faust#define VECTOR_DECLARATION(name)     \
991*c217d954SCole Faust    __global uchar *name##_ptr,      \
992*c217d954SCole Faust    uint        name##_stride_x, \
993*c217d954SCole Faust    uint        name##_step_x,   \
994*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
995*c217d954SCole Faust
996*c217d954SCole Faust#define IMAGE_DECLARATION(name)      \
997*c217d954SCole Faust    __global uchar *name##_ptr,      \
998*c217d954SCole Faust    uint        name##_stride_x, \
999*c217d954SCole Faust    uint        name##_step_x,   \
1000*c217d954SCole Faust    uint        name##_stride_y, \
1001*c217d954SCole Faust    uint        name##_step_y,   \
1002*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
1003*c217d954SCole Faust
1004*c217d954SCole Faust#define TENSOR3D_DECLARATION(name)   \
1005*c217d954SCole Faust    __global uchar *name##_ptr,      \
1006*c217d954SCole Faust    uint        name##_stride_x, \
1007*c217d954SCole Faust    uint        name##_step_x,   \
1008*c217d954SCole Faust    uint        name##_stride_y, \
1009*c217d954SCole Faust    uint        name##_step_y,   \
1010*c217d954SCole Faust    uint        name##_stride_z, \
1011*c217d954SCole Faust    uint        name##_step_z,   \
1012*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
1013*c217d954SCole Faust
1014*c217d954SCole Faust#define TENSOR4D_DECLARATION(name)   \
1015*c217d954SCole Faust    __global uchar *name##_ptr,      \
1016*c217d954SCole Faust    uint        name##_stride_x, \
1017*c217d954SCole Faust    uint        name##_step_x,   \
1018*c217d954SCole Faust    uint        name##_stride_y, \
1019*c217d954SCole Faust    uint        name##_step_y,   \
1020*c217d954SCole Faust    uint        name##_stride_z, \
1021*c217d954SCole Faust    uint        name##_step_z,   \
1022*c217d954SCole Faust    uint        name##_stride_w, \
1023*c217d954SCole Faust    uint        name##_step_w,   \
1024*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
1025*c217d954SCole Faust
1026*c217d954SCole Faust#define TENSOR5D_DECLARATION(name)   \
1027*c217d954SCole Faust    __global uchar *name##_ptr,      \
1028*c217d954SCole Faust    uint        name##_stride_x, \
1029*c217d954SCole Faust    uint        name##_step_x,   \
1030*c217d954SCole Faust    uint        name##_stride_y, \
1031*c217d954SCole Faust    uint        name##_step_y,   \
1032*c217d954SCole Faust    uint        name##_stride_z, \
1033*c217d954SCole Faust    uint        name##_step_z,   \
1034*c217d954SCole Faust    uint        name##_stride_w, \
1035*c217d954SCole Faust    uint        name##_step_w,   \
1036*c217d954SCole Faust    uint        name##_stride_v, \
1037*c217d954SCole Faust    uint        name##_step_v,   \
1038*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
1039*c217d954SCole Faust
1040*c217d954SCole Faust#define CONVERT_TO_VECTOR_STRUCT(name) \
1041*c217d954SCole Faust    update_vector_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x)
1042*c217d954SCole Faust
1043*c217d954SCole Faust#define CONVERT_TO_VECTOR_STRUCT_NO_STEP(name) \
1044*c217d954SCole Faust    update_vector_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0)
1045*c217d954SCole Faust
1046*c217d954SCole Faust#define CONVERT_TO_IMAGE_STRUCT(name) \
1047*c217d954SCole Faust    update_image_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y)
1048*c217d954SCole Faust
1049*c217d954SCole Faust#define CONVERT_TO_IMAGE_STRUCT_NO_STEP(name) \
1050*c217d954SCole Faust    update_image_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0, name##_stride_y, 0)
1051*c217d954SCole Faust
1052*c217d954SCole Faust#define CONVERT_TENSOR3D_TO_IMAGE_STRUCT(name) \
1053*c217d954SCole Faust    update_image_from_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, name##_stride_z, name##_step_z)
1054*c217d954SCole Faust
1055*c217d954SCole Faust#define CONVERT_TENSOR3D_TO_IMAGE_STRUCT_NO_STEP(name) \
1056*c217d954SCole Faust    update_image_from_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0, name##_stride_y, 0, name##_stride_z, name##_step_z)
1057*c217d954SCole Faust
1058*c217d954SCole Faust#define CONVERT_TENSOR3D_TO_IMAGE_STRUCT(name) \
1059*c217d954SCole Faust    update_image_from_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, name##_stride_z, name##_step_z)
1060*c217d954SCole Faust
1061*c217d954SCole Faust#define CONVERT_TO_TENSOR3D_STRUCT(name)                                                                                                           \
1062*c217d954SCole Faust    update_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, \
1063*c217d954SCole Faust                                 name##_stride_z, name##_step_z)
1064*c217d954SCole Faust
1065*c217d954SCole Faust#define CONVERT_TO_TENSOR3D_STRUCT_NO_STEP(name) \
1066*c217d954SCole Faust    update_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0, name##_stride_y, 0, name##_stride_z, 0)
1067*c217d954SCole Faust
1068*c217d954SCole Faust#define CONVERT_TO_TENSOR4D_STRUCT(name, mod_size)                                                                                                 \
1069*c217d954SCole Faust    update_tensor4D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, \
1070*c217d954SCole Faust                                 name##_stride_z, name##_step_z, name##_stride_w, name##_step_w, mod_size)
1071*c217d954SCole Faust
1072*c217d954SCole Faust#define CONVERT_TO_TENSOR4D_STRUCT_NO_STEP(name, mod_size) \
1073*c217d954SCole Faust    update_tensor4D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0, name##_stride_y, 0, name##_stride_z, 0, name##_stride_w, 0, mod_size)
1074*c217d954SCole Faust
1075*c217d954SCole Faust#define CONVERT_TO_TENSOR3D_STRUCT_NO_UPDATE_PTR(name)                                                                                       \
1076*c217d954SCole Faust    tensor3D_ptr_no_update(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, \
1077*c217d954SCole Faust                           name##_stride_z, name##_step_z)
1078*c217d954SCole Faust
1079*c217d954SCole Faust
1080*c217d954SCole Fausttypedef struct Vector
1081*c217d954SCole Faust{
1082*c217d954SCole Faust    __global uchar *ptr;
1083*c217d954SCole Faust    int             offset_first_element_in_bytes;
1084*c217d954SCole Faust    int             stride_x;
1085*c217d954SCole Faust} Vector;
1086*c217d954SCole Faust
1087*c217d954SCole Faust
1088*c217d954SCole Fausttypedef struct Image
1089*c217d954SCole Faust{
1090*c217d954SCole Faust    __global uchar *ptr;
1091*c217d954SCole Faust    int             offset_first_element_in_bytes;
1092*c217d954SCole Faust    int             stride_x;
1093*c217d954SCole Faust    int             stride_y;
1094*c217d954SCole Faust} Image;
1095*c217d954SCole Faust
1096*c217d954SCole Faust
1097*c217d954SCole Fausttypedef struct Tensor3D
1098*c217d954SCole Faust{
1099*c217d954SCole Faust    __global uchar *ptr;
1100*c217d954SCole Faust    int             offset_first_element_in_bytes;
1101*c217d954SCole Faust    int             stride_x;
1102*c217d954SCole Faust    int             stride_y;
1103*c217d954SCole Faust    int             stride_z;
1104*c217d954SCole Faust} Tensor3D;
1105*c217d954SCole Faust
1106*c217d954SCole Faust
1107*c217d954SCole Fausttypedef struct Tensor4D
1108*c217d954SCole Faust{
1109*c217d954SCole Faust    __global uchar *ptr;
1110*c217d954SCole Faust    int             offset_first_element_in_bytes;
1111*c217d954SCole Faust    int             stride_x;
1112*c217d954SCole Faust    int             stride_y;
1113*c217d954SCole Faust    int             stride_z;
1114*c217d954SCole Faust    int             stride_w;
1115*c217d954SCole Faust} Tensor4D;
1116*c217d954SCole Faust
1117*c217d954SCole Faust
1118*c217d954SCole Faustinline Vector update_vector_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x)
1119*c217d954SCole Faust{
1120*c217d954SCole Faust    Vector vector =
1121*c217d954SCole Faust    {
1122*c217d954SCole Faust        .ptr                           = ptr,
1123*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
1124*c217d954SCole Faust        .stride_x                      = stride_x,
1125*c217d954SCole Faust    };
1126*c217d954SCole Faust    vector.ptr += vector.offset_first_element_in_bytes + get_global_id(0) * step_x;
1127*c217d954SCole Faust    return vector;
1128*c217d954SCole Faust}
1129*c217d954SCole Faust
1130*c217d954SCole Faust
1131*c217d954SCole Faustinline Image update_image_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y)
1132*c217d954SCole Faust{
1133*c217d954SCole Faust    Image img =
1134*c217d954SCole Faust    {
1135*c217d954SCole Faust        .ptr                           = ptr,
1136*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
1137*c217d954SCole Faust        .stride_x                      = stride_x,
1138*c217d954SCole Faust        .stride_y                      = stride_y
1139*c217d954SCole Faust    };
1140*c217d954SCole Faust    img.ptr += img.offset_first_element_in_bytes + get_global_id(0) * step_x + get_global_id(1) * step_y;
1141*c217d954SCole Faust    return img;
1142*c217d954SCole Faust}
1143*c217d954SCole Faust
1144*c217d954SCole Faust
1145*c217d954SCole Faustinline Image update_image_from_tensor3D_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y, uint stride_z, uint step_z)
1146*c217d954SCole Faust{
1147*c217d954SCole Faust    Image img =
1148*c217d954SCole Faust    {
1149*c217d954SCole Faust        .ptr                           = ptr,
1150*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
1151*c217d954SCole Faust        .stride_x                      = stride_x,
1152*c217d954SCole Faust        .stride_y                      = stride_y
1153*c217d954SCole Faust    };
1154*c217d954SCole Faust    img.ptr += img.offset_first_element_in_bytes + get_global_id(0) * step_x + get_global_id(1) * step_y + get_global_id(2) * step_z;
1155*c217d954SCole Faust    return img;
1156*c217d954SCole Faust}
1157*c217d954SCole Faust
1158*c217d954SCole Faust
1159*c217d954SCole Faustinline Tensor3D update_tensor3D_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y, uint stride_z, uint step_z)
1160*c217d954SCole Faust{
1161*c217d954SCole Faust    Tensor3D tensor =
1162*c217d954SCole Faust    {
1163*c217d954SCole Faust        .ptr                           = ptr,
1164*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
1165*c217d954SCole Faust        .stride_x                      = stride_x,
1166*c217d954SCole Faust        .stride_y                      = stride_y,
1167*c217d954SCole Faust        .stride_z                      = stride_z
1168*c217d954SCole Faust    };
1169*c217d954SCole Faust    tensor.ptr += tensor.offset_first_element_in_bytes + get_global_id(0) * step_x + get_global_id(1) * step_y + get_global_id(2) * step_z;
1170*c217d954SCole Faust    return tensor;
1171*c217d954SCole Faust}
1172*c217d954SCole Faust
1173*c217d954SCole Faust
1174*c217d954SCole Faustinline Tensor3D tensor3D_ptr_no_update(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y, uint stride_z, uint step_z)
1175*c217d954SCole Faust{
1176*c217d954SCole Faust    Tensor3D tensor =
1177*c217d954SCole Faust    {
1178*c217d954SCole Faust        .ptr                           = ptr,
1179*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
1180*c217d954SCole Faust        .stride_x                      = stride_x,
1181*c217d954SCole Faust        .stride_y                      = stride_y,
1182*c217d954SCole Faust        .stride_z                      = stride_z
1183*c217d954SCole Faust    };
1184*c217d954SCole Faust    return tensor;
1185*c217d954SCole Faust}
1186*c217d954SCole Faust
1187*c217d954SCole Faustinline Tensor4D update_tensor4D_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y, uint stride_z, uint step_z, uint stride_w,
1188*c217d954SCole Faust                                             uint step_w,
1189*c217d954SCole Faust                                             uint mod_size)
1190*c217d954SCole Faust{
1191*c217d954SCole Faust    Tensor4D tensor =
1192*c217d954SCole Faust    {
1193*c217d954SCole Faust        .ptr                           = ptr,
1194*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
1195*c217d954SCole Faust        .stride_x                      = stride_x,
1196*c217d954SCole Faust        .stride_y                      = stride_y,
1197*c217d954SCole Faust        .stride_z                      = stride_z,
1198*c217d954SCole Faust        .stride_w                      = stride_w
1199*c217d954SCole Faust    };
1200*c217d954SCole Faust
1201*c217d954SCole Faust    tensor.ptr += tensor.offset_first_element_in_bytes + get_global_id(0) * step_x + get_global_id(1) * step_y + (get_global_id(2) % mod_size) * step_z + (get_global_id(2) / mod_size) * step_w;
1202*c217d954SCole Faust    return tensor;
1203*c217d954SCole Faust}
1204*c217d954SCole Faust
1205*c217d954SCole Faust
1206*c217d954SCole Faustinline __global const uchar *vector_offset(const Vector *vec, int x)
1207*c217d954SCole Faust{
1208*c217d954SCole Faust    return vec->ptr + x * vec->stride_x;
1209*c217d954SCole Faust}
1210*c217d954SCole Faust
1211*c217d954SCole Faust
1212*c217d954SCole Faustinline __global uchar *offset(const Image *img, int x, int y)
1213*c217d954SCole Faust{
1214*c217d954SCole Faust    return img->ptr + x * img->stride_x + y * img->stride_y;
1215*c217d954SCole Faust}
1216*c217d954SCole Faust
1217*c217d954SCole Faust
1218*c217d954SCole Faustinline __global const uchar *tensor3D_offset(const Tensor3D *tensor, int x, int y, int z)
1219*c217d954SCole Faust{
1220*c217d954SCole Faust    return tensor->ptr + x * tensor->stride_x + y * tensor->stride_y + z * tensor->stride_z;
1221*c217d954SCole Faust}
1222*c217d954SCole Faust
1223*c217d954SCole Faust
1224*c217d954SCole Faustinline __global const uchar *tensor4D_offset(const Tensor4D *tensor, int x, int y, int z, int w)
1225*c217d954SCole Faust{
1226*c217d954SCole Faust    return tensor->ptr + x * tensor->stride_x + y * tensor->stride_y + z * tensor->stride_z + w * tensor->stride_w;
1227*c217d954SCole Faust}
1228*c217d954SCole Faust
1229*c217d954SCole Faust
1230*c217d954SCole Faustinline __global const uchar *tensor3D_index2ptr(const Tensor3D *tensor, uint width, uint height, uint depth, uint index)
1231*c217d954SCole Faust{
1232*c217d954SCole Faust    uint num_elements = width * height;
1233*c217d954SCole Faust
1234*c217d954SCole Faust    const uint z = index / num_elements;
1235*c217d954SCole Faust
1236*c217d954SCole Faust    index %= num_elements;
1237*c217d954SCole Faust
1238*c217d954SCole Faust    const uint y = index / width;
1239*c217d954SCole Faust
1240*c217d954SCole Faust    index %= width;
1241*c217d954SCole Faust
1242*c217d954SCole Faust    const uint x = index;
1243*c217d954SCole Faust
1244*c217d954SCole Faust    return tensor->ptr + x * tensor->stride_x + y * tensor->stride_y + z * tensor->stride_z + tensor->offset_first_element_in_bytes;
1245*c217d954SCole Faust}
1246*c217d954SCole Faust
1247*c217d954SCole Faust#endif
1248*c217d954SCole Faust
1249*c217d954SCole Faust#if GPU_ARCH == GPU_ARCH_BIFROST
1250*c217d954SCole Faust#define MLA(a, b, c) (fma(c, b, a))
1251*c217d954SCole Faust#else
1252*c217d954SCole Faust#define MLA(a, b, c) ((b) * (c) + (a))
1253*c217d954SCole Faust#endif
1254*c217d954SCole Faust
1255*c217d954SCole Faust
1256*c217d954SCole Faust#define hard_swish_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (x * ((min(max((x + (DATA_TYPE)3.0), (DATA_TYPE)0.0), (DATA_TYPE)6.0)) * (DATA_TYPE)0.166666667))
1257*c217d954SCole Faust
1258*c217d954SCole Faust
1259*c217d954SCole Faust#define logistic_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) ((DATA_TYPE)1.0 / ((DATA_TYPE)1.0 + exp(-x)))
1260*c217d954SCole Faust
1261*c217d954SCole Faust
1262*c217d954SCole Faust#define tanh_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) ((DATA_TYPE)A_VAL * tanh((DATA_TYPE)B_VAL * x))
1263*c217d954SCole Faust
1264*c217d954SCole Faust
1265*c217d954SCole Faust#define relu_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (max((DATA_TYPE)0.0, x))
1266*c217d954SCole Faust
1267*c217d954SCole Faust
1268*c217d954SCole Faust#define brelu_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (min((DATA_TYPE)A_VAL, max((DATA_TYPE)0.0, x)))
1269*c217d954SCole Faust
1270*c217d954SCole Faust
1271*c217d954SCole Faust#define lu_brelu_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (min(max(x, (DATA_TYPE)B_VAL), (DATA_TYPE)A_VAL))
1272*c217d954SCole Faust
1273*c217d954SCole Faust
1274*c217d954SCole Faust#define lrelu_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) ((min(x, (DATA_TYPE)0.0) * (DATA_TYPE)A_VAL) + max(x, (DATA_TYPE)0.0))
1275*c217d954SCole Faust
1276*c217d954SCole Faust
1277*c217d954SCole Faust#define srelu_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (log((DATA_TYPE)1.0 + exp(x)))
1278*c217d954SCole Faust
1279*c217d954SCole Faust
1280*c217d954SCole Faust#define elu_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (select(((DATA_TYPE)A_VAL * (exp(x) - (DATA_TYPE)1.0)), x, (SELECT_VEC_DATA_TYPE(DATA_TYPE, VEC_SIZE))isgreaterequal(x, (DATA_TYPE)0.0)))
1281*c217d954SCole Faust
1282*c217d954SCole Faust
1283*c217d954SCole Faust#define abs_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (fabs(x))
1284*c217d954SCole Faust
1285*c217d954SCole Faust
1286*c217d954SCole Faust#define square_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (x * x)
1287*c217d954SCole Faust
1288*c217d954SCole Faust
1289*c217d954SCole Faust#define sqrt_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (sqrt(x))
1290*c217d954SCole Faust
1291*c217d954SCole Faust
1292*c217d954SCole Faust#define linear_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (MLA((DATA_TYPE)B_VAL, (DATA_TYPE)A_VAL, x))
1293*c217d954SCole Faust
1294*c217d954SCole Faust
1295*c217d954SCole Faust#define gelu_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (x * (DATA_TYPE)0.5 * ((DATA_TYPE)1.0 + erf(x / (DATA_TYPE)1.41421356237)))
1296*c217d954SCole Faust
1297*c217d954SCole Faust
1298*c217d954SCole Faust#define identity_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) (x)
1299*c217d954SCole Faust
1300*c217d954SCole Faust#define ACT_OP(op, DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) op##_op(DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL)
1301*c217d954SCole Faust
1302*c217d954SCole Faust#define ACTIVATION(op, DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL) ACT_OP(op, DATA_TYPE, VEC_SIZE, x, A_VAL, B_VAL)
1303*c217d954SCole Faust
1304*c217d954SCole Faust#ifndef ARM_COMPUTE_HELPER_H
1305*c217d954SCole Faust#define ARM_COMPUTE_HELPER_H
1306*c217d954SCole Faust
1307*c217d954SCole Faust
1308*c217d954SCole Faust
1309*c217d954SCole Faust
1310*c217d954SCole Faust#define STORE_ROW_1(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1311*c217d954SCole Faust    VSTORE(N0)                                                 \
1312*c217d954SCole Faust    (BASENAME##0, 0, (__global DATA_TYPE *)(PTR + 0 * STRIDE_Y + Z##0));
1313*c217d954SCole Faust
1314*c217d954SCole Faust#define STORE_ROW_2(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1315*c217d954SCole Faust    STORE_ROW_1(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1316*c217d954SCole Faust    VSTORE(N0)                                                 \
1317*c217d954SCole Faust    (BASENAME##1, 0, (__global DATA_TYPE *)(PTR + 1 * STRIDE_Y + Z##1));
1318*c217d954SCole Faust
1319*c217d954SCole Faust#define STORE_ROW_3(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1320*c217d954SCole Faust    STORE_ROW_2(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1321*c217d954SCole Faust    VSTORE(N0)                                                 \
1322*c217d954SCole Faust    (BASENAME##2, 0, (__global DATA_TYPE *)(PTR + 2 * STRIDE_Y + Z##2));
1323*c217d954SCole Faust
1324*c217d954SCole Faust#define STORE_ROW_4(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1325*c217d954SCole Faust    STORE_ROW_3(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1326*c217d954SCole Faust    VSTORE(N0)                                                 \
1327*c217d954SCole Faust    (BASENAME##3, 0, (__global DATA_TYPE *)(PTR + 3 * STRIDE_Y + Z##3));
1328*c217d954SCole Faust
1329*c217d954SCole Faust#define STORE_ROW_5(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1330*c217d954SCole Faust    STORE_ROW_4(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1331*c217d954SCole Faust    VSTORE(N0)                                                 \
1332*c217d954SCole Faust    (BASENAME##4, 0, (__global DATA_TYPE *)(PTR + 4 * STRIDE_Y + Z##4));
1333*c217d954SCole Faust
1334*c217d954SCole Faust#define STORE_ROW_6(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1335*c217d954SCole Faust    STORE_ROW_5(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1336*c217d954SCole Faust    VSTORE(N0)                                                 \
1337*c217d954SCole Faust    (BASENAME##5, 0, (__global DATA_TYPE *)(PTR + 5 * STRIDE_Y + Z##5));
1338*c217d954SCole Faust
1339*c217d954SCole Faust#define STORE_ROW_7(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1340*c217d954SCole Faust    STORE_ROW_6(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1341*c217d954SCole Faust    VSTORE(N0)                                                 \
1342*c217d954SCole Faust    (BASENAME##6, 0, (__global DATA_TYPE *)(PTR + 6 * STRIDE_Y + Z##6));
1343*c217d954SCole Faust
1344*c217d954SCole Faust#define STORE_ROW_8(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1345*c217d954SCole Faust    STORE_ROW_7(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1346*c217d954SCole Faust    VSTORE(N0)                                                 \
1347*c217d954SCole Faust    (BASENAME##7, 0, (__global DATA_TYPE *)(PTR + 7 * STRIDE_Y + Z##7));
1348*c217d954SCole Faust
1349*c217d954SCole Faust#define STORE_ROW_9(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1350*c217d954SCole Faust    STORE_ROW_8(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1351*c217d954SCole Faust    VSTORE(N0)                                                 \
1352*c217d954SCole Faust    (BASENAME##8, 0, (__global DATA_TYPE *)(PTR + 8 * STRIDE_Y + Z##8));
1353*c217d954SCole Faust
1354*c217d954SCole Faust#define STORE_ROW_10(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1355*c217d954SCole Faust    STORE_ROW_9(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)      \
1356*c217d954SCole Faust    VSTORE(N0)                                                  \
1357*c217d954SCole Faust    (BASENAME##9, 0, (__global DATA_TYPE *)(PTR + 9 * STRIDE_Y + Z##9));
1358*c217d954SCole Faust
1359*c217d954SCole Faust#define STORE_ROW_11(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1360*c217d954SCole Faust    STORE_ROW_10(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1361*c217d954SCole Faust    VSTORE(N0)                                                  \
1362*c217d954SCole Faust    (BASENAME##A, 0, (__global DATA_TYPE *)(PTR + 10 * STRIDE_Y + Z##A));
1363*c217d954SCole Faust
1364*c217d954SCole Faust#define STORE_ROW_12(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1365*c217d954SCole Faust    STORE_ROW_11(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1366*c217d954SCole Faust    VSTORE(N0)                                                  \
1367*c217d954SCole Faust    (BASENAME##B, 0, (__global DATA_TYPE *)(PTR + 11 * STRIDE_Y + Z##B));
1368*c217d954SCole Faust
1369*c217d954SCole Faust#define STORE_ROW_13(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1370*c217d954SCole Faust    STORE_ROW_12(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1371*c217d954SCole Faust    VSTORE(N0)                                                  \
1372*c217d954SCole Faust    (BASENAME##C, 0, (__global DATA_TYPE *)(PTR + 12 * STRIDE_Y + Z##C));
1373*c217d954SCole Faust
1374*c217d954SCole Faust#define STORE_ROW_14(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1375*c217d954SCole Faust    STORE_ROW_13(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1376*c217d954SCole Faust    VSTORE(N0)                                                  \
1377*c217d954SCole Faust    (BASENAME##D, 0, (__global DATA_TYPE *)(PTR + 13 * STRIDE_Y + Z##D));
1378*c217d954SCole Faust
1379*c217d954SCole Faust#define STORE_ROW_15(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1380*c217d954SCole Faust    STORE_ROW_14(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1381*c217d954SCole Faust    VSTORE(N0)                                                  \
1382*c217d954SCole Faust    (BASENAME##E, 0, (__global DATA_TYPE *)(PTR + 14 * STRIDE_Y + Z##E));
1383*c217d954SCole Faust
1384*c217d954SCole Faust#define STORE_ROW_16(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1385*c217d954SCole Faust    STORE_ROW_15(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1386*c217d954SCole Faust    VSTORE(N0)                                                  \
1387*c217d954SCole Faust    (BASENAME##F, 0, (__global DATA_TYPE *)(PTR + 15 * STRIDE_Y + Z##F));
1388*c217d954SCole Faust
1389*c217d954SCole Faust
1390*c217d954SCole Faust
1391*c217d954SCole Faust#define CONVERT_STORE_ROW_1(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1392*c217d954SCole Faust    VSTORE(N0)                                                         \
1393*c217d954SCole Faust    (CONVERT_SAT((BASENAME##0), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 0 * STRIDE_Y + Z##0));
1394*c217d954SCole Faust
1395*c217d954SCole Faust#define CONVERT_STORE_ROW_2(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1396*c217d954SCole Faust    CONVERT_STORE_ROW_1(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1397*c217d954SCole Faust    VSTORE(N0)                                                         \
1398*c217d954SCole Faust    (CONVERT_SAT((BASENAME##1), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 1 * STRIDE_Y + Z##1));
1399*c217d954SCole Faust
1400*c217d954SCole Faust#define CONVERT_STORE_ROW_3(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1401*c217d954SCole Faust    CONVERT_STORE_ROW_2(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1402*c217d954SCole Faust    VSTORE(N0)                                                         \
1403*c217d954SCole Faust    (CONVERT_SAT((BASENAME##2), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 2 * STRIDE_Y + Z##2));
1404*c217d954SCole Faust
1405*c217d954SCole Faust#define CONVERT_STORE_ROW_4(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1406*c217d954SCole Faust    CONVERT_STORE_ROW_3(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1407*c217d954SCole Faust    VSTORE(N0)                                                         \
1408*c217d954SCole Faust    (CONVERT_SAT((BASENAME##3), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 3 * STRIDE_Y + Z##3));
1409*c217d954SCole Faust
1410*c217d954SCole Faust#define CONVERT_STORE_ROW_5(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1411*c217d954SCole Faust    CONVERT_STORE_ROW_4(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1412*c217d954SCole Faust    VSTORE(N0)                                                         \
1413*c217d954SCole Faust    (CONVERT_SAT((BASENAME##4), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 4 * STRIDE_Y + Z##4));
1414*c217d954SCole Faust
1415*c217d954SCole Faust#define CONVERT_STORE_ROW_6(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1416*c217d954SCole Faust    CONVERT_STORE_ROW_5(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1417*c217d954SCole Faust    VSTORE(N0)                                                         \
1418*c217d954SCole Faust    (CONVERT_SAT((BASENAME##5), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 5 * STRIDE_Y + Z##5));
1419*c217d954SCole Faust
1420*c217d954SCole Faust#define CONVERT_STORE_ROW_7(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1421*c217d954SCole Faust    CONVERT_STORE_ROW_6(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1422*c217d954SCole Faust    VSTORE(N0)                                                         \
1423*c217d954SCole Faust    (CONVERT_SAT((BASENAME##6), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 6 * STRIDE_Y + Z##6));
1424*c217d954SCole Faust
1425*c217d954SCole Faust#define CONVERT_STORE_ROW_8(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1426*c217d954SCole Faust    CONVERT_STORE_ROW_7(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1427*c217d954SCole Faust    VSTORE(N0)                                                         \
1428*c217d954SCole Faust    (CONVERT_SAT((BASENAME##7), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 7 * STRIDE_Y + Z##7));
1429*c217d954SCole Faust
1430*c217d954SCole Faust#define CONVERT_STORE_ROW_9(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1431*c217d954SCole Faust    CONVERT_STORE_ROW_8(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1432*c217d954SCole Faust    VSTORE(N0)                                                         \
1433*c217d954SCole Faust    (CONVERT_SAT((BASENAME##8), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 8 * STRIDE_Y + Z##8));
1434*c217d954SCole Faust
1435*c217d954SCole Faust#define CONVERT_STORE_ROW_10(N0, DATA, BASENAME, PTR, STRIDE_Y, Z) \
1436*c217d954SCole Faust    CONVERT_STORE_ROW_9(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1437*c217d954SCole Faust    VSTORE(N0)                                                     \
1438*c217d954SCole Faust    (CONVERT_SAT((BASENAME##9), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 9 * STRIDE_Y + Z##9));
1439*c217d954SCole Faust
1440*c217d954SCole Faust#define CONVERT_STORE_ROW_11(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1441*c217d954SCole Faust    CONVERT_STORE_ROW_10(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1442*c217d954SCole Faust    VSTORE(N0)                                                          \
1443*c217d954SCole Faust    (CONVERT_SAT((BASENAME##A), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 10 * STRIDE_Y + Z##A));
1444*c217d954SCole Faust
1445*c217d954SCole Faust#define CONVERT_STORE_ROW_12(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1446*c217d954SCole Faust    CONVERT_STORE_ROW_11(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1447*c217d954SCole Faust    VSTORE(N0)                                                          \
1448*c217d954SCole Faust    (CONVERT_SAT((BASENAME##B), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 11 * STRIDE_Y + Z##B));
1449*c217d954SCole Faust
1450*c217d954SCole Faust#define CONVERT_STORE_ROW_13(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1451*c217d954SCole Faust    CONVERT_STORE_ROW_12(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1452*c217d954SCole Faust    VSTORE(N0)                                                          \
1453*c217d954SCole Faust    (CONVERT_SAT((BASENAME##C), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 12 * STRIDE_Y + Z##C));
1454*c217d954SCole Faust
1455*c217d954SCole Faust#define CONVERT_STORE_ROW_14(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1456*c217d954SCole Faust    CONVERT_STORE_ROW_13(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1457*c217d954SCole Faust    VSTORE(N0)                                                          \
1458*c217d954SCole Faust    (CONVERT_SAT((BASENAME##D), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 13 * STRIDE_Y + Z##D));
1459*c217d954SCole Faust
1460*c217d954SCole Faust#define CONVERT_STORE_ROW_15(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1461*c217d954SCole Faust    CONVERT_STORE_ROW_14(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1462*c217d954SCole Faust    VSTORE(N0)                                                          \
1463*c217d954SCole Faust    (CONVERT_SAT((BASENAME##E), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 14 * STRIDE_Y + Z##E));
1464*c217d954SCole Faust
1465*c217d954SCole Faust#define CONVERT_STORE_ROW_16(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1466*c217d954SCole Faust    CONVERT_STORE_ROW_15(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1467*c217d954SCole Faust    VSTORE(N0)                                                          \
1468*c217d954SCole Faust    (CONVERT_SAT((BASENAME##F), VEC_DATA_TYPE(DATA_TYPE, N0)), 0, (__global DATA_TYPE *)(PTR + 15 * STRIDE_Y + Z##F));
1469*c217d954SCole Faust
1470*c217d954SCole Faust
1471*c217d954SCole Faust
1472*c217d954SCole Faust
1473*c217d954SCole Faust#define STORE_BLOCK_STR(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) STORE_ROW_##M0(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
1474*c217d954SCole Faust#define STORE_BLOCK(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) STORE_BLOCK_STR(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
1475*c217d954SCole Faust
1476*c217d954SCole Faust
1477*c217d954SCole Faust
1478*c217d954SCole Faust#define CONVERT_STORE_BLOCK_STR(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) CONVERT_STORE_ROW_##M0(N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
1479*c217d954SCole Faust#define CONVERT_STORE_BLOCK(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) CONVERT_STORE_BLOCK_STR(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
1480*c217d954SCole Faust
1481*c217d954SCole Faust
1482*c217d954SCole Faust
1483*c217d954SCole Faust#define STORE_ROW_PARTIAL_1(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1484*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1485*c217d954SCole Faust    (BASENAME##0, 0, (__global DATA_TYPE *)(PTR + 0 * STRIDE_Y + Z##0));
1486*c217d954SCole Faust
1487*c217d954SCole Faust#define STORE_ROW_PARTIAL_2(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1488*c217d954SCole Faust    STORE_ROW_PARTIAL_1(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1489*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1490*c217d954SCole Faust    (BASENAME##1, 0, (__global DATA_TYPE *)(PTR + 1 * STRIDE_Y + Z##1));
1491*c217d954SCole Faust
1492*c217d954SCole Faust#define STORE_ROW_PARTIAL_3(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1493*c217d954SCole Faust    STORE_ROW_PARTIAL_2(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1494*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1495*c217d954SCole Faust    (BASENAME##2, 0, (__global DATA_TYPE *)(PTR + 2 * STRIDE_Y + Z##2));
1496*c217d954SCole Faust
1497*c217d954SCole Faust#define STORE_ROW_PARTIAL_4(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1498*c217d954SCole Faust    STORE_ROW_PARTIAL_3(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1499*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1500*c217d954SCole Faust    (BASENAME##3, 0, (__global DATA_TYPE *)(PTR + 3 * STRIDE_Y + Z##3));
1501*c217d954SCole Faust
1502*c217d954SCole Faust#define STORE_ROW_PARTIAL_5(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1503*c217d954SCole Faust    STORE_ROW_PARTIAL_4(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1504*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1505*c217d954SCole Faust    (BASENAME##4, 0, (__global DATA_TYPE *)(PTR + 4 * STRIDE_Y + Z##4));
1506*c217d954SCole Faust
1507*c217d954SCole Faust#define STORE_ROW_PARTIAL_6(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1508*c217d954SCole Faust    STORE_ROW_PARTIAL_5(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1509*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1510*c217d954SCole Faust    (BASENAME##5, 0, (__global DATA_TYPE *)(PTR + 5 * STRIDE_Y + Z##5));
1511*c217d954SCole Faust
1512*c217d954SCole Faust#define STORE_ROW_PARTIAL_7(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1513*c217d954SCole Faust    STORE_ROW_PARTIAL_6(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1514*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1515*c217d954SCole Faust    (BASENAME##6, 0, (__global DATA_TYPE *)(PTR + 6 * STRIDE_Y + Z##6));
1516*c217d954SCole Faust
1517*c217d954SCole Faust#define STORE_ROW_PARTIAL_8(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1518*c217d954SCole Faust    STORE_ROW_PARTIAL_7(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1519*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1520*c217d954SCole Faust    (BASENAME##7, 0, (__global DATA_TYPE *)(PTR + 7 * STRIDE_Y + Z##7));
1521*c217d954SCole Faust
1522*c217d954SCole Faust#define STORE_ROW_PARTIAL_9(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1523*c217d954SCole Faust    STORE_ROW_PARTIAL_8(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1524*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                 \
1525*c217d954SCole Faust    (BASENAME##8, 0, (__global DATA_TYPE *)(PTR + 8 * STRIDE_Y + Z##8));
1526*c217d954SCole Faust
1527*c217d954SCole Faust#define STORE_ROW_PARTIAL_10(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1528*c217d954SCole Faust    STORE_ROW_PARTIAL_9(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)      \
1529*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
1530*c217d954SCole Faust    (BASENAME##9, 0, (__global DATA_TYPE *)(PTR + 9 * STRIDE_Y + Z##9));
1531*c217d954SCole Faust
1532*c217d954SCole Faust#define STORE_ROW_PARTIAL_11(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1533*c217d954SCole Faust    STORE_ROW_PARTIAL_10(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1534*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
1535*c217d954SCole Faust    (BASENAME##A, 0, (__global DATA_TYPE *)(PTR + 10 * STRIDE_Y + Z##A));
1536*c217d954SCole Faust
1537*c217d954SCole Faust#define STORE_ROW_PARTIAL_12(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1538*c217d954SCole Faust    STORE_ROW_PARTIAL_11(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1539*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
1540*c217d954SCole Faust    (BASENAME##B, 0, (__global DATA_TYPE *)(PTR + 11 * STRIDE_Y + Z##B));
1541*c217d954SCole Faust
1542*c217d954SCole Faust#define STORE_ROW_PARTIAL_13(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1543*c217d954SCole Faust    STORE_ROW_PARTIAL_12(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1544*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
1545*c217d954SCole Faust    (BASENAME##C, 0, (__global DATA_TYPE *)(PTR + 12 * STRIDE_Y + Z##C));
1546*c217d954SCole Faust
1547*c217d954SCole Faust#define STORE_ROW_PARTIAL_14(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1548*c217d954SCole Faust    STORE_ROW_PARTIAL_13(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1549*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
1550*c217d954SCole Faust    (BASENAME##D, 0, (__global DATA_TYPE *)(PTR + 13 * STRIDE_Y + Z##D));
1551*c217d954SCole Faust
1552*c217d954SCole Faust#define STORE_ROW_PARTIAL_15(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1553*c217d954SCole Faust    STORE_ROW_PARTIAL_14(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1554*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
1555*c217d954SCole Faust    (BASENAME##E, 0, (__global DATA_TYPE *)(PTR + 14 * STRIDE_Y + Z##E));
1556*c217d954SCole Faust
1557*c217d954SCole Faust#define STORE_ROW_PARTIAL_16(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) \
1558*c217d954SCole Faust    STORE_ROW_PARTIAL_15(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)     \
1559*c217d954SCole Faust    VSTORE_PARTIAL(N0, STORE_N0)                                                  \
1560*c217d954SCole Faust    (BASENAME##F, 0, (__global DATA_TYPE *)(PTR + 15 * STRIDE_Y + Z##F));
1561*c217d954SCole Faust
1562*c217d954SCole Faust
1563*c217d954SCole Faust
1564*c217d954SCole Faust#define STORE_BLOCK_PARTIAL_STR(STORE_M0, STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) STORE_ROW_PARTIAL_##STORE_M0(N0, STORE_N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
1565*c217d954SCole Faust#define STORE_BLOCK_PARTIAL(STORE_M0, STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z) STORE_BLOCK_PARTIAL_STR(STORE_M0, STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
1566*c217d954SCole Faust
1567*c217d954SCole Faust#define STORE_BLOCK_PARTIAL_IN_X_AND_Y(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
1568*c217d954SCole Faust    if(!(PARTIAL_COND_X) && !(PARTIAL_COND_Y))                                                                                                            \
1569*c217d954SCole Faust    {                                                                                                                                                     \
1570*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                                                           \
1571*c217d954SCole Faust    }                                                                                                                                                     \
1572*c217d954SCole Faust    else if((PARTIAL_COND_Y) && !(PARTIAL_COND_X))                                                                                                        \
1573*c217d954SCole Faust    {                                                                                                                                                     \
1574*c217d954SCole Faust        STORE_BLOCK_PARTIAL(PARTIAL_STORE_M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                                             \
1575*c217d954SCole Faust    }                                                                                                                                                     \
1576*c217d954SCole Faust    else if(!(PARTIAL_COND_Y) && (PARTIAL_COND_X))                                                                                                        \
1577*c217d954SCole Faust    {                                                                                                                                                     \
1578*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, PARTIAL_STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                                             \
1579*c217d954SCole Faust    }                                                                                                                                                     \
1580*c217d954SCole Faust    else                                                                                                                                                  \
1581*c217d954SCole Faust    {                                                                                                                                                     \
1582*c217d954SCole Faust        STORE_BLOCK_PARTIAL(PARTIAL_STORE_M0, PARTIAL_STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                               \
1583*c217d954SCole Faust    }
1584*c217d954SCole Faust
1585*c217d954SCole Faust#define STORE_BLOCK_PARTIAL_IN_X(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_N0, PARTIAL_COND_X) \
1586*c217d954SCole Faust    if(!(PARTIAL_COND_X))                                                                                         \
1587*c217d954SCole Faust    {                                                                                                             \
1588*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                   \
1589*c217d954SCole Faust    }                                                                                                             \
1590*c217d954SCole Faust    else                                                                                                          \
1591*c217d954SCole Faust    {                                                                                                             \
1592*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, PARTIAL_STORE_N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                     \
1593*c217d954SCole Faust    }
1594*c217d954SCole Faust
1595*c217d954SCole Faust#define STORE_BLOCK_PARTIAL_IN_Y(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_COND_Y) \
1596*c217d954SCole Faust    if(!(PARTIAL_COND_Y))                                                                                         \
1597*c217d954SCole Faust    {                                                                                                             \
1598*c217d954SCole Faust        STORE_BLOCK_PARTIAL(M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                                   \
1599*c217d954SCole Faust    }                                                                                                             \
1600*c217d954SCole Faust    else                                                                                                          \
1601*c217d954SCole Faust    {                                                                                                             \
1602*c217d954SCole Faust        STORE_BLOCK_PARTIAL(PARTIAL_STORE_M0, N0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z);                     \
1603*c217d954SCole Faust    }
1604*c217d954SCole Faust
1605*c217d954SCole Faust
1606*c217d954SCole Faust#if defined(PARTIAL_STORE_M0) && defined(PARTIAL_STORE_N0)
1607*c217d954SCole Faust
1608*c217d954SCole Faust
1609*c217d954SCole Faust#if PARTIAL_STORE_M0 == 0 && PARTIAL_STORE_N0 == 0
1610*c217d954SCole Faust
1611*c217d954SCole Faust#define STORE_BLOCK_BOUNDARY_AWARE(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
1612*c217d954SCole Faust    STORE_BLOCK(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z)
1613*c217d954SCole Faust
1614*c217d954SCole Faust#elif PARTIAL_STORE_M0 > 0 && PARTIAL_STORE_N0 == 0
1615*c217d954SCole Faust
1616*c217d954SCole Faust#define STORE_BLOCK_BOUNDARY_AWARE(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
1617*c217d954SCole Faust    STORE_BLOCK_PARTIAL_IN_Y(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_COND_Y)
1618*c217d954SCole Faust
1619*c217d954SCole Faust#elif PARTIAL_STORE_M0 == 0 && PARTIAL_STORE_N0 > 0
1620*c217d954SCole Faust
1621*c217d954SCole Faust#define STORE_BLOCK_BOUNDARY_AWARE(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
1622*c217d954SCole Faust    STORE_BLOCK_PARTIAL_IN_X(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_N0, PARTIAL_COND_X)
1623*c217d954SCole Faust
1624*c217d954SCole Faust#else
1625*c217d954SCole Faust
1626*c217d954SCole Faust#define STORE_BLOCK_BOUNDARY_AWARE(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X) \
1627*c217d954SCole Faust    STORE_BLOCK_PARTIAL_IN_X_AND_Y(M0, N0, DATA_TYPE, BASENAME, PTR, STRIDE_Y, Z, PARTIAL_STORE_M0, PARTIAL_STORE_N0, PARTIAL_COND_Y, PARTIAL_COND_X)
1628*c217d954SCole Faust
1629*c217d954SCole Faust#endif
1630*c217d954SCole Faust
1631*c217d954SCole Faust#endif
1632*c217d954SCole Faust
1633*c217d954SCole Faust
1634*c217d954SCole Faust#if defined(PARTIAL_STORE_M0)
1635*c217d954SCole Faust
1636*c217d954SCole Faust#define COMPUTE_M0_START_ROW(y, M0, PARTIAL_STORE_M0) \
1637*c217d954SCole Faust    ((uint)(max(0, (int)(y * M0) - (int)((M0 - PARTIAL_STORE_M0) % M0))))
1638*c217d954SCole Faust#else
1639*c217d954SCole Faust#define COMPUTE_M0_START_ROW(y, M0, PARTIAL_STORE_M0) \
1640*c217d954SCole Faust    ((uint)(y * M0))
1641*c217d954SCole Faust#endif
1642*c217d954SCole Faust
1643*c217d954SCole Faust
1644*c217d954SCole Faust
1645*c217d954SCole Faust#define STORE_VECTOR_SELECT(basename, data_type, ptr, vec_size, leftover, cond) \
1646*c217d954SCole Faust    STORE_BLOCK_PARTIAL_IN_X(1, vec_size, data_type, basename, ptr, 0, 0, leftover, cond)
1647*c217d954SCole Faust
1648*c217d954SCole Faust
1649*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_FP16_ENABLED) && defined(cl_khr_fp16)
1650*c217d954SCole Faust#pragma OPENCL EXTENSION cl_khr_fp16 : enable
1651*c217d954SCole Faust#endif
1652*c217d954SCole Faust
1653*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_DOT8_ENABLED) && defined(cl_arm_integer_dot_product_int8)
1654*c217d954SCole Faust#pragma OPENCL EXTENSION cl_arm_integer_dot_product_int8 : enable
1655*c217d954SCole Faust#endif
1656*c217d954SCole Faust
1657*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_DOT8_ACC_ENABLED) && defined(cl_arm_integer_dot_product_accumulate_int8)
1658*c217d954SCole Faust#pragma OPENCL EXTENSION cl_arm_integer_dot_product_accumulate_int8 : enable
1659*c217d954SCole Faust#endif
1660*c217d954SCole Faust
1661*c217d954SCole Faust#if defined(ARM_COMPUTE_DEBUG_ENABLED) && defined(cl_arm_printf)
1662*c217d954SCole Faust#pragma OPENCL EXTENSION cl_arm_printf : enable
1663*c217d954SCole Faust#endif
1664*c217d954SCole Faust
1665*c217d954SCole Faust#define GPU_ARCH_MIDGARD 0x100
1666*c217d954SCole Faust#define GPU_ARCH_BIFROST 0x200
1667*c217d954SCole Faust#define GPU_ARCH_VALHALL 0x300
1668*c217d954SCole Faust
1669*c217d954SCole Faust
1670*c217d954SCole Faust#define CONCAT(a, b) a##b
1671*c217d954SCole Faust
1672*c217d954SCole Faust
1673*c217d954SCole Faust#define EXPAND(x) x
1674*c217d954SCole Faust
1675*c217d954SCole Faust
1676*c217d954SCole Faust#define CLAMP(x, min_val, max_val) min(max(x, min_val), max_val)
1677*c217d954SCole Faust
1678*c217d954SCole Faust
1679*c217d954SCole Faust#define REV1(x) ((x))
1680*c217d954SCole Faust#define REV2(x) ((x).s10)
1681*c217d954SCole Faust#define REV3(x) ((x).s210)
1682*c217d954SCole Faust#define REV4(x) ((x).s3210)
1683*c217d954SCole Faust#define REV8(x) ((x).s76543210)
1684*c217d954SCole Faust#define REV16(x) ((x).sFEDCBA9876543210)
1685*c217d954SCole Faust
1686*c217d954SCole Faust
1687*c217d954SCole Faust
1688*c217d954SCole Faust#define REVERSE_STR(x, s) REV##s((x))
1689*c217d954SCole Faust#define REVERSE(x, s) REVERSE_STR(x, s)
1690*c217d954SCole Faust
1691*c217d954SCole Faust
1692*c217d954SCole Faust
1693*c217d954SCole Faust#define ROT1_0(x) ((x))
1694*c217d954SCole Faust#define ROT1_1(x) ((x))
1695*c217d954SCole Faust
1696*c217d954SCole Faust#define ROT2_0(x) ((x))
1697*c217d954SCole Faust#define ROT2_1(x) ((x).s10)
1698*c217d954SCole Faust#define ROT2_2(x) ((x))
1699*c217d954SCole Faust
1700*c217d954SCole Faust#define ROT3_0(x) ((x))
1701*c217d954SCole Faust#define ROT3_1(x) ((x).s201)
1702*c217d954SCole Faust#define ROT3_2(x) ((x).s120)
1703*c217d954SCole Faust#define ROT3_3(x) ((x))
1704*c217d954SCole Faust
1705*c217d954SCole Faust#define ROT4_0(x) ((x))
1706*c217d954SCole Faust#define ROT4_1(x) ((x).s3012)
1707*c217d954SCole Faust#define ROT4_2(x) ((x).s2301)
1708*c217d954SCole Faust#define ROT4_3(x) ((x).s1230)
1709*c217d954SCole Faust#define ROT4_4(x) ((x))
1710*c217d954SCole Faust
1711*c217d954SCole Faust#define ROT8_0(x) ((x))
1712*c217d954SCole Faust#define ROT8_1(x) ((x).s70123456)
1713*c217d954SCole Faust#define ROT8_2(x) ((x).s67012345)
1714*c217d954SCole Faust#define ROT8_3(x) ((x).s56701234)
1715*c217d954SCole Faust#define ROT8_4(x) ((x).s45670123)
1716*c217d954SCole Faust#define ROT8_5(x) ((x).s34567012)
1717*c217d954SCole Faust#define ROT8_6(x) ((x).s23456701)
1718*c217d954SCole Faust#define ROT8_7(x) ((x).s12345670)
1719*c217d954SCole Faust#define ROT8_8(x) ((x))
1720*c217d954SCole Faust
1721*c217d954SCole Faust#define ROT16_0(x) ((x))
1722*c217d954SCole Faust#define ROT16_1(x) ((x).sF0123456789ABCDE)
1723*c217d954SCole Faust#define ROT16_2(x) ((x).sEF0123456789ABCD)
1724*c217d954SCole Faust#define ROT16_3(x) ((x).sDEF0123456789ABC)
1725*c217d954SCole Faust#define ROT16_4(x) ((x).sCDEF0123456789AB)
1726*c217d954SCole Faust#define ROT16_5(x) ((x).sBCDEF0123456789A)
1727*c217d954SCole Faust#define ROT16_6(x) ((x).sABCDEF0123456789)
1728*c217d954SCole Faust#define ROT16_7(x) ((x).s9ABCDEF012345678)
1729*c217d954SCole Faust#define ROT16_8(x) ((x).s89ABCDEF01234567)
1730*c217d954SCole Faust#define ROT16_9(x) ((x).s789ABCDEF0123456)
1731*c217d954SCole Faust#define ROT16_10(x) ((x).s6789ABCDEF012345)
1732*c217d954SCole Faust#define ROT16_11(x) ((x).s56789ABCDEF01234)
1733*c217d954SCole Faust#define ROT16_12(x) ((x).s456789ABCDEF0123)
1734*c217d954SCole Faust#define ROT16_13(x) ((x).s3456789ABCDEF012)
1735*c217d954SCole Faust#define ROT16_14(x) ((x).s23456789ABCDEF01)
1736*c217d954SCole Faust#define ROT16_15(x) ((x).s123456789ABCDEF0)
1737*c217d954SCole Faust#define ROT16_16(x) ((x))
1738*c217d954SCole Faust
1739*c217d954SCole Faust
1740*c217d954SCole Faust
1741*c217d954SCole Faust#define ROTATE_STR(x, s, n) ROT##s##_##n(x)
1742*c217d954SCole Faust#define ROTATE(x, s, n) ROTATE_STR(x, s, n)
1743*c217d954SCole Faust
1744*c217d954SCole Faust
1745*c217d954SCole Faust
1746*c217d954SCole Faust#define V_OFFS1(dt) (dt##1)(0)
1747*c217d954SCole Faust#define V_OFFS2(dt) (dt##2)(0, 1)
1748*c217d954SCole Faust#define V_OFFS3(dt) (dt##3)(0, 1, 2)
1749*c217d954SCole Faust#define V_OFFS4(dt) (dt##4)(0, 1, 2, 3)
1750*c217d954SCole Faust#define V_OFFS8(dt) (dt##8)(0, 1, 2, 3, 4, 5, 6, 7)
1751*c217d954SCole Faust#define V_OFFS16(dt) (dt##16)(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15)
1752*c217d954SCole Faust
1753*c217d954SCole Faust
1754*c217d954SCole Faust
1755*c217d954SCole Faust#define VEC_OFFS_STR(dt, s) V_OFFS##s(dt)
1756*c217d954SCole Faust#define VEC_OFFS(dt, s) VEC_OFFS_STR(dt, s)
1757*c217d954SCole Faust
1758*c217d954SCole Faust
1759*c217d954SCole Faust#define VLOAD_STR(size) vload##size
1760*c217d954SCole Faust#define VLOAD(size) VLOAD_STR(size)
1761*c217d954SCole Faust
1762*c217d954SCole Faust
1763*c217d954SCole Faust#define VLOAD_PARTIAL_STR(size, load_size) vload_partial_##size##_##load_size
1764*c217d954SCole Faust#define VLOAD_PARTIAL(size, load_size) VLOAD_PARTIAL_STR(size, load_size)
1765*c217d954SCole Faust
1766*c217d954SCole Faust#define NO_LOAD(data, offs, ptr) \
1767*c217d954SCole Faust    {                            \
1768*c217d954SCole Faust    }
1769*c217d954SCole Faust
1770*c217d954SCole Faust
1771*c217d954SCole Faust#define vload_partial_1_0 NO_LOAD
1772*c217d954SCole Faust#define vload_partial_1_1 vload1
1773*c217d954SCole Faust#define vload_partial_1_2 NO_LOAD
1774*c217d954SCole Faust#define vload_partial_1_3 NO_LOAD
1775*c217d954SCole Faust#define vload_partial_1_4 NO_LOAD
1776*c217d954SCole Faust#define vload_partial_1_5 NO_LOAD
1777*c217d954SCole Faust#define vload_partial_1_6 NO_LOAD
1778*c217d954SCole Faust#define vload_partial_1_7 NO_LOAD
1779*c217d954SCole Faust#define vload_partial_1_8 NO_LOAD
1780*c217d954SCole Faust#define vload_partial_1_9 NO_LOAD
1781*c217d954SCole Faust#define vload_partial_1_10 NO_LOAD
1782*c217d954SCole Faust#define vload_partial_1_11 NO_LOAD
1783*c217d954SCole Faust#define vload_partial_1_12 NO_LOAD
1784*c217d954SCole Faust#define vload_partial_1_13 NO_LOAD
1785*c217d954SCole Faust#define vload_partial_1_14 NO_LOAD
1786*c217d954SCole Faust#define vload_partial_1_15 NO_LOAD
1787*c217d954SCole Faust#define vload_partial_1_16 NO_LOAD
1788*c217d954SCole Faust
1789*c217d954SCole Faust#define vload_partial_2_0 NO_LOAD
1790*c217d954SCole Faust#define vload_partial_2_1 vload_partial_1
1791*c217d954SCole Faust#define vload_partial_2_2 vload_partial_2
1792*c217d954SCole Faust#define vload_partial_2_3 NO_LOAD
1793*c217d954SCole Faust#define vload_partial_2_4 NO_LOAD
1794*c217d954SCole Faust#define vload_partial_2_5 NO_LOAD
1795*c217d954SCole Faust#define vload_partial_2_6 NO_LOAD
1796*c217d954SCole Faust#define vload_partial_2_7 NO_LOAD
1797*c217d954SCole Faust#define vload_partial_2_8 NO_LOAD
1798*c217d954SCole Faust#define vload_partial_2_9 NO_LOAD
1799*c217d954SCole Faust#define vload_partial_2_10 NO_LOAD
1800*c217d954SCole Faust#define vload_partial_2_11 NO_LOAD
1801*c217d954SCole Faust#define vload_partial_2_12 NO_LOAD
1802*c217d954SCole Faust#define vload_partial_2_13 NO_LOAD
1803*c217d954SCole Faust#define vload_partial_2_14 NO_LOAD
1804*c217d954SCole Faust#define vload_partial_2_15 NO_LOAD
1805*c217d954SCole Faust#define vload_partial_2_16 NO_LOAD
1806*c217d954SCole Faust
1807*c217d954SCole Faust#define vload_partial_3_0 NO_LOAD
1808*c217d954SCole Faust#define vload_partial_3_1 vload_partial_1
1809*c217d954SCole Faust#define vload_partial_3_2 vload_partial_2
1810*c217d954SCole Faust#define vload_partial_3_3 vload_partial_3
1811*c217d954SCole Faust#define vload_partial_3_4 NO_LOAD
1812*c217d954SCole Faust#define vload_partial_3_5 NO_LOAD
1813*c217d954SCole Faust#define vload_partial_3_6 NO_LOAD
1814*c217d954SCole Faust#define vload_partial_3_7 NO_LOAD
1815*c217d954SCole Faust#define vload_partial_3_8 NO_LOAD
1816*c217d954SCole Faust#define vload_partial_3_9 NO_LOAD
1817*c217d954SCole Faust#define vload_partial_3_10 NO_LOAD
1818*c217d954SCole Faust#define vload_partial_3_11 NO_LOAD
1819*c217d954SCole Faust#define vload_partial_3_12 NO_LOAD
1820*c217d954SCole Faust#define vload_partial_3_13 NO_LOAD
1821*c217d954SCole Faust#define vload_partial_3_14 NO_LOAD
1822*c217d954SCole Faust#define vload_partial_3_15 NO_LOAD
1823*c217d954SCole Faust#define vload_partial_3_16 NO_LOAD
1824*c217d954SCole Faust
1825*c217d954SCole Faust#define vload_partial_4_0 NO_LOAD
1826*c217d954SCole Faust#define vload_partial_4_1 vload_partial_1
1827*c217d954SCole Faust#define vload_partial_4_2 vload_partial_2
1828*c217d954SCole Faust#define vload_partial_4_3 vload_partial_3
1829*c217d954SCole Faust#define vload_partial_4_4 vload_partial_4
1830*c217d954SCole Faust#define vload_partial_4_5 NO_LOAD
1831*c217d954SCole Faust#define vload_partial_4_6 NO_LOAD
1832*c217d954SCole Faust#define vload_partial_4_7 NO_LOAD
1833*c217d954SCole Faust#define vload_partial_4_8 NO_LOAD
1834*c217d954SCole Faust#define vload_partial_4_9 NO_LOAD
1835*c217d954SCole Faust#define vload_partial_4_10 NO_LOAD
1836*c217d954SCole Faust#define vload_partial_4_11 NO_LOAD
1837*c217d954SCole Faust#define vload_partial_4_12 NO_LOAD
1838*c217d954SCole Faust#define vload_partial_4_13 NO_LOAD
1839*c217d954SCole Faust#define vload_partial_4_14 NO_LOAD
1840*c217d954SCole Faust#define vload_partial_4_15 NO_LOAD
1841*c217d954SCole Faust#define vload_partial_4_16 NO_LOAD
1842*c217d954SCole Faust
1843*c217d954SCole Faust#define vload_partial_8_0 NO_LOAD
1844*c217d954SCole Faust#define vload_partial_8_1 vload_partial_1
1845*c217d954SCole Faust#define vload_partial_8_2 vload_partial_2
1846*c217d954SCole Faust#define vload_partial_8_3 vload_partial_3
1847*c217d954SCole Faust#define vload_partial_8_4 vload_partial_4
1848*c217d954SCole Faust#define vload_partial_8_5 vload_partial_5
1849*c217d954SCole Faust#define vload_partial_8_6 vload_partial_6
1850*c217d954SCole Faust#define vload_partial_8_7 vload_partial_7
1851*c217d954SCole Faust#define vload_partial_8_8 vload_partial_8
1852*c217d954SCole Faust#define vload_partial_8_9 NO_LOAD
1853*c217d954SCole Faust#define vload_partial_8_10 NO_LOAD
1854*c217d954SCole Faust#define vload_partial_8_11 NO_LOAD
1855*c217d954SCole Faust#define vload_partial_8_12 NO_LOAD
1856*c217d954SCole Faust#define vload_partial_8_13 NO_LOAD
1857*c217d954SCole Faust#define vload_partial_8_14 NO_LOAD
1858*c217d954SCole Faust#define vload_partial_8_15 NO_LOAD
1859*c217d954SCole Faust#define vload_partial_8_16 NO_LOAD
1860*c217d954SCole Faust
1861*c217d954SCole Faust#define vload_partial_16_0 NO_LOAD
1862*c217d954SCole Faust#define vload_partial_16_1 vload_partial_1
1863*c217d954SCole Faust#define vload_partial_16_2 vload_partial_2
1864*c217d954SCole Faust#define vload_partial_16_3 vload_partial_3
1865*c217d954SCole Faust#define vload_partial_16_4 vload_partial_4
1866*c217d954SCole Faust#define vload_partial_16_5 vload_partial_5
1867*c217d954SCole Faust#define vload_partial_16_6 vload_partial_6
1868*c217d954SCole Faust#define vload_partial_16_7 vload_partial_7
1869*c217d954SCole Faust#define vload_partial_16_8 vload_partial_8
1870*c217d954SCole Faust#define vload_partial_16_9 vload_partial_9
1871*c217d954SCole Faust#define vload_partial_16_10 vload_partial_10
1872*c217d954SCole Faust#define vload_partial_16_11 vload_partial_11
1873*c217d954SCole Faust#define vload_partial_16_12 vload_partial_12
1874*c217d954SCole Faust#define vload_partial_16_13 vload_partial_13
1875*c217d954SCole Faust#define vload_partial_16_14 vload_partial_14
1876*c217d954SCole Faust#define vload_partial_16_15 vload_partial_15
1877*c217d954SCole Faust#define vload_partial_16_16 vload_partial_16
1878*c217d954SCole Faust
1879*c217d954SCole Faust
1880*c217d954SCole Faust#define vload_partial_1(DATA, OFFSET, PTR) \
1881*c217d954SCole Faust    DATA.s0 = vload1(OFFSET, PTR);
1882*c217d954SCole Faust
1883*c217d954SCole Faust#define vload_partial_2(DATA, OFFSET, PTR) \
1884*c217d954SCole Faust    DATA.s01 = vload2(OFFSET, PTR);
1885*c217d954SCole Faust
1886*c217d954SCole Faust#define vload_partial_3(DATA, OFFSET, PTR) \
1887*c217d954SCole Faust    DATA.s012 = vload3(OFFSET, PTR);
1888*c217d954SCole Faust
1889*c217d954SCole Faust#define vload_partial_4(DATA, OFFSET, PTR) \
1890*c217d954SCole Faust    DATA.s0123 = vload4(OFFSET, PTR);
1891*c217d954SCole Faust
1892*c217d954SCole Faust#define vload_partial_5(DATA, OFFSET, PTR)    \
1893*c217d954SCole Faust    vload_partial_4(DATA.s0123, OFFSET, PTR); \
1894*c217d954SCole Faust    DATA.s4 = vload1(OFFSET, PTR + 4);
1895*c217d954SCole Faust
1896*c217d954SCole Faust#define vload_partial_6(DATA, OFFSET, PTR)    \
1897*c217d954SCole Faust    vload_partial_4(DATA.s0123, OFFSET, PTR); \
1898*c217d954SCole Faust    vload_partial_2(DATA.s45, OFFSET, PTR + 4);
1899*c217d954SCole Faust
1900*c217d954SCole Faust#define vload_partial_7(DATA, OFFSET, PTR)    \
1901*c217d954SCole Faust    vload_partial_4(DATA.s0123, OFFSET, PTR); \
1902*c217d954SCole Faust    vload_partial_3(DATA.s456, OFFSET, PTR + 4);
1903*c217d954SCole Faust
1904*c217d954SCole Faust#define vload_partial_8(DATA, OFFSET, PTR) \
1905*c217d954SCole Faust    DATA.s01234567 = vload8(OFFSET, PTR);
1906*c217d954SCole Faust
1907*c217d954SCole Faust#define vload_partial_9(DATA, OFFSET, PTR)        \
1908*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
1909*c217d954SCole Faust    DATA.s8 = vload1(OFFSET, PTR + 8);
1910*c217d954SCole Faust
1911*c217d954SCole Faust#define vload_partial_10(DATA, OFFSET, PTR)       \
1912*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
1913*c217d954SCole Faust    vload_partial_2(DATA.s89, OFFSET, PTR + 8);
1914*c217d954SCole Faust
1915*c217d954SCole Faust#define vload_partial_11(DATA, OFFSET, PTR)       \
1916*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
1917*c217d954SCole Faust    vload_partial_3(DATA.s89A, OFFSET, PTR + 8);
1918*c217d954SCole Faust
1919*c217d954SCole Faust#define vload_partial_12(DATA, OFFSET, PTR)       \
1920*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
1921*c217d954SCole Faust    vload_partial_4(DATA.s89AB, OFFSET, PTR + 8);
1922*c217d954SCole Faust
1923*c217d954SCole Faust#define vload_partial_13(DATA, OFFSET, PTR)       \
1924*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
1925*c217d954SCole Faust    vload_partial_5(DATA.s89ABCDEF, OFFSET, PTR + 8);
1926*c217d954SCole Faust
1927*c217d954SCole Faust#define vload_partial_14(DATA, OFFSET, PTR)       \
1928*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
1929*c217d954SCole Faust    vload_partial_6(DATA.s89ABCDEF, OFFSET, PTR + 8);
1930*c217d954SCole Faust
1931*c217d954SCole Faust#define vload_partial_15(DATA, OFFSET, PTR)       \
1932*c217d954SCole Faust    vload_partial_8(DATA.s01234567, OFFSET, PTR); \
1933*c217d954SCole Faust    vload_partial_7(DATA.s89ABCDEF, OFFSET, PTR + 8);
1934*c217d954SCole Faust
1935*c217d954SCole Faust#define vload_partial_16(DATA, OFFSET, PTR) \
1936*c217d954SCole Faust    DATA = vload16(OFFSET, PTR);
1937*c217d954SCole Faust
1938*c217d954SCole Faust
1939*c217d954SCole Faust
1940*c217d954SCole Faust#define PIXEL_UNIT4 1
1941*c217d954SCole Faust#define PIXEL_UNIT8 2
1942*c217d954SCole Faust#define PIXEL_UNIT16 4
1943*c217d954SCole Faust
1944*c217d954SCole Faust
1945*c217d954SCole Faust#define CONVERT_VECTOR_SIZE_TO_PIXEL_UNIT_STR(vec_size) PIXEL_UNIT##vec_size
1946*c217d954SCole Faust#define CONVERT_VECTOR_SIZE_TO_PIXEL_UNIT(vec_size) CONVERT_VECTOR_SIZE_TO_PIXEL_UNIT_STR(vec_size)
1947*c217d954SCole Faust
1948*c217d954SCole Faust
1949*c217d954SCole Faust#define read_image2d_floatx1(img, x_coord, y_coord) (float4)(read_imagef(img, (int2)(x_coord, y_coord)));
1950*c217d954SCole Faust#define read_image2d_floatx2(img, x_coord, y_coord) (float8)(read_imagef(img, (int2)(x_coord, y_coord)), read_imagef(img, (int2)(x_coord + 1, y_coord)));
1951*c217d954SCole Faust#define read_image2d_floatx4(img, x_coord, y_coord) (float16)(read_imagef(img, (int2)(x_coord, y_coord)), read_imagef(img, (int2)(x_coord + 1, y_coord)), read_imagef(img, (int2)(x_coord + 2, y_coord)), read_imagef(img, (int2)(x_coord + 3, y_coord)));
1952*c217d954SCole Faust
1953*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_FP16_ENABLED) && defined(cl_khr_fp16)
1954*c217d954SCole Faust#define read_image2d_halfx1(img, x_coord, y_coord) (half4)(read_imageh(img, (int2)(x_coord, y_coord)));
1955*c217d954SCole Faust#define read_image2d_halfx2(img, x_coord, y_coord) (half8)(read_imageh(img, (int2)(x_coord, y_coord)), read_imageh(img, (int2)(x_coord + 1, y_coord)));
1956*c217d954SCole Faust#define read_image2d_halfx4(img, x_coord, y_coord) (half16)(read_imageh(img, (int2)(x_coord, y_coord)), read_imageh(img, (int2)(x_coord + 1, y_coord)), read_imageh(img, (int2)(x_coord + 2, y_coord)), read_imageh(img, (int2)(x_coord + 3, y_coord)));
1957*c217d954SCole Faust#endif
1958*c217d954SCole Faust
1959*c217d954SCole Faust#define write_image2d_floatx1(img, x_coord, y_coord, values) (write_imagef(img, (int2)(x_coord, y_coord), values));
1960*c217d954SCole Faust#define write_image2d_floatx2(img, x_coord, y_coord, values) (write_imagef(img, (int2)(x_coord, y_coord), values.s0123), write_imagef(img, (int2)(x_coord + 1, y_coord), values.s4567));
1961*c217d954SCole Faust#define write_image2d_floatx4(img, x_coord, y_coord, values) (write_imagef(img, (int2)(x_coord, y_coord), values.s0123), write_imagef(img, (int2)(x_coord + 1, y_coord), values.s4567), write_imagef(img, (int2)(x_coord + 2, y_coord), values.s89AB), write_imagef(img, (int2)(x_coord + 3, y_coord), values.sCDEF));
1962*c217d954SCole Faust
1963*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_FP16_ENABLED) && defined(cl_khr_fp16)
1964*c217d954SCole Faust#define write_image2d_halfx1(img, x_coord, y_coord, values) (write_imageh(img, (int2)(x_coord, y_coord), values));
1965*c217d954SCole Faust#define write_image2d_halfx2(img, x_coord, y_coord, values) (write_imageh(img, (int2)(x_coord, y_coord), values.s0123), write_imageh(img, (int2)(x_coord + 1, y_coord), values.s4567));
1966*c217d954SCole Faust#define write_image2d_halfx4(img, x_coord, y_coord, values) (write_imageh(img, (int2)(x_coord, y_coord), values.s0123), write_imageh(img, (int2)(x_coord + 1, y_coord), values.s4567), write_imageh(img, (int2)(x_coord + 2, y_coord), values.s89AB), write_imageh(img, (int2)(x_coord + 3, y_coord), values.sCDEF));
1967*c217d954SCole Faust#endif
1968*c217d954SCole Faust
1969*c217d954SCole Faust
1970*c217d954SCole Faust#define READ_IMAGE2D_STR(data_type, n0, img, x_coord, y_coord) read_image2d_##data_type##x##n0(img, x_coord, y_coord)
1971*c217d954SCole Faust#define READ_IMAGE2D(data_type, n0, img, x_coord, y_coord) READ_IMAGE2D_STR(data_type, n0, img, x_coord, y_coord)
1972*c217d954SCole Faust
1973*c217d954SCole Faust
1974*c217d954SCole Faust#define WRITE_IMAGE2D_STR(data_type, n0, img, x_coord, y_coord, values) write_image2d_##data_type##x##n0(img, x_coord, y_coord, values)
1975*c217d954SCole Faust#define WRITE_IMAGE2D(data_type, n0, img, x_coord, y_coord, values) WRITE_IMAGE2D_STR(data_type, n0, img, x_coord, y_coord, values)
1976*c217d954SCole Faust
1977*c217d954SCole Faust#define VSTORE_STR(size) vstore##size
1978*c217d954SCole Faust#define VSTORE(size) VSTORE_STR(size)
1979*c217d954SCole Faust
1980*c217d954SCole Faust#define float1 float
1981*c217d954SCole Faust#define half1 half
1982*c217d954SCole Faust#define char1 char
1983*c217d954SCole Faust#define uchar1 uchar
1984*c217d954SCole Faust#define short1 short
1985*c217d954SCole Faust#define ushort1 ushort
1986*c217d954SCole Faust#define int1 int
1987*c217d954SCole Faust#define uint1 uint
1988*c217d954SCole Faust#define long1 long
1989*c217d954SCole Faust#define ulong1 ulong
1990*c217d954SCole Faust#define double1 double
1991*c217d954SCole Faust
1992*c217d954SCole Faust#define vload1(OFFSET, PTR) *(OFFSET + PTR)
1993*c217d954SCole Faust#define vstore1(DATA, OFFSET, PTR) *(OFFSET + PTR) = DATA
1994*c217d954SCole Faust
1995*c217d954SCole Faust
1996*c217d954SCole Faust#define VSTORE_PARTIAL_STR(size, store_size) vstore_partial_##size##_##store_size
1997*c217d954SCole Faust#define VSTORE_PARTIAL(size, store_size) VSTORE_PARTIAL_STR(size, store_size)
1998*c217d954SCole Faust
1999*c217d954SCole Faust#define NO_STORE(data, offs, ptr) \
2000*c217d954SCole Faust    {                             \
2001*c217d954SCole Faust    }
2002*c217d954SCole Faust
2003*c217d954SCole Faust
2004*c217d954SCole Faust#define vstore_partial_1_0 NO_STORE
2005*c217d954SCole Faust#define vstore_partial_1_1 vstore1
2006*c217d954SCole Faust#define vstore_partial_1_2 NO_STORE
2007*c217d954SCole Faust#define vstore_partial_1_3 NO_STORE
2008*c217d954SCole Faust#define vstore_partial_1_4 NO_STORE
2009*c217d954SCole Faust#define vstore_partial_1_5 NO_STORE
2010*c217d954SCole Faust#define vstore_partial_1_6 NO_STORE
2011*c217d954SCole Faust#define vstore_partial_1_7 NO_STORE
2012*c217d954SCole Faust#define vstore_partial_1_8 NO_STORE
2013*c217d954SCole Faust#define vstore_partial_1_9 NO_STORE
2014*c217d954SCole Faust#define vstore_partial_1_10 NO_STORE
2015*c217d954SCole Faust#define vstore_partial_1_11 NO_STORE
2016*c217d954SCole Faust#define vstore_partial_1_12 NO_STORE
2017*c217d954SCole Faust#define vstore_partial_1_13 NO_STORE
2018*c217d954SCole Faust#define vstore_partial_1_14 NO_STORE
2019*c217d954SCole Faust#define vstore_partial_1_15 NO_STORE
2020*c217d954SCole Faust#define vstore_partial_1_16 NO_STORE
2021*c217d954SCole Faust
2022*c217d954SCole Faust#define vstore_partial_2_0 NO_STORE
2023*c217d954SCole Faust#define vstore_partial_2_1 vstore_partial_1
2024*c217d954SCole Faust#define vstore_partial_2_2 vstore_partial_2
2025*c217d954SCole Faust#define vstore_partial_2_3 NO_STORE
2026*c217d954SCole Faust#define vstore_partial_2_4 NO_STORE
2027*c217d954SCole Faust#define vstore_partial_2_5 NO_STORE
2028*c217d954SCole Faust#define vstore_partial_2_6 NO_STORE
2029*c217d954SCole Faust#define vstore_partial_2_7 NO_STORE
2030*c217d954SCole Faust#define vstore_partial_2_8 NO_STORE
2031*c217d954SCole Faust#define vstore_partial_2_9 NO_STORE
2032*c217d954SCole Faust#define vstore_partial_2_10 NO_STORE
2033*c217d954SCole Faust#define vstore_partial_2_11 NO_STORE
2034*c217d954SCole Faust#define vstore_partial_2_12 NO_STORE
2035*c217d954SCole Faust#define vstore_partial_2_13 NO_STORE
2036*c217d954SCole Faust#define vstore_partial_2_14 NO_STORE
2037*c217d954SCole Faust#define vstore_partial_2_15 NO_STORE
2038*c217d954SCole Faust#define vstore_partial_2_16 NO_STORE
2039*c217d954SCole Faust
2040*c217d954SCole Faust#define vstore_partial_3_0 NO_STORE
2041*c217d954SCole Faust#define vstore_partial_3_1 vstore_partial_1
2042*c217d954SCole Faust#define vstore_partial_3_2 vstore_partial_2
2043*c217d954SCole Faust#define vstore_partial_3_3 vstore_partial_3
2044*c217d954SCole Faust#define vstore_partial_3_4 NO_STORE
2045*c217d954SCole Faust#define vstore_partial_3_5 NO_STORE
2046*c217d954SCole Faust#define vstore_partial_3_6 NO_STORE
2047*c217d954SCole Faust#define vstore_partial_3_7 NO_STORE
2048*c217d954SCole Faust#define vstore_partial_3_8 NO_STORE
2049*c217d954SCole Faust#define vstore_partial_3_9 NO_STORE
2050*c217d954SCole Faust#define vstore_partial_3_10 NO_STORE
2051*c217d954SCole Faust#define vstore_partial_3_11 NO_STORE
2052*c217d954SCole Faust#define vstore_partial_3_12 NO_STORE
2053*c217d954SCole Faust#define vstore_partial_3_13 NO_STORE
2054*c217d954SCole Faust#define vstore_partial_3_14 NO_STORE
2055*c217d954SCole Faust#define vstore_partial_3_15 NO_STORE
2056*c217d954SCole Faust#define vstore_partial_3_16 NO_STORE
2057*c217d954SCole Faust
2058*c217d954SCole Faust#define vstore_partial_4_0 NO_STORE
2059*c217d954SCole Faust#define vstore_partial_4_1 vstore_partial_1
2060*c217d954SCole Faust#define vstore_partial_4_2 vstore_partial_2
2061*c217d954SCole Faust#define vstore_partial_4_3 vstore_partial_3
2062*c217d954SCole Faust#define vstore_partial_4_4 vstore_partial_4
2063*c217d954SCole Faust#define vstore_partial_4_5 NO_STORE
2064*c217d954SCole Faust#define vstore_partial_4_6 NO_STORE
2065*c217d954SCole Faust#define vstore_partial_4_7 NO_STORE
2066*c217d954SCole Faust#define vstore_partial_4_8 NO_STORE
2067*c217d954SCole Faust#define vstore_partial_4_9 NO_STORE
2068*c217d954SCole Faust#define vstore_partial_4_10 NO_STORE
2069*c217d954SCole Faust#define vstore_partial_4_11 NO_STORE
2070*c217d954SCole Faust#define vstore_partial_4_12 NO_STORE
2071*c217d954SCole Faust#define vstore_partial_4_13 NO_STORE
2072*c217d954SCole Faust#define vstore_partial_4_14 NO_STORE
2073*c217d954SCole Faust#define vstore_partial_4_15 NO_STORE
2074*c217d954SCole Faust#define vstore_partial_4_16 NO_STORE
2075*c217d954SCole Faust
2076*c217d954SCole Faust#define vstore_partial_8_0 NO_STORE
2077*c217d954SCole Faust#define vstore_partial_8_1 vstore_partial_1
2078*c217d954SCole Faust#define vstore_partial_8_2 vstore_partial_2
2079*c217d954SCole Faust#define vstore_partial_8_3 vstore_partial_3
2080*c217d954SCole Faust#define vstore_partial_8_4 vstore_partial_4
2081*c217d954SCole Faust#define vstore_partial_8_5 vstore_partial_5
2082*c217d954SCole Faust#define vstore_partial_8_6 vstore_partial_6
2083*c217d954SCole Faust#define vstore_partial_8_7 vstore_partial_7
2084*c217d954SCole Faust#define vstore_partial_8_8 vstore_partial_8
2085*c217d954SCole Faust#define vstore_partial_8_9 NO_STORE
2086*c217d954SCole Faust#define vstore_partial_8_10 NO_STORE
2087*c217d954SCole Faust#define vstore_partial_8_11 NO_STORE
2088*c217d954SCole Faust#define vstore_partial_8_12 NO_STORE
2089*c217d954SCole Faust#define vstore_partial_8_13 NO_STORE
2090*c217d954SCole Faust#define vstore_partial_8_14 NO_STORE
2091*c217d954SCole Faust#define vstore_partial_8_15 NO_STORE
2092*c217d954SCole Faust#define vstore_partial_8_16 NO_STORE
2093*c217d954SCole Faust
2094*c217d954SCole Faust#define vstore_partial_16_0 NO_STORE
2095*c217d954SCole Faust#define vstore_partial_16_1 vstore_partial_1
2096*c217d954SCole Faust#define vstore_partial_16_2 vstore_partial_2
2097*c217d954SCole Faust#define vstore_partial_16_3 vstore_partial_3
2098*c217d954SCole Faust#define vstore_partial_16_4 vstore_partial_4
2099*c217d954SCole Faust#define vstore_partial_16_5 vstore_partial_5
2100*c217d954SCole Faust#define vstore_partial_16_6 vstore_partial_6
2101*c217d954SCole Faust#define vstore_partial_16_7 vstore_partial_7
2102*c217d954SCole Faust#define vstore_partial_16_8 vstore_partial_8
2103*c217d954SCole Faust#define vstore_partial_16_9 vstore_partial_9
2104*c217d954SCole Faust#define vstore_partial_16_10 vstore_partial_10
2105*c217d954SCole Faust#define vstore_partial_16_11 vstore_partial_11
2106*c217d954SCole Faust#define vstore_partial_16_12 vstore_partial_12
2107*c217d954SCole Faust#define vstore_partial_16_13 vstore_partial_13
2108*c217d954SCole Faust#define vstore_partial_16_14 vstore_partial_14
2109*c217d954SCole Faust#define vstore_partial_16_15 vstore_partial_15
2110*c217d954SCole Faust#define vstore_partial_16_16 vstore_partial_16
2111*c217d954SCole Faust
2112*c217d954SCole Faust
2113*c217d954SCole Faust#define vstore_partial_1(DATA, OFFSET, PTR) \
2114*c217d954SCole Faust    vstore1(DATA.s0, OFFSET, PTR);
2115*c217d954SCole Faust
2116*c217d954SCole Faust#define vstore_partial_2(DATA, OFFSET, PTR) \
2117*c217d954SCole Faust    vstore2(DATA.s01, OFFSET, PTR);
2118*c217d954SCole Faust
2119*c217d954SCole Faust#define vstore_partial_3(DATA, OFFSET, PTR) \
2120*c217d954SCole Faust    vstore3(DATA.s012, OFFSET, PTR);
2121*c217d954SCole Faust
2122*c217d954SCole Faust#define vstore_partial_4(DATA, OFFSET, PTR) \
2123*c217d954SCole Faust    vstore4(DATA.s0123, OFFSET, PTR);
2124*c217d954SCole Faust
2125*c217d954SCole Faust#define vstore_partial_5(DATA, OFFSET, PTR)    \
2126*c217d954SCole Faust    vstore_partial_4(DATA.s0123, OFFSET, PTR); \
2127*c217d954SCole Faust    vstore1(DATA.s4, OFFSET, PTR + 4);
2128*c217d954SCole Faust
2129*c217d954SCole Faust#define vstore_partial_6(DATA, OFFSET, PTR)    \
2130*c217d954SCole Faust    vstore_partial_4(DATA.s0123, OFFSET, PTR); \
2131*c217d954SCole Faust    vstore_partial_2(DATA.s45, OFFSET, PTR + 4);
2132*c217d954SCole Faust
2133*c217d954SCole Faust#define vstore_partial_7(DATA, OFFSET, PTR)    \
2134*c217d954SCole Faust    vstore_partial_4(DATA.s0123, OFFSET, PTR); \
2135*c217d954SCole Faust    vstore_partial_3(DATA.s456, OFFSET, PTR + 4);
2136*c217d954SCole Faust
2137*c217d954SCole Faust#define vstore_partial_8(DATA, OFFSET, PTR) \
2138*c217d954SCole Faust    vstore8(DATA.s01234567, OFFSET, PTR);
2139*c217d954SCole Faust
2140*c217d954SCole Faust#define vstore_partial_9(DATA, OFFSET, PTR)        \
2141*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
2142*c217d954SCole Faust    vstore1(DATA.s8, OFFSET, PTR + 8);
2143*c217d954SCole Faust
2144*c217d954SCole Faust#define vstore_partial_10(DATA, OFFSET, PTR)       \
2145*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
2146*c217d954SCole Faust    vstore_partial_2(DATA.s89, OFFSET, PTR + 8);
2147*c217d954SCole Faust
2148*c217d954SCole Faust#define vstore_partial_11(DATA, OFFSET, PTR)       \
2149*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
2150*c217d954SCole Faust    vstore_partial_3(DATA.s89a, OFFSET, PTR + 8);
2151*c217d954SCole Faust
2152*c217d954SCole Faust#define vstore_partial_12(DATA, OFFSET, PTR)       \
2153*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
2154*c217d954SCole Faust    vstore_partial_4(DATA.s89ab, OFFSET, PTR + 8);
2155*c217d954SCole Faust
2156*c217d954SCole Faust#define vstore_partial_13(DATA, OFFSET, PTR)       \
2157*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
2158*c217d954SCole Faust    vstore_partial_5(DATA.s89abcdef, OFFSET, PTR + 8);
2159*c217d954SCole Faust
2160*c217d954SCole Faust#define vstore_partial_14(DATA, OFFSET, PTR)       \
2161*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
2162*c217d954SCole Faust    vstore_partial_6(DATA.s89abcdef, OFFSET, PTR + 8);
2163*c217d954SCole Faust
2164*c217d954SCole Faust#define vstore_partial_15(DATA, OFFSET, PTR)       \
2165*c217d954SCole Faust    vstore_partial_8(DATA.s01234567, OFFSET, PTR); \
2166*c217d954SCole Faust    vstore_partial_7(DATA.s89abcdef, OFFSET, PTR + 8);
2167*c217d954SCole Faust
2168*c217d954SCole Faust#define vstore_partial_16(DATA, OFFSET, PTR) \
2169*c217d954SCole Faust    vstore16(DATA, OFFSET, PTR);
2170*c217d954SCole Faust
2171*c217d954SCole Faust
2172*c217d954SCole Faust
2173*c217d954SCole Faust
2174*c217d954SCole Faust
2175*c217d954SCole Faust#define convert_float_sat convert_float
2176*c217d954SCole Faust#define convert_float1_sat convert_float
2177*c217d954SCole Faust#define convert_float2_sat convert_float2
2178*c217d954SCole Faust#define convert_float3_sat convert_float3
2179*c217d954SCole Faust#define convert_float4_sat convert_float4
2180*c217d954SCole Faust#define convert_float8_sat convert_float8
2181*c217d954SCole Faust#define convert_float16_sat convert_float16
2182*c217d954SCole Faust#define convert_half_sat convert_float
2183*c217d954SCole Faust#define convert_half1_sat convert_half
2184*c217d954SCole Faust#define convert_half2_sat convert_half2
2185*c217d954SCole Faust#define convert_half3_sat convert_half3
2186*c217d954SCole Faust#define convert_half4_sat convert_half4
2187*c217d954SCole Faust#define convert_half8_sat convert_half8
2188*c217d954SCole Faust#define convert_half16_sat convert_half16
2189*c217d954SCole Faust
2190*c217d954SCole Faust#define convert_float1 convert_float
2191*c217d954SCole Faust#define convert_half1 convert_half
2192*c217d954SCole Faust#define convert_char1 convert_char
2193*c217d954SCole Faust#define convert_uchar1 convert_uchar
2194*c217d954SCole Faust#define convert_short1 convert_short
2195*c217d954SCole Faust#define convert_ushort1 convert_ushort
2196*c217d954SCole Faust#define convert_int1 convert_int
2197*c217d954SCole Faust#define convert_uint1 convert_uint
2198*c217d954SCole Faust#define convert_long1 convert_long
2199*c217d954SCole Faust#define convert_ulong1 convert_ulong
2200*c217d954SCole Faust#define convert_double1 convert_double
2201*c217d954SCole Faust
2202*c217d954SCole Faust#define convert_char1_sat convert_char_sat
2203*c217d954SCole Faust#define convert_uchar1_sat convert_uchar_sat
2204*c217d954SCole Faust#define convert_uchar2_sat convert_uchar2_sat
2205*c217d954SCole Faust#define convert_uchar3_sat convert_uchar3_sat
2206*c217d954SCole Faust#define convert_uchar4_sat convert_uchar4_sat
2207*c217d954SCole Faust#define convert_uchar8_sat convert_uchar8_sat
2208*c217d954SCole Faust#define convert_uchar16_sat convert_uchar16_sat
2209*c217d954SCole Faust#define convert_short1_sat convert_short_sat
2210*c217d954SCole Faust#define convert_ushort1_sat convert_ushort_sat
2211*c217d954SCole Faust#define convert_int1_sat convert_int_sat
2212*c217d954SCole Faust#define convert_uint1_sat convert_uint_sat
2213*c217d954SCole Faust#define convert_long1_sat convert_long_sat
2214*c217d954SCole Faust#define convert_ulong1_sat convert_ulong_sat
2215*c217d954SCole Faust#define convert_double1_sat convert_double_sat
2216*c217d954SCole Faust
2217*c217d954SCole Faust#define VEC_DATA_TYPE_STR(type, size) type##size
2218*c217d954SCole Faust#define VEC_DATA_TYPE(type, size) VEC_DATA_TYPE_STR(type, size)
2219*c217d954SCole Faust
2220*c217d954SCole Faust#define CONVERT_STR(x, type) (convert_##type((x)))
2221*c217d954SCole Faust#define CONVERT(x, type) CONVERT_STR(x, type)
2222*c217d954SCole Faust
2223*c217d954SCole Faust#define CONVERT_SAT_STR(x, type) (convert_##type##_sat((x)))
2224*c217d954SCole Faust#define CONVERT_SAT(x, type) CONVERT_SAT_STR(x, type)
2225*c217d954SCole Faust
2226*c217d954SCole Faust#define CONVERT_SAT_ROUND_STR(x, type, round) (convert_##type##_sat_##round((x)))
2227*c217d954SCole Faust#define CONVERT_SAT_ROUND(x, type, round) CONVERT_SAT_ROUND_STR(x, type, round)
2228*c217d954SCole Faust
2229*c217d954SCole Faust#define select_vec_dt_uchar(size) uchar##size
2230*c217d954SCole Faust#define select_vec_dt_char(size) char##size
2231*c217d954SCole Faust#define select_vec_dt_ushort(size) ushort##size
2232*c217d954SCole Faust#define select_vec_dt_short(size) short##size
2233*c217d954SCole Faust#define select_vec_dt_half(size) short##size
2234*c217d954SCole Faust#define select_vec_dt_uint(size) uint##size
2235*c217d954SCole Faust#define select_vec_dt_int(size) int##size
2236*c217d954SCole Faust#define select_vec_dt_float(size) int##size
2237*c217d954SCole Faust#define select_vec_dt_ulong(size) ulong##size
2238*c217d954SCole Faust#define select_vec_dt_long(size) long##size
2239*c217d954SCole Faust
2240*c217d954SCole Faust#define SELECT_VEC_DATA_TYPE_STR(type, size) select_vec_dt_##type(size)
2241*c217d954SCole Faust#define SELECT_VEC_DATA_TYPE(type, size) SELECT_VEC_DATA_TYPE_STR(type, size)
2242*c217d954SCole Faust#define SELECT_DATA_TYPE(type) SELECT_VEC_DATA_TYPE_STR(type, 1)
2243*c217d954SCole Faust
2244*c217d954SCole Faust#define signed_int_vec_dt_uchar(size) char##size
2245*c217d954SCole Faust#define signed_int_vec_dt_char(size) char##size
2246*c217d954SCole Faust#define signed_int_vec_dt_ushort(size) short##size
2247*c217d954SCole Faust#define signed_int_vec_dt_short(size) short##size
2248*c217d954SCole Faust#define signed_int_vec_dt_half(size) short##size
2249*c217d954SCole Faust#define signed_int_vec_dt_uint(size) int##size
2250*c217d954SCole Faust#define signed_int_vec_dt_int(size) int##size
2251*c217d954SCole Faust#define signed_int_vec_dt_float(size) int##size
2252*c217d954SCole Faust#define signed_int_vec_dt_ulong(size) long##size
2253*c217d954SCole Faust#define signed_int_vec_dt_long(size) long##size
2254*c217d954SCole Faust
2255*c217d954SCole Faust#define SIGNED_INT_VEC_DATA_TYPE_STR(type, size) signed_int_vec_dt_##type(size)
2256*c217d954SCole Faust#define SIGNED_INT_VEC_DATA_TYPE(type, size) SIGNED_INT_VEC_DATA_TYPE_STR(type, size)
2257*c217d954SCole Faust#define SIGNED_INT_DATA_TYPE(type) SIGNED_INT_VEC_DATA_TYPE_STR(type, 1)
2258*c217d954SCole Faust
2259*c217d954SCole Faust#define sum_reduce_1(x) (x)
2260*c217d954SCole Faust#define sum_reduce_2(x) ((x).s0) + ((x).s1)
2261*c217d954SCole Faust#define sum_reduce_3(x) sum_reduce_2((x).s01) + ((x).s2)
2262*c217d954SCole Faust#define sum_reduce_4(x) sum_reduce_2((x).s01) + sum_reduce_2((x).s23)
2263*c217d954SCole Faust#define sum_reduce_8(x) sum_reduce_4((x).s0123) + sum_reduce_4((x).s4567)
2264*c217d954SCole Faust#define sum_reduce_16(x) sum_reduce_8((x).s01234567) + sum_reduce_8((x).s89ABCDEF)
2265*c217d954SCole Faust
2266*c217d954SCole Faust#define SUM_REDUCE_STR(x, size) sum_reduce_##size(x)
2267*c217d954SCole Faust#define SUM_REDUCE(x, size) SUM_REDUCE_STR(x, size)
2268*c217d954SCole Faust
2269*c217d954SCole Faust#define prod_reduce_1(x) (x)
2270*c217d954SCole Faust#define prod_reduce_2(x) ((x).s0) * ((x).s1)
2271*c217d954SCole Faust#define prod_reduce_3(x) prod_reduce_2((x).s01) * ((x).s2)
2272*c217d954SCole Faust#define prod_reduce_4(x) prod_reduce_2((x).s01) * prod_reduce_2((x).s23)
2273*c217d954SCole Faust#define prod_reduce_8(x) prod_reduce_4((x).s0123) * prod_reduce_4((x).s4567)
2274*c217d954SCole Faust#define prod_reduce_16(x) prod_reduce_8((x).s01234567) * prod_reduce_8((x).s89ABCDEF)
2275*c217d954SCole Faust
2276*c217d954SCole Faust#define PROD_REDUCE_STR(x, size) prod_reduce_##size(x)
2277*c217d954SCole Faust#define PROD_REDUCE(x, size) PROD_REDUCE_STR(x, size)
2278*c217d954SCole Faust
2279*c217d954SCole Faust#define max_reduce_1(x) (x)
2280*c217d954SCole Faust#define max_reduce_2(x) max(((x).s0), ((x).s1))
2281*c217d954SCole Faust#define max_reduce_3(x) max(max_reduce_2((x).s01), ((x).s2))
2282*c217d954SCole Faust#define max_reduce_4(x) max(max_reduce_2((x).s01), max_reduce_2((x).s23))
2283*c217d954SCole Faust#define max_reduce_8(x) max(max_reduce_4((x).s0123), max_reduce_4((x).s4567))
2284*c217d954SCole Faust#define max_reduce_16(x) max(max_reduce_8((x).s01234567), max_reduce_8((x).s89ABCDEF))
2285*c217d954SCole Faust
2286*c217d954SCole Faust#define MAX_REDUCE_STR(x, size) max_reduce_##size(x)
2287*c217d954SCole Faust#define MAX_REDUCE(x, size) MAX_REDUCE_STR(x, size)
2288*c217d954SCole Faust
2289*c217d954SCole Faust#define VECTOR_DECLARATION(name)     \
2290*c217d954SCole Faust    __global uchar *name##_ptr,      \
2291*c217d954SCole Faust    uint        name##_stride_x, \
2292*c217d954SCole Faust    uint        name##_step_x,   \
2293*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2294*c217d954SCole Faust
2295*c217d954SCole Faust#define IMAGE_DECLARATION(name)      \
2296*c217d954SCole Faust    __global uchar *name##_ptr,      \
2297*c217d954SCole Faust    uint        name##_stride_x, \
2298*c217d954SCole Faust    uint        name##_step_x,   \
2299*c217d954SCole Faust    uint        name##_stride_y, \
2300*c217d954SCole Faust    uint        name##_step_y,   \
2301*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2302*c217d954SCole Faust
2303*c217d954SCole Faust#define TENSOR3D_DECLARATION(name)   \
2304*c217d954SCole Faust    __global uchar *name##_ptr,      \
2305*c217d954SCole Faust    uint        name##_stride_x, \
2306*c217d954SCole Faust    uint        name##_step_x,   \
2307*c217d954SCole Faust    uint        name##_stride_y, \
2308*c217d954SCole Faust    uint        name##_step_y,   \
2309*c217d954SCole Faust    uint        name##_stride_z, \
2310*c217d954SCole Faust    uint        name##_step_z,   \
2311*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2312*c217d954SCole Faust
2313*c217d954SCole Faust#define TENSOR4D_DECLARATION(name)   \
2314*c217d954SCole Faust    __global uchar *name##_ptr,      \
2315*c217d954SCole Faust    uint        name##_stride_x, \
2316*c217d954SCole Faust    uint        name##_step_x,   \
2317*c217d954SCole Faust    uint        name##_stride_y, \
2318*c217d954SCole Faust    uint        name##_step_y,   \
2319*c217d954SCole Faust    uint        name##_stride_z, \
2320*c217d954SCole Faust    uint        name##_step_z,   \
2321*c217d954SCole Faust    uint        name##_stride_w, \
2322*c217d954SCole Faust    uint        name##_step_w,   \
2323*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2324*c217d954SCole Faust
2325*c217d954SCole Faust#define TENSOR5D_DECLARATION(name)   \
2326*c217d954SCole Faust    __global uchar *name##_ptr,      \
2327*c217d954SCole Faust    uint        name##_stride_x, \
2328*c217d954SCole Faust    uint        name##_step_x,   \
2329*c217d954SCole Faust    uint        name##_stride_y, \
2330*c217d954SCole Faust    uint        name##_step_y,   \
2331*c217d954SCole Faust    uint        name##_stride_z, \
2332*c217d954SCole Faust    uint        name##_step_z,   \
2333*c217d954SCole Faust    uint        name##_stride_w, \
2334*c217d954SCole Faust    uint        name##_step_w,   \
2335*c217d954SCole Faust    uint        name##_stride_v, \
2336*c217d954SCole Faust    uint        name##_step_v,   \
2337*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2338*c217d954SCole Faust
2339*c217d954SCole Faust#define CONVERT_TO_VECTOR_STRUCT(name) \
2340*c217d954SCole Faust    update_vector_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x)
2341*c217d954SCole Faust
2342*c217d954SCole Faust#define CONVERT_TO_VECTOR_STRUCT_NO_STEP(name) \
2343*c217d954SCole Faust    update_vector_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0)
2344*c217d954SCole Faust
2345*c217d954SCole Faust#define CONVERT_TO_IMAGE_STRUCT(name) \
2346*c217d954SCole Faust    update_image_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y)
2347*c217d954SCole Faust
2348*c217d954SCole Faust#define CONVERT_TO_IMAGE_STRUCT_NO_STEP(name) \
2349*c217d954SCole Faust    update_image_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0, name##_stride_y, 0)
2350*c217d954SCole Faust
2351*c217d954SCole Faust#define CONVERT_TENSOR3D_TO_IMAGE_STRUCT(name) \
2352*c217d954SCole Faust    update_image_from_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, name##_stride_z, name##_step_z)
2353*c217d954SCole Faust
2354*c217d954SCole Faust#define CONVERT_TENSOR3D_TO_IMAGE_STRUCT_NO_STEP(name) \
2355*c217d954SCole Faust    update_image_from_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0, name##_stride_y, 0, name##_stride_z, name##_step_z)
2356*c217d954SCole Faust
2357*c217d954SCole Faust#define CONVERT_TENSOR3D_TO_IMAGE_STRUCT(name) \
2358*c217d954SCole Faust    update_image_from_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, name##_stride_z, name##_step_z)
2359*c217d954SCole Faust
2360*c217d954SCole Faust#define CONVERT_TO_TENSOR3D_STRUCT(name)                                                                                                           \
2361*c217d954SCole Faust    update_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, \
2362*c217d954SCole Faust                                 name##_stride_z, name##_step_z)
2363*c217d954SCole Faust
2364*c217d954SCole Faust#define CONVERT_TO_TENSOR3D_STRUCT_NO_STEP(name) \
2365*c217d954SCole Faust    update_tensor3D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0, name##_stride_y, 0, name##_stride_z, 0)
2366*c217d954SCole Faust
2367*c217d954SCole Faust#define CONVERT_TO_TENSOR4D_STRUCT(name, mod_size)                                                                                                 \
2368*c217d954SCole Faust    update_tensor4D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, \
2369*c217d954SCole Faust                                 name##_stride_z, name##_step_z, name##_stride_w, name##_step_w, mod_size)
2370*c217d954SCole Faust
2371*c217d954SCole Faust#define CONVERT_TO_TENSOR4D_STRUCT_NO_STEP(name, mod_size) \
2372*c217d954SCole Faust    update_tensor4D_workitem_ptr(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, 0, name##_stride_y, 0, name##_stride_z, 0, name##_stride_w, 0, mod_size)
2373*c217d954SCole Faust
2374*c217d954SCole Faust#define CONVERT_TO_TENSOR3D_STRUCT_NO_UPDATE_PTR(name)                                                                                       \
2375*c217d954SCole Faust    tensor3D_ptr_no_update(name##_ptr, name##_offset_first_element_in_bytes, name##_stride_x, name##_step_x, name##_stride_y, name##_step_y, \
2376*c217d954SCole Faust                           name##_stride_z, name##_step_z)
2377*c217d954SCole Faust
2378*c217d954SCole Faust
2379*c217d954SCole Fausttypedef struct Vector
2380*c217d954SCole Faust{
2381*c217d954SCole Faust    __global uchar *ptr;
2382*c217d954SCole Faust    int             offset_first_element_in_bytes;
2383*c217d954SCole Faust    int             stride_x;
2384*c217d954SCole Faust} Vector;
2385*c217d954SCole Faust
2386*c217d954SCole Faust
2387*c217d954SCole Fausttypedef struct Image
2388*c217d954SCole Faust{
2389*c217d954SCole Faust    __global uchar *ptr;
2390*c217d954SCole Faust    int             offset_first_element_in_bytes;
2391*c217d954SCole Faust    int             stride_x;
2392*c217d954SCole Faust    int             stride_y;
2393*c217d954SCole Faust} Image;
2394*c217d954SCole Faust
2395*c217d954SCole Faust
2396*c217d954SCole Fausttypedef struct Tensor3D
2397*c217d954SCole Faust{
2398*c217d954SCole Faust    __global uchar *ptr;
2399*c217d954SCole Faust    int             offset_first_element_in_bytes;
2400*c217d954SCole Faust    int             stride_x;
2401*c217d954SCole Faust    int             stride_y;
2402*c217d954SCole Faust    int             stride_z;
2403*c217d954SCole Faust} Tensor3D;
2404*c217d954SCole Faust
2405*c217d954SCole Faust
2406*c217d954SCole Fausttypedef struct Tensor4D
2407*c217d954SCole Faust{
2408*c217d954SCole Faust    __global uchar *ptr;
2409*c217d954SCole Faust    int             offset_first_element_in_bytes;
2410*c217d954SCole Faust    int             stride_x;
2411*c217d954SCole Faust    int             stride_y;
2412*c217d954SCole Faust    int             stride_z;
2413*c217d954SCole Faust    int             stride_w;
2414*c217d954SCole Faust} Tensor4D;
2415*c217d954SCole Faust
2416*c217d954SCole Faust
2417*c217d954SCole Faustinline Vector update_vector_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x)
2418*c217d954SCole Faust{
2419*c217d954SCole Faust    Vector vector =
2420*c217d954SCole Faust    {
2421*c217d954SCole Faust        .ptr                           = ptr,
2422*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
2423*c217d954SCole Faust        .stride_x                      = stride_x,
2424*c217d954SCole Faust    };
2425*c217d954SCole Faust    vector.ptr += vector.offset_first_element_in_bytes + get_global_id(0) * step_x;
2426*c217d954SCole Faust    return vector;
2427*c217d954SCole Faust}
2428*c217d954SCole Faust
2429*c217d954SCole Faust
2430*c217d954SCole Faustinline Image update_image_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y)
2431*c217d954SCole Faust{
2432*c217d954SCole Faust    Image img =
2433*c217d954SCole Faust    {
2434*c217d954SCole Faust        .ptr                           = ptr,
2435*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
2436*c217d954SCole Faust        .stride_x                      = stride_x,
2437*c217d954SCole Faust        .stride_y                      = stride_y
2438*c217d954SCole Faust    };
2439*c217d954SCole Faust    img.ptr += img.offset_first_element_in_bytes + get_global_id(0) * step_x + get_global_id(1) * step_y;
2440*c217d954SCole Faust    return img;
2441*c217d954SCole Faust}
2442*c217d954SCole Faust
2443*c217d954SCole Faust
2444*c217d954SCole Faustinline Image update_image_from_tensor3D_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y, uint stride_z, uint step_z)
2445*c217d954SCole Faust{
2446*c217d954SCole Faust    Image img =
2447*c217d954SCole Faust    {
2448*c217d954SCole Faust        .ptr                           = ptr,
2449*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
2450*c217d954SCole Faust        .stride_x                      = stride_x,
2451*c217d954SCole Faust        .stride_y                      = stride_y
2452*c217d954SCole Faust    };
2453*c217d954SCole Faust    img.ptr += img.offset_first_element_in_bytes + get_global_id(0) * step_x + get_global_id(1) * step_y + get_global_id(2) * step_z;
2454*c217d954SCole Faust    return img;
2455*c217d954SCole Faust}
2456*c217d954SCole Faust
2457*c217d954SCole Faust
2458*c217d954SCole Faustinline Tensor3D update_tensor3D_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y, uint stride_z, uint step_z)
2459*c217d954SCole Faust{
2460*c217d954SCole Faust    Tensor3D tensor =
2461*c217d954SCole Faust    {
2462*c217d954SCole Faust        .ptr                           = ptr,
2463*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
2464*c217d954SCole Faust        .stride_x                      = stride_x,
2465*c217d954SCole Faust        .stride_y                      = stride_y,
2466*c217d954SCole Faust        .stride_z                      = stride_z
2467*c217d954SCole Faust    };
2468*c217d954SCole Faust    tensor.ptr += tensor.offset_first_element_in_bytes + get_global_id(0) * step_x + get_global_id(1) * step_y + get_global_id(2) * step_z;
2469*c217d954SCole Faust    return tensor;
2470*c217d954SCole Faust}
2471*c217d954SCole Faust
2472*c217d954SCole Faust
2473*c217d954SCole Faustinline Tensor3D tensor3D_ptr_no_update(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y, uint stride_z, uint step_z)
2474*c217d954SCole Faust{
2475*c217d954SCole Faust    Tensor3D tensor =
2476*c217d954SCole Faust    {
2477*c217d954SCole Faust        .ptr                           = ptr,
2478*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
2479*c217d954SCole Faust        .stride_x                      = stride_x,
2480*c217d954SCole Faust        .stride_y                      = stride_y,
2481*c217d954SCole Faust        .stride_z                      = stride_z
2482*c217d954SCole Faust    };
2483*c217d954SCole Faust    return tensor;
2484*c217d954SCole Faust}
2485*c217d954SCole Faust
2486*c217d954SCole Faustinline Tensor4D update_tensor4D_workitem_ptr(__global uchar *ptr, uint offset_first_element_in_bytes, uint stride_x, uint step_x, uint stride_y, uint step_y, uint stride_z, uint step_z, uint stride_w,
2487*c217d954SCole Faust                                             uint step_w,
2488*c217d954SCole Faust                                             uint mod_size)
2489*c217d954SCole Faust{
2490*c217d954SCole Faust    Tensor4D tensor =
2491*c217d954SCole Faust    {
2492*c217d954SCole Faust        .ptr                           = ptr,
2493*c217d954SCole Faust        .offset_first_element_in_bytes = offset_first_element_in_bytes,
2494*c217d954SCole Faust        .stride_x                      = stride_x,
2495*c217d954SCole Faust        .stride_y                      = stride_y,
2496*c217d954SCole Faust        .stride_z                      = stride_z,
2497*c217d954SCole Faust        .stride_w                      = stride_w
2498*c217d954SCole Faust    };
2499*c217d954SCole Faust
2500*c217d954SCole Faust    tensor.ptr += tensor.offset_first_element_in_bytes + get_global_id(0) * step_x + get_global_id(1) * step_y + (get_global_id(2) % mod_size) * step_z + (get_global_id(2) / mod_size) * step_w;
2501*c217d954SCole Faust    return tensor;
2502*c217d954SCole Faust}
2503*c217d954SCole Faust
2504*c217d954SCole Faust
2505*c217d954SCole Faustinline __global const uchar *vector_offset(const Vector *vec, int x)
2506*c217d954SCole Faust{
2507*c217d954SCole Faust    return vec->ptr + x * vec->stride_x;
2508*c217d954SCole Faust}
2509*c217d954SCole Faust
2510*c217d954SCole Faust
2511*c217d954SCole Faustinline __global uchar *offset(const Image *img, int x, int y)
2512*c217d954SCole Faust{
2513*c217d954SCole Faust    return img->ptr + x * img->stride_x + y * img->stride_y;
2514*c217d954SCole Faust}
2515*c217d954SCole Faust
2516*c217d954SCole Faust
2517*c217d954SCole Faustinline __global const uchar *tensor3D_offset(const Tensor3D *tensor, int x, int y, int z)
2518*c217d954SCole Faust{
2519*c217d954SCole Faust    return tensor->ptr + x * tensor->stride_x + y * tensor->stride_y + z * tensor->stride_z;
2520*c217d954SCole Faust}
2521*c217d954SCole Faust
2522*c217d954SCole Faust
2523*c217d954SCole Faustinline __global const uchar *tensor4D_offset(const Tensor4D *tensor, int x, int y, int z, int w)
2524*c217d954SCole Faust{
2525*c217d954SCole Faust    return tensor->ptr + x * tensor->stride_x + y * tensor->stride_y + z * tensor->stride_z + w * tensor->stride_w;
2526*c217d954SCole Faust}
2527*c217d954SCole Faust
2528*c217d954SCole Faust
2529*c217d954SCole Faustinline __global const uchar *tensor3D_index2ptr(const Tensor3D *tensor, uint width, uint height, uint depth, uint index)
2530*c217d954SCole Faust{
2531*c217d954SCole Faust    uint num_elements = width * height;
2532*c217d954SCole Faust
2533*c217d954SCole Faust    const uint z = index / num_elements;
2534*c217d954SCole Faust
2535*c217d954SCole Faust    index %= num_elements;
2536*c217d954SCole Faust
2537*c217d954SCole Faust    const uint y = index / width;
2538*c217d954SCole Faust
2539*c217d954SCole Faust    index %= width;
2540*c217d954SCole Faust
2541*c217d954SCole Faust    const uint x = index;
2542*c217d954SCole Faust
2543*c217d954SCole Faust    return tensor->ptr + x * tensor->stride_x + y * tensor->stride_y + z * tensor->stride_z + tensor->offset_first_element_in_bytes;
2544*c217d954SCole Faust}
2545*c217d954SCole Faust
2546*c217d954SCole Faust#endif
2547*c217d954SCole Faust
2548*c217d954SCole Faust#ifndef SRC_CORE_CL_CL_KERNELS_TILE_HELPERS
2549*c217d954SCole Faust#define SRC_CORE_CL_CL_KERNELS_TILE_HELPERS
2550*c217d954SCole Faust
2551*c217d954SCole Faust
2552*c217d954SCole Faust
2553*c217d954SCole Faust
2554*c217d954SCole Faust#define TILE_VECTOR_SIZE1 1
2555*c217d954SCole Faust#define TILE_VECTOR_SIZE2 2
2556*c217d954SCole Faust#define TILE_VECTOR_SIZE3 3
2557*c217d954SCole Faust#define TILE_VECTOR_SIZE4 4
2558*c217d954SCole Faust#define TILE_VECTOR_SIZE5 8
2559*c217d954SCole Faust#define TILE_VECTOR_SIZE6 8
2560*c217d954SCole Faust#define TILE_VECTOR_SIZE7 8
2561*c217d954SCole Faust#define TILE_VECTOR_SIZE8 8
2562*c217d954SCole Faust#define TILE_VECTOR_SIZE9 16
2563*c217d954SCole Faust#define TILE_VECTOR_SIZE10 16
2564*c217d954SCole Faust#define TILE_VECTOR_SIZE11 16
2565*c217d954SCole Faust#define TILE_VECTOR_SIZE12 16
2566*c217d954SCole Faust#define TILE_VECTOR_SIZE13 16
2567*c217d954SCole Faust#define TILE_VECTOR_SIZE14 16
2568*c217d954SCole Faust#define TILE_VECTOR_SIZE15 16
2569*c217d954SCole Faust#define TILE_VECTOR_SIZE16 16
2570*c217d954SCole Faust
2571*c217d954SCole Faust#define TILE_VECTOR_TYPE1(DATA_TYPE) DATA_TYPE##1
2572*c217d954SCole Faust#define TILE_VECTOR_TYPE2(DATA_TYPE) DATA_TYPE##2
2573*c217d954SCole Faust#define TILE_VECTOR_TYPE3(DATA_TYPE) DATA_TYPE##3
2574*c217d954SCole Faust#define TILE_VECTOR_TYPE4(DATA_TYPE) DATA_TYPE##4
2575*c217d954SCole Faust#define TILE_VECTOR_TYPE5(DATA_TYPE) DATA_TYPE##8
2576*c217d954SCole Faust#define TILE_VECTOR_TYPE6(DATA_TYPE) DATA_TYPE##8
2577*c217d954SCole Faust#define TILE_VECTOR_TYPE7(DATA_TYPE) DATA_TYPE##8
2578*c217d954SCole Faust#define TILE_VECTOR_TYPE8(DATA_TYPE) DATA_TYPE##8
2579*c217d954SCole Faust#define TILE_VECTOR_TYPE9(DATA_TYPE) DATA_TYPE##16
2580*c217d954SCole Faust#define TILE_VECTOR_TYPE10(DATA_TYPE) DATA_TYPE##16
2581*c217d954SCole Faust#define TILE_VECTOR_TYPE11(DATA_TYPE) DATA_TYPE##16
2582*c217d954SCole Faust#define TILE_VECTOR_TYPE12(DATA_TYPE) DATA_TYPE##16
2583*c217d954SCole Faust#define TILE_VECTOR_TYPE13(DATA_TYPE) DATA_TYPE##16
2584*c217d954SCole Faust#define TILE_VECTOR_TYPE14(DATA_TYPE) DATA_TYPE##16
2585*c217d954SCole Faust#define TILE_VECTOR_TYPE15(DATA_TYPE) DATA_TYPE##16
2586*c217d954SCole Faust#define TILE_VECTOR_TYPE16(DATA_TYPE) DATA_TYPE##16
2587*c217d954SCole Faust
2588*c217d954SCole Faust
2589*c217d954SCole Faust#define TILE(DATA_TYPE, H, W, BASENAME) TILE_STR(DATA_TYPE, H, W, BASENAME)
2590*c217d954SCole Faust#define TILE_STR(DATA_TYPE, H, W, BASENAME) \
2591*c217d954SCole Faust    union {                                 \
2592*c217d954SCole Faust        DATA_TYPE                      s[TILE_VECTOR_SIZE##W];                  \
2593*c217d954SCole Faust        TILE_VECTOR_TYPE##W(DATA_TYPE) v;                     \
2594*c217d954SCole Faust    } BASENAME[H]
2595*c217d954SCole Faust
2596*c217d954SCole Faust#define TENSOR4D_IMAGE(name)          \
2597*c217d954SCole Faust    __read_only image2d_t name##_img, \
2598*c217d954SCole Faust    __global uchar *name##_ptr,       \
2599*c217d954SCole Faust    uint            name##_stride_x,  \
2600*c217d954SCole Faust    uint            name##_step_x,    \
2601*c217d954SCole Faust    uint            name##_stride_y,  \
2602*c217d954SCole Faust    uint            name##_step_y,    \
2603*c217d954SCole Faust    uint            name##_stride_z,  \
2604*c217d954SCole Faust    uint            name##_step_z,    \
2605*c217d954SCole Faust    uint            name##_stride_w,  \
2606*c217d954SCole Faust    uint            name##_step_w,    \
2607*c217d954SCole Faust    uint            name##_offset_first_element_in_bytes
2608*c217d954SCole Faust
2609*c217d954SCole Faust#define TENSOR4D_BUFFER(name)    \
2610*c217d954SCole Faust    __global uchar *name##_ptr,  \
2611*c217d954SCole Faust    uint        name##_stride_x, \
2612*c217d954SCole Faust    uint        name##_step_x,   \
2613*c217d954SCole Faust    uint        name##_stride_y, \
2614*c217d954SCole Faust    uint        name##_step_y,   \
2615*c217d954SCole Faust    uint        name##_stride_z, \
2616*c217d954SCole Faust    uint        name##_step_z,   \
2617*c217d954SCole Faust    uint        name##_stride_w, \
2618*c217d954SCole Faust    uint        name##_step_w,   \
2619*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2620*c217d954SCole Faust
2621*c217d954SCole Faust#define TENSOR4D_STR(name, type) TENSOR4D_##type(name)
2622*c217d954SCole Faust#define TENSOR4D(name, type) TENSOR4D_STR(name, type)
2623*c217d954SCole Faust
2624*c217d954SCole Faust#define TENSOR4D_T_IMAGE(name)          \
2625*c217d954SCole Faust    __read_only image2d_t name##_img, \
2626*c217d954SCole Faust    __global uchar *name##_ptr,       \
2627*c217d954SCole Faust    uint        name##_stride_y, \
2628*c217d954SCole Faust    uint        name##_stride_z, \
2629*c217d954SCole Faust    uint        name##_stride_w, \
2630*c217d954SCole Faust    uint        name##_c,   \
2631*c217d954SCole Faust    uint        name##_w,   \
2632*c217d954SCole Faust    uint        name##_h,   \
2633*c217d954SCole Faust    uint        name##_n,   \
2634*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2635*c217d954SCole Faust
2636*c217d954SCole Faust#define TENSOR4D_T_BUFFER(name)    \
2637*c217d954SCole Faust    __global uchar *name##_ptr,  \
2638*c217d954SCole Faust    uint        name##_stride_y, \
2639*c217d954SCole Faust    uint        name##_stride_z, \
2640*c217d954SCole Faust    uint        name##_stride_w, \
2641*c217d954SCole Faust    uint        name##_c,   \
2642*c217d954SCole Faust    uint        name##_w,   \
2643*c217d954SCole Faust    uint        name##_h,   \
2644*c217d954SCole Faust    uint        name##_n,   \
2645*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2646*c217d954SCole Faust
2647*c217d954SCole Faust#define TENSOR4D_T_STR(name, type) TENSOR4D_T_##type(name)
2648*c217d954SCole Faust
2649*c217d954SCole Faust
2650*c217d954SCole Faust#define TENSOR4D_T(name, type) TENSOR4D_T_STR(name, type)
2651*c217d954SCole Faust
2652*c217d954SCole Faust#define TENSOR4D_RO_T_IMAGE(name)          \
2653*c217d954SCole Faust    __read_only image2d_t name##_img, \
2654*c217d954SCole Faust    TENSOR4D_T_BUFFER(name)
2655*c217d954SCole Faust
2656*c217d954SCole Faust#define TENSOR4D_RO_T_BUFFER(name) TENSOR4D_T_BUFFER(name)
2657*c217d954SCole Faust
2658*c217d954SCole Faust#define TENSOR4D_RO_T_STR(name, type) TENSOR4D_RO_T_##type(name)
2659*c217d954SCole Faust
2660*c217d954SCole Faust
2661*c217d954SCole Faust#define TENSOR4D_RO_T(name, type) TENSOR4D_RO_T_STR(name, type)
2662*c217d954SCole Faust
2663*c217d954SCole Faust#define TENSOR4D_WO_T_IMAGE(name)          \
2664*c217d954SCole Faust    __write_only image2d_t name##_img, \
2665*c217d954SCole Faust    TENSOR4D_T_BUFFER(name)
2666*c217d954SCole Faust
2667*c217d954SCole Faust#define TENSOR4D_WO_T_BUFFER(name) TENSOR4D_T_BUFFER(name)
2668*c217d954SCole Faust
2669*c217d954SCole Faust#define TENSOR4D_WO_T_STR(name, type) TENSOR4D_WO_T_##type(name)
2670*c217d954SCole Faust
2671*c217d954SCole Faust
2672*c217d954SCole Faust#define TENSOR4D_WO_T(name, type) TENSOR4D_WO_T_STR(name, type)
2673*c217d954SCole Faust
2674*c217d954SCole Faust#define TENSOR3D_T_IMAGE(name)          \
2675*c217d954SCole Faust    __read_only image2d_t name##_img, \
2676*c217d954SCole Faust    __global uchar *name##_ptr,       \
2677*c217d954SCole Faust    uint        name##_stride_y, \
2678*c217d954SCole Faust    uint        name##_stride_z, \
2679*c217d954SCole Faust    uint        name##_w,   \
2680*c217d954SCole Faust    uint        name##_h,   \
2681*c217d954SCole Faust    uint        name##_n,   \
2682*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2683*c217d954SCole Faust
2684*c217d954SCole Faust#define TENSOR3D_T_BUFFER(name)    \
2685*c217d954SCole Faust    __global uchar *name##_ptr,  \
2686*c217d954SCole Faust    uint        name##_stride_y, \
2687*c217d954SCole Faust    uint        name##_stride_z, \
2688*c217d954SCole Faust    uint        name##_w,   \
2689*c217d954SCole Faust    uint        name##_h,   \
2690*c217d954SCole Faust    uint        name##_n,   \
2691*c217d954SCole Faust    uint        name##_offset_first_element_in_bytes
2692*c217d954SCole Faust
2693*c217d954SCole Faust#define TENSOR3D_T_STR(name, type) TENSOR3D_T_##type(name)
2694*c217d954SCole Faust#define TENSOR3D_T(name, type) TENSOR3D_T_STR(name, type)
2695*c217d954SCole Faust
2696*c217d954SCole Faust#if !defined(UNROLL_WITH_PRAGMA)
2697*c217d954SCole Faust#define UNROLL_INCR(idx, step, macro) idx += (step); (macro)
2698*c217d954SCole Faust
2699*c217d954SCole Faust#define LOOP_UNROLLING_1(idx, step, macro) (macro)
2700*c217d954SCole Faust#define LOOP_UNROLLING_2(idx, step, macro) LOOP_UNROLLING_1(idx, step, macro); UNROLL_INCR(idx, step, macro)
2701*c217d954SCole Faust#define LOOP_UNROLLING_3(idx, step, macro) LOOP_UNROLLING_2(idx, step, macro); UNROLL_INCR(idx, step, macro)
2702*c217d954SCole Faust#define LOOP_UNROLLING_4(idx, step, macro) LOOP_UNROLLING_3(idx, step, macro); UNROLL_INCR(idx, step, macro)
2703*c217d954SCole Faust#define LOOP_UNROLLING_5(idx, step, macro) LOOP_UNROLLING_4(idx, step, macro); UNROLL_INCR(idx, step, macro)
2704*c217d954SCole Faust#define LOOP_UNROLLING_6(idx, step, macro) LOOP_UNROLLING_5(idx, step, macro); UNROLL_INCR(idx, step, macro)
2705*c217d954SCole Faust#define LOOP_UNROLLING_7(idx, step, macro) LOOP_UNROLLING_6(idx, step, macro); UNROLL_INCR(idx, step, macro)
2706*c217d954SCole Faust#define LOOP_UNROLLING_8(idx, step, macro) LOOP_UNROLLING_7(idx, step, macro); UNROLL_INCR(idx, step, macro)
2707*c217d954SCole Faust#define LOOP_UNROLLING_9(idx, step, macro) LOOP_UNROLLING_8(idx, step, macro); UNROLL_INCR(idx, step, macro)
2708*c217d954SCole Faust#define LOOP_UNROLLING_10(idx, step, macro) LOOP_UNROLLING_9(idx, step, macro); UNROLL_INCR(idx, step, macro)
2709*c217d954SCole Faust#define LOOP_UNROLLING_11(idx, step, macro) LOOP_UNROLLING_10(idx, step, macro); UNROLL_INCR(idx, step, macro)
2710*c217d954SCole Faust#define LOOP_UNROLLING_12(idx, step, macro) LOOP_UNROLLING_11(idx, step, macro); UNROLL_INCR(idx, step, macro)
2711*c217d954SCole Faust#define LOOP_UNROLLING_13(idx, step, macro) LOOP_UNROLLING_12(idx, step, macro); UNROLL_INCR(idx, step, macro)
2712*c217d954SCole Faust#define LOOP_UNROLLING_14(idx, step, macro) LOOP_UNROLLING_13(idx, step, macro); UNROLL_INCR(idx, step, macro)
2713*c217d954SCole Faust#define LOOP_UNROLLING_15(idx, step, macro) LOOP_UNROLLING_14(idx, step, macro); UNROLL_INCR(idx, step, macro)
2714*c217d954SCole Faust#define LOOP_UNROLLING_16(idx, step, macro) LOOP_UNROLLING_15(idx, step, macro); UNROLL_INCR(idx, step, macro)
2715*c217d954SCole Faust#define LOOP_UNROLLING_17(idx, step, macro) LOOP_UNROLLING_16(idx, step, macro); UNROLL_INCR(idx, step, macro)
2716*c217d954SCole Faust#define LOOP_UNROLLING_18(idx, step, macro) LOOP_UNROLLING_17(idx, step, macro); UNROLL_INCR(idx, step, macro)
2717*c217d954SCole Faust#define LOOP_UNROLLING_19(idx, step, macro) LOOP_UNROLLING_18(idx, step, macro); UNROLL_INCR(idx, step, macro)
2718*c217d954SCole Faust#define LOOP_UNROLLING_20(idx, step, macro) LOOP_UNROLLING_19(idx, step, macro); UNROLL_INCR(idx, step, macro)
2719*c217d954SCole Faust#define LOOP_UNROLLING_21(idx, step, macro) LOOP_UNROLLING_20(idx, step, macro); UNROLL_INCR(idx, step, macro)
2720*c217d954SCole Faust#define LOOP_UNROLLING_22(idx, step, macro) LOOP_UNROLLING_21(idx, step, macro); UNROLL_INCR(idx, step, macro)
2721*c217d954SCole Faust#define LOOP_UNROLLING_23(idx, step, macro) LOOP_UNROLLING_22(idx, step, macro); UNROLL_INCR(idx, step, macro)
2722*c217d954SCole Faust#define LOOP_UNROLLING_24(idx, step, macro) LOOP_UNROLLING_23(idx, step, macro); UNROLL_INCR(idx, step, macro)
2723*c217d954SCole Faust#define LOOP_UNROLLING_25(idx, step, macro) LOOP_UNROLLING_24(idx, step, macro); UNROLL_INCR(idx, step, macro)
2724*c217d954SCole Faust#define LOOP_UNROLLING_26(idx, step, macro) LOOP_UNROLLING_25(idx, step, macro); UNROLL_INCR(idx, step, macro)
2725*c217d954SCole Faust#define LOOP_UNROLLING_27(idx, step, macro) LOOP_UNROLLING_26(idx, step, macro); UNROLL_INCR(idx, step, macro)
2726*c217d954SCole Faust#define LOOP_UNROLLING_28(idx, step, macro) LOOP_UNROLLING_27(idx, step, macro); UNROLL_INCR(idx, step, macro)
2727*c217d954SCole Faust#define LOOP_UNROLLING_29(idx, step, macro) LOOP_UNROLLING_28(idx, step, macro); UNROLL_INCR(idx, step, macro)
2728*c217d954SCole Faust#define LOOP_UNROLLING_30(idx, step, macro) LOOP_UNROLLING_29(idx, step, macro); UNROLL_INCR(idx, step, macro)
2729*c217d954SCole Faust#define LOOP_UNROLLING_31(idx, step, macro) LOOP_UNROLLING_30(idx, step, macro); UNROLL_INCR(idx, step, macro)
2730*c217d954SCole Faust#define LOOP_UNROLLING_32(idx, step, macro) LOOP_UNROLLING_31(idx, step, macro); UNROLL_INCR(idx, step, macro)
2731*c217d954SCole Faust#define LOOP_UNROLLING_33(idx, step, macro) LOOP_UNROLLING_32(idx, step, macro); UNROLL_INCR(idx, step, macro)
2732*c217d954SCole Faust#define LOOP_UNROLLING_34(idx, step, macro) LOOP_UNROLLING_33(idx, step, macro); UNROLL_INCR(idx, step, macro)
2733*c217d954SCole Faust#define LOOP_UNROLLING_35(idx, step, macro) LOOP_UNROLLING_34(idx, step, macro); UNROLL_INCR(idx, step, macro)
2734*c217d954SCole Faust#define LOOP_UNROLLING_36(idx, step, macro) LOOP_UNROLLING_35(idx, step, macro); UNROLL_INCR(idx, step, macro)
2735*c217d954SCole Faust#define LOOP_UNROLLING_37(idx, step, macro) LOOP_UNROLLING_36(idx, step, macro); UNROLL_INCR(idx, step, macro)
2736*c217d954SCole Faust#define LOOP_UNROLLING_38(idx, step, macro) LOOP_UNROLLING_37(idx, step, macro); UNROLL_INCR(idx, step, macro)
2737*c217d954SCole Faust#define LOOP_UNROLLING_39(idx, step, macro) LOOP_UNROLLING_38(idx, step, macro); UNROLL_INCR(idx, step, macro)
2738*c217d954SCole Faust#define LOOP_UNROLLING_40(idx, step, macro) LOOP_UNROLLING_39(idx, step, macro); UNROLL_INCR(idx, step, macro)
2739*c217d954SCole Faust#define LOOP_UNROLLING_41(idx, step, macro) LOOP_UNROLLING_40(idx, step, macro); UNROLL_INCR(idx, step, macro)
2740*c217d954SCole Faust#define LOOP_UNROLLING_42(idx, step, macro) LOOP_UNROLLING_41(idx, step, macro); UNROLL_INCR(idx, step, macro)
2741*c217d954SCole Faust#define LOOP_UNROLLING_43(idx, step, macro) LOOP_UNROLLING_42(idx, step, macro); UNROLL_INCR(idx, step, macro)
2742*c217d954SCole Faust#define LOOP_UNROLLING_44(idx, step, macro) LOOP_UNROLLING_43(idx, step, macro); UNROLL_INCR(idx, step, macro)
2743*c217d954SCole Faust#define LOOP_UNROLLING_45(idx, step, macro) LOOP_UNROLLING_44(idx, step, macro); UNROLL_INCR(idx, step, macro)
2744*c217d954SCole Faust#define LOOP_UNROLLING_46(idx, step, macro) LOOP_UNROLLING_45(idx, step, macro); UNROLL_INCR(idx, step, macro)
2745*c217d954SCole Faust#define LOOP_UNROLLING_47(idx, step, macro) LOOP_UNROLLING_46(idx, step, macro); UNROLL_INCR(idx, step, macro)
2746*c217d954SCole Faust#define LOOP_UNROLLING_48(idx, step, macro) LOOP_UNROLLING_47(idx, step, macro); UNROLL_INCR(idx, step, macro)
2747*c217d954SCole Faust#define LOOP_UNROLLING_49(idx, step, macro) LOOP_UNROLLING_48(idx, step, macro); UNROLL_INCR(idx, step, macro)
2748*c217d954SCole Faust#define LOOP_UNROLLING_50(idx, step, macro) LOOP_UNROLLING_49(idx, step, macro); UNROLL_INCR(idx, step, macro)
2749*c217d954SCole Faust#define LOOP_UNROLLING_51(idx, step, macro) LOOP_UNROLLING_50(idx, step, macro); UNROLL_INCR(idx, step, macro)
2750*c217d954SCole Faust#define LOOP_UNROLLING_52(idx, step, macro) LOOP_UNROLLING_51(idx, step, macro); UNROLL_INCR(idx, step, macro)
2751*c217d954SCole Faust#define LOOP_UNROLLING_53(idx, step, macro) LOOP_UNROLLING_52(idx, step, macro); UNROLL_INCR(idx, step, macro)
2752*c217d954SCole Faust#define LOOP_UNROLLING_54(idx, step, macro) LOOP_UNROLLING_53(idx, step, macro); UNROLL_INCR(idx, step, macro)
2753*c217d954SCole Faust#define LOOP_UNROLLING_55(idx, step, macro) LOOP_UNROLLING_54(idx, step, macro); UNROLL_INCR(idx, step, macro)
2754*c217d954SCole Faust#define LOOP_UNROLLING_56(idx, step, macro) LOOP_UNROLLING_55(idx, step, macro); UNROLL_INCR(idx, step, macro)
2755*c217d954SCole Faust#define LOOP_UNROLLING_57(idx, step, macro) LOOP_UNROLLING_56(idx, step, macro); UNROLL_INCR(idx, step, macro)
2756*c217d954SCole Faust#define LOOP_UNROLLING_58(idx, step, macro) LOOP_UNROLLING_57(idx, step, macro); UNROLL_INCR(idx, step, macro)
2757*c217d954SCole Faust#define LOOP_UNROLLING_59(idx, step, macro) LOOP_UNROLLING_58(idx, step, macro); UNROLL_INCR(idx, step, macro)
2758*c217d954SCole Faust#define LOOP_UNROLLING_60(idx, step, macro) LOOP_UNROLLING_59(idx, step, macro); UNROLL_INCR(idx, step, macro)
2759*c217d954SCole Faust#define LOOP_UNROLLING_61(idx, step, macro) LOOP_UNROLLING_60(idx, step, macro); UNROLL_INCR(idx, step, macro)
2760*c217d954SCole Faust#define LOOP_UNROLLING_62(idx, step, macro) LOOP_UNROLLING_61(idx, step, macro); UNROLL_INCR(idx, step, macro)
2761*c217d954SCole Faust#define LOOP_UNROLLING_63(idx, step, macro) LOOP_UNROLLING_62(idx, step, macro); UNROLL_INCR(idx, step, macro)
2762*c217d954SCole Faust#define LOOP_UNROLLING_64(idx, step, macro) LOOP_UNROLLING_63(idx, step, macro); UNROLL_INCR(idx, step, macro)
2763*c217d954SCole Faust#define LOOP_UNROLLING_65(idx, step, macro) LOOP_UNROLLING_64(idx, step, macro); UNROLL_INCR(idx, step, macro)
2764*c217d954SCole Faust#define LOOP_UNROLLING_66(idx, step, macro) LOOP_UNROLLING_65(idx, step, macro); UNROLL_INCR(idx, step, macro)
2765*c217d954SCole Faust#define LOOP_UNROLLING_67(idx, step, macro) LOOP_UNROLLING_66(idx, step, macro); UNROLL_INCR(idx, step, macro)
2766*c217d954SCole Faust#define LOOP_UNROLLING_68(idx, step, macro) LOOP_UNROLLING_67(idx, step, macro); UNROLL_INCR(idx, step, macro)
2767*c217d954SCole Faust#define LOOP_UNROLLING_69(idx, step, macro) LOOP_UNROLLING_68(idx, step, macro); UNROLL_INCR(idx, step, macro)
2768*c217d954SCole Faust#define LOOP_UNROLLING_70(idx, step, macro) LOOP_UNROLLING_69(idx, step, macro); UNROLL_INCR(idx, step, macro)
2769*c217d954SCole Faust#define LOOP_UNROLLING_71(idx, step, macro) LOOP_UNROLLING_70(idx, step, macro); UNROLL_INCR(idx, step, macro)
2770*c217d954SCole Faust#define LOOP_UNROLLING_72(idx, step, macro) LOOP_UNROLLING_71(idx, step, macro); UNROLL_INCR(idx, step, macro)
2771*c217d954SCole Faust#define LOOP_UNROLLING_73(idx, step, macro) LOOP_UNROLLING_72(idx, step, macro); UNROLL_INCR(idx, step, macro)
2772*c217d954SCole Faust#define LOOP_UNROLLING_74(idx, step, macro) LOOP_UNROLLING_73(idx, step, macro); UNROLL_INCR(idx, step, macro)
2773*c217d954SCole Faust#define LOOP_UNROLLING_75(idx, step, macro) LOOP_UNROLLING_74(idx, step, macro); UNROLL_INCR(idx, step, macro)
2774*c217d954SCole Faust#define LOOP_UNROLLING_76(idx, step, macro) LOOP_UNROLLING_75(idx, step, macro); UNROLL_INCR(idx, step, macro)
2775*c217d954SCole Faust#define LOOP_UNROLLING_77(idx, step, macro) LOOP_UNROLLING_76(idx, step, macro); UNROLL_INCR(idx, step, macro)
2776*c217d954SCole Faust#define LOOP_UNROLLING_78(idx, step, macro) LOOP_UNROLLING_77(idx, step, macro); UNROLL_INCR(idx, step, macro)
2777*c217d954SCole Faust#define LOOP_UNROLLING_79(idx, step, macro) LOOP_UNROLLING_78(idx, step, macro); UNROLL_INCR(idx, step, macro)
2778*c217d954SCole Faust#define LOOP_UNROLLING_80(idx, step, macro) LOOP_UNROLLING_79(idx, step, macro); UNROLL_INCR(idx, step, macro)
2779*c217d954SCole Faust#define LOOP_UNROLLING_81(idx, step, macro) LOOP_UNROLLING_80(idx, step, macro); UNROLL_INCR(idx, step, macro)
2780*c217d954SCole Faust#define LOOP_UNROLLING_82(idx, step, macro) LOOP_UNROLLING_81(idx, step, macro); UNROLL_INCR(idx, step, macro)
2781*c217d954SCole Faust#define LOOP_UNROLLING_83(idx, step, macro) LOOP_UNROLLING_82(idx, step, macro); UNROLL_INCR(idx, step, macro)
2782*c217d954SCole Faust#define LOOP_UNROLLING_84(idx, step, macro) LOOP_UNROLLING_83(idx, step, macro); UNROLL_INCR(idx, step, macro)
2783*c217d954SCole Faust#define LOOP_UNROLLING_85(idx, step, macro) LOOP_UNROLLING_84(idx, step, macro); UNROLL_INCR(idx, step, macro)
2784*c217d954SCole Faust#define LOOP_UNROLLING_86(idx, step, macro) LOOP_UNROLLING_85(idx, step, macro); UNROLL_INCR(idx, step, macro)
2785*c217d954SCole Faust#define LOOP_UNROLLING_87(idx, step, macro) LOOP_UNROLLING_86(idx, step, macro); UNROLL_INCR(idx, step, macro)
2786*c217d954SCole Faust#define LOOP_UNROLLING_88(idx, step, macro) LOOP_UNROLLING_87(idx, step, macro); UNROLL_INCR(idx, step, macro)
2787*c217d954SCole Faust#define LOOP_UNROLLING_89(idx, step, macro) LOOP_UNROLLING_88(idx, step, macro); UNROLL_INCR(idx, step, macro)
2788*c217d954SCole Faust#define LOOP_UNROLLING_90(idx, step, macro) LOOP_UNROLLING_89(idx, step, macro); UNROLL_INCR(idx, step, macro)
2789*c217d954SCole Faust#define LOOP_UNROLLING_91(idx, step, macro) LOOP_UNROLLING_90(idx, step, macro); UNROLL_INCR(idx, step, macro)
2790*c217d954SCole Faust#define LOOP_UNROLLING_92(idx, step, macro) LOOP_UNROLLING_91(idx, step, macro); UNROLL_INCR(idx, step, macro)
2791*c217d954SCole Faust#define LOOP_UNROLLING_93(idx, step, macro) LOOP_UNROLLING_92(idx, step, macro); UNROLL_INCR(idx, step, macro)
2792*c217d954SCole Faust#define LOOP_UNROLLING_94(idx, step, macro) LOOP_UNROLLING_93(idx, step, macro); UNROLL_INCR(idx, step, macro)
2793*c217d954SCole Faust#define LOOP_UNROLLING_95(idx, step, macro) LOOP_UNROLLING_94(idx, step, macro); UNROLL_INCR(idx, step, macro)
2794*c217d954SCole Faust#define LOOP_UNROLLING_96(idx, step, macro) LOOP_UNROLLING_95(idx, step, macro); UNROLL_INCR(idx, step, macro)
2795*c217d954SCole Faust#define LOOP_UNROLLING_97(idx, step, macro) LOOP_UNROLLING_96(idx, step, macro); UNROLL_INCR(idx, step, macro)
2796*c217d954SCole Faust#define LOOP_UNROLLING_98(idx, step, macro) LOOP_UNROLLING_97(idx, step, macro); UNROLL_INCR(idx, step, macro)
2797*c217d954SCole Faust#define LOOP_UNROLLING_99(idx, step, macro) LOOP_UNROLLING_98(idx, step, macro); UNROLL_INCR(idx, step, macro)
2798*c217d954SCole Faust#define LOOP_UNROLLING_100(idx, step, macro) LOOP_UNROLLING_99(idx, step, macro); UNROLL_INCR(idx, step, macro)
2799*c217d954SCole Faust#define LOOP_UNROLLING_101(idx, step, macro) LOOP_UNROLLING_100(idx, step, macro); UNROLL_INCR(idx, step, macro)
2800*c217d954SCole Faust#define LOOP_UNROLLING_102(idx, step, macro) LOOP_UNROLLING_101(idx, step, macro); UNROLL_INCR(idx, step, macro)
2801*c217d954SCole Faust#define LOOP_UNROLLING_103(idx, step, macro) LOOP_UNROLLING_102(idx, step, macro); UNROLL_INCR(idx, step, macro)
2802*c217d954SCole Faust#define LOOP_UNROLLING_104(idx, step, macro) LOOP_UNROLLING_103(idx, step, macro); UNROLL_INCR(idx, step, macro)
2803*c217d954SCole Faust#define LOOP_UNROLLING_105(idx, step, macro) LOOP_UNROLLING_104(idx, step, macro); UNROLL_INCR(idx, step, macro)
2804*c217d954SCole Faust#define LOOP_UNROLLING_106(idx, step, macro) LOOP_UNROLLING_105(idx, step, macro); UNROLL_INCR(idx, step, macro)
2805*c217d954SCole Faust#define LOOP_UNROLLING_107(idx, step, macro) LOOP_UNROLLING_106(idx, step, macro); UNROLL_INCR(idx, step, macro)
2806*c217d954SCole Faust#define LOOP_UNROLLING_108(idx, step, macro) LOOP_UNROLLING_107(idx, step, macro); UNROLL_INCR(idx, step, macro)
2807*c217d954SCole Faust#define LOOP_UNROLLING_109(idx, step, macro) LOOP_UNROLLING_108(idx, step, macro); UNROLL_INCR(idx, step, macro)
2808*c217d954SCole Faust#define LOOP_UNROLLING_110(idx, step, macro) LOOP_UNROLLING_109(idx, step, macro); UNROLL_INCR(idx, step, macro)
2809*c217d954SCole Faust#define LOOP_UNROLLING_111(idx, step, macro) LOOP_UNROLLING_110(idx, step, macro); UNROLL_INCR(idx, step, macro)
2810*c217d954SCole Faust#define LOOP_UNROLLING_112(idx, step, macro) LOOP_UNROLLING_111(idx, step, macro); UNROLL_INCR(idx, step, macro)
2811*c217d954SCole Faust#define LOOP_UNROLLING_113(idx, step, macro) LOOP_UNROLLING_112(idx, step, macro); UNROLL_INCR(idx, step, macro)
2812*c217d954SCole Faust#define LOOP_UNROLLING_114(idx, step, macro) LOOP_UNROLLING_113(idx, step, macro); UNROLL_INCR(idx, step, macro)
2813*c217d954SCole Faust#define LOOP_UNROLLING_115(idx, step, macro) LOOP_UNROLLING_114(idx, step, macro); UNROLL_INCR(idx, step, macro)
2814*c217d954SCole Faust#define LOOP_UNROLLING_116(idx, step, macro) LOOP_UNROLLING_115(idx, step, macro); UNROLL_INCR(idx, step, macro)
2815*c217d954SCole Faust#define LOOP_UNROLLING_117(idx, step, macro) LOOP_UNROLLING_116(idx, step, macro); UNROLL_INCR(idx, step, macro)
2816*c217d954SCole Faust#define LOOP_UNROLLING_118(idx, step, macro) LOOP_UNROLLING_117(idx, step, macro); UNROLL_INCR(idx, step, macro)
2817*c217d954SCole Faust#define LOOP_UNROLLING_119(idx, step, macro) LOOP_UNROLLING_118(idx, step, macro); UNROLL_INCR(idx, step, macro)
2818*c217d954SCole Faust#define LOOP_UNROLLING_120(idx, step, macro) LOOP_UNROLLING_119(idx, step, macro); UNROLL_INCR(idx, step, macro)
2819*c217d954SCole Faust#define LOOP_UNROLLING_121(idx, step, macro) LOOP_UNROLLING_120(idx, step, macro); UNROLL_INCR(idx, step, macro)
2820*c217d954SCole Faust#define LOOP_UNROLLING_122(idx, step, macro) LOOP_UNROLLING_121(idx, step, macro); UNROLL_INCR(idx, step, macro)
2821*c217d954SCole Faust#define LOOP_UNROLLING_123(idx, step, macro) LOOP_UNROLLING_122(idx, step, macro); UNROLL_INCR(idx, step, macro)
2822*c217d954SCole Faust#define LOOP_UNROLLING_124(idx, step, macro) LOOP_UNROLLING_123(idx, step, macro); UNROLL_INCR(idx, step, macro)
2823*c217d954SCole Faust#define LOOP_UNROLLING_125(idx, step, macro) LOOP_UNROLLING_124(idx, step, macro); UNROLL_INCR(idx, step, macro)
2824*c217d954SCole Faust#define LOOP_UNROLLING_126(idx, step, macro) LOOP_UNROLLING_125(idx, step, macro); UNROLL_INCR(idx, step, macro)
2825*c217d954SCole Faust#define LOOP_UNROLLING_127(idx, step, macro) LOOP_UNROLLING_126(idx, step, macro); UNROLL_INCR(idx, step, macro)
2826*c217d954SCole Faust#define LOOP_UNROLLING_128(idx, step, macro) LOOP_UNROLLING_127(idx, step, macro); UNROLL_INCR(idx, step, macro)
2827*c217d954SCole Faust
2828*c217d954SCole Faust#define LOOP_UNROLLING_STR(type, idx, start, step, num, macro) \
2829*c217d954SCole Faust    {                                                          \
2830*c217d954SCole Faust        type idx = start;                                      \
2831*c217d954SCole Faust        LOOP_UNROLLING_##num(idx, step, macro);                \
2832*c217d954SCole Faust    }
2833*c217d954SCole Faust#else
2834*c217d954SCole Faust#define LOOP_UNROLLING_STR(type, idx, start, step, num, macro) \
2835*c217d954SCole Faust    {                                                          \
2836*c217d954SCole Faust        _Pragma("unroll")                                      \
2837*c217d954SCole Faust        for(type idx = start; idx < (num * step); idx += step) \
2838*c217d954SCole Faust        {                                                      \
2839*c217d954SCole Faust            (macro);                                           \
2840*c217d954SCole Faust        }                                                      \
2841*c217d954SCole Faust    }
2842*c217d954SCole Faust#endif
2843*c217d954SCole Faust#define LOOP_UNROLLING(type, idx, start, step, num, macro) LOOP_UNROLLING_STR(type, idx, start, step, num, macro)
2844*c217d954SCole Faust
2845*c217d954SCole Faust
2846*c217d954SCole Faust#define GET_SPATIAL_IDX(IDX, N0, PARTIAL_N0) (max((int)(get_global_id(IDX) * N0 - (N0 - PARTIAL_N0) % N0), 0))
2847*c217d954SCole Faust
2848*c217d954SCole Faust
2849*c217d954SCole Faust#define DOT_PRODUCT_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, K0, a, b, c) DOT_PRODUCT_INTEGER8_STR(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, K0, a, b, c)
2850*c217d954SCole Faust#define DOT_PRODUCT_INTEGER8_STR(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, K0, a, b, c) DOT_PRODUCT##K0##_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c)
2851*c217d954SCole Faust#define DOT_PRODUCT1_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2852*c217d954SCole Faust    ({                                                \
2853*c217d954SCole Faust        c += (C_DATA_TYPE)(a) * (C_DATA_TYPE)(b);     \
2854*c217d954SCole Faust    })
2855*c217d954SCole Faust#if defined(ARM_COMPUTE_OPENCL_DOT8_ENABLED) && defined(cl_khr_integer_dot_product)
2856*c217d954SCole Faust#define DOT_PRODUCT2_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c += dot((A_DATA_TYPE##4)((a).s01, (A_DATA_TYPE##2)(0)), (B_DATA_TYPE##4)(((b).s01), (B_DATA_TYPE##2)(0)));
2857*c217d954SCole Faust#define DOT_PRODUCT3_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c += dot((A_DATA_TYPE##4)((a).s012, (A_DATA_TYPE)0), (B_DATA_TYPE##4)(((b).s012), (B_DATA_TYPE)0));
2858*c217d954SCole Faust#define DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c += dot((a), (b));
2859*c217d954SCole Faust#elif defined(ARM_COMPUTE_OPENCL_DOT8_ACC_ENABLED) && defined(cl_arm_integer_dot_product_accumulate_int8)
2860*c217d954SCole Faust#define DOT_PRODUCT2_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c = arm_dot_acc((A_DATA_TYPE##4)((a).s01, (A_DATA_TYPE##2)(0)), (B_DATA_TYPE##4)(((b).s01), (B_DATA_TYPE##2)(0)), (c));
2861*c217d954SCole Faust#define DOT_PRODUCT3_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c = arm_dot_acc((A_DATA_TYPE##4)((a).s012, (A_DATA_TYPE)0), (B_DATA_TYPE##4)(((b).s012), (B_DATA_TYPE)0), (c));
2862*c217d954SCole Faust#define DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c = arm_dot_acc((a), (b), (c));
2863*c217d954SCole Faust#elif defined(ARM_COMPUTE_OPENCL_DOT8_ENABLED) && defined(cl_arm_integer_dot_product_int8)
2864*c217d954SCole Faust#define DOT_PRODUCT2_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c += arm_dot((A_DATA_TYPE##4)((a).s01, (A_DATA_TYPE##2)(0)), (B_DATA_TYPE##4)(((b).s01), (B_DATA_TYPE##2)(0)));
2865*c217d954SCole Faust#define DOT_PRODUCT3_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c += arm_dot((A_DATA_TYPE##4)((a).s012, (A_DATA_TYPE)0), (B_DATA_TYPE##4)(((b).s012), (B_DATA_TYPE)0));
2866*c217d954SCole Faust#define DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) c += arm_dot((a), (b));
2867*c217d954SCole Faust#else
2868*c217d954SCole Faust#define DOT_PRODUCT2_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c)   \
2869*c217d954SCole Faust    ({                                                  \
2870*c217d954SCole Faust        c += (C_DATA_TYPE)(a).s0 * (C_DATA_TYPE)(b).s0; \
2871*c217d954SCole Faust        c += (C_DATA_TYPE)(a).s1 * (C_DATA_TYPE)(b).s1; \
2872*c217d954SCole Faust    })
2873*c217d954SCole Faust#define DOT_PRODUCT3_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c)   \
2874*c217d954SCole Faust    ({                                                  \
2875*c217d954SCole Faust        DOT_PRODUCT2_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c);  \
2876*c217d954SCole Faust        c += (C_DATA_TYPE)(a).s2 * (C_DATA_TYPE)(b).s2; \
2877*c217d954SCole Faust    })
2878*c217d954SCole Faust#define DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, x, y, val)   \
2879*c217d954SCole Faust    ({                                                    \
2880*c217d954SCole Faust        val += (C_DATA_TYPE)(x).s0 * (C_DATA_TYPE)(y).s0; \
2881*c217d954SCole Faust        val += (C_DATA_TYPE)(x).s1 * (C_DATA_TYPE)(y).s1; \
2882*c217d954SCole Faust        val += (C_DATA_TYPE)(x).s2 * (C_DATA_TYPE)(y).s2; \
2883*c217d954SCole Faust        val += (C_DATA_TYPE)(x).s3 * (C_DATA_TYPE)(y).s3; \
2884*c217d954SCole Faust    })
2885*c217d954SCole Faust#endif
2886*c217d954SCole Faust#define DOT_PRODUCT5_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2887*c217d954SCole Faust    ({                                                \
2888*c217d954SCole Faust        DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s0123), ((b).s0123), c);     \
2889*c217d954SCole Faust        DOT_PRODUCT1_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s4), ((b).s4), c);     \
2890*c217d954SCole Faust    })
2891*c217d954SCole Faust#define DOT_PRODUCT6_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2892*c217d954SCole Faust    ({                                                \
2893*c217d954SCole Faust        DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s0123), ((b).s0123), c);     \
2894*c217d954SCole Faust        DOT_PRODUCT2_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s45), ((b).s45), c);     \
2895*c217d954SCole Faust    })
2896*c217d954SCole Faust#define DOT_PRODUCT7_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2897*c217d954SCole Faust    ({                                                \
2898*c217d954SCole Faust        DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s0123), ((b).s0123), c);     \
2899*c217d954SCole Faust        DOT_PRODUCT3_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s456), ((b).s456), c);     \
2900*c217d954SCole Faust    })
2901*c217d954SCole Faust#define DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2902*c217d954SCole Faust    ({                                                \
2903*c217d954SCole Faust        DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).lo), ((b).lo), c);     \
2904*c217d954SCole Faust        DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).hi), ((b).hi), c);     \
2905*c217d954SCole Faust    })
2906*c217d954SCole Faust#define DOT_PRODUCT9_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2907*c217d954SCole Faust    ({                                                \
2908*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s01234567), ((b).s01234567), c);     \
2909*c217d954SCole Faust        DOT_PRODUCT1_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s8), ((b).s8), c);     \
2910*c217d954SCole Faust    })
2911*c217d954SCole Faust#define DOT_PRODUCT10_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2912*c217d954SCole Faust    ({                                                \
2913*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s01234567), ((b).s01234567), c);     \
2914*c217d954SCole Faust        DOT_PRODUCT2_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s89), ((b).s89), c);     \
2915*c217d954SCole Faust    })
2916*c217d954SCole Faust#define DOT_PRODUCT11_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2917*c217d954SCole Faust    ({                                                \
2918*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s01234567), ((b).s01234567), c);     \
2919*c217d954SCole Faust        DOT_PRODUCT3_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s89A), ((b).s89A), c);     \
2920*c217d954SCole Faust    })
2921*c217d954SCole Faust#define DOT_PRODUCT12_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2922*c217d954SCole Faust    ({                                                \
2923*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s01234567), ((b).s01234567), c);     \
2924*c217d954SCole Faust        DOT_PRODUCT4_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s89AB), ((b).s89AB), c);     \
2925*c217d954SCole Faust    })
2926*c217d954SCole Faust#define DOT_PRODUCT13_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2927*c217d954SCole Faust    ({                                                \
2928*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s01234567), ((b).s01234567), c);     \
2929*c217d954SCole Faust        DOT_PRODUCT5_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s89ABC), ((b).s89ABC), c);     \
2930*c217d954SCole Faust    })
2931*c217d954SCole Faust#define DOT_PRODUCT14_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2932*c217d954SCole Faust    ({                                                \
2933*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s01234567), ((b).s01234567), c);     \
2934*c217d954SCole Faust        DOT_PRODUCT6_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s89ABCD), ((b).s89ABCD), c);     \
2935*c217d954SCole Faust    })
2936*c217d954SCole Faust#define DOT_PRODUCT15_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2937*c217d954SCole Faust    ({                                                \
2938*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s01234567), ((b).s01234567), c);     \
2939*c217d954SCole Faust        DOT_PRODUCT7_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).s89ABCDE), ((b).s89ABCDE), c);     \
2940*c217d954SCole Faust    })
2941*c217d954SCole Faust#define DOT_PRODUCT16_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, a, b, c) \
2942*c217d954SCole Faust    ({                                                 \
2943*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).lo), ((b).lo), c);      \
2944*c217d954SCole Faust        DOT_PRODUCT8_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, ((a).hi), ((b).hi), c);      \
2945*c217d954SCole Faust    })
2946*c217d954SCole Faust
2947*c217d954SCole Faust
2948*c217d954SCole Faust#define REDUCE_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, K0, a, c) REDUCE_INTEGER8_STR(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, K0, a, c)
2949*c217d954SCole Faust#define REDUCE_INTEGER8_STR(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, K0, a, c) DOT_PRODUCT_INTEGER8(A_DATA_TYPE, B_DATA_TYPE, C_DATA_TYPE, K0, a, (TILE_VECTOR_TYPE##K0(B_DATA_TYPE))1, c)
2950*c217d954SCole Faust
2951*c217d954SCole Faust
2952*c217d954SCole Faust#define V_LOAD(DATA_TYPE, WIDTH, TENSOR_TYPE, TENSOR, X, Y, STRIDE_Y) V_LOAD_STR(DATA_TYPE, WIDTH, TENSOR_TYPE, TENSOR, X, Y, STRIDE_Y)
2953*c217d954SCole Faust#define V_LOAD_STR(DATA_TYPE, WIDTH, TENSOR_TYPE, TENSOR, X, Y, STRIDE_Y) V_LOAD_##TENSOR_TYPE(DATA_TYPE, WIDTH, TENSOR, X, Y, STRIDE_Y)
2954*c217d954SCole Faust#define V_LOAD_BUFFER(DATA_TYPE, WIDTH, TENSOR, X, Y, STRIDE_Y) \
2955*c217d954SCole Faust    VLOAD(WIDTH)                                                \
2956*c217d954SCole Faust    (0, (__global DATA_TYPE *)(TENSOR##_ptr + TENSOR##_offset_first_element_in_bytes + (X) * sizeof(DATA_TYPE) + (Y) * (STRIDE_Y)))
2957*c217d954SCole Faust#define V_LOAD_IMAGE(DATA_TYPE, WIDTH, TENSOR, X, Y, STRIDE_Y) READ_IMAGE2D(DATA_TYPE, CONVERT_VECTOR_SIZE_TO_PIXEL_UNIT(WIDTH), TENSOR##_img, (X) / 4, (Y))
2958*c217d954SCole Faust
2959*c217d954SCole Faust
2960*c217d954SCole Faust#define V_STORE(DATA_TYPE, WIDTH, TENSOR_TYPE, TENSOR, X, Y, STRIDE_Y, VALUES) V_STORE_STR(DATA_TYPE, WIDTH, TENSOR_TYPE, TENSOR, X, Y, STRIDE_Y, VALUES)
2961*c217d954SCole Faust#define V_STORE_STR(DATA_TYPE, WIDTH, TENSOR_TYPE, TENSOR, X, Y, STRIDE_Y, VALUES) V_STORE_##TENSOR_TYPE(DATA_TYPE, WIDTH, TENSOR, X, Y, STRIDE_Y, VALUES)
2962*c217d954SCole Faust#define V_STORE_BUFFER(DATA_TYPE, WIDTH, TENSOR, X, Y, STRIDE_Y, VALUES) \
2963*c217d954SCole Faust    VSTORE(WIDTH)                                                \
2964*c217d954SCole Faust    (VALUES, 0, (__global DATA_TYPE *)(TENSOR##_ptr + TENSOR##_offset_first_element_in_bytes + (X) * sizeof(DATA_TYPE) + (Y) * (STRIDE_Y)))
2965*c217d954SCole Faust#define V_STORE_IMAGE(DATA_TYPE, WIDTH, TENSOR, X, Y, STRIDE_Y, VALUES) WRITE_IMAGE2D(DATA_TYPE, CONVERT_VECTOR_SIZE_TO_PIXEL_UNIT(WIDTH), TENSOR##_img, (X) / 4, (Y), VALUES)
2966*c217d954SCole Faust
2967*c217d954SCole Faust
2968*c217d954SCole Faust#define T_LOAD(DATA_TYPE, HEIGHT, WIDTH, TENSOR_TYPE, TENSOR, X, Y, YI_MULTIPLIER, STRIDE_Y, dst)                      \
2969*c217d954SCole Faust    ({                                                                                                                 \
2970*c217d954SCole Faust        LOOP_UNROLLING(int, _i, 0, 1, HEIGHT,                                                                          \
2971*c217d954SCole Faust        {                                                                                                              \
2972*c217d954SCole Faust            dst[_i].v = V_LOAD(DATA_TYPE, WIDTH, TENSOR_TYPE, TENSOR, X, ((Y) + _i * (int)(YI_MULTIPLIER)), STRIDE_Y); \
2973*c217d954SCole Faust        })                                                                                                             \
2974*c217d954SCole Faust    })
2975*c217d954SCole Faust
2976*c217d954SCole Faust
2977*c217d954SCole Faust#define T_LOAD_INDIRECT(DATA_TYPE, HEIGHT, WIDTH, TENSOR_TYPE, TENSOR, X, STRIDE_Y, indirect_y, dst)    \
2978*c217d954SCole Faust    ({                                                                                                  \
2979*c217d954SCole Faust        LOOP_UNROLLING(int, _i, 0, 1, HEIGHT,                                                           \
2980*c217d954SCole Faust        {                                                                                               \
2981*c217d954SCole Faust            dst[_i].v = V_LOAD(DATA_TYPE, WIDTH, TENSOR_TYPE, TENSOR, X, (indirect_y[_i].v), STRIDE_Y); \
2982*c217d954SCole Faust        })                                                                                              \
2983*c217d954SCole Faust    })
2984*c217d954SCole Faust
2985*c217d954SCole Faust
2986*c217d954SCole Faust#define T_LOAD_INDIRECT_WIDTH_SELECT(DATA_TYPE, HEIGHT, WIDTH0, WIDTH1, TENSOR_TYPE, TENSOR, X, STRIDE_Y, WIDTH1_CONDITION, dst, indirect_y)                                                      \
2987*c217d954SCole Faust    ({                                                                                                                                                                                             \
2988*c217d954SCole Faust        if(WIDTH1_CONDITION)                                                                                                                                                                       \
2989*c217d954SCole Faust        {                                                                                                                                                                                          \
2990*c217d954SCole Faust            LOOP_UNROLLING(int, _i, 0, 1, HEIGHT,                                                                                                                                                  \
2991*c217d954SCole Faust            {                                                                                                                                                                                      \
2992*c217d954SCole Faust                VLOAD_PARTIAL(WIDTH0, WIDTH1)                                                         \
2993*c217d954SCole Faust                (dst[HEIGHT - 1 - _i].v, 0, (__global DATA_TYPE *)(TENSOR##_ptr + TENSOR##_offset_first_element_in_bytes + (X) * sizeof(DATA_TYPE) + (indirect_y[HEIGHT - 1 - _i].v) * STRIDE_Y));               \
2994*c217d954SCole Faust            })                                                                                                                                                                                     \
2995*c217d954SCole Faust        }                                                                                                                                                                                          \
2996*c217d954SCole Faust        else                                                                                                                                                                                       \
2997*c217d954SCole Faust        {                                                                                                                                                                                          \
2998*c217d954SCole Faust            LOOP_UNROLLING(int, _i, 0, 1, HEIGHT,                                                                                                                                                  \
2999*c217d954SCole Faust            {                                                                                                                                                                                      \
3000*c217d954SCole Faust                dst[HEIGHT - 1 - _i].v = V_LOAD(DATA_TYPE, WIDTH0, TENSOR_TYPE, TENSOR, X, (indirect_y[HEIGHT - 1 - _i].v), STRIDE_Y); \
3001*c217d954SCole Faust            })                                                                                                                                                                                     \
3002*c217d954SCole Faust        }                                                                                                                                                                                          \
3003*c217d954SCole Faust    })
3004*c217d954SCole Faust
3005*c217d954SCole Faust#define T_LOAD_NHWC(DATA_TYPE, TILE_HEIGHT, TILE_WIDTH, TILE_CHANNELS, TENSOR_TYPE, TENSOR, B, Y, X, C, TENSOR_WIDTH, TENSOR_HEIGHT, STRIDE_Y, dst)   \
3006*c217d954SCole Faust    ({                                                                                                                                                \
3007*c217d954SCole Faust        LOOP_UNROLLING(int, _yk, 0, 1, TILE_HEIGHT,                                                                                                   \
3008*c217d954SCole Faust        {                                                                                                                                             \
3009*c217d954SCole Faust            LOOP_UNROLLING(int, _xk, 0, 1, TILE_WIDTH,                                                                                                \
3010*c217d954SCole Faust            {                                                                                                                                         \
3011*c217d954SCole Faust                int _src_y = (X) + _xk + ((Y) + _yk) * (TENSOR_WIDTH);                                                                                \
3012*c217d954SCole Faust                _src_y    += (B) * (int)(TENSOR_WIDTH) * (int)(TENSOR_HEIGHT);                                                                        \
3013*c217d954SCole Faust                int _src_valid_y = (((X) + _xk) >= 0 && ((X) + _xk) < (int)(TENSOR_WIDTH) && ((Y) + _yk) >= 0 && ((Y) + _yk) < (int)(TENSOR_HEIGHT)); \
3014*c217d954SCole Faust                if(_src_valid_y != 0)                                                                                                                 \
3015*c217d954SCole Faust                {                                                                                                                                     \
3016*c217d954SCole Faust                    dst[_xk + _yk * (TILE_WIDTH)].v = V_LOAD(DATA_TYPE, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, _src_y, STRIDE_Y);                     \
3017*c217d954SCole Faust                }                                                                                                                                     \
3018*c217d954SCole Faust            })                                                                                                                                        \
3019*c217d954SCole Faust        })                                                                                                                                            \
3020*c217d954SCole Faust    })
3021*c217d954SCole Faust
3022*c217d954SCole Faust
3023*c217d954SCole Faust#define T_LOAD_NHWC_WITH_DILATION(DATA_TYPE, TILE_HEIGHT, TILE_WIDTH, TILE_CHANNELS, TENSOR_TYPE, TENSOR, B, Y, X, C, TENSOR_WIDTH, TENSOR_HEIGHT, DILATION_X, DILATION_Y, BOUNDARY_CHECK, dst)         \
3024*c217d954SCole Faust    ({ \
3025*c217d954SCole Faust        LOOP_UNROLLING(int, _yk, 0, 1, TILE_HEIGHT, \
3026*c217d954SCole Faust        { \
3027*c217d954SCole Faust            LOOP_UNROLLING(int, _xk, 0, 1, TILE_WIDTH, \
3028*c217d954SCole Faust            { \
3029*c217d954SCole Faust                int _src_y = (X) + _xk * (DILATION_X); \
3030*c217d954SCole Faust                int _src_z = ((Y) + _yk * (DILATION_Y)); \
3031*c217d954SCole Faust                int _src_w    = (B); \
3032*c217d954SCole Faust                bool _src_valid_y = (((X) + _xk * (DILATION_X)) >= 0) && (((X) + _xk * (DILATION_X)) < (int)(TENSOR_WIDTH)) && (((Y) + _yk * (DILATION_Y)) >= 0) && (((Y) + _yk * (DILATION_Y)) < (int)(TENSOR_HEIGHT)); \
3033*c217d954SCole Faust                if(!(BOUNDARY_CHECK)) \
3034*c217d954SCole Faust                { \
3035*c217d954SCole Faust                    dst[_xk + _yk * (TILE_WIDTH)].v = VLOAD(TILE_CHANNELS)                                                \
3036*c217d954SCole Faust                    (0, (__global DATA_TYPE *)(TENSOR##_ptr + TENSOR##_offset_first_element_in_bytes + (C) * sizeof(DATA_TYPE) + (_src_y) * (TENSOR##_stride_y) + (_src_z) * (TENSOR##_stride_z) + (_src_w) * (TENSOR##_stride_w))); \
3037*c217d954SCole Faust                } \
3038*c217d954SCole Faust                else \
3039*c217d954SCole Faust                { \
3040*c217d954SCole Faust                    if(_src_valid_y) \
3041*c217d954SCole Faust                    { \
3042*c217d954SCole Faust                        dst[_xk + _yk * (TILE_WIDTH)].v = VLOAD(TILE_CHANNELS)                                                \
3043*c217d954SCole Faust                    (0, (__global DATA_TYPE *)(TENSOR##_ptr + TENSOR##_offset_first_element_in_bytes + (C) * sizeof(DATA_TYPE) + (_src_y) * (TENSOR##_stride_y) + (_src_z) * (TENSOR##_stride_z) + (_src_w) * (TENSOR##_stride_w))); \
3044*c217d954SCole Faust                    }                                                                                                                                                                                                 \
3045*c217d954SCole Faust                } \
3046*c217d954SCole Faust            })                                                                                                                                                                                                             \
3047*c217d954SCole Faust        })                                                                                                                                                                                                             \
3048*c217d954SCole Faust    })
3049*c217d954SCole Faust
3050*c217d954SCole Faust
3051*c217d954SCole Faust#define T_LOAD_NHWC_INDIRECT(DATA_TYPE, TILE_AREA, TILE_CHANNELS, TENSOR_TYPE, TENSOR, B, Y, X, C, TENSOR_WIDTH, TENSOR_HEIGHT, STRIDE_Y, xi, yi, dst)                \
3052*c217d954SCole Faust    ({                                                                                                                                                                \
3053*c217d954SCole Faust        LOOP_UNROLLING(int, _i, 0, 1, TILE_AREA,                                                                                                                      \
3054*c217d954SCole Faust        {                                                                                                                                                             \
3055*c217d954SCole Faust            int _src_y = (X) + xi[_i].v + ((Y) + yi[_i].v) * (TENSOR_WIDTH);                                                                                          \
3056*c217d954SCole Faust            _src_y += (B) * (int)(TENSOR_WIDTH) * (int)(TENSOR_HEIGHT);                                                                                               \
3057*c217d954SCole Faust            int _src_valid_y = (((X) + xi[_i].v) >= 0 && ((X) + xi[_i].v) < (int)(TENSOR_WIDTH) && ((Y) + yi[_i].v) >= 0 && ((Y) + yi[_i].v) < (int)(TENSOR_HEIGHT)); \
3058*c217d954SCole Faust            if(_src_valid_y != 0)                                                                                                                                     \
3059*c217d954SCole Faust            {                                                                                                                                                         \
3060*c217d954SCole Faust                dst[_i].v = V_LOAD(DATA_TYPE, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, _src_y, STRIDE_Y);                                                               \
3061*c217d954SCole Faust            }                                                                                                                                                         \
3062*c217d954SCole Faust        })                                                                                                                                                            \
3063*c217d954SCole Faust    })
3064*c217d954SCole Faust
3065*c217d954SCole Faust
3066*c217d954SCole Faust#define T_LOAD2D_INDIRECT(DATA_TYPE, TILE_AREA, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, STRIDE_Y, yi, dst) T_LOAD2D_INDIRECT_STR(DATA_TYPE, TILE_AREA, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, STRIDE_Y, yi, dst)
3067*c217d954SCole Faust#define T_LOAD2D_INDIRECT_STR(DATA_TYPE, TILE_AREA, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, STRIDE_Y, yi, dst) T_LOAD2D_INDIRECT_##TENSOR_TYPE(DATA_TYPE, TILE_AREA, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, STRIDE_Y, yi, dst)
3068*c217d954SCole Faust#define T_LOAD2D_INDIRECT_BUFFER(DATA_TYPE, TILE_AREA, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, STRIDE_Y, yi, dst) \
3069*c217d954SCole Faust    ({ \
3070*c217d954SCole Faust        LOOP_UNROLLING(int, _i, 0, 1, TILE_AREA, \
3071*c217d954SCole Faust        { \
3072*c217d954SCole Faust            if(yi[0].s[_i] >= 0) \
3073*c217d954SCole Faust            { \
3074*c217d954SCole Faust                dst[_i].v = V_LOAD(DATA_TYPE, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, yi[0].s[_i], STRIDE_Y); \
3075*c217d954SCole Faust            } \
3076*c217d954SCole Faust        }) \
3077*c217d954SCole Faust    })
3078*c217d954SCole Faust
3079*c217d954SCole Faust#define T_LOAD2D_INDIRECT_IMAGE(DATA_TYPE, TILE_AREA, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, STRIDE_Y, yi, dst) \
3080*c217d954SCole Faust    ({ \
3081*c217d954SCole Faust        LOOP_UNROLLING(int, _i, 0, 1, TILE_AREA, \
3082*c217d954SCole Faust        { \
3083*c217d954SCole Faust            dst[_i].v = V_LOAD(DATA_TYPE, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, yi[0].s[_i], STRIDE_Y); \
3084*c217d954SCole Faust        }) \
3085*c217d954SCole Faust    })
3086*c217d954SCole Faust
3087*c217d954SCole Faust
3088*c217d954SCole Faust#define T_LOAD_NDHWC_INDIRECT(DATA_TYPE, TILE_AREA, TILE_CHANNELS, TENSOR_TYPE, TENSOR, B, Z, Y, X, C, TENSOR_WIDTH, TENSOR_HEIGHT, TENSOR_DEPTH, STRIDE_Y, xi, yi, zi, dst) \
3089*c217d954SCole Faust    ({                                                                                                                                                                \
3090*c217d954SCole Faust        LOOP_UNROLLING(int, _i, 0, 1, TILE_AREA,                                                                                                                      \
3091*c217d954SCole Faust        {                                                                                                                                                             \
3092*c217d954SCole Faust            int _src_y = (X) + xi[_i].v + ((Y) + yi[_i].v) * (TENSOR_WIDTH) + ((Z) + zi[_i].v) * (TENSOR_WIDTH * TENSOR_HEIGHT);                                      \
3093*c217d954SCole Faust            _src_y += (B) * (int)(TENSOR_WIDTH) * (int)(TENSOR_HEIGHT) * (int)(TENSOR_DEPTH);                                                                         \
3094*c217d954SCole Faust            int _src_valid_y = (((X) + xi[_i].v) >= 0 && ((X) + xi[_i].v) < (int)(TENSOR_WIDTH) && ((Y) + yi[_i].v) >= 0 && ((Y) + yi[_i].v) < (int)(TENSOR_HEIGHT)   \
3095*c217d954SCole Faust                             && ((Z) + zi[_i].v) >= 0 && ((Z) + zi[_i].v) < (int)(TENSOR_DEPTH));                                                                     \
3096*c217d954SCole Faust            if(_src_valid_y != 0)                                                                                                                                     \
3097*c217d954SCole Faust            {                                                                                                                                                         \
3098*c217d954SCole Faust                dst[_i].v = V_LOAD(DATA_TYPE, TILE_CHANNELS, TENSOR_TYPE, TENSOR, C, _src_y, STRIDE_Y);                                                               \
3099*c217d954SCole Faust            }                                                                                                                                                         \
3100*c217d954SCole Faust        })                                                                                                                                                            \
3101*c217d954SCole Faust    })
3102*c217d954SCole Faust
3103*c217d954SCole Faust
3104*c217d954SCole Faust#define T_STORE_INDIRECT_WIDTH_SELECT(DATA_TYPE, HEIGHT, WIDTH0, WIDTH1, TENSOR_TYPE, TENSOR, X, STRIDE_Y, WIDTH1_CONDITION, src, indirect_y)                                                      \
3105*c217d954SCole Faust    ({                                                                                                                                                                                             \
3106*c217d954SCole Faust        if(WIDTH1_CONDITION)                                                                                                                                                                       \
3107*c217d954SCole Faust        {                                                                                                                                                                                          \
3108*c217d954SCole Faust            LOOP_UNROLLING(int, _i, 0, 1, HEIGHT,                                                                                                                                                  \
3109*c217d954SCole Faust            {                                                                                                                                                                                      \
3110*c217d954SCole Faust                VSTORE_PARTIAL(WIDTH0, WIDTH1)                                                                                                                                                     \
3111*c217d954SCole Faust                (CONVERT(src[HEIGHT - 1 - _i].v, VEC_DATA_TYPE(DATA_TYPE, WIDTH0)), 0, (__global DATA_TYPE *)(TENSOR##_ptr + TENSOR##_offset_first_element_in_bytes + (X) * sizeof(DATA_TYPE) + (indirect_y[HEIGHT - 1 - _i].v) * STRIDE_Y)); \
3112*c217d954SCole Faust            })                                                                                                                                                                                     \
3113*c217d954SCole Faust        }                                                                                                                                                                                          \
3114*c217d954SCole Faust        else                                                                                                                                                                                       \
3115*c217d954SCole Faust        {                                                                                                                                                                                          \
3116*c217d954SCole Faust            LOOP_UNROLLING(int, _i, 0, 1, HEIGHT,                                                                                                                                                  \
3117*c217d954SCole Faust            {                                                                                                                                                                                      \
3118*c217d954SCole Faust                VSTORE(WIDTH0)                                                                                                                                                                     \
3119*c217d954SCole Faust                (CONVERT(src[HEIGHT - 1 - _i].v, VEC_DATA_TYPE(DATA_TYPE, WIDTH0)), 0, (__global DATA_TYPE *)(TENSOR##_ptr + TENSOR##_offset_first_element_in_bytes + (X) * sizeof(DATA_TYPE) + (indirect_y[HEIGHT - 1 - _i].v) * STRIDE_Y)); \
3120*c217d954SCole Faust            })                                                                                                                                                                                     \
3121*c217d954SCole Faust        }                                                                                                                                                                                          \
3122*c217d954SCole Faust    })
3123*c217d954SCole Faust
3124*c217d954SCole Faust
3125*c217d954SCole Faust#define T_OFFSET_CORRECTION(ACC_DATA_TYPE, M0, N0, K0, SRC_OFFSET, WEI_OFFSET, lhs, rhs, dst)        \
3126*c217d954SCole Faust    ({                                                                                               \
3127*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                                                           \
3128*c217d954SCole Faust        {                                                                                            \
3129*c217d954SCole Faust            ACC_DATA_TYPE _tm = 0;                                                                   \
3130*c217d954SCole Faust            LOOP_UNROLLING(int, _k0, 0, 1, K0,                                                       \
3131*c217d954SCole Faust            {                                                                                        \
3132*c217d954SCole Faust                _tm += ((ACC_DATA_TYPE)lhs[_m0].s[_k0] * (ACC_DATA_TYPE)WEI_OFFSET);                 \
3133*c217d954SCole Faust            })                                                                                       \
3134*c217d954SCole Faust            LOOP_UNROLLING(int, _n0, 0, 1, N0,                                                       \
3135*c217d954SCole Faust            {                                                                                        \
3136*c217d954SCole Faust                dst[_m0].s[_n0] += _tm;                                                              \
3137*c217d954SCole Faust                LOOP_UNROLLING(int, _k0, 0, 1, K0,                                                   \
3138*c217d954SCole Faust                {                                                                                    \
3139*c217d954SCole Faust                    dst[_m0].s[_n0] += ((ACC_DATA_TYPE)rhs[_n0].s[_k0] * (ACC_DATA_TYPE)SRC_OFFSET); \
3140*c217d954SCole Faust                })                                                                                   \
3141*c217d954SCole Faust            })                                                                                       \
3142*c217d954SCole Faust        })                                                                                          \
3143*c217d954SCole Faust    })
3144*c217d954SCole Faust
3145*c217d954SCole Faust
3146*c217d954SCole Faust#define T_QUANTIZE8(SRC_DATA_TYPE, DST_DATA_TYPE, QUANTIZATION_TYPE, M0, N0, DST_OFFSET, DST_SHIFT, DST_MULTIPLIER, src, dst_multipliers, dst_shifts, dst) T_QUANTIZE8_STR(SRC_DATA_TYPE, DST_DATA_TYPE, QUANTIZATION_TYPE, M0, N0, DST_OFFSET, DST_SHIFT, DST_MULTIPLIER, src, dst_multipliers, dst_shifts, dst)
3147*c217d954SCole Faust#define T_QUANTIZE8_STR(SRC_DATA_TYPE, DST_DATA_TYPE, QUANTIZATION_TYPE, M0, N0, DST_OFFSET, DST_SHIFT, DST_MULTIPLIER, src, dst_multipliers, dst_shifts, dst) T_QUANTIZE8_##QUANTIZATION_TYPE(SRC_DATA_TYPE, DST_DATA_TYPE, M0, N0, DST_OFFSET, DST_SHIFT, DST_MULTIPLIER, src, dst_multipliers, dst_shifts, dst)
3148*c217d954SCole Faust
3149*c217d954SCole Faust
3150*c217d954SCole Faust#define T_QUANTIZE8_PER_TENSOR(SRC_DATA_TYPE, DST_DATA_TYPE, M0, N0, DST_OFFSET, DST_SHIFT, DST_MULTIPLIER, src, dst_multipliers, dst_shifts, dst)                          \
3151*c217d954SCole Faust    ({ \
3152*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0, \
3153*c217d954SCole Faust        { \
3154*c217d954SCole Faust            LOOP_UNROLLING(int, _n0, 0, 1, N0, \
3155*c217d954SCole Faust            { \
3156*c217d954SCole Faust                SRC_DATA_TYPE _tmp = 0; \
3157*c217d954SCole Faust                SRC_DATA_TYPE _src = src[_m0].s[_n0]; \
3158*c217d954SCole Faust                _src *= select((SRC_DATA_TYPE)1, ((SRC_DATA_TYPE)1 << (SRC_DATA_TYPE)(-DST_SHIFT)), ((SRC_DATA_TYPE)DST_SHIFT < (SRC_DATA_TYPE)0)); \
3159*c217d954SCole Faust                SRC_DATA_TYPE overflow = _src == DST_MULTIPLIER && _src == INT_MIN; \
3160*c217d954SCole Faust                long a_64 = (long)(_src); \
3161*c217d954SCole Faust                long b_64 = (long)(DST_MULTIPLIER); \
3162*c217d954SCole Faust                long ab_64 = a_64 * b_64; \
3163*c217d954SCole Faust                long mask1 = 1 << 30; \
3164*c217d954SCole Faust                long mask2 = 1 - (1 << 30); \
3165*c217d954SCole Faust                long is_positive_or_zero = ab_64 >= 0; \
3166*c217d954SCole Faust                long nudge = select(mask2, mask1, is_positive_or_zero); \
3167*c217d954SCole Faust                SRC_DATA_TYPE ab_x2_high32 = CONVERT((ab_64 + nudge) / (long)(1ll << 31), SRC_DATA_TYPE); \
3168*c217d954SCole Faust                _tmp = select(ab_x2_high32, (SRC_DATA_TYPE)INT_MAX, overflow); \
3169*c217d954SCole Faust                if(DST_SHIFT >= 0) \
3170*c217d954SCole Faust                { \
3171*c217d954SCole Faust                    long mask = ((((int)1) << DST_SHIFT) - (long)1); \
3172*c217d954SCole Faust                    long threshold = _tmp < (int)0 ? (mask >> 1) + (long)1 : (mask >> 1) + 0; \
3173*c217d954SCole Faust                    _tmp = (_tmp & mask) > threshold ? (_tmp >> DST_SHIFT) + (int)1 : (_tmp >> DST_SHIFT); \
3174*c217d954SCole Faust                } \
3175*c217d954SCole Faust                _tmp += DST_OFFSET; \
3176*c217d954SCole Faust                dst[_m0].s[_n0] = CONVERT_SAT(_tmp, DST_DATA_TYPE);                                                                            \
3177*c217d954SCole Faust            })                                                                                                                                          \
3178*c217d954SCole Faust        })                                                                                                                                          \
3179*c217d954SCole Faust    })
3180*c217d954SCole Faust
3181*c217d954SCole Faust
3182*c217d954SCole Faust#define T_QUANTIZE8_PER_CHANNEL(SRC_DATA_TYPE, DST_DATA_TYPE, M0, N0, DST_OFFSET, DST_SHIFT, DST_MULTIPLIER, src, dst_multipliers, dst_shifts, dst)                          \
3183*c217d954SCole Faust    ({ \
3184*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0, \
3185*c217d954SCole Faust        { \
3186*c217d954SCole Faust            LOOP_UNROLLING(int, _n0, 0, 1, N0, \
3187*c217d954SCole Faust            { \
3188*c217d954SCole Faust                SRC_DATA_TYPE _tmp = 0; \
3189*c217d954SCole Faust                SRC_DATA_TYPE _tmp2 = 0; \
3190*c217d954SCole Faust                SRC_DATA_TYPE _src = src[_m0].s[_n0]; \
3191*c217d954SCole Faust                SRC_DATA_TYPE _dst_multiplier = dst_multipliers[0].s[_n0]; \
3192*c217d954SCole Faust                SRC_DATA_TYPE _dst_shift = dst_shifts[0].s[_n0]; \
3193*c217d954SCole Faust                _src *= select((SRC_DATA_TYPE)1, ((SRC_DATA_TYPE)1 << (SRC_DATA_TYPE)(-_dst_shift)), ((SRC_DATA_TYPE)_dst_shift < (SRC_DATA_TYPE)0)); \
3194*c217d954SCole Faust                SRC_DATA_TYPE overflow = _src == _dst_multiplier && _src == INT_MIN; \
3195*c217d954SCole Faust                long a_64 = (long)(_src); \
3196*c217d954SCole Faust                long b_64 = (long)(_dst_multiplier); \
3197*c217d954SCole Faust                long ab_64 = a_64 * b_64; \
3198*c217d954SCole Faust                long mask1 = 1 << 30; \
3199*c217d954SCole Faust                long mask2 = 1 - (1 << 30); \
3200*c217d954SCole Faust                long is_positive_or_zero = ab_64 >= 0; \
3201*c217d954SCole Faust                long nudge = select(mask2, mask1, is_positive_or_zero); \
3202*c217d954SCole Faust                SRC_DATA_TYPE ab_x2_high32 = CONVERT((ab_64 + nudge) / (long)(1ll << 31), SRC_DATA_TYPE); \
3203*c217d954SCole Faust                _tmp = select(ab_x2_high32, (SRC_DATA_TYPE)INT_MAX, overflow); \
3204*c217d954SCole Faust                long mask = ((((int)1) << _dst_shift) - (int)1); \
3205*c217d954SCole Faust                long threshold = (mask >> 1) + any(_tmp); \
3206*c217d954SCole Faust                _tmp2 = _tmp >> _dst_shift; \
3207*c217d954SCole Faust                _tmp2 += select(0, 1, (_tmp & mask) > threshold); \
3208*c217d954SCole Faust                _tmp = select(_tmp, _tmp2, _dst_shift >= 0); \
3209*c217d954SCole Faust                _tmp += DST_OFFSET; \
3210*c217d954SCole Faust                dst[_m0].s[_n0] = CONVERT_SAT(_tmp, DST_DATA_TYPE);                                                                            \
3211*c217d954SCole Faust            })                                                                                                                                          \
3212*c217d954SCole Faust        })                                                                                                                                         \
3213*c217d954SCole Faust    })
3214*c217d954SCole Faust
3215*c217d954SCole Faust
3216*c217d954SCole Faust#define T_QUANTIZE8_ASYMMETRIC(SRC_DATA_TYPE, DST_DATA_TYPE, M0, N0, DST_OFFSET, DST_SHIFT, DST_MULTIPLIER, src, dst)                          \
3217*c217d954SCole Faust    ({ \
3218*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0, \
3219*c217d954SCole Faust        { \
3220*c217d954SCole Faust            LOOP_UNROLLING(int, _n0, 0, 1, N0, \
3221*c217d954SCole Faust            { \
3222*c217d954SCole Faust                SRC_DATA_TYPE _tmp = 0; \
3223*c217d954SCole Faust                SRC_DATA_TYPE _src = src[_m0].s[_n0]; \
3224*c217d954SCole Faust                _src *= select((SRC_DATA_TYPE)1, ((SRC_DATA_TYPE)1 << (SRC_DATA_TYPE)(-DST_SHIFT)), ((SRC_DATA_TYPE)DST_SHIFT < (SRC_DATA_TYPE)0)); \
3225*c217d954SCole Faust                SRC_DATA_TYPE overflow = _src == DST_MULTIPLIER && _src == INT_MIN; \
3226*c217d954SCole Faust                long a_64 = (long)(_src); \
3227*c217d954SCole Faust                long b_64 = (long)(DST_MULTIPLIER); \
3228*c217d954SCole Faust                long ab_64 = a_64 * b_64; \
3229*c217d954SCole Faust                long mask1 = 1 << 30; \
3230*c217d954SCole Faust                long mask2 = 1 - (1 << 30); \
3231*c217d954SCole Faust                long is_positive_or_zero = ab_64 >= 0; \
3232*c217d954SCole Faust                long nudge = select(mask2, mask1, is_positive_or_zero); \
3233*c217d954SCole Faust                SRC_DATA_TYPE ab_x2_high32 = CONVERT((ab_64 + nudge) / (long)(1ll << 31), SRC_DATA_TYPE); \
3234*c217d954SCole Faust                _tmp = select(ab_x2_high32, (SRC_DATA_TYPE)INT_MAX, overflow); \
3235*c217d954SCole Faust                if(DST_SHIFT >= 0) \
3236*c217d954SCole Faust                { \
3237*c217d954SCole Faust                    long mask = ((((int)1) << DST_SHIFT) - (int)1); \
3238*c217d954SCole Faust                    long threshold = _tmp < (int)0 ? (mask >> 1) + (long)1 : (mask >> 1) + 0; \
3239*c217d954SCole Faust                    _tmp = (_tmp & mask) > threshold ? (_tmp >> DST_SHIFT) + (int)1 : (_tmp >> DST_SHIFT); \
3240*c217d954SCole Faust                } \
3241*c217d954SCole Faust                _tmp += DST_OFFSET; \
3242*c217d954SCole Faust                dst[_m0].s[_n0] = CONVERT_SAT(_tmp, DST_DATA_TYPE);                                                                            \
3243*c217d954SCole Faust            })                                                                                                                                          \
3244*c217d954SCole Faust        })                                                                                                                                          \
3245*c217d954SCole Faust    })
3246*c217d954SCole Faust
3247*c217d954SCole Faust
3248*c217d954SCole Faust#define T_ROWSET_MASK(DATA_TYPE, M0, N0, VALUE_TO_SET, a, mask)                                                                                            \
3249*c217d954SCole Faust    ({                                                                                                                                                     \
3250*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                                                                                                                 \
3251*c217d954SCole Faust        {                                                                                                                                                  \
3252*c217d954SCole Faust            LOOP_UNROLLING(int, _n0, 0, 1, N0,                                                                                                             \
3253*c217d954SCole Faust            {                                                                                                                                              \
3254*c217d954SCole Faust                a[_m0].s[_n0] = select((DATA_TYPE)(a[_m0].s[_n0]), (DATA_TYPE)(VALUE_TO_SET), (SELECT_DATA_TYPE(DATA_TYPE))(mask[_m0].v == (DATA_TYPE)0)); \
3255*c217d954SCole Faust            })                                                                                                                                             \
3256*c217d954SCole Faust        })                                                                                                                                                 \
3257*c217d954SCole Faust    })
3258*c217d954SCole Faust
3259*c217d954SCole Faust
3260*c217d954SCole Faust#define T_ACTIVATION(DATA_TYPE, M0, N0, ACTIVATION_TYPE, A_VAL, B_VAL, src, dst)               \
3261*c217d954SCole Faust    ({                                                                                         \
3262*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                                                     \
3263*c217d954SCole Faust        {                                                                                      \
3264*c217d954SCole Faust            dst[_m0].v = ACTIVATION(ACTIVATION_TYPE, DATA_TYPE, N0, src[_m0].v, A_VAL, B_VAL); \
3265*c217d954SCole Faust        })                                                                                     \
3266*c217d954SCole Faust    })
3267*c217d954SCole Faust
3268*c217d954SCole Faust
3269*c217d954SCole Faust#define relu_op_quantized(DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x) (max((DATA_TYPE)ZERO_VALUE, x))
3270*c217d954SCole Faust
3271*c217d954SCole Faust#define brelu_op_quantized(DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x) (min((DATA_TYPE)A_VAL, max((DATA_TYPE)ZERO_VALUE, x)))
3272*c217d954SCole Faust
3273*c217d954SCole Faust#define lu_brelu_op_quantized(DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x) (min(max(x, (DATA_TYPE)B_VAL), (DATA_TYPE)A_VAL))
3274*c217d954SCole Faust
3275*c217d954SCole Faust#define hard_swish_op_quantized(DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x) (x * ((min(max((DATA_TYPE)(x + (DATA_TYPE)3.f), (DATA_TYPE)0.f), (DATA_TYPE)6.f)) * (DATA_TYPE)0.166666667f))
3276*c217d954SCole Faust
3277*c217d954SCole Faust#define identity_op_quantized(DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x) (x)
3278*c217d954SCole Faust
3279*c217d954SCole Faust#define ACT_OP_QUANTIZED(op, DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x) op##_op_quantized(DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x)
3280*c217d954SCole Faust#define ACTIVATION_QUANTIZED(op, DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x) ACT_OP_QUANTIZED(op, DATA_TYPE, VEC_SIZE, ZERO_VALUE, A_VAL, B_VAL, x)
3281*c217d954SCole Faust
3282*c217d954SCole Faust#define V_ADD(A_VAL, B_VAL) ((A_VAL) + (B_VAL))
3283*c217d954SCole Faust#define V_SUB(A_VAL, B_VAL) ((A_VAL) - (B_VAL))
3284*c217d954SCole Faust#define V_DIV(A_VAL, B_VAL) ((A_VAL) / (B_VAL))
3285*c217d954SCole Faust#define V_MUL(A_VAL, B_VAL) ((A_VAL) * (B_VAL))
3286*c217d954SCole Faust
3287*c217d954SCole Faust
3288*c217d954SCole Faust#define T_ACTIVATION_QUANTIZED(DATA_TYPE, M0, N0, ACTIVATION_TYPE, ZERO_VALUE, A_VAL, B_VAL, src, dst)               \
3289*c217d954SCole Faust    ({ \
3290*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0, \
3291*c217d954SCole Faust        { \
3292*c217d954SCole Faust            dst[_m0].v = ACTIVATION_QUANTIZED(ACTIVATION_TYPE, DATA_TYPE, N0, ZERO_VALUE, A_VAL, B_VAL, src[_m0].v); \
3293*c217d954SCole Faust        })                                                                                          \
3294*c217d954SCole Faust    })
3295*c217d954SCole Faust
3296*c217d954SCole Faust
3297*c217d954SCole Faust#define T_ADD(DATA_TYPE, M0, N0, lhs, rhs, dst) \
3298*c217d954SCole Faust    ({                                                            \
3299*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                        \
3300*c217d954SCole Faust        {                                                         \
3301*c217d954SCole Faust            dst[_m0].v = lhs[_m0].v + rhs[_m0].v; \
3302*c217d954SCole Faust        })                                                        \
3303*c217d954SCole Faust    })
3304*c217d954SCole Faust
3305*c217d954SCole Faust
3306*c217d954SCole Faust#define T_ADD_CONSTANT(DATA_TYPE, M0, N0, lhs, rhs_constant, dst) \
3307*c217d954SCole Faust    ({                                                            \
3308*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                        \
3309*c217d954SCole Faust        {                                                         \
3310*c217d954SCole Faust            dst[_m0].v = lhs[_m0].v + (DATA_TYPE)rhs_constant;               \
3311*c217d954SCole Faust        })                                                        \
3312*c217d954SCole Faust    })
3313*c217d954SCole Faust
3314*c217d954SCole Faust#define T_ELTWISE_BROADCAST_ADD_X(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE_BROADCAST_X(V_ADD, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3315*c217d954SCole Faust#define T_ELTWISE_BROADCAST_LHS_X_ADD(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE_BROADCAST_LHS_X(V_ADD, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3316*c217d954SCole Faust#define T_ELTWISE_BROADCAST_RHS_X_ADD(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE_BROADCAST_X(V_ADD, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3317*c217d954SCole Faust
3318*c217d954SCole Faust#define T_ELTWISE_BROADCAST_LHS_X_SUB(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE_BROADCAST_LHS_X(V_SUB, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3319*c217d954SCole Faust#define T_ELTWISE_BROADCAST_RHS_X_SUB(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE_BROADCAST_X(V_SUB, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3320*c217d954SCole Faust
3321*c217d954SCole Faust#define T_ELTWISE_BROADCAST_DIV_X(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE_BROADCAST_X(V_DIV, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3322*c217d954SCole Faust
3323*c217d954SCole Faust#define T_ELTWISE_BROADCAST_LHS_X_MUL(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE_BROADCAST_LHS_X(V_MUL, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3324*c217d954SCole Faust#define T_ELTWISE_BROADCAST_RHS_X_MUL(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE_BROADCAST_X(V_MUL, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3325*c217d954SCole Faust
3326*c217d954SCole Faust
3327*c217d954SCole Faust#define T_SCALE_CONSTANT(DATA_TYPE, M0, N0, lhs, rhs_constant, dst) \
3328*c217d954SCole Faust    ({                                                            \
3329*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                        \
3330*c217d954SCole Faust        {                                                         \
3331*c217d954SCole Faust            dst[_m0].v = lhs[_m0].v * (DATA_TYPE)rhs_constant; \
3332*c217d954SCole Faust        })                                                        \
3333*c217d954SCole Faust    })
3334*c217d954SCole Faust
3335*c217d954SCole Faust
3336*c217d954SCole Faust#define T_ELTWISE_BROADCAST_X(T_ELWISE_OP, DST_DATA_TYPE, M0, N0, lhs, rhs, dst) \
3337*c217d954SCole Faust    ({                                                      \
3338*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                  \
3339*c217d954SCole Faust        {                                                   \
3340*c217d954SCole Faust            dst[_m0].v = T_ELWISE_OP(CONVERT(lhs[_m0].v, VEC_DATA_TYPE(DST_DATA_TYPE, N0)), CONVERT(rhs[0].v, VEC_DATA_TYPE(DST_DATA_TYPE, N0)));             \
3341*c217d954SCole Faust        })                                                  \
3342*c217d954SCole Faust    })
3343*c217d954SCole Faust
3344*c217d954SCole Faust
3345*c217d954SCole Faust#define T_ELTWISE_BROADCAST_LHS_X(T_ELWISE_OP, DST_DATA_TYPE, M0, N0, lhs, rhs, dst) \
3346*c217d954SCole Faust    ({                                                      \
3347*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                  \
3348*c217d954SCole Faust        {                                                   \
3349*c217d954SCole Faust            dst[_m0].v = T_ELWISE_OP(CONVERT(lhs[0].v, VEC_DATA_TYPE(DST_DATA_TYPE, N0)), CONVERT(rhs[_m0].v, VEC_DATA_TYPE(DST_DATA_TYPE, N0)));             \
3350*c217d954SCole Faust        })                                                  \
3351*c217d954SCole Faust    })
3352*c217d954SCole Faust
3353*c217d954SCole Faust#define T_ELTWISE_ADD(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE(V_ADD, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3354*c217d954SCole Faust#define T_ELTWISE_SUB(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE(V_SUB, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3355*c217d954SCole Faust#define T_ELTWISE_DIV(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE(V_DIV, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3356*c217d954SCole Faust#define T_ELTWISE_MUL(DST_DATA_TYPE, M0, N0, lhs, rhs, dst) T_ELTWISE(V_MUL, DST_DATA_TYPE, M0, N0, lhs, rhs, dst)
3357*c217d954SCole Faust
3358*c217d954SCole Faust
3359*c217d954SCole Faust#define T_ELTWISE(T_ELWISE_OP, DST_DATA_TYPE, M0, N0, lhs, rhs, dst) \
3360*c217d954SCole Faust    ({                                                      \
3361*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                  \
3362*c217d954SCole Faust        {                                                   \
3363*c217d954SCole Faust            dst[_m0].v = T_ELWISE_OP(CONVERT(lhs[_m0].v, VEC_DATA_TYPE(DST_DATA_TYPE, N0)), CONVERT(rhs[_m0].v, VEC_DATA_TYPE(DST_DATA_TYPE, N0)));             \
3364*c217d954SCole Faust        })                                                  \
3365*c217d954SCole Faust    })
3366*c217d954SCole Faust
3367*c217d954SCole Faust
3368*c217d954SCole Faust#define T_FLOOR(DST_DATA_TYPE, M0, N0, src, dst) \
3369*c217d954SCole Faust    ({                                                      \
3370*c217d954SCole Faust        LOOP_UNROLLING(int, _m0, 0, 1, M0,                  \
3371*c217d954SCole Faust        {                                                   \
3372*c217d954SCole Faust            dst[_m0].v = floor(CONVERT(src[_m0].v, VEC_DATA_TYPE(DST_DATA_TYPE, N0)));             \
3373*c217d954SCole Faust        })                                                  \
3374*c217d954SCole Faust    })
3375*c217d954SCole Faust
3376*c217d954SCole Faust
3377*c217d954SCole Faust#define T_MMUL(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, LHS_LAYOUT, RHS_LAYOUT, lhs, rhs, dst) T_MMUL_##LHS_LAYOUT##_##RHS_LAYOUT(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)
3378*c217d954SCole Faust#define T_MMUL_NT_T(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst) T_MMUL_NT_T_##LHS_DATA_TYPE##_##RHS_DATA_TYPE##_##DST_DATA_TYPE(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)
3379*c217d954SCole Faust#define T_MMUL_NT_T_float_float_float(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst) T_MMUL_NT_T_FLOAT(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)
3380*c217d954SCole Faust#define T_MMUL_NT_T_half_half_float(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst) T_MMUL_NT_T_FLOAT(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)
3381*c217d954SCole Faust#define T_MMUL_NT_T_half_half_half(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst) T_MMUL_NT_T_FLOAT(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)
3382*c217d954SCole Faust#define T_MMUL_NT_T_char_char_int(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst) T_MMUL_NT_T_INTEGER8(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)
3383*c217d954SCole Faust#define T_MMUL_NT_T_uchar_uchar_uint(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst) T_MMUL_NT_T_INTEGER8(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)
3384*c217d954SCole Faust#define T_MMUL_NT_T_uchar_uchar_int(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst) T_MMUL_NT_T_INTEGER8(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)
3385*c217d954SCole Faust#define T_MMUL_NT_T_FLOAT(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)                       \
3386*c217d954SCole Faust    {                                                                                     \
3387*c217d954SCole Faust        LOOP_UNROLLING(int, _m, 0, 1, M0,                                                 \
3388*c217d954SCole Faust        {                                                                                 \
3389*c217d954SCole Faust            LOOP_UNROLLING(int, _n, 0, 1, N0,                                             \
3390*c217d954SCole Faust            {                                                                             \
3391*c217d954SCole Faust                LOOP_UNROLLING(int, _k, 0, 1, K0,                                         \
3392*c217d954SCole Faust                {                                                                         \
3393*c217d954SCole Faust                    dst[_m].s[_n] = fma((DST_DATA_TYPE)(lhs[_m].s[_k]), (DST_DATA_TYPE)(rhs[_n].s[_k]), dst[_m].s[_n]); \
3394*c217d954SCole Faust                })                                                                        \
3395*c217d954SCole Faust            })                                                                            \
3396*c217d954SCole Faust        })                                                                                \
3397*c217d954SCole Faust    }
3398*c217d954SCole Faust
3399*c217d954SCole Faust#define T_MMUL_NT_T_INTEGER8(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, lhs, rhs, dst)                            \
3400*c217d954SCole Faust    ({ \
3401*c217d954SCole Faust        LOOP_UNROLLING(int, _m, 0, 1, M0, \
3402*c217d954SCole Faust        { \
3403*c217d954SCole Faust            LOOP_UNROLLING(int, _n, 0, 1, N0, \
3404*c217d954SCole Faust            { \
3405*c217d954SCole Faust                DOT_PRODUCT_INTEGER8(LHS_DATA_TYPE, RHS_DATA_TYPE, DST_DATA_TYPE, K0, (lhs[_m].v), (rhs[_n].v), dst[_m].s[_n]); \
3406*c217d954SCole Faust            })                                                                                             \
3407*c217d954SCole Faust        })                                                                                             \
3408*c217d954SCole Faust    })
3409*c217d954SCole Faust
3410*c217d954SCole Faust#endif
3411*c217d954SCole Faust
3412*c217d954SCole Faust#if defined(INDIRECT_CONVOLUTION_ADDRESS_PRECALCULATION)
3413*c217d954SCole Faust
3414*c217d954SCole Faust
3415*c217d954SCole Faust
3416*c217d954SCole Faust__kernel void indirect_convolution_address_precalculation(
3417*c217d954SCole Faust    TENSOR4D_WO_T(dst, DST_TENSOR_TYPE))
3418*c217d954SCole Faust{
3419*c217d954SCole Faust    const int x = get_global_id(0);
3420*c217d954SCole Faust    const int y = get_global_id(1);
3421*c217d954SCole Faust    const int z = get_global_id(2);
3422*c217d954SCole Faust
3423*c217d954SCole Faust
3424*c217d954SCole Faust
3425*c217d954SCole Faust
3426*c217d954SCole Faust    const int mi = x % M0;
3427*c217d954SCole Faust
3428*c217d954SCole Faust    const int ki = x / M0;
3429*c217d954SCole Faust
3430*c217d954SCole Faust    const int xk = ki % WEI_CONV_WIDTH;
3431*c217d954SCole Faust
3432*c217d954SCole Faust    const int yk = ki / WEI_CONV_WIDTH;
3433*c217d954SCole Faust
3434*c217d954SCole Faust    TILE(DST_DATA_TYPE, 1, 1, xi);
3435*c217d954SCole Faust    TILE(DST_DATA_TYPE, 1, 1, yi);
3436*c217d954SCole Faust    TILE(DST_DATA_TYPE, 1, 1, my);
3437*c217d954SCole Faust
3438*c217d954SCole Faust    const int mout = y * M0;
3439*c217d954SCole Faust
3440*c217d954SCole Faust    xi[0].s[0] = ((mout + mi) % DST_CONV_WIDTH) * STRIDE_X;
3441*c217d954SCole Faust    yi[0].s[0] = ((mout + mi) / DST_CONV_WIDTH) * STRIDE_Y;
3442*c217d954SCole Faust    xi[0].s[0] -= PAD_LEFT;
3443*c217d954SCole Faust    yi[0].s[0] -= PAD_TOP;
3444*c217d954SCole Faust
3445*c217d954SCole Faust    const int x_s = xi[0].s[0] + xk;
3446*c217d954SCole Faust    const int y_s = yi[0].s[0] + yk;
3447*c217d954SCole Faust    my[0].s[0]    = x_s + y_s * SRC_CONV_WIDTH;
3448*c217d954SCole Faust    my[0].s[0]    = my[0].s[0] + z * (int)(SRC_CONV_WIDTH * SRC_CONV_HEIGHT);
3449*c217d954SCole Faust    my[0].s[0]    = select(-1, my[0].s[0], x_s >= 0);
3450*c217d954SCole Faust    my[0].s[0]    = select(-1, my[0].s[0], x_s < SRC_CONV_WIDTH);
3451*c217d954SCole Faust    my[0].s[0]    = select(-1, my[0].s[0], y_s >= 0);
3452*c217d954SCole Faust    my[0].s[0]    = select(-1, my[0].s[0], y_s < SRC_CONV_HEIGHT);
3453*c217d954SCole Faust
3454*c217d954SCole Faust    VSTORE(1)
3455*c217d954SCole Faust    (my[0].s[0], 0, (__global DST_DATA_TYPE *)(dst_ptr + dst_offset_first_element_in_bytes + x * sizeof(DST_DATA_TYPE) + y * dst_stride_y + z * dst_stride_z));
3456*c217d954SCole Faust}
3457*c217d954SCole Faust#endif
3458*c217d954SCole Faust
3459*c217d954SCole Faust#if defined(INDIRECT_CONVOLUTION_NHWC)
3460*c217d954SCole Faust
3461*c217d954SCole Faust
3462*c217d954SCole Faust
3463*c217d954SCole Faust__kernel void indirect_convolution_nhwc(
3464*c217d954SCole Faust    TENSOR4D_RO_T(src, SRC_TENSOR_TYPE),
3465*c217d954SCole Faust    TENSOR4D_RO_T(off, OFF_TENSOR_TYPE),
3466*c217d954SCole Faust    TENSOR4D_WO_T(dst, DST_TENSOR_TYPE),
3467*c217d954SCole Faust    TENSOR4D_RO_T(wei, WEI_TENSOR_TYPE)
3468*c217d954SCole Faust#if defined(HAS_BIAS)
3469*c217d954SCole Faust    ,
3470*c217d954SCole Faust    VECTOR_DECLARATION(bia)
3471*c217d954SCole Faust#endif
3472*c217d954SCole Faust)
3473*c217d954SCole Faust{
3474*c217d954SCole Faust
3475*c217d954SCole Faust
3476*c217d954SCole Faust#define _IWEI_WIDTH WEI_WIDTH
3477*c217d954SCole Faust#define _IWEI_HEIGHT WEI_HEIGHT
3478*c217d954SCole Faust#define _ISRC_CHANNELS SRC_CHANNELS
3479*c217d954SCole Faust#define _IDST_WIDTH DST_WIDTH
3480*c217d954SCole Faust#define _IDST_HEIGHT DST_HEIGHT
3481*c217d954SCole Faust#define _IY_MULTIPLIER (_IWEI_WIDTH * _IWEI_HEIGHT)
3482*c217d954SCole Faust
3483*c217d954SCole Faust    const int cout = GET_SPATIAL_IDX(0, N0, PARTIAL_N0);
3484*c217d954SCole Faust    const int mout = GET_SPATIAL_IDX(1, M0, 0);
3485*c217d954SCole Faust    const int bout = GET_SPATIAL_IDX(2, 1, 0);
3486*c217d954SCole Faust
3487*c217d954SCole Faust    off_offset_first_element_in_bytes += get_global_id(1) * off_stride_y;
3488*c217d954SCole Faust    off_offset_first_element_in_bytes += bout * off_stride_z;
3489*c217d954SCole Faust
3490*c217d954SCole Faust
3491*c217d954SCole Faust    TILE(DST_DATA_TYPE, M0, N0, c);
3492*c217d954SCole Faust
3493*c217d954SCole Faust    LOOP_UNROLLING(int, i, 0, 1, M0,
3494*c217d954SCole Faust    {
3495*c217d954SCole Faust        c[i].v = 0;
3496*c217d954SCole Faust    })
3497*c217d954SCole Faust
3498*c217d954SCole Faust    for(int i = 0; i < (_IWEI_WIDTH * _IWEI_HEIGHT); ++i)
3499*c217d954SCole Faust    {
3500*c217d954SCole Faust        TILE(int, 1, IND_BUFF_VEC_SIZE, my);
3501*c217d954SCole Faust        T_LOAD(int, 1, IND_BUFF_VEC_SIZE, OFF_TENSOR_TYPE, off, i * M0, 0, 1, 0, my);
3502*c217d954SCole Faust
3503*c217d954SCole Faust        int ck = 0;
3504*c217d954SCole Faust        for(; ck <= (_ISRC_CHANNELS - K0); ck += K0)
3505*c217d954SCole Faust        {
3506*c217d954SCole Faust            TILE(SRC_DATA_TYPE, M0, K0, a);
3507*c217d954SCole Faust            TILE(WEI_DATA_TYPE, N0, K0, b);
3508*c217d954SCole Faust
3509*c217d954SCole Faust
3510*c217d954SCole Faust            LOOP_UNROLLING(int, i, 0, 1, M0,
3511*c217d954SCole Faust            {
3512*c217d954SCole Faust                a[i].v = 0.0;
3513*c217d954SCole Faust            })
3514*c217d954SCole Faust
3515*c217d954SCole Faust            LOOP_UNROLLING(int, i, 0, 1, N0,
3516*c217d954SCole Faust            {
3517*c217d954SCole Faust                b[i].v = 0.0;
3518*c217d954SCole Faust            })
3519*c217d954SCole Faust
3520*c217d954SCole Faust
3521*c217d954SCole Faust            T_LOAD2D_INDIRECT(SRC_DATA_TYPE, M0, K0, SRC_TENSOR_TYPE, src, ck, src_stride_y, my, a);
3522*c217d954SCole Faust
3523*c217d954SCole Faust
3524*c217d954SCole Faust            T_LOAD(WEI_DATA_TYPE, N0, K0, WEI_TENSOR_TYPE, wei, ck, cout * _IY_MULTIPLIER + i, _IY_MULTIPLIER, wei_stride_y, b);
3525*c217d954SCole Faust
3526*c217d954SCole Faust
3527*c217d954SCole Faust            T_MMUL(SRC_DATA_TYPE, WEI_DATA_TYPE, DST_DATA_TYPE, M0, N0, K0, NT, T, a, b, c);
3528*c217d954SCole Faust        }
3529*c217d954SCole Faust
3530*c217d954SCole Faust
3531*c217d954SCole Faust#if defined(LEFTOVER_LOOP)
3532*c217d954SCole Faust
3533*c217d954SCole Faust        for(; ck < _ISRC_CHANNELS; ++ck)
3534*c217d954SCole Faust        {
3535*c217d954SCole Faust            TILE(SRC_DATA_TYPE, M0, 1, a);
3536*c217d954SCole Faust            TILE(WEI_DATA_TYPE, N0, 1, b);
3537*c217d954SCole Faust
3538*c217d954SCole Faust
3539*c217d954SCole Faust            LOOP_UNROLLING(int, i, 0, 1, M0,
3540*c217d954SCole Faust            {
3541*c217d954SCole Faust                a[i].v = 0.0;
3542*c217d954SCole Faust            })
3543*c217d954SCole Faust
3544*c217d954SCole Faust            LOOP_UNROLLING(int, i, 0, 1, N0,
3545*c217d954SCole Faust            {
3546*c217d954SCole Faust                b[i].v = 0.0;
3547*c217d954SCole Faust            })
3548*c217d954SCole Faust
3549*c217d954SCole Faust
3550*c217d954SCole Faust            T_LOAD2D_INDIRECT(SRC_DATA_TYPE, M0, 1, SRC_TENSOR_TYPE, src, ck, src_stride_y, my, a);
3551*c217d954SCole Faust
3552*c217d954SCole Faust
3553*c217d954SCole Faust
3554*c217d954SCole Faust            T_LOAD(WEI_DATA_TYPE, N0, 1, BUFFER, wei, ck, cout * _IY_MULTIPLIER + i, _IY_MULTIPLIER, wei_stride_y, b);
3555*c217d954SCole Faust
3556*c217d954SCole Faust
3557*c217d954SCole Faust            T_MMUL(SRC_DATA_TYPE, WEI_DATA_TYPE, DST_DATA_TYPE, M0, N0, 1, NT, T, a, b, c);
3558*c217d954SCole Faust        }
3559*c217d954SCole Faust#endif
3560*c217d954SCole Faust    }
3561*c217d954SCole Faust
3562*c217d954SCole Faust#if defined(HAS_BIAS)
3563*c217d954SCole Faust    TILE(BIA_DATA_TYPE, 1, N0, bias0);
3564*c217d954SCole Faust
3565*c217d954SCole Faust    T_LOAD(BIA_DATA_TYPE, 1, N0, BUFFER, bia, cout, 0, 1, 0, bias0);
3566*c217d954SCole Faust
3567*c217d954SCole Faust
3568*c217d954SCole Faust    T_ELTWISE_BROADCAST_ADD_X(DST_DATA_TYPE, M0, N0, c, bias0, c);
3569*c217d954SCole Faust
3570*c217d954SCole Faust#endif
3571*c217d954SCole Faust
3572*c217d954SCole Faust
3573*c217d954SCole Faust    T_ACTIVATION(DST_DATA_TYPE, M0, N0, ACTIVATION_TYPE, A_VAL, B_VAL, c, c);
3574*c217d954SCole Faust
3575*c217d954SCole Faust    TILE(uint, M0, 1, dst_indirect_y);
3576*c217d954SCole Faust
3577*c217d954SCole Faust
3578*c217d954SCole Faust    LOOP_UNROLLING(int, i, 0, 1, M0,
3579*c217d954SCole Faust    {
3580*c217d954SCole Faust        dst_indirect_y[i].v = (uint)min(mout + i, (int)(_IDST_WIDTH * _IDST_HEIGHT) - 1);
3581*c217d954SCole Faust        dst_indirect_y[i].v += bout * (int)(_IDST_WIDTH * _IDST_HEIGHT);
3582*c217d954SCole Faust    })
3583*c217d954SCole Faust
3584*c217d954SCole Faust    const bool x_cond = PARTIAL_N0 != 0 && get_global_id(0) == 0;
3585*c217d954SCole Faust
3586*c217d954SCole Faust
3587*c217d954SCole Faust    T_STORE_INDIRECT_WIDTH_SELECT(DST_DATA_TYPE, M0, N0, PARTIAL_N0, DST_TENSOR_TYPE, dst, cout, dst_stride_y, x_cond, c, dst_indirect_y);
3588*c217d954SCole Faust
3589*c217d954SCole Faust#undef _IWEI_WIDTH
3590*c217d954SCole Faust#undef _IWEI_HEIGHT
3591*c217d954SCole Faust#undef _ISRC_CHANNELS
3592*c217d954SCole Faust#undef _IDST_WIDTH
3593*c217d954SCole Faust#undef _IDST_HEIGHT
3594*c217d954SCole Faust#undef _IY_MULTIPLIER
3595*c217d954SCole Faust}
3596*c217d954SCole Faust#endif  )"