10#ifndef __RISCV_PACKED_SIMD_H
11#define __RISCV_PACKED_SIMD_H
15#if defined(__cplusplus)
33#define __DEFAULT_FN_ATTRS __attribute__((__always_inline__, __nodebug__))
35#define __packed_splat2(ty, x) ((ty){(x), (x)})
36#define __packed_splat4(ty, x) ((ty){(x), (x), (x), (x)})
37#define __packed_splat8(ty, x) ((ty){(x), (x), (x), (x), (x), (x), (x), (x)})
39#define __packed_splat(name, ty, scalar_ty, splat) \
40 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(scalar_ty __x) { \
41 return splat(ty, __x); \
44#define __packed_scalar_binary_op(name, ty, scalar_ty, op, splat) \
45 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
47 return __rs1 op splat(ty, __rs2); \
50#define __packed_binary_op(name, ty, op) \
51 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
52 return __rs1 op __rs2; \
55#define __packed_unary_op(name, ty, op) \
56 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
60#define __packed_extract(name, rty, ty, max_idx) \
61 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __v, \
63 __attribute__((__enable_if__( \
65 "index must be a constant integer from 0 to " #max_idx))) { \
69#define __packed_subvector_extract8(name, rty, ty) \
70 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __v, \
72 __attribute__((__enable_if__( \
73 __idx <= 1, "index must be a constant integer from 0 to 1"))) { \
74 return __idx ? __builtin_shufflevector(__v, __v, 4, 5, 6, 7) \
75 : __builtin_shufflevector(__v, __v, 0, 1, 2, 3); \
77#define __packed_subvector_extract4(name, rty, ty) \
78 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __v, \
80 __attribute__((__enable_if__( \
81 __idx <= 1, "index must be a constant integer from 0 to 1"))) { \
82 return __idx ? __builtin_shufflevector(__v, __v, 2, 3) \
83 : __builtin_shufflevector(__v, __v, 0, 1); \
86#define __packed_store(name, ty, elt_ty) \
87 static __inline__ void __DEFAULT_FN_ATTRS __riscv_##name(elt_ty *__p, \
89 typedef ty __attribute__((__aligned__(1))) ua_ty; \
90 *(ua_ty *)__p = __v; \
93#define __packed_binary_builtin(name, ty, builtin) \
94 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
95 return builtin(__rs1, __rs2); \
98#define __packed_binary_builtin_mixed(name, rty, ty1, ty2, builtin) \
99 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty1 __rs1, \
101 return builtin(__rs1, __rs2); \
104#define __packed_ternary_builtin(name, ty, builtin) \
105 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rd, ty __rs1, \
107 return builtin(__rd, __rs1, __rs2); \
110#define __packed_ternary_builtin_mixed(name, rty, ty1, ty2, builtin) \
111 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(rty __rd, ty1 __rs1, \
113 return builtin(__rd, __rs1, __rs2); \
116#define __packed_sh1add(name, ty) \
117 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
118 return (__rs1 << 1) + __rs2; \
125#define __packed_sh1sadd(name, ty) \
126 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
127 return __builtin_elementwise_add_sat( \
128 __builtin_elementwise_add_sat(__rs1, __rs1), __rs2); \
131#define __packed_cmp(name, ty, rty, op) \
132 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
134 return (rty)(__rs1 op __rs2); \
137#define __packed_pabs(name, ty, rty) \
138 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
139 return (rty)__builtin_elementwise_abs(__rs1); \
142#define __packed_binary_builtin_cast(name, ty, rty, builtin) \
143 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
145 return (rty)builtin(__rs1, __rs2); \
148#define __packed_reduction(name, rty, ty, builtin) \
149 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
151 return builtin(__rs1, __rs2); \
154#define __packed_merge_builtin(name, ty, mask_ty, builtin) \
155 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2, \
157 return (ty)builtin(__rs1, __rs2, __rd); \
160#define __packed_unary_builtin(name, ty, builtin) \
161 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
162 return builtin(__rs1); \
165#define __packed_widen_convert(name, rty, ty) \
166 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
167 return __builtin_convertvector(__rs1, rty); \
169#define __packed_widen_binary_op(name, rty, ty, op) \
170 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
172 return __builtin_convertvector(__rs1, rty) \
173 op __builtin_convertvector(__rs2, rty); \
175#define __packed_widen_binary_acc_op(name, rty, ty, op) \
176 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(rty __rd, ty __rs1, \
178 return __rd op __builtin_convertvector(__rs1, rty) \
179 op __builtin_convertvector(__rs2, rty); \
181#define __packed_widen_sub_acc_op(name, rty, ty) \
182 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(rty __rd, ty __rs1, \
184 return __rd + __builtin_convertvector(__rs1, rty) - \
185 __builtin_convertvector(__rs2, rty); \
187#define __packed_widen_mul(name, rty, ty) \
188 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
190 return __builtin_convertvector(__rs1, rty) * \
191 __builtin_convertvector(__rs2, rty); \
193#define __packed_widen_mulsu(name, rty, ty1, ty2, uty) \
194 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty1 __rs1, \
196 return __builtin_convertvector(__rs1, rty) * \
197 (rty) __builtin_convertvector(__rs2, uty); \
199#define __packed_widen_high2(name, rty, ty) \
200 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
201 return (rty)__builtin_shufflevector((ty){0}, __rs1, 0, 2, 1, 3); \
203#define __packed_widen_high4(name, rty, ty) \
204 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
205 return (rty)__builtin_shufflevector((ty){0}, __rs1, 0, 4, 1, 5, 2, 6, 3, \
209#define __packed_narrow_even2(name, rty, ty, sty) \
210 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
211 return __builtin_convertvector(__rs1, rty); \
213#define __packed_narrow_even4(name, rty, ty, sty) \
214 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
215 return __builtin_convertvector(__rs1, rty); \
217#define __packed_narrow_odd2(name, rty, ty, sty, uty) \
218 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
219 return __builtin_shufflevector((sty)__rs1, (sty)__rs1, 1, 3); \
221#define __packed_narrow_odd4(name, rty, ty, sty, uty) \
222 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
223 return __builtin_shufflevector((sty)__rs1, (sty)__rs1, 1, 3, 5, 7); \
228#define __packed_reverse2(name, ty) \
229 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
230 return __builtin_shufflevector(__rs1, __rs1, 1, 0); \
232#define __packed_reverse4(name, ty) \
233 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
234 return __builtin_shufflevector(__rs1, __rs1, 3, 2, 1, 0); \
236#define __packed_reverse8(name, ty) \
237 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
238 return __builtin_shufflevector(__rs1, __rs1, 7, 6, 5, 4, 3, 2, 1, 0); \
241#define __packed_zip2(name, rty, ty) \
242 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
244 return __builtin_shufflevector(__rs1, __rs2, 0, 2, 1, 3); \
246#define __packed_zip4(name, rty, ty) \
247 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
249 return __builtin_shufflevector(__rs1, __rs2, 0, 4, 1, 5, 2, 6, 3, 7); \
251#define __packed_unzipe2(name, rty, ty) \
252 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
253 return __builtin_shufflevector(__rs1, __rs1, 0, 2); \
255#define __packed_unzipe4(name, rty, ty) \
256 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
257 return __builtin_shufflevector(__rs1, __rs1, 0, 2, 4, 6); \
259#define __packed_unzipo2(name, rty, ty) \
260 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
261 return __builtin_shufflevector(__rs1, __rs1, 1, 3); \
263#define __packed_unzipo4(name, rty, ty) \
264 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
265 return __builtin_shufflevector(__rs1, __rs1, 1, 3, 5, 7); \
268#define __packed_concat2(name, rty, ty) \
269 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __lo, ty __hi) { \
270 return __builtin_shufflevector(__lo, __hi, 0, 1, 2, 3); \
272#define __packed_concat4(name, rty, ty) \
273 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __lo, ty __hi) { \
274 return __builtin_shufflevector(__lo, __hi, 0, 1, 2, 3, 4, 5, 6, 7); \
277#define __packed_slide1(name, ty, elt_ty, ...) \
278 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rd, \
280 return __builtin_shufflevector(__rd, (ty){__rs1}, __VA_ARGS__); \
283#define __packed_pair_ee2(name, ty) \
284 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
285 return __builtin_shufflevector(__rs1, __rs2, 0, 2); \
287#define __packed_pair_eo2(name, ty) \
288 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
289 return __builtin_shufflevector(__rs1, __rs2, 0, 3); \
291#define __packed_pair_oe2(name, ty) \
292 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
293 return __builtin_shufflevector(__rs1, __rs2, 1, 2); \
295#define __packed_pair_oo2(name, ty) \
296 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
297 return __builtin_shufflevector(__rs1, __rs2, 1, 3); \
299#define __packed_pair_ee4(name, ty) \
300 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
301 return __builtin_shufflevector(__rs1, __rs2, 0, 4, 2, 6); \
303#define __packed_pair_eo4(name, ty) \
304 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
305 return __builtin_shufflevector(__rs1, __rs2, 0, 5, 2, 7); \
307#define __packed_pair_oe4(name, ty) \
308 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
309 return __builtin_shufflevector(__rs1, __rs2, 1, 4, 3, 6); \
311#define __packed_pair_oo4(name, ty) \
312 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
313 return __builtin_shufflevector(__rs1, __rs2, 1, 5, 3, 7); \
315#define __packed_pair_ee8(name, ty) \
316 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
317 return __builtin_shufflevector(__rs1, __rs2, 0, 8, 2, 10, 4, 12, 6, 14); \
319#define __packed_pair_eo8(name, ty) \
320 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
321 return __builtin_shufflevector(__rs1, __rs2, 0, 9, 2, 11, 4, 13, 6, 15); \
323#define __packed_pair_oe8(name, ty) \
324 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
325 return __builtin_shufflevector(__rs1, __rs2, 1, 8, 3, 10, 5, 12, 7, 14); \
327#define __packed_pair_oo8(name, ty) \
328 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \
329 return __builtin_shufflevector(__rs1, __rs2, 1, 9, 3, 11, 5, 13, 7, 15); \
332#define __packed_nzip2(name, rty, ty) \
333 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
335 return __builtin_shufflevector((rty)__rs1, (rty)__rs2, 0, 4, 2, 6); \
337#define __packed_nzip4(name, rty, ty) \
338 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
340 return __builtin_shufflevector((rty)__rs1, (rty)__rs2, 0, 8, 2, 10, 4, 12, \
343#define __packed_nziph2(name, rty, ty) \
344 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
346 return __builtin_shufflevector((rty)__rs1, (rty)__rs2, 1, 5, 3, 7); \
348#define __packed_nziph4(name, rty, ty) \
349 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
351 return __builtin_shufflevector((rty)__rs1, (rty)__rs2, 1, 9, 3, 11, 5, 13, \
355#define __packed_wunzip_ext(name, rty, ty, ext) \
356 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
360#define __packed_wunzip_shift(name, rty, ty, op, shamt) \
361 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
362 return (rty)__rs1 op shamt; \
365#define __packed_wunzip_odd_hi(name, rty, ty, zip) \
366 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) { \
367 return (rty)zip((rty){0}, (rty)__rs1); \
370#define __packed_abdsum(name, rty, ty, builtin) \
371 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, \
373 return builtin(__rs1, __rs2); \
376#define __packed_ternary_builtin_cast(name, rty, ty, builtin) \
377 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(rty __rd, ty __rs1, \
379 return builtin(__rd, __rs1, __rs2); \
382#define __packed_reinterpret(name, rty, ty) \
383 static __inline__ rty __DEFAULT_FN_ATTRS __riscv_preinterpret_##name( \
385 return __builtin_bit_cast(rty, __x); \
388#define __packed_insert(name, ty, elt_ty, max_idx) \
389 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __v, elt_ty __e, \
391 __attribute__((__enable_if__( \
392 __idx <= (max_idx), \
393 "index must be a constant integer from 0 to " #max_idx))) { \
398#define __packed_join2(name, ty, elt_ty) \
399 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(elt_ty __e0, \
401 return (ty){__e0, __e1}; \
404#define __packed_join4(name, ty, elt_ty) \
405 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name( \
406 elt_ty __e0, elt_ty __e1, elt_ty __e2, elt_ty __e3) { \
407 return (ty){__e0, __e1, __e2, __e3}; \
410#define __packed_load(name, ty, elt_ty) \
411 static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(elt_ty *__p) { \
412 typedef ty __attribute__((__aligned__(1))) ua_ty; \
413 return *(ua_ty *)__p; \
421#if __riscv_xlen == 64
433#if __riscv_xlen == 64
438#define __packed_mulh_builtin(name, res_ty, ty1, ty2) \
439 static __inline__ res_ty __DEFAULT_FN_ATTRS __riscv_##name(ty1 __rs1, \
441 return __builtin_riscv_##name(__rs1, __rs2); \
449#undef __packed_mulh_builtin
702#define __riscv_pusati_u16x2(rs1, width) \
703 __builtin_riscv_pusati_u16x2(rs1, width)
704#define __riscv_psati_i16x2(rs1, width) __builtin_riscv_psati_i16x2(rs1, width)
707#define __riscv_pusati_u16x4(rs1, width) \
708 __builtin_riscv_pusati_u16x4(rs1, width)
709#define __riscv_pusati_u32x2(rs1, width) \
710 __builtin_riscv_pusati_u32x2(rs1, width)
711#define __riscv_psati_i16x4(rs1, width) __builtin_riscv_psati_i16x4(rs1, width)
712#define __riscv_psati_i32x2(rs1, width) __builtin_riscv_psati_i32x2(rs1, width)
796 __builtin_riscv_pwsll_s_u16x4)
798 __builtin_riscv_pwsll_s_u32x2)
800 __builtin_riscv_pwsla_s_i16x4)
802 __builtin_riscv_pwsla_s_i32x2)
806 __builtin_riscv_pnsrl_s_u8x4)
808 __builtin_riscv_pnsrl_s_u16x2)
810 __builtin_riscv_pnsra_s_i8x4)
812 __builtin_riscv_pnsra_s_i16x2)
814 __builtin_riscv_pnsrar_s_i8x4)
816 __builtin_riscv_pnsrar_s_i16x2)
851 __builtin_riscv_pmqwacc_i32x2)
853 __builtin_riscv_pmqrwacc_i32x2)
1151 __riscv_psext_b_i16x2)
1154 __riscv_pzext_b_u16x2)
1159 __riscv_pnziph_i8x4)
1161 __riscv_pnziph_u8x4)
1165 __riscv_psext_b_i16x4)
1168 __riscv_pzext_b_u16x4)
1171 __riscv_psext_h_i32x2)
1174 __riscv_pzext_h_u32x2)
1178 __riscv_pnziph_i8x8)
1181 __riscv_pnziph_u8x8)
1184 __riscv_pnziph_i16x4)
1187 __riscv_pnziph_u16x4)
1344__packed_slide1(pslide1up_i8x8,
int8x8_t,
int8_t, 8, 0, 1, 2, 3, 4, 5, 6)
1345__packed_slide1(pslide1up_u8x8,
uint8x8_t,
uint8_t, 8, 0, 1, 2, 3, 4, 5, 6)
1350__packed_slide1(pslide1down_i8x8,
int8x8_t,
int8_t, 1, 2, 3, 4, 5, 6, 7, 8)
1351__packed_slide1(pslide1down_u8x8,
uint8x8_t,
uint8_t, 1, 2, 3, 4, 5, 6, 7, 8)
1366 __attribute__((__enable_if__(__idx <= 1,
"index must be a constant integer "
1368 return __idx ? __riscv_pjoin2_i8x8(
1369 __builtin_shufflevector(
__v,
__v, 0, 1, 2, 3), __s)
1370 : __riscv_pjoin2_i8x8(
1371 __s, __builtin_shufflevector(
__v,
__v, 4, 5, 6, 7));
1375 __attribute__((__enable_if__(__idx <= 1,
"index must be a constant integer "
1377 return __idx ? __riscv_pjoin2_u8x8(
1378 __builtin_shufflevector(
__v,
__v, 0, 1, 2, 3), __s)
1379 : __riscv_pjoin2_u8x8(
1380 __s, __builtin_shufflevector(
__v,
__v, 4, 5, 6, 7));
1384 __attribute__((__enable_if__(__idx <= 1,
"index must be a constant integer "
1386 return __idx ? __riscv_pjoin2_i16x4(
1387 __builtin_shufflevector(
__v,
__v, 0, 1), __s)
1388 : __riscv_pjoin2_i16x4(
1389 __s, __builtin_shufflevector(
__v,
__v, 2, 3));
1393 __attribute__((__enable_if__(__idx <= 1,
"index must be a constant integer "
1395 return __idx ? __riscv_pjoin2_u16x4(
1396 __builtin_shufflevector(
__v,
__v, 0, 1), __s)
1397 : __riscv_pjoin2_u16x4(
1398 __s, __builtin_shufflevector(
__v,
__v, 2, 3));
1535#undef __packed_splat2
1536#undef __packed_splat4
1537#undef __packed_splat8
1538#undef __packed_splat
1539#undef __packed_scalar_binary_op
1540#undef __packed_binary_op
1541#undef __packed_unary_op
1542#undef __packed_store
1543#undef __packed_binary_builtin
1544#undef __packed_binary_builtin_mixed
1545#undef __packed_ternary_builtin
1546#undef __packed_ternary_builtin_mixed
1547#undef __packed_sh1add
1548#undef __packed_sh1sadd
1551#undef __packed_binary_builtin_cast
1552#undef __packed_reduction
1553#undef __packed_merge_builtin
1554#undef __packed_unary_builtin
1555#undef __packed_widen_convert
1556#undef __packed_widen_binary_op
1557#undef __packed_widen_binary_acc_op
1558#undef __packed_widen_sub_acc_op
1559#undef __packed_widen_mul
1560#undef __packed_widen_mulsu
1561#undef __packed_widen_high2
1562#undef __packed_widen_high4
1563#undef __packed_narrow_even2
1564#undef __packed_narrow_even4
1565#undef __packed_narrow_odd2
1566#undef __packed_narrow_odd4
1567#undef __packed_reverse2
1568#undef __packed_reverse4
1569#undef __packed_reverse8
1572#undef __packed_unzipe2
1573#undef __packed_unzipe4
1574#undef __packed_unzipo2
1575#undef __packed_unzipo4
1576#undef __packed_concat2
1577#undef __packed_concat4
1578#undef __packed_slide1
1579#undef __packed_pair_ee2
1580#undef __packed_pair_eo2
1581#undef __packed_pair_oe2
1582#undef __packed_pair_oo2
1583#undef __packed_pair_ee4
1584#undef __packed_pair_eo4
1585#undef __packed_pair_oe4
1586#undef __packed_pair_oo4
1587#undef __packed_pair_ee8
1588#undef __packed_pair_eo8
1589#undef __packed_pair_oe8
1590#undef __packed_pair_oo8
1591#undef __packed_nzip2
1592#undef __packed_nzip4
1593#undef __packed_nziph2
1594#undef __packed_nziph4
1595#undef __packed_wunzip_ext
1596#undef __packed_wunzip_shift
1597#undef __packed_wunzip_odd_hi
1598#undef __packed_abdsum
1599#undef __packed_ternary_builtin_cast
1600#undef __packed_extract
1601#undef __packed_subvector_extract8
1602#undef __packed_subvector_extract4
1603#undef __packed_insert
1604#undef __packed_join2
1605#undef __packed_join4
1607#undef __packed_reinterpret
1608#undef __DEFAULT_FN_ATTRS
1610#if defined(__cplusplus)
_Float16 __2f16 __attribute__((ext_vector_type(2)))
Zeroes the upper 128 bits (bits 255:128) of all YMM registers.
#define __DEFAULT_FN_ATTRS
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 __packed_splat4 __packed_splat2 uint8x8_t
#define __packed_insert(name, ty, elt_ty, max_idx)
#define __packed_pair_oe8(name, ty)
#define __packed_cmp(name, ty, rty, op)
#define __packed_widen_convert(name, rty, ty)
#define __packed_wunzip_ext(name, rty, ty, ext)
#define __packed_subvector_extract8(name, rty, ty)
#define __packed_unary_builtin(name, ty, builtin)
#define __packed_widen_mulsu(name, rty, ty1, ty2, uty)
#define __packed_unzipo2(name, rty, ty)
#define __packed_binary_op(name, ty, op)
#define __packed_pair_ee2(name, ty)
#define __packed_reinterpret(name, rty, ty)
#define __packed_splat2(ty, x)
int8_t int8x4_t __attribute__((__vector_size__(4)))
#define __packed_reduction(name, rty, ty, builtin)
#define __packed_binary_builtin_mixed(name, rty, ty1, ty2, builtin)
#define __packed_wunzip_odd_hi(name, rty, ty, zip)
#define __packed_nziph2(name, rty, ty)
#define __packed_reverse2(name, ty)
#define __packed_pair_ee4(name, ty)
#define __packed_splat8(ty, x)
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 __packed_splat4 __packed_splat2 uint8_t
#define __packed_merge_builtin(name, ty, mask_ty, builtin)
#define __packed_scalar_binary_op(name, ty, scalar_ty, op, splat)
#define __packed_concat4(name, rty, ty)
#define __packed_pair_eo4(name, ty)
#define __packed_nzip2(name, rty, ty)
#define __packed_pair_eo8(name, ty)
#define __packed_reverse8(name, ty)
#define __packed_binary_builtin(name, ty, builtin)
#define __packed_join4(name, ty, elt_ty)
#define __packed_pair_oe4(name, ty)
#define __packed_nziph4(name, rty, ty)
#define __packed_wunzip_shift(name, rty, ty, op, shamt)
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 __packed_splat4 __packed_splat2 uint8x4_t
#define __packed_narrow_even2(name, rty, ty, sty)
#define __packed_widen_high4(name, rty, ty)
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 __packed_splat4 int32x2_t
#define __packed_ternary_builtin(name, ty, builtin)
#define __packed_concat2(name, rty, ty)
#define __packed_unary_op(name, ty, op)
#define __packed_extract(name, rty, ty, max_idx)
#define __packed_pair_oo4(name, ty)
#define __packed_pair_ee8(name, ty)
#define __packed_unzipe4(name, rty, ty)
#define __packed_slide1(name, ty, elt_ty,...)
int32_t uint32_t uint32_t int8_t
#define __packed_sh1add(name, ty)
#define __packed_pabs(name, ty, rty)
#define __packed_sh1sadd(name, ty)
#define __packed_pair_oe2(name, ty)
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 int8x8_t
int32_t uint32_t uint32_t __packed_splat4 int16_t
#define __packed_unzipe2(name, rty, ty)
#define __packed_unzipo4(name, rty, ty)
#define __packed_widen_high2(name, rty, ty)
#define __packed_ternary_builtin_mixed(name, rty, ty1, ty2, builtin)
#define __packed_pair_oo8(name, ty)
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 __packed_splat4 __packed_splat2 __packed_splat4 uint16_t
int32_t uint32_t uint32_t __packed_splat4 int16x2_t
#define __packed_nzip4(name, rty, ty)
#define __packed_zip4(name, rty, ty)
#define __packed_splat(name, ty, scalar_ty, splat)
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 int16x4_t
#define __packed_widen_binary_acc_op(name, rty, ty, op)
#define __packed_narrow_odd2(name, rty, ty, sty, uty)
#define __packed_abdsum(name, rty, ty, builtin)
#define __packed_widen_binary_op(name, rty, ty, op)
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 __packed_splat4 __packed_splat2 uint32x2_t
#define __packed_binary_builtin_cast(name, ty, rty, builtin)
#define __packed_ternary_builtin_cast(name, rty, ty, builtin)
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 __packed_splat4 __packed_splat2 uint16x2_t
#define __packed_subvector_extract4(name, rty, ty)
#define __packed_narrow_even4(name, rty, ty, sty)
#define __packed_narrow_odd4(name, rty, ty, sty, uty)
#define __packed_mulh_builtin(name, res_ty, ty1, ty2)
#define __packed_widen_sub_acc_op(name, rty, ty)
#define __packed_pair_eo2(name, ty)
#define __packed_widen_mul(name, rty, ty)
#define __packed_zip2(name, rty, ty)
#define __packed_pair_oo2(name, ty)
#define __packed_join2(name, ty, elt_ty)
#define __packed_store(name, ty, elt_ty)
#define __packed_splat4(ty, x)
#define __packed_load(name, ty, elt_ty)
int32_t uint32_t uint32_t int8x4_t
int32_t uint32_t uint32_t __packed_splat4 __packed_splat2 __packed_splat8 __packed_splat4 __packed_splat2 uint16x4_t
#define __packed_reverse4(name, ty)