#include <pmmintrin.h>

Macros
#define	__DEFAULT_FN_ATTRS
#define	__trunc64(x)
#define	__zext128(x)
#define	__DEFAULT_FN_ATTRS_CONSTEXPR __DEFAULT_FN_ATTRS
#define	_mm_alignr_epi8(a, b, n)
	Concatenates the two 128-bit integer vector operands, and right-shifts the result by the number of bytes specified in the immediate operand.
#define	_mm_alignr_pi8(a, b, n)
	Concatenates the two 64-bit integer vector operands, and right-shifts the result by the number of bytes specified in the immediate operand.

Functions
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_abs_pi8 (__m64 __a)
	Computes the absolute value of each of the packed 8-bit signed integers in the source operand and stores the 8-bit unsigned integer results in the destination.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_abs_epi8 (__m128i __a)
	Computes the absolute value of each of the packed 8-bit signed integers in the source operand and stores the 8-bit unsigned integer results in the destination.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_abs_pi16 (__m64 __a)
	Computes the absolute value of each of the packed 16-bit signed integers in the source operand and stores the 16-bit unsigned integer results in the destination.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_abs_epi16 (__m128i __a)
	Computes the absolute value of each of the packed 16-bit signed integers in the source operand and stores the 16-bit unsigned integer results in the destination.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_abs_pi32 (__m64 __a)
	Computes the absolute value of each of the packed 32-bit signed integers in the source operand and stores the 32-bit unsigned integer results in the destination.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_abs_epi32 (__m128i __a)
	Computes the absolute value of each of the packed 32-bit signed integers in the source operand and stores the 32-bit unsigned integer results in the destination.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hadd_epi16 (__m128i __a, __m128i __b)
	Horizontally adds the adjacent pairs of values contained in 2 packed 128-bit vectors of [8 x i16].
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hadd_epi32 (__m128i __a, __m128i __b)
	Horizontally adds the adjacent pairs of values contained in 2 packed 128-bit vectors of [4 x i32].
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hadd_pi16 (__m64 __a, __m64 __b)
	Horizontally adds the adjacent pairs of values contained in 2 packed 64-bit vectors of [4 x i16].
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hadd_pi32 (__m64 __a, __m64 __b)
	Horizontally adds the adjacent pairs of values contained in 2 packed 64-bit vectors of [2 x i32].
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hadds_epi16 (__m128i __a, __m128i __b)
	Horizontally adds, with saturation, the adjacent pairs of values contained in two packed 128-bit vectors of [8 x i16].
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hadds_pi16 (__m64 __a, __m64 __b)
	Horizontally adds, with saturation, the adjacent pairs of values contained in two packed 64-bit vectors of [4 x i16].
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hsub_epi16 (__m128i __a, __m128i __b)
	Horizontally subtracts the adjacent pairs of values contained in 2 packed 128-bit vectors of [8 x i16].
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hsub_epi32 (__m128i __a, __m128i __b)
	Horizontally subtracts the adjacent pairs of values contained in 2 packed 128-bit vectors of [4 x i32].
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hsub_pi16 (__m64 __a, __m64 __b)
	Horizontally subtracts the adjacent pairs of values contained in 2 packed 64-bit vectors of [4 x i16].
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hsub_pi32 (__m64 __a, __m64 __b)
	Horizontally subtracts the adjacent pairs of values contained in 2 packed 64-bit vectors of [2 x i32].
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hsubs_epi16 (__m128i __a, __m128i __b)
	Horizontally subtracts, with saturation, the adjacent pairs of values contained in two packed 128-bit vectors of [8 x i16].
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_hsubs_pi16 (__m64 __a, __m64 __b)
	Horizontally subtracts, with saturation, the adjacent pairs of values contained in two packed 64-bit vectors of [4 x i16].
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_maddubs_epi16 (__m128i __a, __m128i __b)
	Multiplies corresponding pairs of packed 8-bit unsigned integer values contained in the first source operand and packed 8-bit signed integer values contained in the second source operand, adds pairs of contiguous products with signed saturation, and writes the 16-bit sums to the corresponding bits in the destination.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_maddubs_pi16 (__m64 __a, __m64 __b)
	Multiplies corresponding pairs of packed 8-bit unsigned integer values contained in the first source operand and packed 8-bit signed integer values contained in the second source operand, adds pairs of contiguous products with signed saturation, and writes the 16-bit sums to the corresponding bits in the destination.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_mulhrs_epi16 (__m128i __a, __m128i __b)
	Multiplies packed 16-bit signed integer values, truncates the 32-bit products to the 18 most significant bits by right-shifting, rounds the truncated value by adding 1, and writes bits [16:1] to the destination.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_mulhrs_pi16 (__m64 __a, __m64 __b)
	Multiplies packed 16-bit signed integer values, truncates the 32-bit products to the 18 most significant bits by right-shifting, rounds the truncated value by adding 1, and writes bits [16:1] to the destination.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_shuffle_epi8 (__m128i __a, __m128i __b)
	Copies the 8-bit integers from a 128-bit integer vector to the destination or clears 8-bit values in the destination, as specified by the second source operand.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_shuffle_pi8 (__m64 __a, __m64 __b)
	Copies the 8-bit integers from a 64-bit integer vector to the destination or clears 8-bit values in the destination, as specified by the second source operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_sign_epi8 (__m128i __a, __m128i __b)
	For each 8-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_sign_epi16 (__m128i __a, __m128i __b)
	For each 16-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.
static __inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_sign_epi32 (__m128i __a, __m128i __b)
	For each 32-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_sign_pi8 (__m64 __a, __m64 __b)
	For each 8-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_sign_pi16 (__m64 __a, __m64 __b)
	For each 16-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.
static __inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR	_mm_sign_pi32 (__m64 __a, __m64 __b)
	For each 32-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.

Macro Definition Documentation

◆ __DEFAULT_FN_ATTRS

#define __DEFAULT_FN_ATTRS

Value:

__attribute__((__always_inline__, __nodebug__, __target__("ssse3"), \

__min_vector_width__(128)))

__attribute__

_Float16 __2f16 __attribute__((ext_vector_type(2)))

Zeroes the upper 128 bits (bits 255:128) of all YMM registers.

Definition __clang_hip_libdevice_declares.h:285

Definition at line 20 of file tmmintrin.h.

◆ __DEFAULT_FN_ATTRS_CONSTEXPR

#define __DEFAULT_FN_ATTRS_CONSTEXPR __DEFAULT_FN_ATTRS

Definition at line 33 of file tmmintrin.h.

◆ __trunc64

#define __trunc64 ( x )

Value:

(__m64) __builtin_shufflevector((__v2di)(x), __extension__(__v2di){}, 0)

Definition at line 24 of file tmmintrin.h.

Referenced by _mm_hadd_pi16(), _mm_hadd_pi32(), _mm_hadds_pi16(), _mm_hsub_pi16(), _mm_hsub_pi32(), _mm_hsubs_pi16(), _mm_maddubs_pi16(), _mm_mulhrs_pi16(), _mm_shuffle_pi8(), _mm_sign_pi16(), _mm_sign_pi32(), and _mm_sign_pi8().

◆ __zext128

#define __zext128 ( x )

Value:

(__m128i) __builtin_shufflevector((__v2si)(x), __extension__(__v2si){}, 0, \

1, 2, 3)

Definition at line 26 of file tmmintrin.h.

Referenced by _mm_maddubs_pi16(), _mm_mulhrs_pi16(), _mm_shuffle_pi8(), _mm_sign_pi16(), _mm_sign_pi32(), and _mm_sign_pi8().

◆ _mm_alignr_epi8

#define _mm_alignr_epi8	(	a,
		b,
		n )

Value:

((__m128i)__builtin_ia32_palignr128((__v16qi)(__m128i)(a), \

(__v16qi)(__m128i)(b), (n)))

b

__device__ __2f16 b

Definition __clang_hip_libdevice_declares.h:295

Concatenates the two 128-bit integer vector operands, and right-shifts the result by the number of bytes specified in the immediate operand.

__m128i _mm_alignr_epi8(__m128i a, __m128i b, const int n);

_mm_alignr_epi8

#define _mm_alignr_epi8(a, b, n)

Concatenates the two 128-bit integer vector operands, and right-shifts the result by the number of by...

Definition tmmintrin.h:155

This intrinsic corresponds to the PALIGNR instruction.

Parameters

a	A 128-bit vector of [16 x i8] containing one of the source operands.
b	A 128-bit vector of [16 x i8] containing one of the source operands.
n	An immediate operand specifying how many bytes to right-shift the result.

Returns: A 128-bit integer vector containing the concatenated right-shifted value.

Definition at line 155 of file tmmintrin.h.

◆ _mm_alignr_pi8

#define _mm_alignr_pi8	(	a,
		b,
		n )

Value:

  ((__m64)__builtin_shufflevector(                                             \
      (__v2di)__builtin_ia32_psrldqi128_byteshift(                             \
          (__v16qi)__builtin_shufflevector((__v1di)(a), (__v1di)(b), 1, 0),    \
          (n)),                                                                \
      __extension__(__v2di){}, 0))

Concatenates the two 64-bit integer vector operands, and right-shifts the result by the number of bytes specified in the immediate operand.

__m64 _mm_alignr_pi8(__m64 a, __m64 b, const int n);

_mm_alignr_pi8

#define _mm_alignr_pi8(a, b, n)

Concatenates the two 64-bit integer vector operands, and right-shifts the result by the number of byt...

Definition tmmintrin.h:178

This intrinsic corresponds to the PALIGNR instruction.

Parameters

a	A 64-bit vector of [8 x i8] containing one of the source operands.
b	A 64-bit vector of [8 x i8] containing one of the source operands.
n	An immediate operand specifying how many bytes to right-shift the result.

Returns: A 64-bit integer vector containing the concatenated right-shifted value.

Definition at line 178 of file tmmintrin.h.

Function Documentation

◆ _mm_abs_epi16()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_abs_epi16 ( __m128i __a )

static

Computes the absolute value of each of the packed 16-bit signed integers in the source operand and stores the 16-bit unsigned integer results in the destination.

This intrinsic corresponds to the VPABSW instruction.

Parameters

__a	A 128-bit vector of [8 x i16].

Returns: A 128-bit integer vector containing the absolute values of the elements in the operand.

Definition at line 98 of file tmmintrin.h.

References __a.

Referenced by _mm_mask_abs_epi16(), and _mm_maskz_abs_epi16().

◆ _mm_abs_epi32()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_abs_epi32 ( __m128i __a )

static

Computes the absolute value of each of the packed 32-bit signed integers in the source operand and stores the 32-bit unsigned integer results in the destination.

This intrinsic corresponds to the VPABSD instruction.

Parameters

__a	A 128-bit vector of [4 x i32].

Returns: A 128-bit integer vector containing the absolute values of the elements in the operand.

Definition at line 131 of file tmmintrin.h.

References __a.

Referenced by _mm_mask_abs_epi32(), and _mm_maskz_abs_epi32().

◆ _mm_abs_epi8()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_abs_epi8 ( __m128i __a )

static

Computes the absolute value of each of the packed 8-bit signed integers in the source operand and stores the 8-bit unsigned integer results in the destination.

This intrinsic corresponds to the VPABSB instruction.

Parameters

__a	A 128-bit vector of [16 x i8].

Returns: A 128-bit integer vector containing the absolute values of the elements in the operand.

Definition at line 65 of file tmmintrin.h.

References __a.

Referenced by _mm_mask_abs_epi8(), and _mm_maskz_abs_epi8().

◆ _mm_abs_pi16()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_abs_pi16 ( __m64 __a )

static

Computes the absolute value of each of the packed 16-bit signed integers in the source operand and stores the 16-bit unsigned integer results in the destination.

This intrinsic corresponds to the PABSW instruction.

Parameters

__a	A 64-bit vector of [4 x i16].

Returns: A 64-bit integer vector containing the absolute values of the elements in the operand.

Definition at line 81 of file tmmintrin.h.

References __a, and __DEFAULT_FN_ATTRS_CONSTEXPR.

◆ _mm_abs_pi32()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_abs_pi32 ( __m64 __a )

static

Computes the absolute value of each of the packed 32-bit signed integers in the source operand and stores the 32-bit unsigned integer results in the destination.

This intrinsic corresponds to the PABSD instruction.

Parameters

__a	A 64-bit vector of [2 x i32].

Returns: A 64-bit integer vector containing the absolute values of the elements in the operand.

Definition at line 114 of file tmmintrin.h.

References __a, and __DEFAULT_FN_ATTRS_CONSTEXPR.

◆ _mm_abs_pi8()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_abs_pi8 ( __m64 __a )

static

Computes the absolute value of each of the packed 8-bit signed integers in the source operand and stores the 8-bit unsigned integer results in the destination.

This intrinsic corresponds to the PABSB instruction.

Parameters

__a	A 64-bit vector of [8 x i8].

Returns: A 64-bit integer vector containing the absolute values of the elements in the operand.

Definition at line 48 of file tmmintrin.h.

References __a, and __DEFAULT_FN_ATTRS_CONSTEXPR.

◆ _mm_hadd_epi16()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hadd_epi16	(	__m128i	__a,
		__m128i	__b )

static

Horizontally adds the adjacent pairs of values contained in 2 packed 128-bit vectors of [8 x i16].

This intrinsic corresponds to the VPHADDW instruction.

Parameters

__a	A 128-bit vector of [8 x i16] containing one of the source operands. The horizontal sums of the values are stored in the lower bits of the destination.
__b	A 128-bit vector of [8 x i16] containing one of the source operands. The horizontal sums of the values are stored in the upper bits of the destination.

Returns: A 128-bit vector of [8 x i16] containing the horizontal sums of both operands.

Definition at line 203 of file tmmintrin.h.

References __a, and __b.

◆ _mm_hadd_epi32()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hadd_epi32	(	__m128i	__a,
		__m128i	__b )

static

Horizontally adds the adjacent pairs of values contained in 2 packed 128-bit vectors of [4 x i32].

This intrinsic corresponds to the VPHADDD instruction.

Parameters

__a	A 128-bit vector of [4 x i32] containing one of the source operands. The horizontal sums of the values are stored in the lower bits of the destination.
__b	A 128-bit vector of [4 x i32] containing one of the source operands. The horizontal sums of the values are stored in the upper bits of the destination.

Returns: A 128-bit vector of [4 x i32] containing the horizontal sums of both operands.

Definition at line 225 of file tmmintrin.h.

References __a, and __b.

◆ _mm_hadd_pi16()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hadd_pi16	(	__m64	__a,
		__m64	__b )

static

Horizontally adds the adjacent pairs of values contained in 2 packed 64-bit vectors of [4 x i16].

This intrinsic corresponds to the PHADDW instruction.

Parameters

__a	A 64-bit vector of [4 x i16] containing one of the source operands. The horizontal sums of the values are stored in the lower bits of the destination.
__b	A 64-bit vector of [4 x i16] containing one of the source operands. The horizontal sums of the values are stored in the upper bits of the destination.

Returns: A 64-bit vector of [4 x i16] containing the horizontal sums of both operands.

Definition at line 246 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, and __trunc64.

◆ _mm_hadd_pi32()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hadd_pi32	(	__m64	__a,
		__m64	__b )

static

Horizontally adds the adjacent pairs of values contained in 2 packed 64-bit vectors of [2 x i32].

This intrinsic corresponds to the PHADDD instruction.

Parameters

__a	A 64-bit vector of [2 x i32] containing one of the source operands. The horizontal sums of the values are stored in the lower bits of the destination.
__b	A 64-bit vector of [2 x i32] containing one of the source operands. The horizontal sums of the values are stored in the upper bits of the destination.

Returns: A 64-bit vector of [2 x i32] containing the horizontal sums of both operands.

Definition at line 269 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, and __trunc64.

◆ _mm_hadds_epi16()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hadds_epi16	(	__m128i	__a,
		__m128i	__b )

static

Horizontally adds, with saturation, the adjacent pairs of values contained in two packed 128-bit vectors of [8 x i16].

Positive sums greater than 0x7FFF are saturated to 0x7FFF. Negative sums less than 0x8000 are saturated to 0x8000.

This intrinsic corresponds to the VPHADDSW instruction.

Parameters

__a	A 128-bit vector of [8 x i16] containing one of the source operands. The horizontal sums of the values are stored in the lower bits of the destination.
__b	A 128-bit vector of [8 x i16] containing one of the source operands. The horizontal sums of the values are stored in the upper bits of the destination.

Returns: A 128-bit vector of [8 x i16] containing the horizontal saturated sums of both operands.

Definition at line 296 of file tmmintrin.h.

References __a, and __b.

◆ _mm_hadds_pi16()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hadds_pi16	(	__m64	__a,
		__m64	__b )

static

Horizontally adds, with saturation, the adjacent pairs of values contained in two packed 64-bit vectors of [4 x i16].

Positive sums greater than 0x7FFF are saturated to 0x7FFF. Negative sums less than 0x8000 are saturated to 0x8000.

This intrinsic corresponds to the PHADDSW instruction.

Parameters

__a	A 64-bit vector of [4 x i16] containing one of the source operands. The horizontal sums of the values are stored in the lower bits of the destination.
__b	A 64-bit vector of [4 x i16] containing one of the source operands. The horizontal sums of the values are stored in the upper bits of the destination.

Returns: A 64-bit vector of [4 x i16] containing the horizontal saturated sums of both operands.

Definition at line 320 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, and __trunc64.

◆ _mm_hsub_epi16()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hsub_epi16	(	__m128i	__a,
		__m128i	__b )

static

Horizontally subtracts the adjacent pairs of values contained in 2 packed 128-bit vectors of [8 x i16].

This intrinsic corresponds to the VPHSUBW instruction.

Parameters

__a	A 128-bit vector of [8 x i16] containing one of the source operands. The horizontal differences between the values are stored in the lower bits of the destination.
__b	A 128-bit vector of [8 x i16] containing one of the source operands. The horizontal differences between the values are stored in the upper bits of the destination.

Returns: A 128-bit vector of [8 x i16] containing the horizontal differences of both operands.

Definition at line 344 of file tmmintrin.h.

References __a, and __b.

◆ _mm_hsub_epi32()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hsub_epi32	(	__m128i	__a,
		__m128i	__b )

static

Horizontally subtracts the adjacent pairs of values contained in 2 packed 128-bit vectors of [4 x i32].

This intrinsic corresponds to the VPHSUBD instruction.

Parameters

__a	A 128-bit vector of [4 x i32] containing one of the source operands. The horizontal differences between the values are stored in the lower bits of the destination.
__b	A 128-bit vector of [4 x i32] containing one of the source operands. The horizontal differences between the values are stored in the upper bits of the destination.

Returns: A 128-bit vector of [4 x i32] containing the horizontal differences of both operands.

Definition at line 366 of file tmmintrin.h.

References __a, and __b.

◆ _mm_hsub_pi16()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hsub_pi16	(	__m64	__a,
		__m64	__b )

static

Horizontally subtracts the adjacent pairs of values contained in 2 packed 64-bit vectors of [4 x i16].

This intrinsic corresponds to the PHSUBW instruction.

Parameters

__a	A 64-bit vector of [4 x i16] containing one of the source operands. The horizontal differences between the values are stored in the lower bits of the destination.
__b	A 64-bit vector of [4 x i16] containing one of the source operands. The horizontal differences between the values are stored in the upper bits of the destination.

Returns: A 64-bit vector of [4 x i16] containing the horizontal differences of both operands.

Definition at line 387 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, and __trunc64.

◆ _mm_hsub_pi32()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hsub_pi32	(	__m64	__a,
		__m64	__b )

static

Horizontally subtracts the adjacent pairs of values contained in 2 packed 64-bit vectors of [2 x i32].

This intrinsic corresponds to the PHSUBD instruction.

Parameters

__a	A 64-bit vector of [2 x i32] containing one of the source operands. The horizontal differences between the values are stored in the lower bits of the destination.
__b	A 64-bit vector of [2 x i32] containing one of the source operands. The horizontal differences between the values are stored in the upper bits of the destination.

Returns: A 64-bit vector of [2 x i32] containing the horizontal differences of both operands.

Definition at line 410 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, and __trunc64.

◆ _mm_hsubs_epi16()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hsubs_epi16	(	__m128i	__a,
		__m128i	__b )

static

Horizontally subtracts, with saturation, the adjacent pairs of values contained in two packed 128-bit vectors of [8 x i16].

Positive differences greater than 0x7FFF are saturated to 0x7FFF. Negative differences less than 0x8000 are saturated to 0x8000.

This intrinsic corresponds to the VPHSUBSW instruction.

Parameters

__a	A 128-bit vector of [8 x i16] containing one of the source operands. The horizontal differences between the values are stored in the lower bits of the destination.
__b	A 128-bit vector of [8 x i16] containing one of the source operands. The horizontal differences between the values are stored in the upper bits of the destination.

Returns: A 128-bit vector of [8 x i16] containing the horizontal saturated differences of both operands.

Definition at line 437 of file tmmintrin.h.

References __a, and __b.

◆ _mm_hsubs_pi16()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hsubs_pi16	(	__m64	__a,
		__m64	__b )

static

Horizontally subtracts, with saturation, the adjacent pairs of values contained in two packed 64-bit vectors of [4 x i16].

Positive differences greater than 0x7FFF are saturated to 0x7FFF. Negative differences less than 0x8000 are saturated to 0x8000.

This intrinsic corresponds to the PHSUBSW instruction.

Parameters

__a	A 64-bit vector of [4 x i16] containing one of the source operands. The horizontal differences between the values are stored in the lower bits of the destination.
__b	A 64-bit vector of [4 x i16] containing one of the source operands. The horizontal differences between the values are stored in the upper bits of the destination.

Returns: A 64-bit vector of [4 x i16] containing the horizontal saturated differences of both operands.

Definition at line 461 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, and __trunc64.

◆ _mm_maddubs_epi16()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_maddubs_epi16	(	__m128i	__a,
		__m128i	__b )

static

Multiplies corresponding pairs of packed 8-bit unsigned integer values contained in the first source operand and packed 8-bit signed integer values contained in the second source operand, adds pairs of contiguous products with signed saturation, and writes the 16-bit sums to the corresponding bits in the destination.

For example, bits [7:0] of both operands are multiplied, bits [15:8] of both operands are multiplied, and the sum of both results is written to bits [15:0] of the destination.

This intrinsic corresponds to the VPMADDUBSW instruction.

Parameters

__a	A 128-bit integer vector containing the first source operand.
__b	A 128-bit integer vector containing the second source operand.

Returns: A 128-bit integer vector containing the sums of products of both operands:
R0 := (__a0 * __b0) + (__a1 * __b1)
R1 := (__a2 * __b2) + (__a3 * __b3)
R2 := (__a4 * __b4) + (__a5 * __b5)
R3 := (__a6 * __b6) + (__a7 * __b7)
R4 := (__a8 * __b8) + (__a9 * __b9)
R5 := (__a10 * __b10) + (__a11 * __b11)
R6 := (__a12 * __b12) + (__a13 * __b13)
R7 := (__a14 * __b14) + (__a15 * __b15)

Definition at line 496 of file tmmintrin.h.

References __a, and __b.

Referenced by _mm_mask_maddubs_epi16(), and _mm_maskz_maddubs_epi16().

◆ _mm_maddubs_pi16()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_maddubs_pi16	(	__m64	__a,
		__m64	__b )

static

Multiplies corresponding pairs of packed 8-bit unsigned integer values contained in the first source operand and packed 8-bit signed integer values contained in the second source operand, adds pairs of contiguous products with signed saturation, and writes the 16-bit sums to the corresponding bits in the destination.

For example, bits [7:0] of both operands are multiplied, bits [15:8] of both operands are multiplied, and the sum of both results is written to bits [15:0] of the destination.

This intrinsic corresponds to the PMADDUBSW instruction.

Parameters

__a	A 64-bit integer vector containing the first source operand.
__b	A 64-bit integer vector containing the second source operand.

Returns: A 64-bit integer vector containing the sums of products of both operands:
R0 := (__a0 * __b0) + (__a1 * __b1)
R1 := (__a2 * __b2) + (__a3 * __b3)
R2 := (__a4 * __b4) + (__a5 * __b5)
R3 := (__a6 * __b6) + (__a7 * __b7)

Definition at line 525 of file tmmintrin.h.

References __a, __b, __trunc64, and __zext128.

◆ _mm_mulhrs_epi16()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_mulhrs_epi16	(	__m128i	__a,
		__m128i	__b )

static

Multiplies packed 16-bit signed integer values, truncates the 32-bit products to the 18 most significant bits by right-shifting, rounds the truncated value by adding 1, and writes bits [16:1] to the destination.

This intrinsic corresponds to the VPMULHRSW instruction.

Parameters

__a	A 128-bit vector of [8 x i16] containing one of the source operands.
__b	A 128-bit vector of [8 x i16] containing one of the source operands.

Returns: A 128-bit vector of [8 x i16] containing the rounded and scaled products of both operands.

Definition at line 545 of file tmmintrin.h.

References __a, and __b.

Referenced by _mm_mask_mulhrs_epi16(), and _mm_maskz_mulhrs_epi16().

◆ _mm_mulhrs_pi16()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_mulhrs_pi16	(	__m64	__a,
		__m64	__b )

static

Multiplies packed 16-bit signed integer values, truncates the 32-bit products to the 18 most significant bits by right-shifting, rounds the truncated value by adding 1, and writes bits [16:1] to the destination.

This intrinsic corresponds to the PMULHRSW instruction.

Parameters

__a	A 64-bit vector of [4 x i16] containing one of the source operands.
__b	A 64-bit vector of [4 x i16] containing one of the source operands.

Returns: A 64-bit vector of [4 x i16] containing the rounded and scaled products of both operands.

Definition at line 564 of file tmmintrin.h.

References __a, __b, __trunc64, and __zext128.

◆ _mm_shuffle_epi8()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_shuffle_epi8	(	__m128i	__a,
		__m128i	__b )

static

Copies the 8-bit integers from a 128-bit integer vector to the destination or clears 8-bit values in the destination, as specified by the second source operand.

This intrinsic corresponds to the VPSHUFB instruction.

Parameters

__a	A 128-bit integer vector containing the values to be copied.
__b	A 128-bit integer vector containing control bytes corresponding to positions in the destination: Bit 7: 1: Clear the corresponding byte in the destination. 0: Copy the selected source byte to the corresponding byte in the destination. Bits [6:4] Reserved. Bits [3:0] select the source byte to be copied.

Returns: A 128-bit integer vector containing the copied or cleared values.

Definition at line 590 of file tmmintrin.h.

References __a, and __b.

Referenced by _mm_mask_shuffle_epi8(), and _mm_maskz_shuffle_epi8().

◆ _mm_shuffle_pi8()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_shuffle_pi8	(	__m64	__a,
		__m64	__b )

static

Copies the 8-bit integers from a 64-bit integer vector to the destination or clears 8-bit values in the destination, as specified by the second source operand.

This intrinsic corresponds to the PSHUFB instruction.

Parameters

__a	A 64-bit integer vector containing the values to be copied.
__b	A 64-bit integer vector containing control bytes corresponding to positions in the destination: Bit 7: 1: Clear the corresponding byte in the destination. 0: Copy the selected source byte to the corresponding byte in the destination. Bits [2:0] select the source byte to be copied.

Returns: A 64-bit integer vector containing the copied or cleared values.

Definition at line 614 of file tmmintrin.h.

References __a, __b, __trunc64, and __zext128.

◆ _mm_sign_epi16()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_sign_epi16	(	__m128i	__a,
		__m128i	__b )

static

For each 16-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.

If the word in the second source is negative, calculate the two's complement of the corresponding word in the first source, and write that value to the destination. If the word in the second source is positive, copy the corresponding word from the first source to the destination. If the word in the second source is zero, clear the corresponding word in the destination.

This intrinsic corresponds to the VPSIGNW instruction.

Parameters

__a	A 128-bit integer vector containing the values to be copied.
__b	A 128-bit integer vector containing control words corresponding to positions in the destination.

Returns: A 128-bit integer vector containing the resultant values.

Definition at line 667 of file tmmintrin.h.

References __a, and __b.

◆ _mm_sign_epi32()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_sign_epi32	(	__m128i	__a,
		__m128i	__b )

static

For each 32-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.

If the doubleword in the second source is negative, calculate the two's complement of the corresponding word in the first source, and write that value to the destination. If the doubleword in the second source is positive, copy the corresponding word from the first source to the destination. If the doubleword in the second source is zero, clear the corresponding word in the destination.

This intrinsic corresponds to the VPSIGND instruction.

Parameters

__a	A 128-bit integer vector containing the values to be copied.
__b	A 128-bit integer vector containing control doublewords corresponding to positions in the destination.

Returns: A 128-bit integer vector containing the resultant values.

Definition at line 692 of file tmmintrin.h.

References __a, and __b.

◆ _mm_sign_epi8()

__inline__ __m128i __DEFAULT_FN_ATTRS_CONSTEXPR _mm_sign_epi8	(	__m128i	__a,
		__m128i	__b )

static

For each 8-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.

If the byte in the second source is negative, calculate the two's complement of the corresponding byte in the first source, and write that value to the destination. If the byte in the second source is positive, copy the corresponding byte from the first source to the destination. If the byte in the second source is zero, clear the corresponding byte in the destination.

This intrinsic corresponds to the VPSIGNB instruction.

Parameters

__a	A 128-bit integer vector containing the values to be copied.
__b	A 128-bit integer vector containing control bytes corresponding to positions in the destination.

Returns: A 128-bit integer vector containing the resultant values.

Definition at line 642 of file tmmintrin.h.

References __a, and __b.

◆ _mm_sign_pi16()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_sign_pi16	(	__m64	__a,
		__m64	__b )

static

For each 16-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.

If the word in the second source is negative, calculate the two's complement of the corresponding word in the first source, and write that value to the destination. If the word in the second source is positive, copy the corresponding word from the first source to the destination. If the word in the second source is zero, clear the corresponding word in the destination.

This intrinsic corresponds to the PSIGNW instruction.

Parameters

__a	A 64-bit integer vector containing the values to be copied.
__b	A 64-bit integer vector containing control words corresponding to positions in the destination.

Returns: A 64-bit integer vector containing the resultant values.

Definition at line 742 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, __trunc64, and __zext128.

◆ _mm_sign_pi32()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_sign_pi32	(	__m64	__a,
		__m64	__b )

static

For each 32-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.

If the doubleword in the second source is negative, calculate the two's complement of the corresponding doubleword in the first source, and write that value to the destination. If the doubleword in the second source is positive, copy the corresponding doubleword from the first source to the destination. If the doubleword in the second source is zero, clear the corresponding doubleword in the destination.

This intrinsic corresponds to the PSIGND instruction.

Parameters

__a	A 64-bit integer vector containing the values to be copied.
__b	A 64-bit integer vector containing two control doublewords corresponding to positions in the destination.

Returns: A 64-bit integer vector containing the resultant values.

Definition at line 768 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, __trunc64, and __zext128.

◆ _mm_sign_pi8()

__inline__ __m64 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_sign_pi8	(	__m64	__a,
		__m64	__b )

static

For each 8-bit integer in the first source operand, perform one of the following actions as specified by the second source operand.

If the byte in the second source is negative, calculate the two's complement of the corresponding byte in the first source, and write that value to the destination. If the byte in the second source is positive, copy the corresponding byte from the first source to the destination. If the byte in the second source is zero, clear the corresponding byte in the destination.

This intrinsic corresponds to the PSIGNB instruction.

Parameters

__a	A 64-bit integer vector containing the values to be copied.
__b	A 64-bit integer vector containing control bytes corresponding to positions in the destination.

Returns: A 64-bit integer vector containing the resultant values.

Definition at line 716 of file tmmintrin.h.

References __a, __b, __DEFAULT_FN_ATTRS_CONSTEXPR, __trunc64, and __zext128.

Macros

Functions

Macro Definition Documentation

◆ __DEFAULT_FN_ATTRS

◆ __DEFAULT_FN_ATTRS_CONSTEXPR

◆ __trunc64

◆ __zext128

◆ _mm_alignr_epi8

◆ _mm_alignr_pi8

Function Documentation

◆ _mm_abs_epi16()

◆ _mm_abs_epi32()

◆ _mm_abs_epi8()

◆ _mm_abs_pi16()

◆ _mm_abs_pi32()

◆ _mm_abs_pi8()

◆ _mm_hadd_epi16()

◆ _mm_hadd_epi32()

◆ _mm_hadd_pi16()

◆ _mm_hadd_pi32()

◆ _mm_hadds_epi16()

◆ _mm_hadds_pi16()

◆ _mm_hsub_epi16()

◆ _mm_hsub_epi32()

◆ _mm_hsub_pi16()

◆ _mm_hsub_pi32()

◆ _mm_hsubs_epi16()

◆ _mm_hsubs_pi16()

◆ _mm_maddubs_epi16()

◆ _mm_maddubs_pi16()

◆ _mm_mulhrs_epi16()

◆ _mm_mulhrs_pi16()

◆ _mm_shuffle_epi8()

◆ _mm_shuffle_pi8()

◆ _mm_sign_epi16()

◆ _mm_sign_epi32()

◆ _mm_sign_epi8()

◆ _mm_sign_pi16()

◆ _mm_sign_pi32()

◆ _mm_sign_pi8()