doxygen/ppc__wrappers_2pmmintrin_8h_source.html

/*===---- pmmintrin.h - Implementation of SSE3 intrinsics on PowerPC -------===

 *

 * Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.

 * See https://llvm.org/LICENSE.txt for license information.

 * SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception

 *

 *===-----------------------------------------------------------------------===

 */


/* Implemented from the specification included in the Intel C++ Compiler

   User Guide and Reference, version 9.0.  */


#ifndef NO_WARN_X86_INTRINSICS

/* This header is distributed to simplify porting x86_64 code that

   makes explicit use of Intel intrinsics to powerpc64le.

   It is the user's responsibility to determine if the results are

   acceptable and make additional changes as necessary.

   Note that much code that uses Intel intrinsics can be rewritten in

   standard C or GNU C extensions, which are more portable and better

   optimized across multiple targets.


   In the specific case of X86 SSE3 intrinsics, the PowerPC VMX/VSX ISA

   is a good match for most SIMD operations.  However the Horizontal

   add/sub requires the data pairs be permuted into a separate

   registers with vertical even/odd alignment for the operation.

   And the addsub operation requires the sign of only the even numbered

   elements be flipped (xored with -0.0).

   For larger blocks of code using these intrinsic implementations,

   the compiler be should be able to schedule instructions to avoid

   additional latency.


   In the specific case of the monitor and mwait instructions there are

   no direct equivalent in the PowerISA at this time.  So those

   intrinsics are not implemented.  */

#error                                                                         \

    "Please read comment above.  Use -DNO_WARN_X86_INTRINSICS to disable this warning."

#endif


#ifndef PMMINTRIN_H_

#define PMMINTRIN_H_


#if defined(__powerpc64__) &&                                                  \

    (defined(__linux__) || defined(__FreeBSD__) || defined(_AIX))


/* We need definitions from the SSE2 and SSE header files*/

#include <emmintrin.h>


extern __inline __m128

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_addsub_ps(__m128 __X, __m128 __Y) {

  const __v4sf __even_n0 = {-0.0, 0.0, -0.0, 0.0};

  __v4sf __even_neg_Y = vec_xor(__Y, __even_n0);

  return (__m128)vec_add(__X, __even_neg_Y);

}


extern __inline __m128d

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_addsub_pd(__m128d __X, __m128d __Y) {

  const __v2df __even_n0 = {-0.0, 0.0};

  __v2df __even_neg_Y = vec_xor(__Y, __even_n0);

  return (__m128d)vec_add(__X, __even_neg_Y);

}


extern __inline __m128

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_hadd_ps(__m128 __X, __m128 __Y) {

  __vector unsigned char __xform2 = {0x00, 0x01, 0x02, 0x03, 0x08, 0x09,

                                     0x0A, 0x0B, 0x10, 0x11, 0x12, 0x13,

                                     0x18, 0x19, 0x1A, 0x1B};

  __vector unsigned char __xform1 = {0x04, 0x05, 0x06, 0x07, 0x0C, 0x0D,

                                     0x0E, 0x0F, 0x14, 0x15, 0x16, 0x17,

                                     0x1C, 0x1D, 0x1E, 0x1F};

  return (__m128)vec_add(vec_perm((__v4sf)__X, (__v4sf)__Y, __xform2),

                         vec_perm((__v4sf)__X, (__v4sf)__Y, __xform1));

}


extern __inline __m128

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_hsub_ps(__m128 __X, __m128 __Y) {

  __vector unsigned char __xform2 = {0x00, 0x01, 0x02, 0x03, 0x08, 0x09,

                                     0x0A, 0x0B, 0x10, 0x11, 0x12, 0x13,

                                     0x18, 0x19, 0x1A, 0x1B};

  __vector unsigned char __xform1 = {0x04, 0x05, 0x06, 0x07, 0x0C, 0x0D,

                                     0x0E, 0x0F, 0x14, 0x15, 0x16, 0x17,

                                     0x1C, 0x1D, 0x1E, 0x1F};

  return (__m128)vec_sub(vec_perm((__v4sf)__X, (__v4sf)__Y, __xform2),

                         vec_perm((__v4sf)__X, (__v4sf)__Y, __xform1));

}


extern __inline __m128d

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_hadd_pd(__m128d __X, __m128d __Y) {

  return (__m128d)vec_add(vec_mergeh((__v2df)__X, (__v2df)__Y),

                          vec_mergel((__v2df)__X, (__v2df)__Y));

}


extern __inline __m128d

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_hsub_pd(__m128d __X, __m128d __Y) {

  return (__m128d)vec_sub(vec_mergeh((__v2df)__X, (__v2df)__Y),

                          vec_mergel((__v2df)__X, (__v2df)__Y));

}


#ifdef _ARCH_PWR8

extern __inline __m128

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_movehdup_ps(__m128 __X) {

  return (__m128)vec_mergeo((__v4su)__X, (__v4su)__X);

}

#endif


#ifdef _ARCH_PWR8

extern __inline __m128

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_moveldup_ps(__m128 __X) {

  return (__m128)vec_mergee((__v4su)__X, (__v4su)__X);

}

#endif


extern __inline __m128d

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_loaddup_pd(double const *__P) {

  return (__m128d)vec_splats(*__P);

}


extern __inline __m128d

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_movedup_pd(__m128d __X) {

  return _mm_shuffle_pd(__X, __X, _MM_SHUFFLE2(0, 0));

}


extern __inline __m128i

    __attribute__((__gnu_inline__, __always_inline__, __artificial__))

    _mm_lddqu_si128(__m128i const *__P) {

  return (__m128i)(vec_vsx_ld(0, (signed int const *)__P));

}


/* POWER8 / POWER9 have no equivalent for _mm_monitor nor _mm_wait.  */


#else

#include_next <pmmintrin.h>

#endif /* defined(__powerpc64__) &&                                            \

        *   (defined(__linux__) || defined(__FreeBSD__) || defined(_AIX)) */


#endif /* PMMINTRIN_H_ */

__attribute__
_Float16 __2f16 __attribute__((ext_vector_type(2)))
Zeroes the upper 128 bits (bits 255:128) of all YMM registers.
Definition __clang_hip_libdevice_declares.h:285

vec_splats
static __inline__ vector signed char __ATTRS_o_ai vec_splats(signed char __a)
Definition altivec.h:14737

vec_mergel
static __inline__ vector signed char __ATTRS_o_ai vec_mergel(vector signed char __a, vector signed char __b)
Definition altivec.h:5361

vec_perm
static __inline__ vector signed char __ATTRS_o_ai vec_perm(vector signed char __a, vector signed char __b, vector unsigned char __c)
Definition altivec.h:7962

vec_mergeh
static __inline__ vector signed char __ATTRS_o_ai vec_mergeh(vector signed char __a, vector signed char __b)
Definition altivec.h:5091

vec_add
static __inline__ vector signed char __ATTRS_o_ai vec_add(vector signed char __a, vector signed char __b)
Definition altivec.h:200

vec_xor
static __inline__ vector unsigned char __ATTRS_o_ai vec_xor(vector unsigned char __a, vector unsigned char __b)
Definition altivec.h:13207

vec_sub
static __inline__ vector signed char __ATTRS_o_ai vec_sub(vector signed char __a, vector signed char __b)
Definition altivec.h:11869

_MM_SHUFFLE2
#define _MM_SHUFFLE2(x, y)
Definition emmintrin.h:4929

_mm_shuffle_pd
#define _mm_shuffle_pd(a, b, i)
Constructs a 128-bit floating-point vector of [2 x double] from two 128-bit vector parameters of [2 x...
Definition emmintrin.h:4735

_mm_addsub_pd
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_addsub_pd(__m128d __a, __m128d __b)
Adds the even-indexed values and subtracts the odd-indexed values of two 128-bit vectors of [2 x doub...
Definition pmmintrin.h:170

_mm_loaddup_pd
#define _mm_loaddup_pd(dp)
Moves and duplicates one double-precision value to double-precision values stored in a 128-bit vector...
Definition pmmintrin.h:233

_mm_hsub_ps
static __inline__ __m128 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hsub_ps(__m128 __a, __m128 __b)
Horizontally subtracts the adjacent pairs of values contained in two 128-bit vectors of [4 x float].
Definition pmmintrin.h:108

_mm_movedup_pd
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_movedup_pd(__m128d __a)
Moves and duplicates the double-precision value in the lower bits of a 128-bit vector of [2 x double]...
Definition pmmintrin.h:249

_mm_moveldup_ps
static __inline__ __m128 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_moveldup_ps(__m128 __a)
Duplicates even-indexed values from a 128-bit vector of [4 x float] to float values stored in a 128-b...
Definition pmmintrin.h:151

_mm_hadd_ps
static __inline__ __m128 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hadd_ps(__m128 __a, __m128 __b)
Horizontally adds the adjacent pairs of values contained in two 128-bit vectors of [4 x float].
Definition pmmintrin.h:86

_mm_addsub_ps
static __inline__ __m128 __DEFAULT_FN_ATTRS _mm_addsub_ps(__m128 __a, __m128 __b)
Adds the even-indexed values and subtracts the odd-indexed values of two 128-bit vectors of [4 x floa...
Definition pmmintrin.h:64

_mm_hsub_pd
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hsub_pd(__m128d __a, __m128d __b)
Horizontally subtracts the pairs of values contained in two 128-bit vectors of [2 x double].
Definition pmmintrin.h:214

_mm_lddqu_si128
static __inline__ __m128i __DEFAULT_FN_ATTRS _mm_lddqu_si128(__m128i_u const *__p)
Loads data from an unaligned memory location to elements in a 128-bit vector.
Definition pmmintrin.h:45

_mm_movehdup_ps
static __inline__ __m128 __DEFAULT_FN_ATTRS_CONSTEXPR _mm_movehdup_ps(__m128 __a)
Moves and duplicates odd-indexed values from a 128-bit vector of [4 x float] to float values stored i...
Definition pmmintrin.h:130

_mm_hadd_pd
static __inline__ __m128d __DEFAULT_FN_ATTRS_CONSTEXPR _mm_hadd_pd(__m128d __a, __m128d __b)
Horizontally adds the pairs of values contained in two 128-bit vectors of [2 x double].
Definition pmmintrin.h:192

__P
__inline unsigned int unsigned int unsigned int * __P
Definition bmi2intrin.h:25

__Y
__inline unsigned int unsigned int __Y
Definition bmi2intrin.h:19

emmintrin.h