summaryrefslogtreecommitdiffstats
path: root/clang/lib
diff options
context:
space:
mode:
authorCraig Topper <craig.topper@intel.com>2018-05-26 18:55:26 +0000
committerCraig Topper <craig.topper@intel.com>2018-05-26 18:55:26 +0000
commit387b1423db4df0ef2b387bf0cb517c11d62d37d9 (patch)
tree09d38b38a5d0fd5b9ce7f0f306aa5cc31266845c /clang/lib
parente091523c8297ba915d23002bc39b592ae5115e3f (diff)
downloadbcm5719-llvm-387b1423db4df0ef2b387bf0cb517c11d62d37d9.tar.gz
bcm5719-llvm-387b1423db4df0ef2b387bf0cb517c11d62d37d9.zip
[X86] Remove mask from avx512ifma builtins. Use a select instruction instead.
This reduces from 12 builtins to 6 since we no longer need a mask and maskz version. llvm-svn: 333348
Diffstat (limited to 'clang/lib')
-rw-r--r--clang/lib/Headers/avx512ifmaintrin.h46
-rw-r--r--clang/lib/Headers/avx512ifmavlintrin.h86
2 files changed, 52 insertions, 80 deletions
diff --git a/clang/lib/Headers/avx512ifmaintrin.h b/clang/lib/Headers/avx512ifmaintrin.h
index 5defbaea8bc..0018ef5de9a 100644
--- a/clang/lib/Headers/avx512ifmaintrin.h
+++ b/clang/lib/Headers/avx512ifmaintrin.h
@@ -34,57 +34,47 @@
static __inline__ __m512i __DEFAULT_FN_ATTRS
_mm512_madd52hi_epu64 (__m512i __X, __m512i __Y, __m512i __Z)
{
- return (__m512i) __builtin_ia32_vpmadd52huq512_mask ((__v8di) __X,
- (__v8di) __Y,
- (__v8di) __Z,
- (__mmask8) -1);
+ return (__m512i)__builtin_ia32_vpmadd52huq512((__v8di) __X, (__v8di) __Y,
+ (__v8di) __Z);
}
static __inline__ __m512i __DEFAULT_FN_ATTRS
-_mm512_mask_madd52hi_epu64 (__m512i __W, __mmask8 __M, __m512i __X,
- __m512i __Y)
+_mm512_mask_madd52hi_epu64 (__m512i __W, __mmask8 __M, __m512i __X, __m512i __Y)
{
- return (__m512i) __builtin_ia32_vpmadd52huq512_mask ((__v8di) __W,
- (__v8di) __X,
- (__v8di) __Y,
- (__mmask8) __M);
+ return (__m512i)__builtin_ia32_selectq_512(__M,
+ (__v8di)_mm512_madd52hi_epu64(__W, __X, __Y),
+ (__v8di)__W);
}
static __inline__ __m512i __DEFAULT_FN_ATTRS
_mm512_maskz_madd52hi_epu64 (__mmask8 __M, __m512i __X, __m512i __Y, __m512i __Z)
{
- return (__m512i) __builtin_ia32_vpmadd52huq512_maskz ((__v8di) __X,
- (__v8di) __Y,
- (__v8di) __Z,
- (__mmask8) __M);
+ return (__m512i)__builtin_ia32_selectq_512(__M,
+ (__v8di)_mm512_madd52hi_epu64(__X, __Y, __Z),
+ (__v8di)_mm512_setzero_si512());
}
static __inline__ __m512i __DEFAULT_FN_ATTRS
_mm512_madd52lo_epu64 (__m512i __X, __m512i __Y, __m512i __Z)
{
- return (__m512i) __builtin_ia32_vpmadd52luq512_mask ((__v8di) __X,
- (__v8di) __Y,
- (__v8di) __Z,
- (__mmask8) -1);
+ return (__m512i)__builtin_ia32_vpmadd52luq512((__v8di) __X, (__v8di) __Y,
+ (__v8di) __Z);
}
static __inline__ __m512i __DEFAULT_FN_ATTRS
-_mm512_mask_madd52lo_epu64 (__m512i __W, __mmask8 __M, __m512i __X,
- __m512i __Y)
+_mm512_mask_madd52lo_epu64 (__m512i __W, __mmask8 __M, __m512i __X, __m512i __Y)
{
- return (__m512i) __builtin_ia32_vpmadd52luq512_mask ((__v8di) __W,
- (__v8di) __X,
- (__v8di) __Y,
- (__mmask8) __M);
+ return (__m512i)__builtin_ia32_selectq_512(__M,
+ (__v8di)_mm512_madd52lo_epu64(__W, __X, __Y),
+ (__v8di)__W);
}
static __inline__ __m512i __DEFAULT_FN_ATTRS
_mm512_maskz_madd52lo_epu64 (__mmask8 __M, __m512i __X, __m512i __Y, __m512i __Z)
{
- return (__m512i) __builtin_ia32_vpmadd52luq512_maskz ((__v8di) __X,
- (__v8di) __Y,
- (__v8di) __Z,
- (__mmask8) __M);
+ return (__m512i)__builtin_ia32_selectq_512(__M,
+ (__v8di)_mm512_madd52lo_epu64(__X, __Y, __Z),
+ (__v8di)_mm512_setzero_si512());
}
#undef __DEFAULT_FN_ATTRS
diff --git a/clang/lib/Headers/avx512ifmavlintrin.h b/clang/lib/Headers/avx512ifmavlintrin.h
index 131ee5cb4f8..934cfb433f9 100644
--- a/clang/lib/Headers/avx512ifmavlintrin.h
+++ b/clang/lib/Headers/avx512ifmavlintrin.h
@@ -36,111 +36,93 @@
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_madd52hi_epu64 (__m128i __X, __m128i __Y, __m128i __Z)
{
- return (__m128i) __builtin_ia32_vpmadd52huq128_mask ((__v2di) __X,
- (__v2di) __Y,
- (__v2di) __Z,
- (__mmask8) -1);
+ return (__m128i)__builtin_ia32_vpmadd52huq128((__v2di) __X, (__v2di) __Y,
+ (__v2di) __Z);
}
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_mask_madd52hi_epu64 (__m128i __W, __mmask8 __M, __m128i __X, __m128i __Y)
{
- return (__m128i) __builtin_ia32_vpmadd52huq128_mask ((__v2di) __W,
- (__v2di) __X,
- (__v2di) __Y,
- (__mmask8) __M);
+ return (__m128i)__builtin_ia32_selectq_128(__M,
+ (__v2di)_mm_madd52hi_epu64(__W, __X, __Y),
+ (__v2di)__W);
}
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_maskz_madd52hi_epu64 (__mmask8 __M, __m128i __X, __m128i __Y, __m128i __Z)
{
- return (__m128i) __builtin_ia32_vpmadd52huq128_maskz ((__v2di) __X,
- (__v2di) __Y,
- (__v2di) __Z,
- (__mmask8) __M);
+ return (__m128i)__builtin_ia32_selectq_128(__M,
+ (__v2di)_mm_madd52hi_epu64(__X, __Y, __Z),
+ (__v2di)_mm_setzero_si128());
}
static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_madd52hi_epu64 (__m256i __X, __m256i __Y, __m256i __Z)
{
- return (__m256i) __builtin_ia32_vpmadd52huq256_mask ((__v4di) __X,
- (__v4di) __Y,
- (__v4di) __Z,
- (__mmask8) -1);
+ return (__m256i)__builtin_ia32_vpmadd52huq256((__v4di)__X, (__v4di)__Y,
+ (__v4di)__Z);
}
static __inline__ __m256i __DEFAULT_FN_ATTRS
-_mm256_mask_madd52hi_epu64 (__m256i __W, __mmask8 __M, __m256i __X,
- __m256i __Y)
+_mm256_mask_madd52hi_epu64 (__m256i __W, __mmask8 __M, __m256i __X, __m256i __Y)
{
- return (__m256i) __builtin_ia32_vpmadd52huq256_mask ((__v4di) __W,
- (__v4di) __X,
- (__v4di) __Y,
- (__mmask8) __M);
+ return (__m256i)__builtin_ia32_selectq_256(__M,
+ (__v4di)_mm256_madd52hi_epu64(__W, __X, __Y),
+ (__v4di)__W);
}
static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_maskz_madd52hi_epu64 (__mmask8 __M, __m256i __X, __m256i __Y, __m256i __Z)
{
- return (__m256i) __builtin_ia32_vpmadd52huq256_maskz ((__v4di) __X,
- (__v4di) __Y,
- (__v4di) __Z,
- (__mmask8) __M);
+ return (__m256i)__builtin_ia32_selectq_256(__M,
+ (__v4di)_mm256_madd52hi_epu64(__X, __Y, __Z),
+ (__v4di)_mm256_setzero_si256());
}
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_madd52lo_epu64 (__m128i __X, __m128i __Y, __m128i __Z)
{
- return (__m128i) __builtin_ia32_vpmadd52luq128_mask ((__v2di) __X,
- (__v2di) __Y,
- (__v2di) __Z,
- (__mmask8) -1);
+ return (__m128i)__builtin_ia32_vpmadd52luq128((__v2di)__X, (__v2di)__Y,
+ (__v2di)__Z);
}
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_mask_madd52lo_epu64 (__m128i __W, __mmask8 __M, __m128i __X, __m128i __Y)
{
- return (__m128i) __builtin_ia32_vpmadd52luq128_mask ((__v2di) __W,
- (__v2di) __X,
- (__v2di) __Y,
- (__mmask8) __M);
+ return (__m128i)__builtin_ia32_selectq_128(__M,
+ (__v2di)_mm_madd52lo_epu64(__W, __X, __Y),
+ (__v2di)__W);
}
static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_maskz_madd52lo_epu64 (__mmask8 __M, __m128i __X, __m128i __Y, __m128i __Z)
{
- return (__m128i) __builtin_ia32_vpmadd52luq128_maskz ((__v2di) __X,
- (__v2di) __Y,
- (__v2di) __Z,
- (__mmask8) __M);
+ return (__m128i)__builtin_ia32_selectq_128(__M,
+ (__v2di)_mm_madd52lo_epu64(__X, __Y, __Z),
+ (__v2di)_mm_setzero_si128());
}
static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_madd52lo_epu64 (__m256i __X, __m256i __Y, __m256i __Z)
{
- return (__m256i) __builtin_ia32_vpmadd52luq256_mask ((__v4di) __X,
- (__v4di) __Y,
- (__v4di) __Z,
- (__mmask8) -1);
+ return (__m256i)__builtin_ia32_vpmadd52luq256((__v4di)__X, (__v4di)__Y,
+ (__v4di)__Z);
}
static __inline__ __m256i __DEFAULT_FN_ATTRS
-_mm256_mask_madd52lo_epu64 (__m256i __W, __mmask8 __M, __m256i __X,
- __m256i __Y)
+_mm256_mask_madd52lo_epu64 (__m256i __W, __mmask8 __M, __m256i __X, __m256i __Y)
{
- return (__m256i) __builtin_ia32_vpmadd52luq256_mask ((__v4di) __W,
- (__v4di) __X,
- (__v4di) __Y,
- (__mmask8) __M);
+ return (__m256i)__builtin_ia32_selectq_256(__M,
+ (__v4di)_mm256_madd52lo_epu64(__W, __X, __Y),
+ (__v4di)__W);
}
static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_maskz_madd52lo_epu64 (__mmask8 __M, __m256i __X, __m256i __Y, __m256i __Z)
{
- return (__m256i) __builtin_ia32_vpmadd52luq256_maskz ((__v4di) __X,
- (__v4di) __Y,
- (__v4di) __Z,
- (__mmask8) __M);
+ return (__m256i)__builtin_ia32_selectq_256(__M,
+ (__v4di)_mm256_madd52lo_epu64(__X, __Y, __Z),
+ (__v4di)_mm256_setzero_si256());
}
OpenPOWER on IntegriCloud