1 /*===------------- avx512ifmavlintrin.h - IFMA intrinsics ------------------===
2 *
3 *
4 * Permission is hereby granted, free of charge, to any person obtaining a copy
5 * of this software and associated documentation files (the "Software"), to deal
6 * in the Software without restriction, including without limitation the rights
7 * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
8 * copies of the Software, and to permit persons to whom the Software is
9 * furnished to do so, subject to the following conditions:
10 *
11 * The above copyright notice and this permission notice shall be included in
12 * all copies or substantial portions of the Software.
13 *
14 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
15 * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
16 * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
17 * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
18 * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
19 * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
20 * THE SOFTWARE.
21 *
22 *===-----------------------------------------------------------------------===
23 */
24 #ifndef __IMMINTRIN_H
25 #error "Never use <avx512ifmavlintrin.h> directly; include <immintrin.h> instead."
26 #endif
27
28 #ifndef __IFMAVLINTRIN_H
29 #define __IFMAVLINTRIN_H
30
31 /* Define the default attributes for the functions in this file. */
32 #define __DEFAULT_FN_ATTRS __attribute__((__always_inline__, __nodebug__, __target__("avx512ifma,avx512vl")))
33
34
35
36 static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_madd52hi_epu64(__m128i __X,__m128i __Y,__m128i __Z)37 _mm_madd52hi_epu64 (__m128i __X, __m128i __Y, __m128i __Z)
38 {
39 return (__m128i) __builtin_ia32_vpmadd52huq128_mask ((__v2di) __X,
40 (__v2di) __Y,
41 (__v2di) __Z,
42 (__mmask8) -1);
43 }
44
45 static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_mask_madd52hi_epu64(__m128i __W,__mmask8 __M,__m128i __X,__m128i __Y)46 _mm_mask_madd52hi_epu64 (__m128i __W, __mmask8 __M, __m128i __X, __m128i __Y)
47 {
48 return (__m128i) __builtin_ia32_vpmadd52huq128_mask ((__v2di) __W,
49 (__v2di) __X,
50 (__v2di) __Y,
51 (__mmask8) __M);
52 }
53
54 static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_maskz_madd52hi_epu64(__mmask8 __M,__m128i __X,__m128i __Y,__m128i __Z)55 _mm_maskz_madd52hi_epu64 (__mmask8 __M, __m128i __X, __m128i __Y, __m128i __Z)
56 {
57 return (__m128i) __builtin_ia32_vpmadd52huq128_maskz ((__v2di) __X,
58 (__v2di) __Y,
59 (__v2di) __Z,
60 (__mmask8) __M);
61 }
62
63 static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_madd52hi_epu64(__m256i __X,__m256i __Y,__m256i __Z)64 _mm256_madd52hi_epu64 (__m256i __X, __m256i __Y, __m256i __Z)
65 {
66 return (__m256i) __builtin_ia32_vpmadd52huq256_mask ((__v4di) __X,
67 (__v4di) __Y,
68 (__v4di) __Z,
69 (__mmask8) -1);
70 }
71
72 static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_mask_madd52hi_epu64(__m256i __W,__mmask8 __M,__m256i __X,__m256i __Y)73 _mm256_mask_madd52hi_epu64 (__m256i __W, __mmask8 __M, __m256i __X,
74 __m256i __Y)
75 {
76 return (__m256i) __builtin_ia32_vpmadd52huq256_mask ((__v4di) __W,
77 (__v4di) __X,
78 (__v4di) __Y,
79 (__mmask8) __M);
80 }
81
82 static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_maskz_madd52hi_epu64(__mmask8 __M,__m256i __X,__m256i __Y,__m256i __Z)83 _mm256_maskz_madd52hi_epu64 (__mmask8 __M, __m256i __X, __m256i __Y, __m256i __Z)
84 {
85 return (__m256i) __builtin_ia32_vpmadd52huq256_maskz ((__v4di) __X,
86 (__v4di) __Y,
87 (__v4di) __Z,
88 (__mmask8) __M);
89 }
90
91 static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_madd52lo_epu64(__m128i __X,__m128i __Y,__m128i __Z)92 _mm_madd52lo_epu64 (__m128i __X, __m128i __Y, __m128i __Z)
93 {
94 return (__m128i) __builtin_ia32_vpmadd52luq128_mask ((__v2di) __X,
95 (__v2di) __Y,
96 (__v2di) __Z,
97 (__mmask8) -1);
98 }
99
100 static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_mask_madd52lo_epu64(__m128i __W,__mmask8 __M,__m128i __X,__m128i __Y)101 _mm_mask_madd52lo_epu64 (__m128i __W, __mmask8 __M, __m128i __X, __m128i __Y)
102 {
103 return (__m128i) __builtin_ia32_vpmadd52luq128_mask ((__v2di) __W,
104 (__v2di) __X,
105 (__v2di) __Y,
106 (__mmask8) __M);
107 }
108
109 static __inline__ __m128i __DEFAULT_FN_ATTRS
_mm_maskz_madd52lo_epu64(__mmask8 __M,__m128i __X,__m128i __Y,__m128i __Z)110 _mm_maskz_madd52lo_epu64 (__mmask8 __M, __m128i __X, __m128i __Y, __m128i __Z)
111 {
112 return (__m128i) __builtin_ia32_vpmadd52luq128_maskz ((__v2di) __X,
113 (__v2di) __Y,
114 (__v2di) __Z,
115 (__mmask8) __M);
116 }
117
118 static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_madd52lo_epu64(__m256i __X,__m256i __Y,__m256i __Z)119 _mm256_madd52lo_epu64 (__m256i __X, __m256i __Y, __m256i __Z)
120 {
121 return (__m256i) __builtin_ia32_vpmadd52luq256_mask ((__v4di) __X,
122 (__v4di) __Y,
123 (__v4di) __Z,
124 (__mmask8) -1);
125 }
126
127 static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_mask_madd52lo_epu64(__m256i __W,__mmask8 __M,__m256i __X,__m256i __Y)128 _mm256_mask_madd52lo_epu64 (__m256i __W, __mmask8 __M, __m256i __X,
129 __m256i __Y)
130 {
131 return (__m256i) __builtin_ia32_vpmadd52luq256_mask ((__v4di) __W,
132 (__v4di) __X,
133 (__v4di) __Y,
134 (__mmask8) __M);
135 }
136
137 static __inline__ __m256i __DEFAULT_FN_ATTRS
_mm256_maskz_madd52lo_epu64(__mmask8 __M,__m256i __X,__m256i __Y,__m256i __Z)138 _mm256_maskz_madd52lo_epu64 (__mmask8 __M, __m256i __X, __m256i __Y, __m256i __Z)
139 {
140 return (__m256i) __builtin_ia32_vpmadd52luq256_maskz ((__v4di) __X,
141 (__v4di) __Y,
142 (__v4di) __Z,
143 (__mmask8) __M);
144 }
145
146
147 #undef __DEFAULT_FN_ATTRS
148
149 #endif
150