Skip to main content

core/stdarch/crates/core_arch/src/x86/
sse2.rs

1//! Streaming SIMD Extensions 2 (SSE2)
2
3#[cfg(test)]
4use stdarch_test::assert_instr;
5
6use crate::{
7    core_arch::{simd::*, x86::*},
8    intrinsics::simd::*,
9    intrinsics::sqrtf64,
10    mem, ptr,
11};
12
13/// Provides a hint to the processor that the code sequence is a spin-wait loop.
14///
15/// This can help improve the performance and power consumption of spin-wait
16/// loops.
17///
18/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_pause)
19#[inline]
20#[cfg_attr(all(test, target_feature = "sse2"), assert_instr(pause))]
21#[stable(feature = "simd_x86", since = "1.27.0")]
22pub fn _mm_pause() {
23    // note: `pause` is guaranteed to be interpreted as a `nop` by CPUs without
24    // the SSE2 target-feature - therefore it does not require any target features
25    unsafe { pause() }
26}
27
28/// Invalidates and flushes the cache line that contains `p` from all levels of
29/// the cache hierarchy.
30///
31/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_clflush)
32///
33/// # Safety
34///
35/// Unlike the prefetch intrinsics, `CLFLUSH` is subject to all the permission
36/// checking and faults associated with a byte load, so `p` must point to a
37/// byte that is valid for reads.
38#[inline]
39#[target_feature(enable = "sse2")]
40#[cfg_attr(test, assert_instr(clflush))]
41#[stable(feature = "simd_x86", since = "1.27.0")]
42pub unsafe fn _mm_clflush(p: *const u8) {
43    clflush(p)
44}
45
46/// Performs a serializing operation on all load-from-memory instructions
47/// that were issued prior to this instruction.
48///
49/// Guarantees that every load instruction that precedes, in program order, is
50/// globally visible before any load instruction which follows the fence in
51/// program order.
52///
53/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_lfence)
54#[inline]
55#[target_feature(enable = "sse2")]
56#[cfg_attr(test, assert_instr(lfence))]
57#[stable(feature = "simd_x86", since = "1.27.0")]
58pub fn _mm_lfence() {
59    unsafe { lfence() }
60}
61
62/// Performs a serializing operation on all load-from-memory and store-to-memory
63/// instructions that were issued prior to this instruction.
64///
65/// Guarantees that every memory access that precedes, in program order, the
66/// memory fence instruction is globally visible before any memory instruction
67/// which follows the fence in program order.
68///
69/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_mfence)
70#[inline]
71#[target_feature(enable = "sse2")]
72#[cfg_attr(test, assert_instr(mfence))]
73#[stable(feature = "simd_x86", since = "1.27.0")]
74pub fn _mm_mfence() {
75    unsafe { mfence() }
76}
77
78/// Adds packed 8-bit integers in `a` and `b`.
79///
80/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_add_epi8)
81#[inline]
82#[target_feature(enable = "sse2")]
83#[cfg_attr(test, assert_instr(paddb))]
84#[stable(feature = "simd_x86", since = "1.27.0")]
85#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
86pub const fn _mm_add_epi8(a: __m128i, b: __m128i) -> __m128i {
87    unsafe { transmute(simd_add(a.as_i8x16(), b.as_i8x16())) }
88}
89
90/// Adds packed 16-bit integers in `a` and `b`.
91///
92/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_add_epi16)
93#[inline]
94#[target_feature(enable = "sse2")]
95#[cfg_attr(test, assert_instr(paddw))]
96#[stable(feature = "simd_x86", since = "1.27.0")]
97#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
98pub const fn _mm_add_epi16(a: __m128i, b: __m128i) -> __m128i {
99    unsafe { transmute(simd_add(a.as_i16x8(), b.as_i16x8())) }
100}
101
102/// Adds packed 32-bit integers in `a` and `b`.
103///
104/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_add_epi32)
105#[inline]
106#[target_feature(enable = "sse2")]
107#[cfg_attr(test, assert_instr(paddd))]
108#[stable(feature = "simd_x86", since = "1.27.0")]
109#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
110pub const fn _mm_add_epi32(a: __m128i, b: __m128i) -> __m128i {
111    unsafe { transmute(simd_add(a.as_i32x4(), b.as_i32x4())) }
112}
113
114/// Adds packed 64-bit integers in `a` and `b`.
115///
116/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_add_epi64)
117#[inline]
118#[target_feature(enable = "sse2")]
119#[cfg_attr(test, assert_instr(paddq))]
120#[stable(feature = "simd_x86", since = "1.27.0")]
121#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
122pub const fn _mm_add_epi64(a: __m128i, b: __m128i) -> __m128i {
123    unsafe { transmute(simd_add(a.as_i64x2(), b.as_i64x2())) }
124}
125
126/// Adds packed 8-bit integers in `a` and `b` using saturation.
127///
128/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_adds_epi8)
129#[inline]
130#[target_feature(enable = "sse2")]
131#[cfg_attr(test, assert_instr(paddsb))]
132#[stable(feature = "simd_x86", since = "1.27.0")]
133#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
134pub const fn _mm_adds_epi8(a: __m128i, b: __m128i) -> __m128i {
135    unsafe { transmute(simd_saturating_add(a.as_i8x16(), b.as_i8x16())) }
136}
137
138/// Adds packed 16-bit integers in `a` and `b` using saturation.
139///
140/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_adds_epi16)
141#[inline]
142#[target_feature(enable = "sse2")]
143#[cfg_attr(test, assert_instr(paddsw))]
144#[stable(feature = "simd_x86", since = "1.27.0")]
145#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
146pub const fn _mm_adds_epi16(a: __m128i, b: __m128i) -> __m128i {
147    unsafe { transmute(simd_saturating_add(a.as_i16x8(), b.as_i16x8())) }
148}
149
150/// Adds packed unsigned 8-bit integers in `a` and `b` using saturation.
151///
152/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_adds_epu8)
153#[inline]
154#[target_feature(enable = "sse2")]
155#[cfg_attr(test, assert_instr(paddusb))]
156#[stable(feature = "simd_x86", since = "1.27.0")]
157#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
158pub const fn _mm_adds_epu8(a: __m128i, b: __m128i) -> __m128i {
159    unsafe { transmute(simd_saturating_add(a.as_u8x16(), b.as_u8x16())) }
160}
161
162/// Adds packed unsigned 16-bit integers in `a` and `b` using saturation.
163///
164/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_adds_epu16)
165#[inline]
166#[target_feature(enable = "sse2")]
167#[cfg_attr(test, assert_instr(paddusw))]
168#[stable(feature = "simd_x86", since = "1.27.0")]
169#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
170pub const fn _mm_adds_epu16(a: __m128i, b: __m128i) -> __m128i {
171    unsafe { transmute(simd_saturating_add(a.as_u16x8(), b.as_u16x8())) }
172}
173
174/// Averages packed unsigned 8-bit integers in `a` and `b`.
175///
176/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_avg_epu8)
177#[inline]
178#[target_feature(enable = "sse2")]
179#[cfg_attr(test, assert_instr(pavgb))]
180#[stable(feature = "simd_x86", since = "1.27.0")]
181#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
182pub const fn _mm_avg_epu8(a: __m128i, b: __m128i) -> __m128i {
183    unsafe {
184        let a = simd_cast::<_, u16x16>(a.as_u8x16());
185        let b = simd_cast::<_, u16x16>(b.as_u8x16());
186        let r = simd_shr(simd_add(simd_add(a, b), u16x16::splat(1)), u16x16::splat(1));
187        transmute(simd_cast::<_, u8x16>(r))
188    }
189}
190
191/// Averages packed unsigned 16-bit integers in `a` and `b`.
192///
193/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_avg_epu16)
194#[inline]
195#[target_feature(enable = "sse2")]
196#[cfg_attr(test, assert_instr(pavgw))]
197#[stable(feature = "simd_x86", since = "1.27.0")]
198#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
199pub const fn _mm_avg_epu16(a: __m128i, b: __m128i) -> __m128i {
200    unsafe {
201        let a = simd_cast::<_, u32x8>(a.as_u16x8());
202        let b = simd_cast::<_, u32x8>(b.as_u16x8());
203        let r = simd_shr(simd_add(simd_add(a, b), u32x8::splat(1)), u32x8::splat(1));
204        transmute(simd_cast::<_, u16x8>(r))
205    }
206}
207
208/// Multiplies and then horizontally add signed 16 bit integers in `a` and `b`.
209///
210/// Multiplies packed signed 16-bit integers in `a` and `b`, producing
211/// intermediate signed 32-bit integers. Horizontally add adjacent pairs of
212/// intermediate 32-bit integers.
213///
214/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_madd_epi16)
215#[inline]
216#[target_feature(enable = "sse2")]
217#[cfg_attr(test, assert_instr(pmaddwd))]
218#[stable(feature = "simd_x86", since = "1.27.0")]
219pub fn _mm_madd_epi16(a: __m128i, b: __m128i) -> __m128i {
220    // It's a trick used in the Adler-32 algorithm to perform a widening addition.
221    //
222    // ```rust
223    // #[target_feature(enable = "sse2")]
224    // unsafe fn widening_add(mad: __m128i) -> __m128i {
225    //     _mm_madd_epi16(mad, _mm_set1_epi16(1))
226    // }
227    // ```
228    //
229    // If we implement this using generic vector intrinsics, the optimizer
230    // will eliminate this pattern, and `pmaddwd` will no longer be emitted.
231    // For this reason, we use x86 intrinsics.
232    unsafe { transmute(pmaddwd(a.as_i16x8(), b.as_i16x8())) }
233}
234
235/// Compares packed 16-bit integers in `a` and `b`, and returns the packed
236/// maximum values.
237///
238/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_max_epi16)
239#[inline]
240#[target_feature(enable = "sse2")]
241#[cfg_attr(test, assert_instr(pmaxsw))]
242#[stable(feature = "simd_x86", since = "1.27.0")]
243#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
244pub const fn _mm_max_epi16(a: __m128i, b: __m128i) -> __m128i {
245    unsafe { simd_imax(a.as_i16x8(), b.as_i16x8()).as_m128i() }
246}
247
248/// Compares packed unsigned 8-bit integers in `a` and `b`, and returns the
249/// packed maximum values.
250///
251/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_max_epu8)
252#[inline]
253#[target_feature(enable = "sse2")]
254#[cfg_attr(test, assert_instr(pmaxub))]
255#[stable(feature = "simd_x86", since = "1.27.0")]
256#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
257pub const fn _mm_max_epu8(a: __m128i, b: __m128i) -> __m128i {
258    unsafe { simd_imax(a.as_u8x16(), b.as_u8x16()).as_m128i() }
259}
260
261/// Compares packed 16-bit integers in `a` and `b`, and returns the packed
262/// minimum values.
263///
264/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_min_epi16)
265#[inline]
266#[target_feature(enable = "sse2")]
267#[cfg_attr(test, assert_instr(pminsw))]
268#[stable(feature = "simd_x86", since = "1.27.0")]
269#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
270pub const fn _mm_min_epi16(a: __m128i, b: __m128i) -> __m128i {
271    unsafe { simd_imin(a.as_i16x8(), b.as_i16x8()).as_m128i() }
272}
273
274/// Compares packed unsigned 8-bit integers in `a` and `b`, and returns the
275/// packed minimum values.
276///
277/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_min_epu8)
278#[inline]
279#[target_feature(enable = "sse2")]
280#[cfg_attr(test, assert_instr(pminub))]
281#[stable(feature = "simd_x86", since = "1.27.0")]
282#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
283pub const fn _mm_min_epu8(a: __m128i, b: __m128i) -> __m128i {
284    unsafe { simd_imin(a.as_u8x16(), b.as_u8x16()).as_m128i() }
285}
286
287/// Multiplies the packed 16-bit integers in `a` and `b`.
288///
289/// The multiplication produces intermediate 32-bit integers, and returns the
290/// high 16 bits of the intermediate integers.
291///
292/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_mulhi_epi16)
293#[inline]
294#[target_feature(enable = "sse2")]
295#[cfg_attr(test, assert_instr(pmulhw))]
296#[stable(feature = "simd_x86", since = "1.27.0")]
297#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
298pub const fn _mm_mulhi_epi16(a: __m128i, b: __m128i) -> __m128i {
299    unsafe {
300        let a = simd_cast::<_, i32x8>(a.as_i16x8());
301        let b = simd_cast::<_, i32x8>(b.as_i16x8());
302        let r = simd_shr(simd_mul(a, b), i32x8::splat(16));
303        transmute(simd_cast::<i32x8, i16x8>(r))
304    }
305}
306
307/// Multiplies the packed unsigned 16-bit integers in `a` and `b`.
308///
309/// The multiplication produces intermediate 32-bit integers, and returns the
310/// high 16 bits of the intermediate integers.
311///
312/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_mulhi_epu16)
313#[inline]
314#[target_feature(enable = "sse2")]
315#[cfg_attr(test, assert_instr(pmulhuw))]
316#[stable(feature = "simd_x86", since = "1.27.0")]
317#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
318pub const fn _mm_mulhi_epu16(a: __m128i, b: __m128i) -> __m128i {
319    unsafe {
320        let a = simd_cast::<_, u32x8>(a.as_u16x8());
321        let b = simd_cast::<_, u32x8>(b.as_u16x8());
322        let r = simd_shr(simd_mul(a, b), u32x8::splat(16));
323        transmute(simd_cast::<u32x8, u16x8>(r))
324    }
325}
326
327/// Multiplies the packed 16-bit integers in `a` and `b`.
328///
329/// The multiplication produces intermediate 32-bit integers, and returns the
330/// low 16 bits of the intermediate integers.
331///
332/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_mullo_epi16)
333#[inline]
334#[target_feature(enable = "sse2")]
335#[cfg_attr(test, assert_instr(pmullw))]
336#[stable(feature = "simd_x86", since = "1.27.0")]
337#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
338pub const fn _mm_mullo_epi16(a: __m128i, b: __m128i) -> __m128i {
339    unsafe { transmute(simd_mul(a.as_i16x8(), b.as_i16x8())) }
340}
341
342/// Multiplies the low unsigned 32-bit integers from each packed 64-bit element
343/// in `a` and `b`.
344///
345/// Returns the unsigned 64-bit results.
346///
347/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_mul_epu32)
348#[inline]
349#[target_feature(enable = "sse2")]
350#[cfg_attr(test, assert_instr(pmuludq))]
351#[stable(feature = "simd_x86", since = "1.27.0")]
352#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
353pub const fn _mm_mul_epu32(a: __m128i, b: __m128i) -> __m128i {
354    unsafe {
355        let a = a.as_u64x2();
356        let b = b.as_u64x2();
357        let mask = u64x2::splat(u32::MAX as u64);
358        transmute(simd_mul(simd_and(a, mask), simd_and(b, mask)))
359    }
360}
361
362/// Sum the absolute differences of packed unsigned 8-bit integers.
363///
364/// Computes the absolute differences of packed unsigned 8-bit integers in `a`
365/// and `b`, then horizontally sum each consecutive 8 differences to produce
366/// two unsigned 16-bit integers, and pack these unsigned 16-bit integers in
367/// the low 16 bits of 64-bit elements returned.
368///
369/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sad_epu8)
370#[inline]
371#[target_feature(enable = "sse2")]
372#[cfg_attr(test, assert_instr(psadbw))]
373#[stable(feature = "simd_x86", since = "1.27.0")]
374pub fn _mm_sad_epu8(a: __m128i, b: __m128i) -> __m128i {
375    unsafe { transmute(psadbw(a.as_u8x16(), b.as_u8x16())) }
376}
377
378/// Subtracts packed 8-bit integers in `b` from packed 8-bit integers in `a`.
379///
380/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sub_epi8)
381#[inline]
382#[target_feature(enable = "sse2")]
383#[cfg_attr(test, assert_instr(psubb))]
384#[stable(feature = "simd_x86", since = "1.27.0")]
385#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
386pub const fn _mm_sub_epi8(a: __m128i, b: __m128i) -> __m128i {
387    unsafe { transmute(simd_sub(a.as_i8x16(), b.as_i8x16())) }
388}
389
390/// Subtracts packed 16-bit integers in `b` from packed 16-bit integers in `a`.
391///
392/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sub_epi16)
393#[inline]
394#[target_feature(enable = "sse2")]
395#[cfg_attr(test, assert_instr(psubw))]
396#[stable(feature = "simd_x86", since = "1.27.0")]
397#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
398pub const fn _mm_sub_epi16(a: __m128i, b: __m128i) -> __m128i {
399    unsafe { transmute(simd_sub(a.as_i16x8(), b.as_i16x8())) }
400}
401
402/// Subtract packed 32-bit integers in `b` from packed 32-bit integers in `a`.
403///
404/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sub_epi32)
405#[inline]
406#[target_feature(enable = "sse2")]
407#[cfg_attr(test, assert_instr(psubd))]
408#[stable(feature = "simd_x86", since = "1.27.0")]
409#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
410pub const fn _mm_sub_epi32(a: __m128i, b: __m128i) -> __m128i {
411    unsafe { transmute(simd_sub(a.as_i32x4(), b.as_i32x4())) }
412}
413
414/// Subtract packed 64-bit integers in `b` from packed 64-bit integers in `a`.
415///
416/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sub_epi64)
417#[inline]
418#[target_feature(enable = "sse2")]
419#[cfg_attr(test, assert_instr(psubq))]
420#[stable(feature = "simd_x86", since = "1.27.0")]
421#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
422pub const fn _mm_sub_epi64(a: __m128i, b: __m128i) -> __m128i {
423    unsafe { transmute(simd_sub(a.as_i64x2(), b.as_i64x2())) }
424}
425
426/// Subtract packed 8-bit integers in `b` from packed 8-bit integers in `a`
427/// using saturation.
428///
429/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_subs_epi8)
430#[inline]
431#[target_feature(enable = "sse2")]
432#[cfg_attr(test, assert_instr(psubsb))]
433#[stable(feature = "simd_x86", since = "1.27.0")]
434#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
435pub const fn _mm_subs_epi8(a: __m128i, b: __m128i) -> __m128i {
436    unsafe { transmute(simd_saturating_sub(a.as_i8x16(), b.as_i8x16())) }
437}
438
439/// Subtract packed 16-bit integers in `b` from packed 16-bit integers in `a`
440/// using saturation.
441///
442/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_subs_epi16)
443#[inline]
444#[target_feature(enable = "sse2")]
445#[cfg_attr(test, assert_instr(psubsw))]
446#[stable(feature = "simd_x86", since = "1.27.0")]
447#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
448pub const fn _mm_subs_epi16(a: __m128i, b: __m128i) -> __m128i {
449    unsafe { transmute(simd_saturating_sub(a.as_i16x8(), b.as_i16x8())) }
450}
451
452/// Subtract packed unsigned 8-bit integers in `b` from packed unsigned 8-bit
453/// integers in `a` using saturation.
454///
455/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_subs_epu8)
456#[inline]
457#[target_feature(enable = "sse2")]
458#[cfg_attr(test, assert_instr(psubusb))]
459#[stable(feature = "simd_x86", since = "1.27.0")]
460#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
461pub const fn _mm_subs_epu8(a: __m128i, b: __m128i) -> __m128i {
462    unsafe { transmute(simd_saturating_sub(a.as_u8x16(), b.as_u8x16())) }
463}
464
465/// Subtract packed unsigned 16-bit integers in `b` from packed unsigned 16-bit
466/// integers in `a` using saturation.
467///
468/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_subs_epu16)
469#[inline]
470#[target_feature(enable = "sse2")]
471#[cfg_attr(test, assert_instr(psubusw))]
472#[stable(feature = "simd_x86", since = "1.27.0")]
473#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
474pub const fn _mm_subs_epu16(a: __m128i, b: __m128i) -> __m128i {
475    unsafe { transmute(simd_saturating_sub(a.as_u16x8(), b.as_u16x8())) }
476}
477
478/// Shifts `a` left by `IMM8` bytes while shifting in zeros.
479///
480/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_slli_si128)
481#[inline]
482#[target_feature(enable = "sse2")]
483#[cfg_attr(test, assert_instr(pslldq, IMM8 = 1))]
484#[rustc_legacy_const_generics(1)]
485#[stable(feature = "simd_x86", since = "1.27.0")]
486#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
487pub const fn _mm_slli_si128<const IMM8: i32>(a: __m128i) -> __m128i {
488    static_assert_uimm_bits!(IMM8, 8);
489    unsafe { _mm_slli_si128_impl::<IMM8>(a) }
490}
491
492/// Implementation detail: converts the immediate argument of the
493/// `_mm_slli_si128` intrinsic into a compile-time constant.
494#[inline]
495#[target_feature(enable = "sse2")]
496#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
497const unsafe fn _mm_slli_si128_impl<const IMM8: i32>(a: __m128i) -> __m128i {
498    const fn mask(shift: i32, i: u32) -> u32 {
499        let shift = shift as u32 & 0xff;
500        if shift > 15 { i } else { 16 - shift + i }
501    }
502    transmute::<i8x16, _>(simd_shuffle!(
503        i8x16::ZERO,
504        a.as_i8x16(),
505        [
506            mask(IMM8, 0),
507            mask(IMM8, 1),
508            mask(IMM8, 2),
509            mask(IMM8, 3),
510            mask(IMM8, 4),
511            mask(IMM8, 5),
512            mask(IMM8, 6),
513            mask(IMM8, 7),
514            mask(IMM8, 8),
515            mask(IMM8, 9),
516            mask(IMM8, 10),
517            mask(IMM8, 11),
518            mask(IMM8, 12),
519            mask(IMM8, 13),
520            mask(IMM8, 14),
521            mask(IMM8, 15),
522        ],
523    ))
524}
525
526/// Shifts `a` left by `IMM8` bytes while shifting in zeros.
527///
528/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_bslli_si128)
529#[inline]
530#[target_feature(enable = "sse2")]
531#[cfg_attr(test, assert_instr(pslldq, IMM8 = 1))]
532#[rustc_legacy_const_generics(1)]
533#[stable(feature = "simd_x86", since = "1.27.0")]
534#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
535pub const fn _mm_bslli_si128<const IMM8: i32>(a: __m128i) -> __m128i {
536    unsafe {
537        static_assert_uimm_bits!(IMM8, 8);
538        _mm_slli_si128_impl::<IMM8>(a)
539    }
540}
541
542/// Shifts `a` right by `IMM8` bytes while shifting in zeros.
543///
544/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_bsrli_si128)
545#[inline]
546#[target_feature(enable = "sse2")]
547#[cfg_attr(test, assert_instr(psrldq, IMM8 = 1))]
548#[rustc_legacy_const_generics(1)]
549#[stable(feature = "simd_x86", since = "1.27.0")]
550#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
551pub const fn _mm_bsrli_si128<const IMM8: i32>(a: __m128i) -> __m128i {
552    unsafe {
553        static_assert_uimm_bits!(IMM8, 8);
554        _mm_srli_si128_impl::<IMM8>(a)
555    }
556}
557
558/// Shifts packed 16-bit integers in `a` left by `IMM8` while shifting in zeros.
559///
560/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_slli_epi16)
561#[inline]
562#[target_feature(enable = "sse2")]
563#[cfg_attr(test, assert_instr(psllw, IMM8 = 7))]
564#[rustc_legacy_const_generics(1)]
565#[stable(feature = "simd_x86", since = "1.27.0")]
566#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
567pub const fn _mm_slli_epi16<const IMM8: i32>(a: __m128i) -> __m128i {
568    static_assert_uimm_bits!(IMM8, 8);
569    unsafe {
570        if IMM8 >= 16 {
571            _mm_setzero_si128()
572        } else {
573            transmute(simd_shl(a.as_u16x8(), u16x8::splat(IMM8 as u16)))
574        }
575    }
576}
577
578/// Shifts packed 16-bit integers in `a` left by `count` while shifting in
579/// zeros.
580///
581/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sll_epi16)
582#[inline]
583#[target_feature(enable = "sse2")]
584#[cfg_attr(test, assert_instr(psllw))]
585#[stable(feature = "simd_x86", since = "1.27.0")]
586pub fn _mm_sll_epi16(a: __m128i, count: __m128i) -> __m128i {
587    unsafe { transmute(psllw(a.as_i16x8(), count.as_i16x8())) }
588}
589
590/// Shifts packed 32-bit integers in `a` left by `IMM8` while shifting in zeros.
591///
592/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_slli_epi32)
593#[inline]
594#[target_feature(enable = "sse2")]
595#[cfg_attr(test, assert_instr(pslld, IMM8 = 7))]
596#[rustc_legacy_const_generics(1)]
597#[stable(feature = "simd_x86", since = "1.27.0")]
598#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
599pub const fn _mm_slli_epi32<const IMM8: i32>(a: __m128i) -> __m128i {
600    static_assert_uimm_bits!(IMM8, 8);
601    unsafe {
602        if IMM8 >= 32 {
603            _mm_setzero_si128()
604        } else {
605            transmute(simd_shl(a.as_u32x4(), u32x4::splat(IMM8 as u32)))
606        }
607    }
608}
609
610/// Shifts packed 32-bit integers in `a` left by `count` while shifting in
611/// zeros.
612///
613/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sll_epi32)
614#[inline]
615#[target_feature(enable = "sse2")]
616#[cfg_attr(test, assert_instr(pslld))]
617#[stable(feature = "simd_x86", since = "1.27.0")]
618pub fn _mm_sll_epi32(a: __m128i, count: __m128i) -> __m128i {
619    unsafe { transmute(pslld(a.as_i32x4(), count.as_i32x4())) }
620}
621
622/// Shifts packed 64-bit integers in `a` left by `IMM8` while shifting in zeros.
623///
624/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_slli_epi64)
625#[inline]
626#[target_feature(enable = "sse2")]
627#[cfg_attr(test, assert_instr(psllq, IMM8 = 7))]
628#[rustc_legacy_const_generics(1)]
629#[stable(feature = "simd_x86", since = "1.27.0")]
630#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
631pub const fn _mm_slli_epi64<const IMM8: i32>(a: __m128i) -> __m128i {
632    static_assert_uimm_bits!(IMM8, 8);
633    unsafe {
634        if IMM8 >= 64 {
635            _mm_setzero_si128()
636        } else {
637            transmute(simd_shl(a.as_u64x2(), u64x2::splat(IMM8 as u64)))
638        }
639    }
640}
641
642/// Shifts packed 64-bit integers in `a` left by `count` while shifting in
643/// zeros.
644///
645/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sll_epi64)
646#[inline]
647#[target_feature(enable = "sse2")]
648#[cfg_attr(test, assert_instr(psllq))]
649#[stable(feature = "simd_x86", since = "1.27.0")]
650pub fn _mm_sll_epi64(a: __m128i, count: __m128i) -> __m128i {
651    unsafe { transmute(psllq(a.as_i64x2(), count.as_i64x2())) }
652}
653
654/// Shifts packed 16-bit integers in `a` right by `IMM8` while shifting in sign
655/// bits.
656///
657/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srai_epi16)
658#[inline]
659#[target_feature(enable = "sse2")]
660#[cfg_attr(test, assert_instr(psraw, IMM8 = 1))]
661#[rustc_legacy_const_generics(1)]
662#[stable(feature = "simd_x86", since = "1.27.0")]
663#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
664pub const fn _mm_srai_epi16<const IMM8: i32>(a: __m128i) -> __m128i {
665    static_assert_uimm_bits!(IMM8, 8);
666    unsafe { transmute(simd_shr(a.as_i16x8(), i16x8::splat(IMM8.min(15) as i16))) }
667}
668
669/// Shifts packed 16-bit integers in `a` right by `count` while shifting in sign
670/// bits.
671///
672/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sra_epi16)
673#[inline]
674#[target_feature(enable = "sse2")]
675#[cfg_attr(test, assert_instr(psraw))]
676#[stable(feature = "simd_x86", since = "1.27.0")]
677pub fn _mm_sra_epi16(a: __m128i, count: __m128i) -> __m128i {
678    unsafe { transmute(psraw(a.as_i16x8(), count.as_i16x8())) }
679}
680
681/// Shifts packed 32-bit integers in `a` right by `IMM8` while shifting in sign
682/// bits.
683///
684/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srai_epi32)
685#[inline]
686#[target_feature(enable = "sse2")]
687#[cfg_attr(test, assert_instr(psrad, IMM8 = 1))]
688#[rustc_legacy_const_generics(1)]
689#[stable(feature = "simd_x86", since = "1.27.0")]
690#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
691pub const fn _mm_srai_epi32<const IMM8: i32>(a: __m128i) -> __m128i {
692    static_assert_uimm_bits!(IMM8, 8);
693    unsafe { transmute(simd_shr(a.as_i32x4(), i32x4::splat(IMM8.min(31)))) }
694}
695
696/// Shifts packed 32-bit integers in `a` right by `count` while shifting in sign
697/// bits.
698///
699/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sra_epi32)
700#[inline]
701#[target_feature(enable = "sse2")]
702#[cfg_attr(test, assert_instr(psrad))]
703#[stable(feature = "simd_x86", since = "1.27.0")]
704pub fn _mm_sra_epi32(a: __m128i, count: __m128i) -> __m128i {
705    unsafe { transmute(psrad(a.as_i32x4(), count.as_i32x4())) }
706}
707
708/// Shifts `a` right by `IMM8` bytes while shifting in zeros.
709///
710/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srli_si128)
711#[inline]
712#[target_feature(enable = "sse2")]
713#[cfg_attr(test, assert_instr(psrldq, IMM8 = 1))]
714#[rustc_legacy_const_generics(1)]
715#[stable(feature = "simd_x86", since = "1.27.0")]
716#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
717pub const fn _mm_srli_si128<const IMM8: i32>(a: __m128i) -> __m128i {
718    static_assert_uimm_bits!(IMM8, 8);
719    unsafe { _mm_srli_si128_impl::<IMM8>(a) }
720}
721
722/// Implementation detail: converts the immediate argument of the
723/// `_mm_srli_si128` intrinsic into a compile-time constant.
724#[inline]
725#[target_feature(enable = "sse2")]
726#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
727const unsafe fn _mm_srli_si128_impl<const IMM8: i32>(a: __m128i) -> __m128i {
728    const fn mask(shift: i32, i: u32) -> u32 {
729        if (shift as u32) > 15 {
730            i + 16
731        } else {
732            i + (shift as u32)
733        }
734    }
735    let x: i8x16 = simd_shuffle!(
736        a.as_i8x16(),
737        i8x16::ZERO,
738        [
739            mask(IMM8, 0),
740            mask(IMM8, 1),
741            mask(IMM8, 2),
742            mask(IMM8, 3),
743            mask(IMM8, 4),
744            mask(IMM8, 5),
745            mask(IMM8, 6),
746            mask(IMM8, 7),
747            mask(IMM8, 8),
748            mask(IMM8, 9),
749            mask(IMM8, 10),
750            mask(IMM8, 11),
751            mask(IMM8, 12),
752            mask(IMM8, 13),
753            mask(IMM8, 14),
754            mask(IMM8, 15),
755        ],
756    );
757    transmute(x)
758}
759
760/// Shifts packed 16-bit integers in `a` right by `IMM8` while shifting in
761/// zeros.
762///
763/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srli_epi16)
764#[inline]
765#[target_feature(enable = "sse2")]
766#[cfg_attr(test, assert_instr(psrlw, IMM8 = 1))]
767#[rustc_legacy_const_generics(1)]
768#[stable(feature = "simd_x86", since = "1.27.0")]
769#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
770pub const fn _mm_srli_epi16<const IMM8: i32>(a: __m128i) -> __m128i {
771    static_assert_uimm_bits!(IMM8, 8);
772    unsafe {
773        if IMM8 >= 16 {
774            _mm_setzero_si128()
775        } else {
776            transmute(simd_shr(a.as_u16x8(), u16x8::splat(IMM8 as u16)))
777        }
778    }
779}
780
781/// Shifts packed 16-bit integers in `a` right by `count` while shifting in
782/// zeros.
783///
784/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srl_epi16)
785#[inline]
786#[target_feature(enable = "sse2")]
787#[cfg_attr(test, assert_instr(psrlw))]
788#[stable(feature = "simd_x86", since = "1.27.0")]
789pub fn _mm_srl_epi16(a: __m128i, count: __m128i) -> __m128i {
790    unsafe { transmute(psrlw(a.as_i16x8(), count.as_i16x8())) }
791}
792
793/// Shifts packed 32-bit integers in `a` right by `IMM8` while shifting in
794/// zeros.
795///
796/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srli_epi32)
797#[inline]
798#[target_feature(enable = "sse2")]
799#[cfg_attr(test, assert_instr(psrld, IMM8 = 8))]
800#[rustc_legacy_const_generics(1)]
801#[stable(feature = "simd_x86", since = "1.27.0")]
802#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
803pub const fn _mm_srli_epi32<const IMM8: i32>(a: __m128i) -> __m128i {
804    static_assert_uimm_bits!(IMM8, 8);
805    unsafe {
806        if IMM8 >= 32 {
807            _mm_setzero_si128()
808        } else {
809            transmute(simd_shr(a.as_u32x4(), u32x4::splat(IMM8 as u32)))
810        }
811    }
812}
813
814/// Shifts packed 32-bit integers in `a` right by `count` while shifting in
815/// zeros.
816///
817/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srl_epi32)
818#[inline]
819#[target_feature(enable = "sse2")]
820#[cfg_attr(test, assert_instr(psrld))]
821#[stable(feature = "simd_x86", since = "1.27.0")]
822pub fn _mm_srl_epi32(a: __m128i, count: __m128i) -> __m128i {
823    unsafe { transmute(psrld(a.as_i32x4(), count.as_i32x4())) }
824}
825
826/// Shifts packed 64-bit integers in `a` right by `IMM8` while shifting in
827/// zeros.
828///
829/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srli_epi64)
830#[inline]
831#[target_feature(enable = "sse2")]
832#[cfg_attr(test, assert_instr(psrlq, IMM8 = 1))]
833#[rustc_legacy_const_generics(1)]
834#[stable(feature = "simd_x86", since = "1.27.0")]
835#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
836pub const fn _mm_srli_epi64<const IMM8: i32>(a: __m128i) -> __m128i {
837    static_assert_uimm_bits!(IMM8, 8);
838    unsafe {
839        if IMM8 >= 64 {
840            _mm_setzero_si128()
841        } else {
842            transmute(simd_shr(a.as_u64x2(), u64x2::splat(IMM8 as u64)))
843        }
844    }
845}
846
847/// Shifts packed 64-bit integers in `a` right by `count` while shifting in
848/// zeros.
849///
850/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_srl_epi64)
851#[inline]
852#[target_feature(enable = "sse2")]
853#[cfg_attr(test, assert_instr(psrlq))]
854#[stable(feature = "simd_x86", since = "1.27.0")]
855pub fn _mm_srl_epi64(a: __m128i, count: __m128i) -> __m128i {
856    unsafe { transmute(psrlq(a.as_i64x2(), count.as_i64x2())) }
857}
858
859/// Computes the bitwise AND of 128 bits (representing integer data) in `a` and
860/// `b`.
861///
862/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_and_si128)
863#[inline]
864#[target_feature(enable = "sse2")]
865#[cfg_attr(test, assert_instr(andps))]
866#[stable(feature = "simd_x86", since = "1.27.0")]
867#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
868pub const fn _mm_and_si128(a: __m128i, b: __m128i) -> __m128i {
869    unsafe { simd_and(a, b) }
870}
871
872/// Computes the bitwise NOT of 128 bits (representing integer data) in `a` and
873/// then AND with `b`.
874///
875/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_andnot_si128)
876#[inline]
877#[target_feature(enable = "sse2")]
878#[cfg_attr(test, assert_instr(andnps))]
879#[stable(feature = "simd_x86", since = "1.27.0")]
880#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
881pub const fn _mm_andnot_si128(a: __m128i, b: __m128i) -> __m128i {
882    unsafe { simd_and(simd_xor(_mm_set1_epi8(-1), a), b) }
883}
884
885/// Computes the bitwise OR of 128 bits (representing integer data) in `a` and
886/// `b`.
887///
888/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_or_si128)
889#[inline]
890#[target_feature(enable = "sse2")]
891#[cfg_attr(test, assert_instr(orps))]
892#[stable(feature = "simd_x86", since = "1.27.0")]
893#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
894pub const fn _mm_or_si128(a: __m128i, b: __m128i) -> __m128i {
895    unsafe { simd_or(a, b) }
896}
897
898/// Computes the bitwise XOR of 128 bits (representing integer data) in `a` and
899/// `b`.
900///
901/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_xor_si128)
902#[inline]
903#[target_feature(enable = "sse2")]
904#[cfg_attr(test, assert_instr(xorps))]
905#[stable(feature = "simd_x86", since = "1.27.0")]
906#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
907pub const fn _mm_xor_si128(a: __m128i, b: __m128i) -> __m128i {
908    unsafe { simd_xor(a, b) }
909}
910
911/// Compares packed 8-bit integers in `a` and `b` for equality.
912///
913/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpeq_epi8)
914#[inline]
915#[target_feature(enable = "sse2")]
916#[cfg_attr(test, assert_instr(pcmpeqb))]
917#[stable(feature = "simd_x86", since = "1.27.0")]
918#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
919pub const fn _mm_cmpeq_epi8(a: __m128i, b: __m128i) -> __m128i {
920    unsafe { transmute::<i8x16, _>(simd_eq(a.as_i8x16(), b.as_i8x16())) }
921}
922
923/// Compares packed 16-bit integers in `a` and `b` for equality.
924///
925/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpeq_epi16)
926#[inline]
927#[target_feature(enable = "sse2")]
928#[cfg_attr(test, assert_instr(pcmpeqw))]
929#[stable(feature = "simd_x86", since = "1.27.0")]
930#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
931pub const fn _mm_cmpeq_epi16(a: __m128i, b: __m128i) -> __m128i {
932    unsafe { transmute::<i16x8, _>(simd_eq(a.as_i16x8(), b.as_i16x8())) }
933}
934
935/// Compares packed 32-bit integers in `a` and `b` for equality.
936///
937/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpeq_epi32)
938#[inline]
939#[target_feature(enable = "sse2")]
940#[cfg_attr(test, assert_instr(pcmpeqd))]
941#[stable(feature = "simd_x86", since = "1.27.0")]
942#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
943pub const fn _mm_cmpeq_epi32(a: __m128i, b: __m128i) -> __m128i {
944    unsafe { transmute::<i32x4, _>(simd_eq(a.as_i32x4(), b.as_i32x4())) }
945}
946
947/// Compares packed 8-bit integers in `a` and `b` for greater-than.
948///
949/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpgt_epi8)
950#[inline]
951#[target_feature(enable = "sse2")]
952#[cfg_attr(test, assert_instr(pcmpgtb))]
953#[stable(feature = "simd_x86", since = "1.27.0")]
954#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
955pub const fn _mm_cmpgt_epi8(a: __m128i, b: __m128i) -> __m128i {
956    unsafe { transmute::<i8x16, _>(simd_gt(a.as_i8x16(), b.as_i8x16())) }
957}
958
959/// Compares packed 16-bit integers in `a` and `b` for greater-than.
960///
961/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpgt_epi16)
962#[inline]
963#[target_feature(enable = "sse2")]
964#[cfg_attr(test, assert_instr(pcmpgtw))]
965#[stable(feature = "simd_x86", since = "1.27.0")]
966#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
967pub const fn _mm_cmpgt_epi16(a: __m128i, b: __m128i) -> __m128i {
968    unsafe { transmute::<i16x8, _>(simd_gt(a.as_i16x8(), b.as_i16x8())) }
969}
970
971/// Compares packed 32-bit integers in `a` and `b` for greater-than.
972///
973/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpgt_epi32)
974#[inline]
975#[target_feature(enable = "sse2")]
976#[cfg_attr(test, assert_instr(pcmpgtd))]
977#[stable(feature = "simd_x86", since = "1.27.0")]
978#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
979pub const fn _mm_cmpgt_epi32(a: __m128i, b: __m128i) -> __m128i {
980    unsafe { transmute::<i32x4, _>(simd_gt(a.as_i32x4(), b.as_i32x4())) }
981}
982
983/// Compares packed 8-bit integers in `a` and `b` for less-than.
984///
985/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmplt_epi8)
986#[inline]
987#[target_feature(enable = "sse2")]
988#[cfg_attr(test, assert_instr(pcmpgtb))]
989#[stable(feature = "simd_x86", since = "1.27.0")]
990#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
991pub const fn _mm_cmplt_epi8(a: __m128i, b: __m128i) -> __m128i {
992    unsafe { transmute::<i8x16, _>(simd_lt(a.as_i8x16(), b.as_i8x16())) }
993}
994
995/// Compares packed 16-bit integers in `a` and `b` for less-than.
996///
997/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmplt_epi16)
998#[inline]
999#[target_feature(enable = "sse2")]
1000#[cfg_attr(test, assert_instr(pcmpgtw))]
1001#[stable(feature = "simd_x86", since = "1.27.0")]
1002#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1003pub const fn _mm_cmplt_epi16(a: __m128i, b: __m128i) -> __m128i {
1004    unsafe { transmute::<i16x8, _>(simd_lt(a.as_i16x8(), b.as_i16x8())) }
1005}
1006
1007/// Compares packed 32-bit integers in `a` and `b` for less-than.
1008///
1009/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmplt_epi32)
1010#[inline]
1011#[target_feature(enable = "sse2")]
1012#[cfg_attr(test, assert_instr(pcmpgtd))]
1013#[stable(feature = "simd_x86", since = "1.27.0")]
1014#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1015pub const fn _mm_cmplt_epi32(a: __m128i, b: __m128i) -> __m128i {
1016    unsafe { transmute::<i32x4, _>(simd_lt(a.as_i32x4(), b.as_i32x4())) }
1017}
1018
1019/// Converts the lower two packed 32-bit integers in `a` to packed
1020/// double-precision (64-bit) floating-point elements.
1021///
1022/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtepi32_pd)
1023#[inline]
1024#[target_feature(enable = "sse2")]
1025#[cfg_attr(test, assert_instr(cvtdq2pd))]
1026#[stable(feature = "simd_x86", since = "1.27.0")]
1027#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1028pub const fn _mm_cvtepi32_pd(a: __m128i) -> __m128d {
1029    unsafe {
1030        let a = a.as_i32x4();
1031        simd_cast::<i32x2, __m128d>(simd_shuffle!(a, a, [0, 1]))
1032    }
1033}
1034
1035/// Returns `a` with its lower element replaced by `b` after converting it to
1036/// an `f64`.
1037///
1038/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtsi32_sd)
1039#[inline]
1040#[target_feature(enable = "sse2")]
1041#[cfg_attr(test, assert_instr(cvtsi2sd))]
1042#[stable(feature = "simd_x86", since = "1.27.0")]
1043#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1044pub const fn _mm_cvtsi32_sd(a: __m128d, b: i32) -> __m128d {
1045    unsafe { simd_insert!(a, 0, b as f64) }
1046}
1047
1048/// Converts packed 32-bit integers in `a` to packed single-precision (32-bit)
1049/// floating-point elements.
1050///
1051/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtepi32_ps)
1052#[inline]
1053#[target_feature(enable = "sse2")]
1054#[cfg_attr(test, assert_instr(cvtdq2ps))]
1055#[stable(feature = "simd_x86", since = "1.27.0")]
1056#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1057pub const fn _mm_cvtepi32_ps(a: __m128i) -> __m128 {
1058    unsafe { transmute(simd_cast::<_, f32x4>(a.as_i32x4())) }
1059}
1060
1061/// Converts packed single-precision (32-bit) floating-point elements in `a`
1062/// to packed 32-bit integers.
1063///
1064/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtps_epi32)
1065#[inline]
1066#[target_feature(enable = "sse2")]
1067#[cfg_attr(test, assert_instr(cvtps2dq))]
1068#[stable(feature = "simd_x86", since = "1.27.0")]
1069pub fn _mm_cvtps_epi32(a: __m128) -> __m128i {
1070    unsafe { transmute(cvtps2dq(a)) }
1071}
1072
1073/// Returns a vector whose lowest element is `a` and all higher elements are
1074/// `0`.
1075///
1076/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtsi32_si128)
1077#[inline]
1078#[target_feature(enable = "sse2")]
1079#[stable(feature = "simd_x86", since = "1.27.0")]
1080#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1081pub const fn _mm_cvtsi32_si128(a: i32) -> __m128i {
1082    unsafe { transmute(i32x4::new(a, 0, 0, 0)) }
1083}
1084
1085/// Returns the lowest element of `a`.
1086///
1087/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtsi128_si32)
1088#[inline]
1089#[target_feature(enable = "sse2")]
1090#[stable(feature = "simd_x86", since = "1.27.0")]
1091#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1092pub const fn _mm_cvtsi128_si32(a: __m128i) -> i32 {
1093    unsafe { simd_extract!(a.as_i32x4(), 0) }
1094}
1095
1096/// Sets packed 64-bit integers with the supplied values, from highest to
1097/// lowest.
1098///
1099/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set_epi64x)
1100#[inline]
1101#[target_feature(enable = "sse2")]
1102// no particular instruction to test
1103#[stable(feature = "simd_x86", since = "1.27.0")]
1104#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1105pub const fn _mm_set_epi64x(e1: i64, e0: i64) -> __m128i {
1106    unsafe { transmute(i64x2::new(e0, e1)) }
1107}
1108
1109/// Sets packed 32-bit integers with the supplied values.
1110///
1111/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set_epi32)
1112#[inline]
1113#[target_feature(enable = "sse2")]
1114// no particular instruction to test
1115#[stable(feature = "simd_x86", since = "1.27.0")]
1116#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1117pub const fn _mm_set_epi32(e3: i32, e2: i32, e1: i32, e0: i32) -> __m128i {
1118    unsafe { transmute(i32x4::new(e0, e1, e2, e3)) }
1119}
1120
1121/// Sets packed 16-bit integers with the supplied values.
1122///
1123/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set_epi16)
1124#[inline]
1125#[target_feature(enable = "sse2")]
1126// no particular instruction to test
1127#[stable(feature = "simd_x86", since = "1.27.0")]
1128#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1129pub const fn _mm_set_epi16(
1130    e7: i16,
1131    e6: i16,
1132    e5: i16,
1133    e4: i16,
1134    e3: i16,
1135    e2: i16,
1136    e1: i16,
1137    e0: i16,
1138) -> __m128i {
1139    unsafe { transmute(i16x8::new(e0, e1, e2, e3, e4, e5, e6, e7)) }
1140}
1141
1142/// Sets packed 8-bit integers with the supplied values.
1143///
1144/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set_epi8)
1145#[inline]
1146#[target_feature(enable = "sse2")]
1147// no particular instruction to test
1148#[stable(feature = "simd_x86", since = "1.27.0")]
1149#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1150pub const fn _mm_set_epi8(
1151    e15: i8,
1152    e14: i8,
1153    e13: i8,
1154    e12: i8,
1155    e11: i8,
1156    e10: i8,
1157    e9: i8,
1158    e8: i8,
1159    e7: i8,
1160    e6: i8,
1161    e5: i8,
1162    e4: i8,
1163    e3: i8,
1164    e2: i8,
1165    e1: i8,
1166    e0: i8,
1167) -> __m128i {
1168    unsafe {
1169        #[rustfmt::skip]
1170        transmute(i8x16::new(
1171            e0, e1, e2, e3, e4, e5, e6, e7, e8, e9, e10, e11, e12, e13, e14, e15,
1172        ))
1173    }
1174}
1175
1176/// Broadcasts 64-bit integer `a` to all elements.
1177///
1178/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set1_epi64x)
1179#[inline]
1180#[target_feature(enable = "sse2")]
1181// no particular instruction to test
1182#[stable(feature = "simd_x86", since = "1.27.0")]
1183#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1184pub const fn _mm_set1_epi64x(a: i64) -> __m128i {
1185    i64x2::splat(a).as_m128i()
1186}
1187
1188/// Broadcasts 32-bit integer `a` to all elements.
1189///
1190/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set1_epi32)
1191#[inline]
1192#[target_feature(enable = "sse2")]
1193// no particular instruction to test
1194#[stable(feature = "simd_x86", since = "1.27.0")]
1195#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1196pub const fn _mm_set1_epi32(a: i32) -> __m128i {
1197    i32x4::splat(a).as_m128i()
1198}
1199
1200/// Broadcasts 16-bit integer `a` to all elements.
1201///
1202/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set1_epi16)
1203#[inline]
1204#[target_feature(enable = "sse2")]
1205// no particular instruction to test
1206#[stable(feature = "simd_x86", since = "1.27.0")]
1207#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1208pub const fn _mm_set1_epi16(a: i16) -> __m128i {
1209    i16x8::splat(a).as_m128i()
1210}
1211
1212/// Broadcasts 8-bit integer `a` to all elements.
1213///
1214/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set1_epi8)
1215#[inline]
1216#[target_feature(enable = "sse2")]
1217// no particular instruction to test
1218#[stable(feature = "simd_x86", since = "1.27.0")]
1219#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1220pub const fn _mm_set1_epi8(a: i8) -> __m128i {
1221    i8x16::splat(a).as_m128i()
1222}
1223
1224/// Sets packed 32-bit integers with the supplied values in reverse order.
1225///
1226/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_setr_epi32)
1227#[inline]
1228#[target_feature(enable = "sse2")]
1229// no particular instruction to test
1230#[stable(feature = "simd_x86", since = "1.27.0")]
1231#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1232pub const fn _mm_setr_epi32(e3: i32, e2: i32, e1: i32, e0: i32) -> __m128i {
1233    _mm_set_epi32(e0, e1, e2, e3)
1234}
1235
1236/// Sets packed 16-bit integers with the supplied values in reverse order.
1237///
1238/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_setr_epi16)
1239#[inline]
1240#[target_feature(enable = "sse2")]
1241// no particular instruction to test
1242#[stable(feature = "simd_x86", since = "1.27.0")]
1243#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1244pub const fn _mm_setr_epi16(
1245    e7: i16,
1246    e6: i16,
1247    e5: i16,
1248    e4: i16,
1249    e3: i16,
1250    e2: i16,
1251    e1: i16,
1252    e0: i16,
1253) -> __m128i {
1254    _mm_set_epi16(e0, e1, e2, e3, e4, e5, e6, e7)
1255}
1256
1257/// Sets packed 8-bit integers with the supplied values in reverse order.
1258///
1259/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_setr_epi8)
1260#[inline]
1261#[target_feature(enable = "sse2")]
1262// no particular instruction to test
1263#[stable(feature = "simd_x86", since = "1.27.0")]
1264#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1265pub const fn _mm_setr_epi8(
1266    e15: i8,
1267    e14: i8,
1268    e13: i8,
1269    e12: i8,
1270    e11: i8,
1271    e10: i8,
1272    e9: i8,
1273    e8: i8,
1274    e7: i8,
1275    e6: i8,
1276    e5: i8,
1277    e4: i8,
1278    e3: i8,
1279    e2: i8,
1280    e1: i8,
1281    e0: i8,
1282) -> __m128i {
1283    #[rustfmt::skip]
1284    _mm_set_epi8(
1285        e0, e1, e2, e3, e4, e5, e6, e7, e8, e9, e10, e11, e12, e13, e14, e15,
1286    )
1287}
1288
1289/// Returns a vector with all elements set to zero.
1290///
1291/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_setzero_si128)
1292#[inline]
1293#[target_feature(enable = "sse2")]
1294#[cfg_attr(test, assert_instr(xorps))]
1295#[stable(feature = "simd_x86", since = "1.27.0")]
1296#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1297pub const fn _mm_setzero_si128() -> __m128i {
1298    const { unsafe { mem::zeroed() } }
1299}
1300
1301/// Loads 64-bit integer from memory into first element of returned vector.
1302///
1303/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadl_epi64)
1304#[inline]
1305#[target_feature(enable = "sse2")]
1306#[stable(feature = "simd_x86", since = "1.27.0")]
1307#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1308pub const unsafe fn _mm_loadl_epi64(mem_addr: *const __m128i) -> __m128i {
1309    _mm_set_epi64x(0, ptr::read_unaligned(mem_addr as *const i64))
1310}
1311
1312/// Loads 128-bits of integer data from memory into a new vector.
1313///
1314/// `mem_addr` must be aligned on a 16-byte boundary.
1315///
1316/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_load_si128)
1317#[inline]
1318#[target_feature(enable = "sse2")]
1319#[cfg_attr(
1320    all(test, not(all(target_arch = "x86", target_env = "msvc"))),
1321    assert_instr(movaps)
1322)]
1323#[stable(feature = "simd_x86", since = "1.27.0")]
1324#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1325pub const unsafe fn _mm_load_si128(mem_addr: *const __m128i) -> __m128i {
1326    *mem_addr
1327}
1328
1329/// Loads 128-bits of integer data from memory into a new vector.
1330///
1331/// `mem_addr` does not need to be aligned on any particular boundary.
1332///
1333/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadu_si128)
1334#[inline]
1335#[target_feature(enable = "sse2")]
1336#[cfg_attr(test, assert_instr(movups))]
1337#[stable(feature = "simd_x86", since = "1.27.0")]
1338#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1339pub const unsafe fn _mm_loadu_si128(mem_addr: *const __m128i) -> __m128i {
1340    let mut dst: __m128i = _mm_undefined_si128();
1341    ptr::copy_nonoverlapping(
1342        mem_addr as *const u8,
1343        ptr::addr_of_mut!(dst) as *mut u8,
1344        mem::size_of::<__m128i>(),
1345    );
1346    dst
1347}
1348
1349/// Conditionally store 8-bit integer elements from `a` into memory using
1350/// `mask` flagged as non-temporal (unlikely to be used again soon).
1351///
1352/// Elements are not stored when the highest bit is not set in the
1353/// corresponding element.
1354///
1355/// `mem_addr` should correspond to a 128-bit memory location and does not need
1356/// to be aligned on any particular boundary.
1357///
1358/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_maskmoveu_si128)
1359///
1360/// # Safety of non-temporal stores
1361///
1362/// After using this intrinsic, but before any other access to the memory that this intrinsic
1363/// mutates, a call to [`_mm_sfence`] must be performed by the thread that used the intrinsic. In
1364/// particular, functions that call this intrinsic should generally call `_mm_sfence` before they
1365/// return.
1366///
1367/// See [`_mm_sfence`] for details.
1368#[inline]
1369#[target_feature(enable = "sse2")]
1370#[cfg_attr(test, assert_instr(maskmovdqu))]
1371#[stable(feature = "simd_x86", since = "1.27.0")]
1372pub unsafe fn _mm_maskmoveu_si128(a: __m128i, mask: __m128i, mem_addr: *mut i8) {
1373    maskmovdqu(a.as_i8x16(), mask.as_i8x16(), mem_addr)
1374}
1375
1376/// Stores 128-bits of integer data from `a` into memory.
1377///
1378/// `mem_addr` must be aligned on a 16-byte boundary.
1379///
1380/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_store_si128)
1381#[inline]
1382#[target_feature(enable = "sse2")]
1383#[cfg_attr(
1384    all(test, not(all(target_arch = "x86", target_env = "msvc"))),
1385    assert_instr(movaps)
1386)]
1387#[stable(feature = "simd_x86", since = "1.27.0")]
1388#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1389pub const unsafe fn _mm_store_si128(mem_addr: *mut __m128i, a: __m128i) {
1390    *mem_addr = a;
1391}
1392
1393/// Stores 128-bits of integer data from `a` into memory.
1394///
1395/// `mem_addr` does not need to be aligned on any particular boundary.
1396///
1397/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storeu_si128)
1398#[inline]
1399#[target_feature(enable = "sse2")]
1400#[cfg_attr(test, assert_instr(movups))] // FIXME movdqu expected
1401#[stable(feature = "simd_x86", since = "1.27.0")]
1402#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1403pub const unsafe fn _mm_storeu_si128(mem_addr: *mut __m128i, a: __m128i) {
1404    mem_addr.write_unaligned(a);
1405}
1406
1407/// Stores the lower 64-bit integer `a` to a memory location.
1408///
1409/// `mem_addr` does not need to be aligned on any particular boundary.
1410///
1411/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storel_epi64)
1412#[inline]
1413#[target_feature(enable = "sse2")]
1414#[stable(feature = "simd_x86", since = "1.27.0")]
1415#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1416pub const unsafe fn _mm_storel_epi64(mem_addr: *mut __m128i, a: __m128i) {
1417    ptr::copy_nonoverlapping(ptr::addr_of!(a) as *const u8, mem_addr as *mut u8, 8);
1418}
1419
1420/// Stores a 128-bit integer vector to a 128-bit aligned memory location.
1421/// To minimize caching, the data is flagged as non-temporal (unlikely to be
1422/// used again soon).
1423///
1424/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_stream_si128)
1425///
1426/// # Safety of non-temporal stores
1427///
1428/// After using this intrinsic, but before any other access to the memory that this intrinsic
1429/// mutates, a call to [`_mm_sfence`] must be performed by the thread that used the intrinsic. In
1430/// particular, functions that call this intrinsic should generally call `_mm_sfence` before they
1431/// return.
1432///
1433/// See [`_mm_sfence`] for details.
1434#[inline]
1435#[target_feature(enable = "sse2")]
1436#[cfg_attr(test, assert_instr(movntdq))]
1437#[stable(feature = "simd_x86", since = "1.27.0")]
1438pub unsafe fn _mm_stream_si128(mem_addr: *mut __m128i, a: __m128i) {
1439    // see #1541, we should use inline asm to be sure, because LangRef isn't clear enough
1440    crate::arch::asm!(
1441        vps!("movntdq",  ",{a}"),
1442        p = in(reg) mem_addr,
1443        a = in(xmm_reg) a,
1444        options(nostack, preserves_flags),
1445    );
1446}
1447
1448/// Stores a 32-bit integer value in a 4-byte aligned memory location.
1449/// To minimize caching, the data is flagged as non-temporal (unlikely to be
1450/// used again soon).
1451///
1452/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_stream_si32)
1453///
1454/// # Safety of non-temporal stores
1455///
1456/// After using this intrinsic, but before any other access to the memory that this intrinsic
1457/// mutates, a call to [`_mm_sfence`] must be performed by the thread that used the intrinsic. In
1458/// particular, functions that call this intrinsic should generally call `_mm_sfence` before they
1459/// return.
1460///
1461/// See [`_mm_sfence`] for details.
1462#[inline]
1463#[target_feature(enable = "sse2")]
1464#[cfg_attr(test, assert_instr(movnti))]
1465#[stable(feature = "simd_x86", since = "1.27.0")]
1466pub unsafe fn _mm_stream_si32(mem_addr: *mut i32, a: i32) {
1467    // see #1541, we should use inline asm to be sure, because LangRef isn't clear enough
1468    crate::arch::asm!(
1469        vps!("movnti", ",{a:e}"), // `:e` for 32bit value
1470        p = in(reg) mem_addr,
1471        a = in(reg) a,
1472        options(nostack, preserves_flags),
1473    );
1474}
1475
1476/// Returns a vector where the low element is extracted from `a` and its upper
1477/// element is zero.
1478///
1479/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_move_epi64)
1480#[inline]
1481#[target_feature(enable = "sse2")]
1482// FIXME movd on msvc, movd on i686
1483#[cfg_attr(all(test, target_arch = "x86_64"), assert_instr(movq))]
1484#[stable(feature = "simd_x86", since = "1.27.0")]
1485#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1486pub const fn _mm_move_epi64(a: __m128i) -> __m128i {
1487    unsafe {
1488        let r: i64x2 = simd_shuffle!(a.as_i64x2(), i64x2::ZERO, [0, 2]);
1489        transmute(r)
1490    }
1491}
1492
1493/// Converts packed 16-bit integers from `a` and `b` to packed 8-bit integers
1494/// using signed saturation.
1495///
1496/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_packs_epi16)
1497#[inline]
1498#[target_feature(enable = "sse2")]
1499#[cfg_attr(test, assert_instr(packsswb))]
1500#[stable(feature = "simd_x86", since = "1.27.0")]
1501pub fn _mm_packs_epi16(a: __m128i, b: __m128i) -> __m128i {
1502    unsafe { transmute(packsswb(a.as_i16x8(), b.as_i16x8())) }
1503}
1504
1505/// Converts packed 32-bit integers from `a` and `b` to packed 16-bit integers
1506/// using signed saturation.
1507///
1508/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_packs_epi32)
1509#[inline]
1510#[target_feature(enable = "sse2")]
1511#[cfg_attr(test, assert_instr(packssdw))]
1512#[stable(feature = "simd_x86", since = "1.27.0")]
1513pub fn _mm_packs_epi32(a: __m128i, b: __m128i) -> __m128i {
1514    unsafe { transmute(packssdw(a.as_i32x4(), b.as_i32x4())) }
1515}
1516
1517/// Converts packed 16-bit integers from `a` and `b` to packed 8-bit integers
1518/// using unsigned saturation.
1519///
1520/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_packus_epi16)
1521#[inline]
1522#[target_feature(enable = "sse2")]
1523#[cfg_attr(test, assert_instr(packuswb))]
1524#[stable(feature = "simd_x86", since = "1.27.0")]
1525pub fn _mm_packus_epi16(a: __m128i, b: __m128i) -> __m128i {
1526    unsafe { transmute(packuswb(a.as_i16x8(), b.as_i16x8())) }
1527}
1528
1529/// Returns the `imm8` element of `a`.
1530///
1531/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_extract_epi16)
1532#[inline]
1533#[target_feature(enable = "sse2")]
1534#[cfg_attr(test, assert_instr(pextrw, IMM8 = 7))]
1535#[rustc_legacy_const_generics(1)]
1536#[stable(feature = "simd_x86", since = "1.27.0")]
1537#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1538pub const fn _mm_extract_epi16<const IMM8: i32>(a: __m128i) -> i32 {
1539    static_assert_uimm_bits!(IMM8, 3);
1540    unsafe { simd_extract!(a.as_u16x8(), IMM8 as u32, u16) as i32 }
1541}
1542
1543/// Returns a new vector where the `imm8` element of `a` is replaced with `i`.
1544///
1545/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_insert_epi16)
1546#[inline]
1547#[target_feature(enable = "sse2")]
1548#[cfg_attr(test, assert_instr(pinsrw, IMM8 = 7))]
1549#[rustc_legacy_const_generics(2)]
1550#[stable(feature = "simd_x86", since = "1.27.0")]
1551#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1552pub const fn _mm_insert_epi16<const IMM8: i32>(a: __m128i, i: i32) -> __m128i {
1553    static_assert_uimm_bits!(IMM8, 3);
1554    unsafe { transmute(simd_insert!(a.as_i16x8(), IMM8 as u32, i as i16)) }
1555}
1556
1557/// Returns a mask of the most significant bit of each element in `a`.
1558///
1559/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_movemask_epi8)
1560#[inline]
1561#[target_feature(enable = "sse2")]
1562#[cfg_attr(test, assert_instr(pmovmskb))]
1563#[stable(feature = "simd_x86", since = "1.27.0")]
1564#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1565pub const fn _mm_movemask_epi8(a: __m128i) -> i32 {
1566    unsafe {
1567        let z = i8x16::ZERO;
1568        let m: i8x16 = simd_lt(a.as_i8x16(), z);
1569        simd_bitmask::<_, u16>(m) as u32 as i32
1570    }
1571}
1572
1573/// Shuffles 32-bit integers in `a` using the control in `IMM8`.
1574///
1575/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_shuffle_epi32)
1576#[inline]
1577#[target_feature(enable = "sse2")]
1578#[cfg_attr(test, assert_instr(pshufd, IMM8 = 9))]
1579#[rustc_legacy_const_generics(1)]
1580#[stable(feature = "simd_x86", since = "1.27.0")]
1581#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1582pub const fn _mm_shuffle_epi32<const IMM8: i32>(a: __m128i) -> __m128i {
1583    static_assert_uimm_bits!(IMM8, 8);
1584    unsafe {
1585        let a = a.as_i32x4();
1586        let x: i32x4 = simd_shuffle!(
1587            a,
1588            a,
1589            [
1590                IMM8 as u32 & 0b11,
1591                (IMM8 as u32 >> 2) & 0b11,
1592                (IMM8 as u32 >> 4) & 0b11,
1593                (IMM8 as u32 >> 6) & 0b11,
1594            ],
1595        );
1596        transmute(x)
1597    }
1598}
1599
1600/// Shuffles 16-bit integers in the high 64 bits of `a` using the control in
1601/// `IMM8`.
1602///
1603/// Put the results in the high 64 bits of the returned vector, with the low 64
1604/// bits being copied from `a`.
1605///
1606/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_shufflehi_epi16)
1607#[inline]
1608#[target_feature(enable = "sse2")]
1609#[cfg_attr(test, assert_instr(pshufhw, IMM8 = 9))]
1610#[rustc_legacy_const_generics(1)]
1611#[stable(feature = "simd_x86", since = "1.27.0")]
1612#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1613pub const fn _mm_shufflehi_epi16<const IMM8: i32>(a: __m128i) -> __m128i {
1614    static_assert_uimm_bits!(IMM8, 8);
1615    unsafe {
1616        let a = a.as_i16x8();
1617        let x: i16x8 = simd_shuffle!(
1618            a,
1619            a,
1620            [
1621                0,
1622                1,
1623                2,
1624                3,
1625                (IMM8 as u32 & 0b11) + 4,
1626                ((IMM8 as u32 >> 2) & 0b11) + 4,
1627                ((IMM8 as u32 >> 4) & 0b11) + 4,
1628                ((IMM8 as u32 >> 6) & 0b11) + 4,
1629            ],
1630        );
1631        transmute(x)
1632    }
1633}
1634
1635/// Shuffles 16-bit integers in the low 64 bits of `a` using the control in
1636/// `IMM8`.
1637///
1638/// Put the results in the low 64 bits of the returned vector, with the high 64
1639/// bits being copied from `a`.
1640///
1641/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_shufflelo_epi16)
1642#[inline]
1643#[target_feature(enable = "sse2")]
1644#[cfg_attr(test, assert_instr(pshuflw, IMM8 = 9))]
1645#[rustc_legacy_const_generics(1)]
1646#[stable(feature = "simd_x86", since = "1.27.0")]
1647#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1648pub const fn _mm_shufflelo_epi16<const IMM8: i32>(a: __m128i) -> __m128i {
1649    static_assert_uimm_bits!(IMM8, 8);
1650    unsafe {
1651        let a = a.as_i16x8();
1652        let x: i16x8 = simd_shuffle!(
1653            a,
1654            a,
1655            [
1656                IMM8 as u32 & 0b11,
1657                (IMM8 as u32 >> 2) & 0b11,
1658                (IMM8 as u32 >> 4) & 0b11,
1659                (IMM8 as u32 >> 6) & 0b11,
1660                4,
1661                5,
1662                6,
1663                7,
1664            ],
1665        );
1666        transmute(x)
1667    }
1668}
1669
1670/// Unpacks and interleave 8-bit integers from the high half of `a` and `b`.
1671///
1672/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpackhi_epi8)
1673#[inline]
1674#[target_feature(enable = "sse2")]
1675#[cfg_attr(test, assert_instr(punpckhbw))]
1676#[stable(feature = "simd_x86", since = "1.27.0")]
1677#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1678pub const fn _mm_unpackhi_epi8(a: __m128i, b: __m128i) -> __m128i {
1679    unsafe {
1680        transmute::<i8x16, _>(simd_shuffle!(
1681            a.as_i8x16(),
1682            b.as_i8x16(),
1683            [8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31],
1684        ))
1685    }
1686}
1687
1688/// Unpacks and interleave 16-bit integers from the high half of `a` and `b`.
1689///
1690/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpackhi_epi16)
1691#[inline]
1692#[target_feature(enable = "sse2")]
1693#[cfg_attr(test, assert_instr(punpckhwd))]
1694#[stable(feature = "simd_x86", since = "1.27.0")]
1695#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1696pub const fn _mm_unpackhi_epi16(a: __m128i, b: __m128i) -> __m128i {
1697    unsafe {
1698        let x = simd_shuffle!(a.as_i16x8(), b.as_i16x8(), [4, 12, 5, 13, 6, 14, 7, 15]);
1699        transmute::<i16x8, _>(x)
1700    }
1701}
1702
1703/// Unpacks and interleave 32-bit integers from the high half of `a` and `b`.
1704///
1705/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpackhi_epi32)
1706#[inline]
1707#[target_feature(enable = "sse2")]
1708#[cfg_attr(test, assert_instr(unpckhps))]
1709#[stable(feature = "simd_x86", since = "1.27.0")]
1710#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1711pub const fn _mm_unpackhi_epi32(a: __m128i, b: __m128i) -> __m128i {
1712    unsafe { transmute::<i32x4, _>(simd_shuffle!(a.as_i32x4(), b.as_i32x4(), [2, 6, 3, 7])) }
1713}
1714
1715/// Unpacks and interleave 64-bit integers from the high half of `a` and `b`.
1716///
1717/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpackhi_epi64)
1718#[inline]
1719#[target_feature(enable = "sse2")]
1720#[cfg_attr(test, assert_instr(unpckhpd))]
1721#[stable(feature = "simd_x86", since = "1.27.0")]
1722#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1723pub const fn _mm_unpackhi_epi64(a: __m128i, b: __m128i) -> __m128i {
1724    unsafe { transmute::<i64x2, _>(simd_shuffle!(a.as_i64x2(), b.as_i64x2(), [1, 3])) }
1725}
1726
1727/// Unpacks and interleave 8-bit integers from the low half of `a` and `b`.
1728///
1729/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpacklo_epi8)
1730#[inline]
1731#[target_feature(enable = "sse2")]
1732#[cfg_attr(test, assert_instr(punpcklbw))]
1733#[stable(feature = "simd_x86", since = "1.27.0")]
1734#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1735pub const fn _mm_unpacklo_epi8(a: __m128i, b: __m128i) -> __m128i {
1736    unsafe {
1737        transmute::<i8x16, _>(simd_shuffle!(
1738            a.as_i8x16(),
1739            b.as_i8x16(),
1740            [0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23],
1741        ))
1742    }
1743}
1744
1745/// Unpacks and interleave 16-bit integers from the low half of `a` and `b`.
1746///
1747/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpacklo_epi16)
1748#[inline]
1749#[target_feature(enable = "sse2")]
1750#[cfg_attr(test, assert_instr(punpcklwd))]
1751#[stable(feature = "simd_x86", since = "1.27.0")]
1752#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1753pub const fn _mm_unpacklo_epi16(a: __m128i, b: __m128i) -> __m128i {
1754    unsafe {
1755        let x = simd_shuffle!(a.as_i16x8(), b.as_i16x8(), [0, 8, 1, 9, 2, 10, 3, 11]);
1756        transmute::<i16x8, _>(x)
1757    }
1758}
1759
1760/// Unpacks and interleave 32-bit integers from the low half of `a` and `b`.
1761///
1762/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpacklo_epi32)
1763#[inline]
1764#[target_feature(enable = "sse2")]
1765#[cfg_attr(test, assert_instr(unpcklps))]
1766#[stable(feature = "simd_x86", since = "1.27.0")]
1767#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1768pub const fn _mm_unpacklo_epi32(a: __m128i, b: __m128i) -> __m128i {
1769    unsafe { transmute::<i32x4, _>(simd_shuffle!(a.as_i32x4(), b.as_i32x4(), [0, 4, 1, 5])) }
1770}
1771
1772/// Unpacks and interleave 64-bit integers from the low half of `a` and `b`.
1773///
1774/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpacklo_epi64)
1775#[inline]
1776#[target_feature(enable = "sse2")]
1777#[cfg_attr(test, assert_instr(movlhps))]
1778#[stable(feature = "simd_x86", since = "1.27.0")]
1779#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1780pub const fn _mm_unpacklo_epi64(a: __m128i, b: __m128i) -> __m128i {
1781    unsafe { transmute::<i64x2, _>(simd_shuffle!(a.as_i64x2(), b.as_i64x2(), [0, 2])) }
1782}
1783
1784/// Returns a new vector with the low element of `a` replaced by the sum of the
1785/// low elements of `a` and `b`.
1786///
1787/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_add_sd)
1788#[inline]
1789#[target_feature(enable = "sse2")]
1790#[cfg_attr(test, assert_instr(addsd))]
1791#[stable(feature = "simd_x86", since = "1.27.0")]
1792#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1793pub const fn _mm_add_sd(a: __m128d, b: __m128d) -> __m128d {
1794    unsafe { simd_insert!(a, 0, _mm_cvtsd_f64(a) + _mm_cvtsd_f64(b)) }
1795}
1796
1797/// Adds packed double-precision (64-bit) floating-point elements in `a` and
1798/// `b`.
1799///
1800/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_add_pd)
1801#[inline]
1802#[target_feature(enable = "sse2")]
1803#[cfg_attr(test, assert_instr(addpd))]
1804#[stable(feature = "simd_x86", since = "1.27.0")]
1805#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1806pub const fn _mm_add_pd(a: __m128d, b: __m128d) -> __m128d {
1807    unsafe { simd_add(a, b) }
1808}
1809
1810/// Returns a new vector with the low element of `a` replaced by the result of
1811/// diving the lower element of `a` by the lower element of `b`.
1812///
1813/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_div_sd)
1814#[inline]
1815#[target_feature(enable = "sse2")]
1816#[cfg_attr(test, assert_instr(divsd))]
1817#[stable(feature = "simd_x86", since = "1.27.0")]
1818#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1819pub const fn _mm_div_sd(a: __m128d, b: __m128d) -> __m128d {
1820    unsafe { simd_insert!(a, 0, _mm_cvtsd_f64(a) / _mm_cvtsd_f64(b)) }
1821}
1822
1823/// Divide packed double-precision (64-bit) floating-point elements in `a` by
1824/// packed elements in `b`.
1825///
1826/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_div_pd)
1827#[inline]
1828#[target_feature(enable = "sse2")]
1829#[cfg_attr(test, assert_instr(divpd))]
1830#[stable(feature = "simd_x86", since = "1.27.0")]
1831#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1832pub const fn _mm_div_pd(a: __m128d, b: __m128d) -> __m128d {
1833    unsafe { simd_div(a, b) }
1834}
1835
1836/// Returns a new vector with the low element of `a` replaced by the maximum
1837/// of the lower elements of `a` and `b`.
1838///
1839/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_max_sd)
1840#[inline]
1841#[target_feature(enable = "sse2")]
1842#[cfg_attr(test, assert_instr(maxsd))]
1843#[stable(feature = "simd_x86", since = "1.27.0")]
1844pub fn _mm_max_sd(a: __m128d, b: __m128d) -> __m128d {
1845    unsafe { maxsd(a, b) }
1846}
1847
1848/// Returns a new vector with the maximum values from corresponding elements in
1849/// `a` and `b`.
1850///
1851/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_max_pd)
1852#[inline]
1853#[target_feature(enable = "sse2")]
1854#[cfg_attr(test, assert_instr(maxpd))]
1855#[stable(feature = "simd_x86", since = "1.27.0")]
1856pub fn _mm_max_pd(a: __m128d, b: __m128d) -> __m128d {
1857    unsafe { maxpd(a, b) }
1858}
1859
1860/// Returns a new vector with the low element of `a` replaced by the minimum
1861/// of the lower elements of `a` and `b`.
1862///
1863/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_min_sd)
1864#[inline]
1865#[target_feature(enable = "sse2")]
1866#[cfg_attr(test, assert_instr(minsd))]
1867#[stable(feature = "simd_x86", since = "1.27.0")]
1868pub fn _mm_min_sd(a: __m128d, b: __m128d) -> __m128d {
1869    unsafe { minsd(a, b) }
1870}
1871
1872/// Returns a new vector with the minimum values from corresponding elements in
1873/// `a` and `b`.
1874///
1875/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_min_pd)
1876#[inline]
1877#[target_feature(enable = "sse2")]
1878#[cfg_attr(test, assert_instr(minpd))]
1879#[stable(feature = "simd_x86", since = "1.27.0")]
1880pub fn _mm_min_pd(a: __m128d, b: __m128d) -> __m128d {
1881    unsafe { minpd(a, b) }
1882}
1883
1884/// Returns a new vector with the low element of `a` replaced by multiplying the
1885/// low elements of `a` and `b`.
1886///
1887/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_mul_sd)
1888#[inline]
1889#[target_feature(enable = "sse2")]
1890#[cfg_attr(test, assert_instr(mulsd))]
1891#[stable(feature = "simd_x86", since = "1.27.0")]
1892#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1893pub const fn _mm_mul_sd(a: __m128d, b: __m128d) -> __m128d {
1894    unsafe { simd_insert!(a, 0, _mm_cvtsd_f64(a) * _mm_cvtsd_f64(b)) }
1895}
1896
1897/// Multiplies packed double-precision (64-bit) floating-point elements in `a`
1898/// and `b`.
1899///
1900/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_mul_pd)
1901#[inline]
1902#[target_feature(enable = "sse2")]
1903#[cfg_attr(test, assert_instr(mulpd))]
1904#[stable(feature = "simd_x86", since = "1.27.0")]
1905#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1906pub const fn _mm_mul_pd(a: __m128d, b: __m128d) -> __m128d {
1907    unsafe { simd_mul(a, b) }
1908}
1909
1910/// Returns a new vector with the low element of `a` replaced by the square
1911/// root of the lower element `b`.
1912///
1913/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sqrt_sd)
1914#[inline]
1915#[target_feature(enable = "sse2")]
1916#[cfg_attr(test, assert_instr(sqrtsd))]
1917#[stable(feature = "simd_x86", since = "1.27.0")]
1918pub fn _mm_sqrt_sd(a: __m128d, b: __m128d) -> __m128d {
1919    unsafe { simd_insert!(a, 0, sqrtf64(_mm_cvtsd_f64(b))) }
1920}
1921
1922/// Returns a new vector with the square root of each of the values in `a`.
1923///
1924/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sqrt_pd)
1925#[inline]
1926#[target_feature(enable = "sse2")]
1927#[cfg_attr(test, assert_instr(sqrtpd))]
1928#[stable(feature = "simd_x86", since = "1.27.0")]
1929pub fn _mm_sqrt_pd(a: __m128d) -> __m128d {
1930    unsafe { simd_fsqrt(a) }
1931}
1932
1933/// Returns a new vector with the low element of `a` replaced by subtracting the
1934/// low element by `b` from the low element of `a`.
1935///
1936/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sub_sd)
1937#[inline]
1938#[target_feature(enable = "sse2")]
1939#[cfg_attr(test, assert_instr(subsd))]
1940#[stable(feature = "simd_x86", since = "1.27.0")]
1941#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1942pub const fn _mm_sub_sd(a: __m128d, b: __m128d) -> __m128d {
1943    unsafe { simd_insert!(a, 0, _mm_cvtsd_f64(a) - _mm_cvtsd_f64(b)) }
1944}
1945
1946/// Subtract packed double-precision (64-bit) floating-point elements in `b`
1947/// from `a`.
1948///
1949/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_sub_pd)
1950#[inline]
1951#[target_feature(enable = "sse2")]
1952#[cfg_attr(test, assert_instr(subpd))]
1953#[stable(feature = "simd_x86", since = "1.27.0")]
1954#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1955pub const fn _mm_sub_pd(a: __m128d, b: __m128d) -> __m128d {
1956    unsafe { simd_sub(a, b) }
1957}
1958
1959/// Computes the bitwise AND of packed double-precision (64-bit) floating-point
1960/// elements in `a` and `b`.
1961///
1962/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_and_pd)
1963#[inline]
1964#[target_feature(enable = "sse2")]
1965#[cfg_attr(test, assert_instr(andps))]
1966#[stable(feature = "simd_x86", since = "1.27.0")]
1967#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1968pub const fn _mm_and_pd(a: __m128d, b: __m128d) -> __m128d {
1969    unsafe {
1970        let a: __m128i = transmute(a);
1971        let b: __m128i = transmute(b);
1972        transmute(_mm_and_si128(a, b))
1973    }
1974}
1975
1976/// Computes the bitwise NOT of `a` and then AND with `b`.
1977///
1978/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_andnot_pd)
1979#[inline]
1980#[target_feature(enable = "sse2")]
1981#[cfg_attr(test, assert_instr(andnps))]
1982#[stable(feature = "simd_x86", since = "1.27.0")]
1983#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
1984pub const fn _mm_andnot_pd(a: __m128d, b: __m128d) -> __m128d {
1985    unsafe {
1986        let a: __m128i = transmute(a);
1987        let b: __m128i = transmute(b);
1988        transmute(_mm_andnot_si128(a, b))
1989    }
1990}
1991
1992/// Computes the bitwise OR of `a` and `b`.
1993///
1994/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_or_pd)
1995#[inline]
1996#[target_feature(enable = "sse2")]
1997#[cfg_attr(test, assert_instr(orps))]
1998#[stable(feature = "simd_x86", since = "1.27.0")]
1999#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2000pub const fn _mm_or_pd(a: __m128d, b: __m128d) -> __m128d {
2001    unsafe {
2002        let a: __m128i = transmute(a);
2003        let b: __m128i = transmute(b);
2004        transmute(_mm_or_si128(a, b))
2005    }
2006}
2007
2008/// Computes the bitwise XOR of `a` and `b`.
2009///
2010/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_xor_pd)
2011#[inline]
2012#[target_feature(enable = "sse2")]
2013#[cfg_attr(test, assert_instr(xorps))]
2014#[stable(feature = "simd_x86", since = "1.27.0")]
2015#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2016pub const fn _mm_xor_pd(a: __m128d, b: __m128d) -> __m128d {
2017    unsafe {
2018        let a: __m128i = transmute(a);
2019        let b: __m128i = transmute(b);
2020        transmute(_mm_xor_si128(a, b))
2021    }
2022}
2023
2024/// Returns a new vector with the low element of `a` replaced by the equality
2025/// comparison of the lower elements of `a` and `b`.
2026///
2027/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpeq_sd)
2028#[inline]
2029#[target_feature(enable = "sse2")]
2030#[cfg_attr(test, assert_instr(cmpeqsd))]
2031#[stable(feature = "simd_x86", since = "1.27.0")]
2032pub fn _mm_cmpeq_sd(a: __m128d, b: __m128d) -> __m128d {
2033    unsafe { cmpsd(a, b, 0) }
2034}
2035
2036/// Returns a new vector with the low element of `a` replaced by the less-than
2037/// comparison of the lower elements of `a` and `b`.
2038///
2039/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmplt_sd)
2040#[inline]
2041#[target_feature(enable = "sse2")]
2042#[cfg_attr(test, assert_instr(cmpltsd))]
2043#[stable(feature = "simd_x86", since = "1.27.0")]
2044pub fn _mm_cmplt_sd(a: __m128d, b: __m128d) -> __m128d {
2045    unsafe { cmpsd(a, b, 1) }
2046}
2047
2048/// Returns a new vector with the low element of `a` replaced by the
2049/// less-than-or-equal comparison of the lower elements of `a` and `b`.
2050///
2051/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmple_sd)
2052#[inline]
2053#[target_feature(enable = "sse2")]
2054#[cfg_attr(test, assert_instr(cmplesd))]
2055#[stable(feature = "simd_x86", since = "1.27.0")]
2056pub fn _mm_cmple_sd(a: __m128d, b: __m128d) -> __m128d {
2057    unsafe { cmpsd(a, b, 2) }
2058}
2059
2060/// Returns a new vector with the low element of `a` replaced by the
2061/// greater-than comparison of the lower elements of `a` and `b`.
2062///
2063/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpgt_sd)
2064#[inline]
2065#[target_feature(enable = "sse2")]
2066#[cfg_attr(test, assert_instr(cmpltsd))]
2067#[stable(feature = "simd_x86", since = "1.27.0")]
2068pub fn _mm_cmpgt_sd(a: __m128d, b: __m128d) -> __m128d {
2069    unsafe { simd_insert!(_mm_cmplt_sd(b, a), 1, simd_extract!(a, 1, f64)) }
2070}
2071
2072/// Returns a new vector with the low element of `a` replaced by the
2073/// greater-than-or-equal comparison of the lower elements of `a` and `b`.
2074///
2075/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpge_sd)
2076#[inline]
2077#[target_feature(enable = "sse2")]
2078#[cfg_attr(test, assert_instr(cmplesd))]
2079#[stable(feature = "simd_x86", since = "1.27.0")]
2080pub fn _mm_cmpge_sd(a: __m128d, b: __m128d) -> __m128d {
2081    unsafe { simd_insert!(_mm_cmple_sd(b, a), 1, simd_extract!(a, 1, f64)) }
2082}
2083
2084/// Returns a new vector with the low element of `a` replaced by the result
2085/// of comparing both of the lower elements of `a` and `b` to `NaN`. If
2086/// neither are equal to `NaN` then `0xFFFFFFFFFFFFFFFF` is used and `0`
2087/// otherwise.
2088///
2089/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpord_sd)
2090#[inline]
2091#[target_feature(enable = "sse2")]
2092#[cfg_attr(test, assert_instr(cmpordsd))]
2093#[stable(feature = "simd_x86", since = "1.27.0")]
2094pub fn _mm_cmpord_sd(a: __m128d, b: __m128d) -> __m128d {
2095    unsafe { cmpsd(a, b, 7) }
2096}
2097
2098/// Returns a new vector with the low element of `a` replaced by the result of
2099/// comparing both of the lower elements of `a` and `b` to `NaN`. If either is
2100/// equal to `NaN` then `0xFFFFFFFFFFFFFFFF` is used and `0` otherwise.
2101///
2102/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpunord_sd)
2103#[inline]
2104#[target_feature(enable = "sse2")]
2105#[cfg_attr(test, assert_instr(cmpunordsd))]
2106#[stable(feature = "simd_x86", since = "1.27.0")]
2107pub fn _mm_cmpunord_sd(a: __m128d, b: __m128d) -> __m128d {
2108    unsafe { cmpsd(a, b, 3) }
2109}
2110
2111/// Returns a new vector with the low element of `a` replaced by the not-equal
2112/// comparison of the lower elements of `a` and `b`.
2113///
2114/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpneq_sd)
2115#[inline]
2116#[target_feature(enable = "sse2")]
2117#[cfg_attr(test, assert_instr(cmpneqsd))]
2118#[stable(feature = "simd_x86", since = "1.27.0")]
2119pub fn _mm_cmpneq_sd(a: __m128d, b: __m128d) -> __m128d {
2120    unsafe { cmpsd(a, b, 4) }
2121}
2122
2123/// Returns a new vector with the low element of `a` replaced by the
2124/// not-less-than comparison of the lower elements of `a` and `b`.
2125///
2126/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpnlt_sd)
2127#[inline]
2128#[target_feature(enable = "sse2")]
2129#[cfg_attr(test, assert_instr(cmpnltsd))]
2130#[stable(feature = "simd_x86", since = "1.27.0")]
2131pub fn _mm_cmpnlt_sd(a: __m128d, b: __m128d) -> __m128d {
2132    unsafe { cmpsd(a, b, 5) }
2133}
2134
2135/// Returns a new vector with the low element of `a` replaced by the
2136/// not-less-than-or-equal comparison of the lower elements of `a` and `b`.
2137///
2138/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpnle_sd)
2139#[inline]
2140#[target_feature(enable = "sse2")]
2141#[cfg_attr(test, assert_instr(cmpnlesd))]
2142#[stable(feature = "simd_x86", since = "1.27.0")]
2143pub fn _mm_cmpnle_sd(a: __m128d, b: __m128d) -> __m128d {
2144    unsafe { cmpsd(a, b, 6) }
2145}
2146
2147/// Returns a new vector with the low element of `a` replaced by the
2148/// not-greater-than comparison of the lower elements of `a` and `b`.
2149///
2150/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpngt_sd)
2151#[inline]
2152#[target_feature(enable = "sse2")]
2153#[cfg_attr(test, assert_instr(cmpnltsd))]
2154#[stable(feature = "simd_x86", since = "1.27.0")]
2155pub fn _mm_cmpngt_sd(a: __m128d, b: __m128d) -> __m128d {
2156    unsafe { simd_insert!(_mm_cmpnlt_sd(b, a), 1, simd_extract!(a, 1, f64)) }
2157}
2158
2159/// Returns a new vector with the low element of `a` replaced by the
2160/// not-greater-than-or-equal comparison of the lower elements of `a` and `b`.
2161///
2162/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpnge_sd)
2163#[inline]
2164#[target_feature(enable = "sse2")]
2165#[cfg_attr(test, assert_instr(cmpnlesd))]
2166#[stable(feature = "simd_x86", since = "1.27.0")]
2167pub fn _mm_cmpnge_sd(a: __m128d, b: __m128d) -> __m128d {
2168    unsafe { simd_insert!(_mm_cmpnle_sd(b, a), 1, simd_extract!(a, 1, f64)) }
2169}
2170
2171/// Compares corresponding elements in `a` and `b` for equality.
2172///
2173/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpeq_pd)
2174#[inline]
2175#[target_feature(enable = "sse2")]
2176#[cfg_attr(test, assert_instr(cmpeqpd))]
2177#[stable(feature = "simd_x86", since = "1.27.0")]
2178pub fn _mm_cmpeq_pd(a: __m128d, b: __m128d) -> __m128d {
2179    unsafe { cmppd(a, b, 0) }
2180}
2181
2182/// Compares corresponding elements in `a` and `b` for less-than.
2183///
2184/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmplt_pd)
2185#[inline]
2186#[target_feature(enable = "sse2")]
2187#[cfg_attr(test, assert_instr(cmpltpd))]
2188#[stable(feature = "simd_x86", since = "1.27.0")]
2189pub fn _mm_cmplt_pd(a: __m128d, b: __m128d) -> __m128d {
2190    unsafe { cmppd(a, b, 1) }
2191}
2192
2193/// Compares corresponding elements in `a` and `b` for less-than-or-equal
2194///
2195/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmple_pd)
2196#[inline]
2197#[target_feature(enable = "sse2")]
2198#[cfg_attr(test, assert_instr(cmplepd))]
2199#[stable(feature = "simd_x86", since = "1.27.0")]
2200pub fn _mm_cmple_pd(a: __m128d, b: __m128d) -> __m128d {
2201    unsafe { cmppd(a, b, 2) }
2202}
2203
2204/// Compares corresponding elements in `a` and `b` for greater-than.
2205///
2206/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpgt_pd)
2207#[inline]
2208#[target_feature(enable = "sse2")]
2209#[cfg_attr(test, assert_instr(cmpltpd))]
2210#[stable(feature = "simd_x86", since = "1.27.0")]
2211pub fn _mm_cmpgt_pd(a: __m128d, b: __m128d) -> __m128d {
2212    _mm_cmplt_pd(b, a)
2213}
2214
2215/// Compares corresponding elements in `a` and `b` for greater-than-or-equal.
2216///
2217/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpge_pd)
2218#[inline]
2219#[target_feature(enable = "sse2")]
2220#[cfg_attr(test, assert_instr(cmplepd))]
2221#[stable(feature = "simd_x86", since = "1.27.0")]
2222pub fn _mm_cmpge_pd(a: __m128d, b: __m128d) -> __m128d {
2223    _mm_cmple_pd(b, a)
2224}
2225
2226/// Compares corresponding elements in `a` and `b` to see if neither is `NaN`.
2227///
2228/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpord_pd)
2229#[inline]
2230#[target_feature(enable = "sse2")]
2231#[cfg_attr(test, assert_instr(cmpordpd))]
2232#[stable(feature = "simd_x86", since = "1.27.0")]
2233pub fn _mm_cmpord_pd(a: __m128d, b: __m128d) -> __m128d {
2234    unsafe { cmppd(a, b, 7) }
2235}
2236
2237/// Compares corresponding elements in `a` and `b` to see if either is `NaN`.
2238///
2239/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpunord_pd)
2240#[inline]
2241#[target_feature(enable = "sse2")]
2242#[cfg_attr(test, assert_instr(cmpunordpd))]
2243#[stable(feature = "simd_x86", since = "1.27.0")]
2244pub fn _mm_cmpunord_pd(a: __m128d, b: __m128d) -> __m128d {
2245    unsafe { cmppd(a, b, 3) }
2246}
2247
2248/// Compares corresponding elements in `a` and `b` for not-equal.
2249///
2250/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpneq_pd)
2251#[inline]
2252#[target_feature(enable = "sse2")]
2253#[cfg_attr(test, assert_instr(cmpneqpd))]
2254#[stable(feature = "simd_x86", since = "1.27.0")]
2255pub fn _mm_cmpneq_pd(a: __m128d, b: __m128d) -> __m128d {
2256    unsafe { cmppd(a, b, 4) }
2257}
2258
2259/// Compares corresponding elements in `a` and `b` for not-less-than.
2260///
2261/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpnlt_pd)
2262#[inline]
2263#[target_feature(enable = "sse2")]
2264#[cfg_attr(test, assert_instr(cmpnltpd))]
2265#[stable(feature = "simd_x86", since = "1.27.0")]
2266pub fn _mm_cmpnlt_pd(a: __m128d, b: __m128d) -> __m128d {
2267    unsafe { cmppd(a, b, 5) }
2268}
2269
2270/// Compares corresponding elements in `a` and `b` for not-less-than-or-equal.
2271///
2272/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpnle_pd)
2273#[inline]
2274#[target_feature(enable = "sse2")]
2275#[cfg_attr(test, assert_instr(cmpnlepd))]
2276#[stable(feature = "simd_x86", since = "1.27.0")]
2277pub fn _mm_cmpnle_pd(a: __m128d, b: __m128d) -> __m128d {
2278    unsafe { cmppd(a, b, 6) }
2279}
2280
2281/// Compares corresponding elements in `a` and `b` for not-greater-than.
2282///
2283/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpngt_pd)
2284#[inline]
2285#[target_feature(enable = "sse2")]
2286#[cfg_attr(test, assert_instr(cmpnltpd))]
2287#[stable(feature = "simd_x86", since = "1.27.0")]
2288pub fn _mm_cmpngt_pd(a: __m128d, b: __m128d) -> __m128d {
2289    _mm_cmpnlt_pd(b, a)
2290}
2291
2292/// Compares corresponding elements in `a` and `b` for
2293/// not-greater-than-or-equal.
2294///
2295/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cmpnge_pd)
2296#[inline]
2297#[target_feature(enable = "sse2")]
2298#[cfg_attr(test, assert_instr(cmpnlepd))]
2299#[stable(feature = "simd_x86", since = "1.27.0")]
2300pub fn _mm_cmpnge_pd(a: __m128d, b: __m128d) -> __m128d {
2301    _mm_cmpnle_pd(b, a)
2302}
2303
2304/// Compares the lower element of `a` and `b` for equality.
2305///
2306/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_comieq_sd)
2307#[inline]
2308#[target_feature(enable = "sse2")]
2309#[cfg_attr(test, assert_instr(comisd))]
2310#[stable(feature = "simd_x86", since = "1.27.0")]
2311pub fn _mm_comieq_sd(a: __m128d, b: __m128d) -> i32 {
2312    unsafe { comieqsd(a, b) }
2313}
2314
2315/// Compares the lower element of `a` and `b` for less-than.
2316///
2317/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_comilt_sd)
2318#[inline]
2319#[target_feature(enable = "sse2")]
2320#[cfg_attr(test, assert_instr(comisd))]
2321#[stable(feature = "simd_x86", since = "1.27.0")]
2322pub fn _mm_comilt_sd(a: __m128d, b: __m128d) -> i32 {
2323    unsafe { comiltsd(a, b) }
2324}
2325
2326/// Compares the lower element of `a` and `b` for less-than-or-equal.
2327///
2328/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_comile_sd)
2329#[inline]
2330#[target_feature(enable = "sse2")]
2331#[cfg_attr(test, assert_instr(comisd))]
2332#[stable(feature = "simd_x86", since = "1.27.0")]
2333pub fn _mm_comile_sd(a: __m128d, b: __m128d) -> i32 {
2334    unsafe { comilesd(a, b) }
2335}
2336
2337/// Compares the lower element of `a` and `b` for greater-than.
2338///
2339/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_comigt_sd)
2340#[inline]
2341#[target_feature(enable = "sse2")]
2342#[cfg_attr(test, assert_instr(comisd))]
2343#[stable(feature = "simd_x86", since = "1.27.0")]
2344pub fn _mm_comigt_sd(a: __m128d, b: __m128d) -> i32 {
2345    unsafe { comigtsd(a, b) }
2346}
2347
2348/// Compares the lower element of `a` and `b` for greater-than-or-equal.
2349///
2350/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_comige_sd)
2351#[inline]
2352#[target_feature(enable = "sse2")]
2353#[cfg_attr(test, assert_instr(comisd))]
2354#[stable(feature = "simd_x86", since = "1.27.0")]
2355pub fn _mm_comige_sd(a: __m128d, b: __m128d) -> i32 {
2356    unsafe { comigesd(a, b) }
2357}
2358
2359/// Compares the lower element of `a` and `b` for not-equal.
2360///
2361/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_comineq_sd)
2362#[inline]
2363#[target_feature(enable = "sse2")]
2364#[cfg_attr(test, assert_instr(comisd))]
2365#[stable(feature = "simd_x86", since = "1.27.0")]
2366pub fn _mm_comineq_sd(a: __m128d, b: __m128d) -> i32 {
2367    unsafe { comineqsd(a, b) }
2368}
2369
2370/// Compares the lower element of `a` and `b` for equality.
2371///
2372/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_ucomieq_sd)
2373#[inline]
2374#[target_feature(enable = "sse2")]
2375#[cfg_attr(test, assert_instr(ucomisd))]
2376#[stable(feature = "simd_x86", since = "1.27.0")]
2377pub fn _mm_ucomieq_sd(a: __m128d, b: __m128d) -> i32 {
2378    unsafe { ucomieqsd(a, b) }
2379}
2380
2381/// Compares the lower element of `a` and `b` for less-than.
2382///
2383/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_ucomilt_sd)
2384#[inline]
2385#[target_feature(enable = "sse2")]
2386#[cfg_attr(test, assert_instr(ucomisd))]
2387#[stable(feature = "simd_x86", since = "1.27.0")]
2388pub fn _mm_ucomilt_sd(a: __m128d, b: __m128d) -> i32 {
2389    unsafe { ucomiltsd(a, b) }
2390}
2391
2392/// Compares the lower element of `a` and `b` for less-than-or-equal.
2393///
2394/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_ucomile_sd)
2395#[inline]
2396#[target_feature(enable = "sse2")]
2397#[cfg_attr(test, assert_instr(ucomisd))]
2398#[stable(feature = "simd_x86", since = "1.27.0")]
2399pub fn _mm_ucomile_sd(a: __m128d, b: __m128d) -> i32 {
2400    unsafe { ucomilesd(a, b) }
2401}
2402
2403/// Compares the lower element of `a` and `b` for greater-than.
2404///
2405/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_ucomigt_sd)
2406#[inline]
2407#[target_feature(enable = "sse2")]
2408#[cfg_attr(test, assert_instr(ucomisd))]
2409#[stable(feature = "simd_x86", since = "1.27.0")]
2410pub fn _mm_ucomigt_sd(a: __m128d, b: __m128d) -> i32 {
2411    unsafe { ucomigtsd(a, b) }
2412}
2413
2414/// Compares the lower element of `a` and `b` for greater-than-or-equal.
2415///
2416/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_ucomige_sd)
2417#[inline]
2418#[target_feature(enable = "sse2")]
2419#[cfg_attr(test, assert_instr(ucomisd))]
2420#[stable(feature = "simd_x86", since = "1.27.0")]
2421pub fn _mm_ucomige_sd(a: __m128d, b: __m128d) -> i32 {
2422    unsafe { ucomigesd(a, b) }
2423}
2424
2425/// Compares the lower element of `a` and `b` for not-equal.
2426///
2427/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_ucomineq_sd)
2428#[inline]
2429#[target_feature(enable = "sse2")]
2430#[cfg_attr(test, assert_instr(ucomisd))]
2431#[stable(feature = "simd_x86", since = "1.27.0")]
2432pub fn _mm_ucomineq_sd(a: __m128d, b: __m128d) -> i32 {
2433    unsafe { ucomineqsd(a, b) }
2434}
2435
2436/// Converts packed double-precision (64-bit) floating-point elements in `a` to
2437/// packed single-precision (32-bit) floating-point elements
2438///
2439/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtpd_ps)
2440#[inline]
2441#[target_feature(enable = "sse2")]
2442#[cfg_attr(test, assert_instr(cvtpd2ps))]
2443#[stable(feature = "simd_x86", since = "1.27.0")]
2444#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2445pub const fn _mm_cvtpd_ps(a: __m128d) -> __m128 {
2446    unsafe {
2447        let r = simd_cast::<_, f32x2>(a.as_f64x2());
2448        let zero = f32x2::ZERO;
2449        transmute::<f32x4, _>(simd_shuffle!(r, zero, [0, 1, 2, 3]))
2450    }
2451}
2452
2453/// Converts packed single-precision (32-bit) floating-point elements in `a` to
2454/// packed
2455/// double-precision (64-bit) floating-point elements.
2456///
2457/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtps_pd)
2458#[inline]
2459#[target_feature(enable = "sse2")]
2460#[cfg_attr(test, assert_instr(cvtps2pd))]
2461#[stable(feature = "simd_x86", since = "1.27.0")]
2462#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2463pub const fn _mm_cvtps_pd(a: __m128) -> __m128d {
2464    unsafe {
2465        let a = a.as_f32x4();
2466        transmute(simd_cast::<f32x2, f64x2>(simd_shuffle!(a, a, [0, 1])))
2467    }
2468}
2469
2470/// Converts packed double-precision (64-bit) floating-point elements in `a` to
2471/// packed 32-bit integers.
2472///
2473/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtpd_epi32)
2474#[inline]
2475#[target_feature(enable = "sse2")]
2476#[cfg_attr(test, assert_instr(cvtpd2dq))]
2477#[stable(feature = "simd_x86", since = "1.27.0")]
2478pub fn _mm_cvtpd_epi32(a: __m128d) -> __m128i {
2479    unsafe { transmute(cvtpd2dq(a)) }
2480}
2481
2482/// Converts the lower double-precision (64-bit) floating-point element in a to
2483/// a 32-bit integer.
2484///
2485/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtsd_si32)
2486#[inline]
2487#[target_feature(enable = "sse2")]
2488#[cfg_attr(test, assert_instr(cvtsd2si))]
2489#[stable(feature = "simd_x86", since = "1.27.0")]
2490pub fn _mm_cvtsd_si32(a: __m128d) -> i32 {
2491    unsafe { cvtsd2si(a) }
2492}
2493
2494/// Converts the lower double-precision (64-bit) floating-point element in `b`
2495/// to a single-precision (32-bit) floating-point element, store the result in
2496/// the lower element of the return value, and copies the upper element from `a`
2497/// to the upper element the return value.
2498///
2499/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtsd_ss)
2500#[inline]
2501#[target_feature(enable = "sse2")]
2502#[cfg_attr(test, assert_instr(cvtsd2ss))]
2503#[stable(feature = "simd_x86", since = "1.27.0")]
2504pub fn _mm_cvtsd_ss(a: __m128, b: __m128d) -> __m128 {
2505    unsafe { cvtsd2ss(a, b) }
2506}
2507
2508/// Returns the lower double-precision (64-bit) floating-point element of `a`.
2509///
2510/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtsd_f64)
2511#[inline]
2512#[target_feature(enable = "sse2")]
2513#[stable(feature = "simd_x86", since = "1.27.0")]
2514#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2515pub const fn _mm_cvtsd_f64(a: __m128d) -> f64 {
2516    unsafe { simd_extract!(a, 0) }
2517}
2518
2519/// Converts the lower single-precision (32-bit) floating-point element in `b`
2520/// to a double-precision (64-bit) floating-point element, store the result in
2521/// the lower element of the return value, and copies the upper element from `a`
2522/// to the upper element the return value.
2523///
2524/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvtss_sd)
2525#[inline]
2526#[target_feature(enable = "sse2")]
2527#[cfg_attr(test, assert_instr(cvtss2sd))]
2528#[stable(feature = "simd_x86", since = "1.27.0")]
2529#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2530pub const fn _mm_cvtss_sd(a: __m128d, b: __m128) -> __m128d {
2531    unsafe {
2532        let elt: f32 = simd_extract!(b, 0);
2533        simd_insert!(a, 0, elt as f64)
2534    }
2535}
2536
2537/// Converts packed double-precision (64-bit) floating-point elements in `a` to
2538/// packed 32-bit integers with truncation.
2539///
2540/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvttpd_epi32)
2541#[inline]
2542#[target_feature(enable = "sse2")]
2543#[cfg_attr(test, assert_instr(cvttpd2dq))]
2544#[stable(feature = "simd_x86", since = "1.27.0")]
2545pub fn _mm_cvttpd_epi32(a: __m128d) -> __m128i {
2546    unsafe { transmute(cvttpd2dq(a)) }
2547}
2548
2549/// Converts the lower double-precision (64-bit) floating-point element in `a`
2550/// to a 32-bit integer with truncation.
2551///
2552/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvttsd_si32)
2553#[inline]
2554#[target_feature(enable = "sse2")]
2555#[cfg_attr(test, assert_instr(cvttsd2si))]
2556#[stable(feature = "simd_x86", since = "1.27.0")]
2557pub fn _mm_cvttsd_si32(a: __m128d) -> i32 {
2558    unsafe { cvttsd2si(a) }
2559}
2560
2561/// Converts packed single-precision (32-bit) floating-point elements in `a` to
2562/// packed 32-bit integers with truncation.
2563///
2564/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_cvttps_epi32)
2565#[inline]
2566#[target_feature(enable = "sse2")]
2567#[cfg_attr(test, assert_instr(cvttps2dq))]
2568#[stable(feature = "simd_x86", since = "1.27.0")]
2569pub fn _mm_cvttps_epi32(a: __m128) -> __m128i {
2570    unsafe { transmute(cvttps2dq(a)) }
2571}
2572
2573/// Copies double-precision (64-bit) floating-point element `a` to the lower
2574/// element of the packed 64-bit return value.
2575///
2576/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set_sd)
2577#[inline]
2578#[target_feature(enable = "sse2")]
2579#[stable(feature = "simd_x86", since = "1.27.0")]
2580#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2581pub const fn _mm_set_sd(a: f64) -> __m128d {
2582    _mm_set_pd(0.0, a)
2583}
2584
2585/// Broadcasts double-precision (64-bit) floating-point value a to all elements
2586/// of the return value.
2587///
2588/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set1_pd)
2589#[inline]
2590#[target_feature(enable = "sse2")]
2591#[stable(feature = "simd_x86", since = "1.27.0")]
2592#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2593pub const fn _mm_set1_pd(a: f64) -> __m128d {
2594    _mm_set_pd(a, a)
2595}
2596
2597/// Broadcasts double-precision (64-bit) floating-point value a to all elements
2598/// of the return value.
2599///
2600/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set_pd1)
2601#[inline]
2602#[target_feature(enable = "sse2")]
2603#[stable(feature = "simd_x86", since = "1.27.0")]
2604#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2605pub const fn _mm_set_pd1(a: f64) -> __m128d {
2606    _mm_set_pd(a, a)
2607}
2608
2609/// Sets packed double-precision (64-bit) floating-point elements in the return
2610/// value with the supplied values.
2611///
2612/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_set_pd)
2613#[inline]
2614#[target_feature(enable = "sse2")]
2615#[stable(feature = "simd_x86", since = "1.27.0")]
2616#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2617pub const fn _mm_set_pd(a: f64, b: f64) -> __m128d {
2618    __m128d([b, a])
2619}
2620
2621/// Sets packed double-precision (64-bit) floating-point elements in the return
2622/// value with the supplied values in reverse order.
2623///
2624/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_setr_pd)
2625#[inline]
2626#[target_feature(enable = "sse2")]
2627#[stable(feature = "simd_x86", since = "1.27.0")]
2628#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2629pub const fn _mm_setr_pd(a: f64, b: f64) -> __m128d {
2630    _mm_set_pd(b, a)
2631}
2632
2633/// Returns packed double-precision (64-bit) floating-point elements with all
2634/// zeros.
2635///
2636/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_setzero_pd)
2637#[inline]
2638#[target_feature(enable = "sse2")]
2639#[cfg_attr(test, assert_instr(xorp))]
2640#[stable(feature = "simd_x86", since = "1.27.0")]
2641#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2642pub const fn _mm_setzero_pd() -> __m128d {
2643    const { unsafe { mem::zeroed() } }
2644}
2645
2646/// Returns a mask of the most significant bit of each element in `a`.
2647///
2648/// The mask is stored in the 2 least significant bits of the return value.
2649/// All other bits are set to `0`.
2650///
2651/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_movemask_pd)
2652#[inline]
2653#[target_feature(enable = "sse2")]
2654#[cfg_attr(test, assert_instr(movmskpd))]
2655#[stable(feature = "simd_x86", since = "1.27.0")]
2656#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2657pub const fn _mm_movemask_pd(a: __m128d) -> i32 {
2658    // Propagate the highest bit to the rest, because simd_bitmask
2659    // requires all-1 or all-0.
2660    unsafe {
2661        let mask: i64x2 = simd_lt(transmute(a), i64x2::ZERO);
2662        simd_bitmask::<i64x2, u8>(mask) as i32
2663    }
2664}
2665
2666/// Loads 128-bits (composed of 2 packed double-precision (64-bit)
2667/// floating-point elements) from memory into the returned vector.
2668/// `mem_addr` must be aligned on a 16-byte boundary or a general-protection
2669/// exception may be generated.
2670///
2671/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_load_pd)
2672#[inline]
2673#[target_feature(enable = "sse2")]
2674#[cfg_attr(
2675    all(test, not(all(target_arch = "x86", target_env = "msvc"))),
2676    assert_instr(movaps)
2677)]
2678#[stable(feature = "simd_x86", since = "1.27.0")]
2679#[allow(clippy::cast_ptr_alignment)]
2680#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2681pub const unsafe fn _mm_load_pd(mem_addr: *const f64) -> __m128d {
2682    *(mem_addr as *const __m128d)
2683}
2684
2685/// Loads a 64-bit double-precision value to the low element of a
2686/// 128-bit integer vector and clears the upper element.
2687///
2688/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_load_sd)
2689#[inline]
2690#[target_feature(enable = "sse2")]
2691#[cfg_attr(test, assert_instr(movsd))]
2692#[stable(feature = "simd_x86", since = "1.27.0")]
2693#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2694pub const unsafe fn _mm_load_sd(mem_addr: *const f64) -> __m128d {
2695    _mm_setr_pd(*mem_addr, 0.)
2696}
2697
2698/// Loads a double-precision value into the high-order bits of a 128-bit
2699/// vector of `[2 x double]`. The low-order bits are copied from the low-order
2700/// bits of the first operand.
2701///
2702/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadh_pd)
2703#[inline]
2704#[target_feature(enable = "sse2")]
2705#[cfg_attr(test, assert_instr(movhps))]
2706#[stable(feature = "simd_x86", since = "1.27.0")]
2707#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2708pub const unsafe fn _mm_loadh_pd(a: __m128d, mem_addr: *const f64) -> __m128d {
2709    _mm_setr_pd(simd_extract!(a, 0), *mem_addr)
2710}
2711
2712/// Loads a double-precision value into the low-order bits of a 128-bit
2713/// vector of `[2 x double]`. The high-order bits are copied from the
2714/// high-order bits of the first operand.
2715///
2716/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadl_pd)
2717#[inline]
2718#[target_feature(enable = "sse2")]
2719#[cfg_attr(test, assert_instr(movlps))]
2720#[stable(feature = "simd_x86", since = "1.27.0")]
2721#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2722pub const unsafe fn _mm_loadl_pd(a: __m128d, mem_addr: *const f64) -> __m128d {
2723    _mm_setr_pd(*mem_addr, simd_extract!(a, 1))
2724}
2725
2726/// Stores a 128-bit floating point vector of `[2 x double]` to a 128-bit
2727/// aligned memory location.
2728/// To minimize caching, the data is flagged as non-temporal (unlikely to be
2729/// used again soon).
2730///
2731/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_stream_pd)
2732///
2733/// # Safety of non-temporal stores
2734///
2735/// After using this intrinsic, but before any other access to the memory that this intrinsic
2736/// mutates, a call to [`_mm_sfence`] must be performed by the thread that used the intrinsic. In
2737/// particular, functions that call this intrinsic should generally call `_mm_sfence` before they
2738/// return.
2739///
2740/// See [`_mm_sfence`] for details.
2741#[inline]
2742#[target_feature(enable = "sse2")]
2743#[cfg_attr(test, assert_instr(movntpd))]
2744#[stable(feature = "simd_x86", since = "1.27.0")]
2745#[allow(clippy::cast_ptr_alignment)]
2746pub unsafe fn _mm_stream_pd(mem_addr: *mut f64, a: __m128d) {
2747    // see #1541, we should use inline asm to be sure, because LangRef isn't clear enough
2748    crate::arch::asm!(
2749        vps!("movntpd", ",{a}"),
2750        p = in(reg) mem_addr,
2751        a = in(xmm_reg) a,
2752        options(nostack, preserves_flags),
2753    );
2754}
2755
2756/// Stores the lower 64 bits of a 128-bit vector of `[2 x double]` to a
2757/// memory location.
2758///
2759/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_store_sd)
2760#[inline]
2761#[target_feature(enable = "sse2")]
2762#[cfg_attr(test, assert_instr(movlps))]
2763#[stable(feature = "simd_x86", since = "1.27.0")]
2764#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2765pub const unsafe fn _mm_store_sd(mem_addr: *mut f64, a: __m128d) {
2766    *mem_addr = simd_extract!(a, 0)
2767}
2768
2769/// Stores 128-bits (composed of 2 packed double-precision (64-bit)
2770/// floating-point elements) from `a` into memory. `mem_addr` must be aligned
2771/// on a 16-byte boundary or a general-protection exception may be generated.
2772///
2773/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_store_pd)
2774#[inline]
2775#[target_feature(enable = "sse2")]
2776#[cfg_attr(
2777    all(test, not(all(target_arch = "x86", target_env = "msvc"))),
2778    assert_instr(movaps)
2779)]
2780#[stable(feature = "simd_x86", since = "1.27.0")]
2781#[allow(clippy::cast_ptr_alignment)]
2782#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2783pub const unsafe fn _mm_store_pd(mem_addr: *mut f64, a: __m128d) {
2784    *(mem_addr as *mut __m128d) = a;
2785}
2786
2787/// Stores 128-bits (composed of 2 packed double-precision (64-bit)
2788/// floating-point elements) from `a` into memory.
2789/// `mem_addr` does not need to be aligned on any particular boundary.
2790///
2791/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storeu_pd)
2792#[inline]
2793#[target_feature(enable = "sse2")]
2794#[cfg_attr(test, assert_instr(movups))] // FIXME movupd expected
2795#[stable(feature = "simd_x86", since = "1.27.0")]
2796#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2797pub const unsafe fn _mm_storeu_pd(mem_addr: *mut f64, a: __m128d) {
2798    mem_addr.cast::<__m128d>().write_unaligned(a);
2799}
2800
2801/// Store 16-bit integer from the first element of a into memory.
2802///
2803/// `mem_addr` does not need to be aligned on any particular boundary.
2804///
2805/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storeu_si16)
2806#[inline]
2807#[target_feature(enable = "sse2")]
2808#[stable(feature = "simd_x86_updates", since = "1.82.0")]
2809#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2810pub const unsafe fn _mm_storeu_si16(mem_addr: *mut u8, a: __m128i) {
2811    ptr::write_unaligned(mem_addr as *mut i16, simd_extract(a.as_i16x8(), 0))
2812}
2813
2814/// Store 32-bit integer from the first element of a into memory.
2815///
2816/// `mem_addr` does not need to be aligned on any particular boundary.
2817///
2818/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storeu_si32)
2819#[inline]
2820#[target_feature(enable = "sse2")]
2821#[stable(feature = "simd_x86_updates", since = "1.82.0")]
2822#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2823pub const unsafe fn _mm_storeu_si32(mem_addr: *mut u8, a: __m128i) {
2824    ptr::write_unaligned(mem_addr as *mut i32, simd_extract(a.as_i32x4(), 0))
2825}
2826
2827/// Store 64-bit integer from the first element of a into memory.
2828///
2829/// `mem_addr` does not need to be aligned on any particular boundary.
2830///
2831/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storeu_si64)
2832#[inline]
2833#[target_feature(enable = "sse2")]
2834#[stable(feature = "simd_x86_updates", since = "1.82.0")]
2835#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2836pub const unsafe fn _mm_storeu_si64(mem_addr: *mut u8, a: __m128i) {
2837    ptr::write_unaligned(mem_addr as *mut i64, simd_extract(a.as_i64x2(), 0))
2838}
2839
2840/// Stores the lower double-precision (64-bit) floating-point element from `a`
2841/// into 2 contiguous elements in memory. `mem_addr` must be aligned on a
2842/// 16-byte boundary or a general-protection exception may be generated.
2843///
2844/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_store1_pd)
2845#[inline]
2846#[target_feature(enable = "sse2")]
2847#[stable(feature = "simd_x86", since = "1.27.0")]
2848#[allow(clippy::cast_ptr_alignment)]
2849#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2850pub const unsafe fn _mm_store1_pd(mem_addr: *mut f64, a: __m128d) {
2851    let b: __m128d = simd_shuffle!(a, a, [0, 0]);
2852    *(mem_addr as *mut __m128d) = b;
2853}
2854
2855/// Stores the lower double-precision (64-bit) floating-point element from `a`
2856/// into 2 contiguous elements in memory. `mem_addr` must be aligned on a
2857/// 16-byte boundary or a general-protection exception may be generated.
2858///
2859/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_store_pd1)
2860#[inline]
2861#[target_feature(enable = "sse2")]
2862#[stable(feature = "simd_x86", since = "1.27.0")]
2863#[allow(clippy::cast_ptr_alignment)]
2864#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2865pub const unsafe fn _mm_store_pd1(mem_addr: *mut f64, a: __m128d) {
2866    let b: __m128d = simd_shuffle!(a, a, [0, 0]);
2867    *(mem_addr as *mut __m128d) = b;
2868}
2869
2870/// Stores 2 double-precision (64-bit) floating-point elements from `a` into
2871/// memory in reverse order.
2872/// `mem_addr` must be aligned on a 16-byte boundary or a general-protection
2873/// exception may be generated.
2874///
2875/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storer_pd)
2876#[inline]
2877#[target_feature(enable = "sse2")]
2878#[stable(feature = "simd_x86", since = "1.27.0")]
2879#[allow(clippy::cast_ptr_alignment)]
2880#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2881pub const unsafe fn _mm_storer_pd(mem_addr: *mut f64, a: __m128d) {
2882    let b: __m128d = simd_shuffle!(a, a, [1, 0]);
2883    *(mem_addr as *mut __m128d) = b;
2884}
2885
2886/// Stores the upper 64 bits of a 128-bit vector of `[2 x double]` to a
2887/// memory location.
2888///
2889/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storeh_pd)
2890#[inline]
2891#[target_feature(enable = "sse2")]
2892#[cfg_attr(test, assert_instr(movhps))]
2893#[stable(feature = "simd_x86", since = "1.27.0")]
2894#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2895pub const unsafe fn _mm_storeh_pd(mem_addr: *mut f64, a: __m128d) {
2896    *mem_addr = simd_extract!(a, 1);
2897}
2898
2899/// Stores the lower 64 bits of a 128-bit vector of `[2 x double]` to a
2900/// memory location.
2901///
2902/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_storel_pd)
2903#[inline]
2904#[target_feature(enable = "sse2")]
2905#[cfg_attr(test, assert_instr(movlps))]
2906#[stable(feature = "simd_x86", since = "1.27.0")]
2907#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2908pub const unsafe fn _mm_storel_pd(mem_addr: *mut f64, a: __m128d) {
2909    *mem_addr = simd_extract!(a, 0);
2910}
2911
2912/// Loads a double-precision (64-bit) floating-point element from memory
2913/// into both elements of returned vector.
2914///
2915/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_load1_pd)
2916#[inline]
2917#[target_feature(enable = "sse2")]
2918// #[cfg_attr(test, assert_instr(movapd))] // FIXME LLVM uses different codegen
2919#[stable(feature = "simd_x86", since = "1.27.0")]
2920#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2921pub const unsafe fn _mm_load1_pd(mem_addr: *const f64) -> __m128d {
2922    let d = *mem_addr;
2923    _mm_setr_pd(d, d)
2924}
2925
2926/// Loads a double-precision (64-bit) floating-point element from memory
2927/// into both elements of returned vector.
2928///
2929/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_load_pd1)
2930#[inline]
2931#[target_feature(enable = "sse2")]
2932// #[cfg_attr(test, assert_instr(movapd))] // FIXME same as _mm_load1_pd
2933#[stable(feature = "simd_x86", since = "1.27.0")]
2934#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2935pub const unsafe fn _mm_load_pd1(mem_addr: *const f64) -> __m128d {
2936    _mm_load1_pd(mem_addr)
2937}
2938
2939/// Loads 2 double-precision (64-bit) floating-point elements from memory into
2940/// the returned vector in reverse order. `mem_addr` must be aligned on a
2941/// 16-byte boundary or a general-protection exception may be generated.
2942///
2943/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadr_pd)
2944#[inline]
2945#[target_feature(enable = "sse2")]
2946#[cfg_attr(
2947    all(test, not(all(target_arch = "x86", target_env = "msvc"))),
2948    assert_instr(movaps)
2949)]
2950#[stable(feature = "simd_x86", since = "1.27.0")]
2951#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2952pub const unsafe fn _mm_loadr_pd(mem_addr: *const f64) -> __m128d {
2953    let a = _mm_load_pd(mem_addr);
2954    simd_shuffle!(a, a, [1, 0])
2955}
2956
2957/// Loads 128-bits (composed of 2 packed double-precision (64-bit)
2958/// floating-point elements) from memory into the returned vector.
2959/// `mem_addr` does not need to be aligned on any particular boundary.
2960///
2961/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadu_pd)
2962#[inline]
2963#[target_feature(enable = "sse2")]
2964#[cfg_attr(test, assert_instr(movups))]
2965#[stable(feature = "simd_x86", since = "1.27.0")]
2966#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2967pub const unsafe fn _mm_loadu_pd(mem_addr: *const f64) -> __m128d {
2968    let mut dst = _mm_undefined_pd();
2969    ptr::copy_nonoverlapping(
2970        mem_addr as *const u8,
2971        ptr::addr_of_mut!(dst) as *mut u8,
2972        mem::size_of::<__m128d>(),
2973    );
2974    dst
2975}
2976
2977/// Loads unaligned 16-bits of integer data from memory into new vector.
2978///
2979/// `mem_addr` does not need to be aligned on any particular boundary.
2980///
2981/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadu_si16)
2982#[inline]
2983#[target_feature(enable = "sse2")]
2984#[stable(feature = "simd_x86_updates", since = "1.82.0")]
2985#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
2986pub const unsafe fn _mm_loadu_si16(mem_addr: *const u8) -> __m128i {
2987    transmute(i16x8::new(
2988        ptr::read_unaligned(mem_addr as *const i16),
2989        0,
2990        0,
2991        0,
2992        0,
2993        0,
2994        0,
2995        0,
2996    ))
2997}
2998
2999/// Loads unaligned 32-bits of integer data from memory into new vector.
3000///
3001/// `mem_addr` does not need to be aligned on any particular boundary.
3002///
3003/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadu_si32)
3004#[inline]
3005#[target_feature(enable = "sse2")]
3006#[stable(feature = "simd_x86_updates", since = "1.82.0")]
3007#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3008pub const unsafe fn _mm_loadu_si32(mem_addr: *const u8) -> __m128i {
3009    transmute(i32x4::new(
3010        ptr::read_unaligned(mem_addr as *const i32),
3011        0,
3012        0,
3013        0,
3014    ))
3015}
3016
3017/// Loads unaligned 64-bits of integer data from memory into new vector.
3018///
3019/// `mem_addr` does not need to be aligned on any particular boundary.
3020///
3021/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_loadu_si64)
3022#[inline]
3023#[target_feature(enable = "sse2")]
3024#[stable(feature = "simd_x86_mm_loadu_si64", since = "1.46.0")]
3025#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3026pub const unsafe fn _mm_loadu_si64(mem_addr: *const u8) -> __m128i {
3027    transmute(i64x2::new(ptr::read_unaligned(mem_addr as *const i64), 0))
3028}
3029
3030/// Constructs a 128-bit floating-point vector of `[2 x double]` from two
3031/// 128-bit vector parameters of `[2 x double]`, using the immediate-value
3032/// parameter as a specifier.
3033///
3034/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_shuffle_pd)
3035#[inline]
3036#[target_feature(enable = "sse2")]
3037#[cfg_attr(test, assert_instr(shufps, MASK = 2))]
3038#[rustc_legacy_const_generics(2)]
3039#[stable(feature = "simd_x86", since = "1.27.0")]
3040#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3041pub const fn _mm_shuffle_pd<const MASK: i32>(a: __m128d, b: __m128d) -> __m128d {
3042    static_assert_uimm_bits!(MASK, 8);
3043    unsafe { simd_shuffle!(a, b, [MASK as u32 & 0b1, ((MASK as u32 >> 1) & 0b1) + 2]) }
3044}
3045
3046/// Constructs a 128-bit floating-point vector of `[2 x double]`. The lower
3047/// 64 bits are set to the lower 64 bits of the second parameter. The upper
3048/// 64 bits are set to the upper 64 bits of the first parameter.
3049///
3050/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_move_sd)
3051#[inline]
3052#[target_feature(enable = "sse2")]
3053#[cfg_attr(test, assert_instr(movsd))]
3054#[stable(feature = "simd_x86", since = "1.27.0")]
3055#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3056pub const fn _mm_move_sd(a: __m128d, b: __m128d) -> __m128d {
3057    unsafe { _mm_setr_pd(simd_extract!(b, 0), simd_extract!(a, 1)) }
3058}
3059
3060/// Casts a 128-bit floating-point vector of `[2 x double]` into a 128-bit
3061/// floating-point vector of `[4 x float]`.
3062///
3063/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_castpd_ps)
3064#[inline]
3065#[target_feature(enable = "sse2")]
3066#[stable(feature = "simd_x86", since = "1.27.0")]
3067#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3068pub const fn _mm_castpd_ps(a: __m128d) -> __m128 {
3069    unsafe { transmute(a) }
3070}
3071
3072/// Casts a 128-bit floating-point vector of `[2 x double]` into a 128-bit
3073/// integer vector.
3074///
3075/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_castpd_si128)
3076#[inline]
3077#[target_feature(enable = "sse2")]
3078#[stable(feature = "simd_x86", since = "1.27.0")]
3079#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3080pub const fn _mm_castpd_si128(a: __m128d) -> __m128i {
3081    unsafe { transmute(a) }
3082}
3083
3084/// Casts a 128-bit floating-point vector of `[4 x float]` into a 128-bit
3085/// floating-point vector of `[2 x double]`.
3086///
3087/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_castps_pd)
3088#[inline]
3089#[target_feature(enable = "sse2")]
3090#[stable(feature = "simd_x86", since = "1.27.0")]
3091#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3092pub const fn _mm_castps_pd(a: __m128) -> __m128d {
3093    unsafe { transmute(a) }
3094}
3095
3096/// Casts a 128-bit floating-point vector of `[4 x float]` into a 128-bit
3097/// integer vector.
3098///
3099/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_castps_si128)
3100#[inline]
3101#[target_feature(enable = "sse2")]
3102#[stable(feature = "simd_x86", since = "1.27.0")]
3103#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3104pub const fn _mm_castps_si128(a: __m128) -> __m128i {
3105    unsafe { transmute(a) }
3106}
3107
3108/// Casts a 128-bit integer vector into a 128-bit floating-point vector
3109/// of `[2 x double]`.
3110///
3111/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_castsi128_pd)
3112#[inline]
3113#[target_feature(enable = "sse2")]
3114#[stable(feature = "simd_x86", since = "1.27.0")]
3115#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3116pub const fn _mm_castsi128_pd(a: __m128i) -> __m128d {
3117    unsafe { transmute(a) }
3118}
3119
3120/// Casts a 128-bit integer vector into a 128-bit floating-point vector
3121/// of `[4 x float]`.
3122///
3123/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_castsi128_ps)
3124#[inline]
3125#[target_feature(enable = "sse2")]
3126#[stable(feature = "simd_x86", since = "1.27.0")]
3127#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3128pub const fn _mm_castsi128_ps(a: __m128i) -> __m128 {
3129    unsafe { transmute(a) }
3130}
3131
3132/// Returns vector of type __m128d with indeterminate elements.with indetermination elements.
3133/// Despite using the word "undefined" (following Intel's naming scheme), this non-deterministically
3134/// picks some valid value and is not equivalent to [`mem::MaybeUninit`].
3135/// In practice, this is typically equivalent to [`mem::zeroed`].
3136///
3137/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_undefined_pd)
3138#[inline]
3139#[target_feature(enable = "sse2")]
3140#[stable(feature = "simd_x86", since = "1.27.0")]
3141#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3142pub const fn _mm_undefined_pd() -> __m128d {
3143    const { unsafe { mem::zeroed() } }
3144}
3145
3146/// Returns vector of type __m128i with indeterminate elements.with indetermination elements.
3147/// Despite using the word "undefined" (following Intel's naming scheme), this non-deterministically
3148/// picks some valid value and is not equivalent to [`mem::MaybeUninit`].
3149/// In practice, this is typically equivalent to [`mem::zeroed`].
3150///
3151/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_undefined_si128)
3152#[inline]
3153#[target_feature(enable = "sse2")]
3154#[stable(feature = "simd_x86", since = "1.27.0")]
3155#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3156pub const fn _mm_undefined_si128() -> __m128i {
3157    const { unsafe { mem::zeroed() } }
3158}
3159
3160/// The resulting `__m128d` element is composed by the low-order values of
3161/// the two `__m128d` interleaved input elements, i.e.:
3162///
3163/// * The `[127:64]` bits are copied from the `[127:64]` bits of the second input
3164/// * The `[63:0]` bits are copied from the `[127:64]` bits of the first input
3165///
3166/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpackhi_pd)
3167#[inline]
3168#[target_feature(enable = "sse2")]
3169#[cfg_attr(test, assert_instr(unpckhpd))]
3170#[stable(feature = "simd_x86", since = "1.27.0")]
3171#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3172pub const fn _mm_unpackhi_pd(a: __m128d, b: __m128d) -> __m128d {
3173    unsafe { simd_shuffle!(a, b, [1, 3]) }
3174}
3175
3176/// The resulting `__m128d` element is composed by the high-order values of
3177/// the two `__m128d` interleaved input elements, i.e.:
3178///
3179/// * The `[127:64]` bits are copied from the `[63:0]` bits of the second input
3180/// * The `[63:0]` bits are copied from the `[63:0]` bits of the first input
3181///
3182/// [Intel's documentation](https://www.intel.com/content/www/us/en/docs/intrinsics-guide/index.html#text=_mm_unpacklo_pd)
3183#[inline]
3184#[target_feature(enable = "sse2")]
3185#[cfg_attr(test, assert_instr(movlhps))]
3186#[stable(feature = "simd_x86", since = "1.27.0")]
3187#[rustc_const_unstable(feature = "stdarch_const_x86", issue = "149298")]
3188pub const fn _mm_unpacklo_pd(a: __m128d, b: __m128d) -> __m128d {
3189    unsafe { simd_shuffle!(a, b, [0, 2]) }
3190}
3191
3192#[allow(improper_ctypes)]
3193unsafe extern "unadjusted" {
3194    #[link_name = "llvm.x86.sse2.pause"]
3195    fn pause();
3196    #[link_name = "llvm.x86.sse2.clflush"]
3197    fn clflush(p: *const u8);
3198    #[link_name = "llvm.x86.sse2.lfence"]
3199    fn lfence();
3200    #[link_name = "llvm.x86.sse2.mfence"]
3201    fn mfence();
3202    #[link_name = "llvm.x86.sse2.pmadd.wd"]
3203    fn pmaddwd(a: i16x8, b: i16x8) -> i32x4;
3204    #[link_name = "llvm.x86.sse2.psad.bw"]
3205    fn psadbw(a: u8x16, b: u8x16) -> u64x2;
3206    #[link_name = "llvm.x86.sse2.psll.w"]
3207    fn psllw(a: i16x8, count: i16x8) -> i16x8;
3208    #[link_name = "llvm.x86.sse2.psll.d"]
3209    fn pslld(a: i32x4, count: i32x4) -> i32x4;
3210    #[link_name = "llvm.x86.sse2.psll.q"]
3211    fn psllq(a: i64x2, count: i64x2) -> i64x2;
3212    #[link_name = "llvm.x86.sse2.psra.w"]
3213    fn psraw(a: i16x8, count: i16x8) -> i16x8;
3214    #[link_name = "llvm.x86.sse2.psra.d"]
3215    fn psrad(a: i32x4, count: i32x4) -> i32x4;
3216    #[link_name = "llvm.x86.sse2.psrl.w"]
3217    fn psrlw(a: i16x8, count: i16x8) -> i16x8;
3218    #[link_name = "llvm.x86.sse2.psrl.d"]
3219    fn psrld(a: i32x4, count: i32x4) -> i32x4;
3220    #[link_name = "llvm.x86.sse2.psrl.q"]
3221    fn psrlq(a: i64x2, count: i64x2) -> i64x2;
3222    #[link_name = "llvm.x86.sse2.cvtps2dq"]
3223    fn cvtps2dq(a: __m128) -> i32x4;
3224    #[link_name = "llvm.x86.sse2.maskmov.dqu"]
3225    fn maskmovdqu(a: i8x16, mask: i8x16, mem_addr: *mut i8);
3226    #[link_name = "llvm.x86.sse2.packsswb.128"]
3227    fn packsswb(a: i16x8, b: i16x8) -> i8x16;
3228    #[link_name = "llvm.x86.sse2.packssdw.128"]
3229    fn packssdw(a: i32x4, b: i32x4) -> i16x8;
3230    #[link_name = "llvm.x86.sse2.packuswb.128"]
3231    fn packuswb(a: i16x8, b: i16x8) -> u8x16;
3232    #[link_name = "llvm.x86.sse2.max.sd"]
3233    fn maxsd(a: __m128d, b: __m128d) -> __m128d;
3234    #[link_name = "llvm.x86.sse2.max.pd"]
3235    fn maxpd(a: __m128d, b: __m128d) -> __m128d;
3236    #[link_name = "llvm.x86.sse2.min.sd"]
3237    fn minsd(a: __m128d, b: __m128d) -> __m128d;
3238    #[link_name = "llvm.x86.sse2.min.pd"]
3239    fn minpd(a: __m128d, b: __m128d) -> __m128d;
3240    #[link_name = "llvm.x86.sse2.cmp.sd"]
3241    fn cmpsd(a: __m128d, b: __m128d, imm8: i8) -> __m128d;
3242    #[link_name = "llvm.x86.sse2.cmp.pd"]
3243    fn cmppd(a: __m128d, b: __m128d, imm8: i8) -> __m128d;
3244    #[link_name = "llvm.x86.sse2.comieq.sd"]
3245    fn comieqsd(a: __m128d, b: __m128d) -> i32;
3246    #[link_name = "llvm.x86.sse2.comilt.sd"]
3247    fn comiltsd(a: __m128d, b: __m128d) -> i32;
3248    #[link_name = "llvm.x86.sse2.comile.sd"]
3249    fn comilesd(a: __m128d, b: __m128d) -> i32;
3250    #[link_name = "llvm.x86.sse2.comigt.sd"]
3251    fn comigtsd(a: __m128d, b: __m128d) -> i32;
3252    #[link_name = "llvm.x86.sse2.comige.sd"]
3253    fn comigesd(a: __m128d, b: __m128d) -> i32;
3254    #[link_name = "llvm.x86.sse2.comineq.sd"]
3255    fn comineqsd(a: __m128d, b: __m128d) -> i32;
3256    #[link_name = "llvm.x86.sse2.ucomieq.sd"]
3257    fn ucomieqsd(a: __m128d, b: __m128d) -> i32;
3258    #[link_name = "llvm.x86.sse2.ucomilt.sd"]
3259    fn ucomiltsd(a: __m128d, b: __m128d) -> i32;
3260    #[link_name = "llvm.x86.sse2.ucomile.sd"]
3261    fn ucomilesd(a: __m128d, b: __m128d) -> i32;
3262    #[link_name = "llvm.x86.sse2.ucomigt.sd"]
3263    fn ucomigtsd(a: __m128d, b: __m128d) -> i32;
3264    #[link_name = "llvm.x86.sse2.ucomige.sd"]
3265    fn ucomigesd(a: __m128d, b: __m128d) -> i32;
3266    #[link_name = "llvm.x86.sse2.ucomineq.sd"]
3267    fn ucomineqsd(a: __m128d, b: __m128d) -> i32;
3268    #[link_name = "llvm.x86.sse2.cvtpd2dq"]
3269    fn cvtpd2dq(a: __m128d) -> i32x4;
3270    #[link_name = "llvm.x86.sse2.cvtsd2si"]
3271    fn cvtsd2si(a: __m128d) -> i32;
3272    #[link_name = "llvm.x86.sse2.cvtsd2ss"]
3273    fn cvtsd2ss(a: __m128, b: __m128d) -> __m128;
3274    #[link_name = "llvm.x86.sse2.cvttpd2dq"]
3275    fn cvttpd2dq(a: __m128d) -> i32x4;
3276    #[link_name = "llvm.x86.sse2.cvttsd2si"]
3277    fn cvttsd2si(a: __m128d) -> i32;
3278    #[link_name = "llvm.x86.sse2.cvttps2dq"]
3279    fn cvttps2dq(a: __m128) -> i32x4;
3280}
3281
3282#[cfg(test)]
3283mod tests {
3284    use crate::core_arch::assert_eq_const as assert_eq;
3285    use crate::{
3286        core_arch::{simd::*, x86::*},
3287        hint::black_box,
3288    };
3289    use std::{boxed, mem, ptr};
3290    use stdarch_test::simd_test;
3291
3292    const NAN: f64 = f64::NAN;
3293
3294    #[test]
3295    fn test_mm_pause() {
3296        _mm_pause()
3297    }
3298
3299    #[simd_test(enable = "sse2")]
3300    fn test_mm_clflush() {
3301        let x = 0_u8;
3302        unsafe {
3303            _mm_clflush(ptr::addr_of!(x));
3304        }
3305    }
3306
3307    #[simd_test(enable = "sse2")]
3308    // Miri cannot support this until it is clear how it fits in the Rust memory model
3309    #[cfg_attr(miri, ignore)]
3310    fn test_mm_lfence() {
3311        _mm_lfence();
3312    }
3313
3314    #[simd_test(enable = "sse2")]
3315    // Miri cannot support this until it is clear how it fits in the Rust memory model
3316    #[cfg_attr(miri, ignore)]
3317    fn test_mm_mfence() {
3318        _mm_mfence();
3319    }
3320
3321    #[simd_test(enable = "sse2")]
3322    const fn test_mm_add_epi8() {
3323        let a = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15);
3324        #[rustfmt::skip]
3325        let b = _mm_setr_epi8(
3326            16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31,
3327        );
3328        let r = _mm_add_epi8(a, b);
3329        #[rustfmt::skip]
3330        let e = _mm_setr_epi8(
3331            16, 18, 20, 22, 24, 26, 28, 30, 32, 34, 36, 38, 40, 42, 44, 46,
3332        );
3333        assert_eq_m128i(r, e);
3334    }
3335
3336    #[simd_test(enable = "sse2")]
3337    fn test_mm_add_epi8_overflow() {
3338        let a = _mm_set1_epi8(0x7F);
3339        let b = _mm_set1_epi8(1);
3340        let r = _mm_add_epi8(a, b);
3341        assert_eq_m128i(r, _mm_set1_epi8(-128));
3342    }
3343
3344    #[simd_test(enable = "sse2")]
3345    const fn test_mm_add_epi16() {
3346        let a = _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7);
3347        let b = _mm_setr_epi16(8, 9, 10, 11, 12, 13, 14, 15);
3348        let r = _mm_add_epi16(a, b);
3349        let e = _mm_setr_epi16(8, 10, 12, 14, 16, 18, 20, 22);
3350        assert_eq_m128i(r, e);
3351    }
3352
3353    #[simd_test(enable = "sse2")]
3354    const fn test_mm_add_epi32() {
3355        let a = _mm_setr_epi32(0, 1, 2, 3);
3356        let b = _mm_setr_epi32(4, 5, 6, 7);
3357        let r = _mm_add_epi32(a, b);
3358        let e = _mm_setr_epi32(4, 6, 8, 10);
3359        assert_eq_m128i(r, e);
3360    }
3361
3362    #[simd_test(enable = "sse2")]
3363    const fn test_mm_add_epi64() {
3364        let a = _mm_setr_epi64x(0, 1);
3365        let b = _mm_setr_epi64x(2, 3);
3366        let r = _mm_add_epi64(a, b);
3367        let e = _mm_setr_epi64x(2, 4);
3368        assert_eq_m128i(r, e);
3369    }
3370
3371    #[simd_test(enable = "sse2")]
3372    const fn test_mm_adds_epi8() {
3373        let a = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15);
3374        #[rustfmt::skip]
3375        let b = _mm_setr_epi8(
3376            16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31,
3377        );
3378        let r = _mm_adds_epi8(a, b);
3379        #[rustfmt::skip]
3380        let e = _mm_setr_epi8(
3381            16, 18, 20, 22, 24, 26, 28, 30, 32, 34, 36, 38, 40, 42, 44, 46,
3382        );
3383        assert_eq_m128i(r, e);
3384    }
3385
3386    #[simd_test(enable = "sse2")]
3387    fn test_mm_adds_epi8_saturate_positive() {
3388        let a = _mm_set1_epi8(0x7F);
3389        let b = _mm_set1_epi8(1);
3390        let r = _mm_adds_epi8(a, b);
3391        assert_eq_m128i(r, a);
3392    }
3393
3394    #[simd_test(enable = "sse2")]
3395    fn test_mm_adds_epi8_saturate_negative() {
3396        let a = _mm_set1_epi8(-0x80);
3397        let b = _mm_set1_epi8(-1);
3398        let r = _mm_adds_epi8(a, b);
3399        assert_eq_m128i(r, a);
3400    }
3401
3402    #[simd_test(enable = "sse2")]
3403    const fn test_mm_adds_epi16() {
3404        let a = _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7);
3405        let b = _mm_setr_epi16(8, 9, 10, 11, 12, 13, 14, 15);
3406        let r = _mm_adds_epi16(a, b);
3407        let e = _mm_setr_epi16(8, 10, 12, 14, 16, 18, 20, 22);
3408        assert_eq_m128i(r, e);
3409    }
3410
3411    #[simd_test(enable = "sse2")]
3412    fn test_mm_adds_epi16_saturate_positive() {
3413        let a = _mm_set1_epi16(0x7FFF);
3414        let b = _mm_set1_epi16(1);
3415        let r = _mm_adds_epi16(a, b);
3416        assert_eq_m128i(r, a);
3417    }
3418
3419    #[simd_test(enable = "sse2")]
3420    fn test_mm_adds_epi16_saturate_negative() {
3421        let a = _mm_set1_epi16(-0x8000);
3422        let b = _mm_set1_epi16(-1);
3423        let r = _mm_adds_epi16(a, b);
3424        assert_eq_m128i(r, a);
3425    }
3426
3427    #[simd_test(enable = "sse2")]
3428    const fn test_mm_adds_epu8() {
3429        let a = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15);
3430        #[rustfmt::skip]
3431        let b = _mm_setr_epi8(
3432            16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31,
3433        );
3434        let r = _mm_adds_epu8(a, b);
3435        #[rustfmt::skip]
3436        let e = _mm_setr_epi8(
3437            16, 18, 20, 22, 24, 26, 28, 30, 32, 34, 36, 38, 40, 42, 44, 46,
3438        );
3439        assert_eq_m128i(r, e);
3440    }
3441
3442    #[simd_test(enable = "sse2")]
3443    fn test_mm_adds_epu8_saturate() {
3444        let a = _mm_set1_epi8(!0);
3445        let b = _mm_set1_epi8(1);
3446        let r = _mm_adds_epu8(a, b);
3447        assert_eq_m128i(r, a);
3448    }
3449
3450    #[simd_test(enable = "sse2")]
3451    const fn test_mm_adds_epu16() {
3452        let a = _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7);
3453        let b = _mm_setr_epi16(8, 9, 10, 11, 12, 13, 14, 15);
3454        let r = _mm_adds_epu16(a, b);
3455        let e = _mm_setr_epi16(8, 10, 12, 14, 16, 18, 20, 22);
3456        assert_eq_m128i(r, e);
3457    }
3458
3459    #[simd_test(enable = "sse2")]
3460    fn test_mm_adds_epu16_saturate() {
3461        let a = _mm_set1_epi16(!0);
3462        let b = _mm_set1_epi16(1);
3463        let r = _mm_adds_epu16(a, b);
3464        assert_eq_m128i(r, a);
3465    }
3466
3467    #[simd_test(enable = "sse2")]
3468    const fn test_mm_avg_epu8() {
3469        let (a, b) = (_mm_set1_epi8(3), _mm_set1_epi8(9));
3470        let r = _mm_avg_epu8(a, b);
3471        assert_eq_m128i(r, _mm_set1_epi8(6));
3472    }
3473
3474    #[simd_test(enable = "sse2")]
3475    const fn test_mm_avg_epu16() {
3476        let (a, b) = (_mm_set1_epi16(3), _mm_set1_epi16(9));
3477        let r = _mm_avg_epu16(a, b);
3478        assert_eq_m128i(r, _mm_set1_epi16(6));
3479    }
3480
3481    #[simd_test(enable = "sse2")]
3482    fn test_mm_madd_epi16() {
3483        let a = _mm_setr_epi16(1, 2, 3, 4, 5, 6, 7, 8);
3484        let b = _mm_setr_epi16(9, 10, 11, 12, 13, 14, 15, 16);
3485        let r = _mm_madd_epi16(a, b);
3486        let e = _mm_setr_epi32(29, 81, 149, 233);
3487        assert_eq_m128i(r, e);
3488
3489        // Test large values.
3490        // MIN*MIN+MIN*MIN will overflow into i32::MIN.
3491        let a = _mm_setr_epi16(
3492            i16::MAX,
3493            i16::MAX,
3494            i16::MIN,
3495            i16::MIN,
3496            i16::MIN,
3497            i16::MAX,
3498            0,
3499            0,
3500        );
3501        let b = _mm_setr_epi16(
3502            i16::MAX,
3503            i16::MAX,
3504            i16::MIN,
3505            i16::MIN,
3506            i16::MAX,
3507            i16::MIN,
3508            0,
3509            0,
3510        );
3511        let r = _mm_madd_epi16(a, b);
3512        let e = _mm_setr_epi32(0x7FFE0002, i32::MIN, -0x7FFF0000, 0);
3513        assert_eq_m128i(r, e);
3514    }
3515
3516    #[simd_test(enable = "sse2")]
3517    const fn test_mm_max_epi16() {
3518        let a = _mm_set1_epi16(1);
3519        let b = _mm_set1_epi16(-1);
3520        let r = _mm_max_epi16(a, b);
3521        assert_eq_m128i(r, a);
3522    }
3523
3524    #[simd_test(enable = "sse2")]
3525    const fn test_mm_max_epu8() {
3526        let a = _mm_set1_epi8(1);
3527        let b = _mm_set1_epi8(!0);
3528        let r = _mm_max_epu8(a, b);
3529        assert_eq_m128i(r, b);
3530    }
3531
3532    #[simd_test(enable = "sse2")]
3533    const fn test_mm_min_epi16() {
3534        let a = _mm_set1_epi16(1);
3535        let b = _mm_set1_epi16(-1);
3536        let r = _mm_min_epi16(a, b);
3537        assert_eq_m128i(r, b);
3538    }
3539
3540    #[simd_test(enable = "sse2")]
3541    const fn test_mm_min_epu8() {
3542        let a = _mm_set1_epi8(1);
3543        let b = _mm_set1_epi8(!0);
3544        let r = _mm_min_epu8(a, b);
3545        assert_eq_m128i(r, a);
3546    }
3547
3548    #[simd_test(enable = "sse2")]
3549    const fn test_mm_mulhi_epi16() {
3550        let (a, b) = (_mm_set1_epi16(1000), _mm_set1_epi16(-1001));
3551        let r = _mm_mulhi_epi16(a, b);
3552        assert_eq_m128i(r, _mm_set1_epi16(-16));
3553    }
3554
3555    #[simd_test(enable = "sse2")]
3556    const fn test_mm_mulhi_epu16() {
3557        let (a, b) = (_mm_set1_epi16(1000), _mm_set1_epi16(1001));
3558        let r = _mm_mulhi_epu16(a, b);
3559        assert_eq_m128i(r, _mm_set1_epi16(15));
3560    }
3561
3562    #[simd_test(enable = "sse2")]
3563    const fn test_mm_mullo_epi16() {
3564        let (a, b) = (_mm_set1_epi16(1000), _mm_set1_epi16(-1001));
3565        let r = _mm_mullo_epi16(a, b);
3566        assert_eq_m128i(r, _mm_set1_epi16(-17960));
3567    }
3568
3569    #[simd_test(enable = "sse2")]
3570    const fn test_mm_mul_epu32() {
3571        let a = _mm_setr_epi64x(1_000_000_000, 1 << 34);
3572        let b = _mm_setr_epi64x(1_000_000_000, 1 << 35);
3573        let r = _mm_mul_epu32(a, b);
3574        let e = _mm_setr_epi64x(1_000_000_000 * 1_000_000_000, 0);
3575        assert_eq_m128i(r, e);
3576    }
3577
3578    #[simd_test(enable = "sse2")]
3579    fn test_mm_sad_epu8() {
3580        #[rustfmt::skip]
3581        let a = _mm_setr_epi8(
3582            255u8 as i8, 254u8 as i8, 253u8 as i8, 252u8 as i8,
3583            1, 2, 3, 4,
3584            155u8 as i8, 154u8 as i8, 153u8 as i8, 152u8 as i8,
3585            1, 2, 3, 4,
3586        );
3587        let b = _mm_setr_epi8(0, 0, 0, 0, 2, 1, 2, 1, 1, 1, 1, 1, 1, 2, 1, 2);
3588        let r = _mm_sad_epu8(a, b);
3589        let e = _mm_setr_epi64x(1020, 614);
3590        assert_eq_m128i(r, e);
3591    }
3592
3593    #[simd_test(enable = "sse2")]
3594    const fn test_mm_sub_epi8() {
3595        let (a, b) = (_mm_set1_epi8(5), _mm_set1_epi8(6));
3596        let r = _mm_sub_epi8(a, b);
3597        assert_eq_m128i(r, _mm_set1_epi8(-1));
3598    }
3599
3600    #[simd_test(enable = "sse2")]
3601    const fn test_mm_sub_epi16() {
3602        let (a, b) = (_mm_set1_epi16(5), _mm_set1_epi16(6));
3603        let r = _mm_sub_epi16(a, b);
3604        assert_eq_m128i(r, _mm_set1_epi16(-1));
3605    }
3606
3607    #[simd_test(enable = "sse2")]
3608    const fn test_mm_sub_epi32() {
3609        let (a, b) = (_mm_set1_epi32(5), _mm_set1_epi32(6));
3610        let r = _mm_sub_epi32(a, b);
3611        assert_eq_m128i(r, _mm_set1_epi32(-1));
3612    }
3613
3614    #[simd_test(enable = "sse2")]
3615    const fn test_mm_sub_epi64() {
3616        let (a, b) = (_mm_set1_epi64x(5), _mm_set1_epi64x(6));
3617        let r = _mm_sub_epi64(a, b);
3618        assert_eq_m128i(r, _mm_set1_epi64x(-1));
3619    }
3620
3621    #[simd_test(enable = "sse2")]
3622    const fn test_mm_subs_epi8() {
3623        let (a, b) = (_mm_set1_epi8(5), _mm_set1_epi8(2));
3624        let r = _mm_subs_epi8(a, b);
3625        assert_eq_m128i(r, _mm_set1_epi8(3));
3626    }
3627
3628    #[simd_test(enable = "sse2")]
3629    fn test_mm_subs_epi8_saturate_positive() {
3630        let a = _mm_set1_epi8(0x7F);
3631        let b = _mm_set1_epi8(-1);
3632        let r = _mm_subs_epi8(a, b);
3633        assert_eq_m128i(r, a);
3634    }
3635
3636    #[simd_test(enable = "sse2")]
3637    fn test_mm_subs_epi8_saturate_negative() {
3638        let a = _mm_set1_epi8(-0x80);
3639        let b = _mm_set1_epi8(1);
3640        let r = _mm_subs_epi8(a, b);
3641        assert_eq_m128i(r, a);
3642    }
3643
3644    #[simd_test(enable = "sse2")]
3645    const fn test_mm_subs_epi16() {
3646        let (a, b) = (_mm_set1_epi16(5), _mm_set1_epi16(2));
3647        let r = _mm_subs_epi16(a, b);
3648        assert_eq_m128i(r, _mm_set1_epi16(3));
3649    }
3650
3651    #[simd_test(enable = "sse2")]
3652    fn test_mm_subs_epi16_saturate_positive() {
3653        let a = _mm_set1_epi16(0x7FFF);
3654        let b = _mm_set1_epi16(-1);
3655        let r = _mm_subs_epi16(a, b);
3656        assert_eq_m128i(r, a);
3657    }
3658
3659    #[simd_test(enable = "sse2")]
3660    fn test_mm_subs_epi16_saturate_negative() {
3661        let a = _mm_set1_epi16(-0x8000);
3662        let b = _mm_set1_epi16(1);
3663        let r = _mm_subs_epi16(a, b);
3664        assert_eq_m128i(r, a);
3665    }
3666
3667    #[simd_test(enable = "sse2")]
3668    const fn test_mm_subs_epu8() {
3669        let (a, b) = (_mm_set1_epi8(5), _mm_set1_epi8(2));
3670        let r = _mm_subs_epu8(a, b);
3671        assert_eq_m128i(r, _mm_set1_epi8(3));
3672    }
3673
3674    #[simd_test(enable = "sse2")]
3675    fn test_mm_subs_epu8_saturate() {
3676        let a = _mm_set1_epi8(0);
3677        let b = _mm_set1_epi8(1);
3678        let r = _mm_subs_epu8(a, b);
3679        assert_eq_m128i(r, a);
3680    }
3681
3682    #[simd_test(enable = "sse2")]
3683    const fn test_mm_subs_epu16() {
3684        let (a, b) = (_mm_set1_epi16(5), _mm_set1_epi16(2));
3685        let r = _mm_subs_epu16(a, b);
3686        assert_eq_m128i(r, _mm_set1_epi16(3));
3687    }
3688
3689    #[simd_test(enable = "sse2")]
3690    fn test_mm_subs_epu16_saturate() {
3691        let a = _mm_set1_epi16(0);
3692        let b = _mm_set1_epi16(1);
3693        let r = _mm_subs_epu16(a, b);
3694        assert_eq_m128i(r, a);
3695    }
3696
3697    #[simd_test(enable = "sse2")]
3698    const fn test_mm_slli_si128() {
3699        #[rustfmt::skip]
3700        let a = _mm_setr_epi8(
3701            1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16,
3702        );
3703        let r = _mm_slli_si128::<1>(a);
3704        let e = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15);
3705        assert_eq_m128i(r, e);
3706
3707        #[rustfmt::skip]
3708        let a = _mm_setr_epi8(
3709            1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16,
3710        );
3711        let r = _mm_slli_si128::<15>(a);
3712        let e = _mm_setr_epi8(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1);
3713        assert_eq_m128i(r, e);
3714
3715        #[rustfmt::skip]
3716        let a = _mm_setr_epi8(
3717            1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16,
3718        );
3719        let r = _mm_slli_si128::<16>(a);
3720        assert_eq_m128i(r, _mm_set1_epi8(0));
3721    }
3722
3723    #[simd_test(enable = "sse2")]
3724    const fn test_mm_slli_epi16() {
3725        let a = _mm_setr_epi16(0xCC, -0xCC, 0xDD, -0xDD, 0xEE, -0xEE, 0xFF, -0xFF);
3726        let r = _mm_slli_epi16::<4>(a);
3727        assert_eq_m128i(
3728            r,
3729            _mm_setr_epi16(0xCC0, -0xCC0, 0xDD0, -0xDD0, 0xEE0, -0xEE0, 0xFF0, -0xFF0),
3730        );
3731        let r = _mm_slli_epi16::<16>(a);
3732        assert_eq_m128i(r, _mm_set1_epi16(0));
3733    }
3734
3735    #[simd_test(enable = "sse2")]
3736    fn test_mm_sll_epi16() {
3737        let a = _mm_setr_epi16(0xCC, -0xCC, 0xDD, -0xDD, 0xEE, -0xEE, 0xFF, -0xFF);
3738        let r = _mm_sll_epi16(a, _mm_set_epi64x(0, 4));
3739        assert_eq_m128i(
3740            r,
3741            _mm_setr_epi16(0xCC0, -0xCC0, 0xDD0, -0xDD0, 0xEE0, -0xEE0, 0xFF0, -0xFF0),
3742        );
3743        let r = _mm_sll_epi16(a, _mm_set_epi64x(4, 0));
3744        assert_eq_m128i(r, a);
3745        let r = _mm_sll_epi16(a, _mm_set_epi64x(0, 16));
3746        assert_eq_m128i(r, _mm_set1_epi16(0));
3747        let r = _mm_sll_epi16(a, _mm_set_epi64x(0, i64::MAX));
3748        assert_eq_m128i(r, _mm_set1_epi16(0));
3749    }
3750
3751    #[simd_test(enable = "sse2")]
3752    const fn test_mm_slli_epi32() {
3753        let a = _mm_setr_epi32(0xEEEE, -0xEEEE, 0xFFFF, -0xFFFF);
3754        let r = _mm_slli_epi32::<4>(a);
3755        assert_eq_m128i(r, _mm_setr_epi32(0xEEEE0, -0xEEEE0, 0xFFFF0, -0xFFFF0));
3756        let r = _mm_slli_epi32::<32>(a);
3757        assert_eq_m128i(r, _mm_set1_epi32(0));
3758    }
3759
3760    #[simd_test(enable = "sse2")]
3761    fn test_mm_sll_epi32() {
3762        let a = _mm_setr_epi32(0xEEEE, -0xEEEE, 0xFFFF, -0xFFFF);
3763        let r = _mm_sll_epi32(a, _mm_set_epi64x(0, 4));
3764        assert_eq_m128i(r, _mm_setr_epi32(0xEEEE0, -0xEEEE0, 0xFFFF0, -0xFFFF0));
3765        let r = _mm_sll_epi32(a, _mm_set_epi64x(4, 0));
3766        assert_eq_m128i(r, a);
3767        let r = _mm_sll_epi32(a, _mm_set_epi64x(0, 32));
3768        assert_eq_m128i(r, _mm_set1_epi32(0));
3769        let r = _mm_sll_epi32(a, _mm_set_epi64x(0, i64::MAX));
3770        assert_eq_m128i(r, _mm_set1_epi32(0));
3771    }
3772
3773    #[simd_test(enable = "sse2")]
3774    const fn test_mm_slli_epi64() {
3775        let a = _mm_set_epi64x(0xFFFFFFFF, -0xFFFFFFFF);
3776        let r = _mm_slli_epi64::<4>(a);
3777        assert_eq_m128i(r, _mm_set_epi64x(0xFFFFFFFF0, -0xFFFFFFFF0));
3778        let r = _mm_slli_epi64::<64>(a);
3779        assert_eq_m128i(r, _mm_set1_epi64x(0));
3780    }
3781
3782    #[simd_test(enable = "sse2")]
3783    fn test_mm_sll_epi64() {
3784        let a = _mm_set_epi64x(0xFFFFFFFF, -0xFFFFFFFF);
3785        let r = _mm_sll_epi64(a, _mm_set_epi64x(0, 4));
3786        assert_eq_m128i(r, _mm_set_epi64x(0xFFFFFFFF0, -0xFFFFFFFF0));
3787        let r = _mm_sll_epi64(a, _mm_set_epi64x(4, 0));
3788        assert_eq_m128i(r, a);
3789        let r = _mm_sll_epi64(a, _mm_set_epi64x(0, 64));
3790        assert_eq_m128i(r, _mm_set1_epi64x(0));
3791        let r = _mm_sll_epi64(a, _mm_set_epi64x(0, i64::MAX));
3792        assert_eq_m128i(r, _mm_set1_epi64x(0));
3793    }
3794
3795    #[simd_test(enable = "sse2")]
3796    const fn test_mm_srai_epi16() {
3797        let a = _mm_setr_epi16(0xCC, -0xCC, 0xDD, -0xDD, 0xEE, -0xEE, 0xFF, -0xFF);
3798        let r = _mm_srai_epi16::<4>(a);
3799        assert_eq_m128i(
3800            r,
3801            _mm_setr_epi16(0xC, -0xD, 0xD, -0xE, 0xE, -0xF, 0xF, -0x10),
3802        );
3803        let r = _mm_srai_epi16::<16>(a);
3804        assert_eq_m128i(r, _mm_setr_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3805    }
3806
3807    #[simd_test(enable = "sse2")]
3808    fn test_mm_sra_epi16() {
3809        let a = _mm_setr_epi16(0xCC, -0xCC, 0xDD, -0xDD, 0xEE, -0xEE, 0xFF, -0xFF);
3810        let r = _mm_sra_epi16(a, _mm_set_epi64x(0, 4));
3811        assert_eq_m128i(
3812            r,
3813            _mm_setr_epi16(0xC, -0xD, 0xD, -0xE, 0xE, -0xF, 0xF, -0x10),
3814        );
3815        let r = _mm_sra_epi16(a, _mm_set_epi64x(4, 0));
3816        assert_eq_m128i(r, a);
3817        let r = _mm_sra_epi16(a, _mm_set_epi64x(0, 16));
3818        assert_eq_m128i(r, _mm_setr_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3819        let r = _mm_sra_epi16(a, _mm_set_epi64x(0, i64::MAX));
3820        assert_eq_m128i(r, _mm_setr_epi16(0, -1, 0, -1, 0, -1, 0, -1));
3821    }
3822
3823    #[simd_test(enable = "sse2")]
3824    const fn test_mm_srai_epi32() {
3825        let a = _mm_setr_epi32(0xEEEE, -0xEEEE, 0xFFFF, -0xFFFF);
3826        let r = _mm_srai_epi32::<4>(a);
3827        assert_eq_m128i(r, _mm_setr_epi32(0xEEE, -0xEEF, 0xFFF, -0x1000));
3828        let r = _mm_srai_epi32::<32>(a);
3829        assert_eq_m128i(r, _mm_setr_epi32(0, -1, 0, -1));
3830    }
3831
3832    #[simd_test(enable = "sse2")]
3833    fn test_mm_sra_epi32() {
3834        let a = _mm_setr_epi32(0xEEEE, -0xEEEE, 0xFFFF, -0xFFFF);
3835        let r = _mm_sra_epi32(a, _mm_set_epi64x(0, 4));
3836        assert_eq_m128i(r, _mm_setr_epi32(0xEEE, -0xEEF, 0xFFF, -0x1000));
3837        let r = _mm_sra_epi32(a, _mm_set_epi64x(4, 0));
3838        assert_eq_m128i(r, a);
3839        let r = _mm_sra_epi32(a, _mm_set_epi64x(0, 32));
3840        assert_eq_m128i(r, _mm_setr_epi32(0, -1, 0, -1));
3841        let r = _mm_sra_epi32(a, _mm_set_epi64x(0, i64::MAX));
3842        assert_eq_m128i(r, _mm_setr_epi32(0, -1, 0, -1));
3843    }
3844
3845    #[simd_test(enable = "sse2")]
3846    const fn test_mm_srli_si128() {
3847        #[rustfmt::skip]
3848        let a = _mm_setr_epi8(
3849            1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16,
3850        );
3851        let r = _mm_srli_si128::<1>(a);
3852        #[rustfmt::skip]
3853        let e = _mm_setr_epi8(
3854            2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 0,
3855        );
3856        assert_eq_m128i(r, e);
3857
3858        #[rustfmt::skip]
3859        let a = _mm_setr_epi8(
3860            1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16,
3861        );
3862        let r = _mm_srli_si128::<15>(a);
3863        let e = _mm_setr_epi8(16, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
3864        assert_eq_m128i(r, e);
3865
3866        #[rustfmt::skip]
3867        let a = _mm_setr_epi8(
3868            1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16,
3869        );
3870        let r = _mm_srli_si128::<16>(a);
3871        assert_eq_m128i(r, _mm_set1_epi8(0));
3872    }
3873
3874    #[simd_test(enable = "sse2")]
3875    const fn test_mm_srli_epi16() {
3876        let a = _mm_setr_epi16(0xCC, -0xCC, 0xDD, -0xDD, 0xEE, -0xEE, 0xFF, -0xFF);
3877        let r = _mm_srli_epi16::<4>(a);
3878        assert_eq_m128i(
3879            r,
3880            _mm_setr_epi16(0xC, 0xFF3, 0xD, 0xFF2, 0xE, 0xFF1, 0xF, 0xFF0),
3881        );
3882        let r = _mm_srli_epi16::<16>(a);
3883        assert_eq_m128i(r, _mm_set1_epi16(0));
3884    }
3885
3886    #[simd_test(enable = "sse2")]
3887    fn test_mm_srl_epi16() {
3888        let a = _mm_setr_epi16(0xCC, -0xCC, 0xDD, -0xDD, 0xEE, -0xEE, 0xFF, -0xFF);
3889        let r = _mm_srl_epi16(a, _mm_set_epi64x(0, 4));
3890        assert_eq_m128i(
3891            r,
3892            _mm_setr_epi16(0xC, 0xFF3, 0xD, 0xFF2, 0xE, 0xFF1, 0xF, 0xFF0),
3893        );
3894        let r = _mm_srl_epi16(a, _mm_set_epi64x(4, 0));
3895        assert_eq_m128i(r, a);
3896        let r = _mm_srl_epi16(a, _mm_set_epi64x(0, 16));
3897        assert_eq_m128i(r, _mm_set1_epi16(0));
3898        let r = _mm_srl_epi16(a, _mm_set_epi64x(0, i64::MAX));
3899        assert_eq_m128i(r, _mm_set1_epi16(0));
3900    }
3901
3902    #[simd_test(enable = "sse2")]
3903    const fn test_mm_srli_epi32() {
3904        let a = _mm_setr_epi32(0xEEEE, -0xEEEE, 0xFFFF, -0xFFFF);
3905        let r = _mm_srli_epi32::<4>(a);
3906        assert_eq_m128i(r, _mm_setr_epi32(0xEEE, 0xFFFF111, 0xFFF, 0xFFFF000));
3907        let r = _mm_srli_epi32::<32>(a);
3908        assert_eq_m128i(r, _mm_set1_epi32(0));
3909    }
3910
3911    #[simd_test(enable = "sse2")]
3912    fn test_mm_srl_epi32() {
3913        let a = _mm_setr_epi32(0xEEEE, -0xEEEE, 0xFFFF, -0xFFFF);
3914        let r = _mm_srl_epi32(a, _mm_set_epi64x(0, 4));
3915        assert_eq_m128i(r, _mm_setr_epi32(0xEEE, 0xFFFF111, 0xFFF, 0xFFFF000));
3916        let r = _mm_srl_epi32(a, _mm_set_epi64x(4, 0));
3917        assert_eq_m128i(r, a);
3918        let r = _mm_srl_epi32(a, _mm_set_epi64x(0, 32));
3919        assert_eq_m128i(r, _mm_set1_epi32(0));
3920        let r = _mm_srl_epi32(a, _mm_set_epi64x(0, i64::MAX));
3921        assert_eq_m128i(r, _mm_set1_epi32(0));
3922    }
3923
3924    #[simd_test(enable = "sse2")]
3925    const fn test_mm_srli_epi64() {
3926        let a = _mm_set_epi64x(0xFFFFFFFF, -0xFFFFFFFF);
3927        let r = _mm_srli_epi64::<4>(a);
3928        assert_eq_m128i(r, _mm_set_epi64x(0xFFFFFFF, 0xFFFFFFFF0000000));
3929        let r = _mm_srli_epi64::<64>(a);
3930        assert_eq_m128i(r, _mm_set1_epi64x(0));
3931    }
3932
3933    #[simd_test(enable = "sse2")]
3934    fn test_mm_srl_epi64() {
3935        let a = _mm_set_epi64x(0xFFFFFFFF, -0xFFFFFFFF);
3936        let r = _mm_srl_epi64(a, _mm_set_epi64x(0, 4));
3937        assert_eq_m128i(r, _mm_set_epi64x(0xFFFFFFF, 0xFFFFFFFF0000000));
3938        let r = _mm_srl_epi64(a, _mm_set_epi64x(4, 0));
3939        assert_eq_m128i(r, a);
3940        let r = _mm_srl_epi64(a, _mm_set_epi64x(0, 64));
3941        assert_eq_m128i(r, _mm_set1_epi64x(0));
3942        let r = _mm_srl_epi64(a, _mm_set_epi64x(0, i64::MAX));
3943        assert_eq_m128i(r, _mm_set1_epi64x(0));
3944    }
3945
3946    #[simd_test(enable = "sse2")]
3947    const fn test_mm_and_si128() {
3948        let a = _mm_set1_epi8(5);
3949        let b = _mm_set1_epi8(3);
3950        let r = _mm_and_si128(a, b);
3951        assert_eq_m128i(r, _mm_set1_epi8(1));
3952    }
3953
3954    #[simd_test(enable = "sse2")]
3955    const fn test_mm_andnot_si128() {
3956        let a = _mm_set1_epi8(5);
3957        let b = _mm_set1_epi8(3);
3958        let r = _mm_andnot_si128(a, b);
3959        assert_eq_m128i(r, _mm_set1_epi8(2));
3960    }
3961
3962    #[simd_test(enable = "sse2")]
3963    const fn test_mm_or_si128() {
3964        let a = _mm_set1_epi8(5);
3965        let b = _mm_set1_epi8(3);
3966        let r = _mm_or_si128(a, b);
3967        assert_eq_m128i(r, _mm_set1_epi8(7));
3968    }
3969
3970    #[simd_test(enable = "sse2")]
3971    const fn test_mm_xor_si128() {
3972        let a = _mm_set1_epi8(5);
3973        let b = _mm_set1_epi8(3);
3974        let r = _mm_xor_si128(a, b);
3975        assert_eq_m128i(r, _mm_set1_epi8(6));
3976    }
3977
3978    #[simd_test(enable = "sse2")]
3979    const fn test_mm_cmpeq_epi8() {
3980        let a = _mm_setr_epi8(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15);
3981        let b = _mm_setr_epi8(15, 14, 2, 12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1, 0);
3982        let r = _mm_cmpeq_epi8(a, b);
3983        #[rustfmt::skip]
3984        assert_eq_m128i(
3985            r,
3986            _mm_setr_epi8(
3987                0, 0, 0xFFu8 as i8, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0
3988            )
3989        );
3990    }
3991
3992    #[simd_test(enable = "sse2")]
3993    const fn test_mm_cmpeq_epi16() {
3994        let a = _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7);
3995        let b = _mm_setr_epi16(7, 6, 2, 4, 3, 2, 1, 0);
3996        let r = _mm_cmpeq_epi16(a, b);
3997        assert_eq_m128i(r, _mm_setr_epi16(0, 0, !0, 0, 0, 0, 0, 0));
3998    }
3999
4000    #[simd_test(enable = "sse2")]
4001    const fn test_mm_cmpeq_epi32() {
4002        let a = _mm_setr_epi32(0, 1, 2, 3);
4003        let b = _mm_setr_epi32(3, 2, 2, 0);
4004        let r = _mm_cmpeq_epi32(a, b);
4005        assert_eq_m128i(r, _mm_setr_epi32(0, 0, !0, 0));
4006    }
4007
4008    #[simd_test(enable = "sse2")]
4009    const fn test_mm_cmpgt_epi8() {
4010        let a = _mm_set_epi8(5, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
4011        let b = _mm_set1_epi8(0);
4012        let r = _mm_cmpgt_epi8(a, b);
4013        let e = _mm_set_epi8(!0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
4014        assert_eq_m128i(r, e);
4015    }
4016
4017    #[simd_test(enable = "sse2")]
4018    const fn test_mm_cmpgt_epi16() {
4019        let a = _mm_set_epi16(5, 0, 0, 0, 0, 0, 0, 0);
4020        let b = _mm_set1_epi16(0);
4021        let r = _mm_cmpgt_epi16(a, b);
4022        let e = _mm_set_epi16(!0, 0, 0, 0, 0, 0, 0, 0);
4023        assert_eq_m128i(r, e);
4024    }
4025
4026    #[simd_test(enable = "sse2")]
4027    const fn test_mm_cmpgt_epi32() {
4028        let a = _mm_set_epi32(5, 0, 0, 0);
4029        let b = _mm_set1_epi32(0);
4030        let r = _mm_cmpgt_epi32(a, b);
4031        assert_eq_m128i(r, _mm_set_epi32(!0, 0, 0, 0));
4032    }
4033
4034    #[simd_test(enable = "sse2")]
4035    const fn test_mm_cmplt_epi8() {
4036        let a = _mm_set1_epi8(0);
4037        let b = _mm_set_epi8(5, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
4038        let r = _mm_cmplt_epi8(a, b);
4039        let e = _mm_set_epi8(!0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
4040        assert_eq_m128i(r, e);
4041    }
4042
4043    #[simd_test(enable = "sse2")]
4044    const fn test_mm_cmplt_epi16() {
4045        let a = _mm_set1_epi16(0);
4046        let b = _mm_set_epi16(5, 0, 0, 0, 0, 0, 0, 0);
4047        let r = _mm_cmplt_epi16(a, b);
4048        let e = _mm_set_epi16(!0, 0, 0, 0, 0, 0, 0, 0);
4049        assert_eq_m128i(r, e);
4050    }
4051
4052    #[simd_test(enable = "sse2")]
4053    const fn test_mm_cmplt_epi32() {
4054        let a = _mm_set1_epi32(0);
4055        let b = _mm_set_epi32(5, 0, 0, 0);
4056        let r = _mm_cmplt_epi32(a, b);
4057        assert_eq_m128i(r, _mm_set_epi32(!0, 0, 0, 0));
4058    }
4059
4060    #[simd_test(enable = "sse2")]
4061    const fn test_mm_cvtepi32_pd() {
4062        let a = _mm_set_epi32(35, 25, 15, 5);
4063        let r = _mm_cvtepi32_pd(a);
4064        assert_eq_m128d(r, _mm_setr_pd(5.0, 15.0));
4065    }
4066
4067    #[simd_test(enable = "sse2")]
4068    const fn test_mm_cvtsi32_sd() {
4069        let a = _mm_set1_pd(3.5);
4070        let r = _mm_cvtsi32_sd(a, 5);
4071        assert_eq_m128d(r, _mm_setr_pd(5.0, 3.5));
4072    }
4073
4074    #[simd_test(enable = "sse2")]
4075    const fn test_mm_cvtepi32_ps() {
4076        let a = _mm_setr_epi32(1, 2, 3, 4);
4077        let r = _mm_cvtepi32_ps(a);
4078        assert_eq_m128(r, _mm_setr_ps(1.0, 2.0, 3.0, 4.0));
4079    }
4080
4081    #[simd_test(enable = "sse2")]
4082    fn test_mm_cvtps_epi32() {
4083        let a = _mm_setr_ps(1.0, 2.0, 3.0, 4.0);
4084        let r = _mm_cvtps_epi32(a);
4085        assert_eq_m128i(r, _mm_setr_epi32(1, 2, 3, 4));
4086    }
4087
4088    #[simd_test(enable = "sse2")]
4089    const fn test_mm_cvtsi32_si128() {
4090        let r = _mm_cvtsi32_si128(5);
4091        assert_eq_m128i(r, _mm_setr_epi32(5, 0, 0, 0));
4092    }
4093
4094    #[simd_test(enable = "sse2")]
4095    const fn test_mm_cvtsi128_si32() {
4096        let r = _mm_cvtsi128_si32(_mm_setr_epi32(5, 0, 0, 0));
4097        assert_eq!(r, 5);
4098    }
4099
4100    #[simd_test(enable = "sse2")]
4101    const fn test_mm_set_epi64x() {
4102        let r = _mm_set_epi64x(0, 1);
4103        assert_eq_m128i(r, _mm_setr_epi64x(1, 0));
4104    }
4105
4106    #[simd_test(enable = "sse2")]
4107    const fn test_mm_set_epi32() {
4108        let r = _mm_set_epi32(0, 1, 2, 3);
4109        assert_eq_m128i(r, _mm_setr_epi32(3, 2, 1, 0));
4110    }
4111
4112    #[simd_test(enable = "sse2")]
4113    const fn test_mm_set_epi16() {
4114        let r = _mm_set_epi16(0, 1, 2, 3, 4, 5, 6, 7);
4115        assert_eq_m128i(r, _mm_setr_epi16(7, 6, 5, 4, 3, 2, 1, 0));
4116    }
4117
4118    #[simd_test(enable = "sse2")]
4119    const fn test_mm_set_epi8() {
4120        #[rustfmt::skip]
4121        let r = _mm_set_epi8(
4122            0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
4123        );
4124        #[rustfmt::skip]
4125        let e = _mm_setr_epi8(
4126            15, 14, 13, 12, 11, 10, 9, 8,
4127            7, 6, 5, 4, 3, 2, 1, 0,
4128        );
4129        assert_eq_m128i(r, e);
4130    }
4131
4132    #[simd_test(enable = "sse2")]
4133    const fn test_mm_set1_epi64x() {
4134        let r = _mm_set1_epi64x(1);
4135        assert_eq_m128i(r, _mm_set1_epi64x(1));
4136    }
4137
4138    #[simd_test(enable = "sse2")]
4139    const fn test_mm_set1_epi32() {
4140        let r = _mm_set1_epi32(1);
4141        assert_eq_m128i(r, _mm_set1_epi32(1));
4142    }
4143
4144    #[simd_test(enable = "sse2")]
4145    const fn test_mm_set1_epi16() {
4146        let r = _mm_set1_epi16(1);
4147        assert_eq_m128i(r, _mm_set1_epi16(1));
4148    }
4149
4150    #[simd_test(enable = "sse2")]
4151    const fn test_mm_set1_epi8() {
4152        let r = _mm_set1_epi8(1);
4153        assert_eq_m128i(r, _mm_set1_epi8(1));
4154    }
4155
4156    #[simd_test(enable = "sse2")]
4157    const fn test_mm_setr_epi32() {
4158        let r = _mm_setr_epi32(0, 1, 2, 3);
4159        assert_eq_m128i(r, _mm_setr_epi32(0, 1, 2, 3));
4160    }
4161
4162    #[simd_test(enable = "sse2")]
4163    const fn test_mm_setr_epi16() {
4164        let r = _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7);
4165        assert_eq_m128i(r, _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7));
4166    }
4167
4168    #[simd_test(enable = "sse2")]
4169    const fn test_mm_setr_epi8() {
4170        #[rustfmt::skip]
4171        let r = _mm_setr_epi8(
4172            0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
4173        );
4174        #[rustfmt::skip]
4175        let e = _mm_setr_epi8(
4176            0, 1, 2, 3, 4, 5, 6, 7,
4177            8, 9, 10, 11, 12, 13, 14, 15,
4178        );
4179        assert_eq_m128i(r, e);
4180    }
4181
4182    #[simd_test(enable = "sse2")]
4183    const fn test_mm_setzero_si128() {
4184        let r = _mm_setzero_si128();
4185        assert_eq_m128i(r, _mm_set1_epi64x(0));
4186    }
4187
4188    #[simd_test(enable = "sse2")]
4189    const fn test_mm_loadl_epi64() {
4190        let a = _mm_setr_epi64x(6, 5);
4191        let r = unsafe { _mm_loadl_epi64(ptr::addr_of!(a)) };
4192        assert_eq_m128i(r, _mm_setr_epi64x(6, 0));
4193    }
4194
4195    #[simd_test(enable = "sse2")]
4196    const fn test_mm_load_si128() {
4197        let a = _mm_set_epi64x(5, 6);
4198        let r = unsafe { _mm_load_si128(ptr::addr_of!(a) as *const _) };
4199        assert_eq_m128i(a, r);
4200    }
4201
4202    #[simd_test(enable = "sse2")]
4203    const fn test_mm_loadu_si128() {
4204        let a = _mm_set_epi64x(5, 6);
4205        let r = unsafe { _mm_loadu_si128(ptr::addr_of!(a) as *const _) };
4206        assert_eq_m128i(a, r);
4207    }
4208
4209    #[simd_test(enable = "sse2")]
4210    // Miri cannot support this until it is clear how it fits in the Rust memory model
4211    // (non-temporal store)
4212    #[cfg_attr(miri, ignore)]
4213    fn test_mm_maskmoveu_si128() {
4214        let a = _mm_set1_epi8(9);
4215        #[rustfmt::skip]
4216        let mask = _mm_set_epi8(
4217            0, 0, 0x80u8 as i8, 0, 0, 0, 0, 0,
4218            0, 0, 0, 0, 0, 0, 0, 0,
4219        );
4220        let mut r = _mm_set1_epi8(0);
4221        unsafe {
4222            _mm_maskmoveu_si128(a, mask, ptr::addr_of_mut!(r) as *mut i8);
4223        }
4224        _mm_sfence();
4225        let e = _mm_set_epi8(0, 0, 9, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0);
4226        assert_eq_m128i(r, e);
4227    }
4228
4229    #[simd_test(enable = "sse2")]
4230    const fn test_mm_store_si128() {
4231        let a = _mm_set1_epi8(9);
4232        let mut r = _mm_set1_epi8(0);
4233        unsafe {
4234            _mm_store_si128(&mut r, a);
4235        }
4236        assert_eq_m128i(r, a);
4237    }
4238
4239    #[simd_test(enable = "sse2")]
4240    const fn test_mm_storeu_si128() {
4241        let a = _mm_set1_epi8(9);
4242        let mut r = _mm_set1_epi8(0);
4243        unsafe {
4244            _mm_storeu_si128(&mut r, a);
4245        }
4246        assert_eq_m128i(r, a);
4247    }
4248
4249    #[simd_test(enable = "sse2")]
4250    const fn test_mm_storel_epi64() {
4251        let a = _mm_setr_epi64x(2, 9);
4252        let mut r = _mm_set1_epi8(0);
4253        unsafe {
4254            _mm_storel_epi64(&mut r, a);
4255        }
4256        assert_eq_m128i(r, _mm_setr_epi64x(2, 0));
4257    }
4258
4259    #[simd_test(enable = "sse2")]
4260    // Miri cannot support this until it is clear how it fits in the Rust memory model
4261    // (non-temporal store)
4262    #[cfg_attr(miri, ignore)]
4263    fn test_mm_stream_si128() {
4264        let a = _mm_setr_epi32(1, 2, 3, 4);
4265        let mut r = _mm_undefined_si128();
4266        unsafe {
4267            _mm_stream_si128(ptr::addr_of_mut!(r), a);
4268        }
4269        _mm_sfence();
4270        assert_eq_m128i(r, a);
4271    }
4272
4273    #[simd_test(enable = "sse2")]
4274    // Miri cannot support this until it is clear how it fits in the Rust memory model
4275    // (non-temporal store)
4276    #[cfg_attr(miri, ignore)]
4277    fn test_mm_stream_si32() {
4278        let a: i32 = 7;
4279        let mut mem = boxed::Box::<i32>::new(-1);
4280        unsafe {
4281            _mm_stream_si32(ptr::addr_of_mut!(*mem), a);
4282        }
4283        _mm_sfence();
4284        assert_eq!(a, *mem);
4285    }
4286
4287    #[simd_test(enable = "sse2")]
4288    const fn test_mm_move_epi64() {
4289        let a = _mm_setr_epi64x(5, 6);
4290        let r = _mm_move_epi64(a);
4291        assert_eq_m128i(r, _mm_setr_epi64x(5, 0));
4292    }
4293
4294    #[simd_test(enable = "sse2")]
4295    fn test_mm_packs_epi16() {
4296        let a = _mm_setr_epi16(0x80, -0x81, 0, 0, 0, 0, 0, 0);
4297        let b = _mm_setr_epi16(0, 0, 0, 0, 0, 0, -0x81, 0x80);
4298        let r = _mm_packs_epi16(a, b);
4299        #[rustfmt::skip]
4300        assert_eq_m128i(
4301            r,
4302            _mm_setr_epi8(
4303                0x7F, -0x80, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, -0x80, 0x7F
4304            )
4305        );
4306    }
4307
4308    #[simd_test(enable = "sse2")]
4309    fn test_mm_packs_epi32() {
4310        let a = _mm_setr_epi32(0x8000, -0x8001, 0, 0);
4311        let b = _mm_setr_epi32(0, 0, -0x8001, 0x8000);
4312        let r = _mm_packs_epi32(a, b);
4313        assert_eq_m128i(
4314            r,
4315            _mm_setr_epi16(0x7FFF, -0x8000, 0, 0, 0, 0, -0x8000, 0x7FFF),
4316        );
4317    }
4318
4319    #[simd_test(enable = "sse2")]
4320    fn test_mm_packus_epi16() {
4321        let a = _mm_setr_epi16(0x100, -1, 0, 0, 0, 0, 0, 0);
4322        let b = _mm_setr_epi16(0, 0, 0, 0, 0, 0, -1, 0x100);
4323        let r = _mm_packus_epi16(a, b);
4324        assert_eq_m128i(
4325            r,
4326            _mm_setr_epi8(!0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, !0),
4327        );
4328    }
4329
4330    #[simd_test(enable = "sse2")]
4331    const fn test_mm_extract_epi16() {
4332        let a = _mm_setr_epi16(-1, 1, 2, 3, 4, 5, 6, 7);
4333        let r1 = _mm_extract_epi16::<0>(a);
4334        let r2 = _mm_extract_epi16::<3>(a);
4335        assert_eq!(r1, 0xFFFF);
4336        assert_eq!(r2, 3);
4337    }
4338
4339    #[simd_test(enable = "sse2")]
4340    const fn test_mm_insert_epi16() {
4341        let a = _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7);
4342        let r = _mm_insert_epi16::<0>(a, 9);
4343        let e = _mm_setr_epi16(9, 1, 2, 3, 4, 5, 6, 7);
4344        assert_eq_m128i(r, e);
4345    }
4346
4347    #[simd_test(enable = "sse2")]
4348    const fn test_mm_movemask_epi8() {
4349        #[rustfmt::skip]
4350        let a = _mm_setr_epi8(
4351            0b1000_0000u8 as i8, 0b0, 0b1000_0000u8 as i8, 0b01,
4352            0b0101, 0b1111_0000u8 as i8, 0, 0,
4353            0, 0b1011_0101u8 as i8, 0b1111_0000u8 as i8, 0b0101,
4354            0b01, 0b1000_0000u8 as i8, 0b0, 0b1000_0000u8 as i8,
4355        );
4356        let r = _mm_movemask_epi8(a);
4357        assert_eq!(r, 0b10100110_00100101);
4358    }
4359
4360    #[simd_test(enable = "sse2")]
4361    const fn test_mm_shuffle_epi32() {
4362        let a = _mm_setr_epi32(5, 10, 15, 20);
4363        let r = _mm_shuffle_epi32::<0b00_01_01_11>(a);
4364        let e = _mm_setr_epi32(20, 10, 10, 5);
4365        assert_eq_m128i(r, e);
4366    }
4367
4368    #[simd_test(enable = "sse2")]
4369    const fn test_mm_shufflehi_epi16() {
4370        let a = _mm_setr_epi16(1, 2, 3, 4, 5, 10, 15, 20);
4371        let r = _mm_shufflehi_epi16::<0b00_01_01_11>(a);
4372        let e = _mm_setr_epi16(1, 2, 3, 4, 20, 10, 10, 5);
4373        assert_eq_m128i(r, e);
4374    }
4375
4376    #[simd_test(enable = "sse2")]
4377    const fn test_mm_shufflelo_epi16() {
4378        let a = _mm_setr_epi16(5, 10, 15, 20, 1, 2, 3, 4);
4379        let r = _mm_shufflelo_epi16::<0b00_01_01_11>(a);
4380        let e = _mm_setr_epi16(20, 10, 10, 5, 1, 2, 3, 4);
4381        assert_eq_m128i(r, e);
4382    }
4383
4384    #[simd_test(enable = "sse2")]
4385    const fn test_mm_unpackhi_epi8() {
4386        #[rustfmt::skip]
4387        let a = _mm_setr_epi8(
4388            0, 1, 2, 3, 4, 5, 6, 7,
4389            8, 9, 10, 11, 12, 13, 14, 15,
4390        );
4391        #[rustfmt::skip]
4392        let b = _mm_setr_epi8(
4393            16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31,
4394        );
4395        let r = _mm_unpackhi_epi8(a, b);
4396        #[rustfmt::skip]
4397        let e = _mm_setr_epi8(
4398            8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31,
4399        );
4400        assert_eq_m128i(r, e);
4401    }
4402
4403    #[simd_test(enable = "sse2")]
4404    const fn test_mm_unpackhi_epi16() {
4405        let a = _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7);
4406        let b = _mm_setr_epi16(8, 9, 10, 11, 12, 13, 14, 15);
4407        let r = _mm_unpackhi_epi16(a, b);
4408        let e = _mm_setr_epi16(4, 12, 5, 13, 6, 14, 7, 15);
4409        assert_eq_m128i(r, e);
4410    }
4411
4412    #[simd_test(enable = "sse2")]
4413    const fn test_mm_unpackhi_epi32() {
4414        let a = _mm_setr_epi32(0, 1, 2, 3);
4415        let b = _mm_setr_epi32(4, 5, 6, 7);
4416        let r = _mm_unpackhi_epi32(a, b);
4417        let e = _mm_setr_epi32(2, 6, 3, 7);
4418        assert_eq_m128i(r, e);
4419    }
4420
4421    #[simd_test(enable = "sse2")]
4422    const fn test_mm_unpackhi_epi64() {
4423        let a = _mm_setr_epi64x(0, 1);
4424        let b = _mm_setr_epi64x(2, 3);
4425        let r = _mm_unpackhi_epi64(a, b);
4426        let e = _mm_setr_epi64x(1, 3);
4427        assert_eq_m128i(r, e);
4428    }
4429
4430    #[simd_test(enable = "sse2")]
4431    const fn test_mm_unpacklo_epi8() {
4432        #[rustfmt::skip]
4433        let a = _mm_setr_epi8(
4434            0, 1, 2, 3, 4, 5, 6, 7,
4435            8, 9, 10, 11, 12, 13, 14, 15,
4436        );
4437        #[rustfmt::skip]
4438        let b = _mm_setr_epi8(
4439            16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31,
4440        );
4441        let r = _mm_unpacklo_epi8(a, b);
4442        #[rustfmt::skip]
4443        let e = _mm_setr_epi8(
4444            0, 16, 1, 17, 2, 18, 3, 19,
4445            4, 20, 5, 21, 6, 22, 7, 23,
4446        );
4447        assert_eq_m128i(r, e);
4448    }
4449
4450    #[simd_test(enable = "sse2")]
4451    const fn test_mm_unpacklo_epi16() {
4452        let a = _mm_setr_epi16(0, 1, 2, 3, 4, 5, 6, 7);
4453        let b = _mm_setr_epi16(8, 9, 10, 11, 12, 13, 14, 15);
4454        let r = _mm_unpacklo_epi16(a, b);
4455        let e = _mm_setr_epi16(0, 8, 1, 9, 2, 10, 3, 11);
4456        assert_eq_m128i(r, e);
4457    }
4458
4459    #[simd_test(enable = "sse2")]
4460    const fn test_mm_unpacklo_epi32() {
4461        let a = _mm_setr_epi32(0, 1, 2, 3);
4462        let b = _mm_setr_epi32(4, 5, 6, 7);
4463        let r = _mm_unpacklo_epi32(a, b);
4464        let e = _mm_setr_epi32(0, 4, 1, 5);
4465        assert_eq_m128i(r, e);
4466    }
4467
4468    #[simd_test(enable = "sse2")]
4469    const fn test_mm_unpacklo_epi64() {
4470        let a = _mm_setr_epi64x(0, 1);
4471        let b = _mm_setr_epi64x(2, 3);
4472        let r = _mm_unpacklo_epi64(a, b);
4473        let e = _mm_setr_epi64x(0, 2);
4474        assert_eq_m128i(r, e);
4475    }
4476
4477    #[simd_test(enable = "sse2")]
4478    const fn test_mm_add_sd() {
4479        let a = _mm_setr_pd(1.0, 2.0);
4480        let b = _mm_setr_pd(5.0, 10.0);
4481        let r = _mm_add_sd(a, b);
4482        assert_eq_m128d(r, _mm_setr_pd(6.0, 2.0));
4483    }
4484
4485    #[simd_test(enable = "sse2")]
4486    const fn test_mm_add_pd() {
4487        let a = _mm_setr_pd(1.0, 2.0);
4488        let b = _mm_setr_pd(5.0, 10.0);
4489        let r = _mm_add_pd(a, b);
4490        assert_eq_m128d(r, _mm_setr_pd(6.0, 12.0));
4491    }
4492
4493    #[simd_test(enable = "sse2")]
4494    const fn test_mm_div_sd() {
4495        let a = _mm_setr_pd(1.0, 2.0);
4496        let b = _mm_setr_pd(5.0, 10.0);
4497        let r = _mm_div_sd(a, b);
4498        assert_eq_m128d(r, _mm_setr_pd(0.2, 2.0));
4499    }
4500
4501    #[simd_test(enable = "sse2")]
4502    const fn test_mm_div_pd() {
4503        let a = _mm_setr_pd(1.0, 2.0);
4504        let b = _mm_setr_pd(5.0, 10.0);
4505        let r = _mm_div_pd(a, b);
4506        assert_eq_m128d(r, _mm_setr_pd(0.2, 0.2));
4507    }
4508
4509    #[simd_test(enable = "sse2")]
4510    fn test_mm_max_sd() {
4511        let a = _mm_setr_pd(1.0, 2.0);
4512        let b = _mm_setr_pd(5.0, 10.0);
4513        let r = _mm_max_sd(a, b);
4514        assert_eq_m128d(r, _mm_setr_pd(5.0, 2.0));
4515    }
4516
4517    #[simd_test(enable = "sse2")]
4518    fn test_mm_max_pd() {
4519        let a = _mm_setr_pd(1.0, 2.0);
4520        let b = _mm_setr_pd(5.0, 10.0);
4521        let r = _mm_max_pd(a, b);
4522        assert_eq_m128d(r, _mm_setr_pd(5.0, 10.0));
4523
4524        // Check SSE(2)-specific semantics for -0.0 handling.
4525        let a = _mm_setr_pd(-0.0, 0.0);
4526        let b = _mm_setr_pd(0.0, 0.0);
4527        // Cast to __m128i to compare exact bit patterns
4528        let r1 = _mm_castpd_si128(_mm_max_pd(a, b));
4529        let r2 = _mm_castpd_si128(_mm_max_pd(b, a));
4530        let a = _mm_castpd_si128(a);
4531        let b = _mm_castpd_si128(b);
4532        assert_eq_m128i(r1, b);
4533        assert_eq_m128i(r2, a);
4534        assert_ne!(a.as_u8x16(), b.as_u8x16()); // sanity check that -0.0 is actually present
4535    }
4536
4537    #[simd_test(enable = "sse2")]
4538    fn test_mm_min_sd() {
4539        let a = _mm_setr_pd(1.0, 2.0);
4540        let b = _mm_setr_pd(5.0, 10.0);
4541        let r = _mm_min_sd(a, b);
4542        assert_eq_m128d(r, _mm_setr_pd(1.0, 2.0));
4543    }
4544
4545    #[simd_test(enable = "sse2")]
4546    fn test_mm_min_pd() {
4547        let a = _mm_setr_pd(1.0, 2.0);
4548        let b = _mm_setr_pd(5.0, 10.0);
4549        let r = _mm_min_pd(a, b);
4550        assert_eq_m128d(r, _mm_setr_pd(1.0, 2.0));
4551
4552        // Check SSE(2)-specific semantics for -0.0 handling.
4553        let a = _mm_setr_pd(-0.0, 0.0);
4554        let b = _mm_setr_pd(0.0, 0.0);
4555        // Cast to __m128i to compare exact bit patterns
4556        let r1 = _mm_castpd_si128(_mm_min_pd(a, b));
4557        let r2 = _mm_castpd_si128(_mm_min_pd(b, a));
4558        let a = _mm_castpd_si128(a);
4559        let b = _mm_castpd_si128(b);
4560        assert_eq_m128i(r1, b);
4561        assert_eq_m128i(r2, a);
4562        assert_ne!(a.as_u8x16(), b.as_u8x16()); // sanity check that -0.0 is actually present
4563    }
4564
4565    #[simd_test(enable = "sse2")]
4566    const fn test_mm_mul_sd() {
4567        let a = _mm_setr_pd(1.0, 2.0);
4568        let b = _mm_setr_pd(5.0, 10.0);
4569        let r = _mm_mul_sd(a, b);
4570        assert_eq_m128d(r, _mm_setr_pd(5.0, 2.0));
4571    }
4572
4573    #[simd_test(enable = "sse2")]
4574    const fn test_mm_mul_pd() {
4575        let a = _mm_setr_pd(1.0, 2.0);
4576        let b = _mm_setr_pd(5.0, 10.0);
4577        let r = _mm_mul_pd(a, b);
4578        assert_eq_m128d(r, _mm_setr_pd(5.0, 20.0));
4579    }
4580
4581    #[simd_test(enable = "sse2")]
4582    fn test_mm_sqrt_sd() {
4583        let a = _mm_setr_pd(1.0, 2.0);
4584        let b = _mm_setr_pd(5.0, 10.0);
4585        let r = _mm_sqrt_sd(a, b);
4586        assert_eq_m128d(r, _mm_setr_pd(5.0f64.sqrt(), 2.0));
4587    }
4588
4589    #[simd_test(enable = "sse2")]
4590    fn test_mm_sqrt_pd() {
4591        let r = _mm_sqrt_pd(_mm_setr_pd(1.0, 2.0));
4592        assert_eq_m128d(r, _mm_setr_pd(1.0f64.sqrt(), 2.0f64.sqrt()));
4593    }
4594
4595    #[simd_test(enable = "sse2")]
4596    const fn test_mm_sub_sd() {
4597        let a = _mm_setr_pd(1.0, 2.0);
4598        let b = _mm_setr_pd(5.0, 10.0);
4599        let r = _mm_sub_sd(a, b);
4600        assert_eq_m128d(r, _mm_setr_pd(-4.0, 2.0));
4601    }
4602
4603    #[simd_test(enable = "sse2")]
4604    const fn test_mm_sub_pd() {
4605        let a = _mm_setr_pd(1.0, 2.0);
4606        let b = _mm_setr_pd(5.0, 10.0);
4607        let r = _mm_sub_pd(a, b);
4608        assert_eq_m128d(r, _mm_setr_pd(-4.0, -8.0));
4609    }
4610
4611    #[simd_test(enable = "sse2")]
4612    const fn test_mm_and_pd() {
4613        let a = f64x2::from_bits(u64x2::splat(5)).as_m128d();
4614        let b = f64x2::from_bits(u64x2::splat(3)).as_m128d();
4615        let r = _mm_and_pd(a, b);
4616        let e = f64x2::from_bits(u64x2::splat(1)).as_m128d();
4617        assert_eq_m128d(r, e);
4618    }
4619
4620    #[simd_test(enable = "sse2")]
4621    const fn test_mm_andnot_pd() {
4622        let a = f64x2::from_bits(u64x2::splat(5)).as_m128d();
4623        let b = f64x2::from_bits(u64x2::splat(3)).as_m128d();
4624        let r = _mm_andnot_pd(a, b);
4625        let e = f64x2::from_bits(u64x2::splat(2)).as_m128d();
4626        assert_eq_m128d(r, e);
4627    }
4628
4629    #[simd_test(enable = "sse2")]
4630    const fn test_mm_or_pd() {
4631        let a = f64x2::from_bits(u64x2::splat(5)).as_m128d();
4632        let b = f64x2::from_bits(u64x2::splat(3)).as_m128d();
4633        let r = _mm_or_pd(a, b);
4634        let e = f64x2::from_bits(u64x2::splat(7)).as_m128d();
4635        assert_eq_m128d(r, e);
4636    }
4637
4638    #[simd_test(enable = "sse2")]
4639    const fn test_mm_xor_pd() {
4640        let a = f64x2::from_bits(u64x2::splat(5)).as_m128d();
4641        let b = f64x2::from_bits(u64x2::splat(3)).as_m128d();
4642        let r = _mm_xor_pd(a, b);
4643        let e = f64x2::from_bits(u64x2::splat(6)).as_m128d();
4644        assert_eq_m128d(r, e);
4645    }
4646
4647    #[simd_test(enable = "sse2")]
4648    fn test_mm_cmpeq_sd() {
4649        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4650        let e = _mm_setr_epi64x(!0, 2.0f64.to_bits() as i64);
4651        let r = _mm_castpd_si128(_mm_cmpeq_sd(a, b));
4652        assert_eq_m128i(r, e);
4653    }
4654
4655    #[simd_test(enable = "sse2")]
4656    fn test_mm_cmplt_sd() {
4657        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(5.0, 3.0));
4658        let e = _mm_setr_epi64x(!0, 2.0f64.to_bits() as i64);
4659        let r = _mm_castpd_si128(_mm_cmplt_sd(a, b));
4660        assert_eq_m128i(r, e);
4661    }
4662
4663    #[simd_test(enable = "sse2")]
4664    fn test_mm_cmple_sd() {
4665        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4666        let e = _mm_setr_epi64x(!0, 2.0f64.to_bits() as i64);
4667        let r = _mm_castpd_si128(_mm_cmple_sd(a, b));
4668        assert_eq_m128i(r, e);
4669    }
4670
4671    #[simd_test(enable = "sse2")]
4672    fn test_mm_cmpgt_sd() {
4673        let (a, b) = (_mm_setr_pd(5.0, 2.0), _mm_setr_pd(1.0, 3.0));
4674        let e = _mm_setr_epi64x(!0, 2.0f64.to_bits() as i64);
4675        let r = _mm_castpd_si128(_mm_cmpgt_sd(a, b));
4676        assert_eq_m128i(r, e);
4677    }
4678
4679    #[simd_test(enable = "sse2")]
4680    fn test_mm_cmpge_sd() {
4681        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4682        let e = _mm_setr_epi64x(!0, 2.0f64.to_bits() as i64);
4683        let r = _mm_castpd_si128(_mm_cmpge_sd(a, b));
4684        assert_eq_m128i(r, e);
4685    }
4686
4687    #[simd_test(enable = "sse2")]
4688    fn test_mm_cmpord_sd() {
4689        let (a, b) = (_mm_setr_pd(NAN, 2.0), _mm_setr_pd(5.0, 3.0));
4690        let e = _mm_setr_epi64x(0, 2.0f64.to_bits() as i64);
4691        let r = _mm_castpd_si128(_mm_cmpord_sd(a, b));
4692        assert_eq_m128i(r, e);
4693    }
4694
4695    #[simd_test(enable = "sse2")]
4696    fn test_mm_cmpunord_sd() {
4697        let (a, b) = (_mm_setr_pd(NAN, 2.0), _mm_setr_pd(5.0, 3.0));
4698        let e = _mm_setr_epi64x(!0, 2.0f64.to_bits() as i64);
4699        let r = _mm_castpd_si128(_mm_cmpunord_sd(a, b));
4700        assert_eq_m128i(r, e);
4701    }
4702
4703    #[simd_test(enable = "sse2")]
4704    fn test_mm_cmpneq_sd() {
4705        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(5.0, 3.0));
4706        let e = _mm_setr_epi64x(!0, 2.0f64.to_bits() as i64);
4707        let r = _mm_castpd_si128(_mm_cmpneq_sd(a, b));
4708        assert_eq_m128i(r, e);
4709    }
4710
4711    #[simd_test(enable = "sse2")]
4712    fn test_mm_cmpnlt_sd() {
4713        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(5.0, 3.0));
4714        let e = _mm_setr_epi64x(0, 2.0f64.to_bits() as i64);
4715        let r = _mm_castpd_si128(_mm_cmpnlt_sd(a, b));
4716        assert_eq_m128i(r, e);
4717    }
4718
4719    #[simd_test(enable = "sse2")]
4720    fn test_mm_cmpnle_sd() {
4721        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4722        let e = _mm_setr_epi64x(0, 2.0f64.to_bits() as i64);
4723        let r = _mm_castpd_si128(_mm_cmpnle_sd(a, b));
4724        assert_eq_m128i(r, e);
4725    }
4726
4727    #[simd_test(enable = "sse2")]
4728    fn test_mm_cmpngt_sd() {
4729        let (a, b) = (_mm_setr_pd(5.0, 2.0), _mm_setr_pd(1.0, 3.0));
4730        let e = _mm_setr_epi64x(0, 2.0f64.to_bits() as i64);
4731        let r = _mm_castpd_si128(_mm_cmpngt_sd(a, b));
4732        assert_eq_m128i(r, e);
4733    }
4734
4735    #[simd_test(enable = "sse2")]
4736    fn test_mm_cmpnge_sd() {
4737        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4738        let e = _mm_setr_epi64x(0, 2.0f64.to_bits() as i64);
4739        let r = _mm_castpd_si128(_mm_cmpnge_sd(a, b));
4740        assert_eq_m128i(r, e);
4741    }
4742
4743    #[simd_test(enable = "sse2")]
4744    fn test_mm_cmpeq_pd() {
4745        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4746        let e = _mm_setr_epi64x(!0, 0);
4747        let r = _mm_castpd_si128(_mm_cmpeq_pd(a, b));
4748        assert_eq_m128i(r, e);
4749    }
4750
4751    #[simd_test(enable = "sse2")]
4752    fn test_mm_cmplt_pd() {
4753        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4754        let e = _mm_setr_epi64x(0, !0);
4755        let r = _mm_castpd_si128(_mm_cmplt_pd(a, b));
4756        assert_eq_m128i(r, e);
4757    }
4758
4759    #[simd_test(enable = "sse2")]
4760    fn test_mm_cmple_pd() {
4761        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4762        let e = _mm_setr_epi64x(!0, !0);
4763        let r = _mm_castpd_si128(_mm_cmple_pd(a, b));
4764        assert_eq_m128i(r, e);
4765    }
4766
4767    #[simd_test(enable = "sse2")]
4768    fn test_mm_cmpgt_pd() {
4769        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4770        let e = _mm_setr_epi64x(0, 0);
4771        let r = _mm_castpd_si128(_mm_cmpgt_pd(a, b));
4772        assert_eq_m128i(r, e);
4773    }
4774
4775    #[simd_test(enable = "sse2")]
4776    fn test_mm_cmpge_pd() {
4777        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4778        let e = _mm_setr_epi64x(!0, 0);
4779        let r = _mm_castpd_si128(_mm_cmpge_pd(a, b));
4780        assert_eq_m128i(r, e);
4781    }
4782
4783    #[simd_test(enable = "sse2")]
4784    fn test_mm_cmpord_pd() {
4785        let (a, b) = (_mm_setr_pd(NAN, 2.0), _mm_setr_pd(5.0, 3.0));
4786        let e = _mm_setr_epi64x(0, !0);
4787        let r = _mm_castpd_si128(_mm_cmpord_pd(a, b));
4788        assert_eq_m128i(r, e);
4789    }
4790
4791    #[simd_test(enable = "sse2")]
4792    fn test_mm_cmpunord_pd() {
4793        let (a, b) = (_mm_setr_pd(NAN, 2.0), _mm_setr_pd(5.0, 3.0));
4794        let e = _mm_setr_epi64x(!0, 0);
4795        let r = _mm_castpd_si128(_mm_cmpunord_pd(a, b));
4796        assert_eq_m128i(r, e);
4797    }
4798
4799    #[simd_test(enable = "sse2")]
4800    fn test_mm_cmpneq_pd() {
4801        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(5.0, 3.0));
4802        let e = _mm_setr_epi64x(!0, !0);
4803        let r = _mm_castpd_si128(_mm_cmpneq_pd(a, b));
4804        assert_eq_m128i(r, e);
4805    }
4806
4807    #[simd_test(enable = "sse2")]
4808    fn test_mm_cmpnlt_pd() {
4809        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(5.0, 3.0));
4810        let e = _mm_setr_epi64x(0, 0);
4811        let r = _mm_castpd_si128(_mm_cmpnlt_pd(a, b));
4812        assert_eq_m128i(r, e);
4813    }
4814
4815    #[simd_test(enable = "sse2")]
4816    fn test_mm_cmpnle_pd() {
4817        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4818        let e = _mm_setr_epi64x(0, 0);
4819        let r = _mm_castpd_si128(_mm_cmpnle_pd(a, b));
4820        assert_eq_m128i(r, e);
4821    }
4822
4823    #[simd_test(enable = "sse2")]
4824    fn test_mm_cmpngt_pd() {
4825        let (a, b) = (_mm_setr_pd(5.0, 2.0), _mm_setr_pd(1.0, 3.0));
4826        let e = _mm_setr_epi64x(0, !0);
4827        let r = _mm_castpd_si128(_mm_cmpngt_pd(a, b));
4828        assert_eq_m128i(r, e);
4829    }
4830
4831    #[simd_test(enable = "sse2")]
4832    fn test_mm_cmpnge_pd() {
4833        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4834        let e = _mm_setr_epi64x(0, !0);
4835        let r = _mm_castpd_si128(_mm_cmpnge_pd(a, b));
4836        assert_eq_m128i(r, e);
4837    }
4838
4839    #[simd_test(enable = "sse2")]
4840    fn test_mm_comieq_sd() {
4841        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4842        assert!(_mm_comieq_sd(a, b) != 0);
4843
4844        let (a, b) = (_mm_setr_pd(NAN, 2.0), _mm_setr_pd(1.0, 3.0));
4845        assert!(_mm_comieq_sd(a, b) == 0);
4846    }
4847
4848    #[simd_test(enable = "sse2")]
4849    fn test_mm_comilt_sd() {
4850        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4851        assert!(_mm_comilt_sd(a, b) == 0);
4852    }
4853
4854    #[simd_test(enable = "sse2")]
4855    fn test_mm_comile_sd() {
4856        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4857        assert!(_mm_comile_sd(a, b) != 0);
4858    }
4859
4860    #[simd_test(enable = "sse2")]
4861    fn test_mm_comigt_sd() {
4862        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4863        assert!(_mm_comigt_sd(a, b) == 0);
4864    }
4865
4866    #[simd_test(enable = "sse2")]
4867    fn test_mm_comige_sd() {
4868        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4869        assert!(_mm_comige_sd(a, b) != 0);
4870    }
4871
4872    #[simd_test(enable = "sse2")]
4873    fn test_mm_comineq_sd() {
4874        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4875        assert!(_mm_comineq_sd(a, b) == 0);
4876    }
4877
4878    #[simd_test(enable = "sse2")]
4879    fn test_mm_ucomieq_sd() {
4880        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4881        assert!(_mm_ucomieq_sd(a, b) != 0);
4882
4883        let (a, b) = (_mm_setr_pd(NAN, 2.0), _mm_setr_pd(NAN, 3.0));
4884        assert!(_mm_ucomieq_sd(a, b) == 0);
4885    }
4886
4887    #[simd_test(enable = "sse2")]
4888    fn test_mm_ucomilt_sd() {
4889        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4890        assert!(_mm_ucomilt_sd(a, b) == 0);
4891    }
4892
4893    #[simd_test(enable = "sse2")]
4894    fn test_mm_ucomile_sd() {
4895        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4896        assert!(_mm_ucomile_sd(a, b) != 0);
4897    }
4898
4899    #[simd_test(enable = "sse2")]
4900    fn test_mm_ucomigt_sd() {
4901        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4902        assert!(_mm_ucomigt_sd(a, b) == 0);
4903    }
4904
4905    #[simd_test(enable = "sse2")]
4906    fn test_mm_ucomige_sd() {
4907        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4908        assert!(_mm_ucomige_sd(a, b) != 0);
4909    }
4910
4911    #[simd_test(enable = "sse2")]
4912    fn test_mm_ucomineq_sd() {
4913        let (a, b) = (_mm_setr_pd(1.0, 2.0), _mm_setr_pd(1.0, 3.0));
4914        assert!(_mm_ucomineq_sd(a, b) == 0);
4915    }
4916
4917    #[simd_test(enable = "sse2")]
4918    const fn test_mm_movemask_pd() {
4919        let r = _mm_movemask_pd(_mm_setr_pd(-1.0, 5.0));
4920        assert_eq!(r, 0b01);
4921
4922        let r = _mm_movemask_pd(_mm_setr_pd(-1.0, -5.0));
4923        assert_eq!(r, 0b11);
4924    }
4925
4926    #[repr(align(16))]
4927    struct Memory {
4928        data: [f64; 4],
4929    }
4930
4931    #[simd_test(enable = "sse2")]
4932    const fn test_mm_load_pd() {
4933        let mem = Memory {
4934            data: [1.0f64, 2.0, 3.0, 4.0],
4935        };
4936        let vals = &mem.data;
4937        let d = vals.as_ptr();
4938
4939        let r = unsafe { _mm_load_pd(d) };
4940        assert_eq_m128d(r, _mm_setr_pd(1.0, 2.0));
4941    }
4942
4943    #[simd_test(enable = "sse2")]
4944    const fn test_mm_load_sd() {
4945        let a = 1.;
4946        let expected = _mm_setr_pd(a, 0.);
4947        let r = unsafe { _mm_load_sd(&a) };
4948        assert_eq_m128d(r, expected);
4949    }
4950
4951    #[simd_test(enable = "sse2")]
4952    const fn test_mm_loadh_pd() {
4953        let a = _mm_setr_pd(1., 2.);
4954        let b = 3.;
4955        let expected = _mm_setr_pd(_mm_cvtsd_f64(a), 3.);
4956        let r = unsafe { _mm_loadh_pd(a, &b) };
4957        assert_eq_m128d(r, expected);
4958    }
4959
4960    #[simd_test(enable = "sse2")]
4961    const fn test_mm_loadl_pd() {
4962        let a = _mm_setr_pd(1., 2.);
4963        let b = 3.;
4964        let expected = _mm_setr_pd(3., get_m128d(a, 1));
4965        let r = unsafe { _mm_loadl_pd(a, &b) };
4966        assert_eq_m128d(r, expected);
4967    }
4968
4969    #[simd_test(enable = "sse2")]
4970    // Miri cannot support this until it is clear how it fits in the Rust memory model
4971    // (non-temporal store)
4972    #[cfg_attr(miri, ignore)]
4973    fn test_mm_stream_pd() {
4974        #[repr(align(128))]
4975        struct Memory {
4976            pub data: [f64; 2],
4977        }
4978        let a = _mm_set1_pd(7.0);
4979        let mut mem = Memory { data: [-1.0; 2] };
4980
4981        unsafe {
4982            _mm_stream_pd(ptr::addr_of_mut!(mem.data[0]), a);
4983        }
4984        _mm_sfence();
4985        for i in 0..2 {
4986            assert_eq!(mem.data[i], get_m128d(a, i));
4987        }
4988    }
4989
4990    #[simd_test(enable = "sse2")]
4991    const fn test_mm_store_sd() {
4992        let mut dest = 0.;
4993        let a = _mm_setr_pd(1., 2.);
4994        unsafe {
4995            _mm_store_sd(&mut dest, a);
4996        }
4997        assert_eq!(dest, _mm_cvtsd_f64(a));
4998    }
4999
5000    #[simd_test(enable = "sse2")]
5001    const fn test_mm_store_pd() {
5002        let mut mem = Memory { data: [0.0f64; 4] };
5003        let vals = &mut mem.data;
5004        let a = _mm_setr_pd(1.0, 2.0);
5005        let d = vals.as_mut_ptr();
5006
5007        unsafe {
5008            _mm_store_pd(d, *black_box(&a));
5009        }
5010        assert_eq!(vals[0], 1.0);
5011        assert_eq!(vals[1], 2.0);
5012    }
5013
5014    #[simd_test(enable = "sse2")]
5015    const fn test_mm_storeu_pd() {
5016        // guaranteed to be aligned to 16 bytes
5017        let mut mem = Memory { data: [0.0f64; 4] };
5018        let vals = &mut mem.data;
5019        let a = _mm_setr_pd(1.0, 2.0);
5020
5021        // so p is *not* aligned to 16 bytes
5022        unsafe {
5023            let p = vals.as_mut_ptr().offset(1);
5024            _mm_storeu_pd(p, *black_box(&a));
5025        }
5026
5027        assert_eq!(*vals, [0.0, 1.0, 2.0, 0.0]);
5028    }
5029
5030    #[simd_test(enable = "sse2")]
5031    const fn test_mm_storeu_si16() {
5032        let a = _mm_setr_epi16(1, 2, 3, 4, 5, 6, 7, 8);
5033        let mut r = _mm_setr_epi16(9, 10, 11, 12, 13, 14, 15, 16);
5034        unsafe {
5035            _mm_storeu_si16(ptr::addr_of_mut!(r).cast(), a);
5036        }
5037        let e = _mm_setr_epi16(1, 10, 11, 12, 13, 14, 15, 16);
5038        assert_eq_m128i(r, e);
5039    }
5040
5041    #[simd_test(enable = "sse2")]
5042    const fn test_mm_storeu_si32() {
5043        let a = _mm_setr_epi32(1, 2, 3, 4);
5044        let mut r = _mm_setr_epi32(5, 6, 7, 8);
5045        unsafe {
5046            _mm_storeu_si32(ptr::addr_of_mut!(r).cast(), a);
5047        }
5048        let e = _mm_setr_epi32(1, 6, 7, 8);
5049        assert_eq_m128i(r, e);
5050    }
5051
5052    #[simd_test(enable = "sse2")]
5053    const fn test_mm_storeu_si64() {
5054        let a = _mm_setr_epi64x(1, 2);
5055        let mut r = _mm_setr_epi64x(3, 4);
5056        unsafe {
5057            _mm_storeu_si64(ptr::addr_of_mut!(r).cast(), a);
5058        }
5059        let e = _mm_setr_epi64x(1, 4);
5060        assert_eq_m128i(r, e);
5061    }
5062
5063    #[simd_test(enable = "sse2")]
5064    const fn test_mm_store1_pd() {
5065        let mut mem = Memory { data: [0.0f64; 4] };
5066        let vals = &mut mem.data;
5067        let a = _mm_setr_pd(1.0, 2.0);
5068        let d = vals.as_mut_ptr();
5069
5070        unsafe {
5071            _mm_store1_pd(d, *black_box(&a));
5072        }
5073        assert_eq!(vals[0], 1.0);
5074        assert_eq!(vals[1], 1.0);
5075    }
5076
5077    #[simd_test(enable = "sse2")]
5078    const fn test_mm_store_pd1() {
5079        let mut mem = Memory { data: [0.0f64; 4] };
5080        let vals = &mut mem.data;
5081        let a = _mm_setr_pd(1.0, 2.0);
5082        let d = vals.as_mut_ptr();
5083
5084        unsafe {
5085            _mm_store_pd1(d, *black_box(&a));
5086        }
5087        assert_eq!(vals[0], 1.0);
5088        assert_eq!(vals[1], 1.0);
5089    }
5090
5091    #[simd_test(enable = "sse2")]
5092    const fn test_mm_storer_pd() {
5093        let mut mem = Memory { data: [0.0f64; 4] };
5094        let vals = &mut mem.data;
5095        let a = _mm_setr_pd(1.0, 2.0);
5096        let d = vals.as_mut_ptr();
5097
5098        unsafe {
5099            _mm_storer_pd(d, *black_box(&a));
5100        }
5101        assert_eq!(vals[0], 2.0);
5102        assert_eq!(vals[1], 1.0);
5103    }
5104
5105    #[simd_test(enable = "sse2")]
5106    const fn test_mm_storeh_pd() {
5107        let mut dest = 0.;
5108        let a = _mm_setr_pd(1., 2.);
5109        unsafe {
5110            _mm_storeh_pd(&mut dest, a);
5111        }
5112        assert_eq!(dest, get_m128d(a, 1));
5113    }
5114
5115    #[simd_test(enable = "sse2")]
5116    const fn test_mm_storel_pd() {
5117        let mut dest = 0.;
5118        let a = _mm_setr_pd(1., 2.);
5119        unsafe {
5120            _mm_storel_pd(&mut dest, a);
5121        }
5122        assert_eq!(dest, _mm_cvtsd_f64(a));
5123    }
5124
5125    #[simd_test(enable = "sse2")]
5126    const fn test_mm_loadr_pd() {
5127        let mut mem = Memory {
5128            data: [1.0f64, 2.0, 3.0, 4.0],
5129        };
5130        let vals = &mut mem.data;
5131        let d = vals.as_ptr();
5132
5133        let r = unsafe { _mm_loadr_pd(d) };
5134        assert_eq_m128d(r, _mm_setr_pd(2.0, 1.0));
5135    }
5136
5137    #[simd_test(enable = "sse2")]
5138    const fn test_mm_loadu_pd() {
5139        // guaranteed to be aligned to 16 bytes
5140        let mut mem = Memory {
5141            data: [1.0f64, 2.0, 3.0, 4.0],
5142        };
5143        let vals = &mut mem.data;
5144
5145        // so this will *not* be aligned to 16 bytes
5146        let d = unsafe { vals.as_ptr().offset(1) };
5147
5148        let r = unsafe { _mm_loadu_pd(d) };
5149        let e = _mm_setr_pd(2.0, 3.0);
5150        assert_eq_m128d(r, e);
5151    }
5152
5153    #[simd_test(enable = "sse2")]
5154    const fn test_mm_loadu_si16() {
5155        let a = _mm_setr_epi16(1, 2, 3, 4, 5, 6, 7, 8);
5156        let r = unsafe { _mm_loadu_si16(ptr::addr_of!(a) as *const _) };
5157        assert_eq_m128i(r, _mm_setr_epi16(1, 0, 0, 0, 0, 0, 0, 0));
5158    }
5159
5160    #[simd_test(enable = "sse2")]
5161    const fn test_mm_loadu_si32() {
5162        let a = _mm_setr_epi32(1, 2, 3, 4);
5163        let r = unsafe { _mm_loadu_si32(ptr::addr_of!(a) as *const _) };
5164        assert_eq_m128i(r, _mm_setr_epi32(1, 0, 0, 0));
5165    }
5166
5167    #[simd_test(enable = "sse2")]
5168    const fn test_mm_loadu_si64() {
5169        let a = _mm_setr_epi64x(5, 6);
5170        let r = unsafe { _mm_loadu_si64(ptr::addr_of!(a) as *const _) };
5171        assert_eq_m128i(r, _mm_setr_epi64x(5, 0));
5172    }
5173
5174    #[simd_test(enable = "sse2")]
5175    const fn test_mm_cvtpd_ps() {
5176        let r = _mm_cvtpd_ps(_mm_setr_pd(-1.0, 5.0));
5177        assert_eq_m128(r, _mm_setr_ps(-1.0, 5.0, 0.0, 0.0));
5178
5179        let r = _mm_cvtpd_ps(_mm_setr_pd(-1.0, -5.0));
5180        assert_eq_m128(r, _mm_setr_ps(-1.0, -5.0, 0.0, 0.0));
5181
5182        let r = _mm_cvtpd_ps(_mm_setr_pd(f64::MAX, f64::MIN));
5183        assert_eq_m128(r, _mm_setr_ps(f32::INFINITY, f32::NEG_INFINITY, 0.0, 0.0));
5184
5185        let r = _mm_cvtpd_ps(_mm_setr_pd(f32::MAX as f64, f32::MIN as f64));
5186        assert_eq_m128(r, _mm_setr_ps(f32::MAX, f32::MIN, 0.0, 0.0));
5187    }
5188
5189    #[simd_test(enable = "sse2")]
5190    const fn test_mm_cvtps_pd() {
5191        let r = _mm_cvtps_pd(_mm_setr_ps(-1.0, 2.0, -3.0, 5.0));
5192        assert_eq_m128d(r, _mm_setr_pd(-1.0, 2.0));
5193
5194        let r = _mm_cvtps_pd(_mm_setr_ps(
5195            f32::MAX,
5196            f32::INFINITY,
5197            f32::NEG_INFINITY,
5198            f32::MIN,
5199        ));
5200        assert_eq_m128d(r, _mm_setr_pd(f32::MAX as f64, f64::INFINITY));
5201    }
5202
5203    #[simd_test(enable = "sse2")]
5204    fn test_mm_cvtpd_epi32() {
5205        let r = _mm_cvtpd_epi32(_mm_setr_pd(-1.0, 5.0));
5206        assert_eq_m128i(r, _mm_setr_epi32(-1, 5, 0, 0));
5207
5208        let r = _mm_cvtpd_epi32(_mm_setr_pd(-1.0, -5.0));
5209        assert_eq_m128i(r, _mm_setr_epi32(-1, -5, 0, 0));
5210
5211        let r = _mm_cvtpd_epi32(_mm_setr_pd(f64::MAX, f64::MIN));
5212        assert_eq_m128i(r, _mm_setr_epi32(i32::MIN, i32::MIN, 0, 0));
5213
5214        let r = _mm_cvtpd_epi32(_mm_setr_pd(f64::INFINITY, f64::NEG_INFINITY));
5215        assert_eq_m128i(r, _mm_setr_epi32(i32::MIN, i32::MIN, 0, 0));
5216
5217        let r = _mm_cvtpd_epi32(_mm_setr_pd(f64::NAN, f64::NAN));
5218        assert_eq_m128i(r, _mm_setr_epi32(i32::MIN, i32::MIN, 0, 0));
5219    }
5220
5221    #[simd_test(enable = "sse2")]
5222    fn test_mm_cvtsd_si32() {
5223        let r = _mm_cvtsd_si32(_mm_setr_pd(-2.0, 5.0));
5224        assert_eq!(r, -2);
5225
5226        let r = _mm_cvtsd_si32(_mm_setr_pd(f64::MAX, f64::MIN));
5227        assert_eq!(r, i32::MIN);
5228
5229        let r = _mm_cvtsd_si32(_mm_setr_pd(f64::NAN, f64::NAN));
5230        assert_eq!(r, i32::MIN);
5231    }
5232
5233    #[simd_test(enable = "sse2")]
5234    fn test_mm_cvtsd_ss() {
5235        let a = _mm_setr_ps(-1.1, -2.2, 3.3, 4.4);
5236        let b = _mm_setr_pd(2.0, -5.0);
5237
5238        let r = _mm_cvtsd_ss(a, b);
5239
5240        assert_eq_m128(r, _mm_setr_ps(2.0, -2.2, 3.3, 4.4));
5241
5242        let a = _mm_setr_ps(-1.1, f32::NEG_INFINITY, f32::MAX, f32::NEG_INFINITY);
5243        let b = _mm_setr_pd(f64::INFINITY, -5.0);
5244
5245        let r = _mm_cvtsd_ss(a, b);
5246
5247        assert_eq_m128(
5248            r,
5249            _mm_setr_ps(
5250                f32::INFINITY,
5251                f32::NEG_INFINITY,
5252                f32::MAX,
5253                f32::NEG_INFINITY,
5254            ),
5255        );
5256    }
5257
5258    #[simd_test(enable = "sse2")]
5259    const fn test_mm_cvtsd_f64() {
5260        let r = _mm_cvtsd_f64(_mm_setr_pd(-1.1, 2.2));
5261        assert_eq!(r, -1.1);
5262    }
5263
5264    #[simd_test(enable = "sse2")]
5265    const fn test_mm_cvtss_sd() {
5266        let a = _mm_setr_pd(-1.1, 2.2);
5267        let b = _mm_setr_ps(1.0, 2.0, 3.0, 4.0);
5268
5269        let r = _mm_cvtss_sd(a, b);
5270        assert_eq_m128d(r, _mm_setr_pd(1.0, 2.2));
5271
5272        let a = _mm_setr_pd(-1.1, f64::INFINITY);
5273        let b = _mm_setr_ps(f32::NEG_INFINITY, 2.0, 3.0, 4.0);
5274
5275        let r = _mm_cvtss_sd(a, b);
5276        assert_eq_m128d(r, _mm_setr_pd(f64::NEG_INFINITY, f64::INFINITY));
5277    }
5278
5279    #[simd_test(enable = "sse2")]
5280    fn test_mm_cvttpd_epi32() {
5281        let a = _mm_setr_pd(-1.1, 2.2);
5282        let r = _mm_cvttpd_epi32(a);
5283        assert_eq_m128i(r, _mm_setr_epi32(-1, 2, 0, 0));
5284
5285        let a = _mm_setr_pd(f64::NEG_INFINITY, f64::NAN);
5286        let r = _mm_cvttpd_epi32(a);
5287        assert_eq_m128i(r, _mm_setr_epi32(i32::MIN, i32::MIN, 0, 0));
5288    }
5289
5290    #[simd_test(enable = "sse2")]
5291    fn test_mm_cvttsd_si32() {
5292        let a = _mm_setr_pd(-1.1, 2.2);
5293        let r = _mm_cvttsd_si32(a);
5294        assert_eq!(r, -1);
5295
5296        let a = _mm_setr_pd(f64::NEG_INFINITY, f64::NAN);
5297        let r = _mm_cvttsd_si32(a);
5298        assert_eq!(r, i32::MIN);
5299    }
5300
5301    #[simd_test(enable = "sse2")]
5302    fn test_mm_cvttps_epi32() {
5303        let a = _mm_setr_ps(-1.1, 2.2, -3.3, 6.6);
5304        let r = _mm_cvttps_epi32(a);
5305        assert_eq_m128i(r, _mm_setr_epi32(-1, 2, -3, 6));
5306
5307        let a = _mm_setr_ps(f32::NEG_INFINITY, f32::INFINITY, f32::MIN, f32::MAX);
5308        let r = _mm_cvttps_epi32(a);
5309        assert_eq_m128i(r, _mm_setr_epi32(i32::MIN, i32::MIN, i32::MIN, i32::MIN));
5310    }
5311
5312    #[simd_test(enable = "sse2")]
5313    const fn test_mm_set_sd() {
5314        let r = _mm_set_sd(-1.0_f64);
5315        assert_eq_m128d(r, _mm_setr_pd(-1.0_f64, 0_f64));
5316    }
5317
5318    #[simd_test(enable = "sse2")]
5319    const fn test_mm_set1_pd() {
5320        let r = _mm_set1_pd(-1.0_f64);
5321        assert_eq_m128d(r, _mm_setr_pd(-1.0_f64, -1.0_f64));
5322    }
5323
5324    #[simd_test(enable = "sse2")]
5325    const fn test_mm_set_pd1() {
5326        let r = _mm_set_pd1(-2.0_f64);
5327        assert_eq_m128d(r, _mm_setr_pd(-2.0_f64, -2.0_f64));
5328    }
5329
5330    #[simd_test(enable = "sse2")]
5331    const fn test_mm_set_pd() {
5332        let r = _mm_set_pd(1.0_f64, 5.0_f64);
5333        assert_eq_m128d(r, _mm_setr_pd(5.0_f64, 1.0_f64));
5334    }
5335
5336    #[simd_test(enable = "sse2")]
5337    const fn test_mm_setr_pd() {
5338        let r = _mm_setr_pd(1.0_f64, -5.0_f64);
5339        assert_eq_m128d(r, _mm_setr_pd(1.0_f64, -5.0_f64));
5340    }
5341
5342    #[simd_test(enable = "sse2")]
5343    const fn test_mm_setzero_pd() {
5344        let r = _mm_setzero_pd();
5345        assert_eq_m128d(r, _mm_setr_pd(0_f64, 0_f64));
5346    }
5347
5348    #[simd_test(enable = "sse2")]
5349    const fn test_mm_load1_pd() {
5350        let d = -5.0;
5351        let r = unsafe { _mm_load1_pd(&d) };
5352        assert_eq_m128d(r, _mm_setr_pd(d, d));
5353    }
5354
5355    #[simd_test(enable = "sse2")]
5356    const fn test_mm_load_pd1() {
5357        let d = -5.0;
5358        let r = unsafe { _mm_load_pd1(&d) };
5359        assert_eq_m128d(r, _mm_setr_pd(d, d));
5360    }
5361
5362    #[simd_test(enable = "sse2")]
5363    const fn test_mm_unpackhi_pd() {
5364        let a = _mm_setr_pd(1.0, 2.0);
5365        let b = _mm_setr_pd(3.0, 4.0);
5366        let r = _mm_unpackhi_pd(a, b);
5367        assert_eq_m128d(r, _mm_setr_pd(2.0, 4.0));
5368    }
5369
5370    #[simd_test(enable = "sse2")]
5371    const fn test_mm_unpacklo_pd() {
5372        let a = _mm_setr_pd(1.0, 2.0);
5373        let b = _mm_setr_pd(3.0, 4.0);
5374        let r = _mm_unpacklo_pd(a, b);
5375        assert_eq_m128d(r, _mm_setr_pd(1.0, 3.0));
5376    }
5377
5378    #[simd_test(enable = "sse2")]
5379    const fn test_mm_shuffle_pd() {
5380        let a = _mm_setr_pd(1., 2.);
5381        let b = _mm_setr_pd(3., 4.);
5382        let expected = _mm_setr_pd(1., 3.);
5383        let r = _mm_shuffle_pd::<0b00_00_00_00>(a, b);
5384        assert_eq_m128d(r, expected);
5385    }
5386
5387    #[simd_test(enable = "sse2")]
5388    const fn test_mm_move_sd() {
5389        let a = _mm_setr_pd(1., 2.);
5390        let b = _mm_setr_pd(3., 4.);
5391        let expected = _mm_setr_pd(3., 2.);
5392        let r = _mm_move_sd(a, b);
5393        assert_eq_m128d(r, expected);
5394    }
5395
5396    #[simd_test(enable = "sse2")]
5397    const fn test_mm_castpd_ps() {
5398        let a = _mm_set1_pd(0.);
5399        let expected = _mm_set1_ps(0.);
5400        let r = _mm_castpd_ps(a);
5401        assert_eq_m128(r, expected);
5402    }
5403
5404    #[simd_test(enable = "sse2")]
5405    const fn test_mm_castpd_si128() {
5406        let a = _mm_set1_pd(0.);
5407        let expected = _mm_set1_epi64x(0);
5408        let r = _mm_castpd_si128(a);
5409        assert_eq_m128i(r, expected);
5410    }
5411
5412    #[simd_test(enable = "sse2")]
5413    const fn test_mm_castps_pd() {
5414        let a = _mm_set1_ps(0.);
5415        let expected = _mm_set1_pd(0.);
5416        let r = _mm_castps_pd(a);
5417        assert_eq_m128d(r, expected);
5418    }
5419
5420    #[simd_test(enable = "sse2")]
5421    const fn test_mm_castps_si128() {
5422        let a = _mm_set1_ps(0.);
5423        let expected = _mm_set1_epi32(0);
5424        let r = _mm_castps_si128(a);
5425        assert_eq_m128i(r, expected);
5426    }
5427
5428    #[simd_test(enable = "sse2")]
5429    const fn test_mm_castsi128_pd() {
5430        let a = _mm_set1_epi64x(0);
5431        let expected = _mm_set1_pd(0.);
5432        let r = _mm_castsi128_pd(a);
5433        assert_eq_m128d(r, expected);
5434    }
5435
5436    #[simd_test(enable = "sse2")]
5437    const fn test_mm_castsi128_ps() {
5438        let a = _mm_set1_epi32(0);
5439        let expected = _mm_set1_ps(0.);
5440        let r = _mm_castsi128_ps(a);
5441        assert_eq_m128(r, expected);
5442    }
5443}