From 9cebef44f9cf7011ad4a0b60af398b670924dbbf Mon Sep 17 00:00:00 2001 From: "Sergey \"Shnatsel\" Davidoff" Date: Thu, 6 Aug 2026 11:39:04 +0100 Subject: [PATCH] Initial pass at moving min/max and min_precise/max_precise to SimdBase --- fearless_simd/src/generated/avx2.rs | 990 +++++++------- fearless_simd/src/generated/avx512.rs | 1520 ++++++++++----------- fearless_simd/src/generated/fallback.rs | 576 ++++---- fearless_simd/src/generated/neon.rs | 464 +++---- fearless_simd/src/generated/simd_trait.rs | 842 ++++++------ fearless_simd/src/generated/simd_types.rs | 858 +++++++----- fearless_simd/src/generated/sse2.rs | 662 ++++----- fearless_simd/src/generated/sse4_2.rs | 464 +++---- fearless_simd/src/generated/wasm.rs | 296 ++-- fearless_simd_gen/src/mk_simd_types.rs | 13 +- fearless_simd_gen/src/ops.rs | 93 +- 11 files changed, 3474 insertions(+), 3304 deletions(-) diff --git a/fearless_simd/src/generated/avx2.rs b/fearless_simd/src/generated/avx2.rs index 0305ce98..ed88a4ad 100644 --- a/fearless_simd/src/generated/avx2.rs +++ b/fearless_simd/src/generated/avx2.rs @@ -230,128 +230,128 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] - fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmpeq_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { + _mm_max_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_lt_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmplt_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { + _mm_min_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_le_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmple_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { + let intermediate = _mm_max_ps(a.into(), b.into()); + let b_is_nan = _mm_cmpunord_ps(b.into(), b.into()); + _mm_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { - _mm_unpacklo_ps(a.into(), b.into()).simd_into(token) + let intermediate = _mm_min_ps(a.into(), b.into()); + let b_is_nan = _mm_cmpunord_ps(b.into(), b.into()); + _mm_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { - _mm_unpackhi_ps(a.into(), b.into()).simd_into(token) + fn kernel(token: Avx2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmpeq_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_lt_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { - _mm_shuffle_ps::<0b10_00_10_00>(a.into(), b.into()).simd_into(token) + fn kernel(token: Avx2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmplt_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_le_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { - _mm_shuffle_ps::<0b11_01_11_01>(a.into(), b.into()).simd_into(token) + fn kernel(token: Avx2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmple_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn interleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { - (self.zip_low_f32x4(a, b), self.zip_high_f32x4(a, b)) - } - #[inline(always)] - fn deinterleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { - (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) - } - #[inline(always)] - fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn zip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { - _mm_max_ps(a.into(), b.into()).simd_into(token) + _mm_unpacklo_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn zip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { - _mm_min_ps(a.into(), b.into()).simd_into(token) + _mm_unpackhi_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn unzip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { - let intermediate = _mm_max_ps(a.into(), b.into()); - let b_is_nan = _mm_cmpunord_ps(b.into(), b.into()); - _mm_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) + _mm_shuffle_ps::<0b10_00_10_00>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn unzip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f32x4, b: f32x4) -> f32x4 { - let intermediate = _mm_min_ps(a.into(), b.into()); - let b_is_nan = _mm_cmpunord_ps(b.into(), b.into()); - _mm_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) + _mm_shuffle_ps::<0b11_01_11_01>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] + fn interleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { + (self.zip_low_f32x4(a, b), self.zip_high_f32x4(a, b)) + } + #[inline(always)] + fn deinterleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { + (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) + } + #[inline(always)] fn mul_add_f32x4(self, a: f32x4, b: f32x4, c: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] @@ -796,6 +796,26 @@ impl Simd for Avx2 { .simd_into(self) } #[inline(always)] + fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i8x16, b: i8x16) -> i8x16 { + _mm_max_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i8x16, b: i8x16) -> i8x16 { + _mm_min_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i8x16(self, a: i8x16, b: i8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -895,26 +915,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i8x16, b: i8x16) -> i8x16 { - _mm_min_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i8x16, b: i8x16) -> i8x16 { - _mm_max_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i8x16(self, a: i8x16, b: i8x16) -> i8x32 { crate::kernel!( #[inline(always)] @@ -1255,6 +1255,26 @@ impl Simd for Avx2 { .simd_into(self) } #[inline(always)] + fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u8x16, b: u8x16) -> u8x16 { + _mm_max_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u8x16, b: u8x16) -> u8x16 { + _mm_min_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u8x16(self, a: u8x16, b: u8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -1360,26 +1380,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u8x16, b: u8x16) -> u8x16 { - _mm_min_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u8x16, b: u8x16) -> u8x16 { - _mm_max_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u8x16(self, a: u8x16, b: u8x16) -> u8x32 { crate::kernel!( #[inline(always)] @@ -1797,6 +1797,26 @@ impl Simd for Avx2 { .simd_into(self) } #[inline(always)] + fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i16x8, b: i16x8) -> i16x8 { + _mm_max_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i16x8, b: i16x8) -> i16x8 { + _mm_min_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i16x8(self, a: i16x8, b: i16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -1896,26 +1916,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i16x8, b: i16x8) -> i16x8 { - _mm_min_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i16x8, b: i16x8) -> i16x8 { - _mm_max_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i16x8(self, a: i16x8, b: i16x8) -> i16x16 { crate::kernel!( #[inline(always)] @@ -2214,6 +2214,26 @@ impl Simd for Avx2 { .simd_into(self) } #[inline(always)] + fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u16x8, b: u16x8) -> u16x8 { + _mm_max_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u16x8, b: u16x8) -> u16x8 { + _mm_min_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -2319,26 +2339,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u16x8, b: u16x8) -> u16x8 { - _mm_min_epu16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u16x8, b: u16x8) -> u16x8 { - _mm_max_epu16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u16x8(self, a: u16x8, b: u16x8) -> u16x16 { crate::kernel!( #[inline(always)] @@ -2788,31 +2788,51 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] - fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: i32x4, b: i32x4) -> mask32x4 { - _mm_cmpeq_epi32(a.into(), b.into()).simd_into(token) + fn kernel(token: Avx2, a: i32x4, b: i32x4) -> i32x4 { + _mm_max_epi32(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_lt_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: i32x4, b: i32x4) -> mask32x4 { - _mm_cmpgt_epi32(b.into(), a.into()).simd_into(token) + fn kernel(token: Avx2, a: i32x4, b: i32x4) -> i32x4 { + _mm_min_epi32(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_le_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: i32x4, b: i32x4) -> mask32x4 { - _mm_cmpeq_epi32(_mm_min_epi32(a.into(), b.into()), a.into()).simd_into(token) + _mm_cmpeq_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn simd_lt_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i32x4, b: i32x4) -> mask32x4 { + _mm_cmpgt_epi32(b.into(), a.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn simd_le_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i32x4, b: i32x4) -> mask32x4 { + _mm_cmpeq_epi32(_mm_min_epi32(a.into(), b.into()), a.into()).simd_into(token) } ); kernel(self, a, b) @@ -2885,26 +2905,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i32x4, b: i32x4) -> i32x4 { - _mm_min_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i32x4, b: i32x4) -> i32x4 { - _mm_max_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i32x4(self, a: i32x4, b: i32x4) -> i32x8 { crate::kernel!( #[inline(always)] @@ -3195,6 +3195,26 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u32x4, b: u32x4) -> u32x4 { + _mm_max_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u32x4, b: u32x4) -> u32x4 { + _mm_min_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u32x4(self, a: u32x4, b: u32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -3298,26 +3318,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u32x4, b: u32x4) -> u32x4 { - _mm_min_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u32x4, b: u32x4) -> u32x4 { - _mm_max_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u32x4(self, a: u32x4, b: u32x4) -> u32x8 { crate::kernel!( #[inline(always)] @@ -3752,128 +3752,128 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] - fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmpeq_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { + _mm_max_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_lt_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmplt_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { + _mm_min_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_le_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmple_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { + let intermediate = _mm_max_pd(a.into(), b.into()); + let b_is_nan = _mm_cmpunord_pd(b.into(), b.into()); + _mm_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { - _mm_unpacklo_pd(a.into(), b.into()).simd_into(token) + let intermediate = _mm_min_pd(a.into(), b.into()); + let b_is_nan = _mm_cmpunord_pd(b.into(), b.into()); + _mm_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { - _mm_unpackhi_pd(a.into(), b.into()).simd_into(token) + fn kernel(token: Avx2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmpeq_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_lt_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { - _mm_shuffle_pd::<0b00>(a.into(), b.into()).simd_into(token) + fn kernel(token: Avx2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmplt_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_le_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { - _mm_shuffle_pd::<0b11>(a.into(), b.into()).simd_into(token) + fn kernel(token: Avx2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmple_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn interleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { - (self.zip_low_f64x2(a, b), self.zip_high_f64x2(a, b)) - } - #[inline(always)] - fn deinterleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { - (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) - } - #[inline(always)] - fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn zip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { - _mm_max_pd(a.into(), b.into()).simd_into(token) + _mm_unpacklo_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn zip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { - _mm_min_pd(a.into(), b.into()).simd_into(token) + _mm_unpackhi_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn unzip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { - let intermediate = _mm_max_pd(a.into(), b.into()); - let b_is_nan = _mm_cmpunord_pd(b.into(), b.into()); - _mm_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) + _mm_shuffle_pd::<0b00>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn unzip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Avx2, a: f64x2, b: f64x2) -> f64x2 { - let intermediate = _mm_min_pd(a.into(), b.into()); - let b_is_nan = _mm_cmpunord_pd(b.into(), b.into()); - _mm_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) + _mm_shuffle_pd::<0b11>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] + fn interleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { + (self.zip_low_f64x2(a, b), self.zip_high_f64x2(a, b)) + } + #[inline(always)] + fn deinterleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { + (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) + } + #[inline(always)] fn mul_add_f64x2(self, a: f64x2, b: f64x2, c: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] @@ -4183,6 +4183,22 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::max(a[0usize], b[0usize]), + i64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::min(a[0usize], b[0usize]), + i64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i64x2(self, a: i64x2, b: i64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -4272,22 +4288,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::min(a[0usize], b[0usize]), - i64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::max(a[0usize], b[0usize]), - i64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i64x2(self, a: i64x2, b: i64x2) -> i64x4 { crate::kernel!( #[inline(always)] @@ -4550,6 +4550,22 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::max(a[0usize], b[0usize]), + u64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::min(a[0usize], b[0usize]), + u64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u64x2(self, a: u64x2, b: u64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -4639,22 +4655,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::min(a[0usize], b[0usize]), - u64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::max(a[0usize], b[0usize]), - u64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u64x2(self, a: u64x2, b: u64x2) -> u64x4 { crate::kernel!( #[inline(always)] @@ -5080,6 +5080,50 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: f32x8, b: f32x8) -> f32x8 { + _mm256_max_ps(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: f32x8, b: f32x8) -> f32x8 { + _mm256_min_ps(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: f32x8, b: f32x8) -> f32x8 { + let intermediate = _mm256_max_ps(a.into(), b.into()); + let b_is_nan = _mm256_cmp_ps::<3i32>(b.into(), b.into()); + _mm256_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: f32x8, b: f32x8) -> f32x8 { + let intermediate = _mm256_min_ps(a.into(), b.into()); + let b_is_nan = _mm256_cmp_ps::<3i32>(b.into(), b.into()); + _mm256_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f32x8(self, a: f32x8, b: f32x8) -> mask32x8 { crate::kernel!( #[inline(always)] @@ -5194,50 +5238,6 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] - fn max_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: f32x8, b: f32x8) -> f32x8 { - _mm256_max_ps(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: f32x8, b: f32x8) -> f32x8 { - _mm256_min_ps(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: f32x8, b: f32x8) -> f32x8 { - let intermediate = _mm256_max_ps(a.into(), b.into()); - let b_is_nan = _mm256_cmp_ps::<3i32>(b.into(), b.into()); - _mm256_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: f32x8, b: f32x8) -> f32x8 { - let intermediate = _mm256_min_ps(a.into(), b.into()); - let b_is_nan = _mm256_cmp_ps::<3i32>(b.into(), b.into()); - _mm256_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f32x8(self, a: f32x8, b: f32x8, c: f32x8) -> f32x8 { crate::kernel!( #[inline(always)] @@ -5665,6 +5665,26 @@ impl Simd for Avx2 { .simd_into(self) } #[inline(always)] + fn max_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i8x32, b: i8x32) -> i8x32 { + _mm256_max_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i8x32, b: i8x32) -> i8x32 { + _mm256_min_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i8x32(self, a: i8x32, b: i8x32) -> mask8x32 { crate::kernel!( #[inline(always)] @@ -5824,26 +5844,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i8x32, b: i8x32) -> i8x32 { - _mm256_min_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i8x32, b: i8x32) -> i8x32 { - _mm256_max_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i8x32(self, a: i8x32, b: i8x32) -> i8x64 { i8x64 { val: crate::support::Aligned512([a.val.0, b.val.0]), @@ -6172,6 +6172,26 @@ impl Simd for Avx2 { .simd_into(self) } #[inline(always)] + fn max_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u8x32, b: u8x32) -> u8x32 { + _mm256_max_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u8x32, b: u8x32) -> u8x32 { + _mm256_min_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u8x32(self, a: u8x32, b: u8x32) -> mask8x32 { crate::kernel!( #[inline(always)] @@ -6337,26 +6357,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u8x32, b: u8x32) -> u8x32 { - _mm256_min_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u8x32, b: u8x32) -> u8x32 { - _mm256_max_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u8x32(self, a: u8x32, b: u8x32) -> u8x64 { u8x64 { val: crate::support::Aligned512([a.val.0, b.val.0]), @@ -6746,6 +6746,26 @@ impl Simd for Avx2 { .simd_into(self) } #[inline(always)] + fn max_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i16x16, b: i16x16) -> i16x16 { + _mm256_max_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i16x16, b: i16x16) -> i16x16 { + _mm256_min_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i16x16(self, a: i16x16, b: i16x16) -> mask16x16 { crate::kernel!( #[inline(always)] @@ -6913,26 +6933,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i16x16, b: i16x16) -> i16x16 { - _mm256_min_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i16x16, b: i16x16) -> i16x16 { - _mm256_max_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i16x16(self, a: i16x16, b: i16x16) -> i16x32 { i16x32 { val: crate::support::Aligned512([a.val.0, b.val.0]), @@ -7190,6 +7190,26 @@ impl Simd for Avx2 { .simd_into(self) } #[inline(always)] + fn max_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u16x16, b: u16x16) -> u16x16 { + _mm256_max_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u16x16, b: u16x16) -> u16x16 { + _mm256_min_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u16x16(self, a: u16x16, b: u16x16) -> mask16x16 { crate::kernel!( #[inline(always)] @@ -7352,35 +7372,15 @@ impl Simd for Avx2 { crate::kernel!( #[inline(always)] fn kernel( - token: Avx2, - a: mask16x16, - b: u16x16, - c: u16x16, - ) -> u16x16 { - _mm256_blendv_epi8(c.into(), b.into(), a.into()).simd_into(token) - } - ); - kernel(self, a, b, c) - } - #[inline(always)] - fn min_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u16x16, b: u16x16) -> u16x16 { - _mm256_min_epu16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u16x16, b: u16x16) -> u16x16 { - _mm256_max_epu16(a.into(), b.into()).simd_into(token) + token: Avx2, + a: mask16x16, + b: u16x16, + c: u16x16, + ) -> u16x16 { + _mm256_blendv_epi8(c.into(), b.into(), a.into()).simd_into(token) } ); - kernel(self, a, b) + kernel(self, a, b, c) } #[inline(always)] fn combine_u16x16(self, a: u16x16, b: u16x16) -> u16x32 { @@ -7791,6 +7791,26 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i32x8, b: i32x8) -> i32x8 { + _mm256_max_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: i32x8, b: i32x8) -> i32x8 { + _mm256_min_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i32x8(self, a: i32x8, b: i32x8) -> mask32x8 { crate::kernel!( #[inline(always)] @@ -7932,26 +7952,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i32x8, b: i32x8) -> i32x8 { - _mm256_min_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: i32x8, b: i32x8) -> i32x8 { - _mm256_max_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i32x8(self, a: i32x8, b: i32x8) -> i32x16 { i32x16 { val: crate::support::Aligned512([a.val.0, b.val.0]), @@ -8195,6 +8195,26 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u32x8, b: u32x8) -> u32x8 { + _mm256_max_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: u32x8, b: u32x8) -> u32x8 { + _mm256_min_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u32x8(self, a: u32x8, b: u32x8) -> mask32x8 { crate::kernel!( #[inline(always)] @@ -8342,26 +8362,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u32x8, b: u32x8) -> u32x8 { - _mm256_min_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: u32x8, b: u32x8) -> u32x8 { - _mm256_max_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u32x8(self, a: u32x8, b: u32x8) -> u32x16 { u32x16 { val: crate::support::Aligned512([a.val.0, b.val.0]), @@ -8768,6 +8768,50 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: f64x4, b: f64x4) -> f64x4 { + _mm256_max_pd(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: f64x4, b: f64x4) -> f64x4 { + _mm256_min_pd(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: f64x4, b: f64x4) -> f64x4 { + let intermediate = _mm256_max_pd(a.into(), b.into()); + let b_is_nan = _mm256_cmp_pd::<3i32>(b.into(), b.into()); + _mm256_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx2, a: f64x4, b: f64x4) -> f64x4 { + let intermediate = _mm256_min_pd(a.into(), b.into()); + let b_is_nan = _mm256_cmp_pd::<3i32>(b.into(), b.into()); + _mm256_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f64x4(self, a: f64x4, b: f64x4) -> mask64x4 { crate::kernel!( #[inline(always)] @@ -8876,50 +8920,6 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] - fn max_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: f64x4, b: f64x4) -> f64x4 { - _mm256_max_pd(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: f64x4, b: f64x4) -> f64x4 { - _mm256_min_pd(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: f64x4, b: f64x4) -> f64x4 { - let intermediate = _mm256_max_pd(a.into(), b.into()); - let b_is_nan = _mm256_cmp_pd::<3i32>(b.into(), b.into()); - _mm256_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx2, a: f64x4, b: f64x4) -> f64x4 { - let intermediate = _mm256_min_pd(a.into(), b.into()); - let b_is_nan = _mm256_cmp_pd::<3i32>(b.into(), b.into()); - _mm256_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f64x4(self, a: f64x4, b: f64x4, c: f64x4) -> f64x4 { crate::kernel!( #[inline(always)] @@ -9199,6 +9199,26 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { + [ + i64::max(a[0usize], b[0usize]), + i64::max(a[1usize], b[1usize]), + i64::max(a[2usize], b[2usize]), + i64::max(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { + [ + i64::min(a[0usize], b[0usize]), + i64::min(a[1usize], b[1usize]), + i64::min(a[2usize], b[2usize]), + i64::min(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i64x4(self, a: i64x4, b: i64x4) -> mask64x4 { crate::kernel!( #[inline(always)] @@ -9322,26 +9342,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { - [ - i64::min(a[0usize], b[0usize]), - i64::min(a[1usize], b[1usize]), - i64::min(a[2usize], b[2usize]), - i64::min(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { - [ - i64::max(a[0usize], b[0usize]), - i64::max(a[1usize], b[1usize]), - i64::max(a[2usize], b[2usize]), - i64::max(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i64x4(self, a: i64x4, b: i64x4) -> i64x8 { i64x8 { val: crate::support::Aligned512([a.val.0, b.val.0]), @@ -9569,6 +9569,26 @@ impl Simd for Avx2 { kernel(self, a, b) } #[inline(always)] + fn max_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { + [ + u64::max(a[0usize], b[0usize]), + u64::max(a[1usize], b[1usize]), + u64::max(a[2usize], b[2usize]), + u64::max(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { + [ + u64::min(a[0usize], b[0usize]), + u64::min(a[1usize], b[1usize]), + u64::min(a[2usize], b[2usize]), + u64::min(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u64x4(self, a: u64x4, b: u64x4) -> mask64x4 { crate::kernel!( #[inline(always)] @@ -9692,26 +9712,6 @@ impl Simd for Avx2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { - [ - u64::min(a[0usize], b[0usize]), - u64::min(a[1usize], b[1usize]), - u64::min(a[2usize], b[2usize]), - u64::min(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { - [ - u64::max(a[0usize], b[0usize]), - u64::max(a[1usize], b[1usize]), - u64::max(a[2usize], b[2usize]), - u64::max(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u64x4(self, a: u64x4, b: u64x4) -> u64x8 { u64x8 { val: crate::support::Aligned512([a.val.0, b.val.0]), diff --git a/fearless_simd/src/generated/avx512.rs b/fearless_simd/src/generated/avx512.rs index ac3176a4..80a64697 100644 --- a/fearless_simd/src/generated/avx512.rs +++ b/fearless_simd/src/generated/avx512.rs @@ -504,6 +504,46 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x4, b: f32x4) -> f32x4 { + _mm_max_ps(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x4, b: f32x4) -> f32x4 { + _mm_min_ps(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x4, b: f32x4) -> f32x4 { + _mm_range_ps::<5i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x4, b: f32x4) -> f32x4 { + _mm_range_ps::<4i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -591,46 +631,6 @@ impl Simd for Avx512 { (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) } #[inline(always)] - fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x4, b: f32x4) -> f32x4 { - _mm_max_ps(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x4, b: f32x4) -> f32x4 { - _mm_min_ps(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x4, b: f32x4) -> f32x4 { - _mm_range_ps::<5i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x4, b: f32x4) -> f32x4 { - _mm_range_ps::<4i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f32x4(self, a: f32x4, b: f32x4, c: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] @@ -1025,6 +1025,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i8x16, b: i8x16) -> i8x16 { + _mm_max_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i8x16, b: i8x16) -> i8x16 { + _mm_min_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i8x16(self, a: i8x16, b: i8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -1162,26 +1182,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i8x16, b: i8x16) -> i8x16 { - _mm_min_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i8x16, b: i8x16) -> i8x16 { - _mm_max_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i8x16(self, a: i8x16, b: i8x16) -> i8x32 { crate::kernel!( #[inline(always)] @@ -1494,6 +1494,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u8x16, b: u8x16) -> u8x16 { + _mm_max_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u8x16, b: u8x16) -> u8x16 { + _mm_min_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u8x16(self, a: u8x16, b: u8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -1631,26 +1651,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u8x16, b: u8x16) -> u8x16 { - _mm_min_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u8x16, b: u8x16) -> u8x16 { - _mm_max_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u8x16(self, a: u8x16, b: u8x16) -> u8x32 { crate::kernel!( #[inline(always)] @@ -1977,6 +1977,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i16x8, b: i16x8) -> i16x8 { + _mm_max_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i16x8, b: i16x8) -> i16x8 { + _mm_min_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i16x8(self, a: i16x8, b: i16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -2106,26 +2126,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i16x8, b: i16x8) -> i16x8 { - _mm_min_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i16x8, b: i16x8) -> i16x8 { - _mm_max_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i16x8(self, a: i16x8, b: i16x8) -> i16x16 { crate::kernel!( #[inline(always)] @@ -2383,6 +2383,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u16x8, b: u16x8) -> u16x8 { + _mm_max_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u16x8, b: u16x8) -> u16x8 { + _mm_min_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -2512,26 +2532,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u16x8, b: u16x8) -> u16x8 { - _mm_min_epu16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u16x8, b: u16x8) -> u16x8 { - _mm_max_epu16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u16x8(self, a: u16x8, b: u16x8) -> u16x16 { crate::kernel!( #[inline(always)] @@ -2889,27 +2889,47 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] - fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx512, a: i32x4, b: i32x4) -> mask32x4 { - mask32x4 { - val: _mm_cmpeq_epi32_mask(a.into(), b.into()), - simd: token, - } + fn kernel(token: Avx512, a: i32x4, b: i32x4) -> i32x4 { + _mm_max_epi32(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_lt_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx512, a: i32x4, b: i32x4) -> mask32x4 { - mask32x4 { - val: _mm_cmplt_epi32_mask(a.into(), b.into()), - simd: token, - } + fn kernel(token: Avx512, a: i32x4, b: i32x4) -> i32x4 { + _mm_min_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i32x4, b: i32x4) -> mask32x4 { + mask32x4 { + val: _mm_cmpeq_epi32_mask(a.into(), b.into()), + simd: token, + } + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn simd_lt_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i32x4, b: i32x4) -> mask32x4 { + mask32x4 { + val: _mm_cmplt_epi32_mask(a.into(), b.into()), + simd: token, + } } ); kernel(self, a, b) @@ -3008,26 +3028,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i32x4, b: i32x4) -> i32x4 { - _mm_min_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i32x4, b: i32x4) -> i32x4 { - _mm_max_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i32x4(self, a: i32x4, b: i32x4) -> i32x8 { crate::kernel!( #[inline(always)] @@ -3289,6 +3289,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u32x4, b: u32x4) -> u32x4 { + _mm_max_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u32x4, b: u32x4) -> u32x4 { + _mm_min_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u32x4(self, a: u32x4, b: u32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -3408,26 +3428,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u32x4, b: u32x4) -> u32x4 { - _mm_min_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u32x4, b: u32x4) -> u32x4 { - _mm_max_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u32x4(self, a: u32x4, b: u32x4) -> u32x8 { crate::kernel!( #[inline(always)] @@ -3778,6 +3778,46 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x2, b: f64x2) -> f64x2 { + _mm_max_pd(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x2, b: f64x2) -> f64x2 { + _mm_min_pd(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x2, b: f64x2) -> f64x2 { + _mm_range_pd::<5i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x2, b: f64x2) -> f64x2 { + _mm_range_pd::<4i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -3865,46 +3905,6 @@ impl Simd for Avx512 { (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) } #[inline(always)] - fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x2, b: f64x2) -> f64x2 { - _mm_max_pd(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x2, b: f64x2) -> f64x2 { - _mm_min_pd(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x2, b: f64x2) -> f64x2 { - _mm_range_pd::<5i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x2, b: f64x2) -> f64x2 { - _mm_range_pd::<4i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f64x2(self, a: f64x2, b: f64x2, c: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] @@ -4204,6 +4204,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i64x2, b: i64x2) -> i64x2 { + _mm_max_epi64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i64x2, b: i64x2) -> i64x2 { + _mm_min_epi64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i64x2(self, a: i64x2, b: i64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -4321,26 +4341,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i64x2, b: i64x2) -> i64x2 { - _mm_min_epi64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i64x2, b: i64x2) -> i64x2 { - _mm_max_epi64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i64x2(self, a: i64x2, b: i64x2) -> i64x4 { crate::kernel!( #[inline(always)] @@ -4574,6 +4574,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u64x2, b: u64x2) -> u64x2 { + _mm_max_epu64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u64x2, b: u64x2) -> u64x2 { + _mm_min_epu64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u64x2(self, a: u64x2, b: u64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -4691,26 +4711,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u64x2, b: u64x2) -> u64x2 { - _mm_min_epu64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u64x2, b: u64x2) -> u64x2 { - _mm_max_epu64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u64x2(self, a: u64x2, b: u64x2) -> u64x4 { crate::kernel!( #[inline(always)] @@ -5072,6 +5072,46 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x8, b: f32x8) -> f32x8 { + _mm256_max_ps(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x8, b: f32x8) -> f32x8 { + _mm256_min_ps(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x8, b: f32x8) -> f32x8 { + _mm256_range_ps::<5i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x8, b: f32x8) -> f32x8 { + _mm256_range_ps::<4i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f32x8(self, a: f32x8, b: f32x8) -> mask32x8 { crate::kernel!( #[inline(always)] @@ -5213,46 +5253,6 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] - fn max_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x8, b: f32x8) -> f32x8 { - _mm256_max_ps(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x8, b: f32x8) -> f32x8 { - _mm256_min_ps(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x8, b: f32x8) -> f32x8 { - _mm256_range_ps::<5i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x8, b: f32x8) -> f32x8 { - _mm256_range_ps::<4i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f32x8(self, a: f32x8, b: f32x8, c: f32x8) -> f32x8 { crate::kernel!( #[inline(always)] @@ -5653,6 +5653,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i8x32, b: i8x32) -> i8x32 { + _mm256_max_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i8x32, b: i8x32) -> i8x32 { + _mm256_min_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i8x32(self, a: i8x32, b: i8x32) -> mask8x32 { crate::kernel!( #[inline(always)] @@ -5849,26 +5869,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i8x32, b: i8x32) -> i8x32 { - _mm256_min_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i8x32, b: i8x32) -> i8x32 { - _mm256_max_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i8x32(self, a: i8x32, b: i8x32) -> i8x64 { crate::kernel!( #[inline(always)] @@ -6175,6 +6175,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u8x32, b: u8x32) -> u8x32 { + _mm256_max_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u8x32, b: u8x32) -> u8x32 { + _mm256_min_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u8x32(self, a: u8x32, b: u8x32) -> mask8x32 { crate::kernel!( #[inline(always)] @@ -6371,26 +6391,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u8x32, b: u8x32) -> u8x32 { - _mm256_min_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u8x32, b: u8x32) -> u8x32 { - _mm256_max_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u8x32(self, a: u8x32, b: u8x32) -> u8x64 { crate::kernel!( #[inline(always)] @@ -6719,6 +6719,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i16x16, b: i16x16) -> i16x16 { + _mm256_max_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i16x16, b: i16x16) -> i16x16 { + _mm256_min_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i16x16(self, a: i16x16, b: i16x16) -> mask16x16 { crate::kernel!( #[inline(always)] @@ -6897,26 +6917,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i16x16, b: i16x16) -> i16x16 { - _mm256_min_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i16x16, b: i16x16) -> i16x16 { - _mm256_max_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i16x16(self, a: i16x16, b: i16x16) -> i16x32 { crate::kernel!( #[inline(always)] @@ -7166,6 +7166,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u16x16, b: u16x16) -> u16x16 { + _mm256_max_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u16x16, b: u16x16) -> u16x16 { + _mm256_min_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u16x16(self, a: u16x16, b: u16x16) -> mask16x16 { crate::kernel!( #[inline(always)] @@ -7331,37 +7351,17 @@ impl Simd for Avx512 { #[inline(always)] fn select_u16x16(self, a: mask16x16, b: u16x16, c: u16x16) -> u16x16 { crate::kernel!( - #[inline(always)] - fn kernel( - token: Avx512, - a: mask16x16, - b: u16x16, - c: u16x16, - ) -> u16x16 { - _mm256_mask_blend_epi16(a.val, c.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b, c) - } - #[inline(always)] - fn min_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u16x16, b: u16x16) -> u16x16 { - _mm256_min_epu16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u16x16, b: u16x16) -> u16x16 { - _mm256_max_epu16(a.into(), b.into()).simd_into(token) + #[inline(always)] + fn kernel( + token: Avx512, + a: mask16x16, + b: u16x16, + c: u16x16, + ) -> u16x16 { + _mm256_mask_blend_epi16(a.val, c.into(), b.into()).simd_into(token) } ); - kernel(self, a, b) + kernel(self, a, b, c) } #[inline(always)] fn combine_u16x16(self, a: u16x16, b: u16x16) -> u16x32 { @@ -7727,6 +7727,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i32x8, b: i32x8) -> i32x8 { + _mm256_max_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i32x8, b: i32x8) -> i32x8 { + _mm256_min_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i32x8(self, a: i32x8, b: i32x8) -> mask32x8 { crate::kernel!( #[inline(always)] @@ -7883,26 +7903,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i32x8, b: i32x8) -> i32x8 { - _mm256_min_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i32x8, b: i32x8) -> i32x8 { - _mm256_max_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i32x8(self, a: i32x8, b: i32x8) -> i32x16 { crate::kernel!( #[inline(always)] @@ -8162,6 +8162,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u32x8, b: u32x8) -> u32x8 { + _mm256_max_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u32x8, b: u32x8) -> u32x8 { + _mm256_min_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u32x8(self, a: u32x8, b: u32x8) -> mask32x8 { crate::kernel!( #[inline(always)] @@ -8318,26 +8338,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u32x8, b: u32x8) -> u32x8 { - _mm256_min_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u32x8, b: u32x8) -> u32x8 { - _mm256_max_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u32x8(self, a: u32x8, b: u32x8) -> u32x16 { crate::kernel!( #[inline(always)] @@ -8703,6 +8703,46 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x4, b: f64x4) -> f64x4 { + _mm256_max_pd(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x4, b: f64x4) -> f64x4 { + _mm256_min_pd(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x4, b: f64x4) -> f64x4 { + _mm256_range_pd::<5i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x4, b: f64x4) -> f64x4 { + _mm256_range_pd::<4i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f64x4(self, a: f64x4, b: f64x4) -> mask64x4 { crate::kernel!( #[inline(always)] @@ -8824,46 +8864,6 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] - fn max_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x4, b: f64x4) -> f64x4 { - _mm256_max_pd(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x4, b: f64x4) -> f64x4 { - _mm256_min_pd(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x4, b: f64x4) -> f64x4 { - _mm256_range_pd::<5i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x4, b: f64x4) -> f64x4 { - _mm256_range_pd::<4i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f64x4(self, a: f64x4, b: f64x4, c: f64x4) -> f64x4 { crate::kernel!( #[inline(always)] @@ -9167,6 +9167,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i64x4, b: i64x4) -> i64x4 { + _mm256_max_epi64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i64x4, b: i64x4) -> i64x4 { + _mm256_min_epi64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i64x4(self, a: i64x4, b: i64x4) -> mask64x4 { crate::kernel!( #[inline(always)] @@ -9307,26 +9327,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i64x4, b: i64x4) -> i64x4 { - _mm256_min_epi64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i64x4, b: i64x4) -> i64x4 { - _mm256_max_epi64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i64x4(self, a: i64x4, b: i64x4) -> i64x8 { crate::kernel!( #[inline(always)] @@ -9562,6 +9562,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u64x4, b: u64x4) -> u64x4 { + _mm256_max_epu64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u64x4, b: u64x4) -> u64x4 { + _mm256_min_epu64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u64x4(self, a: u64x4, b: u64x4) -> mask64x4 { crate::kernel!( #[inline(always)] @@ -9691,35 +9711,15 @@ impl Simd for Avx512 { crate::kernel!( #[inline(always)] fn kernel( - token: Avx512, - a: mask64x4, - b: u64x4, - c: u64x4, - ) -> u64x4 { - _mm256_mask_blend_epi64(a.val, c.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b, c) - } - #[inline(always)] - fn min_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u64x4, b: u64x4) -> u64x4 { - _mm256_min_epu64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u64x4, b: u64x4) -> u64x4 { - _mm256_max_epu64(a.into(), b.into()).simd_into(token) + token: Avx512, + a: mask64x4, + b: u64x4, + c: u64x4, + ) -> u64x4 { + _mm256_mask_blend_epi64(a.val, c.into(), b.into()).simd_into(token) } ); - kernel(self, a, b) + kernel(self, a, b, c) } #[inline(always)] fn combine_u64x4(self, a: u64x4, b: u64x4) -> u64x8 { @@ -10064,6 +10064,46 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x16, b: f32x16) -> f32x16 { + _mm512_max_ps(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x16, b: f32x16) -> f32x16 { + _mm512_min_ps(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x16, b: f32x16) -> f32x16 { + _mm512_range_ps::<5i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f32x16, b: f32x16) -> f32x16 { + _mm512_range_ps::<4i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f32x16(self, a: f32x16, b: f32x16) -> mask32x16 { crate::kernel!( #[inline(always)] @@ -10227,46 +10267,6 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] - fn max_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x16, b: f32x16) -> f32x16 { - _mm512_max_ps(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x16, b: f32x16) -> f32x16 { - _mm512_min_ps(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x16, b: f32x16) -> f32x16 { - _mm512_range_ps::<5i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f32x16, b: f32x16) -> f32x16 { - _mm512_range_ps::<4i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f32x16(self, a: f32x16, b: f32x16, c: f32x16) -> f32x16 { crate::kernel!( #[inline(always)] @@ -10660,6 +10660,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i8x64(self, a: i8x64, b: i8x64) -> i8x64 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i8x64, b: i8x64) -> i8x64 { + _mm512_max_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i8x64(self, a: i8x64, b: i8x64) -> i8x64 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i8x64, b: i8x64) -> i8x64 { + _mm512_min_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i8x64(self, a: i8x64, b: i8x64) -> mask8x64 { crate::kernel!( #[inline(always)] @@ -10872,26 +10892,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i8x64(self, a: i8x64, b: i8x64) -> i8x64 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i8x64, b: i8x64) -> i8x64 { - _mm512_min_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i8x64(self, a: i8x64, b: i8x64) -> i8x64 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i8x64, b: i8x64) -> i8x64 { - _mm512_max_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn split_i8x64(self, a: i8x64) -> (i8x32, i8x32) { crate::kernel!( #[inline(always)] @@ -11189,6 +11189,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u8x64(self, a: u8x64, b: u8x64) -> u8x64 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u8x64, b: u8x64) -> u8x64 { + _mm512_max_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u8x64(self, a: u8x64, b: u8x64) -> u8x64 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u8x64, b: u8x64) -> u8x64 { + _mm512_min_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u8x64(self, a: u8x64, b: u8x64) -> mask8x64 { crate::kernel!( #[inline(always)] @@ -11401,26 +11421,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u8x64(self, a: u8x64, b: u8x64) -> u8x64 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u8x64, b: u8x64) -> u8x64 { - _mm512_min_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u8x64(self, a: u8x64, b: u8x64) -> u8x64 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u8x64, b: u8x64) -> u8x64 { - _mm512_max_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn split_u8x64(self, a: u8x64) -> (u8x32, u8x32) { crate::kernel!( #[inline(always)] @@ -11733,6 +11733,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i16x32(self, a: i16x32, b: i16x32) -> i16x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i16x32, b: i16x32) -> i16x32 { + _mm512_max_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i16x32(self, a: i16x32, b: i16x32) -> i16x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i16x32, b: i16x32) -> i16x32 { + _mm512_min_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i16x32(self, a: i16x32, b: i16x32) -> mask16x32 { crate::kernel!( #[inline(always)] @@ -11929,26 +11949,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i16x32(self, a: i16x32, b: i16x32) -> i16x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i16x32, b: i16x32) -> i16x32 { - _mm512_min_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i16x32(self, a: i16x32, b: i16x32) -> i16x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i16x32, b: i16x32) -> i16x32 { - _mm512_max_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn split_i16x32(self, a: i16x32) -> (i16x16, i16x16) { crate::kernel!( #[inline(always)] @@ -12190,6 +12190,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u16x32(self, a: u16x32, b: u16x32) -> u16x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u16x32, b: u16x32) -> u16x32 { + _mm512_max_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u16x32(self, a: u16x32, b: u16x32) -> u16x32 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u16x32, b: u16x32) -> u16x32 { + _mm512_min_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u16x32(self, a: u16x32, b: u16x32) -> mask16x32 { crate::kernel!( #[inline(always)] @@ -12363,47 +12383,27 @@ impl Simd for Avx512 { 27, 25, 23, 21, 19, 17, 15, 13, 11, 9, 7, 5, 3, 1, ), b, - ) - .simd_into(token), - ) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn select_u16x32(self, a: mask16x32, b: u16x32, c: u16x32) -> u16x32 { - crate::kernel!( - #[inline(always)] - fn kernel( - token: Avx512, - a: mask16x32, - b: u16x32, - c: u16x32, - ) -> u16x32 { - _mm512_mask_blend_epi16(a.val, c.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b, c) - } - #[inline(always)] - fn min_u16x32(self, a: u16x32, b: u16x32) -> u16x32 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u16x32, b: u16x32) -> u16x32 { - _mm512_min_epu16(a.into(), b.into()).simd_into(token) + ) + .simd_into(token), + ) } ); kernel(self, a, b) } #[inline(always)] - fn max_u16x32(self, a: u16x32, b: u16x32) -> u16x32 { + fn select_u16x32(self, a: mask16x32, b: u16x32, c: u16x32) -> u16x32 { crate::kernel!( #[inline(always)] - fn kernel(token: Avx512, a: u16x32, b: u16x32) -> u16x32 { - _mm512_max_epu16(a.into(), b.into()).simd_into(token) + fn kernel( + token: Avx512, + a: mask16x32, + b: u16x32, + c: u16x32, + ) -> u16x32 { + _mm512_mask_blend_epi16(a.val, c.into(), b.into()).simd_into(token) } ); - kernel(self, a, b) + kernel(self, a, b, c) } #[inline(always)] fn split_u16x32(self, a: u16x32) -> (u16x16, u16x16) { @@ -12753,6 +12753,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i32x16(self, a: i32x16, b: i32x16) -> i32x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i32x16, b: i32x16) -> i32x16 { + _mm512_max_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i32x16(self, a: i32x16, b: i32x16) -> i32x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i32x16, b: i32x16) -> i32x16 { + _mm512_min_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i32x16(self, a: i32x16, b: i32x16) -> mask32x16 { crate::kernel!( #[inline(always)] @@ -12931,26 +12951,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i32x16(self, a: i32x16, b: i32x16) -> i32x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i32x16, b: i32x16) -> i32x16 { - _mm512_min_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i32x16(self, a: i32x16, b: i32x16) -> i32x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i32x16, b: i32x16) -> i32x16 { - _mm512_max_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn split_i32x16(self, a: i32x16) -> (i32x8, i32x8) { crate::kernel!( #[inline(always)] @@ -13202,6 +13202,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u32x16(self, a: u32x16, b: u32x16) -> u32x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u32x16, b: u32x16) -> u32x16 { + _mm512_max_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u32x16(self, a: u32x16, b: u32x16) -> u32x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u32x16, b: u32x16) -> u32x16 { + _mm512_min_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u32x16(self, a: u32x16, b: u32x16) -> mask32x16 { crate::kernel!( #[inline(always)] @@ -13380,26 +13400,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u32x16(self, a: u32x16, b: u32x16) -> u32x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u32x16, b: u32x16) -> u32x16 { - _mm512_min_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u32x16(self, a: u32x16, b: u32x16) -> u32x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u32x16, b: u32x16) -> u32x16 { - _mm512_max_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn split_u32x16(self, a: u32x16) -> (u32x8, u32x8) { crate::kernel!( #[inline(always)] @@ -13748,6 +13748,46 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x8, b: f64x8) -> f64x8 { + _mm512_max_pd(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x8, b: f64x8) -> f64x8 { + _mm512_min_pd(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x8, b: f64x8) -> f64x8 { + _mm512_range_pd::<5i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: f64x8, b: f64x8) -> f64x8 { + _mm512_range_pd::<4i32>(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f64x8(self, a: f64x8, b: f64x8) -> mask64x8 { crate::kernel!( #[inline(always)] @@ -13889,46 +13929,6 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] - fn max_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x8, b: f64x8) -> f64x8 { - _mm512_max_pd(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x8, b: f64x8) -> f64x8 { - _mm512_min_pd(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x8, b: f64x8) -> f64x8 { - _mm512_range_pd::<5i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: f64x8, b: f64x8) -> f64x8 { - _mm512_range_pd::<4i32>(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f64x8(self, a: f64x8, b: f64x8, c: f64x8) -> f64x8 { crate::kernel!( #[inline(always)] @@ -14224,6 +14224,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_i64x8(self, a: i64x8, b: i64x8) -> i64x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i64x8, b: i64x8) -> i64x8 { + _mm512_max_epi64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i64x8(self, a: i64x8, b: i64x8) -> i64x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: i64x8, b: i64x8) -> i64x8 { + _mm512_min_epi64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i64x8(self, a: i64x8, b: i64x8) -> mask64x8 { crate::kernel!( #[inline(always)] @@ -14380,26 +14400,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_i64x8(self, a: i64x8, b: i64x8) -> i64x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i64x8, b: i64x8) -> i64x8 { - _mm512_min_epi64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i64x8(self, a: i64x8, b: i64x8) -> i64x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: i64x8, b: i64x8) -> i64x8 { - _mm512_max_epi64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn split_i64x8(self, a: i64x8) -> (i64x4, i64x4) { crate::kernel!( #[inline(always)] @@ -14627,6 +14627,26 @@ impl Simd for Avx512 { kernel(self, a, b) } #[inline(always)] + fn max_u64x8(self, a: u64x8, b: u64x8) -> u64x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u64x8, b: u64x8) -> u64x8 { + _mm512_max_epu64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u64x8(self, a: u64x8, b: u64x8) -> u64x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Avx512, a: u64x8, b: u64x8) -> u64x8 { + _mm512_min_epu64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u64x8(self, a: u64x8, b: u64x8) -> mask64x8 { crate::kernel!( #[inline(always)] @@ -14783,26 +14803,6 @@ impl Simd for Avx512 { kernel(self, a, b, c) } #[inline(always)] - fn min_u64x8(self, a: u64x8, b: u64x8) -> u64x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u64x8, b: u64x8) -> u64x8 { - _mm512_min_epu64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u64x8(self, a: u64x8, b: u64x8) -> u64x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Avx512, a: u64x8, b: u64x8) -> u64x8 { - _mm512_max_epu64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn split_u64x8(self, a: u64x8) -> (u64x4, u64x4) { crate::kernel!( #[inline(always)] diff --git a/fearless_simd/src/generated/fallback.rs b/fearless_simd/src/generated/fallback.rs index 866197d2..83c4dc25 100644 --- a/fearless_simd/src/generated/fallback.rs +++ b/fearless_simd/src/generated/fallback.rs @@ -260,6 +260,46 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + [ + f32::max(a[0usize], b[0usize]), + f32::max(a[1usize], b[1usize]), + f32::max(a[2usize], b[2usize]), + f32::max(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + [ + f32::min(a[0usize], b[0usize]), + f32::min(a[1usize], b[1usize]), + f32::min(a[2usize], b[2usize]), + f32::min(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + [ + f32::max(a[0usize], b[0usize]), + f32::max(a[1usize], b[1usize]), + f32::max(a[2usize], b[2usize]), + f32::max(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + [ + f32::min(a[0usize], b[0usize]), + f32::min(a[1usize], b[1usize]), + f32::min(a[2usize], b[2usize]), + f32::min(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { [ -(f32::eq(&a[0usize], &b[0usize]) as i32), @@ -314,46 +354,6 @@ impl Simd for Fallback { (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) } #[inline(always)] - fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - [ - f32::max(a[0usize], b[0usize]), - f32::max(a[1usize], b[1usize]), - f32::max(a[2usize], b[2usize]), - f32::max(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - [ - f32::min(a[0usize], b[0usize]), - f32::min(a[1usize], b[1usize]), - f32::min(a[2usize], b[2usize]), - f32::min(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - [ - f32::max(a[0usize], b[0usize]), - f32::max(a[1usize], b[1usize]), - f32::max(a[2usize], b[2usize]), - f32::max(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - [ - f32::min(a[0usize], b[0usize]), - f32::min(a[1usize], b[1usize]), - f32::min(a[2usize], b[2usize]), - f32::min(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] fn mul_add_f32x4(self, a: f32x4, b: f32x4, c: f32x4) -> f32x4 { a.mul(b).add(c) } @@ -779,6 +779,50 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + [ + i8::max(a[0usize], b[0usize]), + i8::max(a[1usize], b[1usize]), + i8::max(a[2usize], b[2usize]), + i8::max(a[3usize], b[3usize]), + i8::max(a[4usize], b[4usize]), + i8::max(a[5usize], b[5usize]), + i8::max(a[6usize], b[6usize]), + i8::max(a[7usize], b[7usize]), + i8::max(a[8usize], b[8usize]), + i8::max(a[9usize], b[9usize]), + i8::max(a[10usize], b[10usize]), + i8::max(a[11usize], b[11usize]), + i8::max(a[12usize], b[12usize]), + i8::max(a[13usize], b[13usize]), + i8::max(a[14usize], b[14usize]), + i8::max(a[15usize], b[15usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + [ + i8::min(a[0usize], b[0usize]), + i8::min(a[1usize], b[1usize]), + i8::min(a[2usize], b[2usize]), + i8::min(a[3usize], b[3usize]), + i8::min(a[4usize], b[4usize]), + i8::min(a[5usize], b[5usize]), + i8::min(a[6usize], b[6usize]), + i8::min(a[7usize], b[7usize]), + i8::min(a[8usize], b[8usize]), + i8::min(a[9usize], b[9usize]), + i8::min(a[10usize], b[10usize]), + i8::min(a[11usize], b[11usize]), + i8::min(a[12usize], b[12usize]), + i8::min(a[13usize], b[13usize]), + i8::min(a[14usize], b[14usize]), + i8::min(a[15usize], b[15usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i8x16(self, a: i8x16, b: i8x16) -> mask8x16 { [ -(i8::eq(&a[0usize], &b[0usize]) as i8), @@ -974,50 +1018,6 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] - fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - [ - i8::min(a[0usize], b[0usize]), - i8::min(a[1usize], b[1usize]), - i8::min(a[2usize], b[2usize]), - i8::min(a[3usize], b[3usize]), - i8::min(a[4usize], b[4usize]), - i8::min(a[5usize], b[5usize]), - i8::min(a[6usize], b[6usize]), - i8::min(a[7usize], b[7usize]), - i8::min(a[8usize], b[8usize]), - i8::min(a[9usize], b[9usize]), - i8::min(a[10usize], b[10usize]), - i8::min(a[11usize], b[11usize]), - i8::min(a[12usize], b[12usize]), - i8::min(a[13usize], b[13usize]), - i8::min(a[14usize], b[14usize]), - i8::min(a[15usize], b[15usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - [ - i8::max(a[0usize], b[0usize]), - i8::max(a[1usize], b[1usize]), - i8::max(a[2usize], b[2usize]), - i8::max(a[3usize], b[3usize]), - i8::max(a[4usize], b[4usize]), - i8::max(a[5usize], b[5usize]), - i8::max(a[6usize], b[6usize]), - i8::max(a[7usize], b[7usize]), - i8::max(a[8usize], b[8usize]), - i8::max(a[9usize], b[9usize]), - i8::max(a[10usize], b[10usize]), - i8::max(a[11usize], b[11usize]), - i8::max(a[12usize], b[12usize]), - i8::max(a[13usize], b[13usize]), - i8::max(a[14usize], b[14usize]), - i8::max(a[15usize], b[15usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i8x16(self, a: i8x16, b: i8x16) -> i8x32 { let mut result = [0; 32usize]; result[0..16usize].copy_from_slice(&a.val.0); @@ -1576,6 +1576,50 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + [ + u8::max(a[0usize], b[0usize]), + u8::max(a[1usize], b[1usize]), + u8::max(a[2usize], b[2usize]), + u8::max(a[3usize], b[3usize]), + u8::max(a[4usize], b[4usize]), + u8::max(a[5usize], b[5usize]), + u8::max(a[6usize], b[6usize]), + u8::max(a[7usize], b[7usize]), + u8::max(a[8usize], b[8usize]), + u8::max(a[9usize], b[9usize]), + u8::max(a[10usize], b[10usize]), + u8::max(a[11usize], b[11usize]), + u8::max(a[12usize], b[12usize]), + u8::max(a[13usize], b[13usize]), + u8::max(a[14usize], b[14usize]), + u8::max(a[15usize], b[15usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + [ + u8::min(a[0usize], b[0usize]), + u8::min(a[1usize], b[1usize]), + u8::min(a[2usize], b[2usize]), + u8::min(a[3usize], b[3usize]), + u8::min(a[4usize], b[4usize]), + u8::min(a[5usize], b[5usize]), + u8::min(a[6usize], b[6usize]), + u8::min(a[7usize], b[7usize]), + u8::min(a[8usize], b[8usize]), + u8::min(a[9usize], b[9usize]), + u8::min(a[10usize], b[10usize]), + u8::min(a[11usize], b[11usize]), + u8::min(a[12usize], b[12usize]), + u8::min(a[13usize], b[13usize]), + u8::min(a[14usize], b[14usize]), + u8::min(a[15usize], b[15usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u8x16(self, a: u8x16, b: u8x16) -> mask8x16 { [ -(u8::eq(&a[0usize], &b[0usize]) as i8), @@ -1771,50 +1815,6 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] - fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - [ - u8::min(a[0usize], b[0usize]), - u8::min(a[1usize], b[1usize]), - u8::min(a[2usize], b[2usize]), - u8::min(a[3usize], b[3usize]), - u8::min(a[4usize], b[4usize]), - u8::min(a[5usize], b[5usize]), - u8::min(a[6usize], b[6usize]), - u8::min(a[7usize], b[7usize]), - u8::min(a[8usize], b[8usize]), - u8::min(a[9usize], b[9usize]), - u8::min(a[10usize], b[10usize]), - u8::min(a[11usize], b[11usize]), - u8::min(a[12usize], b[12usize]), - u8::min(a[13usize], b[13usize]), - u8::min(a[14usize], b[14usize]), - u8::min(a[15usize], b[15usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - [ - u8::max(a[0usize], b[0usize]), - u8::max(a[1usize], b[1usize]), - u8::max(a[2usize], b[2usize]), - u8::max(a[3usize], b[3usize]), - u8::max(a[4usize], b[4usize]), - u8::max(a[5usize], b[5usize]), - u8::max(a[6usize], b[6usize]), - u8::max(a[7usize], b[7usize]), - u8::max(a[8usize], b[8usize]), - u8::max(a[9usize], b[9usize]), - u8::max(a[10usize], b[10usize]), - u8::max(a[11usize], b[11usize]), - u8::max(a[12usize], b[12usize]), - u8::max(a[13usize], b[13usize]), - u8::max(a[14usize], b[14usize]), - u8::max(a[15usize], b[15usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u8x16(self, a: u8x16, b: u8x16) -> u8x32 { let mut result = [0; 32usize]; result[0..16usize].copy_from_slice(&a.val.0); @@ -2503,6 +2503,34 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + [ + i16::max(a[0usize], b[0usize]), + i16::max(a[1usize], b[1usize]), + i16::max(a[2usize], b[2usize]), + i16::max(a[3usize], b[3usize]), + i16::max(a[4usize], b[4usize]), + i16::max(a[5usize], b[5usize]), + i16::max(a[6usize], b[6usize]), + i16::max(a[7usize], b[7usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + [ + i16::min(a[0usize], b[0usize]), + i16::min(a[1usize], b[1usize]), + i16::min(a[2usize], b[2usize]), + i16::min(a[3usize], b[3usize]), + i16::min(a[4usize], b[4usize]), + i16::min(a[5usize], b[5usize]), + i16::min(a[6usize], b[6usize]), + i16::min(a[7usize], b[7usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i16x8(self, a: i16x8, b: i16x8) -> mask16x8 { [ -(i16::eq(&a[0usize], &b[0usize]) as i16), @@ -2627,34 +2655,6 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] - fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - [ - i16::min(a[0usize], b[0usize]), - i16::min(a[1usize], b[1usize]), - i16::min(a[2usize], b[2usize]), - i16::min(a[3usize], b[3usize]), - i16::min(a[4usize], b[4usize]), - i16::min(a[5usize], b[5usize]), - i16::min(a[6usize], b[6usize]), - i16::min(a[7usize], b[7usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - [ - i16::max(a[0usize], b[0usize]), - i16::max(a[1usize], b[1usize]), - i16::max(a[2usize], b[2usize]), - i16::max(a[3usize], b[3usize]), - i16::max(a[4usize], b[4usize]), - i16::max(a[5usize], b[5usize]), - i16::max(a[6usize], b[6usize]), - i16::max(a[7usize], b[7usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i16x8(self, a: i16x8, b: i16x8) -> i16x16 { let mut result = [0; 16usize]; result[0..8usize].copy_from_slice(&a.val.0); @@ -3005,6 +3005,34 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + [ + u16::max(a[0usize], b[0usize]), + u16::max(a[1usize], b[1usize]), + u16::max(a[2usize], b[2usize]), + u16::max(a[3usize], b[3usize]), + u16::max(a[4usize], b[4usize]), + u16::max(a[5usize], b[5usize]), + u16::max(a[6usize], b[6usize]), + u16::max(a[7usize], b[7usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + [ + u16::min(a[0usize], b[0usize]), + u16::min(a[1usize], b[1usize]), + u16::min(a[2usize], b[2usize]), + u16::min(a[3usize], b[3usize]), + u16::min(a[4usize], b[4usize]), + u16::min(a[5usize], b[5usize]), + u16::min(a[6usize], b[6usize]), + u16::min(a[7usize], b[7usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { [ -(u16::eq(&a[0usize], &b[0usize]) as i16), @@ -3129,34 +3157,6 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] - fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - [ - u16::min(a[0usize], b[0usize]), - u16::min(a[1usize], b[1usize]), - u16::min(a[2usize], b[2usize]), - u16::min(a[3usize], b[3usize]), - u16::min(a[4usize], b[4usize]), - u16::min(a[5usize], b[5usize]), - u16::min(a[6usize], b[6usize]), - u16::min(a[7usize], b[7usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - [ - u16::max(a[0usize], b[0usize]), - u16::max(a[1usize], b[1usize]), - u16::max(a[2usize], b[2usize]), - u16::max(a[3usize], b[3usize]), - u16::max(a[4usize], b[4usize]), - u16::max(a[5usize], b[5usize]), - u16::max(a[6usize], b[6usize]), - u16::max(a[7usize], b[7usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u16x8(self, a: u16x8, b: u16x8) -> u16x16 { let mut result = [0; 16usize]; result[0..8usize].copy_from_slice(&a.val.0); @@ -3664,6 +3664,26 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + [ + i32::max(a[0usize], b[0usize]), + i32::max(a[1usize], b[1usize]), + i32::max(a[2usize], b[2usize]), + i32::max(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + [ + i32::min(a[0usize], b[0usize]), + i32::min(a[1usize], b[1usize]), + i32::min(a[2usize], b[2usize]), + i32::min(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { [ -(i32::eq(&a[0usize], &b[0usize]) as i32), @@ -3744,26 +3764,6 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] - fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - [ - i32::min(a[0usize], b[0usize]), - i32::min(a[1usize], b[1usize]), - i32::min(a[2usize], b[2usize]), - i32::min(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - [ - i32::max(a[0usize], b[0usize]), - i32::max(a[1usize], b[1usize]), - i32::max(a[2usize], b[2usize]), - i32::max(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i32x4(self, a: i32x4, b: i32x4) -> i32x8 { let mut result = [0; 8usize]; result[0..4usize].copy_from_slice(&a.val.0); @@ -3992,6 +3992,26 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + [ + u32::max(a[0usize], b[0usize]), + u32::max(a[1usize], b[1usize]), + u32::max(a[2usize], b[2usize]), + u32::max(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + [ + u32::min(a[0usize], b[0usize]), + u32::min(a[1usize], b[1usize]), + u32::min(a[2usize], b[2usize]), + u32::min(a[3usize], b[3usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u32x4(self, a: u32x4, b: u32x4) -> mask32x4 { [ -(u32::eq(&a[0usize], &b[0usize]) as i32), @@ -4072,26 +4092,6 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] - fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - [ - u32::min(a[0usize], b[0usize]), - u32::min(a[1usize], b[1usize]), - u32::min(a[2usize], b[2usize]), - u32::min(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - [ - u32::max(a[0usize], b[0usize]), - u32::max(a[1usize], b[1usize]), - u32::max(a[2usize], b[2usize]), - u32::max(a[3usize], b[3usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u32x4(self, a: u32x4, b: u32x4) -> u32x8 { let mut result = [0; 8usize]; result[0..4usize].copy_from_slice(&a.val.0); @@ -4399,6 +4399,38 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + [ + f64::max(a[0usize], b[0usize]), + f64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + [ + f64::min(a[0usize], b[0usize]), + f64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + [ + f64::max(a[0usize], b[0usize]), + f64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + [ + f64::min(a[0usize], b[0usize]), + f64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { [ -(f64::eq(&a[0usize], &b[0usize]) as i64), @@ -4447,38 +4479,6 @@ impl Simd for Fallback { (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) } #[inline(always)] - fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - [ - f64::max(a[0usize], b[0usize]), - f64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - [ - f64::min(a[0usize], b[0usize]), - f64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - [ - f64::max(a[0usize], b[0usize]), - f64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - [ - f64::min(a[0usize], b[0usize]), - f64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn mul_add_f64x2(self, a: f64x2, b: f64x2, c: f64x2) -> f64x2 { a.mul(b).add(c) } @@ -4679,6 +4679,22 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::max(a[0usize], b[0usize]), + i64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::min(a[0usize], b[0usize]), + i64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i64x2(self, a: i64x2, b: i64x2) -> mask64x2 { [ -(i64::eq(&a[0usize], &b[0usize]) as i64), @@ -4743,22 +4759,6 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] - fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::min(a[0usize], b[0usize]), - i64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::max(a[0usize], b[0usize]), - i64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i64x2(self, a: i64x2, b: i64x2) -> i64x4 { let mut result = [0; 4usize]; result[0..2usize].copy_from_slice(&a.val.0); @@ -4922,6 +4922,22 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::max(a[0usize], b[0usize]), + u64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::min(a[0usize], b[0usize]), + u64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u64x2(self, a: u64x2, b: u64x2) -> mask64x2 { [ -(u64::eq(&a[0usize], &b[0usize]) as i64), @@ -4986,22 +5002,6 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] - fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::min(a[0usize], b[0usize]), - u64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::max(a[0usize], b[0usize]), - u64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u64x2(self, a: u64x2, b: u64x2) -> u64x4 { let mut result = [0; 4usize]; result[0..2usize].copy_from_slice(&a.val.0); diff --git a/fearless_simd/src/generated/neon.rs b/fearless_simd/src/generated/neon.rs index 826e22cc..ac780430 100644 --- a/fearless_simd/src/generated/neon.rs +++ b/fearless_simd/src/generated/neon.rs @@ -222,6 +222,46 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f32x4, b: f32x4) -> f32x4 { + vmaxq_f32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f32x4, b: f32x4) -> f32x4 { + vminq_f32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f32x4, b: f32x4) -> f32x4 { + vmaxnmq_f32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f32x4, b: f32x4) -> f32x4 { + vminnmq_f32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -308,46 +348,6 @@ impl Simd for Neon { (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) } #[inline(always)] - fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: f32x4, b: f32x4) -> f32x4 { - vmaxq_f32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: f32x4, b: f32x4) -> f32x4 { - vminq_f32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: f32x4, b: f32x4) -> f32x4 { - vmaxnmq_f32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: f32x4, b: f32x4) -> f32x4 { - vminnmq_f32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f32x4(self, a: f32x4, b: f32x4, c: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] @@ -643,6 +643,26 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: i8x16, b: i8x16) -> i8x16 { + vmaxq_s8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: i8x16, b: i8x16) -> i8x16 { + vminq_s8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i8x16(self, a: i8x16, b: i8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -744,26 +764,6 @@ impl Simd for Neon { kernel(self, a, b, c) } #[inline(always)] - fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: i8x16, b: i8x16) -> i8x16 { - vminq_s8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: i8x16, b: i8x16) -> i8x16 { - vmaxq_s8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i8x16(self, a: i8x16, b: i8x16) -> i8x32 { i8x32 { val: crate::support::Aligned256(int8x16x2_t(a.val.0, b.val.0)), @@ -986,6 +986,26 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: u8x16, b: u8x16) -> u8x16 { + vmaxq_u8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: u8x16, b: u8x16) -> u8x16 { + vminq_u8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u8x16(self, a: u8x16, b: u8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -1087,26 +1107,6 @@ impl Simd for Neon { kernel(self, a, b, c) } #[inline(always)] - fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: u8x16, b: u8x16) -> u8x16 { - vminq_u8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: u8x16, b: u8x16) -> u8x16 { - vmaxq_u8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u8x16(self, a: u8x16, b: u8x16) -> u8x32 { u8x32 { val: crate::support::Aligned256(uint8x16x2_t(a.val.0, b.val.0)), @@ -1463,6 +1463,26 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: i16x8, b: i16x8) -> i16x8 { + vmaxq_s16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: i16x8, b: i16x8) -> i16x8 { + vminq_s16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i16x8(self, a: i16x8, b: i16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -1564,26 +1584,6 @@ impl Simd for Neon { kernel(self, a, b, c) } #[inline(always)] - fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: i16x8, b: i16x8) -> i16x8 { - vminq_s16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: i16x8, b: i16x8) -> i16x8 { - vmaxq_s16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i16x8(self, a: i16x8, b: i16x8) -> i16x16 { i16x16 { val: crate::support::Aligned256(int16x8x2_t(a.val.0, b.val.0)), @@ -1805,6 +1805,26 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: u16x8, b: u16x8) -> u16x8 { + vmaxq_u16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: u16x8, b: u16x8) -> u16x8 { + vminq_u16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -1906,26 +1926,6 @@ impl Simd for Neon { kernel(self, a, b, c) } #[inline(always)] - fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: u16x8, b: u16x8) -> u16x8 { - vminq_u16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: u16x8, b: u16x8) -> u16x8 { - vmaxq_u16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u16x8(self, a: u16x8, b: u16x8) -> u16x16 { u16x16 { val: crate::support::Aligned256(uint16x8x2_t(a.val.0, b.val.0)), @@ -2305,6 +2305,26 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: i32x4, b: i32x4) -> i32x4 { + vmaxq_s32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: i32x4, b: i32x4) -> i32x4 { + vminq_s32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -2406,26 +2426,6 @@ impl Simd for Neon { kernel(self, a, b, c) } #[inline(always)] - fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: i32x4, b: i32x4) -> i32x4 { - vminq_s32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: i32x4, b: i32x4) -> i32x4 { - vmaxq_s32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i32x4(self, a: i32x4, b: i32x4) -> i32x8 { i32x8 { val: crate::support::Aligned256(int32x4x2_t(a.val.0, b.val.0)), @@ -2657,6 +2657,26 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: u32x4, b: u32x4) -> u32x4 { + vmaxq_u32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: u32x4, b: u32x4) -> u32x4 { + vminq_u32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u32x4(self, a: u32x4, b: u32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -2758,26 +2778,6 @@ impl Simd for Neon { kernel(self, a, b, c) } #[inline(always)] - fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: u32x4, b: u32x4) -> u32x4 { - vminq_u32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: u32x4, b: u32x4) -> u32x4 { - vmaxq_u32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u32x4(self, a: u32x4, b: u32x4) -> u32x8 { u32x8 { val: crate::support::Aligned256(uint32x4x2_t(a.val.0, b.val.0)), @@ -3147,6 +3147,46 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f64x2, b: f64x2) -> f64x2 { + vmaxq_f64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f64x2, b: f64x2) -> f64x2 { + vminq_f64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f64x2, b: f64x2) -> f64x2 { + vmaxnmq_f64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f64x2, b: f64x2) -> f64x2 { + vminnmq_f64(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -3233,46 +3273,6 @@ impl Simd for Neon { (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) } #[inline(always)] - fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: f64x2, b: f64x2) -> f64x2 { - vmaxq_f64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: f64x2, b: f64x2) -> f64x2 { - vminq_f64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: f64x2, b: f64x2) -> f64x2 { - vmaxnmq_f64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Neon, a: f64x2, b: f64x2) -> f64x2 { - vminnmq_f64(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn mul_add_f64x2(self, a: f64x2, b: f64x2, c: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] @@ -3543,6 +3543,22 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::max(a[0usize], b[0usize]), + i64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::min(a[0usize], b[0usize]), + i64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i64x2(self, a: i64x2, b: i64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -3644,22 +3660,6 @@ impl Simd for Neon { kernel(self, a, b, c) } #[inline(always)] - fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::min(a[0usize], b[0usize]), - i64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::max(a[0usize], b[0usize]), - i64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i64x2(self, a: i64x2, b: i64x2) -> i64x4 { i64x4 { val: crate::support::Aligned256(int64x2x2_t(a.val.0, b.val.0)), @@ -3866,6 +3866,22 @@ impl Simd for Neon { kernel(self, a, b) } #[inline(always)] + fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::max(a[0usize], b[0usize]), + u64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::min(a[0usize], b[0usize]), + u64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u64x2(self, a: u64x2, b: u64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -3967,22 +3983,6 @@ impl Simd for Neon { kernel(self, a, b, c) } #[inline(always)] - fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::min(a[0usize], b[0usize]), - u64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::max(a[0usize], b[0usize]), - u64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u64x2(self, a: u64x2, b: u64x2) -> u64x4 { u64x4 { val: crate::support::Aligned256(uint64x2x2_t(a.val.0, b.val.0)), diff --git a/fearless_simd/src/generated/simd_trait.rs b/fearless_simd/src/generated/simd_trait.rs index 49011255..7ef30ebe 100644 --- a/fearless_simd/src/generated/simd_trait.rs +++ b/fearless_simd/src/generated/simd_trait.rs @@ -255,6 +255,14 @@ pub trait Simd: fn div_f32x4(self, a: f32x4, b: f32x4) -> f32x4; #[doc = "Return a vector with the magnitude of `a` and the sign of `b` for each element.\n\nThis operation copies the sign bit, so if an input element is NaN, the output element will be a NaN with the same payload and a copied sign bit."] fn copysign_f32x4(self, a: f32x4, b: f32x4) -> f32x4; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor integer vectors, this operation is the same as `max`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor integer vectors, this operation is the same as `min`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -283,14 +291,6 @@ pub trait Simd: fn interleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4); #[doc = "Deinterleave two vectors.\n\nThe first result contains all even-indexed elements from `a` followed by all even-indexed elements from `b`. The second result contains all odd-indexed elements from `a` followed by all odd-indexed elements from `b`.\n\nThe reverse of this operation is `interleave`.\n\nFor vectors `[a0, b0, a1, b1]` and `[a2, b2, a3, b3]`, returns `([a0, a1, a2, a3], [b0, b1, b2, b3])`."] fn deinterleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4); - #[doc = "Return the element-wise maximum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4; - #[doc = "Return the element-wise minimum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4; - #[doc = "Return the element-wise maximum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4; - #[doc = "Return the element-wise minimum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4; #[doc = "Compute `(a * b) + c` (fused multiply-add) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by an add, which will result in two rounding errors."] fn mul_add_f32x4(self, a: f32x4, b: f32x4, c: f32x4) -> f32x4; #[doc = "Compute `(a * b) - c` (fused multiply-subtract) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by a subtract, which will result in two rounding errors."] @@ -379,6 +379,10 @@ pub trait Simd: fn shr_i8x16(self, a: i8x16, shift: u32) -> i8x16; #[doc = "Shift each element right by the corresponding element in another vector.\n\nFor unsigned integers, zeros are shifted in on the left. For signed integers, the sign bit is replicated.\n\nWhen shifting out of bounds (e.g. shifting a 32-bit value by 32 or more), the result is implementation-defined and may vary by platform.\n\nThis operation is not implemented in hardware on all platforms. On WebAssembly, and on x86 platforms without AVX2, this will use a fallback scalar implementation."] fn shrv_i8x16(self, a: i8x16, b: i8x16) -> i8x16; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_i8x16(self, a: i8x16, b: i8x16) -> mask8x16; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -409,10 +413,6 @@ pub trait Simd: fn deinterleave_i8x16(self, a: i8x16, b: i8x16) -> (i8x16, i8x16); #[doc = "Select elements from b and c based on the mask operand a.\n\nThis operation's behavior is unspecified if a was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information."] fn select_i8x16(self, a: mask8x16, b: i8x16, c: i8x16) -> i8x16; - #[doc = "Return the element-wise minimum of two vectors."] - fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16; - #[doc = "Return the element-wise maximum of two vectors."] - fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16; #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_i8x16(self, a: i8x16, b: i8x16) -> i8x32; #[doc = "Negate each element of the vector, wrapping on overflow."] @@ -470,6 +470,10 @@ pub trait Simd: fn shr_u8x16(self, a: u8x16, shift: u32) -> u8x16; #[doc = "Shift each element right by the corresponding element in another vector.\n\nFor unsigned integers, zeros are shifted in on the left. For signed integers, the sign bit is replicated.\n\nWhen shifting out of bounds (e.g. shifting a 32-bit value by 32 or more), the result is implementation-defined and may vary by platform.\n\nThis operation is not implemented in hardware on all platforms. On WebAssembly, and on x86 platforms without AVX2, this will use a fallback scalar implementation."] fn shrv_u8x16(self, a: u8x16, b: u8x16) -> u8x16; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_u8x16(self, a: u8x16, b: u8x16) -> mask8x16; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -500,10 +504,6 @@ pub trait Simd: fn deinterleave_u8x16(self, a: u8x16, b: u8x16) -> (u8x16, u8x16); #[doc = "Select elements from b and c based on the mask operand a.\n\nThis operation's behavior is unspecified if a was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information."] fn select_u8x16(self, a: mask8x16, b: u8x16, c: u8x16) -> u8x16; - #[doc = "Return the element-wise minimum of two vectors."] - fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16; - #[doc = "Return the element-wise maximum of two vectors."] - fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16; #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_u8x16(self, a: u8x16, b: u8x16) -> u8x32; #[doc = "Load four 128-bit vectors from an array with 4-way interleaving.\n\nThis is useful e.g. in image processing to turn interleaved RGBA pixels into vectors of each color component.\n\nFor example, with 32-bit lanes, memory laid out as`[r0, g0, b0, a0, r1, g1, b1, a1, r2, g2, b2, a2, r3, g3, b3, a3]` loads as`[[r0, r1, r2, r3], [g0, g1, g2, g3], [b0, b1, b2, b3], [a0, a1, a2, a3]]`."] @@ -603,6 +603,10 @@ pub trait Simd: fn shr_i16x8(self, a: i16x8, shift: u32) -> i16x8; #[doc = "Shift each element right by the corresponding element in another vector.\n\nFor unsigned integers, zeros are shifted in on the left. For signed integers, the sign bit is replicated.\n\nWhen shifting out of bounds (e.g. shifting a 32-bit value by 32 or more), the result is implementation-defined and may vary by platform.\n\nThis operation is not implemented in hardware on all platforms. On WebAssembly, and on x86 platforms without AVX2, this will use a fallback scalar implementation."] fn shrv_i16x8(self, a: i16x8, b: i16x8) -> i16x8; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_i16x8(self, a: i16x8, b: i16x8) -> mask16x8; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -633,10 +637,6 @@ pub trait Simd: fn deinterleave_i16x8(self, a: i16x8, b: i16x8) -> (i16x8, i16x8); #[doc = "Select elements from b and c based on the mask operand a.\n\nThis operation's behavior is unspecified if a was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information."] fn select_i16x8(self, a: mask16x8, b: i16x8, c: i16x8) -> i16x8; - #[doc = "Return the element-wise minimum of two vectors."] - fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8; - #[doc = "Return the element-wise maximum of two vectors."] - fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8; #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_i16x8(self, a: i16x8, b: i16x8) -> i16x16; #[doc = "Negate each element of the vector, wrapping on overflow."] @@ -709,6 +709,10 @@ pub trait Simd: fn shr_u16x8(self, a: u16x8, shift: u32) -> u16x8; #[doc = "Shift each element right by the corresponding element in another vector.\n\nFor unsigned integers, zeros are shifted in on the left. For signed integers, the sign bit is replicated.\n\nWhen shifting out of bounds (e.g. shifting a 32-bit value by 32 or more), the result is implementation-defined and may vary by platform.\n\nThis operation is not implemented in hardware on all platforms. On WebAssembly, and on x86 platforms without AVX2, this will use a fallback scalar implementation."] fn shrv_u16x8(self, a: u16x8, b: u16x8) -> u16x8; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -739,10 +743,6 @@ pub trait Simd: fn deinterleave_u16x8(self, a: u16x8, b: u16x8) -> (u16x8, u16x8); #[doc = "Select elements from b and c based on the mask operand a.\n\nThis operation's behavior is unspecified if a was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information."] fn select_u16x8(self, a: mask16x8, b: u16x8, c: u16x8) -> u16x8; - #[doc = "Return the element-wise minimum of two vectors."] - fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8; - #[doc = "Return the element-wise maximum of two vectors."] - fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8; #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_u16x8(self, a: u16x8, b: u16x8) -> u16x16; #[doc = "Load four 128-bit vectors from an array with 4-way interleaving.\n\nThis is useful e.g. in image processing to turn interleaved RGBA pixels into vectors of each color component.\n\nFor example, with 32-bit lanes, memory laid out as`[r0, g0, b0, a0, r1, g1, b1, a1, r2, g2, b2, a2, r3, g3, b3, a3]` loads as`[[r0, r1, r2, r3], [g0, g1, g2, g3], [b0, b1, b2, b3], [a0, a1, a2, a3]]`."] @@ -848,6 +848,10 @@ pub trait Simd: fn shr_i32x4(self, a: i32x4, shift: u32) -> i32x4; #[doc = "Shift each element right by the corresponding element in another vector.\n\nFor unsigned integers, zeros are shifted in on the left. For signed integers, the sign bit is replicated.\n\nWhen shifting out of bounds (e.g. shifting a 32-bit value by 32 or more), the result is implementation-defined and may vary by platform.\n\nThis operation is not implemented in hardware on all platforms. On WebAssembly, and on x86 platforms without AVX2, this will use a fallback scalar implementation."] fn shrv_i32x4(self, a: i32x4, b: i32x4) -> i32x4; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -878,10 +882,6 @@ pub trait Simd: fn deinterleave_i32x4(self, a: i32x4, b: i32x4) -> (i32x4, i32x4); #[doc = "Select elements from b and c based on the mask operand a.\n\nThis operation's behavior is unspecified if a was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information."] fn select_i32x4(self, a: mask32x4, b: i32x4, c: i32x4) -> i32x4; - #[doc = "Return the element-wise minimum of two vectors."] - fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4; - #[doc = "Return the element-wise maximum of two vectors."] - fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4; #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_i32x4(self, a: i32x4, b: i32x4) -> i32x8; #[doc = "Negate each element of the vector, wrapping on overflow."] @@ -956,6 +956,10 @@ pub trait Simd: fn shr_u32x4(self, a: u32x4, shift: u32) -> u32x4; #[doc = "Shift each element right by the corresponding element in another vector.\n\nFor unsigned integers, zeros are shifted in on the left. For signed integers, the sign bit is replicated.\n\nWhen shifting out of bounds (e.g. shifting a 32-bit value by 32 or more), the result is implementation-defined and may vary by platform.\n\nThis operation is not implemented in hardware on all platforms. On WebAssembly, and on x86 platforms without AVX2, this will use a fallback scalar implementation."] fn shrv_u32x4(self, a: u32x4, b: u32x4) -> u32x4; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_u32x4(self, a: u32x4, b: u32x4) -> mask32x4; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -986,10 +990,6 @@ pub trait Simd: fn deinterleave_u32x4(self, a: u32x4, b: u32x4) -> (u32x4, u32x4); #[doc = "Select elements from b and c based on the mask operand a.\n\nThis operation's behavior is unspecified if a was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information."] fn select_u32x4(self, a: mask32x4, b: u32x4, c: u32x4) -> u32x4; - #[doc = "Return the element-wise minimum of two vectors."] - fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4; - #[doc = "Return the element-wise maximum of two vectors."] - fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4; #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_u32x4(self, a: u32x4, b: u32x4) -> u32x8; #[doc = "Load four 128-bit vectors from an array with 4-way interleaving.\n\nThis is useful e.g. in image processing to turn interleaved RGBA pixels into vectors of each color component.\n\nFor example, with 32-bit lanes, memory laid out as`[r0, g0, b0, a0, r1, g1, b1, a1, r2, g2, b2, a2, r3, g3, b3, a3]` loads as`[[r0, r1, r2, r3], [g0, g1, g2, g3], [b0, b1, b2, b3], [a0, a1, a2, a3]]`."] @@ -1093,6 +1093,14 @@ pub trait Simd: fn div_f64x2(self, a: f64x2, b: f64x2) -> f64x2; #[doc = "Return a vector with the magnitude of `a` and the sign of `b` for each element.\n\nThis operation copies the sign bit, so if an input element is NaN, the output element will be a NaN with the same payload and a copied sign bit."] fn copysign_f64x2(self, a: f64x2, b: f64x2) -> f64x2; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor integer vectors, this operation is the same as `max`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor integer vectors, this operation is the same as `min`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -1121,14 +1129,6 @@ pub trait Simd: fn interleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2); #[doc = "Deinterleave two vectors.\n\nThe first result contains all even-indexed elements from `a` followed by all even-indexed elements from `b`. The second result contains all odd-indexed elements from `a` followed by all odd-indexed elements from `b`.\n\nThe reverse of this operation is `interleave`.\n\nFor vectors `[a0, b0, a1, b1]` and `[a2, b2, a3, b3]`, returns `([a0, a1, a2, a3], [b0, b1, b2, b3])`."] fn deinterleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2); - #[doc = "Return the element-wise maximum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2; - #[doc = "Return the element-wise minimum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2; - #[doc = "Return the element-wise maximum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2; - #[doc = "Return the element-wise minimum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2; #[doc = "Compute `(a * b) + c` (fused multiply-add) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by an add, which will result in two rounding errors."] fn mul_add_f64x2(self, a: f64x2, b: f64x2, c: f64x2) -> f64x2; #[doc = "Compute `(a * b) - c` (fused multiply-subtract) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by a subtract, which will result in two rounding errors."] @@ -1213,6 +1213,10 @@ pub trait Simd: fn shr_i64x2(self, a: i64x2, shift: u32) -> i64x2; #[doc = "Shift each element right by the corresponding element in another vector.\n\nFor unsigned integers, zeros are shifted in on the left. For signed integers, the sign bit is replicated.\n\nWhen shifting out of bounds (e.g. shifting a 32-bit value by 32 or more), the result is implementation-defined and may vary by platform.\n\nThis operation is not implemented in hardware on all platforms. On WebAssembly, and on x86 platforms without AVX2, this will use a fallback scalar implementation."] fn shrv_i64x2(self, a: i64x2, b: i64x2) -> i64x2; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_i64x2(self, a: i64x2, b: i64x2) -> mask64x2; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -1243,10 +1247,6 @@ pub trait Simd: fn deinterleave_i64x2(self, a: i64x2, b: i64x2) -> (i64x2, i64x2); #[doc = "Select elements from b and c based on the mask operand a.\n\nThis operation's behavior is unspecified if a was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information."] fn select_i64x2(self, a: mask64x2, b: i64x2, c: i64x2) -> i64x2; - #[doc = "Return the element-wise minimum of two vectors."] - fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2; - #[doc = "Return the element-wise maximum of two vectors."] - fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2; #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_i64x2(self, a: i64x2, b: i64x2) -> i64x4; #[doc = "Negate each element of the vector, wrapping on overflow."] @@ -1317,6 +1317,10 @@ pub trait Simd: fn shr_u64x2(self, a: u64x2, shift: u32) -> u64x2; #[doc = "Shift each element right by the corresponding element in another vector.\n\nFor unsigned integers, zeros are shifted in on the left. For signed integers, the sign bit is replicated.\n\nWhen shifting out of bounds (e.g. shifting a 32-bit value by 32 or more), the result is implementation-defined and may vary by platform.\n\nThis operation is not implemented in hardware on all platforms. On WebAssembly, and on x86 platforms without AVX2, this will use a fallback scalar implementation."] fn shrv_u64x2(self, a: u64x2, b: u64x2) -> u64x2; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq_u64x2(self, a: u64x2, b: u64x2) -> mask64x2; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] @@ -1347,10 +1351,6 @@ pub trait Simd: fn deinterleave_u64x2(self, a: u64x2, b: u64x2) -> (u64x2, u64x2); #[doc = "Select elements from b and c based on the mask operand a.\n\nThis operation's behavior is unspecified if a was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information."] fn select_u64x2(self, a: mask64x2, b: u64x2, c: u64x2) -> u64x2; - #[doc = "Return the element-wise minimum of two vectors."] - fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2; - #[doc = "Return the element-wise maximum of two vectors."] - fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2; #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_u64x2(self, a: u64x2, b: u64x2) -> u64x4; #[doc = "Load four 128-bit vectors from an array with 4-way interleaving.\n\nThis is useful e.g. in image processing to turn interleaved RGBA pixels into vectors of each color component.\n\nFor example, with 32-bit lanes, memory laid out as`[r0, g0, b0, a0, r1, g1, b1, a1, r2, g2, b2, a2, r3, g3, b3, a3]` loads as`[[r0, r1, r2, r3], [g0, g1, g2, g3], [b0, b1, b2, b3], [a0, a1, a2, a3]]`."] @@ -1503,6 +1503,40 @@ pub trait Simd: let (b0, b1) = self.split_f32x8(b); self.combine_f32x4(self.copysign_f32x4(a0, b0), self.copysign_f32x4(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + let (a0, a1) = self.split_f32x8(a); + let (b0, b1) = self.split_f32x8(b); + self.combine_f32x4(self.max_f32x4(a0, b0), self.max_f32x4(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + let (a0, a1) = self.split_f32x8(a); + let (b0, b1) = self.split_f32x8(b); + self.combine_f32x4(self.min_f32x4(a0, b0), self.min_f32x4(a1, b1)) + } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor integer vectors, this operation is the same as `max`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + #[inline(always)] + fn max_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + let (a0, a1) = self.split_f32x8(a); + let (b0, b1) = self.split_f32x8(b); + self.combine_f32x4( + self.max_precise_f32x4(a0, b0), + self.max_precise_f32x4(a1, b1), + ) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor integer vectors, this operation is the same as `min`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + #[inline(always)] + fn min_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { + let (a0, a1) = self.split_f32x8(a); + let (b0, b1) = self.split_f32x8(b); + self.combine_f32x4( + self.min_precise_f32x4(a0, b0), + self.min_precise_f32x4(a1, b1), + ) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_f32x8(self, a: f32x8, b: f32x8) -> mask32x8 { @@ -1590,40 +1624,6 @@ pub trait Simd: self.combine_f32x4(lo_odd, hi_odd), ) } - #[doc = "Return the element-wise maximum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - #[inline(always)] - fn max_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - let (a0, a1) = self.split_f32x8(a); - let (b0, b1) = self.split_f32x8(b); - self.combine_f32x4(self.max_f32x4(a0, b0), self.max_f32x4(a1, b1)) - } - #[doc = "Return the element-wise minimum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - #[inline(always)] - fn min_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - let (a0, a1) = self.split_f32x8(a); - let (b0, b1) = self.split_f32x8(b); - self.combine_f32x4(self.min_f32x4(a0, b0), self.min_f32x4(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - #[inline(always)] - fn max_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - let (a0, a1) = self.split_f32x8(a); - let (b0, b1) = self.split_f32x8(b); - self.combine_f32x4( - self.max_precise_f32x4(a0, b0), - self.max_precise_f32x4(a1, b1), - ) - } - #[doc = "Return the element-wise minimum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - #[inline(always)] - fn min_precise_f32x8(self, a: f32x8, b: f32x8) -> f32x8 { - let (a0, a1) = self.split_f32x8(a); - let (b0, b1) = self.split_f32x8(b); - self.combine_f32x4( - self.min_precise_f32x4(a0, b0), - self.min_precise_f32x4(a1, b1), - ) - } #[doc = "Compute `(a * b) + c` (fused multiply-add) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by an add, which will result in two rounding errors."] #[inline(always)] fn mul_add_f32x8(self, a: f32x8, b: f32x8, c: f32x8) -> f32x8 { @@ -1840,6 +1840,20 @@ pub trait Simd: let (b0, b1) = self.split_i8x32(b); self.combine_i8x16(self.shrv_i8x16(a0, b0), self.shrv_i8x16(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { + let (a0, a1) = self.split_i8x32(a); + let (b0, b1) = self.split_i8x32(b); + self.combine_i8x16(self.max_i8x16(a0, b0), self.max_i8x16(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { + let (a0, a1) = self.split_i8x32(a); + let (b0, b1) = self.split_i8x32(b); + self.combine_i8x16(self.min_i8x16(a0, b0), self.min_i8x16(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_i8x32(self, a: i8x32, b: i8x32) -> mask8x32 { @@ -1935,20 +1949,6 @@ pub trait Simd: let (c0, c1) = self.split_i8x32(c); self.combine_i8x16(self.select_i8x16(a0, b0, c0), self.select_i8x16(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { - let (a0, a1) = self.split_i8x32(a); - let (b0, b1) = self.split_i8x32(b); - self.combine_i8x16(self.min_i8x16(a0, b0), self.min_i8x16(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_i8x32(self, a: i8x32, b: i8x32) -> i8x32 { - let (a0, a1) = self.split_i8x32(a); - let (b0, b1) = self.split_i8x32(b); - self.combine_i8x16(self.max_i8x16(a0, b0), self.max_i8x16(a1, b1)) - } #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_i8x32(self, a: i8x32, b: i8x32) -> i8x64; #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] @@ -2077,6 +2077,20 @@ pub trait Simd: let (b0, b1) = self.split_u8x32(b); self.combine_u8x16(self.shrv_u8x16(a0, b0), self.shrv_u8x16(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { + let (a0, a1) = self.split_u8x32(a); + let (b0, b1) = self.split_u8x32(b); + self.combine_u8x16(self.max_u8x16(a0, b0), self.max_u8x16(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { + let (a0, a1) = self.split_u8x32(a); + let (b0, b1) = self.split_u8x32(b); + self.combine_u8x16(self.min_u8x16(a0, b0), self.min_u8x16(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_u8x32(self, a: u8x32, b: u8x32) -> mask8x32 { @@ -2172,20 +2186,6 @@ pub trait Simd: let (c0, c1) = self.split_u8x32(c); self.combine_u8x16(self.select_u8x16(a0, b0, c0), self.select_u8x16(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { - let (a0, a1) = self.split_u8x32(a); - let (b0, b1) = self.split_u8x32(b); - self.combine_u8x16(self.min_u8x16(a0, b0), self.min_u8x16(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_u8x32(self, a: u8x32, b: u8x32) -> u8x32 { - let (a0, a1) = self.split_u8x32(a); - let (b0, b1) = self.split_u8x32(b); - self.combine_u8x16(self.max_u8x16(a0, b0), self.max_u8x16(a1, b1)) - } #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_u8x32(self, a: u8x32, b: u8x32) -> u8x64; #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] @@ -2414,6 +2414,20 @@ pub trait Simd: let (b0, b1) = self.split_i16x16(b); self.combine_i16x8(self.shrv_i16x8(a0, b0), self.shrv_i16x8(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { + let (a0, a1) = self.split_i16x16(a); + let (b0, b1) = self.split_i16x16(b); + self.combine_i16x8(self.max_i16x8(a0, b0), self.max_i16x8(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { + let (a0, a1) = self.split_i16x16(a); + let (b0, b1) = self.split_i16x16(b); + self.combine_i16x8(self.min_i16x8(a0, b0), self.min_i16x8(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_i16x16(self, a: i16x16, b: i16x16) -> mask16x16 { @@ -2509,20 +2523,6 @@ pub trait Simd: let (c0, c1) = self.split_i16x16(c); self.combine_i16x8(self.select_i16x8(a0, b0, c0), self.select_i16x8(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { - let (a0, a1) = self.split_i16x16(a); - let (b0, b1) = self.split_i16x16(b); - self.combine_i16x8(self.min_i16x8(a0, b0), self.min_i16x8(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_i16x16(self, a: i16x16, b: i16x16) -> i16x16 { - let (a0, a1) = self.split_i16x16(a); - let (b0, b1) = self.split_i16x16(b); - self.combine_i16x8(self.max_i16x8(a0, b0), self.max_i16x8(a1, b1)) - } #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_i16x16(self, a: i16x16, b: i16x16) -> i16x32; #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] @@ -2683,6 +2683,20 @@ pub trait Simd: let (b0, b1) = self.split_u16x16(b); self.combine_u16x8(self.shrv_u16x8(a0, b0), self.shrv_u16x8(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { + let (a0, a1) = self.split_u16x16(a); + let (b0, b1) = self.split_u16x16(b); + self.combine_u16x8(self.max_u16x8(a0, b0), self.max_u16x8(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { + let (a0, a1) = self.split_u16x16(a); + let (b0, b1) = self.split_u16x16(b); + self.combine_u16x8(self.min_u16x8(a0, b0), self.min_u16x8(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_u16x16(self, a: u16x16, b: u16x16) -> mask16x16 { @@ -2778,20 +2792,6 @@ pub trait Simd: let (c0, c1) = self.split_u16x16(c); self.combine_u16x8(self.select_u16x8(a0, b0, c0), self.select_u16x8(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { - let (a0, a1) = self.split_u16x16(a); - let (b0, b1) = self.split_u16x16(b); - self.combine_u16x8(self.min_u16x8(a0, b0), self.min_u16x8(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_u16x16(self, a: u16x16, b: u16x16) -> u16x16 { - let (a0, a1) = self.split_u16x16(a); - let (b0, b1) = self.split_u16x16(b); - self.combine_u16x8(self.max_u16x8(a0, b0), self.max_u16x8(a1, b1)) - } #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_u16x16(self, a: u16x16, b: u16x16) -> u16x32; #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] @@ -3043,6 +3043,20 @@ pub trait Simd: let (b0, b1) = self.split_i32x8(b); self.combine_i32x4(self.shrv_i32x4(a0, b0), self.shrv_i32x4(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { + let (a0, a1) = self.split_i32x8(a); + let (b0, b1) = self.split_i32x8(b); + self.combine_i32x4(self.max_i32x4(a0, b0), self.max_i32x4(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { + let (a0, a1) = self.split_i32x8(a); + let (b0, b1) = self.split_i32x8(b); + self.combine_i32x4(self.min_i32x4(a0, b0), self.min_i32x4(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_i32x8(self, a: i32x8, b: i32x8) -> mask32x8 { @@ -3138,20 +3152,6 @@ pub trait Simd: let (c0, c1) = self.split_i32x8(c); self.combine_i32x4(self.select_i32x4(a0, b0, c0), self.select_i32x4(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { - let (a0, a1) = self.split_i32x8(a); - let (b0, b1) = self.split_i32x8(b); - self.combine_i32x4(self.min_i32x4(a0, b0), self.min_i32x4(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_i32x8(self, a: i32x8, b: i32x8) -> i32x8 { - let (a0, a1) = self.split_i32x8(a); - let (b0, b1) = self.split_i32x8(b); - self.combine_i32x4(self.max_i32x4(a0, b0), self.max_i32x4(a1, b1)) - } #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_i32x8(self, a: i32x8, b: i32x8) -> i32x16; #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] @@ -3314,6 +3314,20 @@ pub trait Simd: let (b0, b1) = self.split_u32x8(b); self.combine_u32x4(self.shrv_u32x4(a0, b0), self.shrv_u32x4(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { + let (a0, a1) = self.split_u32x8(a); + let (b0, b1) = self.split_u32x8(b); + self.combine_u32x4(self.max_u32x4(a0, b0), self.max_u32x4(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { + let (a0, a1) = self.split_u32x8(a); + let (b0, b1) = self.split_u32x8(b); + self.combine_u32x4(self.min_u32x4(a0, b0), self.min_u32x4(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_u32x8(self, a: u32x8, b: u32x8) -> mask32x8 { @@ -3409,20 +3423,6 @@ pub trait Simd: let (c0, c1) = self.split_u32x8(c); self.combine_u32x4(self.select_u32x4(a0, b0, c0), self.select_u32x4(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { - let (a0, a1) = self.split_u32x8(a); - let (b0, b1) = self.split_u32x8(b); - self.combine_u32x4(self.min_u32x4(a0, b0), self.min_u32x4(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_u32x8(self, a: u32x8, b: u32x8) -> u32x8 { - let (a0, a1) = self.split_u32x8(a); - let (b0, b1) = self.split_u32x8(b); - self.combine_u32x4(self.max_u32x4(a0, b0), self.max_u32x4(a1, b1)) - } #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_u32x8(self, a: u32x8, b: u32x8) -> u32x16; #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] @@ -3668,12 +3668,46 @@ pub trait Simd: let (b0, b1) = self.split_f64x4(b); self.combine_f64x2(self.copysign_f64x2(a0, b0), self.copysign_f64x2(a1, b1)) } - #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] #[inline(always)] - fn simd_eq_f64x4(self, a: f64x4, b: f64x4) -> mask64x4 { + fn max_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { let (a0, a1) = self.split_f64x4(a); let (b0, b1) = self.split_f64x4(b); - self.combine_mask64x2(self.simd_eq_f64x2(a0, b0), self.simd_eq_f64x2(a1, b1)) + self.combine_f64x2(self.max_f64x2(a0, b0), self.max_f64x2(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + let (a0, a1) = self.split_f64x4(a); + let (b0, b1) = self.split_f64x4(b); + self.combine_f64x2(self.min_f64x2(a0, b0), self.min_f64x2(a1, b1)) + } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor integer vectors, this operation is the same as `max`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + #[inline(always)] + fn max_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + let (a0, a1) = self.split_f64x4(a); + let (b0, b1) = self.split_f64x4(b); + self.combine_f64x2( + self.max_precise_f64x2(a0, b0), + self.max_precise_f64x2(a1, b1), + ) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor integer vectors, this operation is the same as `min`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + #[inline(always)] + fn min_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { + let (a0, a1) = self.split_f64x4(a); + let (b0, b1) = self.split_f64x4(b); + self.combine_f64x2( + self.min_precise_f64x2(a0, b0), + self.min_precise_f64x2(a1, b1), + ) + } + #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] + #[inline(always)] + fn simd_eq_f64x4(self, a: f64x4, b: f64x4) -> mask64x4 { + let (a0, a1) = self.split_f64x4(a); + let (b0, b1) = self.split_f64x4(b); + self.combine_mask64x2(self.simd_eq_f64x2(a0, b0), self.simd_eq_f64x2(a1, b1)) } #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `a` is less than `b`, and false if not."] #[inline(always)] @@ -3755,40 +3789,6 @@ pub trait Simd: self.combine_f64x2(lo_odd, hi_odd), ) } - #[doc = "Return the element-wise maximum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - #[inline(always)] - fn max_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - let (a0, a1) = self.split_f64x4(a); - let (b0, b1) = self.split_f64x4(b); - self.combine_f64x2(self.max_f64x2(a0, b0), self.max_f64x2(a1, b1)) - } - #[doc = "Return the element-wise minimum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - #[inline(always)] - fn min_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - let (a0, a1) = self.split_f64x4(a); - let (b0, b1) = self.split_f64x4(b); - self.combine_f64x2(self.min_f64x2(a0, b0), self.min_f64x2(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - #[inline(always)] - fn max_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - let (a0, a1) = self.split_f64x4(a); - let (b0, b1) = self.split_f64x4(b); - self.combine_f64x2( - self.max_precise_f64x2(a0, b0), - self.max_precise_f64x2(a1, b1), - ) - } - #[doc = "Return the element-wise minimum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - #[inline(always)] - fn min_precise_f64x4(self, a: f64x4, b: f64x4) -> f64x4 { - let (a0, a1) = self.split_f64x4(a); - let (b0, b1) = self.split_f64x4(b); - self.combine_f64x2( - self.min_precise_f64x2(a0, b0), - self.min_precise_f64x2(a1, b1), - ) - } #[doc = "Compute `(a * b) + c` (fused multiply-add) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by an add, which will result in two rounding errors."] #[inline(always)] fn mul_add_f64x4(self, a: f64x4, b: f64x4, c: f64x4) -> f64x4 { @@ -3994,6 +3994,20 @@ pub trait Simd: let (b0, b1) = self.split_i64x4(b); self.combine_i64x2(self.shrv_i64x2(a0, b0), self.shrv_i64x2(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { + let (a0, a1) = self.split_i64x4(a); + let (b0, b1) = self.split_i64x4(b); + self.combine_i64x2(self.max_i64x2(a0, b0), self.max_i64x2(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { + let (a0, a1) = self.split_i64x4(a); + let (b0, b1) = self.split_i64x4(b); + self.combine_i64x2(self.min_i64x2(a0, b0), self.min_i64x2(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_i64x4(self, a: i64x4, b: i64x4) -> mask64x4 { @@ -4089,20 +4103,6 @@ pub trait Simd: let (c0, c1) = self.split_i64x4(c); self.combine_i64x2(self.select_i64x2(a0, b0, c0), self.select_i64x2(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { - let (a0, a1) = self.split_i64x4(a); - let (b0, b1) = self.split_i64x4(b); - self.combine_i64x2(self.min_i64x2(a0, b0), self.min_i64x2(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_i64x4(self, a: i64x4, b: i64x4) -> i64x4 { - let (a0, a1) = self.split_i64x4(a); - let (b0, b1) = self.split_i64x4(b); - self.combine_i64x2(self.max_i64x2(a0, b0), self.max_i64x2(a1, b1)) - } #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_i64x4(self, a: i64x4, b: i64x4) -> i64x8; #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] @@ -4251,6 +4251,20 @@ pub trait Simd: let (b0, b1) = self.split_u64x4(b); self.combine_u64x2(self.shrv_u64x2(a0, b0), self.shrv_u64x2(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { + let (a0, a1) = self.split_u64x4(a); + let (b0, b1) = self.split_u64x4(b); + self.combine_u64x2(self.max_u64x2(a0, b0), self.max_u64x2(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { + let (a0, a1) = self.split_u64x4(a); + let (b0, b1) = self.split_u64x4(b); + self.combine_u64x2(self.min_u64x2(a0, b0), self.min_u64x2(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_u64x4(self, a: u64x4, b: u64x4) -> mask64x4 { @@ -4346,20 +4360,6 @@ pub trait Simd: let (c0, c1) = self.split_u64x4(c); self.combine_u64x2(self.select_u64x2(a0, b0, c0), self.select_u64x2(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { - let (a0, a1) = self.split_u64x4(a); - let (b0, b1) = self.split_u64x4(b); - self.combine_u64x2(self.min_u64x2(a0, b0), self.min_u64x2(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_u64x4(self, a: u64x4, b: u64x4) -> u64x4 { - let (a0, a1) = self.split_u64x4(a); - let (b0, b1) = self.split_u64x4(b); - self.combine_u64x2(self.max_u64x2(a0, b0), self.max_u64x2(a1, b1)) - } #[doc = "Combine two vectors into a single vector with twice the width.\n\n`a` provides the lower elements and `b` provides the upper elements."] fn combine_u64x4(self, a: u64x4, b: u64x4) -> u64x8; #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] @@ -4595,6 +4595,40 @@ pub trait Simd: let (b0, b1) = self.split_f32x16(b); self.combine_f32x8(self.copysign_f32x8(a0, b0), self.copysign_f32x8(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { + let (a0, a1) = self.split_f32x16(a); + let (b0, b1) = self.split_f32x16(b); + self.combine_f32x8(self.max_f32x8(a0, b0), self.max_f32x8(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { + let (a0, a1) = self.split_f32x16(a); + let (b0, b1) = self.split_f32x16(b); + self.combine_f32x8(self.min_f32x8(a0, b0), self.min_f32x8(a1, b1)) + } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor integer vectors, this operation is the same as `max`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + #[inline(always)] + fn max_precise_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { + let (a0, a1) = self.split_f32x16(a); + let (b0, b1) = self.split_f32x16(b); + self.combine_f32x8( + self.max_precise_f32x8(a0, b0), + self.max_precise_f32x8(a1, b1), + ) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor integer vectors, this operation is the same as `min`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + #[inline(always)] + fn min_precise_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { + let (a0, a1) = self.split_f32x16(a); + let (b0, b1) = self.split_f32x16(b); + self.combine_f32x8( + self.min_precise_f32x8(a0, b0), + self.min_precise_f32x8(a1, b1), + ) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_f32x16(self, a: f32x16, b: f32x16) -> mask32x16 { @@ -4682,40 +4716,6 @@ pub trait Simd: self.combine_f32x8(lo_odd, hi_odd), ) } - #[doc = "Return the element-wise maximum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - #[inline(always)] - fn max_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { - let (a0, a1) = self.split_f32x16(a); - let (b0, b1) = self.split_f32x16(b); - self.combine_f32x8(self.max_f32x8(a0, b0), self.max_f32x8(a1, b1)) - } - #[doc = "Return the element-wise minimum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - #[inline(always)] - fn min_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { - let (a0, a1) = self.split_f32x16(a); - let (b0, b1) = self.split_f32x16(b); - self.combine_f32x8(self.min_f32x8(a0, b0), self.min_f32x8(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - #[inline(always)] - fn max_precise_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { - let (a0, a1) = self.split_f32x16(a); - let (b0, b1) = self.split_f32x16(b); - self.combine_f32x8( - self.max_precise_f32x8(a0, b0), - self.max_precise_f32x8(a1, b1), - ) - } - #[doc = "Return the element-wise minimum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - #[inline(always)] - fn min_precise_f32x16(self, a: f32x16, b: f32x16) -> f32x16 { - let (a0, a1) = self.split_f32x16(a); - let (b0, b1) = self.split_f32x16(b); - self.combine_f32x8( - self.min_precise_f32x8(a0, b0), - self.min_precise_f32x8(a1, b1), - ) - } #[doc = "Compute `(a * b) + c` (fused multiply-add) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by an add, which will result in two rounding errors."] #[inline(always)] fn mul_add_f32x16(self, a: f32x16, b: f32x16, c: f32x16) -> f32x16 { @@ -4930,6 +4930,20 @@ pub trait Simd: let (b0, b1) = self.split_i8x64(b); self.combine_i8x32(self.shrv_i8x32(a0, b0), self.shrv_i8x32(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_i8x64(self, a: i8x64, b: i8x64) -> i8x64 { + let (a0, a1) = self.split_i8x64(a); + let (b0, b1) = self.split_i8x64(b); + self.combine_i8x32(self.max_i8x32(a0, b0), self.max_i8x32(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_i8x64(self, a: i8x64, b: i8x64) -> i8x64 { + let (a0, a1) = self.split_i8x64(a); + let (b0, b1) = self.split_i8x64(b); + self.combine_i8x32(self.min_i8x32(a0, b0), self.min_i8x32(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_i8x64(self, a: i8x64, b: i8x64) -> mask8x64 { @@ -5025,20 +5039,6 @@ pub trait Simd: let (c0, c1) = self.split_i8x64(c); self.combine_i8x32(self.select_i8x32(a0, b0, c0), self.select_i8x32(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_i8x64(self, a: i8x64, b: i8x64) -> i8x64 { - let (a0, a1) = self.split_i8x64(a); - let (b0, b1) = self.split_i8x64(b); - self.combine_i8x32(self.min_i8x32(a0, b0), self.min_i8x32(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_i8x64(self, a: i8x64, b: i8x64) -> i8x64 { - let (a0, a1) = self.split_i8x64(a); - let (b0, b1) = self.split_i8x64(b); - self.combine_i8x32(self.max_i8x32(a0, b0), self.max_i8x32(a1, b1)) - } #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] fn split_i8x64(self, a: i8x64) -> (i8x32, i8x32); #[doc = "Negate each element of the vector, wrapping on overflow."] @@ -5165,6 +5165,20 @@ pub trait Simd: let (b0, b1) = self.split_u8x64(b); self.combine_u8x32(self.shrv_u8x32(a0, b0), self.shrv_u8x32(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_u8x64(self, a: u8x64, b: u8x64) -> u8x64 { + let (a0, a1) = self.split_u8x64(a); + let (b0, b1) = self.split_u8x64(b); + self.combine_u8x32(self.max_u8x32(a0, b0), self.max_u8x32(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_u8x64(self, a: u8x64, b: u8x64) -> u8x64 { + let (a0, a1) = self.split_u8x64(a); + let (b0, b1) = self.split_u8x64(b); + self.combine_u8x32(self.min_u8x32(a0, b0), self.min_u8x32(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_u8x64(self, a: u8x64, b: u8x64) -> mask8x64 { @@ -5260,20 +5274,6 @@ pub trait Simd: let (c0, c1) = self.split_u8x64(c); self.combine_u8x32(self.select_u8x32(a0, b0, c0), self.select_u8x32(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_u8x64(self, a: u8x64, b: u8x64) -> u8x64 { - let (a0, a1) = self.split_u8x64(a); - let (b0, b1) = self.split_u8x64(b); - self.combine_u8x32(self.min_u8x32(a0, b0), self.min_u8x32(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_u8x64(self, a: u8x64, b: u8x64) -> u8x64 { - let (a0, a1) = self.split_u8x64(a); - let (b0, b1) = self.split_u8x64(b); - self.combine_u8x32(self.max_u8x32(a0, b0), self.max_u8x32(a1, b1)) - } #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] fn split_u8x64(self, a: u8x64) -> (u8x32, u8x32); #[doc = "Widen every lane into two same-width vectors.\n\nThe first result contains the widened lower lanes and the second contains the widened upper lanes."] @@ -5498,6 +5498,20 @@ pub trait Simd: let (b0, b1) = self.split_i16x32(b); self.combine_i16x16(self.shrv_i16x16(a0, b0), self.shrv_i16x16(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_i16x32(self, a: i16x32, b: i16x32) -> i16x32 { + let (a0, a1) = self.split_i16x32(a); + let (b0, b1) = self.split_i16x32(b); + self.combine_i16x16(self.max_i16x16(a0, b0), self.max_i16x16(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_i16x32(self, a: i16x32, b: i16x32) -> i16x32 { + let (a0, a1) = self.split_i16x32(a); + let (b0, b1) = self.split_i16x32(b); + self.combine_i16x16(self.min_i16x16(a0, b0), self.min_i16x16(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_i16x32(self, a: i16x32, b: i16x32) -> mask16x32 { @@ -5599,20 +5613,6 @@ pub trait Simd: self.select_i16x16(a1, b1, c1), ) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_i16x32(self, a: i16x32, b: i16x32) -> i16x32 { - let (a0, a1) = self.split_i16x32(a); - let (b0, b1) = self.split_i16x32(b); - self.combine_i16x16(self.min_i16x16(a0, b0), self.min_i16x16(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_i16x32(self, a: i16x32, b: i16x32) -> i16x32 { - let (a0, a1) = self.split_i16x32(a); - let (b0, b1) = self.split_i16x32(b); - self.combine_i16x16(self.max_i16x16(a0, b0), self.max_i16x16(a1, b1)) - } #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] fn split_i16x32(self, a: i16x32) -> (i16x16, i16x16); #[doc = "Negate each element of the vector, wrapping on overflow."] @@ -5771,6 +5771,20 @@ pub trait Simd: let (b0, b1) = self.split_u16x32(b); self.combine_u16x16(self.shrv_u16x16(a0, b0), self.shrv_u16x16(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_u16x32(self, a: u16x32, b: u16x32) -> u16x32 { + let (a0, a1) = self.split_u16x32(a); + let (b0, b1) = self.split_u16x32(b); + self.combine_u16x16(self.max_u16x16(a0, b0), self.max_u16x16(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_u16x32(self, a: u16x32, b: u16x32) -> u16x32 { + let (a0, a1) = self.split_u16x32(a); + let (b0, b1) = self.split_u16x32(b); + self.combine_u16x16(self.min_u16x16(a0, b0), self.min_u16x16(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_u16x32(self, a: u16x32, b: u16x32) -> mask16x32 { @@ -5872,20 +5886,6 @@ pub trait Simd: self.select_u16x16(a1, b1, c1), ) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_u16x32(self, a: u16x32, b: u16x32) -> u16x32 { - let (a0, a1) = self.split_u16x32(a); - let (b0, b1) = self.split_u16x32(b); - self.combine_u16x16(self.min_u16x16(a0, b0), self.min_u16x16(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_u16x32(self, a: u16x32, b: u16x32) -> u16x32 { - let (a0, a1) = self.split_u16x32(a); - let (b0, b1) = self.split_u16x32(b); - self.combine_u16x16(self.max_u16x16(a0, b0), self.max_u16x16(a1, b1)) - } #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] fn split_u16x32(self, a: u16x32) -> (u16x16, u16x16); #[doc = "Widen every lane into two same-width vectors.\n\nThe first result contains the widened lower lanes and the second contains the widened upper lanes."] @@ -6140,6 +6140,20 @@ pub trait Simd: let (b0, b1) = self.split_i32x16(b); self.combine_i32x8(self.shrv_i32x8(a0, b0), self.shrv_i32x8(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_i32x16(self, a: i32x16, b: i32x16) -> i32x16 { + let (a0, a1) = self.split_i32x16(a); + let (b0, b1) = self.split_i32x16(b); + self.combine_i32x8(self.max_i32x8(a0, b0), self.max_i32x8(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_i32x16(self, a: i32x16, b: i32x16) -> i32x16 { + let (a0, a1) = self.split_i32x16(a); + let (b0, b1) = self.split_i32x16(b); + self.combine_i32x8(self.min_i32x8(a0, b0), self.min_i32x8(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_i32x16(self, a: i32x16, b: i32x16) -> mask32x16 { @@ -6235,20 +6249,6 @@ pub trait Simd: let (c0, c1) = self.split_i32x16(c); self.combine_i32x8(self.select_i32x8(a0, b0, c0), self.select_i32x8(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_i32x16(self, a: i32x16, b: i32x16) -> i32x16 { - let (a0, a1) = self.split_i32x16(a); - let (b0, b1) = self.split_i32x16(b); - self.combine_i32x8(self.min_i32x8(a0, b0), self.min_i32x8(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_i32x16(self, a: i32x16, b: i32x16) -> i32x16 { - let (a0, a1) = self.split_i32x16(a); - let (b0, b1) = self.split_i32x16(b); - self.combine_i32x8(self.max_i32x8(a0, b0), self.max_i32x8(a1, b1)) - } #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] fn split_i32x16(self, a: i32x16) -> (i32x8, i32x8); #[doc = "Negate each element of the vector, wrapping on overflow."] @@ -6413,6 +6413,20 @@ pub trait Simd: let (b0, b1) = self.split_u32x16(b); self.combine_u32x8(self.shrv_u32x8(a0, b0), self.shrv_u32x8(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_u32x16(self, a: u32x16, b: u32x16) -> u32x16 { + let (a0, a1) = self.split_u32x16(a); + let (b0, b1) = self.split_u32x16(b); + self.combine_u32x8(self.max_u32x8(a0, b0), self.max_u32x8(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_u32x16(self, a: u32x16, b: u32x16) -> u32x16 { + let (a0, a1) = self.split_u32x16(a); + let (b0, b1) = self.split_u32x16(b); + self.combine_u32x8(self.min_u32x8(a0, b0), self.min_u32x8(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_u32x16(self, a: u32x16, b: u32x16) -> mask32x16 { @@ -6508,20 +6522,6 @@ pub trait Simd: let (c0, c1) = self.split_u32x16(c); self.combine_u32x8(self.select_u32x8(a0, b0, c0), self.select_u32x8(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_u32x16(self, a: u32x16, b: u32x16) -> u32x16 { - let (a0, a1) = self.split_u32x16(a); - let (b0, b1) = self.split_u32x16(b); - self.combine_u32x8(self.min_u32x8(a0, b0), self.min_u32x8(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_u32x16(self, a: u32x16, b: u32x16) -> u32x16 { - let (a0, a1) = self.split_u32x16(a); - let (b0, b1) = self.split_u32x16(b); - self.combine_u32x8(self.max_u32x8(a0, b0), self.max_u32x8(a1, b1)) - } #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] fn split_u32x16(self, a: u32x16) -> (u32x8, u32x8); #[doc = "Widen every lane into two same-width vectors.\n\nThe first result contains the widened lower lanes and the second contains the widened upper lanes."] @@ -6763,6 +6763,40 @@ pub trait Simd: let (b0, b1) = self.split_f64x8(b); self.combine_f64x4(self.copysign_f64x4(a0, b0), self.copysign_f64x4(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { + let (a0, a1) = self.split_f64x8(a); + let (b0, b1) = self.split_f64x8(b); + self.combine_f64x4(self.max_f64x4(a0, b0), self.max_f64x4(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { + let (a0, a1) = self.split_f64x8(a); + let (b0, b1) = self.split_f64x8(b); + self.combine_f64x4(self.min_f64x4(a0, b0), self.min_f64x4(a1, b1)) + } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor integer vectors, this operation is the same as `max`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + #[inline(always)] + fn max_precise_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { + let (a0, a1) = self.split_f64x8(a); + let (b0, b1) = self.split_f64x8(b); + self.combine_f64x4( + self.max_precise_f64x4(a0, b0), + self.max_precise_f64x4(a1, b1), + ) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor integer vectors, this operation is the same as `min`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + #[inline(always)] + fn min_precise_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { + let (a0, a1) = self.split_f64x8(a); + let (b0, b1) = self.split_f64x8(b); + self.combine_f64x4( + self.min_precise_f64x4(a0, b0), + self.min_precise_f64x4(a1, b1), + ) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_f64x8(self, a: f64x8, b: f64x8) -> mask64x8 { @@ -6850,40 +6884,6 @@ pub trait Simd: self.combine_f64x4(lo_odd, hi_odd), ) } - #[doc = "Return the element-wise maximum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - #[inline(always)] - fn max_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { - let (a0, a1) = self.split_f64x8(a); - let (b0, b1) = self.split_f64x8(b); - self.combine_f64x4(self.max_f64x4(a0, b0), self.max_f64x4(a1, b1)) - } - #[doc = "Return the element-wise minimum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - #[inline(always)] - fn min_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { - let (a0, a1) = self.split_f64x8(a); - let (b0, b1) = self.split_f64x8(b); - self.combine_f64x4(self.min_f64x4(a0, b0), self.min_f64x4(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - #[inline(always)] - fn max_precise_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { - let (a0, a1) = self.split_f64x8(a); - let (b0, b1) = self.split_f64x8(b); - self.combine_f64x4( - self.max_precise_f64x4(a0, b0), - self.max_precise_f64x4(a1, b1), - ) - } - #[doc = "Return the element-wise minimum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - #[inline(always)] - fn min_precise_f64x8(self, a: f64x8, b: f64x8) -> f64x8 { - let (a0, a1) = self.split_f64x8(a); - let (b0, b1) = self.split_f64x8(b); - self.combine_f64x4( - self.min_precise_f64x4(a0, b0), - self.min_precise_f64x4(a1, b1), - ) - } #[doc = "Compute `(a * b) + c` (fused multiply-add) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by an add, which will result in two rounding errors."] #[inline(always)] fn mul_add_f64x8(self, a: f64x8, b: f64x8, c: f64x8) -> f64x8 { @@ -7087,6 +7087,20 @@ pub trait Simd: let (b0, b1) = self.split_i64x8(b); self.combine_i64x4(self.shrv_i64x4(a0, b0), self.shrv_i64x4(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_i64x8(self, a: i64x8, b: i64x8) -> i64x8 { + let (a0, a1) = self.split_i64x8(a); + let (b0, b1) = self.split_i64x8(b); + self.combine_i64x4(self.max_i64x4(a0, b0), self.max_i64x4(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_i64x8(self, a: i64x8, b: i64x8) -> i64x8 { + let (a0, a1) = self.split_i64x8(a); + let (b0, b1) = self.split_i64x8(b); + self.combine_i64x4(self.min_i64x4(a0, b0), self.min_i64x4(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_i64x8(self, a: i64x8, b: i64x8) -> mask64x8 { @@ -7182,20 +7196,6 @@ pub trait Simd: let (c0, c1) = self.split_i64x8(c); self.combine_i64x4(self.select_i64x4(a0, b0, c0), self.select_i64x4(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_i64x8(self, a: i64x8, b: i64x8) -> i64x8 { - let (a0, a1) = self.split_i64x8(a); - let (b0, b1) = self.split_i64x8(b); - self.combine_i64x4(self.min_i64x4(a0, b0), self.min_i64x4(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_i64x8(self, a: i64x8, b: i64x8) -> i64x8 { - let (a0, a1) = self.split_i64x8(a); - let (b0, b1) = self.split_i64x8(b); - self.combine_i64x4(self.max_i64x4(a0, b0), self.max_i64x4(a1, b1)) - } #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] fn split_i64x8(self, a: i64x8) -> (i64x4, i64x4); #[doc = "Negate each element of the vector, wrapping on overflow."] @@ -7342,6 +7342,20 @@ pub trait Simd: let (b0, b1) = self.split_u64x8(b); self.combine_u64x4(self.shrv_u64x4(a0, b0), self.shrv_u64x4(a1, b1)) } + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn max_u64x8(self, a: u64x8, b: u64x8) -> u64x8 { + let (a0, a1) = self.split_u64x8(a); + let (b0, b1) = self.split_u64x8(b); + self.combine_u64x4(self.max_u64x4(a0, b0), self.max_u64x4(a1, b1)) + } + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + #[inline(always)] + fn min_u64x8(self, a: u64x8, b: u64x8) -> u64x8 { + let (a0, a1) = self.split_u64x8(a); + let (b0, b1) = self.split_u64x8(b); + self.combine_u64x4(self.min_u64x4(a0, b0), self.min_u64x4(a1, b1)) + } #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] #[inline(always)] fn simd_eq_u64x8(self, a: u64x8, b: u64x8) -> mask64x8 { @@ -7437,20 +7451,6 @@ pub trait Simd: let (c0, c1) = self.split_u64x8(c); self.combine_u64x4(self.select_u64x4(a0, b0, c0), self.select_u64x4(a1, b1, c1)) } - #[doc = "Return the element-wise minimum of two vectors."] - #[inline(always)] - fn min_u64x8(self, a: u64x8, b: u64x8) -> u64x8 { - let (a0, a1) = self.split_u64x8(a); - let (b0, b1) = self.split_u64x8(b); - self.combine_u64x4(self.min_u64x4(a0, b0), self.min_u64x4(a1, b1)) - } - #[doc = "Return the element-wise maximum of two vectors."] - #[inline(always)] - fn max_u64x8(self, a: u64x8, b: u64x8) -> u64x8 { - let (a0, a1) = self.split_u64x8(a); - let (b0, b1) = self.split_u64x8(b); - self.combine_u64x4(self.max_u64x4(a0, b0), self.max_u64x4(a1, b1)) - } #[doc = "Split a vector into two vectors of half the width.\n\nReturns a tuple of (lower half, upper half)."] fn split_u64x8(self, a: u64x8) -> (u64x4, u64x4); #[doc = "Truncate the lanes of two vectors and concatenate them into one same-width vector.\n\nEach lane retains its low destination-width bits. `a` provides the lower result lanes and `b` provides the upper result lanes."] @@ -8143,6 +8143,14 @@ pub trait SimdBase: fn swizzle_dyn(self, indices: impl SimdInto) -> Self; #[doc = "Dynamically swizzle this vector's bytes across the whole vector.\n\nThe `indices` operand is a same-width byte vector. For each output byte, index values within the vector's byte length select the corresponding byte from the input vector. Out-of-range indices produce zero."] fn swizzle_dyn_precise(self, indices: impl SimdInto) -> Self; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn max(self, rhs: impl SimdInto) -> Self; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] + fn min(self, rhs: impl SimdInto) -> Self; + #[doc = "Return the element-wise maximum of two vectors.\n\nFor integer vectors, this operation is the same as `max`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + fn max_precise(self, rhs: impl SimdInto) -> Self; + #[doc = "Return the element-wise minimum of two vectors.\n\nFor integer vectors, this operation is the same as `min`.\n\nFor floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] + fn min_precise(self, rhs: impl SimdInto) -> Self; #[doc = "Compare two vectors element-wise for equality.\n\nReturns a mask where each logical lane is true if the corresponding elements are equal, and false if not."] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask; #[doc = "Compare two vectors element-wise for less than.\n\nReturns a mask where each logical lane is true if `self` is less than `rhs`, and false if not."] @@ -8206,14 +8214,6 @@ pub trait SimdFloat: fn approximate_recip(self) -> Self; #[doc = "Return a vector with the magnitude of `self` and the sign of `rhs` for each element.\n\nThis operation copies the sign bit, so if an input element is NaN, the output element will be a NaN with the same payload and a copied sign bit."] fn copysign(self, rhs: impl SimdInto) -> Self; - #[doc = "Return the element-wise maximum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - fn max(self, rhs: impl SimdInto) -> Self; - #[doc = "Return the element-wise minimum of two vectors.\n\nIf either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\nIf one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one."] - fn min(self, rhs: impl SimdInto) -> Self; - #[doc = "Return the element-wise maximum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - fn max_precise(self, rhs: impl SimdInto) -> Self; - #[doc = "Return the element-wise minimum of two vectors.\n\nIf one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\nIf one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\nIf an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\nSignaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them."] - fn min_precise(self, rhs: impl SimdInto) -> Self; #[doc = "Compute `(self * op1) + op2` (fused multiply-add) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by an add, which will result in two rounding errors."] fn mul_add(self, op1: impl SimdInto, op2: impl SimdInto) -> Self; #[doc = "Compute `(self * op1) - op2` (fused multiply-subtract) for each element.\n\nDepending on hardware support, the result may be computed with only one rounding error, or may be implemented as a regular multiply followed by a subtract, which will result in two rounding errors."] @@ -8262,10 +8262,6 @@ pub trait SimdInt: fn to_float>(self) -> T { T::float_from(self) } - #[doc = "Return the element-wise minimum of two vectors."] - fn min(self, rhs: impl SimdInto) -> Self; - #[doc = "Return the element-wise maximum of two vectors."] - fn max(self, rhs: impl SimdInto) -> Self; } #[doc = r" Functionality implemented by SIMD masks."] #[doc = r""] diff --git a/fearless_simd/src/generated/simd_types.rs b/fearless_simd/src/generated/simd_types.rs index 87ffa365..436ca1d9 100644 --- a/fearless_simd/src/generated/simd_types.rs +++ b/fearless_simd/src/generated/simd_types.rs @@ -155,6 +155,22 @@ impl SimdBase for f32x4 { .swizzle_dyn_precise_f32x4(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_f32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_f32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_precise_f32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_precise_f32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_f32x4(self, rhs.simd_into(self.simd)) } @@ -217,22 +233,6 @@ impl crate::SimdFloat for f32x4 { self.simd.copysign_f32x4(self, rhs.simd_into(self.simd)) } #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_f32x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_f32x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max_precise(self, rhs: impl SimdInto) -> Self { - self.simd.max_precise_f32x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min_precise(self, rhs: impl SimdInto) -> Self { - self.simd.min_precise_f32x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] fn mul_add(self, op1: impl SimdInto, op2: impl SimdInto) -> Self { self.simd .mul_add_f32x4(self, op1.simd_into(self.simd), op2.simd_into(self.simd)) @@ -459,6 +459,22 @@ impl SimdBase for i8x16 { .swizzle_dyn_precise_i8x16(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i8x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i8x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i8x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i8x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i8x16(self, rhs.simd_into(self.simd)) } @@ -503,16 +519,7 @@ impl SimdBase for i8x16 { self.simd.deinterleave_i8x16(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i8x16 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i8x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i8x16(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i8x16 {} impl SimdWiden for i8x16 { type Widened = i16x8; #[inline(always)] @@ -695,6 +702,22 @@ impl SimdBase for u8x16 { .swizzle_dyn_precise_u8x16(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u8x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u8x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u8x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u8x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u8x16(self, rhs.simd_into(self.simd)) } @@ -739,16 +762,7 @@ impl SimdBase for u8x16 { self.simd.deinterleave_u8x16(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u8x16 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u8x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u8x16(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u8x16 {} impl SimdWiden for u8x16 { type Widened = u16x8; #[inline(always)] @@ -1017,6 +1031,22 @@ impl SimdBase for i16x8 { .swizzle_dyn_precise_i16x8(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i16x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i16x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i16x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i16x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i16x8(self, rhs.simd_into(self.simd)) } @@ -1061,16 +1091,7 @@ impl SimdBase for i16x8 { self.simd.deinterleave_i16x8(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i16x8 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i16x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i16x8(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i16x8 {} impl SimdWiden for i16x8 { type Widened = i32x4; #[inline(always)] @@ -1260,6 +1281,22 @@ impl SimdBase for u16x8 { .swizzle_dyn_precise_u16x8(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u16x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u16x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u16x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u16x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u16x8(self, rhs.simd_into(self.simd)) } @@ -1304,16 +1341,7 @@ impl SimdBase for u16x8 { self.simd.deinterleave_u16x8(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u16x8 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u16x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u16x8(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u16x8 {} impl SimdWiden for u16x8 { type Widened = u32x4; #[inline(always)] @@ -1585,6 +1613,22 @@ impl SimdBase for i32x4 { .swizzle_dyn_precise_i32x4(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i32x4(self, rhs.simd_into(self.simd)) } @@ -1629,16 +1673,7 @@ impl SimdBase for i32x4 { self.simd.deinterleave_i32x4(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i32x4 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i32x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i32x4(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i32x4 {} impl SimdCvtTruncate> for i32x4 { #[doc = "Convert each floating-point element to a signed 32-bit integer, truncating towards zero.\n\nOut-of-range values or NaN will produce implementation-defined results."] #[inline(always)] @@ -1828,6 +1863,22 @@ impl SimdBase for u32x4 { .swizzle_dyn_precise_u32x4(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u32x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u32x4(self, rhs.simd_into(self.simd)) } @@ -1872,16 +1923,7 @@ impl SimdBase for u32x4 { self.simd.deinterleave_u32x4(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u32x4 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u32x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u32x4(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u32x4 {} impl SimdCvtTruncate> for u32x4 { #[doc = "Convert each floating-point element to an unsigned 32-bit integer, truncating towards zero.\n\nOut-of-range values or NaN will produce implementation-defined results.\n\nOn x86 platforms below AVX-512, this operation will still be slower than converting to `i32`, because there is no native instruction for converting to `u32`.\nIf you know your values fit within range of an `i32`, you should convert to an `i32` and cast to your desired datatype afterwards."] #[inline(always)] @@ -2165,6 +2207,22 @@ impl SimdBase for f64x2 { .swizzle_dyn_precise_f64x2(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_f64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_f64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_precise_f64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_precise_f64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_f64x2(self, rhs.simd_into(self.simd)) } @@ -2227,22 +2285,6 @@ impl crate::SimdFloat for f64x2 { self.simd.copysign_f64x2(self, rhs.simd_into(self.simd)) } #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_f64x2(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_f64x2(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max_precise(self, rhs: impl SimdInto) -> Self { - self.simd.max_precise_f64x2(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min_precise(self, rhs: impl SimdInto) -> Self { - self.simd.min_precise_f64x2(self, rhs.simd_into(self.simd)) - } - #[inline(always)] fn mul_add(self, op1: impl SimdInto, op2: impl SimdInto) -> Self { self.simd .mul_add_f64x2(self, op1.simd_into(self.simd), op2.simd_into(self.simd)) @@ -2443,6 +2485,22 @@ impl SimdBase for i64x2 { .swizzle_dyn_precise_i64x2(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i64x2(self, rhs.simd_into(self.simd)) } @@ -2487,16 +2545,7 @@ impl SimdBase for i64x2 { self.simd.deinterleave_i64x2(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i64x2 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i64x2(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i64x2(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i64x2 {} impl SimdNarrow for i64x2 { type Narrowed = i32x4; #[inline(always)] @@ -2667,6 +2716,22 @@ impl SimdBase for u64x2 { .swizzle_dyn_precise_u64x2(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u64x2(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u64x2(self, rhs.simd_into(self.simd)) } @@ -2711,16 +2776,7 @@ impl SimdBase for u64x2 { self.simd.deinterleave_u64x2(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u64x2 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u64x2(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u64x2(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u64x2 {} impl SimdNarrow for u64x2 { type Narrowed = u32x4; #[inline(always)] @@ -2997,6 +3053,22 @@ impl SimdBase for f32x8 { .swizzle_dyn_precise_f32x8(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_f32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_f32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_precise_f32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_precise_f32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_f32x8(self, rhs.simd_into(self.simd)) } @@ -3047,32 +3119,16 @@ impl crate::SimdFloat for f32x8 { self.simd.abs_f32x8(self) } #[inline(always)] - fn sqrt(self) -> Self { - self.simd.sqrt_f32x8(self) - } - #[inline(always)] - fn approximate_recip(self) -> Self { - self.simd.approximate_recip_f32x8(self) - } - #[inline(always)] - fn copysign(self, rhs: impl SimdInto) -> Self { - self.simd.copysign_f32x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_f32x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_f32x8(self, rhs.simd_into(self.simd)) + fn sqrt(self) -> Self { + self.simd.sqrt_f32x8(self) } #[inline(always)] - fn max_precise(self, rhs: impl SimdInto) -> Self { - self.simd.max_precise_f32x8(self, rhs.simd_into(self.simd)) + fn approximate_recip(self) -> Self { + self.simd.approximate_recip_f32x8(self) } #[inline(always)] - fn min_precise(self, rhs: impl SimdInto) -> Self { - self.simd.min_precise_f32x8(self, rhs.simd_into(self.simd)) + fn copysign(self, rhs: impl SimdInto) -> Self { + self.simd.copysign_f32x8(self, rhs.simd_into(self.simd)) } #[inline(always)] fn mul_add(self, op1: impl SimdInto, op2: impl SimdInto) -> Self { @@ -3324,6 +3380,22 @@ impl SimdBase for i8x32 { .swizzle_dyn_precise_i8x32(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i8x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i8x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i8x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i8x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i8x32(self, rhs.simd_into(self.simd)) } @@ -3368,16 +3440,7 @@ impl SimdBase for i8x32 { self.simd.deinterleave_i8x32(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i8x32 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i8x32(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i8x32(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i8x32 {} impl SimdWiden for i8x32 { type Widened = i16x16; #[inline(always)] @@ -3583,6 +3646,22 @@ impl SimdBase for u8x32 { .swizzle_dyn_precise_u8x32(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u8x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u8x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u8x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u8x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u8x32(self, rhs.simd_into(self.simd)) } @@ -3627,16 +3706,7 @@ impl SimdBase for u8x32 { self.simd.deinterleave_u8x32(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u8x32 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u8x32(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u8x32(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u8x32 {} impl SimdWiden for u8x32 { type Widened = u16x16; #[inline(always)] @@ -3920,6 +3990,22 @@ impl SimdBase for i16x16 { .swizzle_dyn_precise_i16x16(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i16x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i16x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i16x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i16x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i16x16(self, rhs.simd_into(self.simd)) } @@ -3965,16 +4051,7 @@ impl SimdBase for i16x16 { .deinterleave_i16x16(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i16x16 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i16x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i16x16(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i16x16 {} impl SimdWiden for i16x16 { type Widened = i32x8; #[inline(always)] @@ -4179,6 +4256,22 @@ impl SimdBase for u16x16 { .swizzle_dyn_precise_u16x16(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u16x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u16x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u16x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u16x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u16x16(self, rhs.simd_into(self.simd)) } @@ -4224,16 +4317,7 @@ impl SimdBase for u16x16 { .deinterleave_u16x16(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u16x16 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u16x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u16x16(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u16x16 {} impl SimdWiden for u16x16 { type Widened = u32x8; #[inline(always)] @@ -4524,6 +4608,22 @@ impl SimdBase for i32x8 { .swizzle_dyn_precise_i32x8(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i32x8(self, rhs.simd_into(self.simd)) } @@ -4568,16 +4668,7 @@ impl SimdBase for i32x8 { self.simd.deinterleave_i32x8(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i32x8 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i32x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i32x8(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i32x8 {} impl SimdCvtTruncate> for i32x8 { #[doc = "Convert each floating-point element to a signed 32-bit integer, truncating towards zero.\n\nOut-of-range values or NaN will produce implementation-defined results."] #[inline(always)] @@ -4786,6 +4877,22 @@ impl SimdBase for u32x8 { .swizzle_dyn_precise_u32x8(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u32x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u32x8(self, rhs.simd_into(self.simd)) } @@ -4830,16 +4937,7 @@ impl SimdBase for u32x8 { self.simd.deinterleave_u32x8(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u32x8 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u32x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u32x8(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u32x8 {} impl SimdCvtTruncate> for u32x8 { #[doc = "Convert each floating-point element to an unsigned 32-bit integer, truncating towards zero.\n\nOut-of-range values or NaN will produce implementation-defined results.\n\nOn x86 platforms below AVX-512, this operation will still be slower than converting to `i32`, because there is no native instruction for converting to `u32`.\nIf you know your values fit within range of an `i32`, you should convert to an `i32` and cast to your desired datatype afterwards."] #[inline(always)] @@ -5130,6 +5228,22 @@ impl SimdBase for f64x4 { .swizzle_dyn_precise_f64x4(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_f64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_f64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_precise_f64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_precise_f64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_f64x4(self, rhs.simd_into(self.simd)) } @@ -5192,22 +5306,6 @@ impl crate::SimdFloat for f64x4 { self.simd.copysign_f64x4(self, rhs.simd_into(self.simd)) } #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_f64x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_f64x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max_precise(self, rhs: impl SimdInto) -> Self { - self.simd.max_precise_f64x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min_precise(self, rhs: impl SimdInto) -> Self { - self.simd.min_precise_f64x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] fn mul_add(self, op1: impl SimdInto, op2: impl SimdInto) -> Self { self.simd .mul_add_f64x4(self, op1.simd_into(self.simd), op2.simd_into(self.simd)) @@ -5415,6 +5513,22 @@ impl SimdBase for i64x4 { .swizzle_dyn_precise_i64x4(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i64x4(self, rhs.simd_into(self.simd)) } @@ -5459,16 +5573,7 @@ impl SimdBase for i64x4 { self.simd.deinterleave_i64x4(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i64x4 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i64x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i64x4(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i64x4 {} impl SimdNarrow for i64x4 { type Narrowed = i32x8; #[inline(always)] @@ -5646,6 +5751,22 @@ impl SimdBase for u64x4 { .swizzle_dyn_precise_u64x4(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u64x4(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u64x4(self, rhs.simd_into(self.simd)) } @@ -5690,16 +5811,7 @@ impl SimdBase for u64x4 { self.simd.deinterleave_u64x4(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u64x4 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u64x4(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u64x4(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u64x4 {} impl SimdNarrow for u64x4 { type Narrowed = u32x8; #[inline(always)] @@ -5987,9 +6099,25 @@ impl SimdBase for f32x16 { .swizzle_dyn_f32x16(self, indices.simd_into(self.simd)) } #[inline(always)] - fn swizzle_dyn_precise(self, indices: impl SimdInto) -> Self { - self.simd - .swizzle_dyn_precise_f32x16(self, indices.simd_into(self.simd)) + fn swizzle_dyn_precise(self, indices: impl SimdInto) -> Self { + self.simd + .swizzle_dyn_precise_f32x16(self, indices.simd_into(self.simd)) + } + #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_f32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_f32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_precise_f32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_precise_f32x16(self, rhs.simd_into(self.simd)) } #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { @@ -6055,22 +6183,6 @@ impl crate::SimdFloat for f32x16 { self.simd.copysign_f32x16(self, rhs.simd_into(self.simd)) } #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_f32x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_f32x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max_precise(self, rhs: impl SimdInto) -> Self { - self.simd.max_precise_f32x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min_precise(self, rhs: impl SimdInto) -> Self { - self.simd.min_precise_f32x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] fn mul_add(self, op1: impl SimdInto, op2: impl SimdInto) -> Self { self.simd .mul_add_f32x16(self, op1.simd_into(self.simd), op2.simd_into(self.simd)) @@ -6346,6 +6458,22 @@ impl SimdBase for i8x64 { .swizzle_dyn_precise_i8x64(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i8x64(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i8x64(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i8x64(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i8x64(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i8x64(self, rhs.simd_into(self.simd)) } @@ -6390,16 +6518,7 @@ impl SimdBase for i8x64 { self.simd.deinterleave_i8x64(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i8x64 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i8x64(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i8x64(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i8x64 {} impl SimdWiden for i8x64 { type Widened = i16x32; #[inline(always)] @@ -6631,6 +6750,22 @@ impl SimdBase for u8x64 { .swizzle_dyn_precise_u8x64(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u8x64(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u8x64(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u8x64(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u8x64(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u8x64(self, rhs.simd_into(self.simd)) } @@ -6675,16 +6810,7 @@ impl SimdBase for u8x64 { self.simd.deinterleave_u8x64(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u8x64 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u8x64(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u8x64(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u8x64 {} impl SimdWiden for u8x64 { type Widened = u16x32; #[inline(always)] @@ -6978,6 +7104,22 @@ impl SimdBase for i16x32 { .swizzle_dyn_precise_i16x32(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i16x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i16x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i16x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i16x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i16x32(self, rhs.simd_into(self.simd)) } @@ -7023,16 +7165,7 @@ impl SimdBase for i16x32 { .deinterleave_i16x32(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i16x32 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i16x32(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i16x32(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i16x32 {} impl SimdWiden for i16x32 { type Widened = i32x16; #[inline(always)] @@ -7247,6 +7380,22 @@ impl SimdBase for u16x32 { .swizzle_dyn_precise_u16x32(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u16x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u16x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u16x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u16x32(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u16x32(self, rhs.simd_into(self.simd)) } @@ -7292,16 +7441,7 @@ impl SimdBase for u16x32 { .deinterleave_u16x32(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u16x32 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u16x32(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u16x32(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u16x32 {} impl SimdWiden for u16x32 { type Widened = u32x16; #[inline(always)] @@ -7594,6 +7734,22 @@ impl SimdBase for i32x16 { .swizzle_dyn_precise_i32x16(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i32x16(self, rhs.simd_into(self.simd)) } @@ -7639,16 +7795,7 @@ impl SimdBase for i32x16 { .deinterleave_i32x16(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i32x16 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i32x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i32x16(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i32x16 {} impl SimdCvtTruncate> for i32x16 { #[doc = "Convert each floating-point element to a signed 32-bit integer, truncating towards zero.\n\nOut-of-range values or NaN will produce implementation-defined results."] #[inline(always)] @@ -7859,6 +8006,22 @@ impl SimdBase for u32x16 { .swizzle_dyn_precise_u32x16(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u32x16(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u32x16(self, rhs.simd_into(self.simd)) } @@ -7904,16 +8067,7 @@ impl SimdBase for u32x16 { .deinterleave_u32x16(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u32x16 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u32x16(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u32x16(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u32x16 {} impl SimdCvtTruncate> for u32x16 { #[doc = "Convert each floating-point element to an unsigned 32-bit integer, truncating towards zero.\n\nOut-of-range values or NaN will produce implementation-defined results.\n\nOn x86 platforms below AVX-512, this operation will still be slower than converting to `i32`, because there is no native instruction for converting to `u32`.\nIf you know your values fit within range of an `i32`, you should convert to an `i32` and cast to your desired datatype afterwards."] #[inline(always)] @@ -8210,6 +8364,22 @@ impl SimdBase for f64x8 { .swizzle_dyn_precise_f64x8(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_f64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_f64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_precise_f64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_precise_f64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_f64x8(self, rhs.simd_into(self.simd)) } @@ -8272,22 +8442,6 @@ impl crate::SimdFloat for f64x8 { self.simd.copysign_f64x8(self, rhs.simd_into(self.simd)) } #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_f64x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_f64x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max_precise(self, rhs: impl SimdInto) -> Self { - self.simd.max_precise_f64x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn min_precise(self, rhs: impl SimdInto) -> Self { - self.simd.min_precise_f64x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] fn mul_add(self, op1: impl SimdInto, op2: impl SimdInto) -> Self { self.simd .mul_add_f64x8(self, op1.simd_into(self.simd), op2.simd_into(self.simd)) @@ -8501,6 +8655,22 @@ impl SimdBase for i64x8 { .swizzle_dyn_precise_i64x8(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_i64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_i64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_i64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_i64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_i64x8(self, rhs.simd_into(self.simd)) } @@ -8545,16 +8715,7 @@ impl SimdBase for i64x8 { self.simd.deinterleave_i64x8(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for i64x8 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_i64x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_i64x8(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for i64x8 {} impl SimdNarrow for i64x8 { type Narrowed = i32x16; #[inline(always)] @@ -8738,6 +8899,22 @@ impl SimdBase for u64x8 { .swizzle_dyn_precise_u64x8(self, indices.simd_into(self.simd)) } #[inline(always)] + fn max(self, rhs: impl SimdInto) -> Self { + self.simd.max_u64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min(self, rhs: impl SimdInto) -> Self { + self.simd.min_u64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn max_precise(self, rhs: impl SimdInto) -> Self { + self.simd.max_u64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] + fn min_precise(self, rhs: impl SimdInto) -> Self { + self.simd.min_u64x8(self, rhs.simd_into(self.simd)) + } + #[inline(always)] fn simd_eq(self, rhs: impl SimdInto) -> Self::Mask { self.simd.simd_eq_u64x8(self, rhs.simd_into(self.simd)) } @@ -8782,16 +8959,7 @@ impl SimdBase for u64x8 { self.simd.deinterleave_u64x8(self, rhs.simd_into(self.simd)) } } -impl crate::SimdInt for u64x8 { - #[inline(always)] - fn min(self, rhs: impl SimdInto) -> Self { - self.simd.min_u64x8(self, rhs.simd_into(self.simd)) - } - #[inline(always)] - fn max(self, rhs: impl SimdInto) -> Self { - self.simd.max_u64x8(self, rhs.simd_into(self.simd)) - } -} +impl crate::SimdInt for u64x8 {} impl SimdNarrow for u64x8 { type Narrowed = u32x16; #[inline(always)] diff --git a/fearless_simd/src/generated/sse2.rs b/fearless_simd/src/generated/sse2.rs index 061b8b1d..49e3a923 100644 --- a/fearless_simd/src/generated/sse2.rs +++ b/fearless_simd/src/generated/sse2.rs @@ -295,140 +295,140 @@ impl Simd for Sse2 { kernel(self, a, b) } #[inline(always)] - fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmpeq_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { + _mm_max_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_lt_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmplt_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { + _mm_min_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_le_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmple_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { + let a = a.into(); + let b = b.into(); + let intermediate = _mm_max_ps(a, b); + let b_is_nan = _mm_cmpunord_ps(b, b); + _mm_or_ps( + _mm_and_ps(b_is_nan, a), + _mm_andnot_ps(b_is_nan, intermediate), + ) + .simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { - _mm_unpacklo_ps(a.into(), b.into()).simd_into(token) + let a = a.into(); + let b = b.into(); + let intermediate = _mm_min_ps(a, b); + let b_is_nan = _mm_cmpunord_ps(b, b); + _mm_or_ps( + _mm_and_ps(b_is_nan, a), + _mm_andnot_ps(b_is_nan, intermediate), + ) + .simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { - _mm_unpackhi_ps(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmpeq_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_lt_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { - _mm_shuffle_ps::<0b10_00_10_00>(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmplt_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_le_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { - _mm_shuffle_ps::<0b11_01_11_01>(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmple_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn interleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { - (self.zip_low_f32x4(a, b), self.zip_high_f32x4(a, b)) - } - #[inline(always)] - fn deinterleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { - (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) - } - #[inline(always)] - fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn zip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { - _mm_max_ps(a.into(), b.into()).simd_into(token) + _mm_unpacklo_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn zip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { - _mm_min_ps(a.into(), b.into()).simd_into(token) + _mm_unpackhi_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn unzip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { - let a = a.into(); - let b = b.into(); - let intermediate = _mm_max_ps(a, b); - let b_is_nan = _mm_cmpunord_ps(b, b); - _mm_or_ps( - _mm_and_ps(b_is_nan, a), - _mm_andnot_ps(b_is_nan, intermediate), - ) - .simd_into(token) + _mm_shuffle_ps::<0b10_00_10_00>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn unzip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f32x4, b: f32x4) -> f32x4 { - let a = a.into(); - let b = b.into(); - let intermediate = _mm_min_ps(a, b); - let b_is_nan = _mm_cmpunord_ps(b, b); - _mm_or_ps( - _mm_and_ps(b_is_nan, a), - _mm_andnot_ps(b_is_nan, intermediate), - ) - .simd_into(token) + _mm_shuffle_ps::<0b11_01_11_01>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] + fn interleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { + (self.zip_low_f32x4(a, b), self.zip_high_f32x4(a, b)) + } + #[inline(always)] + fn deinterleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { + (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) + } + #[inline(always)] fn mul_add_f32x4(self, a: f32x4, b: f32x4, c: f32x4) -> f32x4 { a * b + c } @@ -821,6 +821,38 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: i8x16, b: i8x16) -> i8x16 { + { + let a = a.into(); + let b = b.into(); + let gt = _mm_cmpgt_epi8(a, b); + _mm_or_si128(_mm_and_si128(gt, a), _mm_andnot_si128(gt, b)) + } + .simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: i8x16, b: i8x16) -> i8x16 { + { + let a = a.into(); + let b = b.into(); + let gt = _mm_cmpgt_epi8(a, b); + _mm_or_si128(_mm_and_si128(gt, b), _mm_andnot_si128(gt, a)) + } + .simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i8x16(self, a: i8x16, b: i8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -920,38 +952,6 @@ impl Simd for Sse2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: i8x16, b: i8x16) -> i8x16 { - { - let a = a.into(); - let b = b.into(); - let gt = _mm_cmpgt_epi8(a, b); - _mm_or_si128(_mm_and_si128(gt, b), _mm_andnot_si128(gt, a)) - } - .simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: i8x16, b: i8x16) -> i8x16 { - { - let a = a.into(); - let b = b.into(); - let gt = _mm_cmpgt_epi8(a, b); - _mm_or_si128(_mm_and_si128(gt, a), _mm_andnot_si128(gt, b)) - } - .simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i8x16(self, a: i8x16, b: i8x16) -> i8x32 { i8x32 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -1404,6 +1404,26 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: u8x16, b: u8x16) -> u8x16 { + _mm_max_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: u8x16, b: u8x16) -> u8x16 { + _mm_min_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u8x16(self, a: u8x16, b: u8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -1517,26 +1537,6 @@ impl Simd for Sse2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: u8x16, b: u8x16) -> u8x16 { - _mm_min_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: u8x16, b: u8x16) -> u8x16 { - _mm_max_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u8x16(self, a: u8x16, b: u8x16) -> u8x32 { u8x32 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -2020,6 +2020,26 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: i16x8, b: i16x8) -> i16x8 { + _mm_max_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: i16x8, b: i16x8) -> i16x8 { + _mm_min_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i16x8(self, a: i16x8, b: i16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -2115,26 +2135,6 @@ impl Simd for Sse2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: i16x8, b: i16x8) -> i16x8 { - _mm_min_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: i16x8, b: i16x8) -> i16x8 { - _mm_max_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i16x8(self, a: i16x8, b: i16x8) -> i16x16 { i16x16 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -2428,15 +2428,57 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] - fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { + fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: u16x8, b: u16x8) -> mask16x8 { - _mm_cmpeq_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } + fn kernel(token: Sse2, a: u16x8, b: u16x8) -> u16x8 { + { + let a = a.into(); + let b = b.into(); + let gt = { + let sign_bit = _mm_set1_epi16(0x8000u16.cast_signed()); + let lhs_signed = _mm_xor_si128(a, sign_bit); + let rhs_signed = _mm_xor_si128(b, sign_bit); + _mm_cmpgt_epi16(lhs_signed, rhs_signed) + }; + _mm_or_si128(_mm_and_si128(gt, a), _mm_andnot_si128(gt, b)) + } + .simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: u16x8, b: u16x8) -> u16x8 { + { + let a = a.into(); + let b = b.into(); + let gt = { + let sign_bit = _mm_set1_epi16(0x8000u16.cast_signed()); + let lhs_signed = _mm_xor_si128(a, sign_bit); + let rhs_signed = _mm_xor_si128(b, sign_bit); + _mm_cmpgt_epi16(lhs_signed, rhs_signed) + }; + _mm_or_si128(_mm_and_si128(gt, b), _mm_andnot_si128(gt, a)) + } + .simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: u16x8, b: u16x8) -> mask16x8 { + _mm_cmpeq_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } #[inline(always)] fn simd_lt_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { crate::kernel!( @@ -2537,48 +2579,6 @@ impl Simd for Sse2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: u16x8, b: u16x8) -> u16x8 { - { - let a = a.into(); - let b = b.into(); - let gt = { - let sign_bit = _mm_set1_epi16(0x8000u16.cast_signed()); - let lhs_signed = _mm_xor_si128(a, sign_bit); - let rhs_signed = _mm_xor_si128(b, sign_bit); - _mm_cmpgt_epi16(lhs_signed, rhs_signed) - }; - _mm_or_si128(_mm_and_si128(gt, b), _mm_andnot_si128(gt, a)) - } - .simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: u16x8, b: u16x8) -> u16x8 { - { - let a = a.into(); - let b = b.into(); - let gt = { - let sign_bit = _mm_set1_epi16(0x8000u16.cast_signed()); - let lhs_signed = _mm_xor_si128(a, sign_bit); - let rhs_signed = _mm_xor_si128(b, sign_bit); - _mm_cmpgt_epi16(lhs_signed, rhs_signed) - }; - _mm_or_si128(_mm_and_si128(gt, a), _mm_andnot_si128(gt, b)) - } - .simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u16x8(self, a: u16x8, b: u16x8) -> u16x16 { u16x16 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -3024,6 +3024,38 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: i32x4, b: i32x4) -> i32x4 { + { + let a = a.into(); + let b = b.into(); + let gt = _mm_cmpgt_epi32(a, b); + _mm_or_si128(_mm_and_si128(gt, a), _mm_andnot_si128(gt, b)) + } + .simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: i32x4, b: i32x4) -> i32x4 { + { + let a = a.into(); + let b = b.into(); + let gt = _mm_cmpgt_epi32(a, b); + _mm_or_si128(_mm_and_si128(gt, b), _mm_andnot_si128(gt, a)) + } + .simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -3129,38 +3161,6 @@ impl Simd for Sse2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: i32x4, b: i32x4) -> i32x4 { - { - let a = a.into(); - let b = b.into(); - let gt = _mm_cmpgt_epi32(a, b); - _mm_or_si128(_mm_and_si128(gt, b), _mm_andnot_si128(gt, a)) - } - .simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: i32x4, b: i32x4) -> i32x4 { - { - let a = a.into(); - let b = b.into(); - let gt = _mm_cmpgt_epi32(a, b); - _mm_or_si128(_mm_and_si128(gt, a), _mm_andnot_si128(gt, b)) - } - .simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i32x4(self, a: i32x4, b: i32x4) -> i32x8 { i32x8 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -3451,6 +3451,48 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: u32x4, b: u32x4) -> u32x4 { + { + let a = a.into(); + let b = b.into(); + let gt = { + let sign_bit = _mm_set1_epi32(0x80000000u32.cast_signed()); + let lhs_signed = _mm_xor_si128(a, sign_bit); + let rhs_signed = _mm_xor_si128(b, sign_bit); + _mm_cmpgt_epi32(lhs_signed, rhs_signed) + }; + _mm_or_si128(_mm_and_si128(gt, a), _mm_andnot_si128(gt, b)) + } + .simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse2, a: u32x4, b: u32x4) -> u32x4 { + { + let a = a.into(); + let b = b.into(); + let gt = { + let sign_bit = _mm_set1_epi32(0x80000000u32.cast_signed()); + let lhs_signed = _mm_xor_si128(a, sign_bit); + let rhs_signed = _mm_xor_si128(b, sign_bit); + _mm_cmpgt_epi32(lhs_signed, rhs_signed) + }; + _mm_or_si128(_mm_and_si128(gt, b), _mm_andnot_si128(gt, a)) + } + .simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u32x4(self, a: u32x4, b: u32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -3570,48 +3612,6 @@ impl Simd for Sse2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: u32x4, b: u32x4) -> u32x4 { - { - let a = a.into(); - let b = b.into(); - let gt = { - let sign_bit = _mm_set1_epi32(0x80000000u32.cast_signed()); - let lhs_signed = _mm_xor_si128(a, sign_bit); - let rhs_signed = _mm_xor_si128(b, sign_bit); - _mm_cmpgt_epi32(lhs_signed, rhs_signed) - }; - _mm_or_si128(_mm_and_si128(gt, b), _mm_andnot_si128(gt, a)) - } - .simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse2, a: u32x4, b: u32x4) -> u32x4 { - { - let a = a.into(); - let b = b.into(); - let gt = { - let sign_bit = _mm_set1_epi32(0x80000000u32.cast_signed()); - let lhs_signed = _mm_xor_si128(a, sign_bit); - let rhs_signed = _mm_xor_si128(b, sign_bit); - _mm_cmpgt_epi32(lhs_signed, rhs_signed) - }; - _mm_or_si128(_mm_and_si128(gt, a), _mm_andnot_si128(gt, b)) - } - .simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u32x4(self, a: u32x4, b: u32x4) -> u32x8 { u32x8 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -4040,140 +4040,140 @@ impl Simd for Sse2 { kernel(self, a, b) } #[inline(always)] - fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmpeq_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { + _mm_max_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_lt_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmplt_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { + _mm_min_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_le_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmple_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { + let a = a.into(); + let b = b.into(); + let intermediate = _mm_max_pd(a, b); + let b_is_nan = _mm_cmpunord_pd(b, b); + _mm_or_pd( + _mm_and_pd(b_is_nan, a), + _mm_andnot_pd(b_is_nan, intermediate), + ) + .simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { - _mm_unpacklo_pd(a.into(), b.into()).simd_into(token) + let a = a.into(); + let b = b.into(); + let intermediate = _mm_min_pd(a, b); + let b_is_nan = _mm_cmpunord_pd(b, b); + _mm_or_pd( + _mm_and_pd(b_is_nan, a), + _mm_andnot_pd(b_is_nan, intermediate), + ) + .simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { - _mm_unpackhi_pd(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmpeq_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_lt_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { - _mm_shuffle_pd::<0b00>(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmplt_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_le_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { - _mm_shuffle_pd::<0b11>(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmple_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn interleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { - (self.zip_low_f64x2(a, b), self.zip_high_f64x2(a, b)) - } - #[inline(always)] - fn deinterleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { - (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) - } - #[inline(always)] - fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn zip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { - _mm_max_pd(a.into(), b.into()).simd_into(token) + _mm_unpacklo_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn zip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { - _mm_min_pd(a.into(), b.into()).simd_into(token) + _mm_unpackhi_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn unzip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { - let a = a.into(); - let b = b.into(); - let intermediate = _mm_max_pd(a, b); - let b_is_nan = _mm_cmpunord_pd(b, b); - _mm_or_pd( - _mm_and_pd(b_is_nan, a), - _mm_andnot_pd(b_is_nan, intermediate), - ) - .simd_into(token) + _mm_shuffle_pd::<0b00>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn unzip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse2, a: f64x2, b: f64x2) -> f64x2 { - let a = a.into(); - let b = b.into(); - let intermediate = _mm_min_pd(a, b); - let b_is_nan = _mm_cmpunord_pd(b, b); - _mm_or_pd( - _mm_and_pd(b_is_nan, a), - _mm_andnot_pd(b_is_nan, intermediate), - ) - .simd_into(token) + _mm_shuffle_pd::<0b11>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] + fn interleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { + (self.zip_low_f64x2(a, b), self.zip_high_f64x2(a, b)) + } + #[inline(always)] + fn deinterleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { + (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) + } + #[inline(always)] fn mul_add_f64x2(self, a: f64x2, b: f64x2, c: f64x2) -> f64x2 { a * b + c } @@ -4445,6 +4445,22 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::max(a[0usize], b[0usize]), + i64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::min(a[0usize], b[0usize]), + i64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i64x2(self, a: i64x2, b: i64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -4543,22 +4559,6 @@ impl Simd for Sse2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::min(a[0usize], b[0usize]), - i64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::max(a[0usize], b[0usize]), - i64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i64x2(self, a: i64x2, b: i64x2) -> i64x4 { i64x4 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -4806,6 +4806,22 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::max(a[0usize], b[0usize]), + u64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::min(a[0usize], b[0usize]), + u64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u64x2(self, a: u64x2, b: u64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -4904,22 +4920,6 @@ impl Simd for Sse2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::min(a[0usize], b[0usize]), - u64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::max(a[0usize], b[0usize]), - u64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u64x2(self, a: u64x2, b: u64x2) -> u64x4 { u64x4 { val: crate::support::Aligned256([a.val.0, b.val.0]), diff --git a/fearless_simd/src/generated/sse4_2.rs b/fearless_simd/src/generated/sse4_2.rs index 6a86673f..bf9265c3 100644 --- a/fearless_simd/src/generated/sse4_2.rs +++ b/fearless_simd/src/generated/sse4_2.rs @@ -228,128 +228,128 @@ impl Simd for Sse4_2 { kernel(self, a, b) } #[inline(always)] - fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmpeq_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { + _mm_max_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_lt_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmplt_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { + _mm_min_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_le_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { + fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> mask32x4 { - _mm_castps_si128(_mm_cmple_ps(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { + let intermediate = _mm_max_ps(a.into(), b.into()); + let b_is_nan = _mm_cmpunord_ps(b.into(), b.into()); + _mm_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { - _mm_unpacklo_ps(a.into(), b.into()).simd_into(token) + let intermediate = _mm_min_ps(a.into(), b.into()); + let b_is_nan = _mm_cmpunord_ps(b.into(), b.into()); + _mm_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { - _mm_unpackhi_ps(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmpeq_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_lt_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { - _mm_shuffle_ps::<0b10_00_10_00>(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmplt_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn simd_le_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { - _mm_shuffle_ps::<0b11_01_11_01>(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> mask32x4 { + _mm_castps_si128(_mm_cmple_ps(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn interleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { - (self.zip_low_f32x4(a, b), self.zip_high_f32x4(a, b)) - } - #[inline(always)] - fn deinterleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { - (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) - } - #[inline(always)] - fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn zip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { - _mm_max_ps(a.into(), b.into()).simd_into(token) + _mm_unpacklo_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn zip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { - _mm_min_ps(a.into(), b.into()).simd_into(token) + _mm_unpackhi_ps(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn unzip_low_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { - let intermediate = _mm_max_ps(a.into(), b.into()); - let b_is_nan = _mm_cmpunord_ps(b.into(), b.into()); - _mm_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) + _mm_shuffle_ps::<0b10_00_10_00>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + fn unzip_high_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f32x4, b: f32x4) -> f32x4 { - let intermediate = _mm_min_ps(a.into(), b.into()); - let b_is_nan = _mm_cmpunord_ps(b.into(), b.into()); - _mm_blendv_ps(intermediate, a.into(), b_is_nan).simd_into(token) + _mm_shuffle_ps::<0b11_01_11_01>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] + fn interleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { + (self.zip_low_f32x4(a, b), self.zip_high_f32x4(a, b)) + } + #[inline(always)] + fn deinterleave_f32x4(self, a: f32x4, b: f32x4) -> (f32x4, f32x4) { + (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) + } + #[inline(always)] fn mul_add_f32x4(self, a: f32x4, b: f32x4, c: f32x4) -> f32x4 { a * b + c } @@ -779,6 +779,26 @@ impl Simd for Sse4_2 { .simd_into(self) } #[inline(always)] + fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: i8x16, b: i8x16) -> i8x16 { + _mm_max_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: i8x16, b: i8x16) -> i8x16 { + _mm_min_epi8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i8x16(self, a: i8x16, b: i8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -878,26 +898,6 @@ impl Simd for Sse4_2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: i8x16, b: i8x16) -> i8x16 { - _mm_min_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: i8x16, b: i8x16) -> i8x16 { - _mm_max_epi8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i8x16(self, a: i8x16, b: i8x16) -> i8x32 { i8x32 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -1239,6 +1239,26 @@ impl Simd for Sse4_2 { .simd_into(self) } #[inline(always)] + fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: u8x16, b: u8x16) -> u8x16 { + _mm_max_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: u8x16, b: u8x16) -> u8x16 { + _mm_min_epu8(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u8x16(self, a: u8x16, b: u8x16) -> mask8x16 { crate::kernel!( #[inline(always)] @@ -1344,26 +1364,6 @@ impl Simd for Sse4_2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: u8x16, b: u8x16) -> u8x16 { - _mm_min_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: u8x16, b: u8x16) -> u8x16 { - _mm_max_epu8(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u8x16(self, a: u8x16, b: u8x16) -> u8x32 { u8x32 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -1779,6 +1779,26 @@ impl Simd for Sse4_2 { .simd_into(self) } #[inline(always)] + fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: i16x8, b: i16x8) -> i16x8 { + _mm_max_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: i16x8, b: i16x8) -> i16x8 { + _mm_min_epi16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i16x8(self, a: i16x8, b: i16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -1878,26 +1898,6 @@ impl Simd for Sse4_2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: i16x8, b: i16x8) -> i16x8 { - _mm_min_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: i16x8, b: i16x8) -> i16x8 { - _mm_max_epi16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i16x8(self, a: i16x8, b: i16x8) -> i16x16 { i16x16 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -2193,6 +2193,26 @@ impl Simd for Sse4_2 { .simd_into(self) } #[inline(always)] + fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: u16x8, b: u16x8) -> u16x8 { + _mm_max_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: u16x8, b: u16x8) -> u16x8 { + _mm_min_epu16(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { crate::kernel!( #[inline(always)] @@ -2298,26 +2318,6 @@ impl Simd for Sse4_2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: u16x8, b: u16x8) -> u16x8 { - _mm_min_epu16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: u16x8, b: u16x8) -> u16x8 { - _mm_max_epu16(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u16x8(self, a: u16x8, b: u16x8) -> u16x16 { u16x16 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -2761,6 +2761,26 @@ impl Simd for Sse4_2 { .simd_into(self) } #[inline(always)] + fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: i32x4, b: i32x4) -> i32x4 { + _mm_max_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: i32x4, b: i32x4) -> i32x4 { + _mm_min_epi32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -2858,26 +2878,6 @@ impl Simd for Sse4_2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: i32x4, b: i32x4) -> i32x4 { - _mm_min_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: i32x4, b: i32x4) -> i32x4 { - _mm_max_epi32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_i32x4(self, a: i32x4, b: i32x4) -> i32x8 { i32x8 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -3165,6 +3165,26 @@ impl Simd for Sse4_2 { .simd_into(self) } #[inline(always)] + fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: u32x4, b: u32x4) -> u32x4 { + _mm_max_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] + fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Sse4_2, a: u32x4, b: u32x4) -> u32x4 { + _mm_min_epu32(a.into(), b.into()).simd_into(token) + } + ); + kernel(self, a, b) + } + #[inline(always)] fn simd_eq_u32x4(self, a: u32x4, b: u32x4) -> mask32x4 { crate::kernel!( #[inline(always)] @@ -3268,26 +3288,6 @@ impl Simd for Sse4_2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: u32x4, b: u32x4) -> u32x4 { - _mm_min_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] - fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - crate::kernel!( - #[inline(always)] - fn kernel(token: Sse4_2, a: u32x4, b: u32x4) -> u32x4 { - _mm_max_epu32(a.into(), b.into()).simd_into(token) - } - ); - kernel(self, a, b) - } - #[inline(always)] fn combine_u32x4(self, a: u32x4, b: u32x4) -> u32x8 { u32x8 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -3716,128 +3716,128 @@ impl Simd for Sse4_2 { kernel(self, a, b) } #[inline(always)] - fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmpeq_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { + _mm_max_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_lt_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmplt_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { + _mm_min_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn simd_le_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { + fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> mask64x2 { - _mm_castpd_si128(_mm_cmple_pd(a.into(), b.into())).simd_into(token) + fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { + let intermediate = _mm_max_pd(a.into(), b.into()); + let b_is_nan = _mm_cmpunord_pd(b.into(), b.into()); + _mm_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { - _mm_unpacklo_pd(a.into(), b.into()).simd_into(token) + let intermediate = _mm_min_pd(a.into(), b.into()); + let b_is_nan = _mm_cmpunord_pd(b.into(), b.into()); + _mm_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn zip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { - _mm_unpackhi_pd(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmpeq_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_lt_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { - _mm_shuffle_pd::<0b00>(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmplt_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn unzip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn simd_le_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { crate::kernel!( #[inline(always)] - fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { - _mm_shuffle_pd::<0b11>(a.into(), b.into()).simd_into(token) + fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> mask64x2 { + _mm_castpd_si128(_mm_cmple_pd(a.into(), b.into())).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn interleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { - (self.zip_low_f64x2(a, b), self.zip_high_f64x2(a, b)) - } - #[inline(always)] - fn deinterleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { - (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) - } - #[inline(always)] - fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn zip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { - _mm_max_pd(a.into(), b.into()).simd_into(token) + _mm_unpacklo_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn zip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { - _mm_min_pd(a.into(), b.into()).simd_into(token) + _mm_unpackhi_pd(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn unzip_low_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { - let intermediate = _mm_max_pd(a.into(), b.into()); - let b_is_nan = _mm_cmpunord_pd(b.into(), b.into()); - _mm_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) + _mm_shuffle_pd::<0b00>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] - fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + fn unzip_high_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] fn kernel(token: Sse4_2, a: f64x2, b: f64x2) -> f64x2 { - let intermediate = _mm_min_pd(a.into(), b.into()); - let b_is_nan = _mm_cmpunord_pd(b.into(), b.into()); - _mm_blendv_pd(intermediate, a.into(), b_is_nan).simd_into(token) + _mm_shuffle_pd::<0b11>(a.into(), b.into()).simd_into(token) } ); kernel(self, a, b) } #[inline(always)] + fn interleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { + (self.zip_low_f64x2(a, b), self.zip_high_f64x2(a, b)) + } + #[inline(always)] + fn deinterleave_f64x2(self, a: f64x2, b: f64x2) -> (f64x2, f64x2) { + (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) + } + #[inline(always)] fn mul_add_f64x2(self, a: f64x2, b: f64x2, c: f64x2) -> f64x2 { a * b + c } @@ -4133,6 +4133,22 @@ impl Simd for Sse4_2 { .simd_into(self) } #[inline(always)] + fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::max(a[0usize], b[0usize]), + i64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::min(a[0usize], b[0usize]), + i64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i64x2(self, a: i64x2, b: i64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -4222,22 +4238,6 @@ impl Simd for Sse4_2 { kernel(self, a, b, c) } #[inline(always)] - fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::min(a[0usize], b[0usize]), - i64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::max(a[0usize], b[0usize]), - i64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i64x2(self, a: i64x2, b: i64x2) -> i64x4 { i64x4 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -4503,6 +4503,22 @@ impl Simd for Sse4_2 { .simd_into(self) } #[inline(always)] + fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::max(a[0usize], b[0usize]), + u64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::min(a[0usize], b[0usize]), + u64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u64x2(self, a: u64x2, b: u64x2) -> mask64x2 { crate::kernel!( #[inline(always)] @@ -4592,22 +4608,6 @@ impl Simd for Sse4_2 { kernel(self, a, b, c) } #[inline(always)] - fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::min(a[0usize], b[0usize]), - u64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::max(a[0usize], b[0usize]), - u64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u64x2(self, a: u64x2, b: u64x2) -> u64x4 { u64x4 { val: crate::support::Aligned256([a.val.0, b.val.0]), diff --git a/fearless_simd/src/generated/wasm.rs b/fearless_simd/src/generated/wasm.rs index ccc8e91a..c370d359 100644 --- a/fearless_simd/src/generated/wasm.rs +++ b/fearless_simd/src/generated/wasm.rs @@ -158,6 +158,40 @@ impl Simd for WasmSimd128 { v128_or(magnitude, sign_bits).simd_into(self) } #[inline(always)] + fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + #[cfg(target_feature = "relaxed-simd")] + { + f32x4_relaxed_max(a.into(), b.into()).simd_into(self) + } + #[cfg(not(target_feature = "relaxed-simd"))] + { + f32x4_max(a.into(), b.into()).simd_into(self) + } + } + #[inline(always)] + fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + #[cfg(target_feature = "relaxed-simd")] + { + f32x4_relaxed_min(a.into(), b.into()).simd_into(self) + } + #[cfg(not(target_feature = "relaxed-simd"))] + { + f32x4_min(a.into(), b.into()).simd_into(self) + } + } + #[inline(always)] + fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + let intermediate = f32x4_pmax(b.into(), a.into()); + let b_is_nan = f32x4_ne(b.into(), b.into()); + v128_bitselect(a.into(), intermediate, b_is_nan).simd_into(self) + } + #[inline(always)] + fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { + let intermediate = f32x4_pmin(b.into(), a.into()); + let b_is_nan = f32x4_ne(b.into(), b.into()); + v128_bitselect(a.into(), intermediate, b_is_nan).simd_into(self) + } + #[inline(always)] fn simd_eq_f32x4(self, a: f32x4, b: f32x4) -> mask32x4 { f32x4_eq(a.into(), b.into()).simd_into(self) } @@ -194,40 +228,6 @@ impl Simd for WasmSimd128 { (self.unzip_low_f32x4(a, b), self.unzip_high_f32x4(a, b)) } #[inline(always)] - fn max_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - #[cfg(target_feature = "relaxed-simd")] - { - f32x4_relaxed_max(a.into(), b.into()).simd_into(self) - } - #[cfg(not(target_feature = "relaxed-simd"))] - { - f32x4_max(a.into(), b.into()).simd_into(self) - } - } - #[inline(always)] - fn min_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - #[cfg(target_feature = "relaxed-simd")] - { - f32x4_relaxed_min(a.into(), b.into()).simd_into(self) - } - #[cfg(not(target_feature = "relaxed-simd"))] - { - f32x4_min(a.into(), b.into()).simd_into(self) - } - } - #[inline(always)] - fn max_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - let intermediate = f32x4_pmax(b.into(), a.into()); - let b_is_nan = f32x4_ne(b.into(), b.into()); - v128_bitselect(a.into(), intermediate, b_is_nan).simd_into(self) - } - #[inline(always)] - fn min_precise_f32x4(self, a: f32x4, b: f32x4) -> f32x4 { - let intermediate = f32x4_pmin(b.into(), a.into()); - let b_is_nan = f32x4_ne(b.into(), b.into()); - v128_bitselect(a.into(), intermediate, b_is_nan).simd_into(self) - } - #[inline(always)] fn mul_add_f32x4(self, a: f32x4, b: f32x4, c: f32x4) -> f32x4 { #[cfg(target_feature = "relaxed-simd")] { @@ -474,6 +474,14 @@ impl Simd for WasmSimd128 { .simd_into(self) } #[inline(always)] + fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + i8x16_max(a.into(), b.into()).simd_into(self) + } + #[inline(always)] + fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { + i8x16_min(a.into(), b.into()).simd_into(self) + } + #[inline(always)] fn simd_eq_i8x16(self, a: i8x16, b: i8x16) -> mask8x16 { i8x16_eq(a.into(), b.into()).simd_into(self) } @@ -534,14 +542,6 @@ impl Simd for WasmSimd128 { } } #[inline(always)] - fn min_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - i8x16_min(a.into(), b.into()).simd_into(self) - } - #[inline(always)] - fn max_i8x16(self, a: i8x16, b: i8x16) -> i8x16 { - i8x16_max(a.into(), b.into()).simd_into(self) - } - #[inline(always)] fn combine_i8x16(self, a: i8x16, b: i8x16) -> i8x32 { i8x32 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -752,6 +752,14 @@ impl Simd for WasmSimd128 { .simd_into(self) } #[inline(always)] + fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + u8x16_max(a.into(), b.into()).simd_into(self) + } + #[inline(always)] + fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { + u8x16_min(a.into(), b.into()).simd_into(self) + } + #[inline(always)] fn simd_eq_u8x16(self, a: u8x16, b: u8x16) -> mask8x16 { u8x16_eq(a.into(), b.into()).simd_into(self) } @@ -812,14 +820,6 @@ impl Simd for WasmSimd128 { } } #[inline(always)] - fn min_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - u8x16_min(a.into(), b.into()).simd_into(self) - } - #[inline(always)] - fn max_u8x16(self, a: u8x16, b: u8x16) -> u8x16 { - u8x16_max(a.into(), b.into()).simd_into(self) - } - #[inline(always)] fn combine_u8x16(self, a: u8x16, b: u8x16) -> u8x32 { u8x32 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -1079,6 +1079,14 @@ impl Simd for WasmSimd128 { .simd_into(self) } #[inline(always)] + fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + i16x8_max(a.into(), b.into()).simd_into(self) + } + #[inline(always)] + fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { + i16x8_min(a.into(), b.into()).simd_into(self) + } + #[inline(always)] fn simd_eq_i16x8(self, a: i16x8, b: i16x8) -> mask16x8 { i16x8_eq(a.into(), b.into()).simd_into(self) } @@ -1126,14 +1134,6 @@ impl Simd for WasmSimd128 { } } #[inline(always)] - fn min_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - i16x8_min(a.into(), b.into()).simd_into(self) - } - #[inline(always)] - fn max_i16x8(self, a: i16x8, b: i16x8) -> i16x8 { - i16x8_max(a.into(), b.into()).simd_into(self) - } - #[inline(always)] fn combine_i16x8(self, a: i16x8, b: i16x8) -> i16x16 { i16x16 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -1305,6 +1305,14 @@ impl Simd for WasmSimd128 { .simd_into(self) } #[inline(always)] + fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + u16x8_max(a.into(), b.into()).simd_into(self) + } + #[inline(always)] + fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { + u16x8_min(a.into(), b.into()).simd_into(self) + } + #[inline(always)] fn simd_eq_u16x8(self, a: u16x8, b: u16x8) -> mask16x8 { u16x8_eq(a.into(), b.into()).simd_into(self) } @@ -1352,14 +1360,6 @@ impl Simd for WasmSimd128 { } } #[inline(always)] - fn min_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - u16x8_min(a.into(), b.into()).simd_into(self) - } - #[inline(always)] - fn max_u16x8(self, a: u16x8, b: u16x8) -> u16x8 { - u16x8_max(a.into(), b.into()).simd_into(self) - } - #[inline(always)] fn combine_u16x8(self, a: u16x8, b: u16x8) -> u16x16 { u16x16 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -1606,6 +1606,14 @@ impl Simd for WasmSimd128 { .simd_into(self) } #[inline(always)] + fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + i32x4_max(a.into(), b.into()).simd_into(self) + } + #[inline(always)] + fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { + i32x4_min(a.into(), b.into()).simd_into(self) + } + #[inline(always)] fn simd_eq_i32x4(self, a: i32x4, b: i32x4) -> mask32x4 { i32x4_eq(a.into(), b.into()).simd_into(self) } @@ -1653,14 +1661,6 @@ impl Simd for WasmSimd128 { } } #[inline(always)] - fn min_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - i32x4_min(a.into(), b.into()).simd_into(self) - } - #[inline(always)] - fn max_i32x4(self, a: i32x4, b: i32x4) -> i32x4 { - i32x4_max(a.into(), b.into()).simd_into(self) - } - #[inline(always)] fn combine_i32x4(self, a: i32x4, b: i32x4) -> i32x8 { i32x8 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -1828,6 +1828,14 @@ impl Simd for WasmSimd128 { .simd_into(self) } #[inline(always)] + fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + u32x4_max(a.into(), b.into()).simd_into(self) + } + #[inline(always)] + fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { + u32x4_min(a.into(), b.into()).simd_into(self) + } + #[inline(always)] fn simd_eq_u32x4(self, a: u32x4, b: u32x4) -> mask32x4 { u32x4_eq(a.into(), b.into()).simd_into(self) } @@ -1875,14 +1883,6 @@ impl Simd for WasmSimd128 { } } #[inline(always)] - fn min_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - u32x4_min(a.into(), b.into()).simd_into(self) - } - #[inline(always)] - fn max_u32x4(self, a: u32x4, b: u32x4) -> u32x4 { - u32x4_max(a.into(), b.into()).simd_into(self) - } - #[inline(always)] fn combine_u32x4(self, a: u32x4, b: u32x4) -> u32x8 { u32x8 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -2116,6 +2116,40 @@ impl Simd for WasmSimd128 { v128_or(magnitude, sign_bits).simd_into(self) } #[inline(always)] + fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + #[cfg(target_feature = "relaxed-simd")] + { + f64x2_relaxed_max(a.into(), b.into()).simd_into(self) + } + #[cfg(not(target_feature = "relaxed-simd"))] + { + f64x2_max(a.into(), b.into()).simd_into(self) + } + } + #[inline(always)] + fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + #[cfg(target_feature = "relaxed-simd")] + { + f64x2_relaxed_min(a.into(), b.into()).simd_into(self) + } + #[cfg(not(target_feature = "relaxed-simd"))] + { + f64x2_min(a.into(), b.into()).simd_into(self) + } + } + #[inline(always)] + fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + let intermediate = f64x2_pmax(b.into(), a.into()); + let b_is_nan = f64x2_ne(b.into(), b.into()); + v128_bitselect(a.into(), intermediate, b_is_nan).simd_into(self) + } + #[inline(always)] + fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { + let intermediate = f64x2_pmin(b.into(), a.into()); + let b_is_nan = f64x2_ne(b.into(), b.into()); + v128_bitselect(a.into(), intermediate, b_is_nan).simd_into(self) + } + #[inline(always)] fn simd_eq_f64x2(self, a: f64x2, b: f64x2) -> mask64x2 { f64x2_eq(a.into(), b.into()).simd_into(self) } @@ -2152,40 +2186,6 @@ impl Simd for WasmSimd128 { (self.unzip_low_f64x2(a, b), self.unzip_high_f64x2(a, b)) } #[inline(always)] - fn max_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - #[cfg(target_feature = "relaxed-simd")] - { - f64x2_relaxed_max(a.into(), b.into()).simd_into(self) - } - #[cfg(not(target_feature = "relaxed-simd"))] - { - f64x2_max(a.into(), b.into()).simd_into(self) - } - } - #[inline(always)] - fn min_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - #[cfg(target_feature = "relaxed-simd")] - { - f64x2_relaxed_min(a.into(), b.into()).simd_into(self) - } - #[cfg(not(target_feature = "relaxed-simd"))] - { - f64x2_min(a.into(), b.into()).simd_into(self) - } - } - #[inline(always)] - fn max_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - let intermediate = f64x2_pmax(b.into(), a.into()); - let b_is_nan = f64x2_ne(b.into(), b.into()); - v128_bitselect(a.into(), intermediate, b_is_nan).simd_into(self) - } - #[inline(always)] - fn min_precise_f64x2(self, a: f64x2, b: f64x2) -> f64x2 { - let intermediate = f64x2_pmin(b.into(), a.into()); - let b_is_nan = f64x2_ne(b.into(), b.into()); - v128_bitselect(a.into(), intermediate, b_is_nan).simd_into(self) - } - #[inline(always)] fn mul_add_f64x2(self, a: f64x2, b: f64x2, c: f64x2) -> f64x2 { #[cfg(target_feature = "relaxed-simd")] { @@ -2373,6 +2373,22 @@ impl Simd for WasmSimd128 { .simd_into(self) } #[inline(always)] + fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::max(a[0usize], b[0usize]), + i64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { + [ + i64::min(a[0usize], b[0usize]), + i64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_i64x2(self, a: i64x2, b: i64x2) -> mask64x2 { i64x2_eq(a.into(), b.into()).simd_into(self) } @@ -2420,22 +2436,6 @@ impl Simd for WasmSimd128 { } } #[inline(always)] - fn min_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::min(a[0usize], b[0usize]), - i64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_i64x2(self, a: i64x2, b: i64x2) -> i64x2 { - [ - i64::max(a[0usize], b[0usize]), - i64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_i64x2(self, a: i64x2, b: i64x2) -> i64x4 { i64x4 { val: crate::support::Aligned256([a.val.0, b.val.0]), @@ -2589,6 +2589,22 @@ impl Simd for WasmSimd128 { .simd_into(self) } #[inline(always)] + fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::max(a[0usize], b[0usize]), + u64::max(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] + fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { + [ + u64::min(a[0usize], b[0usize]), + u64::min(a[1usize], b[1usize]), + ] + .simd_into(self) + } + #[inline(always)] fn simd_eq_u64x2(self, a: u64x2, b: u64x2) -> mask64x2 { [ -(u64::eq(&a[0usize], &b[0usize]) as i64), @@ -2648,22 +2664,6 @@ impl Simd for WasmSimd128 { } } #[inline(always)] - fn min_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::min(a[0usize], b[0usize]), - u64::min(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] - fn max_u64x2(self, a: u64x2, b: u64x2) -> u64x2 { - [ - u64::max(a[0usize], b[0usize]), - u64::max(a[1usize], b[1usize]), - ] - .simd_into(self) - } - #[inline(always)] fn combine_u64x2(self, a: u64x2, b: u64x2) -> u64x4 { u64x4 { val: crate::support::Aligned256([a.val.0, b.val.0]), diff --git a/fearless_simd_gen/src/mk_simd_types.rs b/fearless_simd_gen/src/mk_simd_types.rs index fd72129a..7eddf8d8 100644 --- a/fearless_simd_gen/src/mk_simd_types.rs +++ b/fearless_simd_gen/src/mk_simd_types.rs @@ -444,7 +444,18 @@ fn simd_vec_impl(ty: &VecType) -> TokenStream { let Some(call_args) = sig.forwarding_call_args() else { continue; }; - let trait_method = generic_op_name(method, ty); + // Integer min/max have no precision-related edge cases, so the precise + // variants deliberately forward to the regular backend operations. + let backend_method = if matches!(ty.scalar, ScalarType::Int | ScalarType::Unsigned) { + match method { + "min_precise" => "min", + "max_precise" => "max", + _ => method, + } + } else { + method + }; + let trait_method = generic_op_name(backend_method, ty); let method_sig = op .vec_trait_method_sig() .expect("base trait operation must have a vector method signature"); diff --git a/fearless_simd_gen/src/ops.rs b/fearless_simd_gen/src/ops.rs index 173e40d5..bea5a544 100644 --- a/fearless_simd_gen/src/ops.rs +++ b/fearless_simd_gen/src/ops.rs @@ -576,6 +576,44 @@ const BASE_OPS: &[Op] = &[ ]; const COMMON_BASE_OPS: &[Op] = &[ + Op::new( + "max", + OpKind::BaseTraitMethod, + OpSig::Binary, + "Return the element-wise maximum of two vectors.\n\n\ + For floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\n\ + If one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one.", + ), + Op::new( + "min", + OpKind::BaseTraitMethod, + OpSig::Binary, + "Return the element-wise minimum of two vectors.\n\n\ + For floating-point vectors, if either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\n\ + If one floating-point operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one.", + ), + Op::new( + "max_precise", + OpKind::BaseTraitMethod, + OpSig::Binary, + "Return the element-wise maximum of two vectors.\n\n\ + For integer vectors, this operation is the same as `max`.\n\n\ + For floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\n\ + If one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\n\ + If a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\n\ + Signaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them.", + ), + Op::new( + "min_precise", + OpKind::BaseTraitMethod, + OpSig::Binary, + "Return the element-wise minimum of two vectors.\n\n\ + For integer vectors, this operation is the same as `min`.\n\n\ + For floating-point vectors, if one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\n\ + If one floating-point operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\n\ + If a floating-point operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\n\ + Signaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them.", + ), Op::new( "simd_eq", OpKind::BaseTraitMethod, @@ -761,42 +799,6 @@ const FLOAT_OPS: &[Op] = &[ "Return a vector with the magnitude of `{arg0}` and the sign of `{arg1}` for each element.\n\n\ This operation copies the sign bit, so if an input element is NaN, the output element will be a NaN with the same payload and a copied sign bit.", ), - Op::new( - "max", - OpKind::VecTraitMethod, - OpSig::Binary, - "Return the element-wise maximum of two vectors.\n\n\ - If either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `max_precise` for a version that returns the non-NaN operand if only one is NaN.\n\n\ - If one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one.", - ), - Op::new( - "min", - OpKind::VecTraitMethod, - OpSig::Binary, - "Return the element-wise minimum of two vectors.\n\n\ - If either operand is NaN, the result for that lane is implementation-defined-- it could be either the first or second operand. See `min_precise` for a version that returns the non-NaN operand if only one is NaN.\n\n\ - If one operand is positive zero and the other is negative zero, the result is also implementation-defined, and it could be either one.", - ), - Op::new( - "max_precise", - OpKind::VecTraitMethod, - OpSig::Binary, - "Return the element-wise maximum of two vectors.\n\n\ - If one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\n\ - If one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\n\ - If an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\n\ - Signaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them.", - ), - Op::new( - "min_precise", - OpKind::VecTraitMethod, - OpSig::Binary, - "Return the element-wise minimum of two vectors.\n\n\ - If one operand is a quiet NaN and the other is not, this operation will choose the non-NaN operand.\n\n\ - If one operand is positive zero and the other is negative zero, the result is implementation-defined, and it could be either one.\n\n\ - If an operand is a *signaling* NaN, the result is not just implementation-defined, but fully non-deterministic: it may be either NaN or the non-NaN operand.\n\ - Signaling NaN values are not produced by floating-point math operations, only from manual initialization with specific bit patterns. You probably don't need to worry about them.", - ), Op::new( "mul_add", OpKind::VecTraitMethod, @@ -935,18 +937,6 @@ const INT_OPS: &[Op] = &[ "Select elements from {arg1} and {arg2} based on the mask operand {arg0}.\n\n\ This operation's behavior is unspecified if {arg0} was constructed from signed integer lanes that are neither all-zeroes (integer value 0) nor all-ones (integer value -1). See the [`Select`] trait's documentation for more information.", ), - Op::new( - "min", - OpKind::VecTraitMethod, - OpSig::Binary, - "Return the element-wise minimum of two vectors.", - ), - Op::new( - "max", - OpKind::VecTraitMethod, - OpSig::Binary, - "Return the element-wise maximum of two vectors.", - ), ]; // Long blurb shared between all the mask reduction operations. Needs to be a macro because consts don't work in the @@ -1184,7 +1174,12 @@ pub(crate) fn ops_for_type(ty: &VecType) -> Vec { ScalarType::Mask => false, }; if common_ops_follow { - ops.extend_from_slice(COMMON_BASE_OPS); + // Integer precise min/max are exposed by `SimdBase`, but forward to + // the ordinary integer backend operations in `simd_vec_impl`. + ops.extend(COMMON_BASE_OPS.iter().copied().filter(|op| { + ty.scalar == ScalarType::Float + || !matches!(op.method, "min_precise" | "max_precise") + })); } }