From 3070dd42445cd3687717acff92587e12d991d6a9 Mon Sep 17 00:00:00 2001 From: RunDevelopment Date: Wed, 30 Sep 2026 13:07:49 +0200 Subject: [PATCH 1/3] Add `round` --- fearless_simd/src/generated/avx2.rs | 20 ++++ fearless_simd/src/generated/avx512.rs | 30 +++++ fearless_simd/src/generated/fallback.rs | 14 +++ fearless_simd/src/generated/neon.rs | 20 ++++ fearless_simd/src/generated/simd_trait.rs | 44 +++++-- fearless_simd/src/generated/simd_types.rs | 24 ++++ fearless_simd/src/generated/sse2.rs | 10 ++ fearless_simd/src/generated/sse4_2.rs | 10 ++ fearless_simd/src/generated/wasm.rs | 10 ++ fearless_simd/src/math.rs | 8 ++ fearless_simd_gen/src/arch/fallback.rs | 1 + fearless_simd_gen/src/arch/neon.rs | 1 + fearless_simd_gen/src/generic.rs | 17 +++ fearless_simd_gen/src/mk_wasm.rs | 6 +- fearless_simd_gen/src/mk_x86.rs | 3 +- fearless_simd_gen/src/ops.rs | 11 +- fearless_simd_tests/tests/harness/ops/mod.rs | 1 + .../tests/harness/ops/round.rs | 110 ++++++++++++++++++ 18 files changed, 329 insertions(+), 11 deletions(-) create mode 100644 fearless_simd_tests/tests/harness/ops/round.rs diff --git a/fearless_simd/src/generated/avx2.rs b/fearless_simd/src/generated/avx2.rs index 336edc19..0c20dd2f 100644 --- a/fearless_simd/src/generated/avx2.rs +++ b/fearless_simd/src/generated/avx2.rs @@ -531,6 +531,11 @@ impl Simd for Avx2 { kernel(self, a) } #[inline(always)] + fn round_f32x4(self, a: f32x4) -> f32x4 { + let bias = self.splat_f32x4(const { f32::next_down(0.5) }); + (a + self.copysign_f32x4(bias, a)).trunc() + } + #[inline(always)] fn fract_f32x4(self, a: f32x4) -> f32x4 { a - self.trunc_f32x4(a) } @@ -5207,6 +5212,11 @@ impl Simd for Avx2 { kernel(self, a) } #[inline(always)] + fn round_f64x2(self, a: f64x2) -> f64x2 { + let bias = self.splat_f64x2(const { f64::next_down(0.5) }); + (a + self.copysign_f64x2(bias, a)).trunc() + } + #[inline(always)] fn fract_f64x2(self, a: f64x2) -> f64x2 { a - self.trunc_f64x2(a) } @@ -7036,6 +7046,11 @@ impl Simd for Avx2 { kernel(self, a) } #[inline(always)] + fn round_f32x8(self, a: f32x8) -> f32x8 { + let bias = self.splat_f32x8(const { f32::next_down(0.5) }); + (a + self.copysign_f32x8(bias, a)).trunc() + } + #[inline(always)] fn fract_f32x8(self, a: f32x8) -> f32x8 { a - self.trunc_f32x8(a) } @@ -11373,6 +11388,11 @@ impl Simd for Avx2 { kernel(self, a) } #[inline(always)] + fn round_f64x4(self, a: f64x4) -> f64x4 { + let bias = self.splat_f64x4(const { f64::next_down(0.5) }); + (a + self.copysign_f64x4(bias, a)).trunc() + } + #[inline(always)] fn fract_f64x4(self, a: f64x4) -> f64x4 { a - self.trunc_f64x4(a) } diff --git a/fearless_simd/src/generated/avx512.rs b/fearless_simd/src/generated/avx512.rs index 8092b7ec..c620c5f1 100644 --- a/fearless_simd/src/generated/avx512.rs +++ b/fearless_simd/src/generated/avx512.rs @@ -804,6 +804,11 @@ impl Simd for Avx512 { kernel(self, a) } #[inline(always)] + fn round_f32x4(self, a: f32x4) -> f32x4 { + let bias = self.splat_f32x4(const { f32::next_down(0.5) }); + (a + self.copysign_f32x4(bias, a)).trunc() + } + #[inline(always)] fn fract_f32x4(self, a: f32x4) -> f32x4 { a - self.trunc_f32x4(a) } @@ -5150,6 +5155,11 @@ impl Simd for Avx512 { kernel(self, a) } #[inline(always)] + fn round_f64x2(self, a: f64x2) -> f64x2 { + let bias = self.splat_f64x2(const { f64::next_down(0.5) }); + (a + self.copysign_f64x2(bias, a)).trunc() + } + #[inline(always)] fn fract_f64x2(self, a: f64x2) -> f64x2 { a - self.trunc_f64x2(a) } @@ -6896,6 +6906,11 @@ impl Simd for Avx512 { kernel(self, a) } #[inline(always)] + fn round_f32x8(self, a: f32x8) -> f32x8 { + let bias = self.splat_f32x8(const { f32::next_down(0.5) }); + (a + self.copysign_f32x8(bias, a)).trunc() + } + #[inline(always)] fn fract_f32x8(self, a: f32x8) -> f32x8 { a - self.trunc_f32x8(a) } @@ -11091,6 +11106,11 @@ impl Simd for Avx512 { kernel(self, a) } #[inline(always)] + fn round_f64x4(self, a: f64x4) -> f64x4 { + let bias = self.splat_f64x4(const { f64::next_down(0.5) }); + (a + self.copysign_f64x4(bias, a)).trunc() + } + #[inline(always)] fn fract_f64x4(self, a: f64x4) -> f64x4 { a - self.trunc_f64x4(a) } @@ -12835,6 +12855,11 @@ impl Simd for Avx512 { kernel(self, a) } #[inline(always)] + fn round_f32x16(self, a: f32x16) -> f32x16 { + let bias = self.splat_f32x16(const { f32::next_down(0.5) }); + (a + self.copysign_f32x16(bias, a)).trunc() + } + #[inline(always)] fn fract_f32x16(self, a: f32x16) -> f32x16 { a - self.trunc_f32x16(a) } @@ -17090,6 +17115,11 @@ impl Simd for Avx512 { kernel(self, a) } #[inline(always)] + fn round_f64x8(self, a: f64x8) -> f64x8 { + let bias = self.splat_f64x8(const { f64::next_down(0.5) }); + (a + self.copysign_f64x8(bias, a)).trunc() + } + #[inline(always)] fn fract_f64x8(self, a: f64x8) -> f64x8 { a - self.trunc_f64x8(a) } diff --git a/fearless_simd/src/generated/fallback.rs b/fearless_simd/src/generated/fallback.rs index 2aec237c..9a675323 100644 --- a/fearless_simd/src/generated/fallback.rs +++ b/fearless_simd/src/generated/fallback.rs @@ -403,6 +403,16 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn round_f32x4(self, a: f32x4) -> f32x4 { + [ + f32::round(a[0usize]), + f32::round(a[1usize]), + f32::round(a[2usize]), + f32::round(a[3usize]), + ] + .simd_into(self) + } + #[inline(always)] fn fract_f32x4(self, a: f32x4) -> f32x4 { [ f32::fract(a[0usize]), @@ -5703,6 +5713,10 @@ impl Simd for Fallback { .simd_into(self) } #[inline(always)] + fn round_f64x2(self, a: f64x2) -> f64x2 { + [f64::round(a[0usize]), f64::round(a[1usize])].simd_into(self) + } + #[inline(always)] fn fract_f64x2(self, a: f64x2) -> f64x2 { [f64::fract(a[0usize]), f64::fract(a[1usize])].simd_into(self) } diff --git a/fearless_simd/src/generated/neon.rs b/fearless_simd/src/generated/neon.rs index e964bc51..644ac007 100644 --- a/fearless_simd/src/generated/neon.rs +++ b/fearless_simd/src/generated/neon.rs @@ -481,6 +481,16 @@ impl Simd for Neon { kernel(self, a) } #[inline(always)] + fn round_f32x4(self, a: f32x4) -> f32x4 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f32x4) -> f32x4 { + vrndaq_f32(a.into()).simd_into(token) + } + ); + kernel(self, a) + } + #[inline(always)] fn fract_f32x4(self, a: f32x4) -> f32x4 { crate::kernel!( #[inline(always)] @@ -4262,6 +4272,16 @@ impl Simd for Neon { kernel(self, a) } #[inline(always)] + fn round_f64x2(self, a: f64x2) -> f64x2 { + crate::kernel!( + #[inline(always)] + fn kernel(token: Neon, a: f64x2) -> f64x2 { + vrndaq_f64(a.into()).simd_into(token) + } + ); + kernel(self, a) + } + #[inline(always)] fn fract_f64x2(self, a: f64x2) -> f64x2 { crate::kernel!( #[inline(always)] diff --git a/fearless_simd/src/generated/simd_trait.rs b/fearless_simd/src/generated/simd_trait.rs index bdf3cfef..f307c2d5 100644 --- a/fearless_simd/src/generated/simd_trait.rs +++ b/fearless_simd/src/generated/simd_trait.rs @@ -370,8 +370,10 @@ pub trait Simd: fn floor_f32x4(self, a: f32x4) -> f32x4; #[doc = "Return the smallest integer greater than or equal to each element, that is, round towards positive infinity."] fn ceil_f32x4(self, a: f32x4) -> f32x4; - #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."] + #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] fn round_ties_even_f32x4(self, a: f32x4) -> f32x4; + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + fn round_f32x4(self, a: f32x4) -> f32x4; #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."] fn fract_f32x4(self, a: f32x4) -> f32x4; #[doc = "Return the integer part of each element, rounding towards zero."] @@ -1597,8 +1599,10 @@ pub trait Simd: fn floor_f64x2(self, a: f64x2) -> f64x2; #[doc = "Return the smallest integer greater than or equal to each element, that is, round towards positive infinity."] fn ceil_f64x2(self, a: f64x2) -> f64x2; - #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."] + #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] fn round_ties_even_f64x2(self, a: f64x2) -> f64x2; + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + fn round_f64x2(self, a: f64x2) -> f64x2; #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."] fn fract_f64x2(self, a: f64x2) -> f64x2; #[doc = "Return the integer part of each element, rounding towards zero."] @@ -2349,7 +2353,7 @@ pub trait Simd: let (a0, a1) = self.split_f32x8(a); self.combine_f32x4(self.ceil_f32x4(a0), self.ceil_f32x4(a1)) } - #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."] + #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] #[inline(always)] fn round_ties_even_f32x8(self, a: f32x8) -> f32x8 { let (a0, a1) = self.split_f32x8(a); @@ -2358,6 +2362,12 @@ pub trait Simd: self.round_ties_even_f32x4(a1), ) } + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[inline(always)] + fn round_f32x8(self, a: f32x8) -> f32x8 { + let (a0, a1) = self.split_f32x8(a); + self.combine_f32x4(self.round_f32x4(a0), self.round_f32x4(a1)) + } #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."] #[inline(always)] fn fract_f32x8(self, a: f32x8) -> f32x8 { @@ -5274,7 +5284,7 @@ pub trait Simd: let (a0, a1) = self.split_f64x4(a); self.combine_f64x2(self.ceil_f64x2(a0), self.ceil_f64x2(a1)) } - #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."] + #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] #[inline(always)] fn round_ties_even_f64x4(self, a: f64x4) -> f64x4 { let (a0, a1) = self.split_f64x4(a); @@ -5283,6 +5293,12 @@ pub trait Simd: self.round_ties_even_f64x2(a1), ) } + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[inline(always)] + fn round_f64x4(self, a: f64x4) -> f64x4 { + let (a0, a1) = self.split_f64x4(a); + self.combine_f64x2(self.round_f64x2(a0), self.round_f64x2(a1)) + } #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."] #[inline(always)] fn fract_f64x4(self, a: f64x4) -> f64x4 { @@ -6573,7 +6589,7 @@ pub trait Simd: let (a0, a1) = self.split_f32x16(a); self.combine_f32x8(self.ceil_f32x8(a0), self.ceil_f32x8(a1)) } - #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."] + #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] #[inline(always)] fn round_ties_even_f32x16(self, a: f32x16) -> f32x16 { let (a0, a1) = self.split_f32x16(a); @@ -6582,6 +6598,12 @@ pub trait Simd: self.round_ties_even_f32x8(a1), ) } + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[inline(always)] + fn round_f32x16(self, a: f32x16) -> f32x16 { + let (a0, a1) = self.split_f32x16(a); + self.combine_f32x8(self.round_f32x8(a0), self.round_f32x8(a1)) + } #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."] #[inline(always)] fn fract_f32x16(self, a: f32x16) -> f32x16 { @@ -9501,7 +9523,7 @@ pub trait Simd: let (a0, a1) = self.split_f64x8(a); self.combine_f64x4(self.ceil_f64x4(a0), self.ceil_f64x4(a1)) } - #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."] + #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] #[inline(always)] fn round_ties_even_f64x8(self, a: f64x8) -> f64x8 { let (a0, a1) = self.split_f64x8(a); @@ -9510,6 +9532,12 @@ pub trait Simd: self.round_ties_even_f64x4(a1), ) } + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[inline(always)] + fn round_f64x8(self, a: f64x8) -> f64x8 { + let (a0, a1) = self.split_f64x8(a); + self.combine_f64x4(self.round_f64x4(a0), self.round_f64x4(a1)) + } #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."] #[inline(always)] fn fract_f64x8(self, a: f64x8) -> f64x8 { @@ -11105,8 +11133,10 @@ pub trait SimdFloat: fn floor(self) -> Self; #[doc = "Return the smallest integer greater than or equal to each element, that is, round towards positive infinity."] fn ceil(self) -> Self; - #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."] + #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] fn round_ties_even(self) -> Self; + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + fn round(self) -> Self; #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `self - self.trunc()`."] fn fract(self) -> Self; #[doc = "Return the integer part of each element, rounding towards zero."] diff --git a/fearless_simd/src/generated/simd_types.rs b/fearless_simd/src/generated/simd_types.rs index c8145641..0eb8a81c 100644 --- a/fearless_simd/src/generated/simd_types.rs +++ b/fearless_simd/src/generated/simd_types.rs @@ -321,6 +321,10 @@ impl crate::SimdFloat for f32x4 { self.simd.round_ties_even_f32x4(self) } #[inline(always)] + fn round(self) -> Self { + self.simd.round_f32x4(self) + } + #[inline(always)] fn fract(self) -> Self { self.simd.fract_f32x4(self) } @@ -3098,6 +3102,10 @@ impl crate::SimdFloat for f64x2 { self.simd.round_ties_even_f64x2(self) } #[inline(always)] + fn round(self) -> Self { + self.simd.round_f64x2(self) + } + #[inline(always)] fn fract(self) -> Self { self.simd.fract_f64x2(self) } @@ -4289,6 +4297,10 @@ impl crate::SimdFloat for f32x8 { self.simd.round_ties_even_f32x8(self) } #[inline(always)] + fn round(self) -> Self { + self.simd.round_f32x8(self) + } + #[inline(always)] fn fract(self) -> Self { self.simd.fract_f32x8(self) } @@ -7105,6 +7117,10 @@ impl crate::SimdFloat for f64x4 { self.simd.round_ties_even_f64x4(self) } #[inline(always)] + fn round(self) -> Self { + self.simd.round_f64x4(self) + } + #[inline(always)] fn fract(self) -> Self { self.simd.fract_f64x4(self) } @@ -8291,6 +8307,10 @@ impl crate::SimdFloat for f32x16 { self.simd.round_ties_even_f32x16(self) } #[inline(always)] + fn round(self) -> Self { + self.simd.round_f32x16(self) + } + #[inline(always)] fn fract(self) -> Self { self.simd.fract_f32x16(self) } @@ -11191,6 +11211,10 @@ impl crate::SimdFloat for f64x8 { self.simd.round_ties_even_f64x8(self) } #[inline(always)] + fn round(self) -> Self { + self.simd.round_f64x8(self) + } + #[inline(always)] fn fract(self) -> Self { self.simd.fract_f64x8(self) } diff --git a/fearless_simd/src/generated/sse2.rs b/fearless_simd/src/generated/sse2.rs index 1231fdb5..61ef8889 100644 --- a/fearless_simd/src/generated/sse2.rs +++ b/fearless_simd/src/generated/sse2.rs @@ -542,6 +542,11 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn round_f32x4(self, a: f32x4) -> f32x4 { + let bias = self.splat_f32x4(const { f32::next_down(0.5) }); + (a + self.copysign_f32x4(bias, a)).trunc() + } + #[inline(always)] fn fract_f32x4(self, a: f32x4) -> f32x4 { a - self.trunc_f32x4(a) } @@ -5673,6 +5678,11 @@ impl Simd for Sse2 { .simd_into(self) } #[inline(always)] + fn round_f64x2(self, a: f64x2) -> f64x2 { + let bias = self.splat_f64x2(const { f64::next_down(0.5) }); + (a + self.copysign_f64x2(bias, a)).trunc() + } + #[inline(always)] fn fract_f64x2(self, a: f64x2) -> f64x2 { a - self.trunc_f64x2(a) } diff --git a/fearless_simd/src/generated/sse4_2.rs b/fearless_simd/src/generated/sse4_2.rs index 159b686d..fd17dedd 100644 --- a/fearless_simd/src/generated/sse4_2.rs +++ b/fearless_simd/src/generated/sse4_2.rs @@ -523,6 +523,11 @@ impl Simd for Sse4_2 { kernel(self, a) } #[inline(always)] + fn round_f32x4(self, a: f32x4) -> f32x4 { + let bias = self.splat_f32x4(const { f32::next_down(0.5) }); + (a + self.copysign_f32x4(bias, a)).trunc() + } + #[inline(always)] fn fract_f32x4(self, a: f32x4) -> f32x4 { a - self.trunc_f32x4(a) } @@ -5181,6 +5186,11 @@ impl Simd for Sse4_2 { kernel(self, a) } #[inline(always)] + fn round_f64x2(self, a: f64x2) -> f64x2 { + let bias = self.splat_f64x2(const { f64::next_down(0.5) }); + (a + self.copysign_f64x2(bias, a)).trunc() + } + #[inline(always)] fn fract_f64x2(self, a: f64x2) -> f64x2 { a - self.trunc_f64x2(a) } diff --git a/fearless_simd/src/generated/wasm.rs b/fearless_simd/src/generated/wasm.rs index b0c49536..7dab37f5 100644 --- a/fearless_simd/src/generated/wasm.rs +++ b/fearless_simd/src/generated/wasm.rs @@ -343,6 +343,11 @@ impl Simd for WasmSimd128 { f32x4_nearest(a.into()).simd_into(self) } #[inline(always)] + fn round_f32x4(self, a: f32x4) -> f32x4 { + let bias = self.splat_f32x4(const { f32::next_down(0.5) }); + (a + self.copysign_f32x4(bias, a)).trunc() + } + #[inline(always)] fn fract_f32x4(self, a: f32x4) -> f32x4 { self.sub_f32x4(a, self.trunc_f32x4(a)) } @@ -3063,6 +3068,11 @@ impl Simd for WasmSimd128 { f64x2_nearest(a.into()).simd_into(self) } #[inline(always)] + fn round_f64x2(self, a: f64x2) -> f64x2 { + let bias = self.splat_f64x2(const { f64::next_down(0.5) }); + (a + self.copysign_f64x2(bias, a)).trunc() + } + #[inline(always)] fn fract_f64x2(self, a: f64x2) -> f64x2 { self.sub_f64x2(a, self.trunc_f64x2(a)) } diff --git a/fearless_simd/src/math.rs b/fearless_simd/src/math.rs index f3a053fa..f3b1da2a 100644 --- a/fearless_simd/src/math.rs +++ b/fearless_simd/src/math.rs @@ -35,6 +35,10 @@ impl FloatExt for f32 { libm::rintf(self) } #[inline(always)] + fn round(self) -> Self { + libm::roundf(self) + } + #[inline(always)] fn sqrt(self) -> Self { libm::sqrtf(self) } @@ -66,6 +70,10 @@ impl FloatExt for f64 { libm::rint(self) } #[inline(always)] + fn round(self) -> Self { + libm::round(self) + } + #[inline(always)] fn sqrt(self) -> Self { libm::sqrt(self) } diff --git a/fearless_simd_gen/src/arch/fallback.rs b/fearless_simd_gen/src/arch/fallback.rs index d06b0638..48ead357 100644 --- a/fearless_simd_gen/src/arch/fallback.rs +++ b/fearless_simd_gen/src/arch/fallback.rs @@ -25,6 +25,7 @@ pub(crate) fn translate_op(op: &str, is_float: bool) -> Option<&'static str> { "floor" => "floor", "ceil" => "ceil", "round_ties_even" => "round_ties_even", + "round" => "round", "fract" => "fract", "trunc" => "trunc", "sqrt" => "sqrt", diff --git a/fearless_simd_gen/src/arch/neon.rs b/fearless_simd_gen/src/arch/neon.rs index 3d04b1d7..0aee1333 100644 --- a/fearless_simd_gen/src/arch/neon.rs +++ b/fearless_simd_gen/src/arch/neon.rs @@ -12,6 +12,7 @@ fn translate_op(op: &str) -> Option<&'static str> { "floor" => "vrndm", "ceil" => "vrndp", "round_ties_even" => "vrndn", + "round" => "vrnda", "trunc" => "vrnd", "sqrt" => "vsqrt", "add" => "vadd", diff --git a/fearless_simd_gen/src/generic.rs b/fearless_simd_gen/src/generic.rs index 0df46a85..86602d49 100644 --- a/fearless_simd_gen/src/generic.rs +++ b/fearless_simd_gen/src/generic.rs @@ -779,3 +779,20 @@ pub(crate) fn generic_classify( } } } + +pub(crate) fn generic_round(method_sig: TokenStream, vec_ty: &VecType) -> TokenStream { + assert_eq!(vec_ty.scalar, ScalarType::Float); + + let scalar = vec_ty.scalar.rust(vec_ty.scalar_bits); + + let span = Span::call_site(); + let copysign = Ident::new(&format!("copysign_{}", vec_ty.rust_name()), span); + let splat = Ident::new(&format!("splat_{}", vec_ty.rust_name()), span); + + quote! { + #method_sig { + let bias = self.#splat(const{#scalar::next_down(0.5)}); + (a + self.#copysign(bias, a)).trunc() + } + } +} diff --git a/fearless_simd_gen/src/mk_wasm.rs b/fearless_simd_gen/src/mk_wasm.rs index d4bdbbc1..7604e8ce 100644 --- a/fearless_simd_gen/src/mk_wasm.rs +++ b/fearless_simd_gen/src/mk_wasm.rs @@ -7,7 +7,7 @@ use quote::{format_ident, quote}; use crate::arch::wasm::{arch_prefix, v128_intrinsic}; use crate::generic::{ concat_swizzle_dyn_precise_body, count_zeros_method, fallback_method, generic_block_combine, - generic_block_split, generic_classify, generic_mask_set, generic_op_name, + generic_block_split, generic_classify, generic_mask_set, generic_op_name, generic_round, integer_lane_mask_rotate, integer_lane_mask_splat_arg, recursive_swizzle_dyn_precise_body, reverse_method, reverse_vector_mask_method, }; @@ -508,6 +508,10 @@ impl Level for WasmSimd128 { return count_ones_method(op, vec_ty); } + if method == "round" { + return generic_round(method_sig, vec_ty); + } + let args = [quote! { a.into() }]; let expr = if matches!(method, "fract") { assert_eq!( diff --git a/fearless_simd_gen/src/mk_x86.rs b/fearless_simd_gen/src/mk_x86.rs index 180d9cb9..cd346285 100644 --- a/fearless_simd_gen/src/mk_x86.rs +++ b/fearless_simd_gen/src/mk_x86.rs @@ -9,7 +9,7 @@ use crate::arch::x86::{ use crate::generic::{ concat_swizzle_dyn_precise_body, count_zeros_method, fallback_method, generic_block_combine, generic_block_split, generic_classify, generic_mask_from_bitmask, generic_mask_set, - generic_op_name, integer_lane_mask_rotate, integer_lane_mask_splat_arg, + generic_op_name, generic_round, integer_lane_mask_rotate, integer_lane_mask_splat_arg, recursive_swizzle_dyn_precise_body, reverse_method, reverse_vector_mask_method, }; use crate::level::Level; @@ -2032,6 +2032,7 @@ impl X86 { } match method { + "round" => generic_round(method_sig, vec_ty), "fract" => { let trunc_op = generic_op_name("trunc", vec_ty); quote! { diff --git a/fearless_simd_gen/src/ops.rs b/fearless_simd_gen/src/ops.rs index cfb406e5..52c94003 100644 --- a/fearless_simd_gen/src/ops.rs +++ b/fearless_simd_gen/src/ops.rs @@ -1011,8 +1011,15 @@ const FLOAT_OPS: &[Op] = &[ "round_ties_even", OpKind::VecTraitMethod, OpSig::Unary, - "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\n\ - There is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is.", + "Round each element to the nearest integer, with ties rounding to the nearest even integer.", + ), + Op::new( + "round", + OpKind::VecTraitMethod, + OpSig::Unary, + "Round each element to the nearest integer, with ties rounding away from zero.\n\n\ + ## Performance considerations\n\n\ + Only `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible." ), Op::new( "fract", diff --git a/fearless_simd_tests/tests/harness/ops/mod.rs b/fearless_simd_tests/tests/harness/ops/mod.rs index a84d9881..09d19dd9 100644 --- a/fearless_simd_tests/tests/harness/ops/mod.rs +++ b/fearless_simd_tests/tests/harness/ops/mod.rs @@ -69,6 +69,7 @@ mod reduce_sum; mod reverse; mod rotate_elements_left; mod rotate_elements_right; +mod round; mod round_ties_even; mod saturating_add; mod saturating_sub; diff --git a/fearless_simd_tests/tests/harness/ops/round.rs b/fearless_simd_tests/tests/harness/ops/round.rs new file mode 100644 index 00000000..62764711 --- /dev/null +++ b/fearless_simd_tests/tests/harness/ops/round.rs @@ -0,0 +1,110 @@ +// Copyright 2026 the Fearless_SIMD Authors +// SPDX-License-Identifier: Apache-2.0 OR MIT + +use fearless_simd::*; +use fearless_simd_dev_macros::simd_test; + +// One concrete test row per supported vector type. + +#[simd_test] +fn round_f32x4(simd: S) { + let a = f32x4::from_slice(simd, &[2.3, -3.2, 2.7, -3.6]); + assert_eq!(*a.round(), [2.0, -3.0, 3.0, -4.0]); + let b = f32x4::from_slice(simd, &[-3.5, -2.5, 1.5, 0.5]); + assert_eq!(*b.round(), [-4.0, -3.0, 2.0, 1.0]); +} + +#[simd_test] +fn round_f64x2(simd: S) { + let a = f64x2::from_slice(simd, &[2.3, -3.2]); + assert_eq!(*a.round(), [2.0, -3.0]); + let b = f64x2::from_slice(simd, &[2.7, -3.6]); + assert_eq!(*b.round(), [3.0, -4.0]); + let c = f64x2::from_slice(simd, &[-3.5, -2.5]); + assert_eq!(*c.round(), [-4.0, -3.0]); + let d = f64x2::from_slice(simd, &[1.5, 0.5]); + assert_eq!(*d.round(), [2.0, 1.0]); +} + +#[simd_test] +fn round_f32x8(simd: S) { + let a = f32x8::from_slice(simd, &[2.3, -3.2, 2.7, -3.6, -3.5, -2.5, 1.5, 0.5]); + assert_eq!(*a.round(), [2.0, -3.0, 3.0, -4.0, -4.0, -3.0, 2.0, 1.0]); +} + +#[simd_test] +fn round_f32x16(simd: S) { + let a = f32x16::from_slice( + simd, + &[ + 2.3, -3.2, 2.7, -3.6, -3.5, -2.5, 1.5, 0.5, 2.3, -3.2, 2.7, -3.6, -3.5, -2.5, 1.5, 0.5, + ], + ); + assert_eq!( + *a.round(), + [ + 2.0, -3.0, 3.0, -4.0, -4.0, -3.0, 2.0, 1.0, 2.0, -3.0, 3.0, -4.0, -4.0, -3.0, 2.0, 1.0 + ] + ); +} + +#[simd_test] +fn round_f64x8(simd: S) { + let a = f64x8::from_slice(simd, &[2.3, -3.2, 2.7, -3.6, -3.5, -2.5, 1.5, 0.5]); + assert_eq!(*a.round(), [2.0, -3.0, 3.0, -4.0, -4.0, -3.0, 2.0, 1.0]); +} + +// Generated gap-fill coverage rows. + +#[simd_test] +fn round_f64x4(simd: S) { + let values: [f64; 4] = core::array::from_fn(|i| i as f64 - 3.5_f64); + let a = f64x4::from_slice(simd, &values); + let expected: [f64; 4] = core::array::from_fn(|i| values[i].round()); + let result = simd.round_f64x4(a); + assert_eq!(result.as_slice(), expected.as_slice()); +} + +#[simd_test] +fn round_f32x16_special_bit_patterns(simd: S) { + let values = [ + -0.0, + 0.0, + -0.5, + 0.5, + -1.5, + 1.5, + -2.5, + 2.5, + f32::from_bits(0x8000_0001), + f32::from_bits(0x0000_0001), + -((1_u32 << 23) as f32), + (1_u32 << 23) as f32, + f32::NEG_INFINITY, + f32::INFINITY, + -42.25, + 42.75, + ]; + let expected = values.map(|value| value.round().to_bits()); + let result = simd.round_f32x16(f32x16::from_slice(simd, &values)); + + assert_eq!((*result).map(f32::to_bits), expected); +} + +#[simd_test] +fn round_f64x8_special_bit_patterns(simd: S) { + let values = [ + -0.0, + 0.0, + -0.5, + 0.5, + -1.5, + 1.5, + f64::NEG_INFINITY, + f64::INFINITY, + ]; + let expected = values.map(|value| value.round().to_bits()); + let result = simd.round_f64x8(f64x8::from_slice(simd, &values)); + + assert_eq!((*result).map(f64::to_bits), expected); +} From 00ab9b1e684f186f666c0fc6b95a366c0c727380 Mon Sep 17 00:00:00 2001 From: RunDevelopment Date: Wed, 30 Sep 2026 13:28:19 +0200 Subject: [PATCH 2/3] Oops --- fearless_simd/src/math.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/fearless_simd/src/math.rs b/fearless_simd/src/math.rs index f3b1da2a..b0a569d5 100644 --- a/fearless_simd/src/math.rs +++ b/fearless_simd/src/math.rs @@ -15,6 +15,7 @@ pub(crate) trait FloatExt { fn floor(self) -> Self; fn ceil(self) -> Self; fn round_ties_even(self) -> Self; + fn round(self) -> Self; fn fract(self) -> Self; fn sqrt(self) -> Self; fn trunc(self) -> Self; From 812395b067e6d9fdfff9a53e6296421495958117 Mon Sep 17 00:00:00 2001 From: RunDevelopment Date: Wed, 30 Sep 2026 13:58:24 +0200 Subject: [PATCH 3/3] Clickable link --- fearless_simd/src/generated/simd_trait.rs | 14 +++++++------- fearless_simd_gen/src/ops.rs | 2 +- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/fearless_simd/src/generated/simd_trait.rs b/fearless_simd/src/generated/simd_trait.rs index f307c2d5..579f296b 100644 --- a/fearless_simd/src/generated/simd_trait.rs +++ b/fearless_simd/src/generated/simd_trait.rs @@ -372,7 +372,7 @@ pub trait Simd: fn ceil_f32x4(self, a: f32x4) -> f32x4; #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] fn round_ties_even_f32x4(self, a: f32x4) -> f32x4; - #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."] fn round_f32x4(self, a: f32x4) -> f32x4; #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."] fn fract_f32x4(self, a: f32x4) -> f32x4; @@ -1601,7 +1601,7 @@ pub trait Simd: fn ceil_f64x2(self, a: f64x2) -> f64x2; #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] fn round_ties_even_f64x2(self, a: f64x2) -> f64x2; - #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."] fn round_f64x2(self, a: f64x2) -> f64x2; #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."] fn fract_f64x2(self, a: f64x2) -> f64x2; @@ -2362,7 +2362,7 @@ pub trait Simd: self.round_ties_even_f32x4(a1), ) } - #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."] #[inline(always)] fn round_f32x8(self, a: f32x8) -> f32x8 { let (a0, a1) = self.split_f32x8(a); @@ -5293,7 +5293,7 @@ pub trait Simd: self.round_ties_even_f64x2(a1), ) } - #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."] #[inline(always)] fn round_f64x4(self, a: f64x4) -> f64x4 { let (a0, a1) = self.split_f64x4(a); @@ -6598,7 +6598,7 @@ pub trait Simd: self.round_ties_even_f32x8(a1), ) } - #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."] #[inline(always)] fn round_f32x16(self, a: f32x16) -> f32x16 { let (a0, a1) = self.split_f32x16(a); @@ -9532,7 +9532,7 @@ pub trait Simd: self.round_ties_even_f64x4(a1), ) } - #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."] #[inline(always)] fn round_f64x8(self, a: f64x8) -> f64x8 { let (a0, a1) = self.split_f64x8(a); @@ -11135,7 +11135,7 @@ pub trait SimdFloat: fn ceil(self) -> Self; #[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."] fn round_ties_even(self) -> Self; - #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible."] + #[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."] fn round(self) -> Self; #[doc = "Return the fractional part of each element.\n\nThis is equivalent to `self - self.trunc()`."] fn fract(self) -> Self; diff --git a/fearless_simd_gen/src/ops.rs b/fearless_simd_gen/src/ops.rs index 52c94003..bd134c9d 100644 --- a/fearless_simd_gen/src/ops.rs +++ b/fearless_simd_gen/src/ops.rs @@ -1019,7 +1019,7 @@ const FLOAT_OPS: &[Op] = &[ OpSig::Unary, "Round each element to the nearest integer, with ties rounding away from zero.\n\n\ ## Performance considerations\n\n\ - Only `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using `round_ties_even` if possible." + Only `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible." ), Op::new( "fract",