Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions fearless_simd/src/generated/avx2.rs
Original file line number Diff line number Diff line change
Expand Up @@ -531,6 +531,11 @@ impl Simd for Avx2 {
kernel(self, a)
}
#[inline(always)]
fn round_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
let bias = self.splat_f32x4(const { f32::next_down(0.5) });
(a + self.copysign_f32x4(bias, a)).trunc()
}
#[inline(always)]
fn fract_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
a - self.trunc_f32x4(a)
}
Expand Down Expand Up @@ -5207,6 +5212,11 @@ impl Simd for Avx2 {
kernel(self, a)
}
#[inline(always)]
fn round_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
let bias = self.splat_f64x2(const { f64::next_down(0.5) });
(a + self.copysign_f64x2(bias, a)).trunc()
}
#[inline(always)]
fn fract_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
a - self.trunc_f64x2(a)
}
Expand Down Expand Up @@ -7036,6 +7046,11 @@ impl Simd for Avx2 {
kernel(self, a)
}
#[inline(always)]
fn round_f32x8(self, a: f32x8<Self>) -> f32x8<Self> {
let bias = self.splat_f32x8(const { f32::next_down(0.5) });
(a + self.copysign_f32x8(bias, a)).trunc()
}
#[inline(always)]
fn fract_f32x8(self, a: f32x8<Self>) -> f32x8<Self> {
a - self.trunc_f32x8(a)
}
Expand Down Expand Up @@ -11373,6 +11388,11 @@ impl Simd for Avx2 {
kernel(self, a)
}
#[inline(always)]
fn round_f64x4(self, a: f64x4<Self>) -> f64x4<Self> {
let bias = self.splat_f64x4(const { f64::next_down(0.5) });
(a + self.copysign_f64x4(bias, a)).trunc()
}
#[inline(always)]
fn fract_f64x4(self, a: f64x4<Self>) -> f64x4<Self> {
a - self.trunc_f64x4(a)
}
Expand Down
30 changes: 30 additions & 0 deletions fearless_simd/src/generated/avx512.rs
Original file line number Diff line number Diff line change
Expand Up @@ -804,6 +804,11 @@ impl Simd for Avx512 {
kernel(self, a)
}
#[inline(always)]
fn round_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
let bias = self.splat_f32x4(const { f32::next_down(0.5) });
(a + self.copysign_f32x4(bias, a)).trunc()
}
#[inline(always)]
fn fract_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
a - self.trunc_f32x4(a)
}
Expand Down Expand Up @@ -5150,6 +5155,11 @@ impl Simd for Avx512 {
kernel(self, a)
}
#[inline(always)]
fn round_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
let bias = self.splat_f64x2(const { f64::next_down(0.5) });
(a + self.copysign_f64x2(bias, a)).trunc()
}
#[inline(always)]
fn fract_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
a - self.trunc_f64x2(a)
}
Expand Down Expand Up @@ -6896,6 +6906,11 @@ impl Simd for Avx512 {
kernel(self, a)
}
#[inline(always)]
fn round_f32x8(self, a: f32x8<Self>) -> f32x8<Self> {
let bias = self.splat_f32x8(const { f32::next_down(0.5) });
(a + self.copysign_f32x8(bias, a)).trunc()
}
#[inline(always)]
fn fract_f32x8(self, a: f32x8<Self>) -> f32x8<Self> {
a - self.trunc_f32x8(a)
}
Expand Down Expand Up @@ -11091,6 +11106,11 @@ impl Simd for Avx512 {
kernel(self, a)
}
#[inline(always)]
fn round_f64x4(self, a: f64x4<Self>) -> f64x4<Self> {
let bias = self.splat_f64x4(const { f64::next_down(0.5) });
(a + self.copysign_f64x4(bias, a)).trunc()
}
#[inline(always)]
fn fract_f64x4(self, a: f64x4<Self>) -> f64x4<Self> {
a - self.trunc_f64x4(a)
}
Expand Down Expand Up @@ -12835,6 +12855,11 @@ impl Simd for Avx512 {
kernel(self, a)
}
#[inline(always)]
fn round_f32x16(self, a: f32x16<Self>) -> f32x16<Self> {
let bias = self.splat_f32x16(const { f32::next_down(0.5) });
(a + self.copysign_f32x16(bias, a)).trunc()
}
#[inline(always)]
fn fract_f32x16(self, a: f32x16<Self>) -> f32x16<Self> {
a - self.trunc_f32x16(a)
}
Expand Down Expand Up @@ -17090,6 +17115,11 @@ impl Simd for Avx512 {
kernel(self, a)
}
#[inline(always)]
fn round_f64x8(self, a: f64x8<Self>) -> f64x8<Self> {
let bias = self.splat_f64x8(const { f64::next_down(0.5) });
(a + self.copysign_f64x8(bias, a)).trunc()
}
#[inline(always)]
fn fract_f64x8(self, a: f64x8<Self>) -> f64x8<Self> {
a - self.trunc_f64x8(a)
}
Expand Down
14 changes: 14 additions & 0 deletions fearless_simd/src/generated/fallback.rs
Original file line number Diff line number Diff line change
Expand Up @@ -403,6 +403,16 @@ impl Simd for Fallback {
.simd_into(self)
}
#[inline(always)]
fn round_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
[
f32::round(a[0usize]),
f32::round(a[1usize]),
f32::round(a[2usize]),
f32::round(a[3usize]),
]
.simd_into(self)
}
#[inline(always)]
fn fract_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
[
f32::fract(a[0usize]),
Expand Down Expand Up @@ -5703,6 +5713,10 @@ impl Simd for Fallback {
.simd_into(self)
}
#[inline(always)]
fn round_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
[f64::round(a[0usize]), f64::round(a[1usize])].simd_into(self)
}
#[inline(always)]
fn fract_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
[f64::fract(a[0usize]), f64::fract(a[1usize])].simd_into(self)
}
Expand Down
20 changes: 20 additions & 0 deletions fearless_simd/src/generated/neon.rs
Original file line number Diff line number Diff line change
Expand Up @@ -481,6 +481,16 @@ impl Simd for Neon {
kernel(self, a)
}
#[inline(always)]
fn round_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
crate::kernel!(
#[inline(always)]
fn kernel(token: Neon, a: f32x4<Neon>) -> f32x4<Neon> {
vrndaq_f32(a.into()).simd_into(token)
}
);
kernel(self, a)
}
#[inline(always)]
fn fract_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
crate::kernel!(
#[inline(always)]
Expand Down Expand Up @@ -4262,6 +4272,16 @@ impl Simd for Neon {
kernel(self, a)
}
#[inline(always)]
fn round_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
crate::kernel!(
#[inline(always)]
fn kernel(token: Neon, a: f64x2<Neon>) -> f64x2<Neon> {
vrndaq_f64(a.into()).simd_into(token)
}
);
kernel(self, a)
}
#[inline(always)]
fn fract_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
crate::kernel!(
#[inline(always)]
Expand Down
44 changes: 37 additions & 7 deletions fearless_simd/src/generated/simd_trait.rs
Original file line number Diff line number Diff line change
Expand Up @@ -370,8 +370,10 @@ pub trait Simd:
fn floor_f32x4(self, a: f32x4<Self>) -> f32x4<Self>;
#[doc = "Return the smallest integer greater than or equal to each element, that is, round towards positive infinity."]
fn ceil_f32x4(self, a: f32x4<Self>) -> f32x4<Self>;
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."]
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."]
fn round_ties_even_f32x4(self, a: f32x4<Self>) -> f32x4<Self>;
#[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."]
fn round_f32x4(self, a: f32x4<Self>) -> f32x4<Self>;
#[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."]
fn fract_f32x4(self, a: f32x4<Self>) -> f32x4<Self>;
#[doc = "Return the integer part of each element, rounding towards zero."]
Expand Down Expand Up @@ -1597,8 +1599,10 @@ pub trait Simd:
fn floor_f64x2(self, a: f64x2<Self>) -> f64x2<Self>;
#[doc = "Return the smallest integer greater than or equal to each element, that is, round towards positive infinity."]
fn ceil_f64x2(self, a: f64x2<Self>) -> f64x2<Self>;
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."]
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."]
fn round_ties_even_f64x2(self, a: f64x2<Self>) -> f64x2<Self>;
#[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."]
fn round_f64x2(self, a: f64x2<Self>) -> f64x2<Self>;
#[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."]
fn fract_f64x2(self, a: f64x2<Self>) -> f64x2<Self>;
#[doc = "Return the integer part of each element, rounding towards zero."]
Expand Down Expand Up @@ -2349,7 +2353,7 @@ pub trait Simd:
let (a0, a1) = self.split_f32x8(a);
self.combine_f32x4(self.ceil_f32x4(a0), self.ceil_f32x4(a1))
}
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."]
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."]
#[inline(always)]
fn round_ties_even_f32x8(self, a: f32x8<Self>) -> f32x8<Self> {
let (a0, a1) = self.split_f32x8(a);
Expand All @@ -2358,6 +2362,12 @@ pub trait Simd:
self.round_ties_even_f32x4(a1),
)
}
#[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."]
#[inline(always)]
fn round_f32x8(self, a: f32x8<Self>) -> f32x8<Self> {
let (a0, a1) = self.split_f32x8(a);
self.combine_f32x4(self.round_f32x4(a0), self.round_f32x4(a1))
}
#[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."]
#[inline(always)]
fn fract_f32x8(self, a: f32x8<Self>) -> f32x8<Self> {
Expand Down Expand Up @@ -5274,7 +5284,7 @@ pub trait Simd:
let (a0, a1) = self.split_f64x4(a);
self.combine_f64x2(self.ceil_f64x2(a0), self.ceil_f64x2(a1))
}
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."]
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."]
#[inline(always)]
fn round_ties_even_f64x4(self, a: f64x4<Self>) -> f64x4<Self> {
let (a0, a1) = self.split_f64x4(a);
Expand All @@ -5283,6 +5293,12 @@ pub trait Simd:
self.round_ties_even_f64x2(a1),
)
}
#[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."]
#[inline(always)]
fn round_f64x4(self, a: f64x4<Self>) -> f64x4<Self> {
let (a0, a1) = self.split_f64x4(a);
self.combine_f64x2(self.round_f64x2(a0), self.round_f64x2(a1))
}
#[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."]
#[inline(always)]
fn fract_f64x4(self, a: f64x4<Self>) -> f64x4<Self> {
Expand Down Expand Up @@ -6573,7 +6589,7 @@ pub trait Simd:
let (a0, a1) = self.split_f32x16(a);
self.combine_f32x8(self.ceil_f32x8(a0), self.ceil_f32x8(a1))
}
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."]
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."]
#[inline(always)]
fn round_ties_even_f32x16(self, a: f32x16<Self>) -> f32x16<Self> {
let (a0, a1) = self.split_f32x16(a);
Expand All @@ -6582,6 +6598,12 @@ pub trait Simd:
self.round_ties_even_f32x8(a1),
)
}
#[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."]
#[inline(always)]
fn round_f32x16(self, a: f32x16<Self>) -> f32x16<Self> {
let (a0, a1) = self.split_f32x16(a);
self.combine_f32x8(self.round_f32x8(a0), self.round_f32x8(a1))
}
#[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."]
#[inline(always)]
fn fract_f32x16(self, a: f32x16<Self>) -> f32x16<Self> {
Expand Down Expand Up @@ -9501,7 +9523,7 @@ pub trait Simd:
let (a0, a1) = self.split_f64x8(a);
self.combine_f64x4(self.ceil_f64x4(a0), self.ceil_f64x4(a1))
}
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."]
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."]
#[inline(always)]
fn round_ties_even_f64x8(self, a: f64x8<Self>) -> f64x8<Self> {
let (a0, a1) = self.split_f64x8(a);
Expand All @@ -9510,6 +9532,12 @@ pub trait Simd:
self.round_ties_even_f64x4(a1),
)
}
#[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."]
#[inline(always)]
fn round_f64x8(self, a: f64x8<Self>) -> f64x8<Self> {
let (a0, a1) = self.split_f64x8(a);
self.combine_f64x4(self.round_f64x4(a0), self.round_f64x4(a1))
}
#[doc = "Return the fractional part of each element.\n\nThis is equivalent to `a - a.trunc()`."]
#[inline(always)]
fn fract_f64x8(self, a: f64x8<Self>) -> f64x8<Self> {
Expand Down Expand Up @@ -11105,8 +11133,10 @@ pub trait SimdFloat<S: Simd>:
fn floor(self) -> Self;
#[doc = "Return the smallest integer greater than or equal to each element, that is, round towards positive infinity."]
fn ceil(self) -> Self;
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer.\n\nThere is no corresponding `round` operation. Rust's `round` operation rounds ties away from zero, a behavior it inherited from C. That behavior is not implemented across all platforms, whereas round-ties-even is."]
#[doc = "Round each element to the nearest integer, with ties rounding to the nearest even integer."]
fn round_ties_even(self) -> Self;
#[doc = "Round each element to the nearest integer, with ties rounding away from zero.\n\n## Performance considerations\n\nOnly `AArch64` has native instructions for this operation. `round` has to be emulated on all other platforms which is around 2-4x slower than `round_ties_even`. Prefer using [`round_ties_even`](SimdFloat::round_ties_even) if possible."]
fn round(self) -> Self;
#[doc = "Return the fractional part of each element.\n\nThis is equivalent to `self - self.trunc()`."]
fn fract(self) -> Self;
#[doc = "Return the integer part of each element, rounding towards zero."]
Expand Down
24 changes: 24 additions & 0 deletions fearless_simd/src/generated/simd_types.rs
Original file line number Diff line number Diff line change
Expand Up @@ -321,6 +321,10 @@ impl<S: Simd> crate::SimdFloat<S> for f32x4<S> {
self.simd.round_ties_even_f32x4(self)
}
#[inline(always)]
fn round(self) -> Self {
self.simd.round_f32x4(self)
}
#[inline(always)]
fn fract(self) -> Self {
self.simd.fract_f32x4(self)
}
Expand Down Expand Up @@ -3098,6 +3102,10 @@ impl<S: Simd> crate::SimdFloat<S> for f64x2<S> {
self.simd.round_ties_even_f64x2(self)
}
#[inline(always)]
fn round(self) -> Self {
self.simd.round_f64x2(self)
}
#[inline(always)]
fn fract(self) -> Self {
self.simd.fract_f64x2(self)
}
Expand Down Expand Up @@ -4289,6 +4297,10 @@ impl<S: Simd> crate::SimdFloat<S> for f32x8<S> {
self.simd.round_ties_even_f32x8(self)
}
#[inline(always)]
fn round(self) -> Self {
self.simd.round_f32x8(self)
}
#[inline(always)]
fn fract(self) -> Self {
self.simd.fract_f32x8(self)
}
Expand Down Expand Up @@ -7105,6 +7117,10 @@ impl<S: Simd> crate::SimdFloat<S> for f64x4<S> {
self.simd.round_ties_even_f64x4(self)
}
#[inline(always)]
fn round(self) -> Self {
self.simd.round_f64x4(self)
}
#[inline(always)]
fn fract(self) -> Self {
self.simd.fract_f64x4(self)
}
Expand Down Expand Up @@ -8291,6 +8307,10 @@ impl<S: Simd> crate::SimdFloat<S> for f32x16<S> {
self.simd.round_ties_even_f32x16(self)
}
#[inline(always)]
fn round(self) -> Self {
self.simd.round_f32x16(self)
}
#[inline(always)]
fn fract(self) -> Self {
self.simd.fract_f32x16(self)
}
Expand Down Expand Up @@ -11191,6 +11211,10 @@ impl<S: Simd> crate::SimdFloat<S> for f64x8<S> {
self.simd.round_ties_even_f64x8(self)
}
#[inline(always)]
fn round(self) -> Self {
self.simd.round_f64x8(self)
}
#[inline(always)]
fn fract(self) -> Self {
self.simd.fract_f64x8(self)
}
Expand Down
10 changes: 10 additions & 0 deletions fearless_simd/src/generated/sse2.rs
Original file line number Diff line number Diff line change
Expand Up @@ -542,6 +542,11 @@ impl Simd for Sse2 {
.simd_into(self)
}
#[inline(always)]
fn round_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
let bias = self.splat_f32x4(const { f32::next_down(0.5) });
(a + self.copysign_f32x4(bias, a)).trunc()
}
#[inline(always)]
fn fract_f32x4(self, a: f32x4<Self>) -> f32x4<Self> {
a - self.trunc_f32x4(a)
}
Expand Down Expand Up @@ -5673,6 +5678,11 @@ impl Simd for Sse2 {
.simd_into(self)
}
#[inline(always)]
fn round_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
let bias = self.splat_f64x2(const { f64::next_down(0.5) });
(a + self.copysign_f64x2(bias, a)).trunc()
}
#[inline(always)]
fn fract_f64x2(self, a: f64x2<Self>) -> f64x2<Self> {
a - self.trunc_f64x2(a)
}
Expand Down
Loading
Loading