Change negate_D, negate_D_lazy to impl Neg, negate_lazy

This commit is contained in:
Henry de Valence 2018-06-13 12:32:42 -07:00
parent 14ce6d3da6
commit 00d8b6ea4f
2 changed files with 40 additions and 70 deletions

View file

@ -176,7 +176,7 @@ impl From<ExtendedPoint> for CachedPoint {
x.scale_by_curve_constants(); x.scale_by_curve_constants();
// x = (121666*S2 121666*S3 2*121666*Z2 -2*121665*T2) // x = (121666*S2 121666*S3 2*121666*Z2 -2*121665*T2)
x.negate_D(); x = x.blend(-x, Lanes::D);
CachedPoint(x) CachedPoint(x)
} }
@ -210,10 +210,9 @@ impl<'a> Neg for &'a CachedPoint {
type Output = CachedPoint; type Output = CachedPoint;
fn neg(self) -> CachedPoint { fn neg(self) -> CachedPoint {
let mut neg = *self; let mut coords = self.0;
neg.0.swap_AB(); coords.swap_AB();
neg.0.negate_D_lazy(); CachedPoint(coords.blend(coords.negate_lazy(), Lanes::D))
neg
} }
} }

View file

@ -24,7 +24,7 @@ pub const D_LANES64: u8 = 0b11_00_00_00;
pub const ALL_LANES: u8 = A_LANES | B_LANES | C_LANES | D_LANES; pub const ALL_LANES: u8 = A_LANES | B_LANES | C_LANES | D_LANES;
use core::ops::{Add, Mul}; use core::ops::{Add, Mul, Neg};
use core::simd::{i32x8, u32x8, u64x4, IntoBits}; use core::simd::{i32x8, u32x8, u64x4, IntoBits};
use backend::avx2::constants::{P_TIMES_16_HI, P_TIMES_16_LO, P_TIMES_2_HI, P_TIMES_2_LO}; use backend::avx2::constants::{P_TIMES_16_HI, P_TIMES_16_LO, P_TIMES_2_HI, P_TIMES_2_LO};
@ -195,73 +195,20 @@ impl FieldElement32x4 {
return out; return out;
} }
/// Negate the \\(D\\) variable of \\((A,B,C,D)\\). /// Given \\((A,B,C,D)\\), compute \\((-A,-B,-C,-D)\\), without
/// performing a reduction.
/// ///
/// Input limbs must be less than the limbs of \\(2p\\), i.e., freshly reduced. /// Input limbs must be less than the limbs of \\(2p\\), i.e., freshly reduced.
pub fn negate_D_lazy(&mut self) {
unsafe {
use core::arch::x86_64::_mm256_blend_epi32;
self.0[0] = _mm256_blend_epi32(
self.0[0].into_bits(),
(P_TIMES_2_LO - self.0[0]).into_bits(),
D_LANES as i32,
).into_bits();
self.0[1] = _mm256_blend_epi32(
self.0[1].into_bits(),
(P_TIMES_2_HI - self.0[1]).into_bits(),
D_LANES as i32,
).into_bits();
self.0[2] = _mm256_blend_epi32(
self.0[2].into_bits(),
(P_TIMES_2_HI - self.0[2]).into_bits(),
D_LANES as i32,
).into_bits();
self.0[3] = _mm256_blend_epi32(
self.0[3].into_bits(),
(P_TIMES_2_HI - self.0[3]).into_bits(),
D_LANES as i32,
).into_bits();
self.0[4] = _mm256_blend_epi32(
self.0[4].into_bits(),
(P_TIMES_2_HI - self.0[4]).into_bits(),
D_LANES as i32,
).into_bits();
}
}
/// Negate the \\(D\\) variable of \\((A,B,C,D)\\).
/// ///
/// Input limbs must be less than the limbs of \\(2p\\), i.e., freshly reduced. /// The output limbs are bounded by \\(2p\\).
pub fn negate_D(&mut self) { pub fn negate_lazy(&self) -> FieldElement32x4 {
unsafe { FieldElement32x4([
use core::arch::x86_64::_mm256_blend_epi32; P_TIMES_2_LO - self.0[0],
self.0[0] = _mm256_blend_epi32( P_TIMES_2_HI - self.0[1],
self.0[0].into_bits(), P_TIMES_2_HI - self.0[2],
(P_TIMES_16_LO - self.0[0]).into_bits(), P_TIMES_2_HI - self.0[3],
D_LANES as i32, P_TIMES_2_HI - self.0[4],
).into_bits(); ])
self.0[1] = _mm256_blend_epi32(
self.0[1].into_bits(),
(P_TIMES_16_HI - self.0[1]).into_bits(),
D_LANES as i32,
).into_bits();
self.0[2] = _mm256_blend_epi32(
self.0[2].into_bits(),
(P_TIMES_16_HI - self.0[2]).into_bits(),
D_LANES as i32,
).into_bits();
self.0[3] = _mm256_blend_epi32(
self.0[3].into_bits(),
(P_TIMES_16_HI - self.0[3]).into_bits(),
D_LANES as i32,
).into_bits();
self.0[4] = _mm256_blend_epi32(
self.0[4].into_bits(),
(P_TIMES_16_HI - self.0[4]).into_bits(),
D_LANES as i32,
).into_bits();
}
self.reduce32();
} }
/// Given `self = (A,B,C,D)`, set `self = (B,A,C,D)` /// Given `self = (A,B,C,D)`, set `self = (B,A,C,D)`
@ -639,6 +586,30 @@ impl FieldElement32x4 {
} }
} }
impl Neg for FieldElement32x4 {
type Output = FieldElement32x4;
/// Given \\((A,B,C,D)\\), compute \\((-A,-B,-C,-D)\\), and
/// perform a reduction.
///
/// The input limbs can be any size.
///
/// The output limbs are freshly reduced.
#[inline]
fn neg(self) -> FieldElement32x4 {
let mut neg = FieldElement32x4([
P_TIMES_16_LO - self.0[0],
P_TIMES_16_HI - self.0[1],
P_TIMES_16_HI - self.0[2],
P_TIMES_16_HI - self.0[3],
P_TIMES_16_HI - self.0[4],
]);
neg.reduce32();
neg
}
}
impl Add<FieldElement32x4> for FieldElement32x4 { impl Add<FieldElement32x4> for FieldElement32x4 {
type Output = FieldElement32x4; type Output = FieldElement32x4;
#[inline] #[inline]