Skip to main content

wide/
u64x2_.rs

1use super::*;
2
3pick! {
4  if #[cfg(target_feature="sse2")] {
5    #[derive(Default, Clone, Copy, PartialEq, Eq)]
6    #[repr(C, align(16))]
7    pub struct u64x2 { pub(crate) sse: m128i }
8  } else if #[cfg(target_feature="simd128")] {
9    use core::arch::wasm32::*;
10
11    #[derive(Clone, Copy)]
12    #[repr(transparent)]
13    pub struct u64x2 { pub(crate) simd: v128 }
14
15    impl Default for u64x2 {
16      fn default() -> Self {
17        Self::splat(0)
18      }
19    }
20
21    impl PartialEq for u64x2 {
22      fn eq(&self, other: &Self) -> bool {
23        u64x2_all_true(u64x2_eq(self.simd, other.simd))
24      }
25    }
26
27    impl Eq for u64x2 { }
28  } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
29    use core::arch::aarch64::*;
30    #[repr(C)]
31    #[derive(Copy, Clone)]
32    pub struct u64x2 { pub(crate) neon : uint64x2_t }
33
34    impl Default for u64x2 {
35      #[inline]
36      fn default() -> Self {
37        unsafe { Self { neon: vdupq_n_u64(0)} }
38      }
39    }
40
41    impl PartialEq for u64x2 {
42      #[inline]
43      fn eq(&self, other: &Self) -> bool {
44        unsafe {
45          vgetq_lane_u64(self.neon,0) == vgetq_lane_u64(other.neon,0) &&
46          vgetq_lane_u64(self.neon,1) == vgetq_lane_u64(other.neon,1)
47        }
48      }
49    }
50
51    impl Eq for u64x2 { }
52  } else {
53    #[derive(Default, Clone, Copy, PartialEq, Eq)]
54    #[repr(C, align(16))]
55    pub struct u64x2 { arr: [u64;2] }
56  }
57}
58
59int_uint_consts!(u64, 2, u64x2, 128);
60
61unsafe impl Zeroable for u64x2 {}
62unsafe impl Pod for u64x2 {}
63
64impl AlignTo for u64x2 {
65  type Elem = u64;
66}
67
68impl Add for u64x2 {
69  type Output = Self;
70  #[inline]
71  fn add(self, rhs: Self) -> Self::Output {
72    pick! {
73      if #[cfg(target_feature="sse2")] {
74        Self { sse: add_i64_m128i(self.sse, rhs.sse) }
75      } else if #[cfg(target_feature="simd128")] {
76        Self { simd: u64x2_add(self.simd, rhs.simd) }
77      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
78        unsafe { Self { neon: vaddq_u64(self.neon, rhs.neon) } }
79      } else {
80        Self { arr: [
81          self.arr[0].wrapping_add(rhs.arr[0]),
82          self.arr[1].wrapping_add(rhs.arr[1]),
83        ]}
84      }
85    }
86  }
87}
88
89impl Sub for u64x2 {
90  type Output = Self;
91  #[inline]
92  fn sub(self, rhs: Self) -> Self::Output {
93    pick! {
94      if #[cfg(target_feature="sse2")] {
95        Self { sse: sub_i64_m128i(self.sse, rhs.sse) }
96      } else if #[cfg(target_feature="simd128")] {
97        Self { simd: u64x2_sub(self.simd, rhs.simd) }
98      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
99        unsafe { Self { neon: vsubq_u64(self.neon, rhs.neon) } }
100      } else {
101        Self { arr: [
102          self.arr[0].wrapping_sub(rhs.arr[0]),
103          self.arr[1].wrapping_sub(rhs.arr[1]),
104        ]}
105      }
106    }
107  }
108}
109
110//we should try to implement this on sse2
111impl Mul for u64x2 {
112  type Output = Self;
113  #[inline]
114  fn mul(self, rhs: Self) -> Self::Output {
115    pick! {
116      if #[cfg(target_feature="simd128")] {
117        Self { simd: u64x2_mul(self.simd, rhs.simd) }
118      } else {
119        let arr1: [u64; 2] = cast(self);
120        let arr2: [u64; 2] = cast(rhs);
121        cast([
122          arr1[0].wrapping_mul(arr2[0]),
123          arr1[1].wrapping_mul(arr2[1]),
124        ])
125      }
126    }
127  }
128}
129
130integer_impl_div_rem!(u64, u64x2, [0, 1]);
131
132impl Add<u64> for u64x2 {
133  type Output = Self;
134  #[inline]
135  fn add(self, rhs: u64) -> Self::Output {
136    self.add(Self::splat(rhs))
137  }
138}
139
140impl Sub<u64> for u64x2 {
141  type Output = Self;
142  #[inline]
143  fn sub(self, rhs: u64) -> Self::Output {
144    self.sub(Self::splat(rhs))
145  }
146}
147
148impl Mul<u64> for u64x2 {
149  type Output = Self;
150  #[inline]
151  fn mul(self, rhs: u64) -> Self::Output {
152    self.mul(Self::splat(rhs))
153  }
154}
155
156impl Add<u64x2> for u64 {
157  type Output = u64x2;
158  #[inline]
159  fn add(self, rhs: u64x2) -> Self::Output {
160    u64x2::splat(self).add(rhs)
161  }
162}
163
164impl Sub<u64x2> for u64 {
165  type Output = u64x2;
166  #[inline]
167  fn sub(self, rhs: u64x2) -> Self::Output {
168    u64x2::splat(self).sub(rhs)
169  }
170}
171
172impl Mul<u64x2> for u64 {
173  type Output = u64x2;
174  #[inline]
175  fn mul(self, rhs: u64x2) -> Self::Output {
176    u64x2::splat(self).mul(rhs)
177  }
178}
179
180impl BitAnd for u64x2 {
181  type Output = Self;
182  #[inline]
183  fn bitand(self, rhs: Self) -> Self::Output {
184    pick! {
185      if #[cfg(target_feature="sse2")] {
186        Self { sse: bitand_m128i(self.sse, rhs.sse) }
187      } else if #[cfg(target_feature="simd128")] {
188        Self { simd: v128_and(self.simd, rhs.simd) }
189      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
190        unsafe {Self { neon: vandq_u64(self.neon, rhs.neon) }}
191      } else {
192        Self { arr: [
193          self.arr[0].bitand(rhs.arr[0]),
194          self.arr[1].bitand(rhs.arr[1]),
195        ]}
196      }
197    }
198  }
199}
200
201impl BitOr for u64x2 {
202  type Output = Self;
203  #[inline]
204  fn bitor(self, rhs: Self) -> Self::Output {
205    pick! {
206      if #[cfg(target_feature="sse2")] {
207        Self { sse: bitor_m128i(self.sse, rhs.sse) }
208      } else if #[cfg(target_feature="simd128")] {
209        Self { simd: v128_or(self.simd, rhs.simd) }
210      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
211        unsafe {Self { neon: vorrq_u64(self.neon, rhs.neon) }}
212      } else {
213        Self { arr: [
214          self.arr[0].bitor(rhs.arr[0]),
215          self.arr[1].bitor(rhs.arr[1]),
216        ]}
217      }
218    }
219  }
220}
221
222impl BitXor for u64x2 {
223  type Output = Self;
224  #[inline]
225  fn bitxor(self, rhs: Self) -> Self::Output {
226    pick! {
227      if #[cfg(target_feature="sse2")] {
228        Self { sse: bitxor_m128i(self.sse, rhs.sse) }
229      } else if #[cfg(target_feature="simd128")] {
230        Self { simd: v128_xor(self.simd, rhs.simd) }
231      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
232        unsafe {Self { neon: veorq_u64(self.neon, rhs.neon) }}
233      } else {
234        Self { arr: [
235          self.arr[0].bitxor(rhs.arr[0]),
236          self.arr[1].bitxor(rhs.arr[1]),
237        ]}
238      }
239    }
240  }
241}
242
243/// Shifts lanes by the corresponding lane.
244///
245/// Bitwise shift-left; yields `self << mask(rhs)`, where mask removes any
246/// high-order bits of `rhs` that would cause the shift to exceed the bitwidth
247/// of the type. (same as `wrapping_shl`)
248impl Shl for u64x2 {
249  type Output = Self;
250
251  #[inline]
252  fn shl(self, rhs: Self) -> Self::Output {
253    pick! {
254      if #[cfg(target_feature="avx2")] {
255        // mask the shift count to 63 to have same behavior on all platforms
256        let shift_by = rhs & Self::splat(63);
257        Self { sse: shl_each_u64_m128i(self.sse, shift_by.sse) }
258      } else if #[cfg(all(target_feature="neon", target_arch="aarch64"))] {
259        unsafe {
260          // mask the shift count to 63 to have same behavior on all platforms
261          let shift_by = vreinterpretq_s64_u64(vandq_u64(rhs.neon, vmovq_n_u64(63)));
262          Self { neon: vshlq_u64(self.neon, shift_by) }
263        }
264      } else {
265        let arr: [u64; 2] = cast(self);
266        let rhs: [u64; 2] = cast(rhs);
267        cast([
268          arr[0].wrapping_shl(rhs[0] as u32),
269          arr[1].wrapping_shl(rhs[1] as u32),
270        ])
271      }
272    }
273  }
274}
275
276macro_rules! impl_shl_t_for_u64x2 {
277  ($($shift_type:ty),+ $(,)?) => {
278    $(impl Shl<$shift_type> for u64x2 {
279      type Output = Self;
280      /// Shifts all lanes by the value given.
281      #[inline]
282      fn shl(self, rhs: $shift_type) -> Self::Output {
283        pick! {
284          if #[cfg(target_feature="sse2")] {
285            let shift = cast([rhs as u64, 0]);
286            Self { sse: shl_all_u64_m128i(self.sse, shift) }
287          } else if #[cfg(target_feature="simd128")] {
288            Self { simd: u64x2_shl(self.simd, rhs as u32) }
289          } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
290            unsafe {Self { neon: vshlq_u64(self.neon, vmovq_n_s64(rhs as i64)) }}
291          } else {
292            let u = rhs as u32;
293            Self { arr: [
294              self.arr[0].wrapping_shl(u),
295              self.arr[1].wrapping_shl(u),
296            ]}
297          }
298        }
299      }
300    })+
301  };
302}
303impl_shl_t_for_u64x2!(i8, u8, i16, u16, i32, u32, i64, u64, i128, u128);
304
305/// Shifts lanes by the corresponding lane.
306///
307/// Bitwise shift-right; yields `self >> mask(rhs)`, where mask removes any
308/// high-order bits of `rhs` that would cause the shift to exceed the bitwidth
309/// of the type. (same as `wrapping_shr`)
310impl Shr for u64x2 {
311  type Output = Self;
312
313  #[inline]
314  fn shr(self, rhs: Self) -> Self::Output {
315    pick! {
316      if #[cfg(target_feature="avx2")] {
317        // mask the shift count to 63 to have same behavior on all platforms
318        let shift_by = rhs & Self::splat(63);
319        Self { sse: shr_each_u64_m128i(self.sse, shift_by.sse) }
320      } else if #[cfg(all(target_feature="neon", target_arch="aarch64"))] {
321        unsafe {
322          // mask the shift count to 63 to have same behavior on all platforms
323          // no right shift, have to pass negative value to left shift on neon
324          let shift_by = vnegq_s64(vreinterpretq_s64_u64(vandq_u64(rhs.neon, vmovq_n_u64(63))));
325          Self { neon: vshlq_u64(self.neon, shift_by) }
326        }
327      } else {
328        let arr: [u64; 2] = cast(self);
329        let rhs: [u64; 2] = cast(rhs);
330        cast([
331          arr[0].wrapping_shr(rhs[0] as u32),
332          arr[1].wrapping_shr(rhs[1] as u32),
333        ])
334      }
335    }
336  }
337}
338
339macro_rules! impl_shr_t_for_u64x2 {
340  ($($shift_type:ty),+ $(,)?) => {
341    $(impl Shr<$shift_type> for u64x2 {
342      type Output = Self;
343      /// Shifts all lanes by the value given.
344      #[inline]
345      fn shr(self, rhs: $shift_type) -> Self::Output {
346        pick! {
347          if #[cfg(target_feature="sse2")] {
348            let shift = cast([rhs as u64, 0]);
349            Self { sse: shr_all_u64_m128i(self.sse, shift) }
350          } else if #[cfg(target_feature="simd128")] {
351            Self { simd: u64x2_shr(self.simd, rhs as u32) }
352          } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
353            unsafe {Self { neon: vshlq_u64(self.neon, vmovq_n_s64(-(rhs as i64))) }}
354          } else {
355            let u = rhs as u32;
356            Self { arr: [
357              self.arr[0].wrapping_shr(u),
358              self.arr[1].wrapping_shr(u),
359            ]}
360          }
361        }
362      }
363    })+
364  };
365}
366impl_shr_t_for_u64x2!(i8, u8, i16, u16, i32, u32, i64, u64, i128, u128);
367
368#[expect(deprecated)]
369impl CmpEq for u64x2 {
370  type Output = Self;
371  #[inline]
372  fn simd_eq(self, rhs: Self) -> Self::Output {
373    pick! {
374      if #[cfg(target_feature="sse4.1")] {
375        Self { sse: cmp_eq_mask_i64_m128i(self.sse, rhs.sse) }
376      } else if #[cfg(target_feature="simd128")] {
377        Self { simd: u64x2_eq(self.simd, rhs.simd) }
378      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
379        unsafe {Self { neon: vceqq_u64(self.neon, rhs.neon) } }
380      } else {
381        let s: [u64;2] = cast(self);
382        let r: [u64;2] = cast(rhs);
383        cast([
384          if s[0] == r[0] { -1_i64 } else { 0 },
385          if s[1] == r[1] { -1_i64 } else { 0 },
386        ])
387      }
388    }
389  }
390}
391
392#[expect(deprecated)]
393impl CmpGt for u64x2 {
394  type Output = Self;
395  #[inline]
396  fn simd_gt(self, rhs: Self) -> Self::Output {
397    pick! {
398      if #[cfg(target_feature="sse4.2")] {
399        // no unsigned gt so inverting the high bit will get the correct result
400        let highbit = u64x2::splat(1 << 63);
401        Self { sse: cmp_gt_mask_i64_m128i((self ^ highbit).sse, (rhs ^ highbit).sse) }
402      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
403        unsafe {Self { neon: vcgtq_u64(self.neon, rhs.neon) }}
404      } else {
405        // u64x2_gt on WASM is not a thing. https://github.com/WebAssembly/simd/pull/414
406        let s: [u64;2] = cast(self);
407        let r: [u64;2] = cast(rhs);
408        cast([
409          if s[0] > r[0] { u64::MAX } else { 0 },
410          if s[1] > r[1] { u64::MAX } else { 0 },
411        ])
412      }
413    }
414  }
415}
416
417#[expect(deprecated)]
418impl CmpLt for u64x2 {
419  type Output = Self;
420  #[inline]
421  fn simd_lt(self, rhs: Self) -> Self::Output {
422    // lt is just gt the other way around
423    rhs.simd_gt(self)
424  }
425}
426
427#[expect(deprecated)]
428impl CmpNe for u64x2 {
429  type Output = Self;
430  #[inline]
431  fn simd_ne(self, rhs: Self) -> Self::Output {
432    pick! {
433      if #[cfg(target_feature="sse4.1")] {
434        !self.simd_eq(rhs)
435      } else if #[cfg(target_feature="simd128")] {
436        Self { simd: u64x2_ne(self.simd, rhs.simd) }
437      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
438        !self.simd_eq(rhs)
439      } else {
440        let s: [u64;2] = cast(self);
441        let r: [u64;2] = cast(rhs);
442        cast([
443          if s[0] != r[0] { -1_i64 } else { 0 },
444          if s[1] != r[1] { -1_i64 } else { 0 },
445        ])
446      }
447    }
448  }
449}
450
451#[expect(deprecated)]
452impl CmpLe for u64x2 {
453  type Output = Self;
454  #[inline]
455  fn simd_le(self, rhs: Self) -> Self::Output {
456    pick! {
457      if #[cfg(target_feature="sse4.1")] {
458        !self.simd_gt(rhs)
459      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
460        !self.simd_gt(rhs)
461      } else {
462        let s: [u64;2] = cast(self);
463        let r: [u64;2] = cast(rhs);
464        cast([
465          if s[0] <= r[0] { -1_i64 } else { 0 },
466          if s[1] <= r[1] { -1_i64 } else { 0 },
467        ])
468      }
469    }
470  }
471}
472
473#[expect(deprecated)]
474impl CmpGe for u64x2 {
475  type Output = Self;
476  #[inline]
477  fn simd_ge(self, rhs: Self) -> Self::Output {
478    pick! {
479      if #[cfg(target_feature="sse4.1")] {
480        !self.simd_lt(rhs)
481      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
482        !self.simd_lt(rhs)
483      } else {
484        let s: [u64;2] = cast(self);
485        let r: [u64;2] = cast(rhs);
486        cast([
487          if s[0] >= r[0] { -1_i64 } else { 0 },
488          if s[1] >= r[1] { -1_i64 } else { 0 },
489        ])
490      }
491    }
492  }
493}
494
495impl u64x2 {
496  #[inline]
497  #[must_use]
498  pub const fn new(array: [u64; 2]) -> Self {
499    unsafe { core::mem::transmute(array) }
500  }
501
502  simd_comparison_fns!();
503
504  #[inline]
505  #[must_use]
506  pub fn blend(self, t: Self, f: Self) -> Self {
507    pick! {
508      if #[cfg(target_feature="sse4.1")] {
509        Self { sse: blend_varying_i8_m128i(f.sse, t.sse, self.sse) }
510      } else if #[cfg(target_feature="simd128")] {
511        Self { simd: v128_bitselect(t.simd, f.simd, self.simd) }
512      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
513        unsafe {Self { neon: vbslq_u64(self.neon, t.neon, f.neon) }}
514      } else {
515        generic_bit_blend(self, t, f)
516      }
517    }
518  }
519
520  #[inline]
521  #[must_use]
522  pub fn reduce_add(self) -> u64 {
523    cast(i64x2::reduce_add(cast(self)))
524  }
525
526  #[inline]
527  #[must_use]
528  pub fn reduce_max(self) -> u64 {
529    pick! {
530      if #[cfg(any(target_feature="sse2", target_feature="simd128"))] {
531        let array: [u64; 2] = cast(self);
532        array[0].max(array[1])
533      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
534        unsafe { vgetq_lane_u64(self.neon, 0).max(vgetq_lane_u64(self.neon, 1)) }
535      } else {
536        self.arr[0].max(self.arr[1])
537      }
538    }
539  }
540
541  #[inline]
542  #[must_use]
543  pub fn reduce_min(self) -> u64 {
544    pick! {
545      if #[cfg(any(target_feature="sse2", target_feature="simd128"))] {
546        let array: [u64; 2] = cast(self);
547        array[0].min(array[1])
548      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
549        unsafe { vgetq_lane_u64(self.neon, 0).min(vgetq_lane_u64(self.neon, 1)) }
550      } else {
551        self.arr[0].min(self.arr[1])
552      }
553    }
554  }
555
556  #[inline]
557  #[must_use]
558  #[doc(alias("movemask", "move_mask"))]
559  pub fn to_bitmask(self) -> u32 {
560    i64x2::to_bitmask(cast(self))
561  }
562
563  #[inline]
564  #[must_use]
565  pub fn any(self) -> bool {
566    i64x2::any(cast(self))
567  }
568
569  #[inline]
570  #[must_use]
571  pub fn all(self) -> bool {
572    i64x2::all(cast(self))
573  }
574
575  #[inline]
576  #[must_use]
577  pub fn none(self) -> bool {
578    !self.any()
579  }
580
581  /// Transpose matrix of 2x2 `u64` matrix.
582  #[inline]
583  pub fn transpose(data: [u64x2; 2]) -> [u64x2; 2] {
584    cast(i64x2::transpose(cast(data)))
585  }
586
587  #[inline]
588  pub fn to_array(self) -> [u64; 2] {
589    cast(self)
590  }
591
592  #[inline]
593  pub fn as_array(&self) -> &[u64; 2] {
594    cast_ref(self)
595  }
596
597  #[inline]
598  pub fn as_mut_array(&mut self) -> &mut [u64; 2] {
599    cast_mut(self)
600  }
601
602  #[inline]
603  #[must_use]
604  pub fn min(self, rhs: Self) -> Self {
605    self.simd_lt(rhs).blend(self, rhs)
606  }
607
608  #[inline]
609  #[must_use]
610  pub fn max(self, rhs: Self) -> Self {
611    self.simd_gt(rhs).blend(self, rhs)
612  }
613
614  integer_fn_clamp!();
615
616  #[inline]
617  #[must_use]
618  pub fn saturating_add(self, rhs: Self) -> Self {
619    pick! {
620      if #[cfg(any(target_feature="sse2", target_feature="simd128"))] {
621        let result = self + rhs;
622        result.simd_lt(self).blend(Self::MAX, result)
623      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
624        unsafe { Self { neon: vqaddq_u64(self.neon, rhs.neon) } }
625      } else {
626        Self {
627          arr: [
628            self.arr[0].saturating_add(rhs.arr[0]),
629            self.arr[1].saturating_add(rhs.arr[1]),
630          ],
631        }
632      }
633    }
634  }
635
636  #[inline]
637  #[must_use]
638  pub fn saturating_sub(self, rhs: Self) -> Self {
639    pick! {
640      if #[cfg(any(target_feature="sse2", target_feature="simd128"))] {
641        let result = self - rhs;
642        result.simd_gt(self).blend(Self::MIN, result)
643      } else if #[cfg(all(target_feature="neon",target_arch="aarch64"))]{
644        unsafe { Self { neon: vqsubq_u64(self.neon, rhs.neon) } }
645      } else {
646        Self {
647          arr: [
648            self.arr[0].saturating_sub(rhs.arr[0]),
649            self.arr[1].saturating_sub(rhs.arr[1]),
650          ],
651        }
652      }
653    }
654  }
655
656  /// Lanewise saturating multiply.
657  #[inline]
658  #[must_use]
659  pub fn saturating_mul(self, rhs: Self) -> Self {
660    let self_array = self.to_array();
661    let rhs_array = rhs.to_array();
662
663    Self::new([
664      self_array[0].saturating_mul(rhs_array[0]),
665      self_array[1].saturating_mul(rhs_array[1]),
666    ])
667  }
668
669  integer_fn_saturating_div!([0, 1]);
670
671  #[inline]
672  #[must_use]
673  pub fn mul_keep_high(self, rhs: Self) -> Self {
674    let arr1: [u64; 2] = cast(self);
675    let arr2: [u64; 2] = cast(rhs);
676    cast([
677      ((arr1[0] as u128 * arr2[0] as u128) >> 64) as u64,
678      ((arr1[1] as u128 * arr2[1] as u128) >> 64) as u64,
679    ])
680  }
681}