SuperDex Physics C++ API
Loading...
Searching...
No Matches
x64_simd_double_2_inl.h
Go to the documentation of this file.
1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 *
4 * Licensed under the Apache License, Version 2.0 (the "License");
5 * you may not use this file except in compliance with the License.
6 * You may obtain a copy of the License at
7 *
8 * http://www.apache.org/licenses/LICENSE-2.0
9 *
10 * Unless required by applicable law or agreed to in writing, software
11 * distributed under the License is distributed on an "AS IS" BASIS,
12 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13 * See the License for the specific language governing permissions and
14 * limitations under the License.
15 */
16
17#pragma once
18
19#include "x64_simd_inl.h" // for IntelliSense
20
21#if MOCHI_USE_SIMD && MOCHI_ARCH_X64_AVX2
22
23namespace superdex {
24
25/***********************************************************************************************
26 Simd<double, 2>
27*/
28template <>
29class Simd<double, 2> {
30 public:
31 MOCHI_NATIVE_SIMD_IMPL_BOILERPLATE(double, 2, __m128d);
32 Simd(double a, double b) : raw(_mm_set_pd(b, a)) {} // SSE2
33
34 template <class U, MOCHI_REQUIRES_NON_BOOL_SCALAR(U, Scalar)>
35 Simd(U a) : raw(_mm_set_pd1(a)) {} // SSE2
36
37 template <int i>
38 [[nodiscard]] static MOCHI_FORCE_INLINE double Get(Simd v) {
39 static_assert(i >= 0 && i < 2, "Index out of range");
40 return v[i];
41 }
42
43 [[nodiscard]] MOCHI_FORCE_INLINE Scalar operator[](int i) const {
44 MOCHI_ASSERT_VERBOSE(i >= 0 && i < 2, "Index out of range");
45#if MOCHI_COMPILER_MSVC
46 return raw.m128d_f64[i];
47#else
48 return raw[i];
49#endif
50 }
51
52 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Set(Simd v, int i, Scalar value) {
53 MOCHI_ASSERT_VERBOSE(i >= 0 && i < 2, "Index out of range");
54#if MOCHI_COMPILER_MSVC
55 auto result = v;
56 result.raw.m128d_f64[i] = value;
57 return result;
58#else
59 static constexpr __m128i kMasks[] = {{-1LL, 0LL}, {0LL, -1LL}};
60 return _mm_blendv_pd(v.raw, _mm_set_pd1(value), _mm_castsi128_pd(kMasks[i])); // SSE4.1
61#endif
62 }
63
64 template <int i>
65 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Set(Simd v, Scalar value) {
66 static_assert(i >= 0 && i < 2, "Index out of range");
67 if constexpr (i == 0) {
68 return _mm_shuffle_pd(_mm_set1_pd(value), v.raw, 0x02); // SSE2
69 } else {
70 return _mm_shuffle_pd(v.raw, _mm_set1_pd(value), 0x02); // SSE2
71 }
72 }
73
74 template <int N>
75 [[nodiscard]] static MOCHI_FORCE_INLINE bool AllTrue(Simd v) {
76 static_assert(N >= 1 && N <= kSize, "Unsupported N");
77 auto mask = GetMSBitMask(v); // One bit for each byte in the vector
78 if constexpr (N == kSize) {
79 return mask == 0x0000FFFF;
80 } else {
81 return (mask & 0x000000FF) == 0x000000FF;
82 }
83 }
84
85 template <int N>
86 [[nodiscard]] static MOCHI_FORCE_INLINE bool AnyTrue(Simd v) {
87 static_assert(N >= 1 && N <= kSize, "Unsupported N");
88 int mask = GetMSBitMask(v); // One bit for each byte in the vector
89 if constexpr (N == kSize) {
90 return mask != 0;
91 } else {
92 return (mask & 0x000000FF) != 0;
93 }
94 }
95
96 template <int x, int y>
97 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Blend(Simd a, Simd b) {
98 static_assert(x >= 0 && x < 2 && y >= 0 && y < 2, "invalid blend index");
99 if constexpr (x == 0 && y == 0) {
100 return a;
101 } else if constexpr (x == 1 && y == 1) {
102 return b;
103 } else {
104 return _mm_blend_pd(a.raw, b.raw, x | (y << 1)); // SSE4.1
105 }
106 }
107
108 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Broadcast(Scalar const* p) {
109 return Simd{*p};
110 }
111
112 template <int i>
113 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Broadcast(Simd v) {
114 return Shuffle<i, i>(v);
115 }
116
117 [[nodiscard]] static Simd LoadIndexed(Scalar const* ptr, Simd<int64_t, 2> const& indices) {
118 return _mm_i64gather_pd(ptr, indices.raw, sizeof(double)); // AVX2
119 }
120
121 template <int N = kSize>
122 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Load([[maybe_unused]] Scalar const* ptr) {
123 static_assert(N >= 0 && N <= kSize);
124 if constexpr (N == 0) {
125 return Simd::Zero();
126 } else if constexpr (N == 1) {
127 return Simd{*ptr, 0.0};
128 } else {
129 return _mm_loadu_pd(ptr); // SSE2
130 }
131 }
132
133 static_assert(sizeof(long long) == 8);
134
135#if !MOCHI_ARCH_X64_AVX512
136#if MOCHI_COMPILER_MSVC
137 // MSVC defines __m128i as a union. Byte arrays are required to initialize it this way.
138 // clang-format off
139 static constexpr __m128i kLoadMasks[] = {
140 { 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0},
141 {-1, -1, -1, -1, -1, -1, -1, -1, 0, 0, 0, 0, 0, 0, 0, 0},
142 {-1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1}};
143 // clang-format on
144#else
145 // GCC and Clang define __m128i as 'long long' with special attributes.
146 static constexpr __m128i kLoadMasks[] = {{0LL, 0LL}, {-1LL, 0LL}, {-1LL, -1LL}};
147#endif
148#endif
149
150 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Load(Scalar const* ptr, int n) {
151 MOCHI_ASSERT_VERBOSE(n >= 0 && n <= kSize, "Invalid size parameter");
152#if MOCHI_ARCH_X64_AVX512
153 return _mm_maskz_loadu_pd(x64_simd::kLaneMasksS8[n], ptr); // AVX512VL
154#else
155 return _mm_maskload_pd(ptr, kLoadMasks[n]); // AVX
156#endif
157 }
158
159 template <int kTupleCount = kSize>
160 MOCHI_FORCE_INLINE static void
161 LoadTransposed(Scalar const* ptr, Simd& out0, Simd& out1, Simd& out2) {
162 static_assert(kTupleCount >= 1 && kTupleCount <= kSize, "Invalid kTupleCount");
163 constexpr int kCount1 = Clamp(kTupleCount * 3 - 2, 0, 2);
164 constexpr int kCount2 = Clamp(kTupleCount * 3 - 4, 0, 2);
165 auto a = Simd::Load<2>(ptr).raw; // [0,1]
166 auto b = Simd::Load<kCount1>(kCount1 == 0 ? ptr : ptr + 2).raw; // [2,3]
167 auto c = Simd::Load<kCount2>(kCount2 == 0 ? ptr : ptr + 4).raw; // [4,5]
168 out0.raw = _mm_shuffle_pd(a, b, 0b0010); // [0,3]
169 out1.raw = _mm_shuffle_pd(a, c, 0b0001); // [1,4]
170 out2.raw = _mm_shuffle_pd(b, c, 0b0010); // [2,5]
171 }
172
173 template <int N = kSize>
174 static MOCHI_FORCE_INLINE void Store([[maybe_unused]] Scalar* ptr, [[maybe_unused]] Simd v) {
175 static_assert(N >= 0 && N <= kSize);
176 if constexpr (N == 0) {
177 } else if constexpr (N < kSize) {
178 // About 3X faster than a masked store on older AMD CPUs. About the same on others.
179 memcpy(ptr, &v, sizeof(Scalar) * N);
180 } else {
181 _mm_storeu_pd(ptr, v.raw); // SSE2
182 }
183 }
184
185 static MOCHI_FORCE_INLINE void Store(Scalar* ptr, Simd v, int n) {
186 MOCHI_ASSERT_VERBOSE(n >= 0 && n <= kSize, "Invalid size parameter");
187#if MOCHI_ARCH_X64_AVX512
188 _mm_mask_storeu_pd(ptr, x64_simd::kLaneMasksS8[n], v.raw); // AVX512VL
189#else
190 switch (n) { // clang-format off
191 case 1: Store<1>(ptr, v); break;
192 case 2: Store<2>(ptr, v); break;
193 MOCHI_UNLIKELY default: break;
194 } // clang-format on
195#endif
196 }
197
198 MOCHI_FORCE_INLINE static int StoreSelected(Scalar* ptr, Simd condition, Simd values) {
199 auto mask = _mm_movemask_pd(condition.raw);
200 auto swapped = _mm_shuffle_pd(values.raw, values.raw, 1); // swap halves
201 auto blendMask =
202 _mm_castsi128_pd(_mm_set1_epi32((mask & 1) - 1)); // swap first bit of mask is zero
203 _mm_storeu_pd(ptr, _mm_blendv_pd(values.raw, swapped, blendMask));
204 return _mm_popcnt_u32(mask);
205 }
206
207 template <int kTupleCount = kSize>
208 MOCHI_FORCE_INLINE static void StoreTransposed(Scalar* ptr, Simd a, Simd b, Simd c) {
209 static_assert(kTupleCount >= 1 && kTupleCount <= kSize, "Invalid kTupleCount");
210 // a = [0,3], b = [1,4], c = [2,5]
211 Simd::Store<2>(ptr, _mm_shuffle_pd(a.raw, b.raw, 0b00)); // [0,1]
212 constexpr int kCount1 = Clamp(kTupleCount * 3 - 2, 0, 2);
213 constexpr int kCount2 = Clamp(kTupleCount * 3 - 4, 0, 2);
214 if constexpr (kCount1 > 0) {
215 Simd::Store<kCount1>(ptr + 2, _mm_shuffle_pd(c.raw, a.raw, 0b10)); // [2,3]
216 }
217 if constexpr (kCount2 > 0) {
218 Simd::Store<kCount2>(ptr + 4, _mm_shuffle_pd(b.raw, c.raw, 0b11)); // [4,5]
219 }
220 }
221
222 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Select(Simd mask, Simd a, Simd b) {
223 return _mm_blendv_pd(b.raw, a.raw, mask.raw); // SSE4.1
224 }
225
226 // return Simd{v[x], v[y]}
227 template <int x = 0, int y = 1>
228 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Shuffle(Simd v) {
229 static_assert(x >= 0 && x < 2, "Invalid index");
230 static_assert(y >= 0 && y < 2, "Invalid index");
231 if constexpr (x == 0 && y == 1) {
232 return v;
233 } else {
234 return _mm_shuffle_pd(v.raw, v.raw, x | (y << 1)); // SSE2
235 }
236 }
237
238 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Sqrt(Simd v) {
239 return _mm_sqrt_pd(v.raw); // SSE2
240 }
241
242 [[nodiscard]] static MOCHI_FORCE_INLINE Simd RcpApprox(Simd v) {
243 return Simd{1.0} / v;
244 }
245
246 [[nodiscard]] static MOCHI_FORCE_INLINE Simd RcpSqrtApprox(Simd v) {
247 return Simd{1.0} / Sqrt(v);
248 }
249
250 // Broadcast the value -0.0. Use this in bitwise operations to affect just the sign bit.
251 [[nodiscard]] static MOCHI_FORCE_INLINE Simd SignBitMask() {
252 return _mm_castsi128_pd(_mm_set1_epi64x(0x8000000000000000LL)); // SSE2, SSE2
253 }
254
255 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Abs(Simd v) {
256 return _mm_andnot_pd(SignBitMask().raw, v.raw); // SSE2
257 }
258
259 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Min(Simd a, Simd b) {
260 return _mm_min_pd(a.raw, b.raw); // SSE2
261 }
262
263 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Max(Simd a, Simd b) {
264 return _mm_max_pd(a.raw, b.raw); // SSE2
265 }
266
267 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Floor(Simd a) {
268 return _mm_floor_pd(a.raw); // SSE4.1
269 }
270
271 [[nodiscard]] static MOCHI_FORCE_INLINE Simd FastRound(Simd v) {
272 return _mm_round_pd(v.raw, _MM_FROUND_TO_NEAREST_INT); // SSE4.1
273 }
274
275#if MOCHI_ARCH_X64_SVML
276 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Cos(Simd a) {
277 return _mm_cos_pd(a.raw); // SSE
278 }
279
280 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Sin(Simd a) {
281 return _mm_sin_pd(a.raw); // SSE
282 }
283
284 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Tan(Simd a) {
285 return _mm_tan_pd(a.raw); // SSE
286 }
287
288 [[nodiscard]] static MOCHI_FORCE_INLINE Simd ACos(Simd a) {
289 return _mm_acos_pd(a.raw); // SSE
290 }
291
292 [[nodiscard]] static MOCHI_FORCE_INLINE Simd ASin(Simd a) {
293 return _mm_asin_pd(a.raw); // SSE
294 }
295
296 [[nodiscard]] static MOCHI_FORCE_INLINE Simd ATan(Simd a) {
297 return _mm_atan_pd(a.raw); // SSE
298 }
299
300 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Exp(Simd a) {
301 return _mm_exp_pd(a.raw); // SSE
302 }
303
304 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Ln(Simd a) {
305 return _mm_log_pd(a.raw); // SSE
306 }
307
308 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Tanh(Simd a) {
309 return _mm_tanh_pd(a.raw); // SSE
310 }
311#endif // MOCHI_ARCH_X64_SVML
312
313 [[nodiscard]] static MOCHI_FORCE_INLINE Simd MulAdd(Simd a, Simd b, Simd c) {
314#if MOCHI_ARCH_X64_FMA
315 return {_mm_fmadd_pd(a.raw, b.raw, c.raw)}; // FMA
316#else
317 return (a * b) + c;
318#endif
319 }
320
321 [[nodiscard]] static MOCHI_FORCE_INLINE Simd MulSub(Simd a, Simd b, Simd c) {
322#if MOCHI_ARCH_X64_FMA
323 return _mm_fmsub_pd(a.raw, b.raw, c.raw); // FMA
324#else
325 return (a * b) - c;
326#endif
327 }
328
329 [[nodiscard]] static MOCHI_FORCE_INLINE Simd NegMulAdd(Simd a, Simd b, Simd c) {
330#if MOCHI_ARCH_X64_FMA
331 return _mm_fnmadd_pd(a.raw, b.raw, c.raw); // FMA
332#else
333 return -(a * b) + c;
334#endif
335 }
336
337 [[nodiscard]] static MOCHI_FORCE_INLINE Simd NegMulSub(Simd a, Simd b, Simd c) {
338#if MOCHI_ARCH_X64_FMA
339 return _mm_fnmsub_pd(a.raw, b.raw, c.raw); // FMA
340#else
341 return -(a * b) - c;
342#endif
343 }
344
345 template <int N = 2>
346 [[nodiscard]] static MOCHI_FORCE_INLINE Scalar HMin(Simd a) {
347 static_assert(N == 2, "Unsupported N");
348 return Get<0>(Min(a, Broadcast<1>(a)));
349 }
350
351 template <int N = 2>
352 [[nodiscard]] static MOCHI_FORCE_INLINE Scalar HMax(Simd a) {
353 static_assert(N == 2, "Unsupported N");
354 return Get<0>(Max(a, Broadcast<1>(a)));
355 }
356
357 template <int N>
358 [[nodiscard]] static MOCHI_FORCE_INLINE Scalar HSum(Simd a) {
359 static_assert(N == 2, "Unsupported N");
360 return Get<0>(a) + Get<1>(a);
361 }
362
363 template <int N>
364 [[nodiscard]] static MOCHI_FORCE_INLINE Scalar HProd(Simd a) {
365 static_assert(N == 2, "Unsupported N");
366 return Get<0>(Broadcast<0>(a) * Broadcast<1>(a));
367 }
368
369 template <int N>
370 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Dot(Simd a, Simd b) {
371 static_assert(N == 2, "Unsupported N");
372#if MOCHI_COMPILER_CLANG
373 return _mm_dp_pd(a.raw, b.raw, -1); // SSE4.1
374#else
375 return _mm_dp_pd(a.raw, b.raw, 0xFF); // SSE4.1
376#endif
377 }
378
379 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator<(Simd rhs) const {
380 return _mm_cmplt_pd(this->raw, rhs.raw); // SSE
381 }
382
383 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator>(Simd rhs) const {
384 return _mm_cmpgt_pd(this->raw, rhs.raw); // SSE2
385 }
386
387 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator<=(Simd rhs) const {
388 return _mm_cmple_pd(this->raw, rhs.raw); // SSE2
389 }
390
391 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator>=(Simd rhs) const {
392 return _mm_cmpge_pd(this->raw, rhs.raw); // SSE2
393 }
394
395 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Equal(Simd a, Simd b) {
396 return _mm_cmpeq_pd(a.raw, b.raw); // SSE2
397 }
398
399 [[nodiscard]] static MOCHI_FORCE_INLINE Simd NotEqual(Simd a, Simd b) {
400 return _mm_cmpneq_pd(a.raw, b.raw); // SSE2
401 }
402
403 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Zero() {
404 return _mm_setzero_pd(); // SSE2
405 }
406
407 [[nodiscard]] MOCHI_FORCE_INLINE bool operator==(Simd rhs) const {
408 auto mask = GetMSBitMask(Equal(raw, rhs.raw));
409 return mask == 0xFFFF; // All values equal
410 }
411
412 [[nodiscard]] MOCHI_FORCE_INLINE bool operator!=(Simd rhs) const {
413 auto mask = GetMSBitMask(NotEqual(raw, rhs.raw));
414 return mask != 0; // Any values not equal
415 }
416
417 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator~() const {
418 // _mm_cmpeq_epi32 appears to be the fastest way to fill an SSE register with ones.
419 __m128i dummy{};
420 __m128d ones = _mm_castsi128_pd(_mm_cmpeq_epi32(dummy, dummy)); // SSE2, SSE2
421 return _mm_xor_pd(raw, ones); // SSE2
422 }
423
424 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator-() const {
425 return _mm_xor_pd(raw, SignBitMask().raw); // SSE2, SSE2
426 }
427
428 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator+(Simd rhs) const {
429 return _mm_add_pd(raw, rhs.raw); // SSE2
430 }
431
432 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator-(Simd rhs) const {
433 return _mm_sub_pd(raw, rhs.raw); // SSE2
434 }
435
436 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator*(Simd rhs) const {
437 return _mm_mul_pd(raw, rhs.raw); // SSE2
438 }
439
440 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator/(Simd rhs) const {
441 return _mm_div_pd(raw, rhs.raw); // SSE2
442 }
443
444 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator&(Simd rhs) const {
445 return _mm_and_pd(raw, rhs.raw); // SSE2
446 }
447
448 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator|(Simd rhs) const {
449 return _mm_or_pd(raw, rhs.raw); // SSE2
450 }
451
452 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator^(Simd rhs) const {
453 return _mm_xor_pd(raw, rhs.raw); // SSE2
454 }
455
456 private:
457 // Integer mask with with the most significant bit of each byte in the vector
458 [[nodiscard]] static MOCHI_FORCE_INLINE int GetMSBitMask(Simd a) {
459 return _mm_movemask_epi8(_mm_castpd_si128(a.raw)); // SSE2, SSE2
460 }
461};
462
463} // namespace superdex
464
465#endif // MOCHI_USE_SIMD && MOCHI_ARCH_X64_AVX2
Simd operator&(Simd rhs) const
bool operator==(Simd rhs) const
NativeType raw
Definition simd.h:174
Simd operator>(Simd rhs) const
Simd operator*(Simd rhs) const
Simd operator^(Simd rhs) const
Simd operator-() const
Simd operator>=(Simd rhs) const
Simd operator<(Simd rhs) const
bool operator!=(Simd rhs) const
Simd operator|(Simd rhs) const
static constexpr int kSize
Definition simd.h:96
Simd operator+(Simd rhs) const
Simd operator~() const
Simd operator/(Simd rhs) const
Simd operator<=(Simd rhs) const
Scalar operator[](int i) const
#define MOCHI_ASSERT_VERBOSE(condition_without_side_effects,...)
Definition debug.h:102
#define MOCHI_UNLIKELY
#define MOCHI_FORCE_INLINE
T Dot(Simd< T, N > a, Simd< T, N > b)
Definition simd.h:673
constexpr T ACos(T a)
Simd< T, 2 > Shuffle(Simd< T, 2 > a)
Definition simd_inl.h:273
V LoadIndexed(typename V::Scalar const *ptr, Simd< I, V::kSize > indices)
Definition simd_inl.h:200
constexpr T const & Min(T const &a, T const &b)
constexpr auto Equal(T const &a, T const &b)
constexpr T Sin(T a)
T HSum(Simd< T, N > a)
Definition simd_inl.h:377
T HMin(Simd< T, N > a)
Definition simd_inl.h:389
bool AllTrue(T const &a)
Definition basic_utils.h:60
constexpr auto MulAdd(A a, B b, C c)
Simd< T, N > Tanh(Simd< T, N > a)
Definition simd_inl.h:709
Simd< T, N > Set(Simd< T, N > a, T value)
Definition simd_inl.h:313
constexpr auto NotEqual(T const &a, T const &b)
constexpr T Exp(T a)
constexpr T Cos(T a)
T HMax(Simd< T, N > a)
Definition simd_inl.h:395
V Broadcast(typename V::Scalar a)
Definition simd_inl.h:115
constexpr T Abs(T a)
Definition basic_utils.h:50
constexpr auto MulSub(A a, B b, C c)
Simd< T, N > Blend(Simd< T, N > a, Simd< T, N > b)
Definition simd_inl.h:288
constexpr T Tan(T a)
bool AnyTrue(T const &a)
Definition basic_utils.h:66
constexpr T Select(bool condition, T a, T b)
constexpr T Sqrt(T a)
constexpr T ATan(T a)
Simd< T, N > Ln(Simd< T, N > a)
Definition simd_inl.h:699
constexpr auto NegMulAdd(A a, B b, C c)
T Get(Simd< T, N > v)
Definition simd_inl.h:303
constexpr T Floor(T a)
constexpr T ASin(T a)
constexpr ValT Clamp(ValT value, MinT min, MaxT max)
T HProd(Simd< T, N > a)
Definition simd_inl.h:383
constexpr T const & Max(T const &a, T const &b)
void LoadTransposed(T const *ptr, Simd< T, N > &out0, Simd< T, N > &out1, Simd< T, N > &out2)
Definition simd_inl.h:207
Simd< T, N > FastRound(Simd< T, N > a)
Definition simd_inl.h:416
void StoreTransposed(T *ptr, Simd< T, N > a, Simd< T, N > b, Simd< T, N > c)
Definition simd_inl.h:245
void Store(T *ptr, Simd< T, N > a)
Definition simd_inl.h:213
Simd< T, N > RcpSqrtApprox(Simd< T, N > a)
Definition simd_inl.h:357
constexpr T RcpApprox(T a)
constexpr auto NegMulSub(A a, B b, C c)
int StoreSelected(T *ptr, Simd< MaskT, N > condition, Simd< T, N > values)
Definition simd_inl.h:225
V Load(typename V::Scalar const *ptr)
Definition simd_inl.h:184
#define MOCHI_NATIVE_SIMD_IMPL_BOILERPLATE(T, N, NativeT)