SuperDex Physics C++ API
Loading...
Searching...
No Matches
simd.h
Go to the documentation of this file.
1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 *
4 * Licensed under the Apache License, Version 2.0 (the "License");
5 * you may not use this file except in compliance with the License.
6 * You may obtain a copy of the License at
7 *
8 * http://www.apache.org/licenses/LICENSE-2.0
9 *
10 * Unless required by applicable law or agreed to in writing, software
11 * distributed under the License is distributed on an "AS IS" BASIS,
12 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13 * See the License for the specific language governing permissions and
14 * limitations under the License.
15 */
16
17#pragma once
22
23#include <type_traits>
24#include <utility>
25
26namespace superdex {
27
28// Forward declaration for use in composite StaticCast (full definition is in half.h)
29struct Half;
30
31// Number of values of type T that fit in a native SIMD register
32template <class T>
34
35// Type used with enable_if_t that can be used to select a Simd specialization based on type traits
36struct SimdConcept {};
37
38/***********************************************************************************************
39 Simd<T, N>
40
41 A SIMD vector storing N values of type T.
42
43 This generic template must be fully specialized for supported sizes and types.
44 There currently is no reference implementation.
45*/
46template <class T, int N = kSimdDefaultSize<T>, class Concept = SimdConcept>
47class Simd;
48
49// clang-format off
50namespace details {
51template <class T> constexpr bool IsSimdDef = false;
52template <class T, int N> constexpr bool IsSimdDef<Simd<T, N>> = true;
53template <class T> constexpr bool IsSimdSupportedTypeDef = false;
54template <> inline constexpr bool IsSimdSupportedTypeDef<float> = true;
55template <> inline constexpr bool IsSimdSupportedTypeDef<double> = true;
56template <> inline constexpr bool IsSimdSupportedTypeDef<int> = true;
57template <> inline constexpr bool IsSimdSupportedTypeDef<int64_t> = true;
58template <class T> constexpr bool IsHalfDef = false; // Specialized in half.h
59} // namespace details
60// clang-format on
61
62// Compile-time type traits:
63// - IsSimd<T> is true T is an instantiation of the Simd template class.
64// - IsSimdSupportedType<T> is true if Simd<T, N> is supported for some integer N.
65// - IsHalf<T> is true if T is Half (from half.h).
66#if MOCHI_LANGUAGE_CPP20
67template <class T>
68concept IsSimd = details::IsSimdDef<std::decay_t<T>>;
69template <class T>
70concept IsSimdSupportedType = details::IsSimdSupportedTypeDef<T>;
71template <class T>
72concept IsHalf = details::IsHalfDef<std::decay_t<T>>;
73#else
74template <class T>
75inline constexpr bool IsSimd = details::IsSimdDef<std::decay_t<T>>;
76template <class T>
77inline constexpr bool IsSimdSupportedType = details::IsSimdSupportedTypeDef<T>;
78template <class T>
79inline constexpr bool IsHalf = details::IsHalfDef<std::decay_t<T>>;
80#endif
81
82/***********************************************************************************************
83 Simd<T, N>
84
85 A SIMD vector storing N values of type T.
86
87 This generic template must be fully specialized for supported sizes and types.
88 There currently is no reference implementation.
89*/
90template <class T, int N, class Concept>
91class Simd {
92 public:
93 struct NotSupported {};
95 using Scalar = T;
96 static constexpr int kSize = N;
97 static constexpr bool kIsSupported = false; // Can this combination of T and N function
98 static constexpr bool kIsComposite = false; // Is this type a composite of native sizes
99 static constexpr bool kIsEmulated = false; // Whether it emulates SIMD instructions (if true) or
100 // uses actual native SIMD instructions (if false).
101
102 // Check to prevent misuse of Simd<T>::kIsSupported if queried with a const T.
103 static_assert(!std::is_const_v<T>, "Scalar type must not be const");
104
106 static_assert(std::is_same_v<T, void>, "The requested SIMD size or type is not supported");
107 }
108
109 Simd(Simd const& rhs) = default;
110
111 // Implicit conversion from native storage type.
112 Simd(NativeType rhs) : raw(rhs) {};
113
114 // Broadcast scalar
116
117 // Init from 2 or more scalars. Remaining values will be zero.
118 template <class... Ts>
119 Simd(Scalar a, Scalar b, Ts... vals);
120
121 // Lower-case for compatibility with std::size
122 static constexpr size_t size() {
123 return kSize;
124 }
125
126 // Assignment
129
130 // Comparison operators
131 bool operator==(Simd rhs) const;
132 bool operator!=(Simd rhs) const;
133 Simd operator<(Simd rhs) const;
134 Simd operator<=(Simd rhs) const;
135 Simd operator>(Simd rhs) const;
136 Simd operator>=(Simd rhs) const;
137
138 // Math operators
140 Simd operator+(Simd rhs) const;
141 Simd operator-(Simd rhs) const;
142 Simd operator*(Simd rhs) const;
143 Simd operator/(Simd rhs) const;
152
153 // Bitwise operators
155 Simd operator&(Simd rhs) const;
156 Simd operator|(Simd rhs) const;
157 Simd operator^(Simd rhs) const;
161
162 // Logical operators
163 // Each lane must have all-bits-0 (logical false) or all-bits-1 (logical true).
164 Simd operator&&(Simd rhs) const;
165 Simd operator||(Simd rhs) const;
166
167 // Left shift
168 Simd operator<<(int shift) const;
169
170 // Indexing (read only)
171 Scalar operator[](int i) const;
172
173 // Storage type is implementation specific (e.g. __m256)
175};
176
177/// @brief Reinterpret the bits of one Simd type to another of the same binary size.
178/// @note Example: ReinterpretCast<Vec2d>(myVec4i);
179/// @note The generic version of ReinterpretCast is not valid, but specializations
180/// can be provided for specific pairs of types.
181template <class VecTo, class VecFrom, MOCHI_CONCEPT(IsSimd<VecTo>&& IsSimd<VecFrom>)>
182MOCHI_ANY MOCHI_FORCE_INLINE VecTo ReinterpretCast(VecFrom const& a) {
183 static_assert(
184 std::is_same_v<std::decay_t<VecTo>, std::decay_t<VecFrom>>,
185 "ReinterpretCast not supported for these types.");
186 return a;
187}
188
189/// @brief StaticCast from one Simd type to another of the same size.
190/// @note The generic version is a fallback.
191/// Specializations are provided for specific pairs of types.
192template <class VecTo, class VecFrom, MOCHI_CONCEPT(IsSimd<VecTo>&& IsSimd<VecFrom>)>
194
195} // namespace superdex
197namespace superdex {
198
199// Aliases
217
218// clang-format off
219/***********************************************************************************************
220 Native SIMD Function Support Matrix (please keep up-to-date)
221 Functions supporting Simd<T, N> also support composable wider sizes when
222 every native component supports the function.
223
224 For Simd<Half, N> (Vec8h, Vec16h, Vec32h), include half.h.
225
226************************************************************************************************
227
228 | Vec2d | Vec2l | Vec4d | Vec4f | Vec4i | Vec4l | Vec8d | Vec8f | Vec8h | Vec8i | Vec8l | Vec16f | Vec16h | Vec16i | Vec32h |
229 Abs | x | | x | x | | | x | x | | | | x | | | |
230 ACos | x | | x | x | | | x | x | | | | x | | | |
231 AllTrue | x | x | x | x | x | x | x | x | x | x | x | x | x | x | x |
232 AnyTrue | x | | x | x | x | | x | x | x | x | | x | x | x | x |
233 ASin | x | | x | x | | | x | x | | | | x | | | |
234 ATan | x | | x | x | | | x | x | | | | x | | | |
235 Blend | x | x | x | x | x | | | | | | | | | | |
236 Broadcast | x | x | x | x | x | x | x | x | | x | x | x | | x | |
237 Clamp | x | | x | x | | | x | x | | | | x | | | |
238 Cos | x | | x | x | | | x | x | | | | x | | | |
239 Cross3 | | | x | x | | | | | | | | | | | |
240 Dot | x | | x | x | | | x | x | | | | x | | | |
241 (V)Equal | x | x | x | x | x | x | x | x | x | x | x | x | x | x | x |
242 Exp | x | | x | x | | | x | x | | | | x | | | |
243 FastRound | x | | x | x | | | x | x | | | | x | | | |
244 Floor | x | | x | x | | | x | x | | | | x | | | |
245 Get | x | x | x | x | x | x | x | x | x | x | x | x | x | x | x |
246 Get0 | x | x | x | x | x | x | x | x | x | x | x | x | x | x | x |
247 GetHalf | | | x | | | x | x | x | | x | x | x | x | x | x |
248 HMax | x | x | x | x | x | x | x | x | | x | x | x | | x | |
249 HMin | x | x | x | x | x | x | x | x | | x | x | x | | x | |
250 HProd | x | | x | x | | | x | | | | | | | | |
251 HSum | x | x | x | x | x | x | x | x | | x | x | x | | x | |
252 (V)IsFinite | x | | x | x | | | x | x | | | | x | | | |
253 IsTrue | x | x | x | x | x | x | x | x | | x | x | x | | x | |
254 Lerp | x | | x | x | | | x | x | | | | x | | | |
255 Ln | x | | x | x | | | x | x | | | | x | | | |
256 Load | x | x | x | x | x | x | x | x | x | x | x | x | x | x | x |
257 LoadIndexed | x | | x | x | | | x | x | | | | x | | | |
258 LoadTransposed | x | x | x | x | x | x | x | x | | x | x | x | | x | |
259 Max | x | x | x | x | x | x | x | x | | x | x | x | | x | |
260 Min | x | x | x | x | x | x | x | x | | x | x | x | | x | |
261 MulAdd | x | | x | x | | | x | x | | | | x | | | |
262 MulSub | x | | x | x | | | x | x | | | | x | | | |
263 (V)NearEqual | x | | x | x | | | x | x | | | | x | | | |
264 (V)NearZero | x | | x | x | | | x | x | | | | x | | | |
265 Neg (4 bools) | | | x | x | | | | | | | | | | | |
266 NegMulAdd | x | | x | x | | | x | x | | | | x | | | |
267 NegMulSub | x | | x | x | | | x | x | | | | x | | | |
268 (V)Norm | x | | x | x | | | x | x | | | | x | | | |
269 Normalize | x | | x | x | | | x | x | | | | x | | | |
270 NotEqual | x | | x | x | x | | x | x | x | x | | x | x | x | x |
271 VNotEqual | x | x | x | x | x | x | x | x | x | x | x | x | x | x | x |
272 OrthogonalVector3 | | | x | x | | | | | | | | | | | |
273 RcpApprox | x | | x | x | | | x | x | | | | x | | | |
274 RcpSqrtApprox | x | | x | x | | | x | x | | | | x | | | |
275 Select | x | x | x | x | x | x | x | x | | x | x | x | | x | |
276 Set | x | | x | x | x | | x | x | | x | | x | | x | |
277 Sequence | | x | | | x | x | | | | x | x | | | x | |
278 ShiftRight | | x | | | x | x | | | | x | x | | | x | |
279 Shuffle (1 arg) | x | x | x | x | x | x | | | | | | | | | |
280 Shuffle (2 arg) | | | x | x | x | x | | | | | | | | | |
281 Sign | x | | x | x | | | x | x | | | | x | | | |
282 SignedSqrt | x | | x | x | | | x | x | | | | x | | | |
283 SimdBasisVector | | | x | x | | | | | | | | | | | |
284 SimdMask | x | x | x | x | x | x | x | x | | x | x | x | | x | |
285 SimdZero | x | x | x | x | x | x | x | x | x | x | x | x | x | x | x |
286 Sin | x | | x | x | | | x | x | | | | x | | | |
287 SinCos | x | | x | x | | | x | x | | | | x | | | |
288 Sqr | x | x | x | x | x | x | x | x | | x | x | x | | x | |
289 Sqrt | x | | x | x | | | x | x | | | | x | | | |
290 Store | x | x | x | x | x | x | x | x | x | x | x | x | x | x | x |
291 StoreSelected | x | x | x | x | x | x | x | x | | x | x | x | | x | |
292 StoreTransposed | x | x | x | x | x | x | x | x | | x | x | x | | x | |
293 Tan | x | | x | x | | | x | x | | | | x | | | |
294 Tanh | x | | x | x | | | x | x | | | | x | | | |
295 ToSimd | | | x | x | x | x | | | | | | | | | |
296 ToSimdDirection | | | x | x | | | | | | | | | | | |
297 ToSimdPoint | | | x | x | | | | | | | | | | | |
298
299***********************************************************************************************/
300// clang-format on
301
302// Make kDefaultNearEqualEpsilon<T> "just work" when T is a Simd type
303template <class T, int N>
305
306// Broadcast a single value to all elements of the result
307template <class V, MOCHI_CONCEPT(IsSimd<V>)>
308MOCHI_ANY MOCHI_FORCE_INLINE V Broadcast(typename V::Scalar a);
309
310// Broadcast a single value at address ptr to all elements of the result
311template <class V, MOCHI_CONCEPT(IsSimd<V>)>
312MOCHI_ANY MOCHI_FORCE_INLINE V Broadcast(typename V::Scalar const* ptr);
313
314// Broadcast the ith element of a vector to all elements of the result
315template <int i, class T, int N>
316MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Broadcast(Simd<T, N> v);
317
318// Broadcast the ith element of a vector to all elements of the result.
319// Prefer Broadcast<i> if index is known at compile time.
320template <class T, int N>
321MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Broadcast(Simd<T, N> v, int i);
322
323// Return V{b0 ? TRUE : FALSE, b1 ? TRUE : FALSE, ...} where TRUE is represented by all bits set
324// to 1 (e.g. 0xFFFFFFFF) and FALSE is represented by all bits set to 0 (e.g. 0x00000000). This is
325// the same type of mask returned by comparison operators like <, <=, >, and >=. It can be used with
326// functions like AllTrue, AnyTrue, and Select.
327// Example: SimdMask<Vec4r>(true, false, true, false)
328template <class V, class... MoreBools>
329MOCHI_ANY MOCHI_FORCE_INLINE V SimdMask(bool b0, bool b1, MoreBools... bs);
330
331// Return 0 value
332//
333// The function is templated on the return type.
334template <class V = Vec4r>
336
337// Return a vector with a single component set to 1, and the rest to 0.
338template <int i, class V = Vec4r>
340
341// Return a vector with a single component set to 1, and the rest to 0.
342template <class V = Vec4r>
344
345// Return {a[0], a[1], a[2], 1}. Used for positions that can be both rotated & translated.
346template <class V>
348
349// Return {a[0], a[1], a[2], 0}. Used for direction vectors that can only be rotated.
350template <class V>
352
353// Read V::kSize scalars from the specified address into a vector
354template <class V, MOCHI_CONCEPT(IsSimd<V>)>
355MOCHI_ANY MOCHI_FORCE_INLINE V Load(typename V::Scalar const* ptr);
356
357// Read N scalars from the specified address into a vector.
358template <int N, class V, MOCHI_CONCEPT(IsSimd<V>)>
359MOCHI_ANY MOCHI_FORCE_INLINE V Load(typename V::Scalar const* ptr);
360
361// Read n scalars from the specified address into a vector.
362// Prefer Load<N, V> if the number is a constexpr.
363template <class V, MOCHI_CONCEPT(IsSimd<V>)>
364MOCHI_ANY MOCHI_FORCE_INLINE V Load(typename V::Scalar const* ptr, int n);
365
366// Read V::kSize scalars from an unaligned address offset by Simd<I, kSize>, where I is an integer
367// type.
368template <class V, class I, MOCHI_CONCEPT(IsSimd<V>)>
370LoadIndexed(typename V::Scalar const* ptr, Simd<I, V::kSize> indices);
371
372// Load (kTupleCount * 3) values from memory. Deinterleave into 3 output vectors.
373// By default (kTupleCount == N).
374//
375// If the memory order was:
376// {x0, y0, z0, x1, y1, z1, x2, y2, z2, x3, y3, z3}
377//
378// ...then the output vectors will be:
379// {x0, x1, x2, x3}, {y0, y1, y2, y3}, {z0, z1, z2, z3}
380//
381template <int kTupleCount = -1, class T, int N>
383LoadTransposed(T const* ptr, Simd<T, N>& out0, Simd<T, N>& out1, Simd<T, N>& out2);
384
385// Return the first component of a vector (e.g. a[0])
386template <class T, int N>
387MOCHI_ANY MOCHI_FORCE_INLINE T Get0(Simd<T, N> v);
388
389// Return the ith component of a vector (e.g a[i]), when the index is a constexpr
390template <int i, class T, int N>
391MOCHI_ANY MOCHI_FORCE_INLINE T Get(Simd<T, N> v);
392
393// Return the low or high half of a vector via GetHalf<0>(a) or GetHalf<1>(a)
394template <int iHalf, class T, int N>
395MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N / 2> GetHalf(Simd<T, N> a);
396
397// Return a copy of the vector with the ith component replaced with a new value.
398template <int i, class T, int N>
399[[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Set(Simd<T, N> a, T value);
400
401// Return a copy of the vector with the ith component replaced with a new value.
402template <class T, int N>
403[[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Set(Simd<T, N> a, int i, T value);
404
405// Returns a vector with integer values [0, 1, 2, 3, ...].
406// If you want a different starting value, then add an integer to the result.
407// Example: (Sequence<V>() + 1) results in the sequence [1, 2, 3, ...]
408template <class V>
409[[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE V Sequence();
410
411// Write COUNT scalars from a vector to the specified address. A COUNT of -1 means "all".
412template <int COUNT = -1, class T, int N>
413MOCHI_ANY MOCHI_FORCE_INLINE void Store(T* ptr, Simd<T, N> a);
414
415// Write count scalars from a vector to the specified address.
416// Prefer Store<N> if the number is a constexpr.
417template <class T, int N>
418MOCHI_ANY MOCHI_FORCE_INLINE void Store(T* ptr, Simd<T, N> a, int count);
419
420// Write each value for which condition[i] is true. It may then write unspecified values for a total
421// of N values written, so the destination buffer must be large enough. Returns the number of
422// values for which condition[i] was true.
423//
424// Example - Store all the positive values (branchless):
425// int count = 0;
426// for (int i = 0; i + N <= isize(src); i += N) {
427// auto values = Load<Simd<T, N>>(&src[i]);
428// count += StoreSelected(&dst[count], values >= 0, values);
429// }
430//
431// Warning: If the condition is somewhat predictable (e.g. many false values in a row), then a
432// traditional branching algorithm may be faster. You should always profile it.
433//
434template <class T, int N, class MaskT>
435MOCHI_ANY MOCHI_FORCE_INLINE int StoreSelected(T* ptr, Simd<MaskT, N> condition, Simd<T, N> values);
436
437// Interleave the values of 3 input vectors. Store those (kTupleCount * 3) values to memory.
438// By default (kTupleCount == N).
439//
440// If the input vectors are:
441// {x0, x1, x2, x3}, {y0, y1, y2, y3}, {z0, z1, z2, z3}
442//
443// ...then the order written to memory will be:
444// {x0, y0, z0, x1, y1, z1, x2, y2, z2, x3, y3, z3}
445//
446template <int kTupleCount = -1, class T, int N>
447MOCHI_ANY MOCHI_FORCE_INLINE void StoreTransposed(T* ptr, Simd<T, N> a, Simd<T, N> b, Simd<T, N> c);
448
449// Return result[i] = conditionalMask[i] ? a[i] : b[i]
450// where conditionMask[i] is 0x00000000 (false) or 0xFFFFFFFF (true).
451// Example: Select(a <= b, a, b) is equivalent to Min(a, b)
452template <class T, int N, class MaskT>
454Select(Simd<MaskT, N> conditionMask, Simd<T, N> a, Simd<T, N> b);
455
456// Return a right-shifted integer vector.
457// Signed integer vectors use arithmetic/sign-extending right shift.
458// Unsigned integer vectors are not currently supported.
459template <int kShift, class T, int N>
460MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> ShiftRight(Simd<T, N> a);
461
462// Shuffle elements in a vector. Returns: {a[i0], a[i1], ... }
463template <int x = 0, int y = 1, class T>
464MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, 2> Shuffle(Simd<T, 2> a);
465template <int x = 0, int y = 1, int z = 2, int w = 3, class T>
466MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, 4> Shuffle(Simd<T, 4> a);
467
468// Shuffle elements between two vectors. Returns: { a[x], a[y], b[z], b[w] }
469template <int x, int y, int z, int w, class T>
470MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, 4> Shuffle(Simd<T, 4> a, Simd<T, 4> b);
471
472// Blend returns { (x ? b[0] : a[0]), (y ? b[1] : a[1]) }
473template <int x, int y, class T, int N>
474MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Blend(Simd<T, N> a, Simd<T, N> b);
475
476// Blend returns { (x ? b[0] : a[0]), (y ? b[1] : a[1]), (z ? b[2] : a[2]), (w ? b[3] : a[3]) }
477template <int x, int y, int z, int w, class T, int N>
478MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Blend(Simd<T, N> a, Simd<T, N> b);
479
480// Math Operations:
481template <bool x, bool y, bool z, bool w, class T, int N>
483 Simd<T, N> a); // { x ? -a[0] : a[0], y ? -a[1] : a[1], z ? -a[2] : a[2], w ? -a[3] : a[3] }
484template <class T, int N>
485MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Sqrt(Simd<T, N> a); // sqrt(a)
486template <class T, int N>
487MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> RcpApprox(Simd<T, N> a); // Fast approximation of 1/a
488template <class T, int N>
490 Simd<T, N> a); // Fast approximation of 1/sqrt(a)
491template <class T, int N>
492MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Abs(Simd<T, N> a); // abs(a)
493
494// NOTE: NaN behavior of Min/Max is undefined (platform-dependent and operand-order-dependent).
495template <class T, int N>
496MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Min(Simd<T, N> a, Simd<T, N> b); // min(a, b)
497template <class T, int N>
498MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Max(Simd<T, N> a, Simd<T, N> b); // max(a, b)
499template <class T, int N>
500MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Floor(Simd<T, N> a); // floor(a)
501
502// Round to the nearest integer. On ARM, exact ties are rounded away from zero, just like with
503// std::round. However, on x64 ties are rounded to the nearest even integer. Use only in situations
504// where this discrepancy is not an issue.
505template <class T, int N>
506MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> FastRound(Simd<T, N> a);
507
508// NOT FAST: Horizontal sum of first COUNT elements of a vector. COUNT of -1 means "all".
509template <int COUNT = -1, class T, int N>
510MOCHI_ANY MOCHI_FORCE_INLINE T HSum(Simd<T, N> a);
511
512// NOT FAST: Horizontal product of first COUNT elements of a vector. COUNT of -1 means "all".
513template <int COUNT = -1, class T, int N>
514MOCHI_ANY MOCHI_FORCE_INLINE T HProd(Simd<T, N> a);
515
516// NOT FAST: Horizontal minimum of first COUNT elements of a vector. COUNT of -1 means "all".
517template <int COUNT = -1, class T, int N>
518MOCHI_ANY MOCHI_FORCE_INLINE T HMin(Simd<T, N> a);
519
520// NOT FAST: Horizontal maximum of first COUNT elements of a vector. COUNT of -1 means "all".
521template <int COUNT = -1, class T, int N>
522MOCHI_ANY MOCHI_FORCE_INLINE T HMax(Simd<T, N> a);
523
524// We have a custom implementation of cosine for single-precision floats. It matches the precision
525// of std::cos to within than 6.0e-8. Performance is a little slower than a single call to std::cos,
526// but faster than 2 calls, and much faster than N calls. For double-precision, we fall back on SVML
527// intrinsics (no faster than our implementation), or std::cos if SVML is not available.
528template <class T, int N>
529MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Cos(Simd<T, N> a);
530
531// We have a custom implementation of sine for single-precision floats. Performance and precision is
532// nearly identical to Cos (see notes above).
533template <class T, int N>
534MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Sin(Simd<T, N> a);
535
536// If you need both the sine and the cosine of an angle, then this function can compute both at the
537// same time. It is faster than calling std::sin + std::cos for a single float. In that time, it
538// computes sine and cosine for all N vector components. Precision matches the standard library
539// functions to within 6.0e-8 for single-precision floats. For double-precision, we fall back on
540// SVML or scalar functions (more precision, but slower).
541template <class T, int N>
542MOCHI_ANY MOCHI_FORCE_INLINE std::pair<Simd<T, N>, Simd<T, N>> SinCos(Simd<T, N> a);
543
544// These trig functions are implemented with intrinsics with the SVML extension is available on x64
545// platforms. Otherwise, these functions are quite slow because they simply call the corresponding
546// std library functions N times.
547template <class T, int N>
548MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Tan(Simd<T, N> a);
549template <class T, int N>
550MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Tanh(Simd<T, N> a);
551template <class T, int N>
552MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> ACos(Simd<T, N> a);
553template <class T, int N>
554MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> ASin(Simd<T, N> a);
555template <class T, int N>
556MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> ATan(Simd<T, N> a);
557
558// Math functions
559template <class T, int N>
560MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Exp(Simd<T, N> a);
561template <class T, int N>
562MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Ln(Simd<T, N> a);
563
564// Simd Comparisons:
565// Comparisons are performed for each Simd elements. They return a Simd mask where result[i] has
566// all bits set to 0 to signify "false" or all bits set to 1 to signify "true". This is true for
567// Simd operators <, <=, >, and >= as well as Simd functions with a "V" prefix.
568//
569// Exception:
570// Simd operator== and operator!= return a bool to match typical user expectation. Specifically
571// operator== returns true if all Simd elements are equal, while operator!= returns true if any
572// Simd elements are not equal.
573//
574// Example:
575// auto a = VEqual(Vec4i{1,2,3,4}, Vec4i{1,0,3,0});
576// MOCHI_ASSERT(a == Vec4i{0xFFFFFFFF, 0, 0xFFFFFFFF, 0});
577//
578
579// Return true if ALL of the first COUNT elements are logical true. All N elements must be either
580// all-bits-0 (false) or all-bits-1 (true). COUNT of -1 means "all". Example: AllTrue<2>(a < b)
581// returns true iff ((a[0] < b[0]) && (a[1] < b[1]))
582template <int COUNT = -1, class T, int N>
583MOCHI_ANY MOCHI_FORCE_INLINE bool AllTrue(Simd<T, N> a);
584
585// Return true if ANY of the first COUNT elements is logical true. All N elements must be either
586// all-bits-0 (false) or all-bits-1 (true). COUNT of -1 means "all". Example: AnyTrue<2>(a < b)
587// returns true iff ((a[0] < b[0]) || (a[1] < b[1]))
588template <int COUNT = -1, class T, int N>
589MOCHI_ANY MOCHI_FORCE_INLINE bool AnyTrue(Simd<T, N> a);
590
591// Returns result[i] = (a[i] == b[i]) ? 0xFFFFFFFF : 0x00000000;
592// See note above regarding Simd comparisons.
593template <class V>
595
596// Return true if (a[i] == b[i]) for ALL of the first COUNT elements. Count of -1 means "all".
597template <int COUNT = -1, class T, int N>
598MOCHI_ANY MOCHI_FORCE_INLINE bool Equal(Simd<T, N> a, Simd<T, N> b);
599
600// Returns result[i] = (a[i] != b[i]) ? 0xFFFFFFFF : 0x00000000;
601// See note above regarding Simd comparisons.
602template <class V>
604
605// Return true if (a[i] != b[i]) for ANY of the first COUNT lanes. Count of -1 means "all".
606template <int COUNT = -1, class T, int N>
607MOCHI_ANY MOCHI_FORCE_INLINE bool NotEqual(Simd<T, N> a, Simd<T, N> b);
608
609// Like NearEqual except that it takes a Simd epsilon and returns a Simd bit mask.
610// Returns result[i] = (Abs(a[i] - b[i]) <= epsilon[i]) ? 0xFFFFFFFF : 0x00000000;
611// See note above regarding Simd comparisons.
612template <class V>
613MOCHI_ANY MOCHI_FORCE_INLINE V VNearEqual(V a, V b, V epsilon);
614
615// Return true if (Abs(a[i] - b[i]) <= epsilon) for ALL of the first N lanes.
616// Count of -1 means "all".
617template <int COUNT = -1, class T, int N, class Eps = T>
620 return AllTrue<COUNT>(VNearEqual(a, b, Simd<T, N>{epsilon}));
621}
622
623// Like NearZero except that it takes a Simd epsilon and returns a Simd bit mask.
624// Returns result[i] = (Abs(a[i]) <= epsilon[i]) ? 0xFFFFFFFF : 0x00000000;
625// See note above regarding Simd comparisons.
626template <class V>
627MOCHI_ANY MOCHI_FORCE_INLINE V VNearZero(V a, V epsilon);
628
629// Return true if (Abs(a[i]) <= epsilon) for ALL of the first N lanes. Count of -1 means "all".
630template <int COUNT = -1, class T, int N, class Eps = T>
636
637// Returns result[i] = IsFinite(a[i]) ? 0xFFFFFFFF : 0x00000000;
638template <class T, int N>
639MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> VIsFinite(Simd<T, N> a);
640
641// Return true if all elements of the Simd vector are finite (not +inf, -inf, nor NaN).
642template <class T, int N>
646
647// Return a non-zero integer if component i of the vector is considered "true" (see AnyTrue,
648// AllTrue). Example: "if (IsTrue<i>(a < b))" is like writing "if (a[i] < b[i])".
649template <int i, class T, int N>
650MOCHI_ANY MOCHI_FORCE_INLINE int IsTrue(Simd<T, N> mask);
651
652// Fused Math Operations:
653template <class T, int N>
655MulAdd(Simd<T, N> a, Simd<T, N> b, Simd<T, N> c); // (a * b) + c
656template <class T, int N>
658MulSub(Simd<T, N> a, Simd<T, N> b, Simd<T, N> c); // (a * b) - c
659template <class T, int N>
661NegMulAdd(Simd<T, N> a, Simd<T, N> b, Simd<T, N> c); // -(a * b) + c
662template <class T, int N>
664NegMulSub(Simd<T, N> a, Simd<T, N> b, Simd<T, N> c); // -(a * b) - c
665
666// Dot product of the first COUNT elements. Count of -1 means "all". Broadcasts the result to all
667// elements. Use "Dot" (not "VDot") if you want a single scalar result.
668template <int COUNT = -1, class V>
670
671// Dot product of the first COUNT elements. Count of -1 means "all".
672template <int COUNT = -1, class T, int N>
676
677// Cross Product (3 component)
678template <class T, int N>
679MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Cross3(Simd<T, N> a, Simd<T, N> b);
680
681// Compute the square of the norm using the first COUNT elements of a Simd vector. A count of -1
682// means "all". The result will be broadcasted to a new vector. Use NormSqr instead if you want a
683// scalar result.
684template <int COUNT = -1, class T, int N>
685MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> VNormSqr(Simd<T, N> a);
686
687// Compute the norm using the first COUNT elements of a Simd vector. A count of -1 means "all". The
688// result will be broadcasted to a new vector. Use Norm instead if you want a scalar result.
689template <int COUNT = -1, class T, int N>
690MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> VNorm(Simd<T, N> a);
691
692// Compute the square of the norm using the first COUNT elements of a Simd vector. A count of -1
693// means "all".
694template <int COUNT = -1, class T, int N>
695MOCHI_ANY MOCHI_FORCE_INLINE T NormSqr(Simd<T, N> a);
696
697// Compute the norm using the first COUNT elements of a Simd vector. Count of -1 means "all".
698template <int COUNT = -1, class T, int N>
699MOCHI_ANY MOCHI_FORCE_INLINE T Norm(Simd<T, N> a);
700
701// Normalizes Simd<T, N> elements introducing an epsilon term (to avoid divide-by-zero).
702// Returns (x / (√n₀ + ε), y / (√n₁ + ε), z / (√n₂ + ε), w / (√n₃ + ε))
703template <int COUNT = -1, class T, int N>
704[[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Normalize(Simd<T, N> a);
705
706// Use this overload if you have the square of the norm as a Simd vector.
707template <class T, int N>
708[[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Normalize(Simd<T, N> a, Simd<T, N> normSqr);
709
710// Use this overload if you have the square of the norm as a scalar.
711template <class T, int N>
712[[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE Simd<T, N> Normalize(Simd<T, N> a, T normSqr);
713
714// Builds an arbitrary vector that is orthogonal to the given one.
715template <typename T, int N>
717
718} // namespace superdex
719
720#include "simd/simd_inl.h"
Simd operator-(Simd rhs) const
Simd(NativeType rhs)
Definition simd.h:112
Simd operator&(Simd rhs) const
Simd & operator^=(Simd rhs)
Simd & operator&=(Simd rhs)
Simd & operator/=(T rhs)
bool operator==(Simd rhs) const
Simd operator&&(Simd rhs) const
Simd operator>(Simd rhs) const
Simd operator<<(int shift) const
Simd operator*(Simd rhs) const
Simd & operator-=(T rhs)
Simd operator^(Simd rhs) const
Simd operator-() const
Simd & operator=(Scalar rhs)
Simd & operator-=(Simd rhs)
Simd(Scalar a, Scalar b, Ts... vals)
Simd(Scalar a)
Simd(Simd const &rhs)=default
Simd operator||(Simd rhs) const
Simd operator>=(Simd rhs) const
static constexpr bool kIsComposite
Definition simd.h:98
Simd operator<(Simd rhs) const
Simd & operator*=(T rhs)
Simd & operator|=(Simd rhs)
bool operator!=(Simd rhs) const
Simd & operator*=(Simd rhs)
static constexpr size_t size()
Definition simd.h:122
static constexpr bool kIsEmulated
Definition simd.h:99
Simd & operator=(Simd rhs)
Simd operator|(Simd rhs) const
static constexpr int kSize
Definition simd.h:96
Simd operator+(Simd rhs) const
static constexpr bool kIsSupported
Definition simd.h:97
NotSupported NativeType
Definition simd.h:94
Simd & operator+=(T rhs)
Simd operator~() const
Simd operator/(Simd rhs) const
Simd & operator/=(Simd rhs)
Simd operator<=(Simd rhs) const
Simd & operator+=(Simd rhs)
Scalar operator[](int i) const
#define MOCHI_SIMD_REGISTER_SIZE_BYTES
#define MOCHI_FORCE_INLINE
#define MOCHI_ANY
T Dot(Simd< T, N > a, Simd< T, N > b)
Definition simd.h:673
constexpr T ACos(T a)
Simd< T, N > VIsFinite(Simd< T, N > a)
Definition simd_inl.h:776
Simd< T, 2 > Shuffle(Simd< T, 2 > a)
Definition simd_inl.h:273
constexpr int kSimdDefaultSize
Definition simd.h:33
V LoadIndexed(typename V::Scalar const *ptr, Simd< I, V::kSize > indices)
Definition simd_inl.h:200
constexpr T const & Min(T const &a, T const &b)
Simd< int, 4 > Vec4i
Definition simd.h:204
T NormSqr(Simd< T, N > a)
Definition simd_inl.h:874
constexpr auto Equal(T const &a, T const &b)
Simd< T, N > Cross3(Simd< T, N > a, Simd< T, N > b)
Definition simd_inl.h:857
constexpr T Sin(T a)
T HSum(Simd< T, N > a)
Definition simd_inl.h:377
constexpr To StaticCast(From const &a)
Definition basic_utils.h:79
constexpr T kDefaultNearEqualEpsilon
V VDot(V a, V b)
Definition simd_inl.h:851
T HMin(Simd< T, N > a)
Definition simd_inl.h:389
T Norm(Simd< T, N > a)
Definition simd_inl.h:879
bool AllTrue(T const &a)
Definition basic_utils.h:60
constexpr auto MulAdd(A a, B b, C c)
V SimdBasisVector()
Definition simd_inl.h:151
Simd< real, 16 > Vec16r
Definition simd.h:216
Simd< T, N > Tanh(Simd< T, N > a)
Definition simd_inl.h:709
Simd< int64_t, 2 > Vec2l
Definition simd.h:201
V VEqual(V a, V b)
Definition simd_inl.h:721
V ToSimdDirection(V a)
Definition simd_inl.h:179
Simd< T, N > Set(Simd< T, N > a, T value)
Definition simd_inl.h:313
V Sequence()
Definition simd_inl.h:323
Simd< double, 8 > Vec8d
Definition simd.h:207
constexpr auto NotEqual(T const &a, T const &b)
constexpr T Exp(T a)
Simd< real, 4 > Vec4r
Definition simd.h:206
constexpr T Cos(T a)
V VNearEqual(V a, V b, V epsilon)
Definition simd_inl.h:754
Simd< int, 8 > Vec8i
Definition simd.h:209
V SimdZero()
Definition simd_inl.h:146
T HMax(Simd< T, N > a)
Definition simd_inl.h:395
Simd< T, N > VNormSqr(Simd< T, N > a)
Definition simd_inl.h:862
int IsTrue(Simd< T, N > mask)
Definition simd_inl.h:821
Simd< int64_t, 4 > Vec4l
Definition simd.h:205
V Broadcast(typename V::Scalar a)
Definition simd_inl.h:115
constexpr T Abs(T a)
Definition basic_utils.h:50
V VNotEqual(V a, V b)
Definition simd_inl.h:726
constexpr auto MulSub(A a, B b, C c)
Simd< T, N > Normalize(Simd< T, N > a)
Definition simd_inl.h:884
Simd< real, 8 > Vec8r
Definition simd.h:211
Simd< float, 4 > Vec4f
Definition simd.h:203
Simd< T, N > Blend(Simd< T, N > a, Simd< T, N > b)
Definition simd_inl.h:288
constexpr T Tan(T a)
bool AnyTrue(T const &a)
Definition basic_utils.h:66
Simd< double, 4 > Vec4d
Definition simd.h:202
constexpr T Select(bool condition, T a, T b)
constexpr T Sqrt(T a)
constexpr T ATan(T a)
Simd< T, N > Ln(Simd< T, N > a)
Definition simd_inl.h:699
std::pair< Simd< T, N >, Simd< T, N > > SinCos(Simd< T, N > a)
Definition simd_inl.h:519
constexpr auto NegMulAdd(A a, B b, C c)
T Get(Simd< T, N > v)
Definition simd_inl.h:303
Simd< int64_t, 16 > Vec16l
Definition simd.h:215
constexpr T Floor(T a)
constexpr T ASin(T a)
Simd< double, 16 > Vec16d
Definition simd.h:212
V SimdMask(bool b0, bool b1, MoreBools... bs)
Definition simd_inl.h:137
Simd< int, 16 > Vec16i
Definition simd.h:214
Simd< T, N/2 > GetHalf(Simd< T, N > a)
Definition simd_inl.h:308
T HProd(Simd< T, N > a)
Definition simd_inl.h:383
Simd< T, 4 > OrthogonalVector3(Simd< T, 4 > a)
T Get0(Simd< T, N > v)
Definition simd_inl.h:298
Simd< T, N > VNorm(Simd< T, N > a)
Definition simd_inl.h:869
constexpr T const & Max(T const &a, T const &b)
void LoadTransposed(T const *ptr, Simd< T, N > &out0, Simd< T, N > &out1, Simd< T, N > &out2)
Definition simd_inl.h:207
Simd< T, N > FastRound(Simd< T, N > a)
Definition simd_inl.h:416
void StoreTransposed(T *ptr, Simd< T, N > a, Simd< T, N > b, Simd< T, N > c)
Definition simd_inl.h:245
void Store(T *ptr, Simd< T, N > a)
Definition simd_inl.h:213
Simd< T, N > RcpSqrtApprox(Simd< T, N > a)
Definition simd_inl.h:357
Simd< float, 16 > Vec16f
Definition simd.h:213
constexpr T RcpApprox(T a)
Simd< int64_t, 8 > Vec8l
Definition simd.h:210
bool NearEqual(TransformRT const &a, TransformRT const &b, real epsilon=kDefaultNearEqualEpsilon< real >)
constexpr auto NegMulSub(A a, B b, C c)
V VNearZero(V a, V epsilon)
Definition simd_inl.h:759
int StoreSelected(T *ptr, Simd< MaskT, N > condition, Simd< T, N > values)
Definition simd_inl.h:225
V Load(typename V::Scalar const *ptr)
Definition simd_inl.h:184
Simd< T, N > Neg(Simd< T, N > a)
Definition simd_inl.h:338
Simd< float, 8 > Vec8f
Definition simd.h:208
V ToSimdPoint(V a)
Definition simd_inl.h:174
Simd< double, 2 > Vec2d
Definition simd.h:200
bool IsFinite(TransformRT const &a)
constexpr auto NearZero(T const &a, T epsilon=kDefaultNearEqualEpsilon< T >)
Simd< T, N > ShiftRight(Simd< T, N > a)
Definition simd_inl.h:260