SuperDex Physics C++ API
Loading...
Searching...
No Matches
simd_arch_emulator_inl.h
Go to the documentation of this file.
1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 *
4 * Licensed under the Apache License, Version 2.0 (the "License");
5 * you may not use this file except in compliance with the License.
6 * You may obtain a copy of the License at
7 *
8 * http://www.apache.org/licenses/LICENSE-2.0
9 *
10 * Unless required by applicable law or agreed to in writing, software
11 * distributed under the License is distributed on an "AS IS" BASIS,
12 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13 * See the License for the specific language governing permissions and
14 * limitations under the License.
15 */
16
17#pragma once
18
19#include "../simd.h" // for Intelisense
20
21#if !MOCHI_USE_SIMD
24
25#include <array>
26#include <bit>
27#include <cmath>
28#include <cstring>
29#include <type_traits>
30#include <utility>
31
32namespace superdex {
33
34template <typename F, typename... Args>
35concept IsInvokableWithRaw = requires(F&& f, Args&&... args) { f(args.raw[0]...); };
36
37// NOTE: This specialization could support any size N, but it is currently restricted to multiples
38// of the default size to better match the behavior of native SIMD implementations.
39//
40// NOTE: Half (16-bit float) is not currently supported for emulation. We could add this
41// functionality in the future for compilers with native support (e.g. using __half for CUDA).
42//
43// PERFORMANCE: The artificial size restriction sometimes causes users to round up to the next
44// multiple of the default size, resulting in wasted loads, stores, and ALU operations. If
45// performance of SIMD emulation (e.g. GPU kernel) is important, then consider removing the
46// restriction.
47//
48template <typename T, int N>
49 requires(IsSimdSupportedType<T> && !IsHalf<T> && (N % kSimdDefaultSize<T> == 0))
51 using UIntT = std::conditional_t<sizeof(T) == 4, uint32_t, uint64_t>;
52 static constexpr T kOnesMask = std::bit_cast<T>(~UIntT{0});
53
54#define MOCHI_SIMD_EMULATOR_OP_1(Op, Expr) \
55 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd operator Op(Simd rhs) const { \
56 auto eval = []<size_t... I>(Simd const& lhs, Simd const& rhs, std::index_sequence<I...>) { \
57 return Simd{(Expr)...}; \
58 }; \
59 return eval(*this, rhs, std::make_index_sequence<N>{}); \
60 }
61
62#define MOCHI_SIMD_EMULATOR_OP_2(Op) \
63 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd operator Op(Simd rhs) const { \
64 auto eval = []<size_t... I>(Simd const& lhs, Simd const& rhs, std::index_sequence<I...>) { \
65 return Simd( \
66 std::bit_cast<Scalar>(std::bit_cast<UIntT>(lhs.raw[I]) \
67 Op std::bit_cast<UIntT>(rhs.raw[I]))...); \
68 }; \
69 return eval(*this, rhs, std::make_index_sequence<N>{}); \
70 }
71
72#define MOCHI_SIMD_EMULATOR_HREDUCE_BOOL(Name, Op) \
73 template <int M = kSize> \
74 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr bool Name(Simd a) { \
75 static_assert(M >= 1 && M <= kSize, "Unsupported M"); \
76 auto eval = []<size_t... I>(Simd const& a, std::index_sequence<I...>) { \
77 return ((std::bit_cast<UIntT>(a.raw[I]) != UIntT{0}) Op...); \
78 }; \
79 return eval(a, std::make_index_sequence<M>{}); \
80 }
81
82 template <int M = N, auto F>
83 MOCHI_ANY MOCHI_FORCE_INLINE static T FoldApply(Simd a) {
84 auto eval = []<size_t... I>(Simd const& x, std::index_sequence<I...>) {
85 return F(x.raw[I]...);
86 };
87 return eval(a, std::make_index_sequence<M>{});
88 }
89
90 using ST = std::conditional_t<std::floating_point<T>, T, float>;
91 template <ST (*F)(ST)>
92 MOCHI_ANY MOCHI_FORCE_INLINE static Simd Xapply(
93 std::conditional_t<std::floating_point<T>, Simd, struct DoesNotExist>& x) {
94 if constexpr (std::floating_point<T>) {
95 return Simd(F, x);
96 } else {
97 return Simd{};
98 }
99 }
100
101 template <typename S, S (*F)(S, S)>
102 MOCHI_ANY MOCHI_FORCE_INLINE static Simd XYapply(Simd const& x, Simd const& y) {
103 return Simd(F, x, y);
104 }
105
106 template <typename S, S (*F)(S, S, S)>
107 MOCHI_ANY MOCHI_FORCE_INLINE static Simd XYZapply(Simd const& x, Simd const& y, Simd const& z) {
108 return Simd(F, x, y, z);
109 }
110
111 public:
112 static constexpr int kSize = N;
113 static constexpr bool kIsSupported = true;
114 static constexpr bool kIsComposite = false;
115 static constexpr bool kIsEmulated = true;
116 using Scalar = T;
117 // Warning: Do not use std::array as it is not supported in CUDA device code.
120
121 MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd() = default;
122 MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd(Simd const& rhs) = default;
123 MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd(NativeType const& rhs) : raw(rhs) {}
124 template <class U, MOCHI_REQUIRES_NON_BOOL_SCALAR(U, Scalar)>
126 for (int i = 0; i < N; ++i) {
127 raw[i] = a;
128 }
129 }
131 requires(N > 2 && N % 2 == 0)
132 {
133 for (int i = 0; i < N / 2; ++i) {
134 raw[i] = a.raw[i];
135 }
136 for (int i = 0; i < N / 2; ++i) {
137 raw[(N / 2) + i] = b[i];
138 }
139 }
140
141 template <typename... Args>
142 requires(
143 sizeof...(Args) > 1 && sizeof...(Args) <= kSize && (std::convertible_to<Args, T> && ...))
144 MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd(Args... args)
145 : raw{static_cast<T>(std::forward<Args>(args))...} {}
146
147 template <typename F, typename... Args>
148 MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd(F const& f, Args const&... args)
149 requires IsInvokableWithRaw<F, Args...>
150 {
151 for (size_t i = 0; i < N; ++i) {
152 raw[i] = f(args.raw[i]...);
153 }
154 }
155
156 MOCHI_ANY MOCHI_FORCE_INLINE static constexpr size_t size() {
157 return static_cast<size_t>(kSize);
158 }
159
160 MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd& operator=(Simd const& rhs) = default;
161
162 template <class U, MOCHI_REQUIRES_NON_BOOL_SCALAR(U, Scalar)>
164 for (int i = 0; i < N; ++i) {
165 raw[i] = a;
166 }
167 return *this;
168 }
169
170 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd Zero() {
171 return NativeType{};
172 }
173
174 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE Scalar constexpr operator[](int i) const {
175 MOCHI_ASSERT_VERBOSE(i >= 0 && i < kSize, "Index out of range.");
176 return raw[i];
177 }
178
179 template <int i>
180 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Scalar Get(Simd v) {
181 static_assert(i >= 0 && i < kSize, "Index out of range");
182 return v[i];
183 }
184
185 template <int i>
186 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd<T, N / 2> GetHalf(Simd v)
187 requires(N >= 2 && N % 2 == 0)
188 {
189 static_assert(i == 0 || i == 1);
190 auto eval = [&v]<size_t... I>(std::index_sequence<I...>) {
191 return Simd<T, N / 2>{v.raw[i * (N / 2) + I]...};
192 };
193 return eval(std::make_index_sequence<N / 2>{});
194 }
195
196 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd
197 Set(Simd v, int i, Scalar value) {
198 MOCHI_ASSERT_VERBOSE(i >= 0 && i < kSize, "Index out of range.");
199 auto result = v;
200 result.raw[i] = value;
201 return result;
202 }
203
204 template <int i>
205 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static Simd Set(Simd v, Scalar value) {
206 static_assert(i >= 0 && i < kSize, "Index out of range");
207 return Set(v, i, value);
208 }
209
210 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd AsPoint(Simd a)
211 requires(N == 4)
212 {
213 return {a.raw[0], a.raw[1], a.raw[2], Scalar{1}};
214 }
215
216 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd AsDirection(Simd a)
217 requires(N == 4)
218 {
219 return {a.raw[0], a.raw[1], a.raw[2], Scalar{0}};
220 }
221
222 template <int i>
223 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd SetBasisVector()
224 requires(N == 4)
225 {
226 static_assert(i >= 0 && i <= 3, "Invalid component index");
227 return {
228 i == 0 ? Scalar{1} : Scalar{0},
229 i == 1 ? Scalar{1} : Scalar{0},
230 i == 2 ? Scalar{1} : Scalar{0},
231 i == 3 ? Scalar{1} : Scalar{0}};
232 }
233
234 template <int... x>
235 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd Blend(Simd a, Simd b)
236 requires(sizeof...(x) == N)
237 {
238 static_assert(((x == 0 || x == 1) && ...), "Invalid index");
239 auto eval = []<size_t... I>(Simd const& a, Simd const& b, std::index_sequence<I...>) {
240 return Simd{(x ? b.raw[I] : a.raw[I])...};
241 };
242 return eval(a, b, std::make_index_sequence<N>{});
243 }
244
245 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd Broadcast(Scalar const* p) {
246 return Simd{*p};
247 }
248
249 template <int i>
250 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd Broadcast(Simd v) {
251 static_assert(i >= 0 && i < kSize, "Index out of range");
252 return Simd{v.raw[i]};
253 }
254
255 template <int x = 0, int y = 1>
256 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd Shuffle(Simd a)
257 requires(N == 2)
258 {
259 static_assert(x >= 0 && x < 2 && y >= 0 && y < 2, "Invalid index");
260 return Simd{a.raw[x], a.raw[y]};
261 }
262
263 template <int x = 0, int y = 1, int z = 2, int w = 3>
264 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd Shuffle(Simd a, Simd b)
265 requires(N == 4)
266 {
267 static_assert(
268 x >= 0 && x < 4 && y >= 0 && y < 4 && z >= 0 && z < 4 && w >= 0 && w < 4, "Invalid index");
269 return Simd{a.raw[x], a.raw[y], b.raw[z], b.raw[w]};
270 }
271
272 template <int x = 0, int y = 1, int z = 2, int w = 3>
273 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd Shuffle(Simd v)
274 requires(N == 4)
275 {
276 return Shuffle<x, y, z, w>(v, v);
277 }
278
279 template <int M = kSize>
280 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static Simd Load([[maybe_unused]] Scalar const* ptr) {
281 static_assert(M >= 0 && M <= kSize);
282 if constexpr (M == 0) {
283 return Zero();
284 } else {
285 Simd result;
286 std::memcpy(result.raw.data(), ptr, M * sizeof(Scalar));
287 if constexpr (M < kSize) {
288 std::memset(result.raw.data() + M, 0, (kSize - M) * sizeof(Scalar));
289 }
290 return result;
291 }
292 }
293
294 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static Simd Load(Scalar const* ptr, int n) {
295 MOCHI_ASSERT_VERBOSE(n >= 0 && n <= kSize, "Invalid load size.");
296 Simd result;
297 for (int i = 0; i < n; ++i) {
298 result.raw[i] = ptr[i];
299 }
300 for (int i = n; i < kSize; ++i) {
301 result.raw[i] = Scalar{0};
302 }
303 return result;
304 }
305
306 template <typename IntT>
308 Scalar const* ptr,
309 Simd<IntT, N> const& indices)
310 requires(std::integral<IntT>)
311 {
312 auto eval = []<size_t... I>(
313 Scalar const* ptr, Simd<IntT, N> const& indices, std::index_sequence<I...>) {
314 return Simd{ptr[indices.raw[I]]...};
315 };
316 return eval(ptr, indices, std::make_index_sequence<N>{});
317 }
318
319 template <int kTupleCount = kSize>
320 MOCHI_ANY MOCHI_FORCE_INLINE static void
321 LoadTransposed(Scalar const* ptr, Simd& out0, Simd& out1, Simd& out2) {
322 static_assert(kTupleCount >= 1 && kTupleCount <= kSize, "Invalid kTupleCount");
323 constexpr Scalar zero{};
324 constexpr int kTupleSize = 3;
325 auto eval = [ptr, zero, &out0, &out1, &out2]<size_t... I>(std::index_sequence<I...>) {
326 out0 = Simd{(I < kTupleCount ? ptr[I * kTupleSize + 0] : zero)...};
327 out1 = Simd{(I < kTupleCount ? ptr[I * kTupleSize + 1] : zero)...};
328 out2 = Simd{(I < kTupleCount ? ptr[I * kTupleSize + 2] : zero)...};
329 };
330 eval(std::make_index_sequence<kSize>{});
331 }
332
333 template <int M = kSize>
335 [[maybe_unused]] Scalar* ptr,
336 [[maybe_unused]] Simd v) {
337 static_assert(M >= 0 && M <= kSize, "Unsupported M");
338 if constexpr (M != 0) {
339 std::memcpy(ptr, v.raw.data(), M * sizeof(Scalar));
340 }
341 }
342
343 MOCHI_ANY MOCHI_FORCE_INLINE static void Store(Scalar* ptr, Simd v, int n) {
344 MOCHI_ASSERT_VERBOSE(n >= 0 && n <= kSize, "Invalid store size.");
345 for (int i = 0; i < n; ++i) {
346 ptr[i] = v.raw[i];
347 }
348 }
349
350 MOCHI_ANY MOCHI_FORCE_INLINE static int StoreSelected(Scalar* ptr, Simd condition, Simd values) {
351 int count = 0;
352 auto eval = [&count]<size_t... I>(
353 Scalar* ptr,
354 Simd<Scalar, N> const& condition,
355 Simd<Scalar, N> const& values,
356 std::index_sequence<I...>) {
357 ((ptr[count] = values.raw[I], count += static_cast<int>(!!condition.raw[I])), ...);
358 };
359 eval(ptr, condition, values, std::make_index_sequence<N>{});
360 return count;
361 }
362
363 template <int kTupleCount = kSize>
364 MOCHI_ANY MOCHI_FORCE_INLINE static void
365 StoreTransposed(Scalar* ptr, Simd out0, Simd out1, Simd out2) {
366 static_assert(kTupleCount >= 1 && kTupleCount <= kSize, "Invalid kTupleCount");
367 auto eval = [ptr, &out0, &out1, &out2]<size_t... I>(std::index_sequence<I...>) {
368 ((I < kTupleCount ? (void)(ptr[I * 3 + 0] = out0.raw[I],
369 ptr[I * 3 + 1] = out1.raw[I],
370 ptr[I * 3 + 2] = out2.raw[I])
371 : (void)0),
372 ...);
373 };
374 eval(std::make_index_sequence<kSize>{});
375 }
376
377 template <int M = kSize>
378 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Scalar HMin(Simd a) {
379 static_assert(M >= 1 && M <= kSize, "Unsupported M");
380 return FoldApply<M, [](auto... v) { return superdex::Min(v...); }>(a);
381 }
382
383 template <int M = kSize>
384 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Scalar HMax(Simd a) {
385 static_assert(M >= 1 && M <= kSize, "Unsupported M");
386 return FoldApply<M, [](auto... v) { return superdex::Max(v...); }>(a);
387 }
388
389 template <int M = kSize>
390 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Scalar HSum(Simd a) {
391 static_assert(M >= 1 && M <= kSize, "Unsupported M");
392 return FoldApply<M, [](auto... v) { return (v + ...); }>(a);
393 }
394
395 template <int M = kSize>
396 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Scalar HProd(Simd a) {
397 static_assert(M >= 1 && M <= kSize, "Unsupported M");
398 return FoldApply<M, [](auto... v) { return (v * ...); }>(a);
399 }
400
401 template <int M = kSize>
402 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd Dot(Simd a, Simd b) {
403 static_assert(M > 0 && M <= kSize, "Unsupported M");
404 auto eval = []<size_t... I>(Simd const& a, Simd const& b, std::index_sequence<I...>) {
405 return Simd{((a.raw[I] * b.raw[I]) + ...)};
406 };
407 return eval(a, b, std::make_index_sequence<M>{});
408 }
409
412
413 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd
414 Select(Simd mask, Simd a, Simd b) {
415 auto eval = []<size_t... I>(
416 auto const& mask, Simd const& a, Simd const& b, std::index_sequence<I...>) {
417 return Simd{(std::bit_cast<UIntT>(mask.raw[I]) != UIntT{0} ? a.raw[I] : b.raw[I])...};
418 };
419 return eval(mask, a, b, std::make_index_sequence<N>{});
420 }
421
422 // Unary functions.
423 // TODO: SinCosImpl in simd_inl.h might be faster than this Cos() and Sin().
424 static constexpr auto Sqrt = Xapply<std::sqrt>;
425 static constexpr auto Abs = Xapply<std::abs>;
426 static constexpr auto Floor = Xapply<std::floor>;
427 static constexpr auto FastRound = Xapply<std::round>;
428 static constexpr auto RcpApprox = Xapply<[](ST a) { return ST{1} / a; }>;
429 static constexpr auto RcpSqrtApprox = Xapply<[](ST a) { return ST{1} / std::sqrt(a); }>;
430 static constexpr auto Cos = Xapply<std::cos>;
431 static constexpr auto Sin = Xapply<std::sin>;
432 static constexpr auto Tan = Xapply<std::tan>;
433 static constexpr auto ACos = Xapply<std::acos>;
434 static constexpr auto ASin = Xapply<std::asin>;
435 static constexpr auto ATan = Xapply<std::atan>;
436 static constexpr auto Exp = Xapply<std::exp>;
437 static constexpr auto Ln = Xapply<std::log>;
438 static constexpr auto Tanh = Xapply<std::tanh>;
439
440 // Binary functions.
441 static constexpr auto Min = XYapply<T const&, superdex::Min<T>>;
442 static constexpr auto Max = XYapply<T const&, superdex::Max<T>>;
443
444 MOCHI_ANY MOCHI_FORCE_INLINE static Simd Equal(Simd const& x, Simd const& y) {
445 return Simd([](T a, T b) { return a == b ? Scalar{kOnesMask} : Scalar{0}; }, x, y);
446 }
447
448 MOCHI_ANY MOCHI_FORCE_INLINE static Simd NotEqual(Simd const& x, Simd const& y) {
449 return Simd([](T a, T b) { return a != b ? Scalar{kOnesMask} : Scalar{0}; }, x, y);
450 }
451
452 // Ternary functions.
453 static constexpr auto MulAdd = XYZapply<T, superdex::MulAdd<T, T, T>>;
454 static constexpr auto MulSub = XYZapply<T, superdex::MulSub<T, T, T>>;
455 static constexpr auto NegMulAdd = XYZapply<T, superdex::NegMulAdd<T, T, T>>;
456 static constexpr auto NegMulSub = XYZapply<T, superdex::NegMulSub<T, T, T>>;
457
458 // Unary operators.
459 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd operator-() const {
460 return Simd([](T a) { return -a; }, *this);
461 }
462
463 // This one uses a function that may not work on CUDA (std::bit_cast)
464 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd operator~() const {
465 return Simd([](T a) { return std::bit_cast<Scalar>(~std::bit_cast<UIntT>(a)); }, *this);
466 }
467
468 // Binary operators.
469 MOCHI_SIMD_EMULATOR_OP_1(<, lhs.raw[I] < rhs.raw[I] ? kOnesMask : Scalar{0});
470 MOCHI_SIMD_EMULATOR_OP_1(>, lhs.raw[I] > rhs.raw[I] ? kOnesMask : Scalar{0});
471 MOCHI_SIMD_EMULATOR_OP_1(<=, lhs.raw[I] <= rhs.raw[I] ? kOnesMask : Scalar{0});
472 MOCHI_SIMD_EMULATOR_OP_1(>=, lhs.raw[I] >= rhs.raw[I] ? kOnesMask : Scalar{0});
473 MOCHI_SIMD_EMULATOR_OP_1(+, lhs.raw[I] + rhs.raw[I]);
474 MOCHI_SIMD_EMULATOR_OP_1(-, lhs.raw[I] - rhs.raw[I]);
475 MOCHI_SIMD_EMULATOR_OP_1(*, lhs.raw[I] * rhs.raw[I]);
476 MOCHI_SIMD_EMULATOR_OP_1(/, lhs.raw[I] / rhs.raw[I]);
482
483 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE constexpr bool operator==(Simd rhs) const {
484 auto eval = []<size_t... I>(Simd const& lhs, Simd const& rhs, std::index_sequence<I...>) {
485 return ((lhs.raw[I] == rhs.raw[I]) && ...);
486 };
487 return eval(*this, rhs, std::make_index_sequence<N>{});
488 }
489
490 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE constexpr bool operator!=(Simd rhs) const {
491 return !(*this == rhs);
492 }
493
494 template <int kShift>
495 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr Simd ShiftRight(Simd a) {
496 static_assert(kShift >= 0 && kShift < (8 * sizeof(Scalar)), "Shift amount out-of-range");
497 if constexpr (std::is_integral_v<Scalar>) {
498 static_assert(std::is_signed_v<Scalar>, "Unsigned integer ShiftRight is not supported.");
499 auto eval = []<size_t... I>(Simd const& v, std::index_sequence<I...>) {
500 return Simd{static_cast<Scalar>(v.raw[I] >> kShift)...};
501 };
502 return eval(a, std::make_index_sequence<N>{});
503 } else {
504 return a >> kShift;
505 }
506 }
507
508#undef MOCHI_SIMD_EMULATOR_OP_1
509#undef MOCHI_SIMD_EMULATOR_OP_2
510#undef MOCHI_SIMD_EMULATOR_HREDUCE_BOOL
511};
512
513template <class To, class FromT, int FromN>
514MOCHI_ANY MOCHI_FORCE_INLINE To ReinterpretCast(Simd<FromT, FromN> const& in)
515 requires(IsSimd<To> && Simd<FromT, FromN>::kIsSupported)
516{
517 if constexpr (std::is_same_v<std::decay_t<To>, Simd<FromT, FromN>>) {
518 return in;
519 } else {
520 static_assert(sizeof(To) == sizeof(FromT) * FromN, "Size mismatch");
521 To out;
522 std::memcpy(&out.raw, &in.raw, sizeof(in.raw));
523 return out;
524 }
525}
526
527template <class To, class FromT, int FromN>
530{
531 if constexpr (std::is_same_v<std::decay_t<To>, Simd<FromT, FromN>>) {
532 return in;
533 } else {
534 static_assert(To::kSize == FromN, "Size mismatch");
535 auto eval = [&in]<size_t... I>(std::index_sequence<I...>) {
536 return To{static_cast<typename To::Scalar>(in.raw[I])...};
537 };
538 return eval(std::make_index_sequence<FromN>{});
539 }
540}
541
542} // namespace superdex
543#endif
Scalar constexpr operator[](int i) const
static constexpr Simd Broadcast(Simd v)
static constexpr Scalar HMax(Simd a)
static constexpr Simd Blend(Simd a, Simd b)
static constexpr auto NegMulSub
static constexpr bool AnyTrue(Simd a)
static int StoreSelected(Scalar *ptr, Simd condition, Simd values)
static constexpr Simd Broadcast(Scalar const *p)
static constexpr Simd Set(Simd v, int i, Scalar value)
constexpr Simd & operator=(U a)
static void Store(Scalar *ptr, Simd v)
constexpr bool operator!=(Simd rhs) const
static void LoadTransposed(Scalar const *ptr, Simd &out0, Simd &out1, Simd &out2)
static constexpr Simd Dot(Simd a, Simd b)
static constexpr Scalar HMin(Simd a)
static constexpr bool AllTrue(Simd a)
static constexpr Simd Zero()
static constexpr auto Floor
static Simd NotEqual(Simd const &x, Simd const &y)
constexpr Simd operator~() const
constexpr Simd(Simd const &rhs)=default
static Simd Load(Scalar const *ptr, int n)
static constexpr Scalar Get(Simd v)
static constexpr bool kIsComposite
static constexpr Simd Select(Simd mask, Simd a, Simd b)
constexpr bool operator==(Simd rhs) const
static constexpr Simd AsPoint(Simd a)
static constexpr auto RcpApprox
static constexpr Simd Shuffle(Simd a, Simd b)
constexpr Simd(Simd< T, N/2 > a, Simd< T, N/2 > b)
static constexpr Simd SetBasisVector()
static constexpr Scalar HProd(Simd a)
constexpr Simd & operator=(Simd const &rhs)=default
static constexpr Simd AsDirection(Simd a)
static void Store(Scalar *ptr, Simd v, int n)
static constexpr bool kIsEmulated
static void StoreTransposed(Scalar *ptr, Simd out0, Simd out1, Simd out2)
static constexpr size_t size()
static constexpr Simd ShiftRight(Simd a)
static constexpr auto MulSub
static constexpr bool kIsSupported
static Simd Load(Scalar const *ptr)
constexpr Simd operator-() const
static constexpr auto RcpSqrtApprox
static constexpr Simd< T, N/2 > GetHalf(Simd v)
constexpr Simd()=default
static Simd Equal(Simd const &x, Simd const &y)
static constexpr Scalar HSum(Simd a)
static constexpr auto FastRound
static Simd LoadIndexed(Scalar const *ptr, Simd< IntT, N > const &indices)
static constexpr Simd Shuffle(Simd v)
static constexpr Simd Shuffle(Simd a)
static constexpr auto MulAdd
constexpr Simd(F const &f, Args const &... args)
static constexpr auto NegMulAdd
constexpr Simd(NativeType const &rhs)
static Simd Set(Simd v, Scalar value)
NativeType raw
Definition simd.h:174
static constexpr bool kIsSupported
Definition simd.h:97
#define MOCHI_ASSERT_VERBOSE(condition_without_side_effects,...)
Definition debug.h:102
#define MOCHI_FORCE_INLINE
#define MOCHI_ANY
Simd< T, 2 > Shuffle(Simd< T, 2 > a)
Definition simd_inl.h:273
constexpr int kSimdDefaultSize
Definition simd.h:33
constexpr T const & Min(T const &a, T const &b)
constexpr To StaticCast(From const &a)
Definition basic_utils.h:79
Simd< T, N > Set(Simd< T, N > a, T value)
Definition simd_inl.h:313
constexpr T const & Max(T const &a, T const &b)
#define MOCHI_SIMD_EMULATOR_OP_2(Op)
#define MOCHI_SIMD_EMULATOR_HREDUCE_BOOL(Name, Op)
#define MOCHI_SIMD_EMULATOR_OP_1(Op, Expr)