34template <
typename F,
typename... Args>
48template <
typename T,
int N>
51 using UIntT = std::conditional_t<
sizeof(T) == 4, uint32_t, uint64_t>;
52 static constexpr T kOnesMask = std::bit_cast<T>(~UIntT{0});
54#define MOCHI_SIMD_EMULATOR_OP_1(Op, Expr) \
55 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd operator Op(Simd rhs) const { \
56 auto eval = []<size_t... I>(Simd const& lhs, Simd const& rhs, std::index_sequence<I...>) { \
57 return Simd{(Expr)...}; \
59 return eval(*this, rhs, std::make_index_sequence<N>{}); \
62#define MOCHI_SIMD_EMULATOR_OP_2(Op) \
63 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE constexpr Simd operator Op(Simd rhs) const { \
64 auto eval = []<size_t... I>(Simd const& lhs, Simd const& rhs, std::index_sequence<I...>) { \
66 std::bit_cast<Scalar>(std::bit_cast<UIntT>(lhs.raw[I]) \
67 Op std::bit_cast<UIntT>(rhs.raw[I]))...); \
69 return eval(*this, rhs, std::make_index_sequence<N>{}); \
72#define MOCHI_SIMD_EMULATOR_HREDUCE_BOOL(Name, Op) \
73 template <int M = kSize> \
74 [[nodiscard]] MOCHI_ANY MOCHI_FORCE_INLINE static constexpr bool Name(Simd a) { \
75 static_assert(M >= 1 && M <= kSize, "Unsupported M"); \
76 auto eval = []<size_t... I>(Simd const& a, std::index_sequence<I...>) { \
77 return ((std::bit_cast<UIntT>(a.raw[I]) != UIntT{0}) Op...); \
79 return eval(a, std::make_index_sequence<M>{}); \
82 template <
int M = N, auto F>
84 auto eval = []<
size_t... I>(Simd
const& x, std::index_sequence<I...>) {
85 return F(x.raw[I]...);
87 return eval(a, std::make_index_sequence<M>{});
90 using ST = std::conditional_t<std::floating_point<T>, T,
float>;
91 template <ST (*F)(ST)>
93 std::conditional_t<std::floating_point<T>,
Simd,
struct DoesNotExist>& x) {
94 if constexpr (std::floating_point<T>) {
101 template <
typename S, S (*F)(S, S)>
103 return Simd(F, x, y);
106 template <
typename S, S (*F)(S, S, S)>
108 return Simd(F, x, y, z);
124 template <
class U, MOCHI_REQUIRES_NON_BOOL_SCALAR(U, Scalar)>
126 for (
int i = 0; i < N; ++i) {
131 requires(N > 2 && N % 2 == 0)
133 for (
int i = 0; i < N / 2; ++i) {
136 for (
int i = 0; i < N / 2; ++i) {
137 raw[(N / 2) + i] = b[i];
141 template <
typename... Args>
143 sizeof...(Args) > 1 &&
sizeof...(Args) <= kSize && (std::convertible_to<Args, T> && ...))
145 :
raw{
static_cast<T
>(std::forward<Args>(args))...} {}
147 template <
typename F,
typename... Args>
151 for (
size_t i = 0; i < N; ++i) {
152 raw[i] = f(args.raw[i]...);
157 return static_cast<size_t>(
kSize);
162 template <
class U, MOCHI_REQUIRES_NON_BOOL_SCALAR(U, Scalar)>
164 for (
int i = 0; i < N; ++i) {
181 static_assert(i >= 0 && i <
kSize,
"Index out of range");
187 requires(N >= 2 && N % 2 == 0)
189 static_assert(i == 0 || i == 1);
190 auto eval = [&v]<
size_t... I>(std::index_sequence<I...>) {
191 return Simd<T, N / 2>{v.raw[i * (N / 2) + I]...};
193 return eval(std::make_index_sequence<N / 2>{});
200 result.
raw[i] = value;
206 static_assert(i >= 0 && i <
kSize,
"Index out of range");
207 return Set(v, i, value);
213 return {a.raw[0], a.raw[1], a.raw[2],
Scalar{1}};
219 return {a.raw[0], a.raw[1], a.raw[2],
Scalar{0}};
226 static_assert(i >= 0 && i <= 3,
"Invalid component index");
236 requires(
sizeof...(x) == N)
238 static_assert(((x == 0 || x == 1) && ...),
"Invalid index");
239 auto eval = []<
size_t... I>(
Simd const& a,
Simd const& b, std::index_sequence<I...>) {
240 return Simd{(x ? b.raw[I] : a.raw[I])...};
242 return eval(a, b, std::make_index_sequence<N>{});
251 static_assert(i >= 0 && i <
kSize,
"Index out of range");
255 template <
int x = 0,
int y = 1>
259 static_assert(x >= 0 && x < 2 && y >= 0 && y < 2,
"Invalid index");
260 return Simd{a.raw[x], a.raw[y]};
263 template <
int x = 0,
int y = 1,
int z = 2,
int w = 3>
268 x >= 0 && x < 4 && y >= 0 && y < 4 && z >= 0 && z < 4 && w >= 0 && w < 4,
"Invalid index");
269 return Simd{a.raw[x], a.raw[y], b.raw[z], b.raw[w]};
272 template <
int x = 0,
int y = 1,
int z = 2,
int w = 3>
279 template <
int M = kSize>
281 static_assert(M >= 0 && M <=
kSize);
282 if constexpr (M == 0) {
286 std::memcpy(result.
raw.data(), ptr, M *
sizeof(
Scalar));
287 if constexpr (M <
kSize) {
288 std::memset(result.
raw.data() + M, 0, (
kSize - M) *
sizeof(
Scalar));
297 for (
int i = 0; i < n; ++i) {
298 result.
raw[i] = ptr[i];
300 for (
int i = n; i <
kSize; ++i) {
306 template <
typename IntT>
310 requires(std::integral<IntT>)
312 auto eval = []<
size_t... I>(
314 return Simd{ptr[indices.raw[I]]...};
316 return eval(ptr, indices, std::make_index_sequence<N>{});
319 template <
int kTupleCount = kSize>
322 static_assert(kTupleCount >= 1 && kTupleCount <=
kSize,
"Invalid kTupleCount");
324 constexpr int kTupleSize = 3;
325 auto eval = [ptr, zero, &out0, &out1, &out2]<
size_t... I>(std::index_sequence<I...>) {
326 out0 =
Simd{(I < kTupleCount ? ptr[I * kTupleSize + 0] : zero)...};
327 out1 =
Simd{(I < kTupleCount ? ptr[I * kTupleSize + 1] : zero)...};
328 out2 =
Simd{(I < kTupleCount ? ptr[I * kTupleSize + 2] : zero)...};
330 eval(std::make_index_sequence<kSize>{});
333 template <
int M = kSize>
335 [[maybe_unused]]
Scalar* ptr,
336 [[maybe_unused]]
Simd v) {
337 static_assert(M >= 0 && M <=
kSize,
"Unsupported M");
338 if constexpr (M != 0) {
339 std::memcpy(ptr, v.raw.data(), M *
sizeof(
Scalar));
345 for (
int i = 0; i < n; ++i) {
352 auto eval = [&count]<
size_t... I>(
356 std::index_sequence<I...>) {
357 ((ptr[count] = values.
raw[I], count +=
static_cast<int>(!!condition.
raw[I])), ...);
359 eval(ptr, condition, values, std::make_index_sequence<N>{});
363 template <
int kTupleCount = kSize>
366 static_assert(kTupleCount >= 1 && kTupleCount <=
kSize,
"Invalid kTupleCount");
367 auto eval = [ptr, &out0, &out1, &out2]<
size_t... I>(std::index_sequence<I...>) {
368 ((I < kTupleCount ? (void)(ptr[I * 3 + 0] = out0.
raw[I],
369 ptr[I * 3 + 1] = out1.
raw[I],
370 ptr[I * 3 + 2] = out2.
raw[I])
374 eval(std::make_index_sequence<kSize>{});
377 template <
int M = kSize>
379 static_assert(M >= 1 && M <=
kSize,
"Unsupported M");
380 return FoldApply<M, [](
auto... v) {
return superdex::Min(v...); }>(a);
383 template <
int M = kSize>
385 static_assert(M >= 1 && M <=
kSize,
"Unsupported M");
386 return FoldApply<M, [](
auto... v) {
return superdex::Max(v...); }>(a);
389 template <
int M = kSize>
391 static_assert(M >= 1 && M <=
kSize,
"Unsupported M");
392 return FoldApply<M, [](
auto... v) {
return (v + ...); }>(a);
395 template <
int M = kSize>
397 static_assert(M >= 1 && M <=
kSize,
"Unsupported M");
398 return FoldApply<M, [](
auto... v) {
return (v * ...); }>(a);
401 template <
int M = kSize>
403 static_assert(M > 0 && M <=
kSize,
"Unsupported M");
404 auto eval = []<
size_t... I>(
Simd const& a,
Simd const& b, std::index_sequence<I...>) {
407 return eval(a, b, std::make_index_sequence<M>{});
415 auto eval = []<
size_t... I>(
416 auto const& mask,
Simd const& a,
Simd const& b, std::index_sequence<I...>) {
417 return Simd{(std::bit_cast<UIntT>(mask.raw[I]) != UIntT{0} ? a.raw[I] : b.raw[I])...};
419 return eval(mask, a, b, std::make_index_sequence<N>{});
424 static constexpr auto Sqrt = Xapply<std::sqrt>;
425 static constexpr auto Abs = Xapply<std::abs>;
426 static constexpr auto Floor = Xapply<std::floor>;
428 static constexpr auto RcpApprox = Xapply<[](ST a) {
return ST{1} / a; }>;
429 static constexpr auto RcpSqrtApprox = Xapply<[](ST a) {
return ST{1} / std::sqrt(a); }>;
430 static constexpr auto Cos = Xapply<std::cos>;
431 static constexpr auto Sin = Xapply<std::sin>;
432 static constexpr auto Tan = Xapply<std::tan>;
433 static constexpr auto ACos = Xapply<std::acos>;
434 static constexpr auto ASin = Xapply<std::asin>;
435 static constexpr auto ATan = Xapply<std::atan>;
436 static constexpr auto Exp = Xapply<std::exp>;
437 static constexpr auto Ln = Xapply<std::log>;
438 static constexpr auto Tanh = Xapply<std::tanh>;
441 static constexpr auto Min = XYapply<T const&, superdex::Min<T>>;
442 static constexpr auto Max = XYapply<T const&, superdex::Max<T>>;
445 return Simd([](T a, T b) {
return a == b ?
Scalar{kOnesMask} :
Scalar{0}; }, x, y);
449 return Simd([](T a, T b) {
return a != b ?
Scalar{kOnesMask} :
Scalar{0}; }, x, y);
453 static constexpr auto MulAdd = XYZapply<T, superdex::MulAdd<T, T, T>>;
454 static constexpr auto MulSub = XYZapply<T, superdex::MulSub<T, T, T>>;
455 static constexpr auto NegMulAdd = XYZapply<T, superdex::NegMulAdd<T, T, T>>;
456 static constexpr auto NegMulSub = XYZapply<T, superdex::NegMulSub<T, T, T>>;
460 return Simd([](T a) {
return -a; }, *
this);
465 return Simd([](T a) {
return std::bit_cast<Scalar>(~std::bit_cast<UIntT>(a)); }, *
this);
484 auto eval = []<
size_t... I>(
Simd const& lhs,
Simd const& rhs, std::index_sequence<I...>) {
485 return ((lhs.raw[I] == rhs.
raw[I]) && ...);
487 return eval(*
this, rhs, std::make_index_sequence<N>{});
491 return !(*
this == rhs);
494 template <
int kShift>
496 static_assert(kShift >= 0 && kShift < (8 *
sizeof(
Scalar)),
"Shift amount out-of-range");
497 if constexpr (std::is_integral_v<Scalar>) {
498 static_assert(std::is_signed_v<Scalar>,
"Unsigned integer ShiftRight is not supported.");
499 auto eval = []<
size_t... I>(
Simd const& v, std::index_sequence<I...>) {
500 return Simd{
static_cast<Scalar>(v.raw[I] >> kShift)...};
502 return eval(a, std::make_index_sequence<N>{});
508#undef MOCHI_SIMD_EMULATOR_OP_1
509#undef MOCHI_SIMD_EMULATOR_OP_2
510#undef MOCHI_SIMD_EMULATOR_HREDUCE_BOOL
513template <
class To,
class FromT,
int FromN>
517 if constexpr (std::is_same_v<std::decay_t<To>, Simd<FromT, FromN>>) {
520 static_assert(
sizeof(To) ==
sizeof(FromT) * FromN,
"Size mismatch");
522 std::memcpy(&out.raw, &in.raw,
sizeof(in.raw));
527template <
class To,
class FromT,
int FromN>
534 static_assert(To::kSize == FromN,
"Size mismatch");
535 auto eval = [&in]<
size_t... I>(std::index_sequence<I...>) {
536 return To{
static_cast<typename To::Scalar
>(in.raw[I])...};
538 return eval(std::make_index_sequence<FromN>{});
Scalar constexpr operator[](int i) const
static constexpr Simd Broadcast(Simd v)
static constexpr auto ASin
static constexpr Scalar HMax(Simd a)
static constexpr Simd Blend(Simd a, Simd b)
static constexpr auto NegMulSub
static constexpr bool AnyTrue(Simd a)
static int StoreSelected(Scalar *ptr, Simd condition, Simd values)
static constexpr Simd Broadcast(Scalar const *p)
static constexpr Simd Set(Simd v, int i, Scalar value)
constexpr Simd & operator=(U a)
static void Store(Scalar *ptr, Simd v)
constexpr bool operator!=(Simd rhs) const
Array< Scalar, N > NativeType
static void LoadTransposed(Scalar const *ptr, Simd &out0, Simd &out1, Simd &out2)
static constexpr Simd Dot(Simd a, Simd b)
static constexpr Scalar HMin(Simd a)
static constexpr bool AllTrue(Simd a)
static constexpr Simd Zero()
static constexpr auto Min
static constexpr auto Tan
static constexpr auto Sqrt
static constexpr auto Floor
static Simd NotEqual(Simd const &x, Simd const &y)
constexpr Simd operator~() const
constexpr Simd(Simd const &rhs)=default
static constexpr auto Sin
static Simd Load(Scalar const *ptr, int n)
static constexpr Scalar Get(Simd v)
static constexpr bool kIsComposite
static constexpr auto Max
static constexpr Simd Select(Simd mask, Simd a, Simd b)
constexpr bool operator==(Simd rhs) const
static constexpr Simd AsPoint(Simd a)
static constexpr auto RcpApprox
static constexpr Simd Shuffle(Simd a, Simd b)
constexpr Simd(Simd< T, N/2 > a, Simd< T, N/2 > b)
static constexpr auto Cos
static constexpr auto Tanh
static constexpr Simd SetBasisVector()
static constexpr Scalar HProd(Simd a)
constexpr Simd & operator=(Simd const &rhs)=default
static constexpr Simd AsDirection(Simd a)
static constexpr auto Abs
static void Store(Scalar *ptr, Simd v, int n)
static constexpr bool kIsEmulated
static constexpr auto ATan
static void StoreTransposed(Scalar *ptr, Simd out0, Simd out1, Simd out2)
static constexpr size_t size()
static constexpr Simd ShiftRight(Simd a)
static constexpr auto MulSub
static constexpr bool kIsSupported
static Simd Load(Scalar const *ptr)
static constexpr auto Exp
constexpr Simd operator-() const
static constexpr auto RcpSqrtApprox
static constexpr Simd< T, N/2 > GetHalf(Simd v)
static Simd Equal(Simd const &x, Simd const &y)
static constexpr Scalar HSum(Simd a)
static constexpr auto FastRound
static Simd LoadIndexed(Scalar const *ptr, Simd< IntT, N > const &indices)
static constexpr Simd Shuffle(Simd v)
static constexpr Simd Shuffle(Simd a)
static constexpr auto MulAdd
static constexpr auto ACos
constexpr Simd(F const &f, Args const &... args)
static constexpr auto NegMulAdd
static constexpr int kSize
constexpr Simd(NativeType const &rhs)
static Simd Set(Simd v, Scalar value)
static constexpr bool kIsSupported
#define MOCHI_ASSERT_VERBOSE(condition_without_side_effects,...)
Simd< T, 2 > Shuffle(Simd< T, 2 > a)
constexpr int kSimdDefaultSize
constexpr T const & Min(T const &a, T const &b)
constexpr To StaticCast(From const &a)
Simd< T, N > Set(Simd< T, N > a, T value)
constexpr T const & Max(T const &a, T const &b)
#define MOCHI_SIMD_EMULATOR_OP_2(Op)
#define MOCHI_SIMD_EMULATOR_HREDUCE_BOOL(Name, Op)
#define MOCHI_SIMD_EMULATOR_OP_1(Op, Expr)