SuperDex Physics C++ API
Loading...
Searching...
No Matches
x64_simd_float_4_inl.h
Go to the documentation of this file.
1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 *
4 * Licensed under the Apache License, Version 2.0 (the "License");
5 * you may not use this file except in compliance with the License.
6 * You may obtain a copy of the License at
7 *
8 * http://www.apache.org/licenses/LICENSE-2.0
9 *
10 * Unless required by applicable law or agreed to in writing, software
11 * distributed under the License is distributed on an "AS IS" BASIS,
12 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13 * See the License for the specific language governing permissions and
14 * limitations under the License.
15 */
16
17#pragma once
18
19#include "x64_simd_inl.h" // for IntelliSense
20
21#if MOCHI_USE_SIMD && MOCHI_ARCH_X64_AVX2
22
23namespace superdex {
24
25/***********************************************************************************************
26 Simd<float, 4>
27*/
28template <>
29class Simd<float, 4> {
30 public:
31 MOCHI_NATIVE_SIMD_IMPL_BOILERPLATE(float, 4, __m128);
32 Simd(float a, float b, float c = 0.0f, float d = 0.0f) : raw(_mm_set_ps(d, c, b, a)) {} // SSE
33 template <class U, MOCHI_REQUIRES_NON_BOOL_SCALAR(U, Scalar)>
34 Simd(U a) : raw(_mm_set_ps1(a)) {} // SSE
35
36 template <int i>
37 [[nodiscard]] static MOCHI_FORCE_INLINE float Get(Simd v) {
38 static_assert(i >= 0 && i < 4, "Index out of range");
39 return v[i];
40 }
41
42 [[nodiscard]] MOCHI_FORCE_INLINE Scalar operator[](int i) const {
43 MOCHI_ASSERT_VERBOSE(i >= 0 && i < kSize, "Index out of range");
44#if MOCHI_COMPILER_MSVC
45 return raw.m128_f32[i];
46#else
47 return raw[i];
48#endif
49 }
50
51 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Set(Simd v, int i, Scalar value) {
52 MOCHI_ASSERT_VERBOSE(i >= 0 && i < kSize, "Index out of range");
53#if MOCHI_COMPILER_MSVC
54 auto result = v;
55 result.raw.m128_f32[i] = value;
56 return result;
57#else
58 static constexpr __m128i kMasks[] = {
59 // clang-format off
60 {static_cast<long long>(0x00000000FFFFFFFFLL), static_cast<long long>(0x0000000000000000LL)},
61 {static_cast<long long>(0xFFFFFFFF00000000LL), static_cast<long long>(0x0000000000000000LL)},
62 {static_cast<long long>(0x0000000000000000LL), static_cast<long long>(0x00000000FFFFFFFFLL)},
63 {static_cast<long long>(0x0000000000000000LL), static_cast<long long>(0xFFFFFFFF00000000LL)}
64 }; // clang-format on
65 return _mm_blendv_ps(v.raw, _mm_set1_ps(value), _mm_castsi128_ps(kMasks[i])); // SSE4.1
66#endif
67 }
68
69 template <int i>
70 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Set(Simd v, Scalar value) {
71 return _mm_insert_ps(v.raw, _mm_set1_ps(value), i << 4); // SSE4.1
72 }
73
74 [[nodiscard]] static MOCHI_FORCE_INLINE Simd AsPoint(Simd a) {
75 // Replace the 3rd component with an integer that has the same bits as 1.0f.
76 auto araw = _mm_castps_si128(a.raw); // SSE2
77 auto v = _mm_insert_epi32(araw, 0x3f800000, 3); // SSE4.1
78 return _mm_castsi128_ps(v); // SSE2
79 }
80
81 [[nodiscard]] static MOCHI_FORCE_INLINE Simd AsDirection(Simd a) {
82 // Replace the 3rd component with an integer that has the same bits as 0.0f.
83 auto araw = _mm_castps_si128(a.raw); // SSE2
84 auto v = _mm_insert_epi32(araw, 0x00000000, 3); // SSE4.1
85 return _mm_castsi128_ps(v); // SSE2
86 }
87
88 template <int x, int y, int z, int w>
89 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Blend(Simd a, Simd b) {
90 static_assert(
91 x >= 0 && x < 2 && y >= 0 && y < 2 && z >= 0 && z < 2 && w >= 0 && w < 2,
92 "invalid blend index");
93 if constexpr (x == 0 && y == 0 && z == 0 && w == 0) {
94 return a;
95 } else if constexpr (x == 1 && y == 1 && z == 1 && w == 1) {
96 return b;
97 } else {
98 return _mm_blend_ps(a.raw, b.raw, x | (y << 1) | (z << 2) | (w << 3)); // SSE4.1
99 }
100 }
101
102 template <int N>
103 [[nodiscard]] static MOCHI_FORCE_INLINE bool AllTrue(Simd v) {
104 static_assert(N >= 1 && N <= kSize, "Unsupported N");
105 int mask = GetMSBitMask(v); // One bit for each byte in the vector
106 if constexpr (N == kSize) {
107 return mask == 0x0000FFFF;
108 } else {
109 int constexpr kNumBits = N * sizeof(Scalar);
110 auto constexpr kMustBeSet = (1UL << kNumBits) - 1;
111 return (mask & kMustBeSet) == kMustBeSet;
112 }
113 }
114
115 template <int N>
116 [[nodiscard]] static MOCHI_FORCE_INLINE bool AnyTrue(Simd v) {
117 static_assert(N >= 1 && N <= kSize, "Unsupported N");
118 int mask = GetMSBitMask(v); // One bit for each byte in the vector
119 if constexpr (N == kSize) {
120 return mask != 0;
121 } else {
122 int constexpr kNumBits = N * sizeof(Scalar);
123 auto constexpr kMayBeSet = (1UL << kNumBits) - 1;
124 return (mask & kMayBeSet) != 0;
125 }
126 }
127
128 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Broadcast(Scalar const* p) {
129 return _mm_broadcast_ss(p); // AVX
130 }
131
132 template <int i>
133 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Broadcast(Simd v) {
134 return Shuffle<i, i, i, i>(v);
135 }
136
137 template <int N = kSize>
138 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Load([[maybe_unused]] Scalar const* ptr) {
139 static_assert(N >= 0 && N <= 4);
140 if constexpr (N == 0) {
141 return Simd::Zero();
142 } else if constexpr (N == 1) {
143 return Simd{*ptr, 0.0f};
144 } else if constexpr (N == 2) {
145 __m128i mask = _mm_set_epi32(0, 0, -1, -1); // SSE2
146 return _mm_maskload_ps(ptr, mask); // AVX
147 } else if constexpr (N == 3) {
148 __m128i mask = _mm_set_epi32(0, -1, -1, -1); // SSE2
149 return _mm_maskload_ps(ptr, mask); // AVX
150 } else if constexpr (N == 4) {
151 return _mm_loadu_ps(ptr); // SSE
152 }
153 }
154
155 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Load(Scalar const* ptr, int n) {
156 MOCHI_ASSERT_VERBOSE(n >= 0 && n <= kSize, "Invalid size parameter");
157#if MOCHI_ARCH_X64_AVX512
158 return _mm_maskz_loadu_ps(x64_simd::kLaneMasksS8[n], ptr); // AVX512VL
159#else
160 return _mm_maskload_ps(ptr, x64_simd::kLoadMasksS4[n]); // AVX
161#endif
162 }
163
164 [[nodiscard]] static Simd LoadIndexed(Scalar const* ptr, Simd<int, 4> const& indices) {
165 return _mm_i32gather_ps(ptr, indices.raw, sizeof(float)); // AVX2
166 }
167
168 template <int kTupleCount = kSize>
169 MOCHI_FORCE_INLINE static void
170 LoadTransposed(Scalar const* ptr, Simd& out0, Simd& out1, Simd& out2) {
171 static_assert(kTupleCount >= 1 && kTupleCount <= kSize, "Invalid kTupleCount");
172 constexpr int kCount0 = Clamp(kTupleCount * 3 - 0, 0, 4);
173 constexpr int kCount1 = Clamp(kTupleCount * 3 - 4, 0, 4);
174 constexpr int kCount2 = Clamp(kTupleCount * 3 - 8, 0, 4);
175 auto a = Simd::Load<kCount0>(ptr).raw; // [0,1,2,3]
176 auto b = Simd::Load<kCount1>(kCount1 == 0 ? ptr : ptr + 4).raw; // [4,5,6,7]
177 auto c = Simd::Load<kCount2>(kCount2 == 0 ? ptr : ptr + 8).raw; // [8,9,10,11]
178
179 auto t0 = _mm_blend_ps(a, b, 0b0100); // [0,_,6,3]
180 auto t1 = _mm_blend_ps(t0, c, 0b0010); // [0,9,6,3]
181 out0 = _mm_shuffle_ps(t1, t1, _MM_SHUFFLE(1, 2, 3, 0)); // [0,3,6,9]
182
183 t0 = _mm_blend_ps(a, b, 0b1001); // [4,1,_,7]
184 t1 = _mm_blend_ps(t0, c, 0b0100); // [4,1,10,7]
185 out1 = _mm_shuffle_ps(t1, t1, _MM_SHUFFLE(2, 3, 0, 1)); // [1,4,7,10]
186
187 t0 = _mm_blend_ps(a, b, 0b0010); // [_,5,2,_]
188 t1 = _mm_blend_ps(c, t0, 0b0110); // [8,5,2,11]
189 out2 = _mm_shuffle_ps(t1, t1, _MM_SHUFFLE(3, 0, 1, 2)); // [2,5,8,11]
190 }
191
192 template <int i>
193 [[nodiscard]] static MOCHI_FORCE_INLINE Simd SetBasisVector() {
194 static_assert(i >= 0 && i <= 3, "Invalid component index");
195 auto zero = _mm_setzero_si128(); // SSE2
196 auto v = _mm_insert_epi32(zero, 0x3f800000, i); // SSE4.1
197 return _mm_castsi128_ps(v); // SSE2
198 }
199
200 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Select(Simd mask, Simd a, Simd b) {
201 return _mm_blendv_ps(b.raw, a.raw, mask.raw); // SSE4.1
202 }
203
204 // return Simd{v[x], v[y], v[z], v[w]}
205 template <int x = 0, int y = 1, int z = 2, int w = 3>
206 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Shuffle(Simd v) {
207 static_assert(
208 x >= 0 && x < 4 && y >= 0 && y < 4 && z >= 0 && z < 4 && w >= 0 && w < 4, "Invalid index");
209 if constexpr (x == 0 && y == 1 && z == 2 && w == 3) {
210 return v;
211 } else {
212 return _mm_shuffle_ps(v.raw, v.raw, x | (y << 2) | (z << 4) | (w << 6)); // SSE
213 }
214 }
215 template <int x = 0, int y = 1, int z = 2, int w = 3>
216 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Shuffle(Simd a, Simd b) {
217 static_assert(
218 x >= 0 && x < 4 && y >= 0 && y < 4 && z >= 0 && z < 4 && w >= 0 && w < 4, "Invalid index");
219 return _mm_shuffle_ps(a.raw, b.raw, x | (y << 2) | (z << 4) | (w << 6)); // SSE
220 }
221
222 template <int N = kSize>
223 static MOCHI_FORCE_INLINE void Store([[maybe_unused]] Scalar* ptr, [[maybe_unused]] Simd v) {
224 static_assert(N >= 0 && N <= kSize);
225 if constexpr (N == 0) {
226 } else if constexpr (N < kSize) {
227 // About 3X faster than a masked store on older AMD CPUs. About the same on others.
228 memcpy(ptr, &v, sizeof(Scalar) * N);
229 } else {
230 _mm_storeu_ps(ptr, v.raw); // SSE
231 }
232 }
233
234 static MOCHI_FORCE_INLINE void Store(Scalar* ptr, Simd v, int n) {
235 MOCHI_ASSERT_VERBOSE(n >= 0 && n <= kSize, "Invalid size parameter");
236#if MOCHI_ARCH_X64_AVX512
237 _mm_mask_storeu_ps(ptr, x64_simd::kLaneMasksS8[n], v.raw); // AVX512VL
238#else
239 // With AVX2, this is faster than masked store for a predictable value of n.
240 // It is much slower for a random value of n.
241 switch (n) { // clang-format off
242 case 1: Store<1>(ptr, v); break;
243 case 2: Store<2>(ptr, v); break;
244 case 3: Store<3>(ptr, v); break;
245 case 4: Store<4>(ptr, v); break;
246 MOCHI_UNLIKELY default: break;
247 } // clang-format on
248#endif
249 }
250
251 MOCHI_FORCE_INLINE static int StoreSelected(Scalar* ptr, Simd condition, Simd values) {
252 auto mask = _mm_movemask_ps(condition.raw);
253 // Load the shuffle pattern from a lookup table.
254 auto const* tableRow =
255 reinterpret_cast<__m128i const*>(x64_simd::kStoreSelectedShuffleTableS4[mask]);
256 auto pattern = _mm_load_si128(tableRow);
257 auto packed = _mm_permutevar_ps(values.raw, pattern);
258 _mm_storeu_ps(ptr, packed);
259 return _mm_popcnt_u32(mask);
260 }
261
262 template <int kTupleCount = kSize>
263 MOCHI_FORCE_INLINE static void StoreTransposed(Scalar* ptr, Simd a, Simd b, Simd c) {
264 static_assert(kTupleCount >= 1 && kTupleCount <= kSize, "Invalid kTupleCount");
265 // a = [0,3,6,9], b = [1,4,7,10], c = [2,5,8,11]
266 auto d = _mm_shuffle_ps(a.raw, a.raw, _MM_SHUFFLE(1, 2, 3, 0)); // [0,9,6,3]
267 auto e = _mm_shuffle_ps(b.raw, b.raw, _MM_SHUFFLE(2, 3, 0, 1)); // [4,1,10,7]
268 auto f = _mm_shuffle_ps(c.raw, c.raw, _MM_SHUFFLE(3, 0, 1, 2)); // [8,5,2,11]
269 constexpr int kCount0 = Clamp(kTupleCount * 3 - 0, 0, 4);
270 constexpr int kCount1 = Clamp(kTupleCount * 3 - 4, 0, 4);
271 constexpr int kCount2 = Clamp(kTupleCount * 3 - 8, 0, 4);
272 Simd::Store<kCount0>(ptr, _mm_blend_ps(_mm_blend_ps(d, e, 0b0010), f, 0b0100)); // [0,1,2,3]
273 if constexpr (kCount1 > 0) {
274 Simd::Store<kCount1>(
275 ptr + 4, _mm_blend_ps(_mm_blend_ps(d, e, 0b1001), f, 0b0010)); // [4,5,6,7]
276 }
277 if constexpr (kCount2 > 0) {
278 Simd::Store<kCount2>(
279 ptr + 8, _mm_blend_ps(_mm_blend_ps(d, e, 0b0100), f, 0b1001)); // [8,9,10,11]
280 }
281 }
282
283 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Sqrt(Simd v) {
284 return _mm_sqrt_ps(v.raw); // SSE
285 }
286
287 [[nodiscard]] static MOCHI_FORCE_INLINE Simd RcpApprox(Simd v) {
288 return _mm_rcp_ps(v.raw); // SSE
289 }
290
291 [[nodiscard]] static MOCHI_FORCE_INLINE Simd RcpSqrtApprox(Simd v) {
292 return _mm_rsqrt_ps(v.raw); // SSE
293 }
294
295 // Broadcast the value -0.0. Use this in bitwise operations to affect just the sign bit.
296 [[nodiscard]] static MOCHI_FORCE_INLINE Simd SignBitMask() {
297 return _mm_castsi128_ps(_mm_set1_epi32(0x80000000)); // SSE2, SSE2
298 }
299
300 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Abs(Simd v) {
301 return _mm_andnot_ps(SignBitMask().raw, v.raw); // SSE
302 }
303
304 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Min(Simd a, Simd b) {
305 return _mm_min_ps(a.raw, b.raw); // SSE
306 }
307
308 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Max(Simd a, Simd b) {
309 return _mm_max_ps(a.raw, b.raw); // SSE
310 }
311
312 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Floor(Simd a) {
313 return _mm_floor_ps(a.raw); // SSE4.1
314 }
315
316 [[nodiscard]] static MOCHI_FORCE_INLINE Simd FastRound(Simd v) {
317 return _mm_round_ps(v.raw, _MM_FROUND_TO_NEAREST_INT); // SSE4.1
318 }
319
320#if MOCHI_ARCH_X64_SVML
321 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Cos(Simd a) {
322 return _mm_cos_ps(a.raw); // SSE
323 }
324
325 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Sin(Simd a) {
326 return _mm_sin_ps(a.raw); // SSE
327 }
328
329 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Tan(Simd a) {
330 return _mm_tan_ps(a.raw); // SSE
331 }
332
333 [[nodiscard]] static MOCHI_FORCE_INLINE Simd ACos(Simd a) {
334 return _mm_acos_ps(a.raw); // SSE
335 }
336
337 [[nodiscard]] static MOCHI_FORCE_INLINE Simd ASin(Simd a) {
338 return _mm_asin_ps(a.raw); // SSE
339 }
340
341 [[nodiscard]] static MOCHI_FORCE_INLINE Simd ATan(Simd a) {
342 return _mm_atan_ps(a.raw); // SSE
343 }
344
345 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Exp(Simd a) {
346 return _mm_exp_ps(a.raw); // SSE
347 }
348
349 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Ln(Simd a) {
350 return _mm_log_ps(a.raw); // SSE
351 }
352
353 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Tanh(Simd a) {
354 return _mm_tanh_ps(a.raw); // SSE
355 }
356#endif // MOCHI_ARCH_X64_SVML
357
358 [[nodiscard]] static MOCHI_FORCE_INLINE Simd MulAdd(Simd a, Simd b, Simd c) {
359#if MOCHI_ARCH_X64_FMA
360 return {_mm_fmadd_ps(a.raw, b.raw, c.raw)}; // FMA
361#else
362 return (a * b) + c;
363#endif
364 }
365
366 [[nodiscard]] static MOCHI_FORCE_INLINE Simd MulSub(Simd a, Simd b, Simd c) {
367#if MOCHI_ARCH_X64_FMA
368 return _mm_fmsub_ps(a.raw, b.raw, c.raw); // FMA
369#else
370 return (a * b) - c;
371#endif
372 }
373
374 [[nodiscard]] static MOCHI_FORCE_INLINE Simd NegMulAdd(Simd a, Simd b, Simd c) {
375#if MOCHI_ARCH_X64_FMA
376 return _mm_fnmadd_ps(a.raw, b.raw, c.raw); // FMA
377#else
378 return -(a * b) + c;
379#endif
380 }
381
382 [[nodiscard]] static MOCHI_FORCE_INLINE Simd NegMulSub(Simd a, Simd b, Simd c) {
383#if MOCHI_ARCH_X64_FMA
384 return _mm_fnmsub_ps(a.raw, b.raw, c.raw); // FMA
385#else
386 return -(a * b) - c;
387#endif
388 }
389
390 template <int N = 4>
391 [[nodiscard]] static MOCHI_FORCE_INLINE Scalar HMin(Simd a) {
392 static_assert(N >= 2 && N <= 4, "Unsupported N");
393 if constexpr (N == 2) {
394 return Get<0>(Min(a, Broadcast<1>(a)));
395 } else if constexpr (N == 3) {
396 return Get<0>(Min(Min(a, Broadcast<1>(a)), Broadcast<2>(a)));
397 } else {
398 auto tmp = Min(a, Shuffle<1, 2, 3, 0>(a));
399 return Get<0>(Min(tmp, Simd::Shuffle<2, 3, 0, 1>(tmp)));
400 }
401 }
402
403 template <int N = 4>
404 [[nodiscard]] static MOCHI_FORCE_INLINE Scalar HMax(Simd a) {
405 static_assert(N >= 2 && N <= 4, "Unsupported N");
406 if constexpr (N == 2) {
407 return Get<0>(Max(a, Broadcast<1>(a)));
408 } else if constexpr (N == 3) {
409 return Get<0>(Max(Max(a, Broadcast<1>(a)), Broadcast<2>(a)));
410 } else {
411 auto tmp = Max(a, Shuffle<1, 2, 3, 0>(a));
412 return Get<0>(Max(tmp, Shuffle<2, 3, 0, 1>(tmp)));
413 }
414 }
415
416 template <int N>
417 [[nodiscard]] static MOCHI_FORCE_INLINE Scalar HSum(Simd a) {
418 static_assert(N >= 2 && N <= 4, "Unsupported N");
419 // Terms are added in the same order as Simd<double, 4>::HSum<N> for consistency.
420 if constexpr (N == 2) {
421 return Get<0>(a) + Get<1>(a); // a[0] + a[1]
422 } else if constexpr (N == 3) {
423 return (Get<0>(a) + Get<2>(a)) + Get<1>(a); // (a[0] + a[2]) + a[1]
424 } else {
425 // PERF NOTE: Alternatively _mm_dp_ps could be used to compute the dot product with
426 // Simd{1.0f}. In comparison, this implementation takes 2 extra instructions, but it had ~50%
427 // higher throughput and ~45% lower latency, when benchmarked on an AMD CPU.
428 return (Get<0>(a) + Get<2>(a)) + (Get<1>(a) + Get<3>(a)); // (a[0] + a[2]) + (a[1] + a[3])
429 }
430 }
431
432 template <int N>
433 [[nodiscard]] static MOCHI_FORCE_INLINE Scalar HProd(Simd a) {
434 static_assert(N >= 2 && N <= 4, "Unsupported N");
435 alignas(alignof(Simd)) Scalar buf[4];
436 Store(buf, a);
437 if constexpr (N == 2) {
438 return buf[0] * buf[1];
439 } else if constexpr (N == 3) {
440 return buf[0] * buf[1] * buf[2];
441 } else if constexpr (N == 4) {
442 return buf[0] * buf[1] * buf[2] * buf[3];
443 }
444 }
445
446 template <int N>
447 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Dot(Simd a, Simd b) {
448 static_assert(N >= 2 && N <= 4, "Unsupported N");
449 if constexpr (N == 2) {
450 return _mm_dp_ps(a.raw, b.raw, 0x3F); // SSE4.1
451 } else if constexpr (N == 3) {
452 return _mm_dp_ps(a.raw, b.raw, 0x7F); // SSE4.1
453 } else if constexpr (N == 4) {
454#if MOCHI_COMPILER_CLANG
455 return _mm_dp_ps(a.raw, b.raw, -1); // SSE4.1
456#else
457 return _mm_dp_ps(a.raw, b.raw, 0xFF); // SSE4.1
458#endif
459 }
460 }
461
462 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator<(Simd rhs) const {
463 return _mm_cmplt_ps(this->raw, rhs.raw); // SSE
464 }
465
466 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator>(Simd rhs) const {
467 return _mm_cmpgt_ps(this->raw, rhs.raw); // SSE
468 }
469
470 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator<=(Simd rhs) const {
471 return _mm_cmple_ps(this->raw, rhs.raw); // SSE
472 }
473
474 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator>=(Simd rhs) const {
475 return _mm_cmpge_ps(this->raw, rhs.raw); // SSE
476 }
477
478 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Equal(Simd a, Simd b) {
479 return _mm_cmpeq_ps(a.raw, b.raw); // SSE
480 }
481
482 [[nodiscard]] static MOCHI_FORCE_INLINE Simd NotEqual(Simd a, Simd b) {
483 return _mm_cmpneq_ps(a.raw, b.raw); // SSE
484 }
485
486 [[nodiscard]] static MOCHI_FORCE_INLINE Simd Zero() {
487 return _mm_setzero_ps(); // SSE
488 }
489
490 [[nodiscard]] MOCHI_FORCE_INLINE bool operator==(Simd rhs) const {
491 auto mask = GetMSBitMask(Equal(*this, rhs));
492 return mask == 0x0000FFFF; // All values equal
493 }
494
495 [[nodiscard]] MOCHI_FORCE_INLINE bool operator!=(Simd rhs) const {
496 auto mask = GetMSBitMask(NotEqual(*this, rhs));
497 return mask != 0; // All values equal
498 }
499
500 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator~() const {
501 // _mm_cmpeq_epi32 appears to be the fastest way to fill an SSE register with ones.
502 __m128i dummy{};
503 __m128 ones = _mm_castsi128_ps(_mm_cmpeq_epi32(dummy, dummy)); // SSE2, SSE2
504 return _mm_xor_ps(raw, ones); // SSE
505 }
506
507 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator-() const {
508 return _mm_xor_ps(raw, SignBitMask().raw); // SSE, SSE
509 }
510
511 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator+(Simd rhs) const {
512 return _mm_add_ps(raw, rhs.raw); // SSE
513 }
514
515 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator-(Simd rhs) const {
516 return _mm_sub_ps(raw, rhs.raw); // SSE
517 }
518
519 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator*(Simd rhs) const {
520 return _mm_mul_ps(raw, rhs.raw); // SSE
521 }
522
523 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator/(Simd rhs) const {
524 return _mm_div_ps(raw, rhs.raw); // SSE
525 }
526
527 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator&(Simd rhs) const {
528 return _mm_and_ps(raw, rhs.raw); // SSE
529 }
530
531 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator|(Simd rhs) const {
532 return _mm_or_ps(raw, rhs.raw); // SSE
533 }
534
535 [[nodiscard]] MOCHI_FORCE_INLINE Simd operator^(Simd rhs) const {
536 return _mm_xor_ps(raw, rhs.raw); // SSE
537 }
538
539 private:
540 // Integer mask with the most significant bit of each byte in the vector
541 [[nodiscard]] static MOCHI_FORCE_INLINE int GetMSBitMask(Simd a) {
542 return _mm_movemask_epi8(_mm_castps_si128(a.raw)); // SSE2, SSE2
543 }
544};
545
546} // namespace superdex
547
548#endif // MOCHI_USE_SIMD && MOCHI_ARCH_X64_AVX2
Simd operator&(Simd rhs) const
bool operator==(Simd rhs) const
NativeType raw
Definition simd.h:174
Simd operator>(Simd rhs) const
Simd operator*(Simd rhs) const
Simd operator^(Simd rhs) const
Simd operator-() const
Simd operator>=(Simd rhs) const
Simd operator<(Simd rhs) const
bool operator!=(Simd rhs) const
Simd operator|(Simd rhs) const
static constexpr int kSize
Definition simd.h:96
Simd operator+(Simd rhs) const
Simd operator~() const
Simd operator/(Simd rhs) const
Simd operator<=(Simd rhs) const
Scalar operator[](int i) const
#define MOCHI_ASSERT_VERBOSE(condition_without_side_effects,...)
Definition debug.h:102
#define MOCHI_UNLIKELY
#define MOCHI_FORCE_INLINE
T Dot(Simd< T, N > a, Simd< T, N > b)
Definition simd.h:673
constexpr T ACos(T a)
Simd< T, 2 > Shuffle(Simd< T, 2 > a)
Definition simd_inl.h:273
V LoadIndexed(typename V::Scalar const *ptr, Simd< I, V::kSize > indices)
Definition simd_inl.h:200
constexpr T const & Min(T const &a, T const &b)
constexpr auto Equal(T const &a, T const &b)
constexpr T Sin(T a)
T HSum(Simd< T, N > a)
Definition simd_inl.h:377
T HMin(Simd< T, N > a)
Definition simd_inl.h:389
bool AllTrue(T const &a)
Definition basic_utils.h:60
constexpr auto MulAdd(A a, B b, C c)
Simd< T, N > Tanh(Simd< T, N > a)
Definition simd_inl.h:709
Simd< T, N > Set(Simd< T, N > a, T value)
Definition simd_inl.h:313
constexpr auto NotEqual(T const &a, T const &b)
constexpr T Exp(T a)
constexpr T Cos(T a)
T HMax(Simd< T, N > a)
Definition simd_inl.h:395
V Broadcast(typename V::Scalar a)
Definition simd_inl.h:115
constexpr T Abs(T a)
Definition basic_utils.h:50
constexpr auto MulSub(A a, B b, C c)
Simd< T, N > Blend(Simd< T, N > a, Simd< T, N > b)
Definition simd_inl.h:288
constexpr T Tan(T a)
bool AnyTrue(T const &a)
Definition basic_utils.h:66
constexpr T Select(bool condition, T a, T b)
constexpr T Sqrt(T a)
constexpr T ATan(T a)
Simd< T, N > Ln(Simd< T, N > a)
Definition simd_inl.h:699
constexpr auto NegMulAdd(A a, B b, C c)
T Get(Simd< T, N > v)
Definition simd_inl.h:303
constexpr T Floor(T a)
constexpr T ASin(T a)
constexpr ValT Clamp(ValT value, MinT min, MaxT max)
T HProd(Simd< T, N > a)
Definition simd_inl.h:383
constexpr T const & Max(T const &a, T const &b)
void LoadTransposed(T const *ptr, Simd< T, N > &out0, Simd< T, N > &out1, Simd< T, N > &out2)
Definition simd_inl.h:207
Simd< T, N > FastRound(Simd< T, N > a)
Definition simd_inl.h:416
void StoreTransposed(T *ptr, Simd< T, N > a, Simd< T, N > b, Simd< T, N > c)
Definition simd_inl.h:245
void Store(T *ptr, Simd< T, N > a)
Definition simd_inl.h:213
Simd< T, N > RcpSqrtApprox(Simd< T, N > a)
Definition simd_inl.h:357
constexpr T RcpApprox(T a)
constexpr auto NegMulSub(A a, B b, C c)
int StoreSelected(T *ptr, Simd< MaskT, N > condition, Simd< T, N > values)
Definition simd_inl.h:225
V Load(typename V::Scalar const *ptr)
Definition simd_inl.h:184
#define MOCHI_NATIVE_SIMD_IMPL_BOILERPLATE(T, N, NativeT)