RavEngine
Loading...
Searching...
No Matches
simd_math_sse-inl.h
1//----------------------------------------------------------------------------//
2// //
3// ozz-animation is hosted at http://github.com/guillaumeblanc/ozz-animation //
4// and distributed under the MIT License (MIT). //
5// //
6// Copyright (c) Guillaume Blanc //
7// //
8// Permission is hereby granted, free of charge, to any person obtaining a //
9// copy of this software and associated documentation files (the "Software"), //
10// to deal in the Software without restriction, including without limitation //
11// the rights to use, copy, modify, merge, publish, distribute, sublicense, //
12// and/or sell copies of the Software, and to permit persons to whom the //
13// Software is furnished to do so, subject to the following conditions: //
14// //
15// The above copyright notice and this permission notice shall be included in //
16// all copies or substantial portions of the Software. //
17// //
18// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR //
19// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, //
20// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL //
21// THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER //
22// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING //
23// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER //
24// DEALINGS IN THE SOFTWARE. //
25// //
26//----------------------------------------------------------------------------//
27
28#ifndef OZZ_OZZ_BASE_MATHS_INTERNAL_SIMD_MATH_SSE_INL_H_
29#define OZZ_OZZ_BASE_MATHS_INTERNAL_SIMD_MATH_SSE_INL_H_
30
31// SIMD SSE2+ implementation, based on scalar floats.
32
33#include <stdint.h>
34
35#include <cassert>
36
37// Temporarly needed while trigonometric functions aren't implemented.
38#include <cmath>
39
40#include "ozz/base/maths/math_constant.h"
41
42namespace ozz {
43namespace math {
44
45namespace simd_float4 {
46
47// Internal macros.
48// Unused components of the result vector are replicated from the first input
49// argument.
50
51#ifdef OZZ_SIMD_AVX
52#define OZZ_SHUFFLE_PS1(_v, _m) _mm_permute_ps(_v, _m)
53#else // OZZ_SIMD_AVX
54#define OZZ_SHUFFLE_PS1(_v, _m) _mm_shuffle_ps(_v, _v, _m)
55#endif // OZZ_SIMD_AVX
56
57#define OZZ_SSE_SPLAT_F(_v, _i) OZZ_SHUFFLE_PS1(_v, _MM_SHUFFLE(_i, _i, _i, _i))
58
59#define OZZ_SSE_SPLAT_I(_v, _i) \
60 _mm_shuffle_epi32(_v, _MM_SHUFFLE(_i, _i, _i, _i))
61
62// _v.x + _v.y, _v.y, _v.z, _v.w
63#define OZZ_SSE_HADD2_F(_v) _mm_add_ss(_v, OZZ_SSE_SPLAT_F(_v, 1))
64
65// _v.x + _v.y + _v.z, _v.y, _v.z, _v.w
66#define OZZ_SSE_HADD3_F(_v) \
67 _mm_add_ss(_mm_add_ss(_v, OZZ_SSE_SPLAT_F(_v, 2)), OZZ_SSE_SPLAT_F(_v, 1))
68
69// _v.x + _v.y + _v.z + _v.w, ?, ?, ?
70#define OZZ_SSE_HADD4_F(_v, _r) \
71 do { \
72 const __m128 haddxyzw = _mm_add_ps(_v, _mm_movehl_ps(_v, _v)); \
73 _r = _mm_add_ss(haddxyzw, OZZ_SSE_SPLAT_F(haddxyzw, 1)); \
74 } while (void(0), 0)
75
76// dot2, ?, ?, ?
77#define OZZ_SSE_DOT2_F(_a, _b, _r) \
78 do { \
79 const __m128 ab = _mm_mul_ps(_a, _b); \
80 _r = _mm_add_ss(ab, OZZ_SSE_SPLAT_F(ab, 1)); \
81 \
82 } while (void(0), 0)
83
84#ifdef OZZ_SIMD_SSE4_1
85// dot3, ?, ?, ?
86#define OZZ_SSE_DOT3_F(_a, _b, _r) \
87 do { \
88 _r = _mm_dp_ps(_a, _b, 0x7f); \
89 } while (void(0), 0)
90
91// dot4, ?, ?, ?
92#define OZZ_SSE_DOT4_F(_a, _b, _r) \
93 do { \
94 _r = _mm_dp_ps(_a, _b, 0xff); \
95 } while (void(0), 0)
96
97#else // OZZ_SIMD_SSE4_1
98// dot3, ?, ?, ?
99#define OZZ_SSE_DOT3_F(_a, _b, _r) \
100 do { \
101 const __m128 ab = _mm_mul_ps(_a, _b); \
102 _r = OZZ_SSE_HADD3_F(ab); \
103 } while (void(0), 0)
104
105// dot4, ?, ?, ?
106#define OZZ_SSE_DOT4_F(_a, _b, _r) \
107 do { \
108 const __m128 ab = _mm_mul_ps(_a, _b); \
109 OZZ_SSE_HADD4_F(ab, _r); \
110 } while (void(0), 0)
111#endif // OZZ_SIMD_SSE4_1
112
113// FMA operations
114#ifdef OZZ_SIMD_FMA
115#define OZZ_MADD(_a, _b, _c) _mm_fmadd_ps(_a, _b, _c)
116#define OZZ_MSUB(_a, _b, _c) _mm_fmsub_ps(_a, _b, _c)
117#define OZZ_NMADD(_a, _b, _c) _mm_fnmadd_ps(_a, _b, _c)
118#define OZZ_NMSUB(_a, _b, _c) _mm_fnmsub_ps(_a, _b, _c)
119#define OZZ_MADDX(_a, _b, _c) _mm_fmadd_ss(_a, _b, _c)
120#define OZZ_MSUBX(_a, _b, _c) _mm_fmsub_ss(_a, _b, _c)
121#define OZZ_NMADDX(_a, _b, _c) _mm_fnmadd_ss(_a, _b, _c)
122#define OZZ_NMSUBX(_a, _b, _c) _mm_fnmsub_ss(_a, _b, _c)
123#else // OZZ_SIMD_FMA
124#define OZZ_MADD(_a, _b, _c) _mm_add_ps(_mm_mul_ps(_a, _b), _c)
125#define OZZ_MSUB(_a, _b, _c) _mm_sub_ps(_mm_mul_ps(_a, _b), _c)
126#define OZZ_NMADD(_a, _b, _c) _mm_sub_ps(_c, _mm_mul_ps(_a, _b))
127#define OZZ_NMSUB(_a, _b, _c) (-_mm_add_ps(_mm_mul_ps(_a, _b), _c))
128#define OZZ_MADDX(_a, _b, _c) _mm_add_ss(_mm_mul_ss(_a, _b), _c)
129#define OZZ_MSUBX(_a, _b, _c) _mm_sub_ss(_mm_mul_ss(_a, _b), _c)
130#define OZZ_NMADDX(_a, _b, _c) _mm_sub_ss(_c, _mm_mul_ss(_a, _b))
131#define OZZ_NMSUBX(_a, _b, _c) (-_mm_add_ss(_mm_mul_ss(_a, _b), _c))
132#endif // OZZ_SIMD_FMA
133
134OZZ_INLINE SimdFloat4 DivX(_SimdFloat4 _a, _SimdFloat4 _b) {
135 return _mm_div_ss(_a, _b);
136}
137
138#ifdef OZZ_SIMD_SSE4_1
139
140#define OZZ_SSE_SELECT_F(_b, _true, _false) \
141 _mm_blendv_ps(_false, _true, _mm_castsi128_ps(_b))
142
143#define OZZ_SSE_SELECT_I(_b, _true, _false) _mm_blendv_epi8(_false, _true, _b)
144
145#else // OZZ_SIMD_SSE4_1
146
147#define OZZ_SSE_SELECT_F(_b, _true, _false) \
148 _mm_or_ps(_mm_and_ps(_true, _mm_castsi128_ps(_b)), \
149 _mm_andnot_ps(_mm_castsi128_ps(_b), _false))
150
151#define OZZ_SSE_SELECT_I(_b, _true, _false) \
152 _mm_or_si128(_mm_and_si128(_true, _b), _mm_andnot_si128(_b, _false))
153
154#endif // OZZ_SIMD_SSE4_1
155
156OZZ_INLINE SimdFloat4 zero() { return _mm_setzero_ps(); }
157
158OZZ_INLINE SimdFloat4 one() {
159 const __m128i zero = _mm_setzero_si128();
160 return _mm_castsi128_ps(
161 _mm_srli_epi32(_mm_slli_epi32(_mm_cmpeq_epi32(zero, zero), 25), 2));
162}
163
164OZZ_INLINE SimdFloat4 x_axis() {
165 const __m128i zero = _mm_setzero_si128();
166 const __m128i one =
167 _mm_srli_epi32(_mm_slli_epi32(_mm_cmpeq_epi32(zero, zero), 25), 2);
168 return _mm_castsi128_ps(_mm_srli_si128(one, 12));
169}
170
171OZZ_INLINE SimdFloat4 y_axis() {
172 const __m128i zero = _mm_setzero_si128();
173 const __m128i one =
174 _mm_srli_epi32(_mm_slli_epi32(_mm_cmpeq_epi32(zero, zero), 25), 2);
175 return _mm_castsi128_ps(_mm_slli_si128(_mm_srli_si128(one, 12), 4));
176}
177
178OZZ_INLINE SimdFloat4 z_axis() {
179 const __m128i zero = _mm_setzero_si128();
180 const __m128i one =
181 _mm_srli_epi32(_mm_slli_epi32(_mm_cmpeq_epi32(zero, zero), 25), 2);
182 return _mm_castsi128_ps(_mm_slli_si128(_mm_srli_si128(one, 12), 8));
183}
184
185OZZ_INLINE SimdFloat4 w_axis() {
186 const __m128i zero = _mm_setzero_si128();
187 const __m128i one =
188 _mm_srli_epi32(_mm_slli_epi32(_mm_cmpeq_epi32(zero, zero), 25), 2);
189 return _mm_castsi128_ps(_mm_slli_si128(one, 12));
190}
191
192OZZ_INLINE SimdFloat4 Load(float _x, float _y, float _z, float _w) {
193 return _mm_set_ps(_w, _z, _y, _x);
194}
195
196OZZ_INLINE SimdFloat4 LoadX(float _x) { return _mm_set_ss(_x); }
197
198OZZ_INLINE SimdFloat4 Load1(float _x) { return _mm_set_ps1(_x); }
199
200OZZ_INLINE SimdFloat4 LoadPtr(const float* _f) {
201 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
202 return _mm_load_ps(_f);
203}
204
205OZZ_INLINE SimdFloat4 LoadPtrU(const float* _f) {
206 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
207 return _mm_loadu_ps(_f);
208}
209
210OZZ_INLINE SimdFloat4 LoadXPtrU(const float* _f) {
211 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
212 return _mm_load_ss(_f);
213}
214
215OZZ_INLINE SimdFloat4 Load1PtrU(const float* _f) {
216 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
217 return _mm_load_ps1(_f);
218}
219
220OZZ_INLINE SimdFloat4 Load2PtrU(const float* _f) {
221 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
222 return _mm_unpacklo_ps(_mm_load_ss(_f + 0), _mm_load_ss(_f + 1));
223}
224
225OZZ_INLINE SimdFloat4 Load3PtrU(const float* _f) {
226 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
227 return _mm_movelh_ps(
228 _mm_unpacklo_ps(_mm_load_ss(_f + 0), _mm_load_ss(_f + 1)),
229 _mm_load_ss(_f + 2));
230}
231
232OZZ_INLINE SimdFloat4 FromInt(_SimdInt4 _i) { return _mm_cvtepi32_ps(_i); }
233} // namespace simd_float4
234
235OZZ_INLINE float GetX(_SimdFloat4 _v) { return _mm_cvtss_f32(_v); }
236
237OZZ_INLINE float GetY(_SimdFloat4 _v) {
238 return _mm_cvtss_f32(OZZ_SSE_SPLAT_F(_v, 1));
239}
240
241OZZ_INLINE float GetZ(_SimdFloat4 _v) {
242 return _mm_cvtss_f32(_mm_movehl_ps(_v, _v));
243}
244
245OZZ_INLINE float GetW(_SimdFloat4 _v) {
246 return _mm_cvtss_f32(OZZ_SSE_SPLAT_F(_v, 3));
247}
248
249OZZ_INLINE SimdFloat4 SetX(_SimdFloat4 _v, _SimdFloat4 _f) {
250 return _mm_move_ss(_v, _f);
251}
252
253OZZ_INLINE SimdFloat4 SetY(_SimdFloat4 _v, _SimdFloat4 _f) {
254 const __m128 xfnn = _mm_unpacklo_ps(_v, _f);
255 return _mm_shuffle_ps(xfnn, _v, _MM_SHUFFLE(3, 2, 1, 0));
256}
257
258OZZ_INLINE SimdFloat4 SetZ(_SimdFloat4 _v, _SimdFloat4 _f) {
259 const __m128 ffww = _mm_shuffle_ps(_f, _v, _MM_SHUFFLE(3, 3, 0, 0));
260 return _mm_shuffle_ps(_v, ffww, _MM_SHUFFLE(2, 0, 1, 0));
261}
262
263OZZ_INLINE SimdFloat4 SetW(_SimdFloat4 _v, _SimdFloat4 _f) {
264 const __m128 ffzz = _mm_shuffle_ps(_f, _v, _MM_SHUFFLE(2, 2, 0, 0));
265 return _mm_shuffle_ps(_v, ffzz, _MM_SHUFFLE(0, 2, 1, 0));
266}
267
268OZZ_INLINE SimdFloat4 SetI(_SimdFloat4 _v, _SimdFloat4 _f, int _ith) {
269 assert(_ith >= 0 && _ith <= 3 && "Invalid index, out of range.");
270 union {
271 SimdFloat4 ret;
272 float af[4];
273 } u = {_v};
274 u.af[_ith] = _mm_cvtss_f32(_f);
275 return u.ret;
276}
277
278OZZ_INLINE void StorePtr(_SimdFloat4 _v, float* _f) {
279 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
280 _mm_store_ps(_f, _v);
281}
282
283OZZ_INLINE void Store1Ptr(_SimdFloat4 _v, float* _f) {
284 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
285 _mm_store_ss(_f, _v);
286}
287
288OZZ_INLINE void Store2Ptr(_SimdFloat4 _v, float* _f) {
289 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
290 _mm_storel_pi(reinterpret_cast<__m64*>(_f), _v);
291}
292
293OZZ_INLINE void Store3Ptr(_SimdFloat4 _v, float* _f) {
294 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
295 _mm_storel_pi(reinterpret_cast<__m64*>(_f), _v);
296 _mm_store_ss(_f + 2, _mm_movehl_ps(_v, _v));
297}
298
299OZZ_INLINE void StorePtrU(_SimdFloat4 _v, float* _f) {
300 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
301 _mm_storeu_ps(_f, _v);
302}
303
304OZZ_INLINE void Store1PtrU(_SimdFloat4 _v, float* _f) {
305 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
306 _mm_store_ss(_f, _v);
307}
308
309OZZ_INLINE void Store2PtrU(_SimdFloat4 _v, float* _f) {
310 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
311 _mm_store_ss(_f + 0, _v);
312 _mm_store_ss(_f + 1, OZZ_SSE_SPLAT_F(_v, 1));
313}
314
315OZZ_INLINE void Store3PtrU(_SimdFloat4 _v, float* _f) {
316 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
317 _mm_store_ss(_f + 0, _v);
318 _mm_store_ss(_f + 1, OZZ_SSE_SPLAT_F(_v, 1));
319 _mm_store_ss(_f + 2, _mm_movehl_ps(_v, _v));
320}
321
322OZZ_INLINE SimdFloat4 SplatX(_SimdFloat4 _v) { return OZZ_SSE_SPLAT_F(_v, 0); }
323
324OZZ_INLINE SimdFloat4 SplatY(_SimdFloat4 _v) { return OZZ_SSE_SPLAT_F(_v, 1); }
325
326OZZ_INLINE SimdFloat4 SplatZ(_SimdFloat4 _v) { return OZZ_SSE_SPLAT_F(_v, 2); }
327
328OZZ_INLINE SimdFloat4 SplatW(_SimdFloat4 _v) { return OZZ_SSE_SPLAT_F(_v, 3); }
329
330template <size_t _X, size_t _Y, size_t _Z, size_t _W>
331OZZ_INLINE SimdFloat4 Swizzle(_SimdFloat4 _v) {
332 static_assert(_X <= 3 && _Y <= 3 && _Z <= 3 && _W <= 3,
333 "Indices must be between 0 and 3");
334 return OZZ_SHUFFLE_PS1(_v, _MM_SHUFFLE(_W, _Z, _Y, _X));
335}
336
337template <>
338OZZ_INLINE SimdFloat4 Swizzle<0, 1, 2, 3>(_SimdFloat4 _v) {
339 return _v;
340}
341
342template <>
343OZZ_INLINE SimdFloat4 Swizzle<0, 1, 0, 1>(_SimdFloat4 _v) {
344 return _mm_movelh_ps(_v, _v);
345}
346
347template <>
348OZZ_INLINE SimdFloat4 Swizzle<2, 3, 2, 3>(_SimdFloat4 _v) {
349 return _mm_movehl_ps(_v, _v);
350}
351
352template <>
353OZZ_INLINE SimdFloat4 Swizzle<0, 0, 1, 1>(_SimdFloat4 _v) {
354 return _mm_unpacklo_ps(_v, _v);
355}
356
357template <>
358OZZ_INLINE SimdFloat4 Swizzle<2, 2, 3, 3>(_SimdFloat4 _v) {
359 return _mm_unpackhi_ps(_v, _v);
360}
361
362OZZ_INLINE void Transpose4x1(const SimdFloat4 _in[4], SimdFloat4 _out[1]) {
363 const __m128 xz = _mm_unpacklo_ps(_in[0], _in[2]);
364 const __m128 yw = _mm_unpacklo_ps(_in[1], _in[3]);
365 _out[0] = _mm_unpacklo_ps(xz, yw);
366}
367
368OZZ_INLINE void Transpose1x4(const SimdFloat4 _in[1], SimdFloat4 _out[4]) {
369 const __m128 zwzw = _mm_movehl_ps(_in[0], _in[0]);
370 const __m128 yyyy = OZZ_SSE_SPLAT_F(_in[0], 1);
371 const __m128 wwww = OZZ_SSE_SPLAT_F(_in[0], 3);
372 const __m128 zero = _mm_setzero_ps();
373 _out[0] = _mm_move_ss(zero, _in[0]);
374 _out[1] = _mm_move_ss(zero, yyyy);
375 _out[2] = _mm_move_ss(zero, zwzw);
376 _out[3] = _mm_move_ss(zero, wwww);
377}
378
379OZZ_INLINE void Transpose4x2(const SimdFloat4 _in[4], SimdFloat4 _out[2]) {
380 const __m128 tmp0 = _mm_unpacklo_ps(_in[0], _in[2]);
381 const __m128 tmp1 = _mm_unpacklo_ps(_in[1], _in[3]);
382 _out[0] = _mm_unpacklo_ps(tmp0, tmp1);
383 _out[1] = _mm_unpackhi_ps(tmp0, tmp1);
384}
385
386OZZ_INLINE void Transpose2x4(const SimdFloat4 _in[2], SimdFloat4 _out[4]) {
387 const __m128 tmp0 = _mm_unpacklo_ps(_in[0], _in[1]);
388 const __m128 tmp1 = _mm_unpackhi_ps(_in[0], _in[1]);
389 const __m128 zero = _mm_setzero_ps();
390 _out[0] = _mm_movelh_ps(tmp0, zero);
391 _out[1] = _mm_movehl_ps(zero, tmp0);
392 _out[2] = _mm_movelh_ps(tmp1, zero);
393 _out[3] = _mm_movehl_ps(zero, tmp1);
394}
395
396OZZ_INLINE void Transpose4x3(const SimdFloat4 _in[4], SimdFloat4 _out[3]) {
397 const __m128 tmp0 = _mm_unpacklo_ps(_in[0], _in[2]);
398 const __m128 tmp1 = _mm_unpacklo_ps(_in[1], _in[3]);
399 const __m128 tmp2 = _mm_unpackhi_ps(_in[0], _in[2]);
400 const __m128 tmp3 = _mm_unpackhi_ps(_in[1], _in[3]);
401 _out[0] = _mm_unpacklo_ps(tmp0, tmp1);
402 _out[1] = _mm_unpackhi_ps(tmp0, tmp1);
403 _out[2] = _mm_unpacklo_ps(tmp2, tmp3);
404}
405
406OZZ_INLINE void Transpose3x4(const SimdFloat4 _in[3], SimdFloat4 _out[4]) {
407 const __m128 zero = _mm_setzero_ps();
408 const __m128 temp0 = _mm_unpacklo_ps(_in[0], _in[1]);
409 const __m128 temp1 = _mm_unpacklo_ps(_in[2], zero);
410 const __m128 temp2 = _mm_unpackhi_ps(_in[0], _in[1]);
411 const __m128 temp3 = _mm_unpackhi_ps(_in[2], zero);
412 _out[0] = _mm_movelh_ps(temp0, temp1);
413 _out[1] = _mm_movehl_ps(temp1, temp0);
414 _out[2] = _mm_movelh_ps(temp2, temp3);
415 _out[3] = _mm_movehl_ps(temp3, temp2);
416}
417
418OZZ_INLINE void Transpose4x4(const SimdFloat4 _in[4], SimdFloat4 _out[4]) {
419 const __m128 tmp0 = _mm_unpacklo_ps(_in[0], _in[2]);
420 const __m128 tmp1 = _mm_unpacklo_ps(_in[1], _in[3]);
421 const __m128 tmp2 = _mm_unpackhi_ps(_in[0], _in[2]);
422 const __m128 tmp3 = _mm_unpackhi_ps(_in[1], _in[3]);
423 _out[0] = _mm_unpacklo_ps(tmp0, tmp1);
424 _out[1] = _mm_unpackhi_ps(tmp0, tmp1);
425 _out[2] = _mm_unpacklo_ps(tmp2, tmp3);
426 _out[3] = _mm_unpackhi_ps(tmp2, tmp3);
427}
428
429OZZ_INLINE void Transpose16x16(const SimdFloat4 _in[16], SimdFloat4 _out[16]) {
430 const __m128 tmp0 = _mm_unpacklo_ps(_in[0], _in[2]);
431 const __m128 tmp1 = _mm_unpacklo_ps(_in[1], _in[3]);
432 _out[0] = _mm_unpacklo_ps(tmp0, tmp1);
433 _out[4] = _mm_unpackhi_ps(tmp0, tmp1);
434 const __m128 tmp2 = _mm_unpackhi_ps(_in[0], _in[2]);
435 const __m128 tmp3 = _mm_unpackhi_ps(_in[1], _in[3]);
436 _out[8] = _mm_unpacklo_ps(tmp2, tmp3);
437 _out[12] = _mm_unpackhi_ps(tmp2, tmp3);
438 const __m128 tmp4 = _mm_unpacklo_ps(_in[4], _in[6]);
439 const __m128 tmp5 = _mm_unpacklo_ps(_in[5], _in[7]);
440 _out[1] = _mm_unpacklo_ps(tmp4, tmp5);
441 _out[5] = _mm_unpackhi_ps(tmp4, tmp5);
442 const __m128 tmp6 = _mm_unpackhi_ps(_in[4], _in[6]);
443 const __m128 tmp7 = _mm_unpackhi_ps(_in[5], _in[7]);
444 _out[9] = _mm_unpacklo_ps(tmp6, tmp7);
445 _out[13] = _mm_unpackhi_ps(tmp6, tmp7);
446 const __m128 tmp8 = _mm_unpacklo_ps(_in[8], _in[10]);
447 const __m128 tmp9 = _mm_unpacklo_ps(_in[9], _in[11]);
448 _out[2] = _mm_unpacklo_ps(tmp8, tmp9);
449 _out[6] = _mm_unpackhi_ps(tmp8, tmp9);
450 const __m128 tmp10 = _mm_unpackhi_ps(_in[8], _in[10]);
451 const __m128 tmp11 = _mm_unpackhi_ps(_in[9], _in[11]);
452 _out[10] = _mm_unpacklo_ps(tmp10, tmp11);
453 _out[14] = _mm_unpackhi_ps(tmp10, tmp11);
454 const __m128 tmp12 = _mm_unpacklo_ps(_in[12], _in[14]);
455 const __m128 tmp13 = _mm_unpacklo_ps(_in[13], _in[15]);
456 _out[3] = _mm_unpacklo_ps(tmp12, tmp13);
457 _out[7] = _mm_unpackhi_ps(tmp12, tmp13);
458 const __m128 tmp14 = _mm_unpackhi_ps(_in[12], _in[14]);
459 const __m128 tmp15 = _mm_unpackhi_ps(_in[13], _in[15]);
460 _out[11] = _mm_unpacklo_ps(tmp14, tmp15);
461 _out[15] = _mm_unpackhi_ps(tmp14, tmp15);
462}
463
464OZZ_INLINE SimdFloat4 MAdd(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
465 return OZZ_MADD(_a, _b, _c);
466}
467
468OZZ_INLINE SimdFloat4 MSub(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
469 return OZZ_MSUB(_a, _b, _c);
470}
471
472OZZ_INLINE SimdFloat4 NMAdd(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
473 return OZZ_NMADD(_a, _b, _c);
474}
475
476OZZ_INLINE SimdFloat4 NMSub(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
477 return OZZ_NMSUB(_a, _b, _c);
478}
479
480OZZ_INLINE SimdFloat4 DivX(_SimdFloat4 _a, _SimdFloat4 _b) {
481 return _mm_div_ss(_a, _b);
482}
483
484OZZ_INLINE SimdFloat4 HAdd2(_SimdFloat4 _v) { return OZZ_SSE_HADD2_F(_v); }
485
486OZZ_INLINE SimdFloat4 HAdd3(_SimdFloat4 _v) { return OZZ_SSE_HADD3_F(_v); }
487
488OZZ_INLINE SimdFloat4 HAdd4(_SimdFloat4 _v) {
489 __m128 hadd4;
490 OZZ_SSE_HADD4_F(_v, hadd4);
491 return hadd4;
492}
493
494OZZ_INLINE SimdFloat4 Dot2(_SimdFloat4 _a, _SimdFloat4 _b) {
495 __m128 dot2;
496 OZZ_SSE_DOT2_F(_a, _b, dot2);
497 return dot2;
498}
499
500OZZ_INLINE SimdFloat4 Dot3(_SimdFloat4 _a, _SimdFloat4 _b) {
501 __m128 dot3;
502 OZZ_SSE_DOT3_F(_a, _b, dot3);
503 return dot3;
504}
505
506OZZ_INLINE SimdFloat4 Dot4(_SimdFloat4 _a, _SimdFloat4 _b) {
507 __m128 dot4;
508 OZZ_SSE_DOT4_F(_a, _b, dot4);
509 return dot4;
510}
511
512OZZ_INLINE SimdFloat4 Cross3(_SimdFloat4 _a, _SimdFloat4 _b) {
513 // Implementation with 3 shuffles only is based on:
514 // https://geometrian.com/programming/tutorials/cross-product
515 const __m128 shufa = OZZ_SHUFFLE_PS1(_a, _MM_SHUFFLE(3, 0, 2, 1));
516 const __m128 shufb = OZZ_SHUFFLE_PS1(_b, _MM_SHUFFLE(3, 0, 2, 1));
517 const __m128 shufc = OZZ_MSUB(_a, shufb, _mm_mul_ps(_b, shufa));
518 return OZZ_SHUFFLE_PS1(shufc, _MM_SHUFFLE(3, 0, 2, 1));
519}
520
521OZZ_INLINE SimdFloat4 RcpEst(_SimdFloat4 _v) { return _mm_rcp_ps(_v); }
522
523OZZ_INLINE SimdFloat4 RcpEstNR(_SimdFloat4 _v) {
524 const __m128 nr = _mm_rcp_ps(_v);
525 // Do one more Newton-Raphson step to improve precision.
526 return OZZ_NMADD(_mm_mul_ps(nr, nr), _v, _mm_add_ps(nr, nr));
527}
528
529OZZ_INLINE SimdFloat4 RcpEstX(_SimdFloat4 _v) { return _mm_rcp_ss(_v); }
530
531OZZ_INLINE SimdFloat4 RcpEstXNR(_SimdFloat4 _v) {
532 const __m128 nr = _mm_rcp_ss(_v);
533 // Do one more Newton-Raphson step to improve precision.
534 return OZZ_NMADDX(_mm_mul_ss(nr, nr), _v, _mm_add_ss(nr, nr));
535}
536
537OZZ_INLINE SimdFloat4 Sqrt(_SimdFloat4 _v) { return _mm_sqrt_ps(_v); }
538
539OZZ_INLINE SimdFloat4 SqrtX(_SimdFloat4 _v) { return _mm_sqrt_ss(_v); }
540
541OZZ_INLINE SimdFloat4 RSqrtEst(_SimdFloat4 _v) { return _mm_rsqrt_ps(_v); }
542
543OZZ_INLINE SimdFloat4 RSqrtEstNR(_SimdFloat4 _v) {
544 const __m128 nr = _mm_rsqrt_ps(_v);
545 // Do one more Newton-Raphson step to improve precision.
546 return _mm_mul_ps(_mm_mul_ps(_mm_set_ps1(.5f), nr),
547 OZZ_NMADD(_mm_mul_ps(_v, nr), nr, _mm_set_ps1(3.f)));
548}
549
550OZZ_INLINE SimdFloat4 RSqrtEstX(_SimdFloat4 _v) { return _mm_rsqrt_ss(_v); }
551
552OZZ_INLINE SimdFloat4 RSqrtEstXNR(_SimdFloat4 _v) {
553 const __m128 nr = _mm_rsqrt_ss(_v);
554 // Do one more Newton-Raphson step to improve precision.
555 return _mm_mul_ss(_mm_mul_ss(_mm_set_ps1(.5f), nr),
556 OZZ_NMADDX(_mm_mul_ss(_v, nr), nr, _mm_set_ps1(3.f)));
557}
558
559OZZ_INLINE SimdFloat4 Abs(_SimdFloat4 _v) {
560 const __m128i zero = _mm_setzero_si128();
561 return _mm_and_ps(
562 _mm_castsi128_ps(_mm_srli_epi32(_mm_cmpeq_epi32(zero, zero), 1)), _v);
563}
564
565OZZ_INLINE SimdInt4 Sign(_SimdFloat4 _v) {
566 return _mm_slli_epi32(_mm_srli_epi32(_mm_castps_si128(_v), 31), 31);
567}
568
569OZZ_INLINE SimdFloat4 Length2(_SimdFloat4 _v) {
570 __m128 sq_len;
571 OZZ_SSE_DOT2_F(_v, _v, sq_len);
572 return _mm_sqrt_ss(sq_len);
573}
574
575OZZ_INLINE SimdFloat4 Length3(_SimdFloat4 _v) {
576 __m128 sq_len;
577 OZZ_SSE_DOT3_F(_v, _v, sq_len);
578 return _mm_sqrt_ss(sq_len);
579}
580
581OZZ_INLINE SimdFloat4 Length4(_SimdFloat4 _v) {
582 __m128 sq_len;
583 OZZ_SSE_DOT4_F(_v, _v, sq_len);
584 return _mm_sqrt_ss(sq_len);
585}
586
587OZZ_INLINE SimdFloat4 Length2Sqr(_SimdFloat4 _v) {
588 __m128 sq_len;
589 OZZ_SSE_DOT2_F(_v, _v, sq_len);
590 return sq_len;
591}
592
593OZZ_INLINE SimdFloat4 Length3Sqr(_SimdFloat4 _v) {
594 __m128 sq_len;
595 OZZ_SSE_DOT3_F(_v, _v, sq_len);
596 return sq_len;
597}
598
599OZZ_INLINE SimdFloat4 Length4Sqr(_SimdFloat4 _v) {
600 __m128 sq_len;
601 OZZ_SSE_DOT4_F(_v, _v, sq_len);
602 return sq_len;
603}
604
605OZZ_INLINE SimdFloat4 Normalize2(_SimdFloat4 _v) {
606 __m128 sq_len;
607 OZZ_SSE_DOT2_F(_v, _v, sq_len);
608 assert(_mm_cvtss_f32(sq_len) != 0.f && "_v is not normalizable");
609 const __m128 inv_len = _mm_div_ss(simd_float4::one(), _mm_sqrt_ss(sq_len));
610 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
611 const __m128 norm = _mm_mul_ps(_v, inv_lenxxxx);
612 return _mm_movelh_ps(norm, _mm_movehl_ps(_v, _v));
613}
614
615OZZ_INLINE SimdFloat4 Normalize3(_SimdFloat4 _v) {
616 __m128 sq_len;
617 OZZ_SSE_DOT3_F(_v, _v, sq_len);
618 assert(_mm_cvtss_f32(sq_len) != 0.f && "_v is not normalizable");
619 const __m128 inv_len = _mm_div_ss(simd_float4::one(), _mm_sqrt_ss(sq_len));
620 const __m128 vwxyz = OZZ_SHUFFLE_PS1(_v, _MM_SHUFFLE(0, 1, 2, 3));
621 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
622 const __m128 normwxyz = _mm_move_ss(_mm_mul_ps(vwxyz, inv_lenxxxx), vwxyz);
623 return OZZ_SHUFFLE_PS1(normwxyz, _MM_SHUFFLE(0, 1, 2, 3));
624}
625
626OZZ_INLINE SimdFloat4 Normalize4(_SimdFloat4 _v) {
627 __m128 sq_len;
628 OZZ_SSE_DOT4_F(_v, _v, sq_len);
629 assert(_mm_cvtss_f32(sq_len) != 0.f && "_v is not normalizable");
630 const __m128 inv_len = _mm_div_ss(simd_float4::one(), _mm_sqrt_ss(sq_len));
631 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
632 return _mm_mul_ps(_v, inv_lenxxxx);
633}
634
635OZZ_INLINE SimdFloat4 NormalizeEst2(_SimdFloat4 _v) {
636 __m128 sq_len;
637 OZZ_SSE_DOT2_F(_v, _v, sq_len);
638 assert(_mm_cvtss_f32(sq_len) != 0.f && "_v is not normalizable");
639 const __m128 inv_len = _mm_rsqrt_ss(sq_len);
640 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
641 const __m128 norm = _mm_mul_ps(_v, inv_lenxxxx);
642 return _mm_movelh_ps(norm, _mm_movehl_ps(_v, _v));
643}
644
645OZZ_INLINE SimdFloat4 NormalizeEst3(_SimdFloat4 _v) {
646 __m128 sq_len;
647 OZZ_SSE_DOT3_F(_v, _v, sq_len);
648 assert(_mm_cvtss_f32(sq_len) != 0.f && "_v is not normalizable");
649 const __m128 inv_len = _mm_rsqrt_ss(sq_len);
650 const __m128 vwxyz = OZZ_SHUFFLE_PS1(_v, _MM_SHUFFLE(0, 1, 2, 3));
651 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
652 const __m128 normwxyz = _mm_move_ss(_mm_mul_ps(vwxyz, inv_lenxxxx), vwxyz);
653 return OZZ_SHUFFLE_PS1(normwxyz, _MM_SHUFFLE(0, 1, 2, 3));
654}
655
656OZZ_INLINE SimdFloat4 NormalizeEst4(_SimdFloat4 _v) {
657 __m128 sq_len;
658 OZZ_SSE_DOT4_F(_v, _v, sq_len);
659 assert(_mm_cvtss_f32(sq_len) != 0.f && "_v is not normalizable");
660 const __m128 inv_len = _mm_rsqrt_ss(sq_len);
661 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
662 return _mm_mul_ps(_v, inv_lenxxxx);
663}
664
665OZZ_INLINE SimdInt4 IsNormalized2(_SimdFloat4 _v) {
666 const __m128 max = _mm_set_ss(1.f + kNormalizationToleranceSq);
667 const __m128 min = _mm_set_ss(1.f - kNormalizationToleranceSq);
668 __m128 dot;
669 OZZ_SSE_DOT2_F(_v, _v, dot);
670 __m128 dotx000 = _mm_move_ss(_mm_setzero_ps(), dot);
671 return _mm_castps_si128(
672 _mm_and_ps(_mm_cmplt_ss(dotx000, max), _mm_cmpgt_ss(dotx000, min)));
673}
674
675OZZ_INLINE SimdInt4 IsNormalized3(_SimdFloat4 _v) {
676 const __m128 max = _mm_set_ss(1.f + kNormalizationToleranceSq);
677 const __m128 min = _mm_set_ss(1.f - kNormalizationToleranceSq);
678 __m128 dot;
679 OZZ_SSE_DOT3_F(_v, _v, dot);
680 __m128 dotx000 = _mm_move_ss(_mm_setzero_ps(), dot);
681 return _mm_castps_si128(
682 _mm_and_ps(_mm_cmplt_ss(dotx000, max), _mm_cmpgt_ss(dotx000, min)));
683}
684
685OZZ_INLINE SimdInt4 IsNormalized4(_SimdFloat4 _v) {
686 const __m128 max = _mm_set_ss(1.f + kNormalizationToleranceSq);
687 const __m128 min = _mm_set_ss(1.f - kNormalizationToleranceSq);
688 __m128 dot;
689 OZZ_SSE_DOT4_F(_v, _v, dot);
690 __m128 dotx000 = _mm_move_ss(_mm_setzero_ps(), dot);
691 return _mm_castps_si128(
692 _mm_and_ps(_mm_cmplt_ss(dotx000, max), _mm_cmpgt_ss(dotx000, min)));
693}
694
695OZZ_INLINE SimdInt4 IsNormalizedEst2(_SimdFloat4 _v) {
696 const __m128 max = _mm_set_ss(1.f + kNormalizationToleranceEstSq);
697 const __m128 min = _mm_set_ss(1.f - kNormalizationToleranceEstSq);
698 __m128 dot;
699 OZZ_SSE_DOT2_F(_v, _v, dot);
700 __m128 dotx000 = _mm_move_ss(_mm_setzero_ps(), dot);
701 return _mm_castps_si128(
702 _mm_and_ps(_mm_cmplt_ss(dotx000, max), _mm_cmpgt_ss(dotx000, min)));
703}
704
705OZZ_INLINE SimdInt4 IsNormalizedEst3(_SimdFloat4 _v) {
706 const __m128 max = _mm_set_ss(1.f + kNormalizationToleranceEstSq);
707 const __m128 min = _mm_set_ss(1.f - kNormalizationToleranceEstSq);
708 __m128 dot;
709 OZZ_SSE_DOT3_F(_v, _v, dot);
710 __m128 dotx000 = _mm_move_ss(_mm_setzero_ps(), dot);
711 return _mm_castps_si128(
712 _mm_and_ps(_mm_cmplt_ss(dotx000, max), _mm_cmpgt_ss(dotx000, min)));
713}
714
715OZZ_INLINE SimdInt4 IsNormalizedEst4(_SimdFloat4 _v) {
716 const __m128 max = _mm_set_ss(1.f + kNormalizationToleranceEstSq);
717 const __m128 min = _mm_set_ss(1.f - kNormalizationToleranceEstSq);
718 __m128 dot;
719 OZZ_SSE_DOT4_F(_v, _v, dot);
720 __m128 dotx000 = _mm_move_ss(_mm_setzero_ps(), dot);
721 return _mm_castps_si128(
722 _mm_and_ps(_mm_cmplt_ss(dotx000, max), _mm_cmpgt_ss(dotx000, min)));
723}
724
725OZZ_INLINE SimdFloat4 NormalizeSafe2(_SimdFloat4 _v, _SimdFloat4 _safe) {
726 // assert(AreAllTrue1(IsNormalized2(_safe)) && "_safe is not normalized");
727 __m128 sq_len;
728 OZZ_SSE_DOT2_F(_v, _v, sq_len);
729 const __m128 inv_len = _mm_div_ss(simd_float4::one(), _mm_sqrt_ss(sq_len));
730 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
731 const __m128 norm = _mm_mul_ps(_v, inv_lenxxxx);
732 const __m128i cond = _mm_castps_si128(
733 _mm_cmple_ps(OZZ_SSE_SPLAT_F(sq_len, 0), _mm_setzero_ps()));
734 const __m128 cfalse = _mm_movelh_ps(norm, _mm_movehl_ps(_v, _v));
735 return OZZ_SSE_SELECT_F(cond, _safe, cfalse);
736}
737
738OZZ_INLINE SimdFloat4 NormalizeSafe3(_SimdFloat4 _v, _SimdFloat4 _safe) {
739 // assert(AreAllTrue1(IsNormalized3(_safe)) && "_safe is not normalized");
740 __m128 sq_len;
741 OZZ_SSE_DOT3_F(_v, _v, sq_len);
742 const __m128 inv_len = _mm_div_ss(simd_float4::one(), _mm_sqrt_ss(sq_len));
743 const __m128 vwxyz = OZZ_SHUFFLE_PS1(_v, _MM_SHUFFLE(0, 1, 2, 3));
744 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
745 const __m128 normwxyz = _mm_move_ss(_mm_mul_ps(vwxyz, inv_lenxxxx), vwxyz);
746 const __m128i cond = _mm_castps_si128(
747 _mm_cmple_ps(OZZ_SSE_SPLAT_F(sq_len, 0), _mm_setzero_ps()));
748 const __m128 cfalse = OZZ_SHUFFLE_PS1(normwxyz, _MM_SHUFFLE(0, 1, 2, 3));
749 return OZZ_SSE_SELECT_F(cond, _safe, cfalse);
750}
751
752OZZ_INLINE SimdFloat4 NormalizeSafe4(_SimdFloat4 _v, _SimdFloat4 _safe) {
753 // assert(AreAllTrue1(IsNormalized4(_safe)) && "_safe is not normalized");
754 __m128 sq_len;
755 OZZ_SSE_DOT4_F(_v, _v, sq_len);
756 const __m128 inv_len = _mm_div_ss(simd_float4::one(), _mm_sqrt_ss(sq_len));
757 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
758 const __m128i cond = _mm_castps_si128(
759 _mm_cmple_ps(OZZ_SSE_SPLAT_F(sq_len, 0), _mm_setzero_ps()));
760 const __m128 cfalse = _mm_mul_ps(_v, inv_lenxxxx);
761 return OZZ_SSE_SELECT_F(cond, _safe, cfalse);
762}
763
764OZZ_INLINE SimdFloat4 NormalizeSafeEst2(_SimdFloat4 _v, _SimdFloat4 _safe) {
765 // assert(AreAllTrue1(IsNormalizedEst2(_safe)) && "_safe is not normalized");
766 __m128 sq_len;
767 OZZ_SSE_DOT2_F(_v, _v, sq_len);
768 const __m128 inv_len = _mm_rsqrt_ss(sq_len);
769 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
770 const __m128 norm = _mm_mul_ps(_v, inv_lenxxxx);
771 const __m128i cond = _mm_castps_si128(
772 _mm_cmple_ps(OZZ_SSE_SPLAT_F(sq_len, 0), _mm_setzero_ps()));
773 const __m128 cfalse = _mm_movelh_ps(norm, _mm_movehl_ps(_v, _v));
774 return OZZ_SSE_SELECT_F(cond, _safe, cfalse);
775}
776
777OZZ_INLINE SimdFloat4 NormalizeSafeEst3(_SimdFloat4 _v, _SimdFloat4 _safe) {
778 // assert(AreAllTrue1(IsNormalizedEst3(_safe)) && "_safe is not normalized");
779 __m128 sq_len;
780 OZZ_SSE_DOT3_F(_v, _v, sq_len);
781 const __m128 inv_len = _mm_rsqrt_ss(sq_len);
782 const __m128 vwxyz = OZZ_SHUFFLE_PS1(_v, _MM_SHUFFLE(0, 1, 2, 3));
783 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
784 const __m128 normwxyz = _mm_move_ss(_mm_mul_ps(vwxyz, inv_lenxxxx), vwxyz);
785 const __m128i cond = _mm_castps_si128(
786 _mm_cmple_ps(OZZ_SSE_SPLAT_F(sq_len, 0), _mm_setzero_ps()));
787 const __m128 cfalse = OZZ_SHUFFLE_PS1(normwxyz, _MM_SHUFFLE(0, 1, 2, 3));
788 return OZZ_SSE_SELECT_F(cond, _safe, cfalse);
789}
790
791OZZ_INLINE SimdFloat4 NormalizeSafeEst4(_SimdFloat4 _v, _SimdFloat4 _safe) {
792 // assert(AreAllTrue1(IsNormalizedEst4(_safe)) && "_safe is not normalized");
793 __m128 sq_len;
794 OZZ_SSE_DOT4_F(_v, _v, sq_len);
795 const __m128 inv_len = _mm_rsqrt_ss(sq_len);
796 const __m128 inv_lenxxxx = OZZ_SSE_SPLAT_F(inv_len, 0);
797 const __m128i cond = _mm_castps_si128(
798 _mm_cmple_ps(OZZ_SSE_SPLAT_F(sq_len, 0), _mm_setzero_ps()));
799 const __m128 cfalse = _mm_mul_ps(_v, inv_lenxxxx);
800 return OZZ_SSE_SELECT_F(cond, _safe, cfalse);
801}
802
803OZZ_INLINE SimdFloat4 Lerp(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _alpha) {
804 return OZZ_MADD(_alpha, _mm_sub_ps(_b, _a), _a);
805}
806
807OZZ_INLINE SimdFloat4 Min(_SimdFloat4 _a, _SimdFloat4 _b) {
808 return _mm_min_ps(_a, _b);
809}
810
811OZZ_INLINE SimdFloat4 Max(_SimdFloat4 _a, _SimdFloat4 _b) {
812 return _mm_max_ps(_a, _b);
813}
814
815OZZ_INLINE SimdFloat4 Min0(_SimdFloat4 _v) {
816 return _mm_min_ps(_mm_setzero_ps(), _v);
817}
818
819OZZ_INLINE SimdFloat4 Max0(_SimdFloat4 _v) {
820 return _mm_max_ps(_mm_setzero_ps(), _v);
821}
822
823OZZ_INLINE SimdFloat4 Clamp(_SimdFloat4 _a, _SimdFloat4 _v, _SimdFloat4 _b) {
824 return _mm_max_ps(_a, _mm_min_ps(_v, _b));
825}
826
827OZZ_INLINE SimdFloat4 Select(_SimdInt4 _b, _SimdFloat4 _true,
828 _SimdFloat4 _false) {
829 return OZZ_SSE_SELECT_F(_b, _true, _false);
830}
831
832OZZ_INLINE SimdInt4 CmpEq(_SimdFloat4 _a, _SimdFloat4 _b) {
833 return _mm_castps_si128(_mm_cmpeq_ps(_a, _b));
834}
835
836OZZ_INLINE SimdInt4 CmpNe(_SimdFloat4 _a, _SimdFloat4 _b) {
837 return _mm_castps_si128(_mm_cmpneq_ps(_a, _b));
838}
839
840OZZ_INLINE SimdInt4 CmpLt(_SimdFloat4 _a, _SimdFloat4 _b) {
841 return _mm_castps_si128(_mm_cmplt_ps(_a, _b));
842}
843
844OZZ_INLINE SimdInt4 CmpLe(_SimdFloat4 _a, _SimdFloat4 _b) {
845 return _mm_castps_si128(_mm_cmple_ps(_a, _b));
846}
847
848OZZ_INLINE SimdInt4 CmpGt(_SimdFloat4 _a, _SimdFloat4 _b) {
849 return _mm_castps_si128(_mm_cmpgt_ps(_a, _b));
850}
851
852OZZ_INLINE SimdInt4 CmpGe(_SimdFloat4 _a, _SimdFloat4 _b) {
853 return _mm_castps_si128(_mm_cmpge_ps(_a, _b));
854}
855
856OZZ_INLINE SimdFloat4 And(_SimdFloat4 _a, _SimdFloat4 _b) {
857 return _mm_and_ps(_a, _b);
858}
859
860OZZ_INLINE SimdFloat4 Or(_SimdFloat4 _a, _SimdFloat4 _b) {
861 return _mm_or_ps(_a, _b);
862}
863
864OZZ_INLINE SimdFloat4 Xor(_SimdFloat4 _a, _SimdFloat4 _b) {
865 return _mm_xor_ps(_a, _b);
866}
867
868OZZ_INLINE SimdFloat4 And(_SimdFloat4 _a, _SimdInt4 _b) {
869 return _mm_and_ps(_a, _mm_castsi128_ps(_b));
870}
871
872OZZ_INLINE SimdFloat4 AndNot(_SimdFloat4 _a, _SimdInt4 _b) {
873 return _mm_andnot_ps(_mm_castsi128_ps(_b), _a);
874}
875
876OZZ_INLINE SimdFloat4 Or(_SimdFloat4 _a, _SimdInt4 _b) {
877 return _mm_or_ps(_a, _mm_castsi128_ps(_b));
878}
879
880OZZ_INLINE SimdFloat4 Xor(_SimdFloat4 _a, _SimdInt4 _b) {
881 return _mm_xor_ps(_a, _mm_castsi128_ps(_b));
882}
883
884OZZ_INLINE SimdFloat4 Cos(_SimdFloat4 _v) {
885 return _mm_set_ps(std::cos(GetW(_v)), std::cos(GetZ(_v)), std::cos(GetY(_v)),
886 std::cos(GetX(_v)));
887}
888
889OZZ_INLINE SimdFloat4 CosX(_SimdFloat4 _v) {
890 return _mm_move_ss(_v, _mm_set_ps1(std::cos(GetX(_v))));
891}
892
893OZZ_INLINE SimdFloat4 ACos(_SimdFloat4 _v) {
894 return _mm_set_ps(std::acos(GetW(_v)), std::acos(GetZ(_v)),
895 std::acos(GetY(_v)), std::acos(GetX(_v)));
896}
897
898OZZ_INLINE SimdFloat4 ACosX(_SimdFloat4 _v) {
899 return _mm_move_ss(_v, _mm_set_ps1(std::acos(GetX(_v))));
900}
901
902OZZ_INLINE SimdFloat4 Sin(_SimdFloat4 _v) {
903 return _mm_set_ps(std::sin(GetW(_v)), std::sin(GetZ(_v)), std::sin(GetY(_v)),
904 std::sin(GetX(_v)));
905}
906
907OZZ_INLINE SimdFloat4 SinX(_SimdFloat4 _v) {
908 return _mm_move_ss(_v, _mm_set_ps1(std::sin(GetX(_v))));
909}
910
911OZZ_INLINE SimdFloat4 ASin(_SimdFloat4 _v) {
912 return _mm_set_ps(std::asin(GetW(_v)), std::asin(GetZ(_v)),
913 std::asin(GetY(_v)), std::asin(GetX(_v)));
914}
915
916OZZ_INLINE SimdFloat4 ASinX(_SimdFloat4 _v) {
917 return _mm_move_ss(_v, _mm_set_ps1(std::asin(GetX(_v))));
918}
919
920OZZ_INLINE SimdFloat4 Tan(_SimdFloat4 _v) {
921 return _mm_set_ps(std::tan(GetW(_v)), std::tan(GetZ(_v)), std::tan(GetY(_v)),
922 std::tan(GetX(_v)));
923}
924
925OZZ_INLINE SimdFloat4 TanX(_SimdFloat4 _v) {
926 return _mm_move_ss(_v, _mm_set_ps1(std::tan(GetX(_v))));
927}
928
929OZZ_INLINE SimdFloat4 ATan(_SimdFloat4 _v) {
930 return _mm_set_ps(std::atan(GetW(_v)), std::atan(GetZ(_v)),
931 std::atan(GetY(_v)), std::atan(GetX(_v)));
932}
933
934OZZ_INLINE SimdFloat4 ATanX(_SimdFloat4 _v) {
935 return _mm_move_ss(_v, _mm_set_ps1(std::atan(GetX(_v))));
936}
937
938namespace simd_int4 {
939
940OZZ_INLINE SimdInt4 zero() { return _mm_setzero_si128(); }
941
942OZZ_INLINE SimdInt4 one() {
943 const __m128i zero = _mm_setzero_si128();
944 return _mm_sub_epi32(zero, _mm_cmpeq_epi32(zero, zero));
945}
946
947OZZ_INLINE SimdInt4 x_axis() {
948 const __m128i zero = _mm_setzero_si128();
949 return _mm_srli_si128(_mm_sub_epi32(zero, _mm_cmpeq_epi32(zero, zero)), 12);
950}
951
952OZZ_INLINE SimdInt4 y_axis() {
953 const __m128i zero = _mm_setzero_si128();
954 return _mm_slli_si128(
955 _mm_srli_si128(_mm_sub_epi32(zero, _mm_cmpeq_epi32(zero, zero)), 12), 4);
956}
957
958OZZ_INLINE SimdInt4 z_axis() {
959 const __m128i zero = _mm_setzero_si128();
960 return _mm_slli_si128(
961 _mm_srli_si128(_mm_sub_epi32(zero, _mm_cmpeq_epi32(zero, zero)), 12), 8);
962}
963
964OZZ_INLINE SimdInt4 w_axis() {
965 const __m128i zero = _mm_setzero_si128();
966 return _mm_slli_si128(_mm_sub_epi32(zero, _mm_cmpeq_epi32(zero, zero)), 12);
967}
968
969OZZ_INLINE SimdInt4 all_true() {
970 const __m128i zero = _mm_setzero_si128();
971 return _mm_cmpeq_epi32(zero, zero);
972}
973
974OZZ_INLINE SimdInt4 all_false() { return _mm_setzero_si128(); }
975
976OZZ_INLINE SimdInt4 mask_sign() {
977 const __m128i zero = _mm_setzero_si128();
978 return _mm_slli_epi32(_mm_cmpeq_epi32(zero, zero), 31);
979}
980
981OZZ_INLINE SimdInt4 mask_sign_xyz() {
982 const __m128i zero = _mm_setzero_si128();
983 return _mm_srli_si128(_mm_slli_epi32(_mm_cmpeq_epi32(zero, zero), 31), 4);
984}
985
986OZZ_INLINE SimdInt4 mask_sign_w() {
987 const __m128i zero = _mm_setzero_si128();
988 return _mm_slli_si128(_mm_slli_epi32(_mm_cmpeq_epi32(zero, zero), 31), 12);
989}
990
991OZZ_INLINE SimdInt4 mask_not_sign() {
992 const __m128i zero = _mm_setzero_si128();
993 return _mm_srli_epi32(_mm_cmpeq_epi32(zero, zero), 1);
994}
995
996OZZ_INLINE SimdInt4 mask_ffff() {
997 const __m128i zero = _mm_setzero_si128();
998 return _mm_cmpeq_epi32(zero, zero);
999}
1000OZZ_INLINE SimdInt4 mask_0000() { return _mm_setzero_si128(); }
1001
1002OZZ_INLINE SimdInt4 mask_fff0() {
1003 const __m128i zero = _mm_setzero_si128();
1004 return _mm_srli_si128(_mm_cmpeq_epi32(zero, zero), 4);
1005}
1006
1007OZZ_INLINE SimdInt4 mask_f000() {
1008 const __m128i zero = _mm_setzero_si128();
1009 return _mm_srli_si128(_mm_cmpeq_epi32(zero, zero), 12);
1010}
1011
1012OZZ_INLINE SimdInt4 mask_0f00() {
1013 const __m128i zero = _mm_setzero_si128();
1014 return _mm_srli_si128(_mm_slli_si128(_mm_cmpeq_epi32(zero, zero), 12), 8);
1015}
1016
1017OZZ_INLINE SimdInt4 mask_00f0() {
1018 const __m128i zero = _mm_setzero_si128();
1019 return _mm_srli_si128(_mm_slli_si128(_mm_cmpeq_epi32(zero, zero), 12), 4);
1020}
1021
1022OZZ_INLINE SimdInt4 mask_000f() {
1023 const __m128i zero = _mm_setzero_si128();
1024 return _mm_slli_si128(_mm_cmpeq_epi32(zero, zero), 12);
1025}
1026
1027OZZ_INLINE SimdInt4 Load(int _x, int _y, int _z, int _w) {
1028 return _mm_set_epi32(_w, _z, _y, _x);
1029}
1030
1031OZZ_INLINE SimdInt4 LoadX(int _x) { return _mm_set_epi32(0, 0, 0, _x); }
1032
1033OZZ_INLINE SimdInt4 Load1(int _x) { return _mm_set1_epi32(_x); }
1034
1035OZZ_INLINE SimdInt4 Load(bool _x, bool _y, bool _z, bool _w) {
1036 return _mm_sub_epi32(_mm_setzero_si128(), _mm_set_epi32(_w, _z, _y, _x));
1037}
1038
1039OZZ_INLINE SimdInt4 LoadX(bool _x) {
1040 return _mm_sub_epi32(_mm_setzero_si128(), _mm_set_epi32(0, 0, 0, _x));
1041}
1042
1043OZZ_INLINE SimdInt4 Load1(bool _x) {
1044 return _mm_sub_epi32(_mm_setzero_si128(), _mm_set1_epi32(_x));
1045}
1046
1047OZZ_INLINE SimdInt4 LoadPtr(const int* _i) {
1048 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1049 return _mm_load_si128(reinterpret_cast<const __m128i*>(_i));
1050}
1051
1052OZZ_INLINE SimdInt4 LoadXPtr(const int* _i) {
1053 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1054 return _mm_cvtsi32_si128(*_i);
1055}
1056
1057OZZ_INLINE SimdInt4 Load1Ptr(const int* _i) {
1058 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1059 return _mm_shuffle_epi32(
1060 _mm_loadl_epi64(reinterpret_cast<const __m128i*>(_i)),
1061 _MM_SHUFFLE(0, 0, 0, 0));
1062}
1063
1064OZZ_INLINE SimdInt4 Load2Ptr(const int* _i) {
1065 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1066 return _mm_loadl_epi64(reinterpret_cast<const __m128i*>(_i));
1067}
1068
1069OZZ_INLINE SimdInt4 Load3Ptr(const int* _i) {
1070 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1071 return _mm_set_epi32(0, _i[2], _i[1], _i[0]);
1072}
1073
1074OZZ_INLINE SimdInt4 LoadPtrU(const int* _i) {
1075 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1076 return _mm_loadu_si128(reinterpret_cast<const __m128i*>(_i));
1077}
1078
1079OZZ_INLINE SimdInt4 LoadXPtrU(const int* _i) {
1080 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1081 return _mm_cvtsi32_si128(*_i);
1082}
1083
1084OZZ_INLINE SimdInt4 Load1PtrU(const int* _i) {
1085 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1086 return _mm_set1_epi32(*_i);
1087}
1088
1089OZZ_INLINE SimdInt4 Load2PtrU(const int* _i) {
1090 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1091 return _mm_set_epi32(0, 0, _i[1], _i[0]);
1092}
1093
1094OZZ_INLINE SimdInt4 Load3PtrU(const int* _i) {
1095 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1096 return _mm_set_epi32(0, _i[2], _i[1], _i[0]);
1097}
1098
1099OZZ_INLINE SimdInt4 FromFloatRound(_SimdFloat4 _f) {
1100 return _mm_cvtps_epi32(_f);
1101}
1102
1103OZZ_INLINE SimdInt4 FromFloatTrunc(_SimdFloat4 _f) {
1104 return _mm_cvttps_epi32(_f);
1105}
1106} // namespace simd_int4
1107
1108OZZ_INLINE int GetX(_SimdInt4 _v) { return _mm_cvtsi128_si32(_v); }
1109
1110OZZ_INLINE int GetY(_SimdInt4 _v) {
1111 return _mm_cvtsi128_si32(OZZ_SSE_SPLAT_I(_v, 1));
1112}
1113
1114OZZ_INLINE int GetZ(_SimdInt4 _v) {
1115 return _mm_cvtsi128_si32(_mm_unpackhi_epi32(_v, _v));
1116}
1117
1118OZZ_INLINE int GetW(_SimdInt4 _v) {
1119 return _mm_cvtsi128_si32(OZZ_SSE_SPLAT_I(_v, 3));
1120}
1121
1122OZZ_INLINE SimdInt4 SetX(_SimdInt4 _v, _SimdInt4 _i) {
1123 return _mm_castps_si128(
1124 _mm_move_ss(_mm_castsi128_ps(_v), _mm_castsi128_ps(_i)));
1125}
1126
1127OZZ_INLINE SimdInt4 SetY(_SimdInt4 _v, _SimdInt4 _i) {
1128 const __m128 xfnn = _mm_castsi128_ps(_mm_unpacklo_epi32(_v, _i));
1129 return _mm_castps_si128(
1130 _mm_shuffle_ps(xfnn, _mm_castsi128_ps(_v), _MM_SHUFFLE(3, 2, 1, 0)));
1131}
1132
1133OZZ_INLINE SimdInt4 SetZ(_SimdInt4 _v, _SimdInt4 _i) {
1134 const __m128 ffww = _mm_shuffle_ps(_mm_castsi128_ps(_i), _mm_castsi128_ps(_v),
1135 _MM_SHUFFLE(3, 3, 0, 0));
1136 return _mm_castps_si128(
1137 _mm_shuffle_ps(_mm_castsi128_ps(_v), ffww, _MM_SHUFFLE(2, 0, 1, 0)));
1138}
1139
1140OZZ_INLINE SimdInt4 SetW(_SimdInt4 _v, _SimdInt4 _i) {
1141 const __m128 ffzz = _mm_shuffle_ps(_mm_castsi128_ps(_i), _mm_castsi128_ps(_v),
1142 _MM_SHUFFLE(2, 2, 0, 0));
1143 return _mm_castps_si128(
1144 _mm_shuffle_ps(_mm_castsi128_ps(_v), ffzz, _MM_SHUFFLE(0, 2, 1, 0)));
1145}
1146
1147OZZ_INLINE SimdInt4 SetI(_SimdInt4 _v, _SimdInt4 _i, int _ith) {
1148 assert(_ith >= 0 && _ith <= 3 && "Invalid index, out of range.");
1149 union {
1150 SimdInt4 ret;
1151 int af[4];
1152 } u = {_v};
1153 u.af[_ith] = GetX(_i);
1154 return u.ret;
1155}
1156
1157OZZ_INLINE void StorePtr(_SimdInt4 _v, int* _i) {
1158 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1159 _mm_store_si128(reinterpret_cast<__m128i*>(_i), _v);
1160}
1161
1162OZZ_INLINE void Store1Ptr(_SimdInt4 _v, int* _i) {
1163 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1164 *_i = _mm_cvtsi128_si32(_v);
1165}
1166
1167OZZ_INLINE void Store2Ptr(_SimdInt4 _v, int* _i) {
1168 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1169 _i[0] = _mm_cvtsi128_si32(_v);
1170 _i[1] = _mm_cvtsi128_si32(OZZ_SSE_SPLAT_I(_v, 1));
1171}
1172
1173OZZ_INLINE void Store3Ptr(_SimdInt4 _v, int* _i) {
1174 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1175 _i[0] = _mm_cvtsi128_si32(_v);
1176 _i[1] = _mm_cvtsi128_si32(OZZ_SSE_SPLAT_I(_v, 1));
1177 _i[2] = _mm_cvtsi128_si32(_mm_unpackhi_epi32(_v, _v));
1178}
1179
1180OZZ_INLINE void StorePtrU(_SimdInt4 _v, int* _i) {
1181 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1182 _mm_storeu_si128(reinterpret_cast<__m128i*>(_i), _v);
1183}
1184
1185OZZ_INLINE void Store1PtrU(_SimdInt4 _v, int* _i) {
1186 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1187 *_i = _mm_cvtsi128_si32(_v);
1188}
1189
1190OZZ_INLINE void Store2PtrU(_SimdInt4 _v, int* _i) {
1191 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1192 _i[0] = _mm_cvtsi128_si32(_v);
1193 _i[1] = _mm_cvtsi128_si32(OZZ_SSE_SPLAT_I(_v, 1));
1194}
1195
1196OZZ_INLINE void Store3PtrU(_SimdInt4 _v, int* _i) {
1197 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1198 _i[0] = _mm_cvtsi128_si32(_v);
1199 _i[1] = _mm_cvtsi128_si32(OZZ_SSE_SPLAT_I(_v, 1));
1200 _i[2] = _mm_cvtsi128_si32(_mm_unpackhi_epi32(_v, _v));
1201}
1202
1203OZZ_INLINE SimdInt4 SplatX(_SimdInt4 _a) { return OZZ_SSE_SPLAT_I(_a, 0); }
1204
1205OZZ_INLINE SimdInt4 SplatY(_SimdInt4 _a) { return OZZ_SSE_SPLAT_I(_a, 1); }
1206
1207OZZ_INLINE SimdInt4 SplatZ(_SimdInt4 _a) { return OZZ_SSE_SPLAT_I(_a, 2); }
1208
1209OZZ_INLINE SimdInt4 SplatW(_SimdInt4 _a) { return OZZ_SSE_SPLAT_I(_a, 3); }
1210
1211template <size_t _X, size_t _Y, size_t _Z, size_t _W>
1212OZZ_INLINE SimdInt4 Swizzle(_SimdInt4 _v) {
1213 static_assert(_X <= 3 && _Y <= 3 && _Z <= 3 && _W <= 3,
1214 "Indices must be between 0 and 3");
1215 return _mm_shuffle_epi32(_v, _MM_SHUFFLE(_W, _Z, _Y, _X));
1216}
1217
1218template <>
1219OZZ_INLINE SimdInt4 Swizzle<0, 1, 2, 3>(_SimdInt4 _v) {
1220 return _v;
1221}
1222
1223OZZ_INLINE int MoveMask(_SimdInt4 _v) {
1224 return _mm_movemask_ps(_mm_castsi128_ps(_v));
1225}
1226
1227OZZ_INLINE bool AreAllTrue(_SimdInt4 _v) {
1228 return _mm_movemask_ps(_mm_castsi128_ps(_v)) == 0xf;
1229}
1230
1231OZZ_INLINE bool AreAllTrue3(_SimdInt4 _v) {
1232 return (_mm_movemask_ps(_mm_castsi128_ps(_v)) & 0x7) == 0x7;
1233}
1234
1235OZZ_INLINE bool AreAllTrue2(_SimdInt4 _v) {
1236 return (_mm_movemask_ps(_mm_castsi128_ps(_v)) & 0x3) == 0x3;
1237}
1238
1239OZZ_INLINE bool AreAllTrue1(_SimdInt4 _v) {
1240 return (_mm_movemask_ps(_mm_castsi128_ps(_v)) & 0x1) == 0x1;
1241}
1242
1243OZZ_INLINE bool AreAllFalse(_SimdInt4 _v) {
1244 return _mm_movemask_ps(_mm_castsi128_ps(_v)) == 0;
1245}
1246
1247OZZ_INLINE bool AreAllFalse3(_SimdInt4 _v) {
1248 return (_mm_movemask_ps(_mm_castsi128_ps(_v)) & 0x7) == 0;
1249}
1250
1251OZZ_INLINE bool AreAllFalse2(_SimdInt4 _v) {
1252 return (_mm_movemask_ps(_mm_castsi128_ps(_v)) & 0x3) == 0;
1253}
1254
1255OZZ_INLINE bool AreAllFalse1(_SimdInt4 _v) {
1256 return (_mm_movemask_ps(_mm_castsi128_ps(_v)) & 0x1) == 0;
1257}
1258
1259OZZ_INLINE SimdInt4 HAdd2(_SimdInt4 _v) {
1260 const __m128i hadd = _mm_add_epi32(_v, OZZ_SSE_SPLAT_I(_v, 1));
1261 return _mm_castps_si128(
1262 _mm_move_ss(_mm_castsi128_ps(_v), _mm_castsi128_ps(hadd)));
1263}
1264
1265OZZ_INLINE SimdInt4 HAdd3(_SimdInt4 _v) {
1266 const __m128i hadd = _mm_add_epi32(_mm_add_epi32(_v, OZZ_SSE_SPLAT_I(_v, 1)),
1267 _mm_unpackhi_epi32(_v, _v));
1268 return _mm_castps_si128(
1269 _mm_move_ss(_mm_castsi128_ps(_v), _mm_castsi128_ps(hadd)));
1270}
1271
1272OZZ_INLINE SimdInt4 HAdd4(_SimdInt4 _v) {
1273 const __m128 v = _mm_castsi128_ps(_v);
1274 const __m128i haddxyzw =
1275 _mm_add_epi32(_v, _mm_castps_si128(_mm_movehl_ps(v, v)));
1276 return _mm_castps_si128(_mm_move_ss(
1277 v,
1278 _mm_castsi128_ps(_mm_add_epi32(haddxyzw, OZZ_SSE_SPLAT_I(haddxyzw, 1)))));
1279}
1280
1281OZZ_INLINE SimdInt4 Abs(_SimdInt4 _v) {
1282#ifdef OZZ_SIMD_SSSE3
1283 return _mm_abs_epi32(_v);
1284#else // OZZ_SIMD_SSSE3
1285 const __m128i zero = _mm_setzero_si128();
1286 return OZZ_SSE_SELECT_I(_mm_cmplt_epi32(_v, zero), _mm_sub_epi32(zero, _v),
1287 _v);
1288#endif // OZZ_SIMD_SSSE3
1289}
1290
1291OZZ_INLINE SimdInt4 Sign(_SimdInt4 _v) {
1292 return _mm_slli_epi32(_mm_srli_epi32(_v, 31), 31);
1293}
1294
1295OZZ_INLINE SimdInt4 Min(_SimdInt4 _a, _SimdInt4 _b) {
1296#ifdef OZZ_SIMD_SSE4_1
1297 return _mm_min_epi32(_a, _b);
1298#else // OZZ_SIMD_SSE4_1
1299 return OZZ_SSE_SELECT_I(_mm_cmplt_epi32(_a, _b), _a, _b);
1300#endif // OZZ_SIMD_SSE4_1
1301}
1302
1303OZZ_INLINE SimdInt4 Max(_SimdInt4 _a, _SimdInt4 _b) {
1304#ifdef OZZ_SIMD_SSE4_1
1305 return _mm_max_epi32(_a, _b);
1306#else // OZZ_SIMD_SSE4_1
1307 return OZZ_SSE_SELECT_I(_mm_cmpgt_epi32(_a, _b), _a, _b);
1308#endif // OZZ_SIMD_SSE4_1
1309}
1310
1311OZZ_INLINE SimdInt4 Min0(_SimdInt4 _v) {
1312 const __m128i zero = _mm_setzero_si128();
1313#ifdef OZZ_SIMD_SSE4_1
1314 return _mm_min_epi32(zero, _v);
1315#else // OZZ_SIMD_SSE4_1
1316 return OZZ_SSE_SELECT_I(_mm_cmplt_epi32(zero, _v), zero, _v);
1317#endif // OZZ_SIMD_SSE4_1
1318}
1319
1320OZZ_INLINE SimdInt4 Max0(_SimdInt4 _v) {
1321 const __m128i zero = _mm_setzero_si128();
1322#ifdef OZZ_SIMD_SSE4_1
1323 return _mm_max_epi32(zero, _v);
1324#else // OZZ_SIMD_SSE4_1
1325 return OZZ_SSE_SELECT_I(_mm_cmpgt_epi32(zero, _v), zero, _v);
1326#endif // OZZ_SIMD_SSE4_1
1327}
1328
1329OZZ_INLINE SimdInt4 Clamp(_SimdInt4 _a, _SimdInt4 _v, _SimdInt4 _b) {
1330#ifdef OZZ_SIMD_SSE4_1
1331 return _mm_min_epi32(_mm_max_epi32(_a, _v), _b);
1332#else // OZZ_SIMD_SSE4_1
1333 const __m128i min = OZZ_SSE_SELECT_I(_mm_cmplt_epi32(_v, _b), _v, _b);
1334 return OZZ_SSE_SELECT_I(_mm_cmpgt_epi32(_a, min), _a, min);
1335#endif // OZZ_SIMD_SSE4_1
1336}
1337
1338OZZ_INLINE SimdInt4 Select(_SimdInt4 _b, _SimdInt4 _true, _SimdInt4 _false) {
1339 return OZZ_SSE_SELECT_I(_b, _true, _false);
1340}
1341
1342OZZ_INLINE SimdInt4 And(_SimdInt4 _a, _SimdInt4 _b) {
1343 return _mm_and_si128(_a, _b);
1344}
1345
1346OZZ_INLINE SimdInt4 AndNot(_SimdInt4 _a, _SimdInt4 _b) {
1347 return _mm_andnot_si128(_b, _a);
1348}
1349
1350OZZ_INLINE SimdInt4 Or(_SimdInt4 _a, _SimdInt4 _b) {
1351 return _mm_or_si128(_a, _b);
1352}
1353
1354OZZ_INLINE SimdInt4 Xor(_SimdInt4 _a, _SimdInt4 _b) {
1355 return _mm_xor_si128(_a, _b);
1356}
1357
1358OZZ_INLINE SimdInt4 Not(_SimdInt4 _v) {
1359 return _mm_xor_si128(_v, _mm_cmpeq_epi32(_v, _v));
1360}
1361
1362OZZ_INLINE SimdInt4 ShiftL(_SimdInt4 _v, int _bits) {
1363 return _mm_slli_epi32(_v, _bits);
1364}
1365
1366OZZ_INLINE SimdInt4 ShiftR(_SimdInt4 _v, int _bits) {
1367 return _mm_srai_epi32(_v, _bits);
1368}
1369
1370OZZ_INLINE SimdInt4 ShiftRu(_SimdInt4 _v, int _bits) {
1371 return _mm_srli_epi32(_v, _bits);
1372}
1373
1374OZZ_INLINE SimdInt4 CmpEq(_SimdInt4 _a, _SimdInt4 _b) {
1375 return _mm_cmpeq_epi32(_a, _b);
1376}
1377
1378OZZ_INLINE SimdInt4 CmpNe(_SimdInt4 _a, _SimdInt4 _b) {
1379 const __m128i eq = _mm_cmpeq_epi32(_a, _b);
1380 return _mm_xor_si128(eq, _mm_cmpeq_epi32(_a, _a));
1381}
1382
1383OZZ_INLINE SimdInt4 CmpLt(_SimdInt4 _a, _SimdInt4 _b) {
1384 return _mm_cmpgt_epi32(_b, _a);
1385}
1386
1387OZZ_INLINE SimdInt4 CmpLe(_SimdInt4 _a, _SimdInt4 _b) {
1388 const __m128i gt = _mm_cmpgt_epi32(_a, _b);
1389 return _mm_xor_si128(gt, _mm_cmpeq_epi32(_a, _a));
1390}
1391
1392OZZ_INLINE SimdInt4 CmpGt(_SimdInt4 _a, _SimdInt4 _b) {
1393 return _mm_cmpgt_epi32(_a, _b);
1394}
1395
1396OZZ_INLINE SimdInt4 CmpGe(_SimdInt4 _a, _SimdInt4 _b) {
1397 const __m128i lt = _mm_cmpgt_epi32(_b, _a);
1398 return _mm_xor_si128(lt, _mm_cmpeq_epi32(_a, _a));
1399}
1400
1401OZZ_INLINE Float4x4 Float4x4::identity() {
1402 const __m128i zero = _mm_setzero_si128();
1403 const __m128i ffff = _mm_cmpeq_epi32(zero, zero);
1404 const __m128i one = _mm_srli_epi32(_mm_slli_epi32(ffff, 25), 2);
1405 const __m128i x = _mm_srli_si128(one, 12);
1406 const Float4x4 ret = {{_mm_castsi128_ps(x),
1407 _mm_castsi128_ps(_mm_slli_si128(x, 4)),
1408 _mm_castsi128_ps(_mm_slli_si128(x, 8)),
1409 _mm_castsi128_ps(_mm_slli_si128(one, 12))}};
1410 return ret;
1411}
1412
1413OZZ_INLINE Float4x4 Transpose(const Float4x4& _m) {
1414 const __m128 tmp0 = _mm_unpacklo_ps(_m.cols[0], _m.cols[2]);
1415 const __m128 tmp1 = _mm_unpacklo_ps(_m.cols[1], _m.cols[3]);
1416 const __m128 tmp2 = _mm_unpackhi_ps(_m.cols[0], _m.cols[2]);
1417 const __m128 tmp3 = _mm_unpackhi_ps(_m.cols[1], _m.cols[3]);
1418 const Float4x4 ret = {
1419 {_mm_unpacklo_ps(tmp0, tmp1), _mm_unpackhi_ps(tmp0, tmp1),
1420 _mm_unpacklo_ps(tmp2, tmp3), _mm_unpackhi_ps(tmp2, tmp3)}};
1421 return ret;
1422}
1423
1424inline Float4x4 Invert(const Float4x4& _m, SimdInt4* _invertible) {
1425 const __m128 _t0 =
1426 _mm_shuffle_ps(_m.cols[0], _m.cols[1], _MM_SHUFFLE(1, 0, 1, 0));
1427 const __m128 _t1 =
1428 _mm_shuffle_ps(_m.cols[2], _m.cols[3], _MM_SHUFFLE(1, 0, 1, 0));
1429 const __m128 _t2 =
1430 _mm_shuffle_ps(_m.cols[0], _m.cols[1], _MM_SHUFFLE(3, 2, 3, 2));
1431 const __m128 _t3 =
1432 _mm_shuffle_ps(_m.cols[2], _m.cols[3], _MM_SHUFFLE(3, 2, 3, 2));
1433 const __m128 c0 = _mm_shuffle_ps(_t0, _t1, _MM_SHUFFLE(2, 0, 2, 0));
1434 const __m128 c1 = _mm_shuffle_ps(_t1, _t0, _MM_SHUFFLE(3, 1, 3, 1));
1435 const __m128 c2 = _mm_shuffle_ps(_t2, _t3, _MM_SHUFFLE(2, 0, 2, 0));
1436 const __m128 c3 = _mm_shuffle_ps(_t3, _t2, _MM_SHUFFLE(3, 1, 3, 1));
1437
1438 __m128 minor0, minor1, minor2, minor3, tmp1, tmp2;
1439 tmp1 = _mm_mul_ps(c2, c3);
1440 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0xB1);
1441 minor0 = _mm_mul_ps(c1, tmp1);
1442 minor1 = _mm_mul_ps(c0, tmp1);
1443 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0x4E);
1444 minor0 = OZZ_MSUB(c1, tmp1, minor0);
1445 minor1 = OZZ_MSUB(c0, tmp1, minor1);
1446 minor1 = OZZ_SHUFFLE_PS1(minor1, 0x4E);
1447
1448 tmp1 = _mm_mul_ps(c1, c2);
1449 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0xB1);
1450 minor0 = OZZ_MADD(c3, tmp1, minor0);
1451 minor3 = _mm_mul_ps(c0, tmp1);
1452 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0x4E);
1453 minor0 = OZZ_NMADD(c3, tmp1, minor0);
1454 minor3 = OZZ_MSUB(c0, tmp1, minor3);
1455 minor3 = OZZ_SHUFFLE_PS1(minor3, 0x4E);
1456
1457 tmp1 = _mm_mul_ps(OZZ_SHUFFLE_PS1(c1, 0x4E), c3);
1458 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0xB1);
1459 tmp2 = OZZ_SHUFFLE_PS1(c2, 0x4E);
1460 minor0 = OZZ_MADD(tmp2, tmp1, minor0);
1461 minor2 = _mm_mul_ps(c0, tmp1);
1462 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0x4E);
1463 minor0 = OZZ_NMADD(tmp2, tmp1, minor0);
1464 minor2 = OZZ_MSUB(c0, tmp1, minor2);
1465 minor2 = OZZ_SHUFFLE_PS1(minor2, 0x4E);
1466
1467 tmp1 = _mm_mul_ps(c0, c1);
1468 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0xB1);
1469 minor2 = OZZ_MADD(c3, tmp1, minor2);
1470 minor3 = OZZ_MSUB(tmp2, tmp1, minor3);
1471 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0x4E);
1472 minor2 = OZZ_MSUB(c3, tmp1, minor2);
1473 minor3 = OZZ_NMADD(tmp2, tmp1, minor3);
1474
1475 tmp1 = _mm_mul_ps(c0, c3);
1476 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0xB1);
1477 minor1 = OZZ_NMADD(tmp2, tmp1, minor1);
1478 minor2 = OZZ_MADD(c1, tmp1, minor2);
1479 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0x4E);
1480 minor1 = OZZ_MADD(tmp2, tmp1, minor1);
1481 minor2 = OZZ_NMADD(c1, tmp1, minor2);
1482
1483 tmp1 = _mm_mul_ps(c0, tmp2);
1484 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0xB1);
1485 minor1 = OZZ_MADD(c3, tmp1, minor1);
1486 minor3 = OZZ_NMADD(c1, tmp1, minor3);
1487 tmp1 = OZZ_SHUFFLE_PS1(tmp1, 0x4E);
1488 minor1 = OZZ_NMADD(c3, tmp1, minor1);
1489 minor3 = OZZ_MADD(c1, tmp1, minor3);
1490
1491 __m128 det;
1492 det = _mm_mul_ps(c0, minor0);
1493 det = _mm_add_ps(OZZ_SHUFFLE_PS1(det, 0x4E), det);
1494 det = _mm_add_ss(OZZ_SHUFFLE_PS1(det, 0xB1), det);
1495 const SimdInt4 invertible = CmpNe(det, simd_float4::zero());
1496 assert((_invertible || AreAllTrue1(invertible)) &&
1497 "Matrix is not invertible");
1498 if (_invertible != nullptr) {
1499 *_invertible = invertible;
1500 }
1501 tmp1 = OZZ_SSE_SELECT_F(invertible, RcpEstNR(det), simd_float4::zero());
1502 det = OZZ_NMADDX(det, _mm_mul_ss(tmp1, tmp1), _mm_add_ss(tmp1, tmp1));
1503 det = OZZ_SHUFFLE_PS1(det, 0x00);
1504
1505 // Copy the final columns
1506 const Float4x4 ret = {{_mm_mul_ps(det, minor0), _mm_mul_ps(det, minor1),
1507 _mm_mul_ps(det, minor2), _mm_mul_ps(det, minor3)}};
1508 return ret;
1509}
1510
1511Float4x4 Float4x4::Translation(_SimdFloat4 _v) {
1512 const __m128i zero = _mm_setzero_si128();
1513 const __m128i ffff = _mm_cmpeq_epi32(zero, zero);
1514 const __m128i mask000f = _mm_slli_si128(ffff, 12);
1515 const __m128i one = _mm_srli_epi32(_mm_slli_epi32(ffff, 25), 2);
1516 const __m128i x = _mm_srli_si128(one, 12);
1517 const Float4x4 ret = {
1518 {_mm_castsi128_ps(x), _mm_castsi128_ps(_mm_slli_si128(x, 4)),
1519 _mm_castsi128_ps(_mm_slli_si128(x, 8)),
1520 OZZ_SSE_SELECT_F(mask000f, _mm_castsi128_ps(one), _v)}};
1521 return ret;
1522} // math
1523
1524Float4x4 Float4x4::Scaling(_SimdFloat4 _v) {
1525 const __m128i zero = _mm_setzero_si128();
1526 const __m128i ffff = _mm_cmpeq_epi32(zero, zero);
1527 const __m128i if000 = _mm_srli_si128(ffff, 12);
1528 const __m128i ione = _mm_srli_epi32(_mm_slli_epi32(ffff, 25), 2);
1529 const Float4x4 ret = {
1530 {_mm_and_ps(_v, _mm_castsi128_ps(if000)),
1531 _mm_and_ps(_v, _mm_castsi128_ps(_mm_slli_si128(if000, 4))),
1532 _mm_and_ps(_v, _mm_castsi128_ps(_mm_slli_si128(if000, 8))),
1533 _mm_castsi128_ps(_mm_slli_si128(ione, 12))}};
1534 return ret;
1535} // math
1536
1537OZZ_INLINE Float4x4 Translate(const Float4x4& _m, _SimdFloat4 _v) {
1538 const __m128 a01 = OZZ_MADD(_m.cols[0], OZZ_SSE_SPLAT_F(_v, 0),
1539 _mm_mul_ps(_m.cols[1], OZZ_SSE_SPLAT_F(_v, 1)));
1540 const __m128 m3 = OZZ_MADD(_m.cols[2], OZZ_SSE_SPLAT_F(_v, 2), _m.cols[3]);
1541 const Float4x4 ret = {
1542 {_m.cols[0], _m.cols[1], _m.cols[2], _mm_add_ps(a01, m3)}};
1543 return ret;
1544}
1545
1546OZZ_INLINE Float4x4 Scale(const Float4x4& _m, _SimdFloat4 _v) {
1547 const Float4x4 ret = {{_mm_mul_ps(_m.cols[0], OZZ_SSE_SPLAT_F(_v, 0)),
1548 _mm_mul_ps(_m.cols[1], OZZ_SSE_SPLAT_F(_v, 1)),
1549 _mm_mul_ps(_m.cols[2], OZZ_SSE_SPLAT_F(_v, 2)),
1550 _m.cols[3]}};
1551 return ret;
1552}
1553
1554OZZ_INLINE Float4x4 ColumnMultiply(const Float4x4& _m, _SimdFloat4 _v) {
1555 const Float4x4 ret = {{_mm_mul_ps(_m.cols[0], _v), _mm_mul_ps(_m.cols[1], _v),
1556 _mm_mul_ps(_m.cols[2], _v),
1557 _mm_mul_ps(_m.cols[3], _v)}};
1558 return ret;
1559}
1560
1561inline SimdInt4 IsNormalized(const Float4x4& _m) {
1562 const __m128 max = _mm_set_ps1(1.f + kNormalizationToleranceSq);
1563 const __m128 min = _mm_set_ps1(1.f - kNormalizationToleranceSq);
1564
1565 const __m128 tmp0 = _mm_unpacklo_ps(_m.cols[0], _m.cols[2]);
1566 const __m128 tmp1 = _mm_unpacklo_ps(_m.cols[1], _m.cols[3]);
1567 const __m128 tmp2 = _mm_unpackhi_ps(_m.cols[0], _m.cols[2]);
1568 const __m128 tmp3 = _mm_unpackhi_ps(_m.cols[1], _m.cols[3]);
1569 const __m128 row0 = _mm_unpacklo_ps(tmp0, tmp1);
1570 const __m128 row1 = _mm_unpackhi_ps(tmp0, tmp1);
1571 const __m128 row2 = _mm_unpacklo_ps(tmp2, tmp3);
1572
1573 const __m128 dot =
1574 OZZ_MADD(row0, row0, OZZ_MADD(row1, row1, _mm_mul_ps(row2, row2)));
1575 const __m128 normalized =
1576 _mm_and_ps(_mm_cmplt_ps(dot, max), _mm_cmpgt_ps(dot, min));
1577 return _mm_castps_si128(
1578 _mm_and_ps(normalized, _mm_castsi128_ps(simd_int4::mask_fff0())));
1579}
1580
1581inline SimdInt4 IsNormalizedEst(const Float4x4& _m) {
1582 const __m128 max = _mm_set_ps1(1.f + kNormalizationToleranceEstSq);
1583 const __m128 min = _mm_set_ps1(1.f - kNormalizationToleranceEstSq);
1584
1585 const __m128 tmp0 = _mm_unpacklo_ps(_m.cols[0], _m.cols[2]);
1586 const __m128 tmp1 = _mm_unpacklo_ps(_m.cols[1], _m.cols[3]);
1587 const __m128 tmp2 = _mm_unpackhi_ps(_m.cols[0], _m.cols[2]);
1588 const __m128 tmp3 = _mm_unpackhi_ps(_m.cols[1], _m.cols[3]);
1589 const __m128 row0 = _mm_unpacklo_ps(tmp0, tmp1);
1590 const __m128 row1 = _mm_unpackhi_ps(tmp0, tmp1);
1591 const __m128 row2 = _mm_unpacklo_ps(tmp2, tmp3);
1592
1593 const __m128 dot =
1594 OZZ_MADD(row0, row0, OZZ_MADD(row1, row1, _mm_mul_ps(row2, row2)));
1595
1596 const __m128 normalized =
1597 _mm_and_ps(_mm_cmplt_ps(dot, max), _mm_cmpgt_ps(dot, min));
1598
1599 return _mm_castps_si128(
1600 _mm_and_ps(normalized, _mm_castsi128_ps(simd_int4::mask_fff0())));
1601}
1602
1603OZZ_INLINE SimdInt4 IsOrthogonal(const Float4x4& _m) {
1604 const __m128 max = _mm_set_ss(1.f + kNormalizationToleranceSq);
1605 const __m128 min = _mm_set_ss(1.f - kNormalizationToleranceSq);
1606 const __m128 zero = _mm_setzero_ps();
1607
1608 // Use simd_float4::zero() if one of the normalization fails. _m will then be
1609 // considered not orthogonal.
1610 const SimdFloat4 cross = NormalizeSafe3(Cross3(_m.cols[0], _m.cols[1]), zero);
1611 const SimdFloat4 at = NormalizeSafe3(_m.cols[2], zero);
1612
1613 SimdFloat4 dot;
1614 OZZ_SSE_DOT3_F(cross, at, dot);
1615 __m128 dotx000 = _mm_move_ss(zero, dot);
1616 return _mm_castps_si128(
1617 _mm_and_ps(_mm_cmplt_ss(dotx000, max), _mm_cmpgt_ss(dotx000, min)));
1618}
1619
1620inline SimdFloat4 ToQuaternion(const Float4x4& _m) {
1621 assert(AreAllTrue3(IsNormalizedEst(_m)));
1622 assert(AreAllTrue1(IsOrthogonal(_m)));
1623
1624 // Prepares constants.
1625 const __m128i zero = _mm_setzero_si128();
1626 const __m128i ffff = _mm_cmpeq_epi32(zero, zero);
1627 const __m128 half = _mm_set1_ps(0.5f);
1628 const __m128i mask_f000 = _mm_srli_si128(ffff, 12);
1629 const __m128i mask_000f = _mm_slli_si128(ffff, 12);
1630 const __m128 one =
1631 _mm_castsi128_ps(_mm_srli_epi32(_mm_slli_epi32(ffff, 25), 2));
1632 const __m128i mask_0f00 = _mm_slli_si128(mask_f000, 4);
1633 const __m128i mask_00f0 = _mm_slli_si128(mask_f000, 8);
1634
1635 const __m128 xx_yy = OZZ_SSE_SELECT_F(mask_0f00, _m.cols[1], _m.cols[0]);
1636 const __m128 xx_yy_0010 = OZZ_SHUFFLE_PS1(xx_yy, _MM_SHUFFLE(0, 0, 1, 0));
1637 const __m128 xx_yy_zz_xx =
1638 OZZ_SSE_SELECT_F(mask_00f0, _m.cols[2], xx_yy_0010);
1639 const __m128 yy_zz_xx_yy =
1640 OZZ_SHUFFLE_PS1(xx_yy_zz_xx, _MM_SHUFFLE(1, 0, 2, 1));
1641 const __m128 zz_xx_yy_zz =
1642 OZZ_SHUFFLE_PS1(xx_yy_zz_xx, _MM_SHUFFLE(2, 1, 0, 2));
1643
1644 const __m128 diag_sum =
1645 _mm_add_ps(_mm_add_ps(xx_yy_zz_xx, yy_zz_xx_yy), zz_xx_yy_zz);
1646 const __m128 diag_diff =
1647 _mm_sub_ps(_mm_sub_ps(xx_yy_zz_xx, yy_zz_xx_yy), zz_xx_yy_zz);
1648 const __m128 radicand =
1649 _mm_add_ps(OZZ_SSE_SELECT_F(mask_000f, diag_sum, diag_diff), one);
1650 const __m128 invSqrt = one / _mm_sqrt_ps(radicand);
1651
1652 __m128 zy_xz_yx = OZZ_SSE_SELECT_F(mask_00f0, _m.cols[1], _m.cols[0]);
1653 zy_xz_yx = OZZ_SHUFFLE_PS1(zy_xz_yx, _MM_SHUFFLE(0, 1, 2, 2));
1654 zy_xz_yx =
1655 OZZ_SSE_SELECT_F(mask_0f00, OZZ_SSE_SPLAT_F(_m.cols[2], 0), zy_xz_yx);
1656 __m128 yz_zx_xy = OZZ_SSE_SELECT_F(mask_f000, _m.cols[1], _m.cols[0]);
1657 yz_zx_xy = OZZ_SHUFFLE_PS1(yz_zx_xy, _MM_SHUFFLE(0, 0, 2, 0));
1658 yz_zx_xy =
1659 OZZ_SSE_SELECT_F(mask_f000, OZZ_SSE_SPLAT_F(_m.cols[2], 1), yz_zx_xy);
1660 const __m128 sum = _mm_add_ps(zy_xz_yx, yz_zx_xy);
1661 const __m128 diff = _mm_sub_ps(zy_xz_yx, yz_zx_xy);
1662 const __m128 scale = _mm_mul_ps(invSqrt, half);
1663
1664 const __m128 sum0 = OZZ_SHUFFLE_PS1(sum, _MM_SHUFFLE(0, 1, 2, 0));
1665 const __m128 sum1 = OZZ_SHUFFLE_PS1(sum, _MM_SHUFFLE(0, 0, 0, 2));
1666 const __m128 sum2 = OZZ_SHUFFLE_PS1(sum, _MM_SHUFFLE(0, 0, 0, 1));
1667 __m128 res0 = OZZ_SSE_SELECT_F(mask_000f, OZZ_SSE_SPLAT_F(diff, 0), sum0);
1668 __m128 res1 = OZZ_SSE_SELECT_F(mask_000f, OZZ_SSE_SPLAT_F(diff, 1), sum1);
1669 __m128 res2 = OZZ_SSE_SELECT_F(mask_000f, OZZ_SSE_SPLAT_F(diff, 2), sum2);
1670 res0 = _mm_mul_ps(OZZ_SSE_SELECT_F(mask_f000, radicand, res0),
1671 OZZ_SSE_SPLAT_F(scale, 0));
1672 res1 = _mm_mul_ps(OZZ_SSE_SELECT_F(mask_0f00, radicand, res1),
1673 OZZ_SSE_SPLAT_F(scale, 1));
1674 res2 = _mm_mul_ps(OZZ_SSE_SELECT_F(mask_00f0, radicand, res2),
1675 OZZ_SSE_SPLAT_F(scale, 2));
1676 __m128 res3 = _mm_mul_ps(OZZ_SSE_SELECT_F(mask_000f, radicand, diff),
1677 OZZ_SSE_SPLAT_F(scale, 3));
1678
1679 const __m128 xx = OZZ_SSE_SPLAT_F(_m.cols[0], 0);
1680 const __m128 yy = OZZ_SSE_SPLAT_F(_m.cols[1], 1);
1681 const __m128 zz = OZZ_SSE_SPLAT_F(_m.cols[2], 2);
1682 const __m128i cond0 = _mm_castps_si128(_mm_cmpgt_ps(yy, xx));
1683 const __m128i cond1 =
1684 _mm_castps_si128(_mm_and_ps(_mm_cmpgt_ps(zz, xx), _mm_cmpgt_ps(zz, yy)));
1685 const __m128i cond2 = _mm_castps_si128(
1686 _mm_cmpgt_ps(OZZ_SSE_SPLAT_F(diag_sum, 0), _mm_castsi128_ps(zero)));
1687 __m128 res = OZZ_SSE_SELECT_F(cond0, res1, res0);
1688 res = OZZ_SSE_SELECT_F(cond1, res2, res);
1689 res = OZZ_SSE_SELECT_F(cond2, res3, res);
1690
1691 assert(AreAllTrue1(IsNormalizedEst4(res)));
1692 return res;
1693}
1694
1695inline bool ToAffine(const Float4x4& _m, SimdFloat4* _translation,
1696 SimdFloat4* _quaternion, SimdFloat4* _scale) {
1697 const __m128 zero = _mm_setzero_ps();
1698 const __m128 one = simd_float4::one();
1699 const __m128i fff0 = simd_int4::mask_fff0();
1700 const __m128 max = _mm_set_ps1(kOrthogonalisationToleranceSq);
1701 const __m128 min = _mm_set_ps1(-kOrthogonalisationToleranceSq);
1702
1703 // Extracts translation.
1704 *_translation = OZZ_SSE_SELECT_F(fff0, _m.cols[3], one);
1705
1706 // Extracts scale.
1707 const __m128 m_tmp0 = _mm_unpacklo_ps(_m.cols[0], _m.cols[2]);
1708 const __m128 m_tmp1 = _mm_unpacklo_ps(_m.cols[1], _m.cols[3]);
1709 const __m128 m_tmp2 = _mm_unpackhi_ps(_m.cols[0], _m.cols[2]);
1710 const __m128 m_tmp3 = _mm_unpackhi_ps(_m.cols[1], _m.cols[3]);
1711 const __m128 m_row0 = _mm_unpacklo_ps(m_tmp0, m_tmp1);
1712 const __m128 m_row1 = _mm_unpackhi_ps(m_tmp0, m_tmp1);
1713 const __m128 m_row2 = _mm_unpacklo_ps(m_tmp2, m_tmp3);
1714
1715 const __m128 dot = OZZ_MADD(
1716 m_row0, m_row0, OZZ_MADD(m_row1, m_row1, _mm_mul_ps(m_row2, m_row2)));
1717 const __m128 abs_scale = _mm_sqrt_ps(dot);
1718
1719 const __m128 zero_axis =
1720 _mm_and_ps(_mm_cmplt_ps(dot, max), _mm_cmpgt_ps(dot, min));
1721
1722 // Builds an orthonormal matrix in order to support quaternion extraction.
1723 Float4x4 orthonormal;
1724 int mask = _mm_movemask_ps(zero_axis);
1725 if (mask & 1) {
1726 if (mask & 6) {
1727 return false;
1728 }
1729 orthonormal.cols[1] = _mm_div_ps(_m.cols[1], OZZ_SSE_SPLAT_F(abs_scale, 1));
1730 orthonormal.cols[0] = Normalize3(Cross3(orthonormal.cols[1], _m.cols[2]));
1731 orthonormal.cols[2] =
1732 Normalize3(Cross3(orthonormal.cols[0], orthonormal.cols[1]));
1733 } else if (mask & 4) {
1734 if (mask & 3) {
1735 return false;
1736 }
1737 orthonormal.cols[0] = _mm_div_ps(_m.cols[0], OZZ_SSE_SPLAT_F(abs_scale, 0));
1738 orthonormal.cols[2] = Normalize3(Cross3(orthonormal.cols[0], _m.cols[1]));
1739 orthonormal.cols[1] =
1740 Normalize3(Cross3(orthonormal.cols[2], orthonormal.cols[0]));
1741 } else { // Favor z axis in the default case
1742 if (mask & 5) {
1743 return false;
1744 }
1745 orthonormal.cols[2] = _mm_div_ps(_m.cols[2], OZZ_SSE_SPLAT_F(abs_scale, 2));
1746 orthonormal.cols[1] = Normalize3(Cross3(orthonormal.cols[2], _m.cols[0]));
1747 orthonormal.cols[0] =
1748 Normalize3(Cross3(orthonormal.cols[1], orthonormal.cols[2]));
1749 }
1750 orthonormal.cols[3] = simd_float4::w_axis();
1751
1752 // Get back scale signs in case of reflexions
1753 const __m128 o_tmp0 =
1754 _mm_unpacklo_ps(orthonormal.cols[0], orthonormal.cols[2]);
1755 const __m128 o_tmp1 =
1756 _mm_unpacklo_ps(orthonormal.cols[1], orthonormal.cols[3]);
1757 const __m128 o_tmp2 =
1758 _mm_unpackhi_ps(orthonormal.cols[0], orthonormal.cols[2]);
1759 const __m128 o_tmp3 =
1760 _mm_unpackhi_ps(orthonormal.cols[1], orthonormal.cols[3]);
1761 const __m128 o_row0 = _mm_unpacklo_ps(o_tmp0, o_tmp1);
1762 const __m128 o_row1 = _mm_unpackhi_ps(o_tmp0, o_tmp1);
1763 const __m128 o_row2 = _mm_unpacklo_ps(o_tmp2, o_tmp3);
1764
1765 const __m128 scale_dot = OZZ_MADD(
1766 o_row0, m_row0, OZZ_MADD(o_row1, m_row1, _mm_mul_ps(o_row2, m_row2)));
1767
1768 const __m128i cond = _mm_castps_si128(_mm_cmpgt_ps(scale_dot, zero));
1769 const __m128 cfalse = _mm_sub_ps(zero, abs_scale);
1770 const __m128 scale = OZZ_SSE_SELECT_F(cond, abs_scale, cfalse);
1771 *_scale = OZZ_SSE_SELECT_F(fff0, scale, one);
1772
1773 // Extracts quaternion.
1774 *_quaternion = ToQuaternion(orthonormal);
1775 return true;
1776}
1777
1778inline Float4x4 Float4x4::FromEuler(_SimdFloat4 _v) {
1779 return Float4x4::FromAxisAngle(simd_float4::y_axis(), SplatX(_v)) *
1780 Float4x4::FromAxisAngle(simd_float4::x_axis(), SplatY(_v)) *
1781 Float4x4::FromAxisAngle(simd_float4::z_axis(), SplatZ(_v));
1782}
1783
1784inline Float4x4 Float4x4::FromAxisAngle(_SimdFloat4 _axis, _SimdFloat4 _angle) {
1785 assert(AreAllTrue1(IsNormalizedEst3(_axis)));
1786
1787 const __m128i zero = _mm_setzero_si128();
1788 const __m128i ffff = _mm_cmpeq_epi32(zero, zero);
1789 const __m128i ione = _mm_srli_epi32(_mm_slli_epi32(ffff, 25), 2);
1790 const __m128 fff0 = _mm_castsi128_ps(_mm_srli_si128(ffff, 4));
1791 const __m128 one = _mm_castsi128_ps(ione);
1792 const __m128 w_axis = _mm_castsi128_ps(_mm_slli_si128(ione, 12));
1793
1794 const __m128 sin = SplatX(SinX(_angle));
1795 const __m128 cos = SplatX(CosX(_angle));
1796 const __m128 one_minus_cos = _mm_sub_ps(one, cos);
1797
1798 const __m128 v0 =
1799 _mm_mul_ps(_mm_mul_ps(one_minus_cos,
1800 OZZ_SHUFFLE_PS1(_axis, _MM_SHUFFLE(3, 0, 2, 1))),
1801 OZZ_SHUFFLE_PS1(_axis, _MM_SHUFFLE(3, 1, 0, 2)));
1802 const __m128 r0 =
1803 _mm_add_ps(_mm_mul_ps(_mm_mul_ps(one_minus_cos, _axis), _axis), cos);
1804 const __m128 r1 = _mm_add_ps(_mm_mul_ps(sin, _axis), v0);
1805 const __m128 r2 = _mm_sub_ps(v0, _mm_mul_ps(sin, _axis));
1806 const __m128 r0fff0 = _mm_and_ps(r0, fff0);
1807 const __m128 r1r22120 = _mm_shuffle_ps(r1, r2, _MM_SHUFFLE(2, 1, 2, 0));
1808 const __m128 v1 = OZZ_SHUFFLE_PS1(r1r22120, _MM_SHUFFLE(0, 3, 2, 1));
1809 const __m128 r1r20011 = _mm_shuffle_ps(r1, r2, _MM_SHUFFLE(0, 0, 1, 1));
1810 const __m128 v2 = OZZ_SHUFFLE_PS1(r1r20011, _MM_SHUFFLE(2, 0, 2, 0));
1811
1812 const __m128 t0 = _mm_shuffle_ps(r0fff0, v1, _MM_SHUFFLE(1, 0, 3, 0));
1813 const __m128 t1 = _mm_shuffle_ps(r0fff0, v1, _MM_SHUFFLE(3, 2, 3, 1));
1814 const Float4x4 ret = {{OZZ_SHUFFLE_PS1(t0, _MM_SHUFFLE(1, 3, 2, 0)),
1815 OZZ_SHUFFLE_PS1(t1, _MM_SHUFFLE(1, 3, 0, 2)),
1816 _mm_shuffle_ps(v2, r0fff0, _MM_SHUFFLE(3, 2, 1, 0)),
1817 w_axis}};
1818 return ret;
1819}
1820
1821inline Float4x4 Float4x4::FromQuaternion(_SimdFloat4 _quaternion) {
1822 assert(AreAllTrue1(IsNormalizedEst4(_quaternion)));
1823
1824 const __m128i zero = _mm_setzero_si128();
1825 const __m128i ffff = _mm_cmpeq_epi32(zero, zero);
1826 const __m128i ione = _mm_srli_epi32(_mm_slli_epi32(ffff, 25), 2);
1827 const __m128 fff0 = _mm_castsi128_ps(_mm_srli_si128(ffff, 4));
1828 const __m128 c1110 = _mm_castsi128_ps(_mm_srli_si128(ione, 4));
1829 const __m128 w_axis = _mm_castsi128_ps(_mm_slli_si128(ione, 12));
1830
1831 const __m128 vsum = _mm_add_ps(_quaternion, _quaternion);
1832 const __m128 vms = _mm_mul_ps(_quaternion, vsum);
1833
1834 const __m128 r0 = _mm_sub_ps(
1835 _mm_sub_ps(
1836 c1110,
1837 _mm_and_ps(OZZ_SHUFFLE_PS1(vms, _MM_SHUFFLE(3, 0, 0, 1)), fff0)),
1838 _mm_and_ps(OZZ_SHUFFLE_PS1(vms, _MM_SHUFFLE(3, 1, 2, 2)), fff0));
1839 const __m128 v0 =
1840 _mm_mul_ps(OZZ_SHUFFLE_PS1(_quaternion, _MM_SHUFFLE(3, 1, 0, 0)),
1841 OZZ_SHUFFLE_PS1(vsum, _MM_SHUFFLE(3, 2, 1, 2)));
1842 const __m128 v1 =
1843 _mm_mul_ps(OZZ_SHUFFLE_PS1(_quaternion, _MM_SHUFFLE(3, 3, 3, 3)),
1844 OZZ_SHUFFLE_PS1(vsum, _MM_SHUFFLE(3, 0, 2, 1)));
1845
1846 const __m128 r1 = _mm_add_ps(v0, v1);
1847 const __m128 r2 = _mm_sub_ps(v0, v1);
1848
1849 const __m128 r1r21021 = _mm_shuffle_ps(r1, r2, _MM_SHUFFLE(1, 0, 2, 1));
1850 const __m128 v2 = OZZ_SHUFFLE_PS1(r1r21021, _MM_SHUFFLE(1, 3, 2, 0));
1851 const __m128 r1r22200 = _mm_shuffle_ps(r1, r2, _MM_SHUFFLE(2, 2, 0, 0));
1852 const __m128 v3 = OZZ_SHUFFLE_PS1(r1r22200, _MM_SHUFFLE(2, 0, 2, 0));
1853
1854 const __m128 q0 = _mm_shuffle_ps(r0, v2, _MM_SHUFFLE(1, 0, 3, 0));
1855 const __m128 q1 = _mm_shuffle_ps(r0, v2, _MM_SHUFFLE(3, 2, 3, 1));
1856 const Float4x4 ret = {{OZZ_SHUFFLE_PS1(q0, _MM_SHUFFLE(1, 3, 2, 0)),
1857 OZZ_SHUFFLE_PS1(q1, _MM_SHUFFLE(1, 3, 0, 2)),
1858 _mm_shuffle_ps(v3, r0, _MM_SHUFFLE(3, 2, 1, 0)),
1859 w_axis}};
1860 return ret;
1861}
1862
1863inline Float4x4 Float4x4::FromAffine(_SimdFloat4 _translation,
1864 _SimdFloat4 _quaternion,
1865 _SimdFloat4 _scale) {
1866 assert(AreAllTrue1(IsNormalizedEst4(_quaternion)));
1867
1868 const __m128i zero = _mm_setzero_si128();
1869 const __m128i ffff = _mm_cmpeq_epi32(zero, zero);
1870 const __m128i ione = _mm_srli_epi32(_mm_slli_epi32(ffff, 25), 2);
1871 const __m128 fff0 = _mm_castsi128_ps(_mm_srli_si128(ffff, 4));
1872 const __m128 c1110 = _mm_castsi128_ps(_mm_srli_si128(ione, 4));
1873
1874 const __m128 vsum = _mm_add_ps(_quaternion, _quaternion);
1875 const __m128 vms = _mm_mul_ps(_quaternion, vsum);
1876
1877 const __m128 r0 = _mm_sub_ps(
1878 _mm_sub_ps(
1879 c1110,
1880 _mm_and_ps(OZZ_SHUFFLE_PS1(vms, _MM_SHUFFLE(3, 0, 0, 1)), fff0)),
1881 _mm_and_ps(OZZ_SHUFFLE_PS1(vms, _MM_SHUFFLE(3, 1, 2, 2)), fff0));
1882 const __m128 v0 =
1883 _mm_mul_ps(OZZ_SHUFFLE_PS1(_quaternion, _MM_SHUFFLE(3, 1, 0, 0)),
1884 OZZ_SHUFFLE_PS1(vsum, _MM_SHUFFLE(3, 2, 1, 2)));
1885 const __m128 v1 =
1886 _mm_mul_ps(OZZ_SHUFFLE_PS1(_quaternion, _MM_SHUFFLE(3, 3, 3, 3)),
1887 OZZ_SHUFFLE_PS1(vsum, _MM_SHUFFLE(3, 0, 2, 1)));
1888
1889 const __m128 r1 = _mm_add_ps(v0, v1);
1890 const __m128 r2 = _mm_sub_ps(v0, v1);
1891
1892 const __m128 r1r21021 = _mm_shuffle_ps(r1, r2, _MM_SHUFFLE(1, 0, 2, 1));
1893 const __m128 v2 = OZZ_SHUFFLE_PS1(r1r21021, _MM_SHUFFLE(1, 3, 2, 0));
1894 const __m128 r1r22200 = _mm_shuffle_ps(r1, r2, _MM_SHUFFLE(2, 2, 0, 0));
1895 const __m128 v3 = OZZ_SHUFFLE_PS1(r1r22200, _MM_SHUFFLE(2, 0, 2, 0));
1896
1897 const __m128 q0 = _mm_shuffle_ps(r0, v2, _MM_SHUFFLE(1, 0, 3, 0));
1898 const __m128 q1 = _mm_shuffle_ps(r0, v2, _MM_SHUFFLE(3, 2, 3, 1));
1899
1900 const Float4x4 ret = {
1901 {_mm_mul_ps(OZZ_SHUFFLE_PS1(q0, _MM_SHUFFLE(1, 3, 2, 0)),
1902 OZZ_SSE_SPLAT_F(_scale, 0)),
1903 _mm_mul_ps(OZZ_SHUFFLE_PS1(q1, _MM_SHUFFLE(1, 3, 0, 2)),
1904 OZZ_SSE_SPLAT_F(_scale, 1)),
1905 _mm_mul_ps(_mm_shuffle_ps(v3, r0, _MM_SHUFFLE(3, 2, 1, 0)),
1906 OZZ_SSE_SPLAT_F(_scale, 2)),
1907 _mm_movelh_ps(_translation, _mm_unpackhi_ps(_translation, c1110))}};
1908 return ret;
1909}
1910
1911OZZ_INLINE ozz::math::SimdFloat4 TransformPoint(const ozz::math::Float4x4& _m,
1913 const __m128 xxxx = _mm_mul_ps(OZZ_SSE_SPLAT_F(_v, 0), _m.cols[0]);
1914 const __m128 a23 = OZZ_MADD(OZZ_SSE_SPLAT_F(_v, 2), _m.cols[2], _m.cols[3]);
1915 const __m128 a01 = OZZ_MADD(OZZ_SSE_SPLAT_F(_v, 1), _m.cols[1], xxxx);
1916 return _mm_add_ps(a01, a23);
1917}
1918
1919OZZ_INLINE ozz::math::SimdFloat4 TransformVector(const ozz::math::Float4x4& _m,
1921 const __m128 xxxx = _mm_mul_ps(_m.cols[0], OZZ_SSE_SPLAT_F(_v, 0));
1922 const __m128 zzzz = _mm_mul_ps(_m.cols[1], OZZ_SSE_SPLAT_F(_v, 1));
1923 const __m128 a21 = OZZ_MADD(_m.cols[2], OZZ_SSE_SPLAT_F(_v, 2), xxxx);
1924 return _mm_add_ps(zzzz, a21);
1925}
1926
1927OZZ_INLINE ozz::math::SimdFloat4 operator*(const ozz::math::Float4x4& _m,
1929 const __m128 xxxx = _mm_mul_ps(OZZ_SSE_SPLAT_F(_v, 0), _m.cols[0]);
1930 const __m128 zzzz = _mm_mul_ps(OZZ_SSE_SPLAT_F(_v, 2), _m.cols[2]);
1931 const __m128 a01 = OZZ_MADD(OZZ_SSE_SPLAT_F(_v, 1), _m.cols[1], xxxx);
1932 const __m128 a23 = OZZ_MADD(OZZ_SSE_SPLAT_F(_v, 3), _m.cols[3], zzzz);
1933 return _mm_add_ps(a01, a23);
1934}
1935
1936inline ozz::math::Float4x4 operator*(const ozz::math::Float4x4& _a,
1937 const ozz::math::Float4x4& _b) {
1939 {
1940 const __m128 xxxx = _mm_mul_ps(OZZ_SSE_SPLAT_F(_b.cols[0], 0), _a.cols[0]);
1941 const __m128 zzzz = _mm_mul_ps(OZZ_SSE_SPLAT_F(_b.cols[0], 2), _a.cols[2]);
1942 const __m128 a01 =
1943 OZZ_MADD(OZZ_SSE_SPLAT_F(_b.cols[0], 1), _a.cols[1], xxxx);
1944 const __m128 a23 =
1945 OZZ_MADD(OZZ_SSE_SPLAT_F(_b.cols[0], 3), _a.cols[3], zzzz);
1946 ret.cols[0] = _mm_add_ps(a01, a23);
1947 }
1948 {
1949 const __m128 xxxx = _mm_mul_ps(OZZ_SSE_SPLAT_F(_b.cols[1], 0), _a.cols[0]);
1950 const __m128 zzzz = _mm_mul_ps(OZZ_SSE_SPLAT_F(_b.cols[1], 2), _a.cols[2]);
1951 const __m128 a01 =
1952 OZZ_MADD(OZZ_SSE_SPLAT_F(_b.cols[1], 1), _a.cols[1], xxxx);
1953 const __m128 a23 =
1954 OZZ_MADD(OZZ_SSE_SPLAT_F(_b.cols[1], 3), _a.cols[3], zzzz);
1955 ret.cols[1] = _mm_add_ps(a01, a23);
1956 }
1957 {
1958 const __m128 xxxx = _mm_mul_ps(OZZ_SSE_SPLAT_F(_b.cols[2], 0), _a.cols[0]);
1959 const __m128 zzzz = _mm_mul_ps(OZZ_SSE_SPLAT_F(_b.cols[2], 2), _a.cols[2]);
1960 const __m128 a01 =
1961 OZZ_MADD(OZZ_SSE_SPLAT_F(_b.cols[2], 1), _a.cols[1], xxxx);
1962 const __m128 a23 =
1963 OZZ_MADD(OZZ_SSE_SPLAT_F(_b.cols[2], 3), _a.cols[3], zzzz);
1964 ret.cols[2] = _mm_add_ps(a01, a23);
1965 }
1966 {
1967 const __m128 xxxx = _mm_mul_ps(OZZ_SSE_SPLAT_F(_b.cols[3], 0), _a.cols[0]);
1968 const __m128 zzzz = _mm_mul_ps(OZZ_SSE_SPLAT_F(_b.cols[3], 2), _a.cols[2]);
1969 const __m128 a01 =
1970 OZZ_MADD(OZZ_SSE_SPLAT_F(_b.cols[3], 1), _a.cols[1], xxxx);
1971 const __m128 a23 =
1972 OZZ_MADD(OZZ_SSE_SPLAT_F(_b.cols[3], 3), _a.cols[3], zzzz);
1973 ret.cols[3] = _mm_add_ps(a01, a23);
1974 }
1975 return ret;
1976}
1977
1978OZZ_INLINE ozz::math::Float4x4 operator+(const ozz::math::Float4x4& _a,
1979 const ozz::math::Float4x4& _b) {
1980 const ozz::math::Float4x4 ret = {
1981 {_mm_add_ps(_a.cols[0], _b.cols[0]), _mm_add_ps(_a.cols[1], _b.cols[1]),
1982 _mm_add_ps(_a.cols[2], _b.cols[2]), _mm_add_ps(_a.cols[3], _b.cols[3])}};
1983 return ret;
1984}
1985
1986OZZ_INLINE ozz::math::Float4x4 operator-(const ozz::math::Float4x4& _a,
1987 const ozz::math::Float4x4& _b) {
1988 const ozz::math::Float4x4 ret = {
1989 {_mm_sub_ps(_a.cols[0], _b.cols[0]), _mm_sub_ps(_a.cols[1], _b.cols[1]),
1990 _mm_sub_ps(_a.cols[2], _b.cols[2]), _mm_sub_ps(_a.cols[3], _b.cols[3])}};
1991 return ret;
1992}
1993} // namespace math
1994} // namespace ozz
1995
1996#if !defined(OZZ_DISABLE_SSE_NATIVE_OPERATORS)
1997OZZ_INLINE ozz::math::SimdFloat4 operator+(ozz::math::_SimdFloat4 _a,
1999 return _mm_add_ps(_a, _b);
2000}
2001
2002OZZ_INLINE ozz::math::SimdFloat4 operator-(ozz::math::_SimdFloat4 _a,
2004 return _mm_sub_ps(_a, _b);
2005}
2006
2007OZZ_INLINE ozz::math::SimdFloat4 operator-(ozz::math::_SimdFloat4 _v) {
2008 return _mm_sub_ps(_mm_setzero_ps(), _v);
2009}
2010
2011OZZ_INLINE ozz::math::SimdFloat4 operator*(ozz::math::_SimdFloat4 _a,
2013 return _mm_mul_ps(_a, _b);
2014}
2015
2016OZZ_INLINE ozz::math::SimdFloat4 operator/(ozz::math::_SimdFloat4 _a,
2018 return _mm_div_ps(_a, _b);
2019}
2020#endif // !defined(OZZ_DISABLE_SSE_NATIVE_OPERATORS)
2021
2022namespace ozz {
2023namespace math {
2024OZZ_INLINE uint16_t FloatToHalf(float _f) {
2025 const int h = _mm_cvtsi128_si32(FloatToHalf(_mm_set1_ps(_f)));
2026 return static_cast<uint16_t>(h);
2027}
2028
2029OZZ_INLINE float HalfToFloat(uint16_t _h) {
2030 return _mm_cvtss_f32(HalfToFloat(_mm_set1_epi32(_h)));
2031}
2032
2033// Half <-> Float implementation is based on:
2034// http://fgiesen.wordpress.com/2012/03/28/half-to-float-done-quic/.
2035inline SimdInt4 FloatToHalf(_SimdFloat4 _f) {
2036 const __m128i mask_sign = _mm_set1_epi32(0x80000000u);
2037 const __m128i mask_round = _mm_set1_epi32(~0xfffu);
2038 const __m128i f32infty = _mm_set1_epi32(255 << 23);
2039 const __m128 magic = _mm_castsi128_ps(_mm_set1_epi32(15 << 23));
2040 const __m128i nanbit = _mm_set1_epi32(0x200);
2041 const __m128i infty_as_fp16 = _mm_set1_epi32(0x7c00);
2042 const __m128 clamp = _mm_castsi128_ps(_mm_set1_epi32((31 << 23) - 0x1000));
2043
2044 const __m128 msign = _mm_castsi128_ps(mask_sign);
2045 const __m128 justsign = _mm_and_ps(msign, _f);
2046 const __m128 absf = _mm_xor_ps(_f, justsign);
2047 const __m128 mround = _mm_castsi128_ps(mask_round);
2048 const __m128i absf_int = _mm_castps_si128(absf);
2049 const __m128i b_isnan = _mm_cmpgt_epi32(absf_int, f32infty);
2050 const __m128i b_isnormal = _mm_cmpgt_epi32(f32infty, _mm_castps_si128(absf));
2051 const __m128i inf_or_nan =
2052 _mm_or_si128(_mm_and_si128(b_isnan, nanbit), infty_as_fp16);
2053 const __m128 fnosticky = _mm_and_ps(absf, mround);
2054 const __m128 scaled = _mm_mul_ps(fnosticky, magic);
2055 // Logically, we want PMINSD on "biased", but this should gen better code
2056 const __m128 clamped = _mm_min_ps(scaled, clamp);
2057 const __m128i biased =
2058 _mm_sub_epi32(_mm_castps_si128(clamped), _mm_castps_si128(mround));
2059 const __m128i shifted = _mm_srli_epi32(biased, 13);
2060 const __m128i normal = _mm_and_si128(shifted, b_isnormal);
2061 const __m128i not_normal = _mm_andnot_si128(b_isnormal, inf_or_nan);
2062 const __m128i joined = _mm_or_si128(normal, not_normal);
2063
2064 const __m128i sign_shift = _mm_srli_epi32(_mm_castps_si128(justsign), 16);
2065 return _mm_or_si128(joined, sign_shift);
2066}
2067
2068OZZ_INLINE SimdFloat4 HalfToFloat(_SimdInt4 _h) {
2069 const __m128i mask_nosign = _mm_set1_epi32(0x7fff);
2070 const __m128 magic = _mm_castsi128_ps(_mm_set1_epi32((254 - 15) << 23));
2071 const __m128i was_infnan = _mm_set1_epi32(0x7bff);
2072 const __m128 exp_infnan = _mm_castsi128_ps(_mm_set1_epi32(255 << 23));
2073
2074 const __m128i expmant = _mm_and_si128(mask_nosign, _h);
2075 const __m128i shifted = _mm_slli_epi32(expmant, 13);
2076 const __m128 scaled = _mm_mul_ps(_mm_castsi128_ps(shifted), magic);
2077 const __m128i b_wasinfnan = _mm_cmpgt_epi32(expmant, was_infnan);
2078 const __m128i sign = _mm_slli_epi32(_mm_xor_si128(_h, expmant), 16);
2079 const __m128 infnanexp =
2080 _mm_and_ps(_mm_castsi128_ps(b_wasinfnan), exp_infnan);
2081 const __m128 sign_inf = _mm_or_ps(_mm_castsi128_ps(sign), infnanexp);
2082 return _mm_or_ps(scaled, sign_inf);
2083}
2084} // namespace math
2085} // namespace ozz
2086
2087#undef OZZ_SHUFFLE_PS1
2088#undef OZZ_SSE_SPLAT_F
2089#undef OZZ_SSE_HADD2_F
2090#undef OZZ_SSE_HADD3_F
2091#undef OZZ_SSE_HADD4_F
2092#undef OZZ_SSE_DOT2_F
2093#undef OZZ_SSE_DOT3_F
2094#undef OZZ_SSE_DOT4_F
2095#undef OZZ_MADD
2096#undef OZZ_MSUB
2097#undef OZZ_NMADD
2098#undef OZZ_NMSUB
2099#undef OZZ_MADDX
2100#undef OZZ_MSUBX
2101#undef OZZ_NMADDX
2102#undef OZZ_NMSUBX
2103#undef OZZ_SSE_SELECT_F
2104#undef OZZ_SSE_SPLAT_I
2105#undef OZZ_SSE_SELECT_I
2106#endif // OZZ_OZZ_BASE_MATHS_INTERNAL_SIMD_MATH_SSE_INL_H_
GLM_FUNC_DECL GLM_CONSTEXPR genType clamp(genType x, genType minVal, genType maxVal)
Definition func_common.inl:505
GLM_FUNC_QUALIFIER vec< 3, T, Q > cross(vec< 3, T, Q > const &x, vec< 3, T, Q > const &y)
Definition func_geometric.inl:175
GLM_FUNC_QUALIFIER vec< L, T, Q > sin(vec< L, T, Q > const &v)
Definition func_trigonometric.inl:41
GLM_FUNC_QUALIFIER vec< L, T, Q > cos(vec< L, T, Q > const &v)
Definition func_trigonometric.inl:50
GLM_FUNC_DECL GLM_CONSTEXPR genType zero()
Definition constants.inl:6
uint16 uint16_t
Definition fwd.hpp:117
Definition simd_math_config.h:121
Definition simd_math.h:1066