28#ifndef OZZ_OZZ_BASE_MATHS_INTERNAL_SIMD_MATH_REF_INL_H_
29#define OZZ_OZZ_BASE_MATHS_INTERNAL_SIMD_MATH_REF_INL_H_
39#include "ozz/base/maths/math_constant.h"
57#define OZZ_RCP_EST(_in, _out) \
59 const float in = _in; \
67 } ui = {(0x3f800000 * 2) - uf.i}; \
68 const float fp = ui.f * (2.f - in * ui.f); \
69 _out = fp * (2.f - in * fp); \
72#define OZZ_RCP_EST_NR(_in, _out) \
75 OZZ_RCP_EST(_in, fp2); \
76 _out = fp2 * (2.f - _in * fp2); \
79#define OZZ_RSQRT_EST(_in, _out) \
81 const float in = _in; \
89 } ui = {0x5f3759df - (uf.i / 2)}; \
90 const float fp = ui.f * (1.5f - (in * .5f * ui.f * ui.f)); \
91 _out = fp * (1.5f - (in * .5f * fp * fp)); \
94#define OZZ_RSQRT_EST_NR(_in, _out) \
97 OZZ_RSQRT_EST(_in, fp2); \
98 _out = fp2 * (1.5f - (_in * .5f * fp2 * fp2)); \
101namespace simd_float4 {
108OZZ_INLINE SimdFloat4 one() {
109 const SimdFloat4 ret = {1.f, 1.f, 1.f, 1.f};
113OZZ_INLINE SimdFloat4 x_axis() {
114 const SimdFloat4 ret = {1.f, 0.f, 0.f, 0.f};
118OZZ_INLINE SimdFloat4 y_axis() {
119 const SimdFloat4 ret = {0.f, 1.f, 0.f, 0.f};
123OZZ_INLINE SimdFloat4 z_axis() {
124 const SimdFloat4 ret = {0.f, 0.f, 1.f, 0.f};
128OZZ_INLINE SimdFloat4 w_axis() {
129 const SimdFloat4 ret = {0.f, 0.f, 0.f, 1.f};
133OZZ_INLINE SimdFloat4 Load(
float _x,
float _y,
float _z,
float _w) {
134 const SimdFloat4 ret = {_x, _y, _z, _w};
138OZZ_INLINE SimdFloat4 LoadX(
float _x) {
139 const SimdFloat4 ret = {_x, 0.f, 0.f, 0.f};
143OZZ_INLINE SimdFloat4 Load1(
float _x) {
144 const SimdFloat4 ret = {_x, _x, _x, _x};
148OZZ_INLINE SimdFloat4 LoadPtr(
const float* _f) {
149 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0xf) &&
"Invalid alignment");
150 const SimdFloat4 ret = {_f[0], _f[1], _f[2], _f[3]};
154OZZ_INLINE SimdFloat4 LoadPtrU(
const float* _f) {
155 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
156 const SimdFloat4 ret = {_f[0], _f[1], _f[2], _f[3]};
160OZZ_INLINE SimdFloat4 LoadXPtrU(
const float* _f) {
161 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
162 const SimdFloat4 ret = {*_f, 0.f, 0.f, 0.f};
166OZZ_INLINE SimdFloat4 Load1PtrU(
const float* _f) {
167 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
168 const SimdFloat4 ret = {*_f, *_f, *_f, *_f};
172OZZ_INLINE SimdFloat4 Load2PtrU(
const float* _f) {
173 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
174 const SimdFloat4 ret = {_f[0], _f[1], 0.f, 0.f};
178OZZ_INLINE SimdFloat4 Load3PtrU(
const float* _f) {
179 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
180 const SimdFloat4 ret = {_f[0], _f[1], _f[2]};
184OZZ_INLINE SimdFloat4 FromInt(_SimdInt4 _i) {
185 const SimdFloat4 ret = {
static_cast<float>(_i.x),
static_cast<float>(_i.y),
186 static_cast<float>(_i.z),
static_cast<float>(_i.w)};
191OZZ_INLINE
float GetX(_SimdFloat4 _v) {
return _v.x; }
193OZZ_INLINE
float GetY(_SimdFloat4 _v) {
return _v.y; }
195OZZ_INLINE
float GetZ(_SimdFloat4 _v) {
return _v.z; }
197OZZ_INLINE
float GetW(_SimdFloat4 _v) {
return _v.w; }
198OZZ_INLINE SimdFloat4 SetX(_SimdFloat4 _v, _SimdFloat4 _f) {
199 const SimdFloat4 ret = {_f.x, _v.y, _v.z, _v.w};
203OZZ_INLINE SimdFloat4 SetY(_SimdFloat4 _v, _SimdFloat4 _f) {
204 const SimdFloat4 ret = {_v.x, _f.x, _v.z, _v.w};
208OZZ_INLINE SimdFloat4 SetZ(_SimdFloat4 _v, _SimdFloat4 _f) {
209 const SimdFloat4 ret = {_v.x, _v.y, _f.x, _v.w};
213OZZ_INLINE SimdFloat4 SetW(_SimdFloat4 _v, _SimdFloat4 _f) {
214 const SimdFloat4 ret = {_v.x, _v.y, _v.z, _f.x};
218OZZ_INLINE SimdFloat4 SetI(_SimdFloat4 _v, _SimdFloat4 _f,
int _ith) {
219 assert(_ith >= 0 && _ith <= 3 &&
"Invalid index, out of range.");
221 (&ret.x)[_ith] = _f.x;
225OZZ_INLINE
void StorePtr(_SimdFloat4 _v,
float* _f) {
226 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0xf) &&
"Invalid alignment");
233OZZ_INLINE
void Store1Ptr(_SimdFloat4 _v,
float* _f) {
234 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0xf) &&
"Invalid alignment");
238OZZ_INLINE
void Store2Ptr(_SimdFloat4 _v,
float* _f) {
239 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0xf) &&
"Invalid alignment");
244OZZ_INLINE
void Store3Ptr(_SimdFloat4 _v,
float* _f) {
245 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0xf) &&
"Invalid alignment");
251OZZ_INLINE
void StorePtrU(_SimdFloat4 _v,
float* _f) {
252 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
259OZZ_INLINE
void Store1PtrU(_SimdFloat4 _v,
float* _f) {
260 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
264OZZ_INLINE
void Store2PtrU(_SimdFloat4 _v,
float* _f) {
265 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
270OZZ_INLINE
void Store3PtrU(_SimdFloat4 _v,
float* _f) {
271 assert(!(
reinterpret_cast<uintptr_t
>(_f) & 0x3) &&
"Invalid alignment");
277OZZ_INLINE SimdFloat4 SplatX(_SimdFloat4 _v) {
278 const SimdFloat4 ret = {_v.x, _v.x, _v.x, _v.x};
282OZZ_INLINE SimdFloat4 SplatY(_SimdFloat4 _v) {
283 const SimdFloat4 ret = {_v.y, _v.y, _v.y, _v.y};
287OZZ_INLINE SimdFloat4 SplatZ(_SimdFloat4 _v) {
288 const SimdFloat4 ret = {_v.z, _v.z, _v.z, _v.z};
292OZZ_INLINE SimdFloat4 SplatW(_SimdFloat4 _v) {
293 const SimdFloat4 ret = {_v.w, _v.w, _v.w, _v.w};
297template <
size_t _X,
size_t _Y,
size_t _Z,
size_t _W>
298OZZ_INLINE SimdFloat4 Swizzle(_SimdFloat4 _v) {
299 static_assert(_X <= 3 && _Y <= 3 && _Z <= 3 && _W <= 3,
300 "Indices must be between 0 and 3");
301 const float* pf = &_v.x;
302 const SimdFloat4 ret = {pf[_X], pf[_Y], pf[_Z], pf[_W]};
306OZZ_INLINE
void Transpose4x1(
const SimdFloat4 _in[4], SimdFloat4 _out[1]) {
307 _out[0].x = _in[0].x;
308 _out[0].y = _in[1].x;
309 _out[0].z = _in[2].x;
310 _out[0].w = _in[3].x;
313OZZ_INLINE
void Transpose1x4(
const SimdFloat4 _in[1], SimdFloat4 _out[4]) {
314 _out[0].x = _in[0].x;
315 _out[0].y = _out[0].z = _out[0].w = 0.f;
316 _out[1].x = _in[0].y;
317 _out[1].y = _out[1].z = _out[1].w = 0.f;
318 _out[2].x = _in[0].z;
319 _out[2].y = _out[2].z = _out[2].w = 0.f;
320 _out[3].x = _in[0].w;
321 _out[3].y = _out[3].z = _out[3].w = 0.f;
324OZZ_INLINE
void Transpose4x2(
const SimdFloat4 _in[4], SimdFloat4 _out[2]) {
325 _out[0].x = _in[0].x;
326 _out[0].y = _in[1].x;
327 _out[0].z = _in[2].x;
328 _out[0].w = _in[3].x;
329 _out[1].x = _in[0].y;
330 _out[1].y = _in[1].y;
331 _out[1].z = _in[2].y;
332 _out[1].w = _in[3].y;
335OZZ_INLINE
void Transpose2x4(
const SimdFloat4 _in[2], SimdFloat4 _out[4]) {
336 _out[0].x = _in[0].x;
337 _out[0].y = _in[1].x;
338 _out[0].z = _out[0].w = 0.f;
339 _out[1].x = _in[0].y;
340 _out[1].y = _in[1].y;
341 _out[1].z = _out[1].w = 0.f;
342 _out[2].x = _in[0].z;
343 _out[2].y = _in[1].z;
344 _out[2].z = _out[2].w = 0.f;
345 _out[3].x = _in[0].w;
346 _out[3].y = _in[1].w;
347 _out[3].z = _out[3].w = 0.f;
350OZZ_INLINE
void Transpose4x3(
const SimdFloat4 _in[4], SimdFloat4 _out[3]) {
351 _out[0].x = _in[0].x;
352 _out[0].y = _in[1].x;
353 _out[0].z = _in[2].x;
354 _out[0].w = _in[3].x;
355 _out[1].x = _in[0].y;
356 _out[1].y = _in[1].y;
357 _out[1].z = _in[2].y;
358 _out[1].w = _in[3].y;
359 _out[2].x = _in[0].z;
360 _out[2].y = _in[1].z;
361 _out[2].z = _in[2].z;
362 _out[2].w = _in[3].z;
365OZZ_INLINE
void Transpose3x4(
const SimdFloat4 _in[3], SimdFloat4 _out[4]) {
366 _out[0].x = _in[0].x;
367 _out[0].y = _in[1].x;
368 _out[0].z = _in[2].x;
370 _out[1].x = _in[0].y;
371 _out[1].y = _in[1].y;
372 _out[1].z = _in[2].y;
374 _out[2].x = _in[0].z;
375 _out[2].y = _in[1].z;
376 _out[2].z = _in[2].z;
378 _out[3].x = _in[0].w;
379 _out[3].y = _in[1].w;
380 _out[3].z = _in[2].w;
384OZZ_INLINE
void Transpose4x4(
const SimdFloat4 _in[4], SimdFloat4 _out[4]) {
385 _out[0].x = _in[0].x;
386 _out[1].x = _in[0].y;
387 _out[2].x = _in[0].z;
388 _out[3].x = _in[0].w;
389 _out[0].y = _in[1].x;
390 _out[1].y = _in[1].y;
391 _out[2].y = _in[1].z;
392 _out[3].y = _in[1].w;
393 _out[0].z = _in[2].x;
394 _out[1].z = _in[2].y;
395 _out[2].z = _in[2].z;
396 _out[3].z = _in[2].w;
397 _out[0].w = _in[3].x;
398 _out[1].w = _in[3].y;
399 _out[2].w = _in[3].z;
400 _out[3].w = _in[3].w;
403OZZ_INLINE
void Transpose16x16(
const SimdFloat4 _in[16], SimdFloat4 _out[16]) {
404 for (
int i = 0; i < 4; ++i) {
405 const int i4 = i * 4;
406 _out[i4 + 0].x = *(&_in[0].x + i);
407 _out[i4 + 0].y = *(&_in[1].x + i);
408 _out[i4 + 0].z = *(&_in[2].x + i);
409 _out[i4 + 0].w = *(&_in[3].x + i);
410 _out[i4 + 1].x = *(&_in[4].x + i);
411 _out[i4 + 1].y = *(&_in[5].x + i);
412 _out[i4 + 1].z = *(&_in[6].x + i);
413 _out[i4 + 1].w = *(&_in[7].x + i);
414 _out[i4 + 2].x = *(&_in[8].x + i);
415 _out[i4 + 2].y = *(&_in[9].x + i);
416 _out[i4 + 2].z = *(&_in[10].x + i);
417 _out[i4 + 2].w = *(&_in[11].x + i);
418 _out[i4 + 3].x = *(&_in[12].x + i);
419 _out[i4 + 3].y = *(&_in[13].x + i);
420 _out[i4 + 3].z = *(&_in[14].x + i);
421 _out[i4 + 3].w = *(&_in[15].x + i);
425OZZ_INLINE SimdFloat4 MAdd(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
426 const SimdFloat4 ret = {_a.x * _b.x + _c.x, _a.y * _b.y + _c.y,
427 _a.z * _b.z + _c.z, _a.w * _b.w + _c.w};
431OZZ_INLINE SimdFloat4 MSub(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
432 const SimdFloat4 ret = {_a.x * _b.x - _c.x, _a.y * _b.y - _c.y,
433 _a.z * _b.z - _c.z, _a.w * _b.w - _c.w};
437OZZ_INLINE SimdFloat4 NMAdd(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
438 const SimdFloat4 ret = {_c.x - _a.x * _b.x, _c.y - _a.y * _b.y,
439 _c.z - _a.z * _b.z, _c.w - _a.w * _b.w};
443OZZ_INLINE SimdFloat4 NMSub(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
444 const SimdFloat4 ret = {-_a.x * _b.x - _c.x, -_a.y * _b.y - _c.y,
445 -_a.z * _b.z - _c.z, -_a.w * _b.w - _c.w};
449OZZ_INLINE SimdFloat4 DivX(_SimdFloat4 _a, _SimdFloat4 _b) {
450 const SimdFloat4 ret = {_a.x / _b.x, _a.y, _a.z, _a.w};
454OZZ_INLINE SimdFloat4 HAdd2(_SimdFloat4 _v) {
455 const SimdFloat4 ret = {_v.x + _v.y, _v.y, _v.z, _v.w};
459OZZ_INLINE SimdFloat4 HAdd3(_SimdFloat4 _v) {
460 const SimdFloat4 ret = {_v.x + _v.y + _v.z, _v.y, _v.z, _v.w};
464OZZ_INLINE SimdFloat4 HAdd4(_SimdFloat4 _v) {
465 const SimdFloat4 ret = {_v.x + _v.y + _v.z + _v.w, _v.x, _v.x, _v.x};
469OZZ_INLINE SimdFloat4 Dot2(_SimdFloat4 _a, _SimdFloat4 _b) {
470 const SimdFloat4 ret = {_a.x * _b.x + _a.y * _b.y, _a.x, _a.x, _a.x};
474OZZ_INLINE SimdFloat4 Dot3(_SimdFloat4 _a, _SimdFloat4 _b) {
475 const SimdFloat4 ret = {_a.x * _b.x + _a.y * _b.y + _a.z * _b.z, _a.x, _a.x,
480OZZ_INLINE SimdFloat4 Dot4(_SimdFloat4 _a, _SimdFloat4 _b) {
481 const SimdFloat4 ret = {_a.x * _b.x + _a.y * _b.y + _a.z * _b.z + _a.w * _b.w,
486OZZ_INLINE SimdFloat4 Cross3(_SimdFloat4 _a, _SimdFloat4 _b) {
487 const SimdFloat4 ret = {_a.y * _b.z - _a.z * _b.y, _a.z * _b.x - _a.x * _b.z,
488 _a.x * _b.y - _a.y * _b.x, _a.x};
492OZZ_INLINE SimdFloat4 RcpEst(_SimdFloat4 _v) {
494 OZZ_RCP_EST(_v.x, ret.x);
495 OZZ_RCP_EST(_v.y, ret.y);
496 OZZ_RCP_EST(_v.z, ret.z);
497 OZZ_RCP_EST(_v.w, ret.w);
501OZZ_INLINE SimdFloat4 RcpEstNR(_SimdFloat4 _v) {
503 OZZ_RCP_EST_NR(_v.x, ret.x);
504 OZZ_RCP_EST_NR(_v.y, ret.y);
505 OZZ_RCP_EST_NR(_v.z, ret.z);
506 OZZ_RCP_EST_NR(_v.w, ret.w);
510OZZ_INLINE SimdFloat4 RcpEstX(_SimdFloat4 _v) {
512 OZZ_RCP_EST(_v.x, ret.x);
519OZZ_INLINE SimdFloat4 RcpEstXNR(_SimdFloat4 _v) {
521 OZZ_RCP_EST(_v.x, ret.x);
528OZZ_INLINE SimdFloat4 Sqrt(_SimdFloat4 _v) {
529 const SimdFloat4 ret = {std::sqrt(_v.x), std::sqrt(_v.y), std::sqrt(_v.z),
534OZZ_INLINE SimdFloat4 SqrtX(_SimdFloat4 _v) {
535 const SimdFloat4 ret = {std::sqrt(_v.x), _v.y, _v.z, _v.w};
539OZZ_INLINE SimdFloat4 RSqrtEst(_SimdFloat4 _v) {
541 OZZ_RSQRT_EST(_v.x, ret.x);
542 OZZ_RSQRT_EST(_v.y, ret.y);
543 OZZ_RSQRT_EST(_v.z, ret.z);
544 OZZ_RSQRT_EST(_v.w, ret.w);
548OZZ_INLINE SimdFloat4 RSqrtEstNR(_SimdFloat4 _v) {
550 OZZ_RSQRT_EST_NR(_v.x, ret.x);
551 OZZ_RSQRT_EST_NR(_v.y, ret.y);
552 OZZ_RSQRT_EST_NR(_v.z, ret.z);
553 OZZ_RSQRT_EST_NR(_v.w, ret.w);
557OZZ_INLINE SimdFloat4 RSqrtEstX(_SimdFloat4 _v) {
559 OZZ_RSQRT_EST(_v.x, ret.x);
566OZZ_INLINE SimdFloat4 RSqrtEstXNR(_SimdFloat4 _v) {
568 OZZ_RSQRT_EST(_v.x, ret.x);
575OZZ_INLINE SimdFloat4 Abs(_SimdFloat4 _v) {
576 const SimdFloat4 ret = {std::abs(_v.x), std::abs(_v.y), std::abs(_v.z),
581OZZ_INLINE SimdInt4 Sign(_SimdFloat4 _v) {
582 internal::SimdFI4 fi = {_v};
583 const SimdInt4 ret = {fi.i.x &
static_cast<int>(0x80000000),
584 fi.i.y &
static_cast<int>(0x80000000),
585 fi.i.z &
static_cast<int>(0x80000000),
586 fi.i.w &
static_cast<int>(0x80000000)};
590OZZ_INLINE SimdFloat4 Length2(_SimdFloat4 _v) {
591 const float sq_len = _v.x * _v.x + _v.y * _v.y;
592 const SimdFloat4 ret = {std::sqrt(sq_len), _v.x, _v.x, _v.x};
596OZZ_INLINE SimdFloat4 Length3(_SimdFloat4 _v) {
597 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
598 const SimdFloat4 ret = {std::sqrt(sq_len), _v.x, _v.x, _v.x};
602OZZ_INLINE SimdFloat4 Length4(_SimdFloat4 _v) {
603 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
604 const SimdFloat4 ret = {std::sqrt(sq_len), _v.x, _v.x, _v.x};
608OZZ_INLINE SimdFloat4 Length2Sqr(_SimdFloat4 _v) {
609 const float sq_len = _v.x * _v.x + _v.y * _v.y;
610 const SimdFloat4 ret = {sq_len, _v.x, _v.x, _v.x};
614OZZ_INLINE SimdFloat4 Length3Sqr(_SimdFloat4 _v) {
615 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
616 const SimdFloat4 ret = {sq_len, _v.x, _v.x, _v.x};
620OZZ_INLINE SimdFloat4 Length4Sqr(_SimdFloat4 _v) {
621 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
622 const SimdFloat4 ret = {sq_len, _v.x, _v.x, _v.x};
626OZZ_INLINE SimdFloat4 Normalize2(_SimdFloat4 _v) {
627 const float sq_len = _v.x * _v.x + _v.y * _v.y;
628 assert(sq_len != 0.f &&
"_v is not normalizable");
629 const float inv_len = 1.f / std::sqrt(sq_len);
630 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z, _v.w};
634OZZ_INLINE SimdFloat4 Normalize3(_SimdFloat4 _v) {
635 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
636 assert(sq_len != 0.f &&
"_v is not normalizable");
637 const float inv_len = 1.f / std::sqrt(sq_len);
638 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len, _v.w};
642OZZ_INLINE SimdFloat4 Normalize4(_SimdFloat4 _v) {
643 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
644 assert(sq_len != 0.f &&
"_v is not normalizable");
645 const float inv_len = 1.f / std::sqrt(sq_len);
646 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len,
651OZZ_INLINE SimdFloat4 NormalizeEst2(_SimdFloat4 _v) {
652 const float sq_len = _v.x * _v.x + _v.y * _v.y;
653 assert(sq_len != 0.f &&
"_v is not normalizable");
655 OZZ_RSQRT_EST(sq_len, inv_len);
656 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z, _v.w};
660OZZ_INLINE SimdFloat4 NormalizeEst3(_SimdFloat4 _v) {
661 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
662 assert(sq_len != 0.f &&
"_v is not normalizable");
664 OZZ_RSQRT_EST(sq_len, inv_len);
665 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len, _v.w};
669OZZ_INLINE SimdFloat4 NormalizeEst4(_SimdFloat4 _v) {
670 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
671 assert(sq_len != 0.f &&
"_v is not normalizable");
673 OZZ_RSQRT_EST(sq_len, inv_len);
674 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len,
679OZZ_INLINE SimdInt4 IsNormalized2(_SimdFloat4 _v) {
680 const float sq_len = _v.x * _v.x + _v.y * _v.y;
681 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceSq;
682 const SimdInt4 ret = {-
static_cast<int>(normalized), 0, 0, 0};
686OZZ_INLINE SimdInt4 IsNormalized3(_SimdFloat4 _v) {
687 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
688 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceSq;
689 const SimdInt4 ret = {-
static_cast<int>(normalized), 0, 0, 0};
693OZZ_INLINE SimdInt4 IsNormalized4(_SimdFloat4 _v) {
694 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
695 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceSq;
696 const SimdInt4 ret = {-
static_cast<int>(normalized), 0, 0, 0};
700OZZ_INLINE SimdInt4 IsNormalizedEst2(_SimdFloat4 _v) {
701 const float sq_len = _v.x * _v.x + _v.y * _v.y;
702 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceEstSq;
703 const SimdInt4 ret = {-
static_cast<int>(normalized), 0, 0, 0};
707OZZ_INLINE SimdInt4 IsNormalizedEst3(_SimdFloat4 _v) {
708 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
709 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceEstSq;
710 const SimdInt4 ret = {-
static_cast<int>(normalized), 0, 0, 0};
714OZZ_INLINE SimdInt4 IsNormalizedEst4(_SimdFloat4 _v) {
715 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
716 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceEstSq;
717 const SimdInt4 ret = {-
static_cast<int>(normalized), 0, 0, 0};
721OZZ_INLINE SimdFloat4 NormalizeSafe2(_SimdFloat4 _v, _SimdFloat4 _safe) {
723 const float sq_len = _v.x * _v.x + _v.y * _v.y;
725 const SimdFloat4 ret = {_safe.x, _safe.y, _v.z, _v.w};
728 const float inv_len = 1.f / std::sqrt(sq_len);
729 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z, _v.w};
733OZZ_INLINE SimdFloat4 NormalizeSafe3(_SimdFloat4 _v, _SimdFloat4 _safe) {
735 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
737 const SimdFloat4 ret = {_safe.x, _safe.y, _safe.z, _v.w};
740 const float inv_len = 1.f / std::sqrt(sq_len);
741 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len, _v.w};
745OZZ_INLINE SimdFloat4 NormalizeSafe4(_SimdFloat4 _v, _SimdFloat4 _safe) {
747 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
751 const float inv_len = 1.f / std::sqrt(sq_len);
752 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len,
757OZZ_INLINE SimdFloat4 NormalizeSafeEst2(_SimdFloat4 _v, _SimdFloat4 _safe) {
759 const float sq_len = _v.x * _v.x + _v.y * _v.y;
761 const SimdFloat4 ret = {_safe.x, _safe.y, _v.z, _v.w};
765 OZZ_RSQRT_EST(sq_len, inv_len);
766 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z, _v.w};
770OZZ_INLINE SimdFloat4 NormalizeSafeEst3(_SimdFloat4 _v, _SimdFloat4 _safe) {
772 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
774 const SimdFloat4 ret = {_safe.x, _safe.y, _safe.z, _v.w};
778 OZZ_RSQRT_EST(sq_len, inv_len);
779 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len, _v.w};
783OZZ_INLINE SimdFloat4 NormalizeSafeEst4(_SimdFloat4 _v, _SimdFloat4 _safe) {
785 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
790 OZZ_RSQRT_EST(sq_len, inv_len);
791 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len,
796OZZ_INLINE SimdFloat4 Lerp(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _alpha) {
797 const SimdFloat4 ret = {
798 (_b.x - _a.x) * _alpha.x + _a.x, (_b.y - _a.y) * _alpha.y + _a.y,
799 (_b.z - _a.z) * _alpha.z + _a.z, (_b.w - _a.w) * _alpha.w + _a.w};
803OZZ_INLINE SimdFloat4 Min(_SimdFloat4 _a, _SimdFloat4 _b) {
804 const SimdFloat4 ret = {_a.x < _b.x ? _a.x : _b.x, _a.y < _b.y ? _a.y : _b.y,
805 _a.z < _b.z ? _a.z : _b.z, _a.w < _b.w ? _a.w : _b.w};
809OZZ_INLINE SimdFloat4 Max(_SimdFloat4 _a, _SimdFloat4 _b) {
810 const SimdFloat4 ret = {_a.x > _b.x ? _a.x : _b.x, _a.y > _b.y ? _a.y : _b.y,
811 _a.z > _b.z ? _a.z : _b.z, _a.w > _b.w ? _a.w : _b.w};
815OZZ_INLINE SimdFloat4 Min0(_SimdFloat4 _v) {
816 const SimdFloat4 ret = {_v.x < 0.f ? _v.x : 0.f, _v.y < 0.f ? _v.y : 0.f,
817 _v.z < 0.f ? _v.z : 0.f, _v.w < 0.f ? _v.w : 0.f};
821OZZ_INLINE SimdFloat4 Max0(_SimdFloat4 _v) {
822 const SimdFloat4 ret = {_v.x > 0.f ? _v.x : 0.f, _v.y > 0.f ? _v.y : 0.f,
823 _v.z > 0.f ? _v.z : 0.f, _v.w > 0.f ? _v.w : 0.f};
827OZZ_INLINE SimdFloat4 Clamp(_SimdFloat4 _a, _SimdFloat4 _v, _SimdFloat4 _b) {
828 const SimdFloat4 min = {_v.x < _b.x ? _v.x : _b.x, _v.y < _b.y ? _v.y : _b.y,
829 _v.z < _b.z ? _v.z : _b.z, _v.w < _b.w ? _v.w : _b.w};
830 const SimdFloat4 r = {
831 _a.x > min.x ? _a.x : min.x, _a.y > min.y ? _a.y : min.y,
832 _a.z > min.z ? _a.z : min.z, _a.w > min.w ? _a.w : min.w};
836OZZ_INLINE SimdFloat4 Select(_SimdInt4 _b, _SimdFloat4 _true,
837 _SimdFloat4 _false) {
838 using internal::SimdFI4;
839 using internal::SimdIF4;
841 const SimdFI4 i_true = {_true};
842 const SimdFI4 i_false = {_false};
843 const SimdIF4 ret = {{i_false.i.x ^ (_b.x & (i_true.i.x ^ i_false.i.x)),
844 i_false.i.y ^ (_b.y & (i_true.i.y ^ i_false.i.y)),
845 i_false.i.z ^ (_b.z & (i_true.i.z ^ i_false.i.z)),
846 i_false.i.w ^ (_b.w & (i_true.i.w ^ i_false.i.w))}};
850OZZ_INLINE SimdInt4 CmpEq(_SimdFloat4 _a, _SimdFloat4 _b) {
851 const SimdInt4 ret = {
852 -
static_cast<int>(_a.x == _b.x), -
static_cast<int>(_a.y == _b.y),
853 -
static_cast<int>(_a.z == _b.z), -
static_cast<int>(_a.w == _b.w)};
857OZZ_INLINE SimdInt4 CmpNe(_SimdFloat4 _a, _SimdFloat4 _b) {
858 const SimdInt4 ret = {
859 -
static_cast<int>(_a.x != _b.x), -
static_cast<int>(_a.y != _b.y),
860 -
static_cast<int>(_a.z != _b.z), -
static_cast<int>(_a.w != _b.w)};
864OZZ_INLINE SimdInt4 CmpLt(_SimdFloat4 _a, _SimdFloat4 _b) {
865 const SimdInt4 ret = {
866 -
static_cast<int>(_a.x < _b.x), -
static_cast<int>(_a.y < _b.y),
867 -
static_cast<int>(_a.z < _b.z), -
static_cast<int>(_a.w < _b.w)};
871OZZ_INLINE SimdInt4 CmpLe(_SimdFloat4 _a, _SimdFloat4 _b) {
872 const SimdInt4 ret = {
873 -
static_cast<int>(_a.x <= _b.x), -
static_cast<int>(_a.y <= _b.y),
874 -
static_cast<int>(_a.z <= _b.z), -
static_cast<int>(_a.w <= _b.w)};
878OZZ_INLINE SimdInt4 CmpGt(_SimdFloat4 _a, _SimdFloat4 _b) {
879 const SimdInt4 ret = {
880 -
static_cast<int>(_a.x > _b.x), -
static_cast<int>(_a.y > _b.y),
881 -
static_cast<int>(_a.z > _b.z), -
static_cast<int>(_a.w > _b.w)};
885OZZ_INLINE SimdInt4 CmpGe(_SimdFloat4 _a, _SimdFloat4 _b) {
886 const SimdInt4 ret = {
887 -
static_cast<int>(_a.x >= _b.x), -
static_cast<int>(_a.y >= _b.y),
888 -
static_cast<int>(_a.z >= _b.z), -
static_cast<int>(_a.w >= _b.w)};
892OZZ_INLINE SimdFloat4 And(_SimdFloat4 _a, _SimdFloat4 _b) {
893 const internal::SimdFI4 a = {_a};
894 const internal::SimdFI4 b = {_b};
895 const internal::SimdIF4 ret = {
896 {a.i.x & b.i.x, a.i.y & b.i.y, a.i.z & b.i.z, a.i.w & b.i.w}};
900OZZ_INLINE SimdFloat4 Or(_SimdFloat4 _a, _SimdFloat4 _b) {
901 const internal::SimdFI4 a = {_a};
902 const internal::SimdFI4 b = {_b};
903 const internal::SimdIF4 ret = {
904 {a.i.x | b.i.x, a.i.y | b.i.y, a.i.z | b.i.z, a.i.w | b.i.w}};
908OZZ_INLINE SimdFloat4 Xor(_SimdFloat4 _a, _SimdFloat4 _b) {
909 const internal::SimdFI4 a = {_a};
910 const internal::SimdFI4 b = {_b};
911 const internal::SimdIF4 ret = {
912 {a.i.x ^ b.i.x, a.i.y ^ b.i.y, a.i.z ^ b.i.z, a.i.w ^ b.i.w}};
916OZZ_INLINE SimdFloat4 And(_SimdFloat4 _a, _SimdInt4 _b) {
917 const internal::SimdFI4 a = {_a};
918 const internal::SimdIF4 ret = {
919 {a.i.x & _b.x, a.i.y & _b.y, a.i.z & _b.z, a.i.w & _b.w}};
923OZZ_INLINE SimdFloat4 AndNot(_SimdFloat4 _a, _SimdInt4 _b) {
924 const internal::SimdFI4 a = {_a};
925 const internal::SimdIF4 ret = {
926 {a.i.x & ~_b.x, a.i.y & ~_b.y, a.i.z & ~_b.z, a.i.w & ~_b.w}};
930OZZ_INLINE SimdFloat4 Or(_SimdFloat4 _a, _SimdInt4 _b) {
931 const internal::SimdFI4 a = {_a};
932 const internal::SimdIF4 ret = {
933 {a.i.x | _b.x, a.i.y | _b.y, a.i.z | _b.z, a.i.w | _b.w}};
937OZZ_INLINE SimdFloat4 Xor(_SimdFloat4 _a, _SimdInt4 _b) {
938 const internal::SimdFI4 a = {_a};
939 const internal::SimdIF4 ret = {
940 {a.i.x ^ _b.x, a.i.y ^ _b.y, a.i.z ^ _b.z, a.i.w ^ _b.w}};
944OZZ_INLINE SimdFloat4 Cos(_SimdFloat4 _v) {
945 const SimdFloat4 ret = {std::cos(_v.x), std::cos(_v.y), std::cos(_v.z),
950OZZ_INLINE SimdFloat4 CosX(_SimdFloat4 _v) {
951 const SimdFloat4 ret = {std::cos(_v.x), _v.y, _v.z, _v.w};
955OZZ_INLINE SimdFloat4 ACos(_SimdFloat4 _v) {
956 const SimdFloat4 ret = {std::acos(_v.x), std::acos(_v.y), std::acos(_v.z),
961OZZ_INLINE SimdFloat4 ACosX(_SimdFloat4 _v) {
962 const SimdFloat4 ret = {std::acos(_v.x), _v.y, _v.z, _v.w};
966OZZ_INLINE SimdFloat4 Sin(_SimdFloat4 _v) {
967 const SimdFloat4 ret = {std::sin(_v.x), std::sin(_v.y), std::sin(_v.z),
972OZZ_INLINE SimdFloat4 SinX(_SimdFloat4 _v) {
973 const SimdFloat4 ret = {std::sin(_v.x), _v.y, _v.z, _v.w};
977OZZ_INLINE SimdFloat4 ASin(_SimdFloat4 _v) {
978 const SimdFloat4 ret = {std::asin(_v.x), std::asin(_v.y), std::asin(_v.z),
983OZZ_INLINE SimdFloat4 ASinX(_SimdFloat4 _v) {
984 const SimdFloat4 ret = {std::asin(_v.x), _v.y, _v.z, _v.w};
988OZZ_INLINE SimdFloat4 Tan(_SimdFloat4 _v) {
989 const SimdFloat4 ret = {std::tan(_v.x), std::tan(_v.y), std::tan(_v.z),
994OZZ_INLINE SimdFloat4 TanX(_SimdFloat4 _v) {
995 const SimdFloat4 ret = {std::tan(_v.x), _v.y, _v.z, _v.w};
999OZZ_INLINE SimdFloat4 ATan(_SimdFloat4 _v) {
1000 const SimdFloat4 ret = {std::atan(_v.x), std::atan(_v.y), std::atan(_v.z),
1005OZZ_INLINE SimdFloat4 ATanX(_SimdFloat4 _v) {
1006 const SimdFloat4 ret = {std::atan(_v.x), _v.y, _v.z, _v.w};
1010namespace simd_int4 {
1012OZZ_INLINE SimdInt4
zero() {
1013 const SimdInt4 ret = {0, 0, 0, 0};
1017OZZ_INLINE SimdInt4 one() {
1018 const SimdInt4 ret = {1, 1, 1, 1};
1022OZZ_INLINE SimdInt4 x_axis() {
1023 const SimdInt4 ret = {1, 0, 0, 0};
1027OZZ_INLINE SimdInt4 y_axis() {
1028 const SimdInt4 ret = {0, 1, 0, 0};
1032OZZ_INLINE SimdInt4 z_axis() {
1033 const SimdInt4 ret = {0, 0, 1, 0};
1037OZZ_INLINE SimdInt4 w_axis() {
1038 const SimdInt4 ret = {0, 0, 0, 1};
1042OZZ_INLINE SimdInt4 all_true() {
1043 const SimdInt4 ret = {~0, ~0, ~0, ~0};
1047OZZ_INLINE SimdInt4 all_false() {
1048 const SimdInt4 ret = {0, 0, 0, 0};
1052OZZ_INLINE SimdInt4 mask_sign() {
1053 const SimdInt4 ret = {
1054 static_cast<int>(0x80000000),
static_cast<int>(0x80000000),
1055 static_cast<int>(0x80000000),
static_cast<int>(0x80000000)};
1059OZZ_INLINE SimdInt4 mask_sign_xyz() {
1060 const SimdInt4 ret = {
1061 static_cast<int>(0x80000000),
static_cast<int>(0x80000000),
1062 static_cast<int>(0x80000000),
static_cast<int>(0x00000000)};
1066OZZ_INLINE SimdInt4 mask_sign_w() {
1067 const SimdInt4 ret = {
1068 static_cast<int>(0x00000000),
static_cast<int>(0x00000000),
1069 static_cast<int>(0x00000000),
static_cast<int>(0x80000000)};
1073OZZ_INLINE SimdInt4 mask_not_sign() {
1074 const SimdInt4 ret = {
1075 static_cast<int>(0x7fffffff),
static_cast<int>(0x7fffffff),
1076 static_cast<int>(0x7fffffff),
static_cast<int>(0x7fffffff)};
1080OZZ_INLINE SimdInt4 mask_ffff() {
1081 const SimdInt4 ret = {~0, ~0, ~0, ~0};
1085OZZ_INLINE SimdInt4 mask_fff0() {
1086 const SimdInt4 ret = {~0, ~0, ~0, 0};
1090OZZ_INLINE SimdInt4 mask_0000() {
1091 const SimdInt4 ret = {0, 0, 0, 0};
1095OZZ_INLINE SimdInt4 mask_f000() {
1096 const SimdInt4 ret = {~0, 0, 0, 0};
1100OZZ_INLINE SimdInt4 mask_0f00() {
1101 const SimdInt4 ret = {0, ~0, 0, 0};
1105OZZ_INLINE SimdInt4 mask_00f0() {
1106 const SimdInt4 ret = {0, 0, ~0, 0};
1110OZZ_INLINE SimdInt4 mask_000f() {
1111 const SimdInt4 ret = {0, 0, 0, ~0};
1115OZZ_INLINE SimdInt4 Load(
int _x,
int _y,
int _z,
int _w) {
1116 const SimdInt4 ret = {_x, _y, _z, _w};
1120OZZ_INLINE SimdInt4 LoadX(
int _x) {
1121 const SimdInt4 ret = {_x, 0, 0, 0};
1125OZZ_INLINE SimdInt4 Load1(
int _x) {
1126 const SimdInt4 ret = {_x, _x, _x, _x};
1130OZZ_INLINE SimdInt4 Load(
bool _x,
bool _y,
bool _z,
bool _w) {
1131 const SimdInt4 ret = {-
static_cast<int>(_x), -
static_cast<int>(_y),
1132 -
static_cast<int>(_z), -
static_cast<int>(_w)};
1136OZZ_INLINE SimdInt4 LoadX(
bool _x) {
1137 const SimdInt4 ret = {-
static_cast<int>(_x), 0, 0, 0};
1141OZZ_INLINE SimdInt4 Load1(
bool _x) {
1142 const int i = -
static_cast<int>(_x);
1143 const SimdInt4 ret = {i, i, i, i};
1147OZZ_INLINE SimdInt4 LoadPtr(
const int* _i) {
1148 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1149 const SimdInt4 ret = {_i[0], _i[1], _i[2], _i[3]};
1153OZZ_INLINE SimdInt4 LoadXPtr(
const int* _i) {
1154 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1155 const SimdInt4 ret = {*_i, 0, 0, 0};
1159OZZ_INLINE SimdInt4 Load1Ptr(
const int* _i) {
1160 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1161 const SimdInt4 ret = {*_i, *_i, *_i, *_i};
1165OZZ_INLINE SimdInt4 Load2Ptr(
const int* _i) {
1166 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1167 const SimdInt4 ret = {_i[0], _i[1], 0, 0};
1171OZZ_INLINE SimdInt4 Load3Ptr(
const int* _i) {
1172 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1173 const SimdInt4 ret = {_i[0], _i[1], _i[2], 0};
1177OZZ_INLINE SimdInt4 LoadPtrU(
const int* _i) {
1178 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1179 const SimdInt4 ret = {_i[0], _i[1], _i[2], _i[3]};
1183OZZ_INLINE SimdInt4 LoadXPtrU(
const int* _i) {
1184 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1185 const SimdInt4 ret = {*_i, 0, 0, 0};
1189OZZ_INLINE SimdInt4 Load1PtrU(
const int* _i) {
1190 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1191 const SimdInt4 ret = {*_i, *_i, *_i, *_i};
1195OZZ_INLINE SimdInt4 Load2PtrU(
const int* _i) {
1196 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1197 const SimdInt4 ret = {_i[0], _i[1], 0, 0};
1201OZZ_INLINE SimdInt4 Load3PtrU(
const int* _i) {
1202 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1203 const SimdInt4 ret = {_i[0], _i[1], _i[2], 0};
1207OZZ_INLINE SimdInt4 FromFloatRound(_SimdFloat4 _f) {
1208 const SimdInt4 ret = {
1209 static_cast<int>(floor(_f.x + .5f)),
static_cast<int>(floor(_f.y + .5f)),
1210 static_cast<int>(floor(_f.z + .5f)),
static_cast<int>(floor(_f.w + .5f))};
1214OZZ_INLINE SimdInt4 FromFloatTrunc(_SimdFloat4 _f) {
1215 const SimdInt4 ret = {
static_cast<int>(_f.x),
static_cast<int>(_f.y),
1216 static_cast<int>(_f.z),
static_cast<int>(_f.w)};
1221OZZ_INLINE
int GetX(_SimdInt4 _v) {
return _v.x; }
1223OZZ_INLINE
int GetY(_SimdInt4 _v) {
return _v.y; }
1225OZZ_INLINE
int GetZ(_SimdInt4 _v) {
return _v.z; }
1227OZZ_INLINE
int GetW(_SimdInt4 _v) {
return _v.w; }
1229OZZ_INLINE SimdInt4 SetX(_SimdInt4 _v, _SimdInt4 _i) {
1230 const SimdInt4 ret = {_i.x, _v.y, _v.z, _v.w};
1234OZZ_INLINE SimdInt4 SetY(_SimdInt4 _v, _SimdInt4 _i) {
1235 const SimdInt4 ret = {_v.x, _i.x, _v.z, _v.w};
1239OZZ_INLINE SimdInt4 SetZ(_SimdInt4 _v, _SimdInt4 _i) {
1240 const SimdInt4 ret = {_v.x, _v.y, _i.x, _v.w};
1244OZZ_INLINE SimdInt4 SetW(_SimdInt4 _v, _SimdInt4 _i) {
1245 const SimdInt4 ret = {_v.x, _v.y, _v.z, _i.x};
1249OZZ_INLINE SimdInt4 SetI(_SimdInt4 _v, _SimdInt4 _i,
int _ith) {
1250 assert(_ith >= 0 && _ith <= 3 &&
"Invalid index, out of range.");
1252 (&ret.x)[_ith] = _i.x;
1256OZZ_INLINE
void StorePtr(_SimdInt4 _v,
int* _i) {
1257 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1264OZZ_INLINE
void Store1Ptr(_SimdInt4 _v,
int* _i) {
1265 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1269OZZ_INLINE
void Store2Ptr(_SimdInt4 _v,
int* _i) {
1270 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1275OZZ_INLINE
void Store3Ptr(_SimdInt4 _v,
int* _i) {
1276 assert(!(uintptr_t(_i) & 0xf) &&
"Invalid alignment");
1282OZZ_INLINE
void StorePtrU(_SimdInt4 _v,
int* _i) {
1283 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1290OZZ_INLINE
void Store1PtrU(_SimdInt4 _v,
int* _i) {
1291 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1295OZZ_INLINE
void Store2PtrU(_SimdInt4 _v,
int* _i) {
1296 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1301OZZ_INLINE
void Store3PtrU(_SimdInt4 _v,
int* _i) {
1302 assert(!(uintptr_t(_i) & 0x3) &&
"Invalid alignment");
1308OZZ_INLINE SimdInt4 SplatX(_SimdInt4 _a) {
1309 const SimdInt4 ret = {_a.x, _a.x, _a.x, _a.x};
1313OZZ_INLINE SimdInt4 SplatY(_SimdInt4 _a) {
1314 const SimdInt4 ret = {_a.y, _a.y, _a.y, _a.y};
1318OZZ_INLINE SimdInt4 SplatZ(_SimdInt4 _a) {
1319 const SimdInt4 ret = {_a.z, _a.z, _a.z, _a.z};
1323OZZ_INLINE SimdInt4 SplatW(_SimdInt4 _a) {
1324 const SimdInt4 ret = {_a.w, _a.w, _a.w, _a.w};
1328template <
size_t _X,
size_t _Y,
size_t _Z,
size_t _W>
1329OZZ_INLINE SimdInt4 Swizzle(_SimdInt4 _v) {
1330 static_assert(_X <= 3 && _Y <= 3 && _Z <= 3 && _W <= 3,
1331 "Indices must be between 0 and 3");
1332 const int*
pi = &_v.x;
1333 const SimdInt4 ret = {
pi[_X],
pi[_Y],
pi[_Z],
pi[_W]};
1337OZZ_INLINE
int MoveMask(_SimdInt4 _v) {
1338 return ((_v.x & 0x80000000) >> 31) | ((_v.y & 0x80000000) >> 30) |
1339 ((_v.z & 0x80000000) >> 29) | ((_v.w & 0x80000000) >> 28);
1342OZZ_INLINE
bool AreAllTrue(_SimdInt4 _v) {
1343 return _v.x != 0 && _v.y != 0 && _v.z != 0 && _v.w != 0;
1346OZZ_INLINE
bool AreAllTrue3(_SimdInt4 _v) {
1347 return _v.x != 0 && _v.y != 0 && _v.z != 0;
1350OZZ_INLINE
bool AreAllTrue2(_SimdInt4 _v) {
return _v.x != 0 && _v.y != 0; }
1352OZZ_INLINE
bool AreAllTrue1(_SimdInt4 _v) {
return _v.x != 0; }
1354OZZ_INLINE
bool AreAllFalse(_SimdInt4 _v) {
1355 return _v.x == 0 && _v.y == 0 && _v.z == 0 && _v.w == 0;
1358OZZ_INLINE
bool AreAllFalse3(_SimdInt4 _v) {
1359 return _v.x == 0 && _v.y == 0 && _v.z == 0;
1362OZZ_INLINE
bool AreAllFalse2(_SimdInt4 _v) {
return _v.x == 0 && _v.y == 0; }
1364OZZ_INLINE
bool AreAllFalse1(_SimdInt4 _v) {
return _v.x == 0; }
1366OZZ_INLINE SimdInt4 MAdd(_SimdInt4 _a, _SimdInt4 _b, _SimdInt4 _addend) {
1367 const SimdInt4 ret = {_a.x * _b.x + _addend.x, _a.y * _b.y + _addend.y,
1368 _a.z * _b.z + _addend.z, _a.w * _b.w + _addend.w};
1372OZZ_INLINE SimdInt4 DivX(_SimdInt4 _a, _SimdInt4 _b) {
1373 const SimdInt4 ret = {_a.x / _b.x, _a.y, _a.z, _a.w};
1377OZZ_INLINE SimdInt4 HAdd2(_SimdInt4 _v) {
1378 const SimdInt4 ret = {_v.x + _v.y, _v.y, _v.z, _v.w};
1382OZZ_INLINE SimdInt4 HAdd3(_SimdInt4 _v) {
1383 const SimdInt4 ret = {_v.x + _v.y + _v.z, _v.y, _v.z, _v.w};
1387OZZ_INLINE SimdInt4 HAdd4(_SimdInt4 _v) {
1388 const SimdInt4 ret = {_v.x + _v.y + _v.z + _v.w, _v.y, _v.z, _v.w};
1392OZZ_INLINE SimdInt4 Dot2(_SimdInt4 _a, _SimdInt4 _b) {
1393 const SimdInt4 ret = {_a.x * _b.x + _a.y * _b.y, _a.y, _a.z, _a.w};
1397OZZ_INLINE SimdInt4 Dot3(_SimdInt4 _a, _SimdInt4 _b) {
1398 const SimdInt4 ret = {_a.x * _b.x + _a.y * _b.y + _a.z * _b.z, _a.y, _a.z,
1403OZZ_INLINE SimdInt4 Dot4(_SimdInt4 _a, _SimdInt4 _b) {
1404 const SimdInt4 ret = {_a.x * _b.x + _a.y * _b.y + _a.z * _b.z + _a.w * _b.w,
1408OZZ_INLINE SimdInt4 Abs(_SimdInt4 _v) {
1409 const SimdInt4 mash = {_v.x >> 31, _v.y >> 31, _v.z >> 31, _v.w >> 31};
1410 const SimdInt4 ret = {
1411 (_v.x + (mash.x)) ^ (mash.x), (_v.y + (mash.y)) ^ (mash.y),
1412 (_v.z + (mash.z)) ^ (mash.z), (_v.w + (mash.w)) ^ (mash.w)};
1416OZZ_INLINE SimdInt4 Sign(_SimdInt4 _v) {
1417 const SimdInt4 ret = {
1418 _v.x &
static_cast<int>(0x80000000), _v.y &
static_cast<int>(0x80000000),
1419 _v.z &
static_cast<int>(0x80000000), _v.w &
static_cast<int>(0x80000000)};
1423OZZ_INLINE SimdInt4 Min(_SimdInt4 _a, _SimdInt4 _b) {
1424 const SimdInt4 ret = {_a.x < _b.x ? _a.x : _b.x, _a.y < _b.y ? _a.y : _b.y,
1425 _a.z < _b.z ? _a.z : _b.z, _a.w < _b.w ? _a.w : _b.w};
1429OZZ_INLINE SimdInt4 Max(_SimdInt4 _a, _SimdInt4 _b) {
1430 const SimdInt4 ret = {_a.x > _b.x ? _a.x : _b.x, _a.y > _b.y ? _a.y : _b.y,
1431 _a.z > _b.z ? _a.z : _b.z, _a.w > _b.w ? _a.w : _b.w};
1435OZZ_INLINE SimdInt4 Min0(_SimdInt4 _v) {
1436 const SimdInt4 ret = {_v.x < 0 ? _v.x : 0, _v.y < 0 ? _v.y : 0,
1437 _v.z < 0 ? _v.z : 0, _v.w < 0 ? _v.w : 0};
1441OZZ_INLINE SimdInt4 Max0(_SimdInt4 _v) {
1442 const SimdInt4 ret = {_v.x > 0 ? _v.x : 0, _v.y > 0 ? _v.y : 0,
1443 _v.z > 0 ? _v.z : 0, _v.w > 0 ? _v.w : 0};
1447OZZ_INLINE SimdInt4 Clamp(_SimdInt4 _a, _SimdInt4 _v, _SimdInt4 _b) {
1448 const SimdInt4 min = {_v.x < _b.x ? _v.x : _b.x, _v.y < _b.y ? _v.y : _b.y,
1449 _v.z < _b.z ? _v.z : _b.z, _v.w < _b.w ? _v.w : _b.w};
1450 const SimdInt4 r = {_a.x > min.x ? _a.x : min.x, _a.y > min.y ? _a.y : min.y,
1451 _a.z > min.z ? _a.z : min.z, _a.w > min.w ? _a.w : min.w};
1455OZZ_INLINE SimdInt4 Select(_SimdInt4 _b, _SimdInt4 _true, _SimdInt4 _false) {
1456 const SimdInt4 ret = {_false.x ^ (_b.x & (_true.x ^ _false.x)),
1457 _false.y ^ (_b.y & (_true.y ^ _false.y)),
1458 _false.z ^ (_b.z & (_true.z ^ _false.z)),
1459 _false.w ^ (_b.w & (_true.w ^ _false.w))};
1463OZZ_INLINE SimdInt4 And(_SimdInt4 _a, _SimdInt4 _b) {
1464 const SimdInt4 ret = {_a.x & _b.x, _a.y & _b.y, _a.z & _b.z, _a.w & _b.w};
1468OZZ_INLINE SimdInt4 AndNot(_SimdInt4 _a, _SimdInt4 _b) {
1469 const SimdInt4 ret = {_a.x & ~_b.x, _a.y & ~_b.y, _a.z & ~_b.z, _a.w & ~_b.w};
1473OZZ_INLINE SimdInt4 Or(_SimdInt4 _a, _SimdInt4 _b) {
1474 const SimdInt4 ret = {_a.x | _b.x, _a.y | _b.y, _a.z | _b.z, _a.w | _b.w};
1478OZZ_INLINE SimdInt4 Xor(_SimdInt4 _a, _SimdInt4 _b) {
1479 const SimdInt4 ret = {_a.x ^ _b.x, _a.y ^ _b.y, _a.z ^ _b.z, _a.w ^ _b.w};
1483OZZ_INLINE SimdInt4 Not(_SimdInt4 _v) {
1484 const SimdInt4 ret = {~_v.x, ~_v.y, ~_v.z, ~_v.w};
1488OZZ_INLINE SimdInt4 ShiftL(_SimdInt4 _v,
int _bits) {
1489 const SimdInt4 ret = {_v.x << _bits, _v.y << _bits, _v.z << _bits,
1494OZZ_INLINE SimdInt4 ShiftR(_SimdInt4 _v,
int _bits) {
1495 const SimdInt4 ret = {_v.x >> _bits, _v.y >> _bits, _v.z >> _bits,
1500OZZ_INLINE SimdInt4 ShiftRu(_SimdInt4 _v,
int _bits) {
1504 } iu = {{_v.x, _v.y, _v.z, _v.w}};
1509 {iu.u[0] >> _bits, iu.u[1] >> _bits, iu.u[2] >> _bits, iu.u[3] >> _bits}};
1510 const SimdInt4 ret = {ui.i[0], ui.i[1], ui.i[2], ui.i[3]};
1514OZZ_INLINE SimdInt4 CmpEq(_SimdInt4 _a, _SimdInt4 _b) {
1515 const SimdInt4 ret = {
1516 -
static_cast<int>(_a.x == _b.x), -
static_cast<int>(_a.y == _b.y),
1517 -
static_cast<int>(_a.z == _b.z), -
static_cast<int>(_a.w == _b.w)};
1521OZZ_INLINE SimdInt4 CmpNe(_SimdInt4 _a, _SimdInt4 _b) {
1522 const SimdInt4 ret = {
1523 -
static_cast<int>(_a.x != _b.x), -
static_cast<int>(_a.y != _b.y),
1524 -
static_cast<int>(_a.z != _b.z), -
static_cast<int>(_a.w != _b.w)};
1528OZZ_INLINE SimdInt4 CmpLt(_SimdInt4 _a, _SimdInt4 _b) {
1529 const SimdInt4 ret = {
1530 -
static_cast<int>(_a.x < _b.x), -
static_cast<int>(_a.y < _b.y),
1531 -
static_cast<int>(_a.z < _b.z), -
static_cast<int>(_a.w < _b.w)};
1535OZZ_INLINE SimdInt4 CmpLe(_SimdInt4 _a, _SimdInt4 _b) {
1536 const SimdInt4 ret = {
1537 -
static_cast<int>(_a.x <= _b.x), -
static_cast<int>(_a.y <= _b.y),
1538 -
static_cast<int>(_a.z <= _b.z), -
static_cast<int>(_a.w <= _b.w)};
1542OZZ_INLINE SimdInt4 CmpGt(_SimdInt4 _a, _SimdInt4 _b) {
1543 const SimdInt4 ret = {
1544 -
static_cast<int>(_a.x > _b.x), -
static_cast<int>(_a.y > _b.y),
1545 -
static_cast<int>(_a.z > _b.z), -
static_cast<int>(_a.w > _b.w)};
1549OZZ_INLINE SimdInt4 CmpGe(_SimdInt4 _a, _SimdInt4 _b) {
1550 const SimdInt4 ret = {
1551 -
static_cast<int>(_a.x >= _b.x), -
static_cast<int>(_a.y >= _b.y),
1552 -
static_cast<int>(_a.z >= _b.z), -
static_cast<int>(_a.w >= _b.w)};
1556OZZ_INLINE Float4x4 Float4x4::identity() {
1557 const Float4x4 ret = {{{1.f, 0.f, 0.f, 0.f},
1558 {0.f, 1.f, 0.f, 0.f},
1559 {0.f, 0.f, 1.f, 0.f},
1560 {0.f, 0.f, 0.f, 1.f}}};
1564OZZ_INLINE Float4x4 Transpose(
const Float4x4& _m) {
1565 const Float4x4 ret = {
1566 {{_m.cols[0].x, _m.cols[1].x, _m.cols[2].x, _m.cols[3].x},
1567 {_m.cols[0].y, _m.cols[1].y, _m.cols[2].y, _m.cols[3].y},
1568 {_m.cols[0].z, _m.cols[1].z, _m.cols[2].z, _m.cols[3].z},
1569 {_m.cols[0].w, _m.cols[1].w, _m.cols[2].w, _m.cols[3].w}}};
1573OZZ_INLINE Float4x4 Invert(
const Float4x4& _m, SimdInt4* _invertible) {
1574 const SimdFloat4* cols = _m.cols;
1575 const float a00 = cols[2].z * cols[3].w - cols[3].z * cols[2].w;
1576 const float a01 = cols[2].y * cols[3].w - cols[3].y * cols[2].w;
1577 const float a02 = cols[2].y * cols[3].z - cols[3].y * cols[2].z;
1578 const float a03 = cols[2].x * cols[3].w - cols[3].x * cols[2].w;
1579 const float a04 = cols[2].x * cols[3].z - cols[3].x * cols[2].z;
1580 const float a05 = cols[2].x * cols[3].y - cols[3].x * cols[2].y;
1581 const float a06 = cols[1].z * cols[3].w - cols[3].z * cols[1].w;
1582 const float a07 = cols[1].y * cols[3].w - cols[3].y * cols[1].w;
1583 const float a08 = cols[1].y * cols[3].z - cols[3].y * cols[1].z;
1584 const float a09 = cols[1].x * cols[3].w - cols[3].x * cols[1].w;
1585 const float a10 = cols[1].x * cols[3].z - cols[3].x * cols[1].z;
1586 const float a11 = cols[1].y * cols[3].w - cols[3].y * cols[1].w;
1587 const float a12 = cols[1].x * cols[3].y - cols[3].x * cols[1].y;
1588 const float a13 = cols[1].z * cols[2].w - cols[2].z * cols[1].w;
1589 const float a14 = cols[1].y * cols[2].w - cols[2].y * cols[1].w;
1590 const float a15 = cols[1].y * cols[2].z - cols[2].y * cols[1].z;
1591 const float a16 = cols[1].x * cols[2].w - cols[2].x * cols[1].w;
1592 const float a17 = cols[1].x * cols[2].z - cols[2].x * cols[1].z;
1593 const float a18 = cols[1].x * cols[2].y - cols[2].x * cols[1].y;
1595 const float b0x = cols[1].y * a00 - cols[1].z * a01 + cols[1].w * a02;
1596 const float b1x = -cols[1].x * a00 + cols[1].z * a03 - cols[1].w * a04;
1597 const float b2x = cols[1].x * a01 - cols[1].y * a03 + cols[1].w * a05;
1598 const float b3x = -cols[1].x * a02 + cols[1].y * a04 - cols[1].z * a05;
1600 const float b0y = -cols[0].y * a00 + cols[0].z * a01 - cols[0].w * a02;
1601 const float b1y = cols[0].x * a00 - cols[0].z * a03 + cols[0].w * a04;
1602 const float b2y = -cols[0].x * a01 + cols[0].y * a03 - cols[0].w * a05;
1603 const float b3y = cols[0].x * a02 - cols[0].y * a04 + cols[0].z * a05;
1605 const float b0z = cols[0].y * a06 - cols[0].z * a07 + cols[0].w * a08;
1606 const float b1z = -cols[0].x * a06 + cols[0].z * a09 - cols[0].w * a10;
1607 const float b2z = cols[0].x * a11 - cols[0].y * a09 + cols[0].w * a12;
1608 const float b3z = -cols[0].x * a08 + cols[0].y * a10 - cols[0].z * a12;
1610 const float b0w = -cols[0].y * a13 + cols[0].z * a14 - cols[0].w * a15;
1611 const float b1w = cols[0].x * a13 - cols[0].z * a16 + cols[0].w * a17;
1612 const float b2w = -cols[0].x * a14 + cols[0].y * a16 - cols[0].w * a18;
1613 const float b3w = cols[0].x * a15 - cols[0].y * a17 + cols[0].z * a18;
1616 cols[0].x * b0x + cols[0].y * b1x + cols[0].z * b2x + cols[0].w * b3x;
1617 const bool invertible = det != 0.f;
1618 assert((_invertible || invertible) &&
"Matrix is not invertible");
1619 if (_invertible !=
nullptr) {
1620 *_invertible = simd_int4::LoadX(invertible);
1622 const float inv_det = invertible ? 1.f / det : 0.f;
1624 const Float4x4 ret = {
1625 {{b0x * inv_det, b0y * inv_det, b0z * inv_det, b0w * inv_det},
1626 {b1x * inv_det, b1y * inv_det, b1z * inv_det, b1w * inv_det},
1627 {b2x * inv_det, b2y * inv_det, b2z * inv_det, b2w * inv_det},
1628 {b3x * inv_det, b3y * inv_det, b3z * inv_det, b3w * inv_det}}};
1632Float4x4 Float4x4::Scaling(_SimdFloat4 _v) {
1633 const Float4x4 ret = {{{_v.x, 0.f, 0.f, 0.f},
1634 {0.f, _v.y, 0.f, 0.f},
1635 {0.f, 0.f, _v.z, 0.f},
1636 {0.f, 0.f, 0.f, 1.f}}};
1640Float4x4 Float4x4::Translation(_SimdFloat4 _v) {
1641 const Float4x4 ret = {{{1.f, 0.f, 0.f, 0.f},
1642 {0.f, 1.f, 0.f, 0.f},
1643 {0.f, 0.f, 1.f, 0.f},
1644 {_v.x, _v.y, _v.z, 1.f}}};
1648OZZ_INLINE Float4x4 Translate(
const Float4x4& _m, _SimdFloat4 _v) {
1649 const Float4x4 ret = {{_m.cols[0],
1652 {_m.cols[0].x * _v.x + _m.cols[1].x * _v.y +
1653 _m.cols[2].x * _v.z + _m.cols[3].x,
1654 _m.cols[0].y * _v.x + _m.cols[1].y * _v.y +
1655 _m.cols[2].y * _v.z + _m.cols[3].y,
1656 _m.cols[0].z * _v.x + _m.cols[1].z * _v.y +
1657 _m.cols[2].z * _v.z + _m.cols[3].z,
1658 _m.cols[0].w * _v.x + _m.cols[1].w * _v.y +
1659 _m.cols[2].w * _v.z + _m.cols[3].w}}};
1663OZZ_INLINE Float4x4 Scale(
const Float4x4& _m, _SimdFloat4 _v) {
1664 const Float4x4 ret = {{{_m.cols[0].x * _v.x, _m.cols[0].y * _v.x,
1665 _m.cols[0].z * _v.x, _m.cols[0].w * _v.x},
1666 {_m.cols[1].x * _v.y, _m.cols[1].y * _v.y,
1667 _m.cols[1].z * _v.y, _m.cols[1].w * _v.y},
1668 {_m.cols[2].x * _v.z, _m.cols[2].y * _v.z,
1669 _m.cols[2].z * _v.z, _m.cols[2].w * _v.z},
1674OZZ_INLINE Float4x4 ColumnMultiply(
const Float4x4& _m, _SimdFloat4 _v) {
1675 const Float4x4 ret = {{{_m.cols[0].x * _v.x, _m.cols[0].y * _v.y,
1676 _m.cols[0].z * _v.z, _m.cols[0].w * _v.w},
1677 {_m.cols[1].x * _v.x, _m.cols[1].y * _v.y,
1678 _m.cols[1].z * _v.z, _m.cols[1].w * _v.w},
1679 {_m.cols[2].x * _v.x, _m.cols[2].y * _v.y,
1680 _m.cols[2].z * _v.z, _m.cols[2].w * _v.w},
1681 {_m.cols[3].x * _v.x, _m.cols[3].y * _v.y,
1682 _m.cols[3].z * _v.z, _m.cols[3].w * _v.w}}};
1686OZZ_INLINE SimdInt4 IsNormalized(
const Float4x4& _m) {
1687 const SimdInt4 ret = {IsNormalized3(_m.cols[0]).x,
1688 IsNormalized3(_m.cols[1]).x,
1689 IsNormalized3(_m.cols[2]).x, 0};
1693OZZ_INLINE SimdInt4 IsNormalizedEst(
const Float4x4& _m) {
1694 const SimdInt4 ret = {IsNormalizedEst3(_m.cols[0]).x,
1695 IsNormalizedEst3(_m.cols[1]).x,
1696 IsNormalizedEst3(_m.cols[2]).x, 0};
1700OZZ_INLINE SimdInt4 IsOrthogonal(
const Float4x4& _m) {
1703 const SimdFloat4
cross =
1704 NormalizeSafe3(Cross3(_m.cols[0], _m.cols[1]), simd_float4::zero());
1705 const SimdFloat4 at = NormalizeSafe3(_m.cols[2], simd_float4::zero());
1707 const float sq_len =
cross.x * at.x +
cross.y * at.y +
cross.z * at.z;
1708 const bool same = std::abs(sq_len - 1.f) < kNormalizationToleranceSq;
1709 const SimdInt4 ret = {-
static_cast<int>(same), 0, 0, 0};
1713OZZ_INLINE SimdFloat4 ToQuaternion(
const Float4x4& _m) {
1714 assert(AreAllTrue3(IsNormalized(_m)));
1715 assert(AreAllTrue1(IsOrthogonal(_m)));
1718 if (_m.cols[0].x + _m.cols[1].y + _m.cols[2].z > .0f) {
1719 const float t = _m.cols[0].x + _m.cols[1].y + _m.cols[2].z + 1.0f;
1720 const float s = (1.f / std::sqrt(t)) * .5f;
1721 ret.x = (_m.cols[1].z - _m.cols[2].y) * s;
1722 ret.y = (_m.cols[2].x - _m.cols[0].z) * s;
1723 ret.z = (_m.cols[0].y - _m.cols[1].x) * s;
1725 }
else if (_m.cols[0].x > _m.cols[1].y && _m.cols[0].x > _m.cols[2].z) {
1726 const float t = _m.cols[0].x - _m.cols[1].y - _m.cols[2].z + 1.0f;
1727 const float s = (1.f / std::sqrt(t)) * .5f;
1729 ret.y = (_m.cols[0].y + _m.cols[1].x) * s;
1730 ret.z = (_m.cols[2].x + _m.cols[0].z) * s;
1731 ret.w = (_m.cols[1].z - _m.cols[2].y) * s;
1732 }
else if (_m.cols[1].y > _m.cols[2].z) {
1733 const float t = -_m.cols[0].x + _m.cols[1].y - _m.cols[2].z + 1.0f;
1734 const float s = (1.f / std::sqrt(t)) * .5f;
1735 ret.x = (_m.cols[0].y + _m.cols[1].x) * s;
1737 ret.z = (_m.cols[1].z + _m.cols[2].y) * s;
1738 ret.w = (_m.cols[2].x - _m.cols[0].z) * s;
1740 const float t = -_m.cols[0].x - _m.cols[1].y + _m.cols[2].z + 1.0f;
1741 const float s = (1.f / std::sqrt(t)) * .5f;
1742 ret.x = (_m.cols[2].x + _m.cols[0].z) * s;
1743 ret.y = (_m.cols[1].z + _m.cols[2].y) * s;
1745 ret.w = (_m.cols[0].y - _m.cols[1].x) * s;
1747 assert(AreAllTrue1(IsNormalizedEst4(ret)));
1751OZZ_INLINE
bool ToAffine(
const Float4x4& _m, SimdFloat4* _translation,
1752 SimdFloat4* _quaternion, SimdFloat4* _scale) {
1753 _translation->x = _m.cols[3].x;
1754 _translation->y = _m.cols[3].y;
1755 _translation->z = _m.cols[3].z;
1756 _translation->w = 1.f;
1759 const float sq_scale_x = Length3Sqr(_m.cols[0]).x;
1760 const float scale_x = std::sqrt(sq_scale_x);
1761 const float sq_scale_y = Length3Sqr(_m.cols[1]).x;
1762 const float scale_y = std::sqrt(sq_scale_y);
1763 const float sq_scale_z = Length3Sqr(_m.cols[2]).x;
1764 const float scale_z = std::sqrt(sq_scale_z);
1767 const bool x_zero = std::abs(sq_scale_x) < kOrthogonalisationToleranceSq;
1768 const bool y_zero = std::abs(sq_scale_y) < kOrthogonalisationToleranceSq;
1769 const bool z_zero = std::abs(sq_scale_z) < kOrthogonalisationToleranceSq;
1771 Float4x4 orthonormal;
1773 if (y_zero || z_zero) {
1776 orthonormal.cols[1].x = _m.cols[1].x / scale_y;
1777 orthonormal.cols[1].y = _m.cols[1].y / scale_y;
1778 orthonormal.cols[1].z = _m.cols[1].z / scale_y;
1779 orthonormal.cols[1].w = 0.f;
1780 orthonormal.cols[0] = Normalize3(Cross3(orthonormal.cols[1], _m.cols[2]));
1781 orthonormal.cols[2] =
1782 Normalize3(Cross3(orthonormal.cols[0], orthonormal.cols[1]));
1783 }
else if (z_zero) {
1784 if (x_zero || y_zero) {
1787 orthonormal.cols[0].x = _m.cols[0].x / scale_x;
1788 orthonormal.cols[0].y = _m.cols[0].y / scale_x;
1789 orthonormal.cols[0].z = _m.cols[0].z / scale_x;
1790 orthonormal.cols[0].w = 0.f;
1791 orthonormal.cols[2] = Normalize3(Cross3(orthonormal.cols[0], _m.cols[1]));
1792 orthonormal.cols[1] =
1793 Normalize3(Cross3(orthonormal.cols[2], orthonormal.cols[0]));
1795 if (x_zero || z_zero) {
1798 orthonormal.cols[2].x = _m.cols[2].x / scale_z;
1799 orthonormal.cols[2].y = _m.cols[2].y / scale_z;
1800 orthonormal.cols[2].z = _m.cols[2].z / scale_z;
1801 orthonormal.cols[2].w = 0.f;
1802 orthonormal.cols[1] = Normalize3(Cross3(orthonormal.cols[2], _m.cols[0]));
1803 orthonormal.cols[0] =
1804 Normalize3(Cross3(orthonormal.cols[1], orthonormal.cols[2]));
1811 Dot3(orthonormal.cols[0], _m.cols[0]).x > 0.f ? scale_x : -scale_x;
1813 Dot3(orthonormal.cols[1], _m.cols[1]).x > 0.f ? scale_y : -scale_y;
1815 Dot3(orthonormal.cols[2], _m.cols[2]).x > 0.f ? scale_z : -scale_z;
1819 *_quaternion = ToQuaternion(orthonormal);
1823OZZ_INLINE Float4x4 Float4x4::FromEuler(_SimdFloat4 _v) {
1824 return Float4x4::FromAxisAngle(simd_float4::y_axis(), SplatX(_v)) *
1825 Float4x4::FromAxisAngle(simd_float4::x_axis(), SplatY(_v)) *
1826 Float4x4::FromAxisAngle(simd_float4::z_axis(), SplatZ(_v));
1829OZZ_INLINE Float4x4 Float4x4::FromAxisAngle(_SimdFloat4 _axis,
1830 _SimdFloat4 _angle) {
1831 assert(AreAllTrue1(IsNormalizedEst3(_axis)));
1833 const float cos = std::cos(_angle.x);
1834 const float sin = std::sin(_angle.x);
1835 const float t = 1.f -
cos;
1837 const float a = _axis.x * _axis.y * t;
1838 const float b = _axis.z *
sin;
1839 const float c = _axis.x * _axis.z * t;
1840 const float d = _axis.y *
sin;
1841 const float e = _axis.y * _axis.z * t;
1842 const float f = _axis.x *
sin;
1844 const Float4x4 ret = {{{
cos + _axis.x * _axis.x * t, a + b, c - d, 0.f},
1845 {a - b,
cos + _axis.y * _axis.y * t, e + f, 0.f},
1846 {c + d, e - f,
cos + _axis.z * _axis.z * t, 0.f},
1847 {0.f, 0.f, 0.f, 1.f}}};
1851OZZ_INLINE Float4x4 Float4x4::FromQuaternion(_SimdFloat4 _v) {
1852 assert(AreAllTrue1(IsNormalizedEst4(_v)));
1854 const float xx = _v.x * _v.x;
1855 const float xy = _v.x * _v.y;
1856 const float xz = _v.x * _v.z;
1857 const float xw = _v.x * _v.w;
1858 const float yy = _v.y * _v.y;
1859 const float yz = _v.y * _v.z;
1860 const float yw = _v.y * _v.w;
1861 const float zz = _v.z * _v.z;
1862 const float zw = _v.z * _v.w;
1864 const Float4x4 ret = {
1865 {{1.f - 2.f * (yy + zz), 2.f * (xy + zw), 2.f * (xz - yw), 0.f},
1866 {2.f * (xy - zw), 1.f - 2.f * (xx + zz), 2.f * (yz + xw), 0.f},
1867 {2.f * (xz + yw), 2.f * (yz - xw), 1.f - 2.f * (xx + yy), 0.f},
1868 {0.f, 0.f, 0.f, 1.f}}};
1872OZZ_INLINE Float4x4 Float4x4::FromAffine(_SimdFloat4 _translation,
1873 _SimdFloat4 _quaternion,
1874 _SimdFloat4 _scale) {
1875 assert(AreAllTrue1(IsNormalizedEst4(_quaternion)));
1877 const float xx = _quaternion.x * _quaternion.x;
1878 const float xy = _quaternion.x * _quaternion.y;
1879 const float xz = _quaternion.x * _quaternion.z;
1880 const float xw = _quaternion.x * _quaternion.w;
1881 const float yy = _quaternion.y * _quaternion.y;
1882 const float yz = _quaternion.y * _quaternion.z;
1883 const float yw = _quaternion.y * _quaternion.w;
1884 const float zz = _quaternion.z * _quaternion.z;
1885 const float zw = _quaternion.z * _quaternion.w;
1887 const Float4x4 ret = {
1888 {{_scale.x * (1.f - 2.f * (yy + zz)), _scale.x * 2.f * (xy + zw),
1889 _scale.x * 2.f * (xz - yw), 0.f},
1890 {_scale.y * 2.f * (xy - zw), _scale.y * (1.f - 2.f * (xx + zz)),
1891 _scale.y * (2.f * (yz + xw)), 0.f},
1892 {_scale.z * 2.f * (xz + yw), _scale.z * 2.f * (yz - xw),
1893 _scale.z * (1.f - 2.f * (xx + yy)), 0.f},
1894 {_translation.x, _translation.y, _translation.z, 1.f}}};
1901 _m.cols[2].x * _v.z + _m.cols[3].x,
1902 _m.cols[0].y * _v.x + _m.cols[1].y * _v.y +
1903 _m.cols[2].y * _v.z + _m.cols[3].y,
1904 _m.cols[0].z * _v.x + _m.cols[1].z * _v.y +
1905 _m.cols[2].z * _v.z + _m.cols[3].z,
1906 _m.cols[0].w * _v.x + _m.cols[1].w * _v.y +
1907 _m.cols[2].w * _v.z + _m.cols[3].w};
1914 _m.cols[0].x * _v.x + _m.cols[1].x * _v.y + _m.cols[2].x * _v.z,
1915 _m.cols[0].y * _v.x + _m.cols[1].y * _v.y + _m.cols[2].y * _v.z,
1916 _m.cols[0].z * _v.x + _m.cols[1].z * _v.y + _m.cols[2].z * _v.z,
1917 _m.cols[0].w * _v.x + _m.cols[1].w * _v.y + _m.cols[2].w * _v.z};
1924 _m.cols[0].x * _v.x + _m.cols[1].x * _v.y + _m.cols[2].x * _v.z +
1925 _m.cols[3].x * _v.w,
1926 _m.cols[0].y * _v.x + _m.cols[1].y * _v.y + _m.cols[2].y * _v.z +
1927 _m.cols[3].y * _v.w,
1928 _m.cols[0].z * _v.x + _m.cols[1].z * _v.y + _m.cols[2].z * _v.z +
1929 _m.cols[3].z * _v.w,
1930 _m.cols[0].w * _v.x + _m.cols[1].w * _v.y + _m.cols[2].w * _v.z +
1931 _m.cols[3].w * _v.w};
1938 {_a * _b.cols[0], _a * _b.cols[1], _a * _b.cols[2], _a * _b.cols[3]}};
1945 {{_a.cols[0].x + _b.cols[0].x, _a.cols[0].y + _b.cols[0].y,
1946 _a.cols[0].z + _b.cols[0].z, _a.cols[0].w + _b.cols[0].w},
1947 {_a.cols[1].x + _b.cols[1].x, _a.cols[1].y + _b.cols[1].y,
1948 _a.cols[1].z + _b.cols[1].z, _a.cols[1].w + _b.cols[1].w},
1949 {_a.cols[2].x + _b.cols[2].x, _a.cols[2].y + _b.cols[2].y,
1950 _a.cols[2].z + _b.cols[2].z, _a.cols[2].w + _b.cols[2].w},
1951 {_a.cols[3].x + _b.cols[3].x, _a.cols[3].y + _b.cols[3].y,
1952 _a.cols[3].z + _b.cols[3].z, _a.cols[3].w + _b.cols[3].w}}};
1959 {{_a.cols[0].x - _b.cols[0].x, _a.cols[0].y - _b.cols[0].y,
1960 _a.cols[0].z - _b.cols[0].z, _a.cols[0].w - _b.cols[0].w},
1961 {_a.cols[1].x - _b.cols[1].x, _a.cols[1].y - _b.cols[1].y,
1962 _a.cols[1].z - _b.cols[1].z, _a.cols[1].w - _b.cols[1].w},
1963 {_a.cols[2].x - _b.cols[2].x, _a.cols[2].y - _b.cols[2].y,
1964 _a.cols[2].z - _b.cols[2].z, _a.cols[2].w - _b.cols[2].w},
1965 {_a.cols[3].x - _b.cols[3].x, _a.cols[3].y - _b.cols[3].y,
1966 _a.cols[3].z - _b.cols[3].z, _a.cols[3].w - _b.cols[3].w}}};
2009OZZ_INLINE
uint16_t FloatToHalf(
float _f) {
2010 const uint32_t f32infty = 255 << 23;
2011 const uint32_t f16infty = 31 << 23;
2015 } magic = {15 << 23};
2016 const uint32_t sign_mask = 0x80000000u;
2017 const uint32_t round_mask = ~0x00000fffu;
2023 const uint32_t sign = f.u & sign_mask;
2024 const uint32_t f_nosign = f.u & ~sign_mask;
2026 if (f_nosign >= f32infty) {
2029 ((f_nosign > f32infty) ? 0x7e00 : 0x7c00) | (sign >> 16);
2030 return static_cast<uint16_t>(result);
2035 } rounded = {f_nosign & round_mask};
2039 }
exp = {rounded.f * magic.f};
2043 ((re_rounded > f16infty ? f16infty : re_rounded) >> 13) | (sign >> 16);
2044 return static_cast<uint16_t>(result);
2048OZZ_INLINE
float HalfToFloat(uint16_t _h) {
2052 } magic = {(254 - 15) << 23};
2056 } infnan = {(127 + 16) << 23};
2062 } exp_mant = {(_h & 0x7fff) << 13};
2066 } adjust = {exp_mant.f * magic.f};
2071 } result = {(adjust.f >= infnan.f ? (adjust.u | 255 << 23) : adjust.u) |
2076OZZ_INLINE SimdInt4 FloatToHalf(_SimdFloat4 _f) {
2078 FloatToHalf(_f.z), FloatToHalf(_f.w)};
2082OZZ_INLINE SimdFloat4 HalfToFloat(_SimdInt4 _h) {
2084 HalfToFloat(_h.x & 0x0000ffff), HalfToFloat(_h.y & 0x0000ffff),
2085 HalfToFloat(_h.z & 0x0000ffff), HalfToFloat(_h.w & 0x0000ffff)};
GLM_FUNC_QUALIFIER vec< L, T, Q > exp(vec< L, T, Q > const &x)
Definition func_exponential.inl:80
GLM_FUNC_QUALIFIER vec< 3, T, Q > cross(vec< 3, T, Q > const &x, vec< 3, T, Q > const &y)
Definition func_geometric.inl:175
GLM_FUNC_QUALIFIER vec< L, T, Q > sin(vec< L, T, Q > const &v)
Definition func_trigonometric.inl:41
GLM_FUNC_QUALIFIER vec< L, T, Q > cos(vec< L, T, Q > const &v)
Definition func_trigonometric.inl:50
GLM_FUNC_DECL GLM_CONSTEXPR genType pi()
Return the pi constant for floating point types.
Definition scalar_constants.inl:13
GLM_FUNC_DECL GLM_CONSTEXPR genType zero()
Definition constants.inl:6
uint32 uint32_t
Definition fwd.hpp:131
int32 int32_t
Definition fwd.hpp:71
uint16 uint16_t
Definition fwd.hpp:117
Definition simd_math_config.h:121
Definition simd_math_config.h:129
Definition simd_math.h:1066
Definition simd_math_ref-inl.h:47
Definition simd_math_ref-inl.h:51