RavEngine
Loading...
Searching...
No Matches
simd_math_ref-inl.h
1//----------------------------------------------------------------------------//
2// //
3// ozz-animation is hosted at http://github.com/guillaumeblanc/ozz-animation //
4// and distributed under the MIT License (MIT). //
5// //
6// Copyright (c) Guillaume Blanc //
7// //
8// Permission is hereby granted, free of charge, to any person obtaining a //
9// copy of this software and associated documentation files (the "Software"), //
10// to deal in the Software without restriction, including without limitation //
11// the rights to use, copy, modify, merge, publish, distribute, sublicense, //
12// and/or sell copies of the Software, and to permit persons to whom the //
13// Software is furnished to do so, subject to the following conditions: //
14// //
15// The above copyright notice and this permission notice shall be included in //
16// all copies or substantial portions of the Software. //
17// //
18// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR //
19// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, //
20// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL //
21// THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER //
22// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING //
23// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER //
24// DEALINGS IN THE SOFTWARE. //
25// //
26//----------------------------------------------------------------------------//
27
28#ifndef OZZ_OZZ_BASE_MATHS_INTERNAL_SIMD_MATH_REF_INL_H_
29#define OZZ_OZZ_BASE_MATHS_INTERNAL_SIMD_MATH_REF_INL_H_
30
31// SIMD refence implementation, based on scalar floats.
32
33#include <stdint.h>
34
35#include <cassert>
36#include <cmath>
37#include <cstddef>
38
39#include "ozz/base/maths/math_constant.h"
40
41namespace ozz {
42namespace math {
43
44namespace internal {
45// Defines union cast helpers that are used internally for binary logical
46// operations.
47union SimdFI4 {
48 SimdFloat4 f;
49 SimdInt4 i;
50};
51union SimdIF4 {
52 SimdInt4 i;
53 SimdFloat4 f;
54};
55} // namespace internal
56
57#define OZZ_RCP_EST(_in, _out) \
58 do { \
59 const float in = _in; \
60 const union { \
61 float f; \
62 int i; \
63 } uf = {in}; \
64 const union { \
65 int i; \
66 float f; \
67 } ui = {(0x3f800000 * 2) - uf.i}; \
68 const float fp = ui.f * (2.f - in * ui.f); \
69 _out = fp * (2.f - in * fp); \
70 } while (void(0), 0)
71
72#define OZZ_RCP_EST_NR(_in, _out) \
73 do { \
74 float fp2; \
75 OZZ_RCP_EST(_in, fp2); \
76 _out = fp2 * (2.f - _in * fp2); \
77 } while (void(0), 0)
78
79#define OZZ_RSQRT_EST(_in, _out) \
80 do { \
81 const float in = _in; \
82 union { \
83 float f; \
84 int i; \
85 } uf = {in}; \
86 union { \
87 int i; \
88 float f; \
89 } ui = {0x5f3759df - (uf.i / 2)}; \
90 const float fp = ui.f * (1.5f - (in * .5f * ui.f * ui.f)); \
91 _out = fp * (1.5f - (in * .5f * fp * fp)); \
92 } while (void(0), 0)
93
94#define OZZ_RSQRT_EST_NR(_in, _out) \
95 do { \
96 float fp2; \
97 OZZ_RSQRT_EST(_in, fp2); \
98 _out = fp2 * (1.5f - (_in * .5f * fp2 * fp2)); \
99 } while (void(0), 0)
100
101namespace simd_float4 {
102
103OZZ_INLINE SimdFloat4 zero() {
104 const SimdFloat4 ret = {0.f, 0.f, 0.f, 0.f};
105 return ret;
106}
107
108OZZ_INLINE SimdFloat4 one() {
109 const SimdFloat4 ret = {1.f, 1.f, 1.f, 1.f};
110 return ret;
111}
112
113OZZ_INLINE SimdFloat4 x_axis() {
114 const SimdFloat4 ret = {1.f, 0.f, 0.f, 0.f};
115 return ret;
116}
117
118OZZ_INLINE SimdFloat4 y_axis() {
119 const SimdFloat4 ret = {0.f, 1.f, 0.f, 0.f};
120 return ret;
121}
122
123OZZ_INLINE SimdFloat4 z_axis() {
124 const SimdFloat4 ret = {0.f, 0.f, 1.f, 0.f};
125 return ret;
126}
127
128OZZ_INLINE SimdFloat4 w_axis() {
129 const SimdFloat4 ret = {0.f, 0.f, 0.f, 1.f};
130 return ret;
131}
132
133OZZ_INLINE SimdFloat4 Load(float _x, float _y, float _z, float _w) {
134 const SimdFloat4 ret = {_x, _y, _z, _w};
135 return ret;
136}
137
138OZZ_INLINE SimdFloat4 LoadX(float _x) {
139 const SimdFloat4 ret = {_x, 0.f, 0.f, 0.f};
140 return ret;
141}
142
143OZZ_INLINE SimdFloat4 Load1(float _x) {
144 const SimdFloat4 ret = {_x, _x, _x, _x};
145 return ret;
146}
147
148OZZ_INLINE SimdFloat4 LoadPtr(const float* _f) {
149 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
150 const SimdFloat4 ret = {_f[0], _f[1], _f[2], _f[3]};
151 return ret;
152}
153
154OZZ_INLINE SimdFloat4 LoadPtrU(const float* _f) {
155 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
156 const SimdFloat4 ret = {_f[0], _f[1], _f[2], _f[3]};
157 return ret;
158}
159
160OZZ_INLINE SimdFloat4 LoadXPtrU(const float* _f) {
161 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
162 const SimdFloat4 ret = {*_f, 0.f, 0.f, 0.f};
163 return ret;
164}
165
166OZZ_INLINE SimdFloat4 Load1PtrU(const float* _f) {
167 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
168 const SimdFloat4 ret = {*_f, *_f, *_f, *_f};
169 return ret;
170}
171
172OZZ_INLINE SimdFloat4 Load2PtrU(const float* _f) {
173 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
174 const SimdFloat4 ret = {_f[0], _f[1], 0.f, 0.f};
175 return ret;
176}
177
178OZZ_INLINE SimdFloat4 Load3PtrU(const float* _f) {
179 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
180 const SimdFloat4 ret = {_f[0], _f[1], _f[2]};
181 return ret;
182}
183
184OZZ_INLINE SimdFloat4 FromInt(_SimdInt4 _i) {
185 const SimdFloat4 ret = {static_cast<float>(_i.x), static_cast<float>(_i.y),
186 static_cast<float>(_i.z), static_cast<float>(_i.w)};
187 return ret;
188}
189} // namespace simd_float4
190
191OZZ_INLINE float GetX(_SimdFloat4 _v) { return _v.x; }
192
193OZZ_INLINE float GetY(_SimdFloat4 _v) { return _v.y; }
194
195OZZ_INLINE float GetZ(_SimdFloat4 _v) { return _v.z; }
196
197OZZ_INLINE float GetW(_SimdFloat4 _v) { return _v.w; }
198OZZ_INLINE SimdFloat4 SetX(_SimdFloat4 _v, _SimdFloat4 _f) {
199 const SimdFloat4 ret = {_f.x, _v.y, _v.z, _v.w};
200 return ret;
201}
202
203OZZ_INLINE SimdFloat4 SetY(_SimdFloat4 _v, _SimdFloat4 _f) {
204 const SimdFloat4 ret = {_v.x, _f.x, _v.z, _v.w};
205 return ret;
206}
207
208OZZ_INLINE SimdFloat4 SetZ(_SimdFloat4 _v, _SimdFloat4 _f) {
209 const SimdFloat4 ret = {_v.x, _v.y, _f.x, _v.w};
210 return ret;
211}
212
213OZZ_INLINE SimdFloat4 SetW(_SimdFloat4 _v, _SimdFloat4 _f) {
214 const SimdFloat4 ret = {_v.x, _v.y, _v.z, _f.x};
215 return ret;
216}
217
218OZZ_INLINE SimdFloat4 SetI(_SimdFloat4 _v, _SimdFloat4 _f, int _ith) {
219 assert(_ith >= 0 && _ith <= 3 && "Invalid index, out of range.");
220 SimdFloat4 ret = _v;
221 (&ret.x)[_ith] = _f.x;
222 return ret;
223}
224
225OZZ_INLINE void StorePtr(_SimdFloat4 _v, float* _f) {
226 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
227 _f[0] = _v.x;
228 _f[1] = _v.y;
229 _f[2] = _v.z;
230 _f[3] = _v.w;
231}
232
233OZZ_INLINE void Store1Ptr(_SimdFloat4 _v, float* _f) {
234 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
235 _f[0] = _v.x;
236}
237
238OZZ_INLINE void Store2Ptr(_SimdFloat4 _v, float* _f) {
239 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
240 _f[0] = _v.x;
241 _f[1] = _v.y;
242}
243
244OZZ_INLINE void Store3Ptr(_SimdFloat4 _v, float* _f) {
245 assert(!(reinterpret_cast<uintptr_t>(_f) & 0xf) && "Invalid alignment");
246 _f[0] = _v.x;
247 _f[1] = _v.y;
248 _f[2] = _v.z;
249}
250
251OZZ_INLINE void StorePtrU(_SimdFloat4 _v, float* _f) {
252 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
253 _f[0] = _v.x;
254 _f[1] = _v.y;
255 _f[2] = _v.z;
256 _f[3] = _v.w;
257}
258
259OZZ_INLINE void Store1PtrU(_SimdFloat4 _v, float* _f) {
260 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
261 _f[0] = _v.x;
262}
263
264OZZ_INLINE void Store2PtrU(_SimdFloat4 _v, float* _f) {
265 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
266 _f[0] = _v.x;
267 _f[1] = _v.y;
268}
269
270OZZ_INLINE void Store3PtrU(_SimdFloat4 _v, float* _f) {
271 assert(!(reinterpret_cast<uintptr_t>(_f) & 0x3) && "Invalid alignment");
272 _f[0] = _v.x;
273 _f[1] = _v.y;
274 _f[2] = _v.z;
275}
276
277OZZ_INLINE SimdFloat4 SplatX(_SimdFloat4 _v) {
278 const SimdFloat4 ret = {_v.x, _v.x, _v.x, _v.x};
279 return ret;
280}
281
282OZZ_INLINE SimdFloat4 SplatY(_SimdFloat4 _v) {
283 const SimdFloat4 ret = {_v.y, _v.y, _v.y, _v.y};
284 return ret;
285}
286
287OZZ_INLINE SimdFloat4 SplatZ(_SimdFloat4 _v) {
288 const SimdFloat4 ret = {_v.z, _v.z, _v.z, _v.z};
289 return ret;
290}
291
292OZZ_INLINE SimdFloat4 SplatW(_SimdFloat4 _v) {
293 const SimdFloat4 ret = {_v.w, _v.w, _v.w, _v.w};
294 return ret;
295}
296
297template <size_t _X, size_t _Y, size_t _Z, size_t _W>
298OZZ_INLINE SimdFloat4 Swizzle(_SimdFloat4 _v) {
299 static_assert(_X <= 3 && _Y <= 3 && _Z <= 3 && _W <= 3,
300 "Indices must be between 0 and 3");
301 const float* pf = &_v.x;
302 const SimdFloat4 ret = {pf[_X], pf[_Y], pf[_Z], pf[_W]};
303 return ret;
304}
305
306OZZ_INLINE void Transpose4x1(const SimdFloat4 _in[4], SimdFloat4 _out[1]) {
307 _out[0].x = _in[0].x;
308 _out[0].y = _in[1].x;
309 _out[0].z = _in[2].x;
310 _out[0].w = _in[3].x;
311}
312
313OZZ_INLINE void Transpose1x4(const SimdFloat4 _in[1], SimdFloat4 _out[4]) {
314 _out[0].x = _in[0].x;
315 _out[0].y = _out[0].z = _out[0].w = 0.f;
316 _out[1].x = _in[0].y;
317 _out[1].y = _out[1].z = _out[1].w = 0.f;
318 _out[2].x = _in[0].z;
319 _out[2].y = _out[2].z = _out[2].w = 0.f;
320 _out[3].x = _in[0].w;
321 _out[3].y = _out[3].z = _out[3].w = 0.f;
322}
323
324OZZ_INLINE void Transpose4x2(const SimdFloat4 _in[4], SimdFloat4 _out[2]) {
325 _out[0].x = _in[0].x;
326 _out[0].y = _in[1].x;
327 _out[0].z = _in[2].x;
328 _out[0].w = _in[3].x;
329 _out[1].x = _in[0].y;
330 _out[1].y = _in[1].y;
331 _out[1].z = _in[2].y;
332 _out[1].w = _in[3].y;
333}
334
335OZZ_INLINE void Transpose2x4(const SimdFloat4 _in[2], SimdFloat4 _out[4]) {
336 _out[0].x = _in[0].x;
337 _out[0].y = _in[1].x;
338 _out[0].z = _out[0].w = 0.f;
339 _out[1].x = _in[0].y;
340 _out[1].y = _in[1].y;
341 _out[1].z = _out[1].w = 0.f;
342 _out[2].x = _in[0].z;
343 _out[2].y = _in[1].z;
344 _out[2].z = _out[2].w = 0.f;
345 _out[3].x = _in[0].w;
346 _out[3].y = _in[1].w;
347 _out[3].z = _out[3].w = 0.f;
348}
349
350OZZ_INLINE void Transpose4x3(const SimdFloat4 _in[4], SimdFloat4 _out[3]) {
351 _out[0].x = _in[0].x;
352 _out[0].y = _in[1].x;
353 _out[0].z = _in[2].x;
354 _out[0].w = _in[3].x;
355 _out[1].x = _in[0].y;
356 _out[1].y = _in[1].y;
357 _out[1].z = _in[2].y;
358 _out[1].w = _in[3].y;
359 _out[2].x = _in[0].z;
360 _out[2].y = _in[1].z;
361 _out[2].z = _in[2].z;
362 _out[2].w = _in[3].z;
363}
364
365OZZ_INLINE void Transpose3x4(const SimdFloat4 _in[3], SimdFloat4 _out[4]) {
366 _out[0].x = _in[0].x;
367 _out[0].y = _in[1].x;
368 _out[0].z = _in[2].x;
369 _out[0].w = 0.f;
370 _out[1].x = _in[0].y;
371 _out[1].y = _in[1].y;
372 _out[1].z = _in[2].y;
373 _out[1].w = 0.f;
374 _out[2].x = _in[0].z;
375 _out[2].y = _in[1].z;
376 _out[2].z = _in[2].z;
377 _out[2].w = 0.f;
378 _out[3].x = _in[0].w;
379 _out[3].y = _in[1].w;
380 _out[3].z = _in[2].w;
381 _out[3].w = 0.f;
382}
383
384OZZ_INLINE void Transpose4x4(const SimdFloat4 _in[4], SimdFloat4 _out[4]) {
385 _out[0].x = _in[0].x;
386 _out[1].x = _in[0].y;
387 _out[2].x = _in[0].z;
388 _out[3].x = _in[0].w;
389 _out[0].y = _in[1].x;
390 _out[1].y = _in[1].y;
391 _out[2].y = _in[1].z;
392 _out[3].y = _in[1].w;
393 _out[0].z = _in[2].x;
394 _out[1].z = _in[2].y;
395 _out[2].z = _in[2].z;
396 _out[3].z = _in[2].w;
397 _out[0].w = _in[3].x;
398 _out[1].w = _in[3].y;
399 _out[2].w = _in[3].z;
400 _out[3].w = _in[3].w;
401}
402
403OZZ_INLINE void Transpose16x16(const SimdFloat4 _in[16], SimdFloat4 _out[16]) {
404 for (int i = 0; i < 4; ++i) {
405 const int i4 = i * 4;
406 _out[i4 + 0].x = *(&_in[0].x + i);
407 _out[i4 + 0].y = *(&_in[1].x + i);
408 _out[i4 + 0].z = *(&_in[2].x + i);
409 _out[i4 + 0].w = *(&_in[3].x + i);
410 _out[i4 + 1].x = *(&_in[4].x + i);
411 _out[i4 + 1].y = *(&_in[5].x + i);
412 _out[i4 + 1].z = *(&_in[6].x + i);
413 _out[i4 + 1].w = *(&_in[7].x + i);
414 _out[i4 + 2].x = *(&_in[8].x + i);
415 _out[i4 + 2].y = *(&_in[9].x + i);
416 _out[i4 + 2].z = *(&_in[10].x + i);
417 _out[i4 + 2].w = *(&_in[11].x + i);
418 _out[i4 + 3].x = *(&_in[12].x + i);
419 _out[i4 + 3].y = *(&_in[13].x + i);
420 _out[i4 + 3].z = *(&_in[14].x + i);
421 _out[i4 + 3].w = *(&_in[15].x + i);
422 }
423}
424
425OZZ_INLINE SimdFloat4 MAdd(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
426 const SimdFloat4 ret = {_a.x * _b.x + _c.x, _a.y * _b.y + _c.y,
427 _a.z * _b.z + _c.z, _a.w * _b.w + _c.w};
428 return ret;
429}
430
431OZZ_INLINE SimdFloat4 MSub(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
432 const SimdFloat4 ret = {_a.x * _b.x - _c.x, _a.y * _b.y - _c.y,
433 _a.z * _b.z - _c.z, _a.w * _b.w - _c.w};
434 return ret;
435}
436
437OZZ_INLINE SimdFloat4 NMAdd(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
438 const SimdFloat4 ret = {_c.x - _a.x * _b.x, _c.y - _a.y * _b.y,
439 _c.z - _a.z * _b.z, _c.w - _a.w * _b.w};
440 return ret;
441}
442
443OZZ_INLINE SimdFloat4 NMSub(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c) {
444 const SimdFloat4 ret = {-_a.x * _b.x - _c.x, -_a.y * _b.y - _c.y,
445 -_a.z * _b.z - _c.z, -_a.w * _b.w - _c.w};
446 return ret;
447}
448
449OZZ_INLINE SimdFloat4 DivX(_SimdFloat4 _a, _SimdFloat4 _b) {
450 const SimdFloat4 ret = {_a.x / _b.x, _a.y, _a.z, _a.w};
451 return ret;
452}
453
454OZZ_INLINE SimdFloat4 HAdd2(_SimdFloat4 _v) {
455 const SimdFloat4 ret = {_v.x + _v.y, _v.y, _v.z, _v.w};
456 return ret;
457}
458
459OZZ_INLINE SimdFloat4 HAdd3(_SimdFloat4 _v) {
460 const SimdFloat4 ret = {_v.x + _v.y + _v.z, _v.y, _v.z, _v.w};
461 return ret;
462}
463
464OZZ_INLINE SimdFloat4 HAdd4(_SimdFloat4 _v) {
465 const SimdFloat4 ret = {_v.x + _v.y + _v.z + _v.w, _v.x, _v.x, _v.x};
466 return ret;
467}
468
469OZZ_INLINE SimdFloat4 Dot2(_SimdFloat4 _a, _SimdFloat4 _b) {
470 const SimdFloat4 ret = {_a.x * _b.x + _a.y * _b.y, _a.x, _a.x, _a.x};
471 return ret;
472}
473
474OZZ_INLINE SimdFloat4 Dot3(_SimdFloat4 _a, _SimdFloat4 _b) {
475 const SimdFloat4 ret = {_a.x * _b.x + _a.y * _b.y + _a.z * _b.z, _a.x, _a.x,
476 _a.x};
477 return ret;
478}
479
480OZZ_INLINE SimdFloat4 Dot4(_SimdFloat4 _a, _SimdFloat4 _b) {
481 const SimdFloat4 ret = {_a.x * _b.x + _a.y * _b.y + _a.z * _b.z + _a.w * _b.w,
482 _a.x, _a.x, _a.x};
483 return ret;
484}
485
486OZZ_INLINE SimdFloat4 Cross3(_SimdFloat4 _a, _SimdFloat4 _b) {
487 const SimdFloat4 ret = {_a.y * _b.z - _a.z * _b.y, _a.z * _b.x - _a.x * _b.z,
488 _a.x * _b.y - _a.y * _b.x, _a.x};
489 return ret;
490}
491
492OZZ_INLINE SimdFloat4 RcpEst(_SimdFloat4 _v) {
493 SimdFloat4 ret;
494 OZZ_RCP_EST(_v.x, ret.x);
495 OZZ_RCP_EST(_v.y, ret.y);
496 OZZ_RCP_EST(_v.z, ret.z);
497 OZZ_RCP_EST(_v.w, ret.w);
498 return ret;
499}
500
501OZZ_INLINE SimdFloat4 RcpEstNR(_SimdFloat4 _v) {
502 SimdFloat4 ret;
503 OZZ_RCP_EST_NR(_v.x, ret.x);
504 OZZ_RCP_EST_NR(_v.y, ret.y);
505 OZZ_RCP_EST_NR(_v.z, ret.z);
506 OZZ_RCP_EST_NR(_v.w, ret.w);
507 return ret;
508}
509
510OZZ_INLINE SimdFloat4 RcpEstX(_SimdFloat4 _v) {
511 SimdFloat4 ret;
512 OZZ_RCP_EST(_v.x, ret.x);
513 ret.y = _v.y;
514 ret.z = _v.z;
515 ret.w = _v.w;
516 return ret;
517}
518
519OZZ_INLINE SimdFloat4 RcpEstXNR(_SimdFloat4 _v) {
520 SimdFloat4 ret;
521 OZZ_RCP_EST(_v.x, ret.x);
522 ret.y = _v.x;
523 ret.z = _v.x;
524 ret.w = _v.x;
525 return ret;
526}
527
528OZZ_INLINE SimdFloat4 Sqrt(_SimdFloat4 _v) {
529 const SimdFloat4 ret = {std::sqrt(_v.x), std::sqrt(_v.y), std::sqrt(_v.z),
530 std::sqrt(_v.w)};
531 return ret;
532}
533
534OZZ_INLINE SimdFloat4 SqrtX(_SimdFloat4 _v) {
535 const SimdFloat4 ret = {std::sqrt(_v.x), _v.y, _v.z, _v.w};
536 return ret;
537}
538
539OZZ_INLINE SimdFloat4 RSqrtEst(_SimdFloat4 _v) {
540 SimdFloat4 ret;
541 OZZ_RSQRT_EST(_v.x, ret.x);
542 OZZ_RSQRT_EST(_v.y, ret.y);
543 OZZ_RSQRT_EST(_v.z, ret.z);
544 OZZ_RSQRT_EST(_v.w, ret.w);
545 return ret;
546}
547
548OZZ_INLINE SimdFloat4 RSqrtEstNR(_SimdFloat4 _v) {
549 SimdFloat4 ret;
550 OZZ_RSQRT_EST_NR(_v.x, ret.x);
551 OZZ_RSQRT_EST_NR(_v.y, ret.y);
552 OZZ_RSQRT_EST_NR(_v.z, ret.z);
553 OZZ_RSQRT_EST_NR(_v.w, ret.w);
554 return ret;
555}
556
557OZZ_INLINE SimdFloat4 RSqrtEstX(_SimdFloat4 _v) {
558 SimdFloat4 ret;
559 OZZ_RSQRT_EST(_v.x, ret.x);
560 ret.y = _v.y;
561 ret.z = _v.z;
562 ret.w = _v.w;
563 return ret;
564}
565
566OZZ_INLINE SimdFloat4 RSqrtEstXNR(_SimdFloat4 _v) {
567 SimdFloat4 ret;
568 OZZ_RSQRT_EST(_v.x, ret.x);
569 ret.y = _v.x;
570 ret.z = _v.x;
571 ret.w = _v.x;
572 return ret;
573}
574
575OZZ_INLINE SimdFloat4 Abs(_SimdFloat4 _v) {
576 const SimdFloat4 ret = {std::abs(_v.x), std::abs(_v.y), std::abs(_v.z),
577 std::abs(_v.w)};
578 return ret;
579}
580
581OZZ_INLINE SimdInt4 Sign(_SimdFloat4 _v) {
582 internal::SimdFI4 fi = {_v};
583 const SimdInt4 ret = {fi.i.x & static_cast<int>(0x80000000),
584 fi.i.y & static_cast<int>(0x80000000),
585 fi.i.z & static_cast<int>(0x80000000),
586 fi.i.w & static_cast<int>(0x80000000)};
587 return ret;
588}
589
590OZZ_INLINE SimdFloat4 Length2(_SimdFloat4 _v) {
591 const float sq_len = _v.x * _v.x + _v.y * _v.y;
592 const SimdFloat4 ret = {std::sqrt(sq_len), _v.x, _v.x, _v.x};
593 return ret;
594}
595
596OZZ_INLINE SimdFloat4 Length3(_SimdFloat4 _v) {
597 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
598 const SimdFloat4 ret = {std::sqrt(sq_len), _v.x, _v.x, _v.x};
599 return ret;
600}
601
602OZZ_INLINE SimdFloat4 Length4(_SimdFloat4 _v) {
603 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
604 const SimdFloat4 ret = {std::sqrt(sq_len), _v.x, _v.x, _v.x};
605 return ret;
606}
607
608OZZ_INLINE SimdFloat4 Length2Sqr(_SimdFloat4 _v) {
609 const float sq_len = _v.x * _v.x + _v.y * _v.y;
610 const SimdFloat4 ret = {sq_len, _v.x, _v.x, _v.x};
611 return ret;
612}
613
614OZZ_INLINE SimdFloat4 Length3Sqr(_SimdFloat4 _v) {
615 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
616 const SimdFloat4 ret = {sq_len, _v.x, _v.x, _v.x};
617 return ret;
618}
619
620OZZ_INLINE SimdFloat4 Length4Sqr(_SimdFloat4 _v) {
621 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
622 const SimdFloat4 ret = {sq_len, _v.x, _v.x, _v.x};
623 return ret;
624}
625
626OZZ_INLINE SimdFloat4 Normalize2(_SimdFloat4 _v) {
627 const float sq_len = _v.x * _v.x + _v.y * _v.y;
628 assert(sq_len != 0.f && "_v is not normalizable");
629 const float inv_len = 1.f / std::sqrt(sq_len);
630 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z, _v.w};
631 return ret;
632}
633
634OZZ_INLINE SimdFloat4 Normalize3(_SimdFloat4 _v) {
635 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
636 assert(sq_len != 0.f && "_v is not normalizable");
637 const float inv_len = 1.f / std::sqrt(sq_len);
638 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len, _v.w};
639 return ret;
640}
641
642OZZ_INLINE SimdFloat4 Normalize4(_SimdFloat4 _v) {
643 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
644 assert(sq_len != 0.f && "_v is not normalizable");
645 const float inv_len = 1.f / std::sqrt(sq_len);
646 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len,
647 _v.w * inv_len};
648 return ret;
649}
650
651OZZ_INLINE SimdFloat4 NormalizeEst2(_SimdFloat4 _v) {
652 const float sq_len = _v.x * _v.x + _v.y * _v.y;
653 assert(sq_len != 0.f && "_v is not normalizable");
654 float inv_len;
655 OZZ_RSQRT_EST(sq_len, inv_len);
656 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z, _v.w};
657 return ret;
658}
659
660OZZ_INLINE SimdFloat4 NormalizeEst3(_SimdFloat4 _v) {
661 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
662 assert(sq_len != 0.f && "_v is not normalizable");
663 float inv_len;
664 OZZ_RSQRT_EST(sq_len, inv_len);
665 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len, _v.w};
666 return ret;
667}
668
669OZZ_INLINE SimdFloat4 NormalizeEst4(_SimdFloat4 _v) {
670 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
671 assert(sq_len != 0.f && "_v is not normalizable");
672 float inv_len;
673 OZZ_RSQRT_EST(sq_len, inv_len);
674 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len,
675 _v.w * inv_len};
676 return ret;
677}
678
679OZZ_INLINE SimdInt4 IsNormalized2(_SimdFloat4 _v) {
680 const float sq_len = _v.x * _v.x + _v.y * _v.y;
681 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceSq;
682 const SimdInt4 ret = {-static_cast<int>(normalized), 0, 0, 0};
683 return ret;
684}
685
686OZZ_INLINE SimdInt4 IsNormalized3(_SimdFloat4 _v) {
687 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
688 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceSq;
689 const SimdInt4 ret = {-static_cast<int>(normalized), 0, 0, 0};
690 return ret;
691}
692
693OZZ_INLINE SimdInt4 IsNormalized4(_SimdFloat4 _v) {
694 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
695 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceSq;
696 const SimdInt4 ret = {-static_cast<int>(normalized), 0, 0, 0};
697 return ret;
698}
699
700OZZ_INLINE SimdInt4 IsNormalizedEst2(_SimdFloat4 _v) {
701 const float sq_len = _v.x * _v.x + _v.y * _v.y;
702 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceEstSq;
703 const SimdInt4 ret = {-static_cast<int>(normalized), 0, 0, 0};
704 return ret;
705}
706
707OZZ_INLINE SimdInt4 IsNormalizedEst3(_SimdFloat4 _v) {
708 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
709 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceEstSq;
710 const SimdInt4 ret = {-static_cast<int>(normalized), 0, 0, 0};
711 return ret;
712}
713
714OZZ_INLINE SimdInt4 IsNormalizedEst4(_SimdFloat4 _v) {
715 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
716 const bool normalized = std::abs(sq_len - 1.f) < kNormalizationToleranceEstSq;
717 const SimdInt4 ret = {-static_cast<int>(normalized), 0, 0, 0};
718 return ret;
719}
720
721OZZ_INLINE SimdFloat4 NormalizeSafe2(_SimdFloat4 _v, _SimdFloat4 _safe) {
722 // assert(AreAllTrue1(IsNormalized2(_safe)) && "_safe is not normalized");
723 const float sq_len = _v.x * _v.x + _v.y * _v.y;
724 if (sq_len == 0.f) {
725 const SimdFloat4 ret = {_safe.x, _safe.y, _v.z, _v.w};
726 return ret;
727 }
728 const float inv_len = 1.f / std::sqrt(sq_len);
729 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z, _v.w};
730 return ret;
731}
732
733OZZ_INLINE SimdFloat4 NormalizeSafe3(_SimdFloat4 _v, _SimdFloat4 _safe) {
734 // assert(AreAllTrue1(IsNormalized3(_safe)) && "_safe is not normalized");
735 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
736 if (sq_len == 0.f) {
737 const SimdFloat4 ret = {_safe.x, _safe.y, _safe.z, _v.w};
738 return ret;
739 }
740 const float inv_len = 1.f / std::sqrt(sq_len);
741 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len, _v.w};
742 return ret;
743}
744
745OZZ_INLINE SimdFloat4 NormalizeSafe4(_SimdFloat4 _v, _SimdFloat4 _safe) {
746 // assert(AreAllTrue1(IsNormalized4(_safe)) && "_safe is not normalized");
747 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
748 if (sq_len == 0.f) {
749 return _safe;
750 }
751 const float inv_len = 1.f / std::sqrt(sq_len);
752 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len,
753 _v.w * inv_len};
754 return ret;
755}
756
757OZZ_INLINE SimdFloat4 NormalizeSafeEst2(_SimdFloat4 _v, _SimdFloat4 _safe) {
758 // assert(AreAllTrue1(IsNormalizedEst2(_safe)) && "_safe is not normalized");
759 const float sq_len = _v.x * _v.x + _v.y * _v.y;
760 if (sq_len == 0.f) {
761 const SimdFloat4 ret = {_safe.x, _safe.y, _v.z, _v.w};
762 return ret;
763 }
764 float inv_len;
765 OZZ_RSQRT_EST(sq_len, inv_len);
766 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z, _v.w};
767 return ret;
768}
769
770OZZ_INLINE SimdFloat4 NormalizeSafeEst3(_SimdFloat4 _v, _SimdFloat4 _safe) {
771 // assert(AreAllTrue1(IsNormalizedEst3(_safe)) && "_safe is not normalized");
772 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z;
773 if (sq_len == 0.f) {
774 const SimdFloat4 ret = {_safe.x, _safe.y, _safe.z, _v.w};
775 return ret;
776 }
777 float inv_len;
778 OZZ_RSQRT_EST(sq_len, inv_len);
779 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len, _v.w};
780 return ret;
781}
782
783OZZ_INLINE SimdFloat4 NormalizeSafeEst4(_SimdFloat4 _v, _SimdFloat4 _safe) {
784 // assert(AreAllTrue1(IsNormalizedEst4(_safe)) && "_safe is not normalized");
785 const float sq_len = _v.x * _v.x + _v.y * _v.y + _v.z * _v.z + _v.w * _v.w;
786 if (sq_len == 0.f) {
787 return _safe;
788 }
789 float inv_len;
790 OZZ_RSQRT_EST(sq_len, inv_len);
791 const SimdFloat4 ret = {_v.x * inv_len, _v.y * inv_len, _v.z * inv_len,
792 _v.w * inv_len};
793 return ret;
794}
795
796OZZ_INLINE SimdFloat4 Lerp(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _alpha) {
797 const SimdFloat4 ret = {
798 (_b.x - _a.x) * _alpha.x + _a.x, (_b.y - _a.y) * _alpha.y + _a.y,
799 (_b.z - _a.z) * _alpha.z + _a.z, (_b.w - _a.w) * _alpha.w + _a.w};
800 return ret;
801}
802
803OZZ_INLINE SimdFloat4 Min(_SimdFloat4 _a, _SimdFloat4 _b) {
804 const SimdFloat4 ret = {_a.x < _b.x ? _a.x : _b.x, _a.y < _b.y ? _a.y : _b.y,
805 _a.z < _b.z ? _a.z : _b.z, _a.w < _b.w ? _a.w : _b.w};
806 return ret;
807}
808
809OZZ_INLINE SimdFloat4 Max(_SimdFloat4 _a, _SimdFloat4 _b) {
810 const SimdFloat4 ret = {_a.x > _b.x ? _a.x : _b.x, _a.y > _b.y ? _a.y : _b.y,
811 _a.z > _b.z ? _a.z : _b.z, _a.w > _b.w ? _a.w : _b.w};
812 return ret;
813}
814
815OZZ_INLINE SimdFloat4 Min0(_SimdFloat4 _v) {
816 const SimdFloat4 ret = {_v.x < 0.f ? _v.x : 0.f, _v.y < 0.f ? _v.y : 0.f,
817 _v.z < 0.f ? _v.z : 0.f, _v.w < 0.f ? _v.w : 0.f};
818 return ret;
819}
820
821OZZ_INLINE SimdFloat4 Max0(_SimdFloat4 _v) {
822 const SimdFloat4 ret = {_v.x > 0.f ? _v.x : 0.f, _v.y > 0.f ? _v.y : 0.f,
823 _v.z > 0.f ? _v.z : 0.f, _v.w > 0.f ? _v.w : 0.f};
824 return ret;
825}
826
827OZZ_INLINE SimdFloat4 Clamp(_SimdFloat4 _a, _SimdFloat4 _v, _SimdFloat4 _b) {
828 const SimdFloat4 min = {_v.x < _b.x ? _v.x : _b.x, _v.y < _b.y ? _v.y : _b.y,
829 _v.z < _b.z ? _v.z : _b.z, _v.w < _b.w ? _v.w : _b.w};
830 const SimdFloat4 r = {
831 _a.x > min.x ? _a.x : min.x, _a.y > min.y ? _a.y : min.y,
832 _a.z > min.z ? _a.z : min.z, _a.w > min.w ? _a.w : min.w};
833 return r;
834}
835
836OZZ_INLINE SimdFloat4 Select(_SimdInt4 _b, _SimdFloat4 _true,
837 _SimdFloat4 _false) {
838 using internal::SimdFI4;
839 using internal::SimdIF4;
840
841 const SimdFI4 i_true = {_true};
842 const SimdFI4 i_false = {_false};
843 const SimdIF4 ret = {{i_false.i.x ^ (_b.x & (i_true.i.x ^ i_false.i.x)),
844 i_false.i.y ^ (_b.y & (i_true.i.y ^ i_false.i.y)),
845 i_false.i.z ^ (_b.z & (i_true.i.z ^ i_false.i.z)),
846 i_false.i.w ^ (_b.w & (i_true.i.w ^ i_false.i.w))}};
847 return ret.f;
848}
849
850OZZ_INLINE SimdInt4 CmpEq(_SimdFloat4 _a, _SimdFloat4 _b) {
851 const SimdInt4 ret = {
852 -static_cast<int>(_a.x == _b.x), -static_cast<int>(_a.y == _b.y),
853 -static_cast<int>(_a.z == _b.z), -static_cast<int>(_a.w == _b.w)};
854 return ret;
855}
856
857OZZ_INLINE SimdInt4 CmpNe(_SimdFloat4 _a, _SimdFloat4 _b) {
858 const SimdInt4 ret = {
859 -static_cast<int>(_a.x != _b.x), -static_cast<int>(_a.y != _b.y),
860 -static_cast<int>(_a.z != _b.z), -static_cast<int>(_a.w != _b.w)};
861 return ret;
862}
863
864OZZ_INLINE SimdInt4 CmpLt(_SimdFloat4 _a, _SimdFloat4 _b) {
865 const SimdInt4 ret = {
866 -static_cast<int>(_a.x < _b.x), -static_cast<int>(_a.y < _b.y),
867 -static_cast<int>(_a.z < _b.z), -static_cast<int>(_a.w < _b.w)};
868 return ret;
869}
870
871OZZ_INLINE SimdInt4 CmpLe(_SimdFloat4 _a, _SimdFloat4 _b) {
872 const SimdInt4 ret = {
873 -static_cast<int>(_a.x <= _b.x), -static_cast<int>(_a.y <= _b.y),
874 -static_cast<int>(_a.z <= _b.z), -static_cast<int>(_a.w <= _b.w)};
875 return ret;
876}
877
878OZZ_INLINE SimdInt4 CmpGt(_SimdFloat4 _a, _SimdFloat4 _b) {
879 const SimdInt4 ret = {
880 -static_cast<int>(_a.x > _b.x), -static_cast<int>(_a.y > _b.y),
881 -static_cast<int>(_a.z > _b.z), -static_cast<int>(_a.w > _b.w)};
882 return ret;
883}
884
885OZZ_INLINE SimdInt4 CmpGe(_SimdFloat4 _a, _SimdFloat4 _b) {
886 const SimdInt4 ret = {
887 -static_cast<int>(_a.x >= _b.x), -static_cast<int>(_a.y >= _b.y),
888 -static_cast<int>(_a.z >= _b.z), -static_cast<int>(_a.w >= _b.w)};
889 return ret;
890}
891
892OZZ_INLINE SimdFloat4 And(_SimdFloat4 _a, _SimdFloat4 _b) {
893 const internal::SimdFI4 a = {_a};
894 const internal::SimdFI4 b = {_b};
895 const internal::SimdIF4 ret = {
896 {a.i.x & b.i.x, a.i.y & b.i.y, a.i.z & b.i.z, a.i.w & b.i.w}};
897 return ret.f;
898}
899
900OZZ_INLINE SimdFloat4 Or(_SimdFloat4 _a, _SimdFloat4 _b) {
901 const internal::SimdFI4 a = {_a};
902 const internal::SimdFI4 b = {_b};
903 const internal::SimdIF4 ret = {
904 {a.i.x | b.i.x, a.i.y | b.i.y, a.i.z | b.i.z, a.i.w | b.i.w}};
905 return ret.f;
906}
907
908OZZ_INLINE SimdFloat4 Xor(_SimdFloat4 _a, _SimdFloat4 _b) {
909 const internal::SimdFI4 a = {_a};
910 const internal::SimdFI4 b = {_b};
911 const internal::SimdIF4 ret = {
912 {a.i.x ^ b.i.x, a.i.y ^ b.i.y, a.i.z ^ b.i.z, a.i.w ^ b.i.w}};
913 return ret.f;
914}
915
916OZZ_INLINE SimdFloat4 And(_SimdFloat4 _a, _SimdInt4 _b) {
917 const internal::SimdFI4 a = {_a};
918 const internal::SimdIF4 ret = {
919 {a.i.x & _b.x, a.i.y & _b.y, a.i.z & _b.z, a.i.w & _b.w}};
920 return ret.f;
921}
922
923OZZ_INLINE SimdFloat4 AndNot(_SimdFloat4 _a, _SimdInt4 _b) {
924 const internal::SimdFI4 a = {_a};
925 const internal::SimdIF4 ret = {
926 {a.i.x & ~_b.x, a.i.y & ~_b.y, a.i.z & ~_b.z, a.i.w & ~_b.w}};
927 return ret.f;
928}
929
930OZZ_INLINE SimdFloat4 Or(_SimdFloat4 _a, _SimdInt4 _b) {
931 const internal::SimdFI4 a = {_a};
932 const internal::SimdIF4 ret = {
933 {a.i.x | _b.x, a.i.y | _b.y, a.i.z | _b.z, a.i.w | _b.w}};
934 return ret.f;
935}
936
937OZZ_INLINE SimdFloat4 Xor(_SimdFloat4 _a, _SimdInt4 _b) {
938 const internal::SimdFI4 a = {_a};
939 const internal::SimdIF4 ret = {
940 {a.i.x ^ _b.x, a.i.y ^ _b.y, a.i.z ^ _b.z, a.i.w ^ _b.w}};
941 return ret.f;
942}
943
944OZZ_INLINE SimdFloat4 Cos(_SimdFloat4 _v) {
945 const SimdFloat4 ret = {std::cos(_v.x), std::cos(_v.y), std::cos(_v.z),
946 std::cos(_v.w)};
947 return ret;
948}
949
950OZZ_INLINE SimdFloat4 CosX(_SimdFloat4 _v) {
951 const SimdFloat4 ret = {std::cos(_v.x), _v.y, _v.z, _v.w};
952 return ret;
953}
954
955OZZ_INLINE SimdFloat4 ACos(_SimdFloat4 _v) {
956 const SimdFloat4 ret = {std::acos(_v.x), std::acos(_v.y), std::acos(_v.z),
957 std::acos(_v.w)};
958 return ret;
959}
960
961OZZ_INLINE SimdFloat4 ACosX(_SimdFloat4 _v) {
962 const SimdFloat4 ret = {std::acos(_v.x), _v.y, _v.z, _v.w};
963 return ret;
964}
965
966OZZ_INLINE SimdFloat4 Sin(_SimdFloat4 _v) {
967 const SimdFloat4 ret = {std::sin(_v.x), std::sin(_v.y), std::sin(_v.z),
968 std::sin(_v.w)};
969 return ret;
970}
971
972OZZ_INLINE SimdFloat4 SinX(_SimdFloat4 _v) {
973 const SimdFloat4 ret = {std::sin(_v.x), _v.y, _v.z, _v.w};
974 return ret;
975}
976
977OZZ_INLINE SimdFloat4 ASin(_SimdFloat4 _v) {
978 const SimdFloat4 ret = {std::asin(_v.x), std::asin(_v.y), std::asin(_v.z),
979 std::asin(_v.w)};
980 return ret;
981}
982
983OZZ_INLINE SimdFloat4 ASinX(_SimdFloat4 _v) {
984 const SimdFloat4 ret = {std::asin(_v.x), _v.y, _v.z, _v.w};
985 return ret;
986}
987
988OZZ_INLINE SimdFloat4 Tan(_SimdFloat4 _v) {
989 const SimdFloat4 ret = {std::tan(_v.x), std::tan(_v.y), std::tan(_v.z),
990 std::tan(_v.w)};
991 return ret;
992}
993
994OZZ_INLINE SimdFloat4 TanX(_SimdFloat4 _v) {
995 const SimdFloat4 ret = {std::tan(_v.x), _v.y, _v.z, _v.w};
996 return ret;
997}
998
999OZZ_INLINE SimdFloat4 ATan(_SimdFloat4 _v) {
1000 const SimdFloat4 ret = {std::atan(_v.x), std::atan(_v.y), std::atan(_v.z),
1001 std::atan(_v.w)};
1002 return ret;
1003}
1004
1005OZZ_INLINE SimdFloat4 ATanX(_SimdFloat4 _v) {
1006 const SimdFloat4 ret = {std::atan(_v.x), _v.y, _v.z, _v.w};
1007 return ret;
1008}
1009
1010namespace simd_int4 {
1011
1012OZZ_INLINE SimdInt4 zero() {
1013 const SimdInt4 ret = {0, 0, 0, 0};
1014 return ret;
1015}
1016
1017OZZ_INLINE SimdInt4 one() {
1018 const SimdInt4 ret = {1, 1, 1, 1};
1019 return ret;
1020}
1021
1022OZZ_INLINE SimdInt4 x_axis() {
1023 const SimdInt4 ret = {1, 0, 0, 0};
1024 return ret;
1025}
1026
1027OZZ_INLINE SimdInt4 y_axis() {
1028 const SimdInt4 ret = {0, 1, 0, 0};
1029 return ret;
1030}
1031
1032OZZ_INLINE SimdInt4 z_axis() {
1033 const SimdInt4 ret = {0, 0, 1, 0};
1034 return ret;
1035}
1036
1037OZZ_INLINE SimdInt4 w_axis() {
1038 const SimdInt4 ret = {0, 0, 0, 1};
1039 return ret;
1040}
1041
1042OZZ_INLINE SimdInt4 all_true() {
1043 const SimdInt4 ret = {~0, ~0, ~0, ~0};
1044 return ret;
1045}
1046
1047OZZ_INLINE SimdInt4 all_false() {
1048 const SimdInt4 ret = {0, 0, 0, 0};
1049 return ret;
1050}
1051
1052OZZ_INLINE SimdInt4 mask_sign() {
1053 const SimdInt4 ret = {
1054 static_cast<int>(0x80000000), static_cast<int>(0x80000000),
1055 static_cast<int>(0x80000000), static_cast<int>(0x80000000)};
1056 return ret;
1057}
1058
1059OZZ_INLINE SimdInt4 mask_sign_xyz() {
1060 const SimdInt4 ret = {
1061 static_cast<int>(0x80000000), static_cast<int>(0x80000000),
1062 static_cast<int>(0x80000000), static_cast<int>(0x00000000)};
1063 return ret;
1064}
1065
1066OZZ_INLINE SimdInt4 mask_sign_w() {
1067 const SimdInt4 ret = {
1068 static_cast<int>(0x00000000), static_cast<int>(0x00000000),
1069 static_cast<int>(0x00000000), static_cast<int>(0x80000000)};
1070 return ret;
1071}
1072
1073OZZ_INLINE SimdInt4 mask_not_sign() {
1074 const SimdInt4 ret = {
1075 static_cast<int>(0x7fffffff), static_cast<int>(0x7fffffff),
1076 static_cast<int>(0x7fffffff), static_cast<int>(0x7fffffff)};
1077 return ret;
1078}
1079
1080OZZ_INLINE SimdInt4 mask_ffff() {
1081 const SimdInt4 ret = {~0, ~0, ~0, ~0};
1082 return ret;
1083}
1084
1085OZZ_INLINE SimdInt4 mask_fff0() {
1086 const SimdInt4 ret = {~0, ~0, ~0, 0};
1087 return ret;
1088}
1089
1090OZZ_INLINE SimdInt4 mask_0000() {
1091 const SimdInt4 ret = {0, 0, 0, 0};
1092 return ret;
1093}
1094
1095OZZ_INLINE SimdInt4 mask_f000() {
1096 const SimdInt4 ret = {~0, 0, 0, 0};
1097 return ret;
1098}
1099
1100OZZ_INLINE SimdInt4 mask_0f00() {
1101 const SimdInt4 ret = {0, ~0, 0, 0};
1102 return ret;
1103}
1104
1105OZZ_INLINE SimdInt4 mask_00f0() {
1106 const SimdInt4 ret = {0, 0, ~0, 0};
1107 return ret;
1108}
1109
1110OZZ_INLINE SimdInt4 mask_000f() {
1111 const SimdInt4 ret = {0, 0, 0, ~0};
1112 return ret;
1113}
1114
1115OZZ_INLINE SimdInt4 Load(int _x, int _y, int _z, int _w) {
1116 const SimdInt4 ret = {_x, _y, _z, _w};
1117 return ret;
1118}
1119
1120OZZ_INLINE SimdInt4 LoadX(int _x) {
1121 const SimdInt4 ret = {_x, 0, 0, 0};
1122 return ret;
1123}
1124
1125OZZ_INLINE SimdInt4 Load1(int _x) {
1126 const SimdInt4 ret = {_x, _x, _x, _x};
1127 return ret;
1128}
1129
1130OZZ_INLINE SimdInt4 Load(bool _x, bool _y, bool _z, bool _w) {
1131 const SimdInt4 ret = {-static_cast<int>(_x), -static_cast<int>(_y),
1132 -static_cast<int>(_z), -static_cast<int>(_w)};
1133 return ret;
1134}
1135
1136OZZ_INLINE SimdInt4 LoadX(bool _x) {
1137 const SimdInt4 ret = {-static_cast<int>(_x), 0, 0, 0};
1138 return ret;
1139}
1140
1141OZZ_INLINE SimdInt4 Load1(bool _x) {
1142 const int i = -static_cast<int>(_x);
1143 const SimdInt4 ret = {i, i, i, i};
1144 return ret;
1145}
1146
1147OZZ_INLINE SimdInt4 LoadPtr(const int* _i) {
1148 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1149 const SimdInt4 ret = {_i[0], _i[1], _i[2], _i[3]};
1150 return ret;
1151}
1152
1153OZZ_INLINE SimdInt4 LoadXPtr(const int* _i) {
1154 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1155 const SimdInt4 ret = {*_i, 0, 0, 0};
1156 return ret;
1157}
1158
1159OZZ_INLINE SimdInt4 Load1Ptr(const int* _i) {
1160 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1161 const SimdInt4 ret = {*_i, *_i, *_i, *_i};
1162 return ret;
1163}
1164
1165OZZ_INLINE SimdInt4 Load2Ptr(const int* _i) {
1166 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1167 const SimdInt4 ret = {_i[0], _i[1], 0, 0};
1168 return ret;
1169}
1170
1171OZZ_INLINE SimdInt4 Load3Ptr(const int* _i) {
1172 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1173 const SimdInt4 ret = {_i[0], _i[1], _i[2], 0};
1174 return ret;
1175}
1176
1177OZZ_INLINE SimdInt4 LoadPtrU(const int* _i) {
1178 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1179 const SimdInt4 ret = {_i[0], _i[1], _i[2], _i[3]};
1180 return ret;
1181}
1182
1183OZZ_INLINE SimdInt4 LoadXPtrU(const int* _i) {
1184 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1185 const SimdInt4 ret = {*_i, 0, 0, 0};
1186 return ret;
1187}
1188
1189OZZ_INLINE SimdInt4 Load1PtrU(const int* _i) {
1190 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1191 const SimdInt4 ret = {*_i, *_i, *_i, *_i};
1192 return ret;
1193}
1194
1195OZZ_INLINE SimdInt4 Load2PtrU(const int* _i) {
1196 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1197 const SimdInt4 ret = {_i[0], _i[1], 0, 0};
1198 return ret;
1199}
1200
1201OZZ_INLINE SimdInt4 Load3PtrU(const int* _i) {
1202 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1203 const SimdInt4 ret = {_i[0], _i[1], _i[2], 0};
1204 return ret;
1205}
1206
1207OZZ_INLINE SimdInt4 FromFloatRound(_SimdFloat4 _f) {
1208 const SimdInt4 ret = {
1209 static_cast<int>(floor(_f.x + .5f)), static_cast<int>(floor(_f.y + .5f)),
1210 static_cast<int>(floor(_f.z + .5f)), static_cast<int>(floor(_f.w + .5f))};
1211 return ret;
1212}
1213
1214OZZ_INLINE SimdInt4 FromFloatTrunc(_SimdFloat4 _f) {
1215 const SimdInt4 ret = {static_cast<int>(_f.x), static_cast<int>(_f.y),
1216 static_cast<int>(_f.z), static_cast<int>(_f.w)};
1217 return ret;
1218}
1219} // namespace simd_int4
1220
1221OZZ_INLINE int GetX(_SimdInt4 _v) { return _v.x; }
1222
1223OZZ_INLINE int GetY(_SimdInt4 _v) { return _v.y; }
1224
1225OZZ_INLINE int GetZ(_SimdInt4 _v) { return _v.z; }
1226
1227OZZ_INLINE int GetW(_SimdInt4 _v) { return _v.w; }
1228
1229OZZ_INLINE SimdInt4 SetX(_SimdInt4 _v, _SimdInt4 _i) {
1230 const SimdInt4 ret = {_i.x, _v.y, _v.z, _v.w};
1231 return ret;
1232}
1233
1234OZZ_INLINE SimdInt4 SetY(_SimdInt4 _v, _SimdInt4 _i) {
1235 const SimdInt4 ret = {_v.x, _i.x, _v.z, _v.w};
1236 return ret;
1237}
1238
1239OZZ_INLINE SimdInt4 SetZ(_SimdInt4 _v, _SimdInt4 _i) {
1240 const SimdInt4 ret = {_v.x, _v.y, _i.x, _v.w};
1241 return ret;
1242}
1243
1244OZZ_INLINE SimdInt4 SetW(_SimdInt4 _v, _SimdInt4 _i) {
1245 const SimdInt4 ret = {_v.x, _v.y, _v.z, _i.x};
1246 return ret;
1247}
1248
1249OZZ_INLINE SimdInt4 SetI(_SimdInt4 _v, _SimdInt4 _i, int _ith) {
1250 assert(_ith >= 0 && _ith <= 3 && "Invalid index, out of range.");
1251 SimdInt4 ret = _v;
1252 (&ret.x)[_ith] = _i.x;
1253 return ret;
1254}
1255
1256OZZ_INLINE void StorePtr(_SimdInt4 _v, int* _i) {
1257 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1258 _i[0] = _v.x;
1259 _i[1] = _v.y;
1260 _i[2] = _v.z;
1261 _i[3] = _v.w;
1262}
1263
1264OZZ_INLINE void Store1Ptr(_SimdInt4 _v, int* _i) {
1265 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1266 _i[0] = _v.x;
1267}
1268
1269OZZ_INLINE void Store2Ptr(_SimdInt4 _v, int* _i) {
1270 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1271 _i[0] = _v.x;
1272 _i[1] = _v.y;
1273}
1274
1275OZZ_INLINE void Store3Ptr(_SimdInt4 _v, int* _i) {
1276 assert(!(uintptr_t(_i) & 0xf) && "Invalid alignment");
1277 _i[0] = _v.x;
1278 _i[1] = _v.y;
1279 _i[2] = _v.z;
1280}
1281
1282OZZ_INLINE void StorePtrU(_SimdInt4 _v, int* _i) {
1283 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1284 _i[0] = _v.x;
1285 _i[1] = _v.y;
1286 _i[2] = _v.z;
1287 _i[3] = _v.w;
1288}
1289
1290OZZ_INLINE void Store1PtrU(_SimdInt4 _v, int* _i) {
1291 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1292 _i[0] = _v.x;
1293}
1294
1295OZZ_INLINE void Store2PtrU(_SimdInt4 _v, int* _i) {
1296 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1297 _i[0] = _v.x;
1298 _i[1] = _v.y;
1299}
1300
1301OZZ_INLINE void Store3PtrU(_SimdInt4 _v, int* _i) {
1302 assert(!(uintptr_t(_i) & 0x3) && "Invalid alignment");
1303 _i[0] = _v.x;
1304 _i[1] = _v.y;
1305 _i[2] = _v.z;
1306}
1307
1308OZZ_INLINE SimdInt4 SplatX(_SimdInt4 _a) {
1309 const SimdInt4 ret = {_a.x, _a.x, _a.x, _a.x};
1310 return ret;
1311}
1312
1313OZZ_INLINE SimdInt4 SplatY(_SimdInt4 _a) {
1314 const SimdInt4 ret = {_a.y, _a.y, _a.y, _a.y};
1315 return ret;
1316}
1317
1318OZZ_INLINE SimdInt4 SplatZ(_SimdInt4 _a) {
1319 const SimdInt4 ret = {_a.z, _a.z, _a.z, _a.z};
1320 return ret;
1321}
1322
1323OZZ_INLINE SimdInt4 SplatW(_SimdInt4 _a) {
1324 const SimdInt4 ret = {_a.w, _a.w, _a.w, _a.w};
1325 return ret;
1326}
1327
1328template <size_t _X, size_t _Y, size_t _Z, size_t _W>
1329OZZ_INLINE SimdInt4 Swizzle(_SimdInt4 _v) {
1330 static_assert(_X <= 3 && _Y <= 3 && _Z <= 3 && _W <= 3,
1331 "Indices must be between 0 and 3");
1332 const int* pi = &_v.x;
1333 const SimdInt4 ret = {pi[_X], pi[_Y], pi[_Z], pi[_W]};
1334 return ret;
1335}
1336
1337OZZ_INLINE int MoveMask(_SimdInt4 _v) {
1338 return ((_v.x & 0x80000000) >> 31) | ((_v.y & 0x80000000) >> 30) |
1339 ((_v.z & 0x80000000) >> 29) | ((_v.w & 0x80000000) >> 28);
1340}
1341
1342OZZ_INLINE bool AreAllTrue(_SimdInt4 _v) {
1343 return _v.x != 0 && _v.y != 0 && _v.z != 0 && _v.w != 0;
1344}
1345
1346OZZ_INLINE bool AreAllTrue3(_SimdInt4 _v) {
1347 return _v.x != 0 && _v.y != 0 && _v.z != 0;
1348}
1349
1350OZZ_INLINE bool AreAllTrue2(_SimdInt4 _v) { return _v.x != 0 && _v.y != 0; }
1351
1352OZZ_INLINE bool AreAllTrue1(_SimdInt4 _v) { return _v.x != 0; }
1353
1354OZZ_INLINE bool AreAllFalse(_SimdInt4 _v) {
1355 return _v.x == 0 && _v.y == 0 && _v.z == 0 && _v.w == 0;
1356}
1357
1358OZZ_INLINE bool AreAllFalse3(_SimdInt4 _v) {
1359 return _v.x == 0 && _v.y == 0 && _v.z == 0;
1360}
1361
1362OZZ_INLINE bool AreAllFalse2(_SimdInt4 _v) { return _v.x == 0 && _v.y == 0; }
1363
1364OZZ_INLINE bool AreAllFalse1(_SimdInt4 _v) { return _v.x == 0; }
1365
1366OZZ_INLINE SimdInt4 MAdd(_SimdInt4 _a, _SimdInt4 _b, _SimdInt4 _addend) {
1367 const SimdInt4 ret = {_a.x * _b.x + _addend.x, _a.y * _b.y + _addend.y,
1368 _a.z * _b.z + _addend.z, _a.w * _b.w + _addend.w};
1369 return ret;
1370}
1371
1372OZZ_INLINE SimdInt4 DivX(_SimdInt4 _a, _SimdInt4 _b) {
1373 const SimdInt4 ret = {_a.x / _b.x, _a.y, _a.z, _a.w};
1374 return ret;
1375}
1376
1377OZZ_INLINE SimdInt4 HAdd2(_SimdInt4 _v) {
1378 const SimdInt4 ret = {_v.x + _v.y, _v.y, _v.z, _v.w};
1379 return ret;
1380}
1381
1382OZZ_INLINE SimdInt4 HAdd3(_SimdInt4 _v) {
1383 const SimdInt4 ret = {_v.x + _v.y + _v.z, _v.y, _v.z, _v.w};
1384 return ret;
1385}
1386
1387OZZ_INLINE SimdInt4 HAdd4(_SimdInt4 _v) {
1388 const SimdInt4 ret = {_v.x + _v.y + _v.z + _v.w, _v.y, _v.z, _v.w};
1389 return ret;
1390}
1391
1392OZZ_INLINE SimdInt4 Dot2(_SimdInt4 _a, _SimdInt4 _b) {
1393 const SimdInt4 ret = {_a.x * _b.x + _a.y * _b.y, _a.y, _a.z, _a.w};
1394 return ret;
1395}
1396
1397OZZ_INLINE SimdInt4 Dot3(_SimdInt4 _a, _SimdInt4 _b) {
1398 const SimdInt4 ret = {_a.x * _b.x + _a.y * _b.y + _a.z * _b.z, _a.y, _a.z,
1399 _a.w};
1400 return ret;
1401}
1402
1403OZZ_INLINE SimdInt4 Dot4(_SimdInt4 _a, _SimdInt4 _b) {
1404 const SimdInt4 ret = {_a.x * _b.x + _a.y * _b.y + _a.z * _b.z + _a.w * _b.w,
1405 _a.y, _a.z, _a.w};
1406 return ret;
1407}
1408OZZ_INLINE SimdInt4 Abs(_SimdInt4 _v) {
1409 const SimdInt4 mash = {_v.x >> 31, _v.y >> 31, _v.z >> 31, _v.w >> 31};
1410 const SimdInt4 ret = {
1411 (_v.x + (mash.x)) ^ (mash.x), (_v.y + (mash.y)) ^ (mash.y),
1412 (_v.z + (mash.z)) ^ (mash.z), (_v.w + (mash.w)) ^ (mash.w)};
1413 return ret;
1414}
1415
1416OZZ_INLINE SimdInt4 Sign(_SimdInt4 _v) {
1417 const SimdInt4 ret = {
1418 _v.x & static_cast<int>(0x80000000), _v.y & static_cast<int>(0x80000000),
1419 _v.z & static_cast<int>(0x80000000), _v.w & static_cast<int>(0x80000000)};
1420 return ret;
1421}
1422
1423OZZ_INLINE SimdInt4 Min(_SimdInt4 _a, _SimdInt4 _b) {
1424 const SimdInt4 ret = {_a.x < _b.x ? _a.x : _b.x, _a.y < _b.y ? _a.y : _b.y,
1425 _a.z < _b.z ? _a.z : _b.z, _a.w < _b.w ? _a.w : _b.w};
1426 return ret;
1427}
1428
1429OZZ_INLINE SimdInt4 Max(_SimdInt4 _a, _SimdInt4 _b) {
1430 const SimdInt4 ret = {_a.x > _b.x ? _a.x : _b.x, _a.y > _b.y ? _a.y : _b.y,
1431 _a.z > _b.z ? _a.z : _b.z, _a.w > _b.w ? _a.w : _b.w};
1432 return ret;
1433}
1434
1435OZZ_INLINE SimdInt4 Min0(_SimdInt4 _v) {
1436 const SimdInt4 ret = {_v.x < 0 ? _v.x : 0, _v.y < 0 ? _v.y : 0,
1437 _v.z < 0 ? _v.z : 0, _v.w < 0 ? _v.w : 0};
1438 return ret;
1439}
1440
1441OZZ_INLINE SimdInt4 Max0(_SimdInt4 _v) {
1442 const SimdInt4 ret = {_v.x > 0 ? _v.x : 0, _v.y > 0 ? _v.y : 0,
1443 _v.z > 0 ? _v.z : 0, _v.w > 0 ? _v.w : 0};
1444 return ret;
1445}
1446
1447OZZ_INLINE SimdInt4 Clamp(_SimdInt4 _a, _SimdInt4 _v, _SimdInt4 _b) {
1448 const SimdInt4 min = {_v.x < _b.x ? _v.x : _b.x, _v.y < _b.y ? _v.y : _b.y,
1449 _v.z < _b.z ? _v.z : _b.z, _v.w < _b.w ? _v.w : _b.w};
1450 const SimdInt4 r = {_a.x > min.x ? _a.x : min.x, _a.y > min.y ? _a.y : min.y,
1451 _a.z > min.z ? _a.z : min.z, _a.w > min.w ? _a.w : min.w};
1452 return r;
1453}
1454
1455OZZ_INLINE SimdInt4 Select(_SimdInt4 _b, _SimdInt4 _true, _SimdInt4 _false) {
1456 const SimdInt4 ret = {_false.x ^ (_b.x & (_true.x ^ _false.x)),
1457 _false.y ^ (_b.y & (_true.y ^ _false.y)),
1458 _false.z ^ (_b.z & (_true.z ^ _false.z)),
1459 _false.w ^ (_b.w & (_true.w ^ _false.w))};
1460 return ret;
1461}
1462
1463OZZ_INLINE SimdInt4 And(_SimdInt4 _a, _SimdInt4 _b) {
1464 const SimdInt4 ret = {_a.x & _b.x, _a.y & _b.y, _a.z & _b.z, _a.w & _b.w};
1465 return ret;
1466}
1467
1468OZZ_INLINE SimdInt4 AndNot(_SimdInt4 _a, _SimdInt4 _b) {
1469 const SimdInt4 ret = {_a.x & ~_b.x, _a.y & ~_b.y, _a.z & ~_b.z, _a.w & ~_b.w};
1470 return ret;
1471}
1472
1473OZZ_INLINE SimdInt4 Or(_SimdInt4 _a, _SimdInt4 _b) {
1474 const SimdInt4 ret = {_a.x | _b.x, _a.y | _b.y, _a.z | _b.z, _a.w | _b.w};
1475 return ret;
1476}
1477
1478OZZ_INLINE SimdInt4 Xor(_SimdInt4 _a, _SimdInt4 _b) {
1479 const SimdInt4 ret = {_a.x ^ _b.x, _a.y ^ _b.y, _a.z ^ _b.z, _a.w ^ _b.w};
1480 return ret;
1481}
1482
1483OZZ_INLINE SimdInt4 Not(_SimdInt4 _v) {
1484 const SimdInt4 ret = {~_v.x, ~_v.y, ~_v.z, ~_v.w};
1485 return ret;
1486}
1487
1488OZZ_INLINE SimdInt4 ShiftL(_SimdInt4 _v, int _bits) {
1489 const SimdInt4 ret = {_v.x << _bits, _v.y << _bits, _v.z << _bits,
1490 _v.w << _bits};
1491 return ret;
1492}
1493
1494OZZ_INLINE SimdInt4 ShiftR(_SimdInt4 _v, int _bits) {
1495 const SimdInt4 ret = {_v.x >> _bits, _v.y >> _bits, _v.z >> _bits,
1496 _v.w >> _bits};
1497 return ret;
1498}
1499
1500OZZ_INLINE SimdInt4 ShiftRu(_SimdInt4 _v, int _bits) {
1501 const union IU {
1502 int i[4];
1503 unsigned int u[4];
1504 } iu = {{_v.x, _v.y, _v.z, _v.w}};
1505 const union UI {
1506 unsigned int u[4];
1507 int i[4];
1508 } ui = {
1509 {iu.u[0] >> _bits, iu.u[1] >> _bits, iu.u[2] >> _bits, iu.u[3] >> _bits}};
1510 const SimdInt4 ret = {ui.i[0], ui.i[1], ui.i[2], ui.i[3]};
1511 return ret;
1512}
1513
1514OZZ_INLINE SimdInt4 CmpEq(_SimdInt4 _a, _SimdInt4 _b) {
1515 const SimdInt4 ret = {
1516 -static_cast<int>(_a.x == _b.x), -static_cast<int>(_a.y == _b.y),
1517 -static_cast<int>(_a.z == _b.z), -static_cast<int>(_a.w == _b.w)};
1518 return ret;
1519}
1520
1521OZZ_INLINE SimdInt4 CmpNe(_SimdInt4 _a, _SimdInt4 _b) {
1522 const SimdInt4 ret = {
1523 -static_cast<int>(_a.x != _b.x), -static_cast<int>(_a.y != _b.y),
1524 -static_cast<int>(_a.z != _b.z), -static_cast<int>(_a.w != _b.w)};
1525 return ret;
1526}
1527
1528OZZ_INLINE SimdInt4 CmpLt(_SimdInt4 _a, _SimdInt4 _b) {
1529 const SimdInt4 ret = {
1530 -static_cast<int>(_a.x < _b.x), -static_cast<int>(_a.y < _b.y),
1531 -static_cast<int>(_a.z < _b.z), -static_cast<int>(_a.w < _b.w)};
1532 return ret;
1533}
1534
1535OZZ_INLINE SimdInt4 CmpLe(_SimdInt4 _a, _SimdInt4 _b) {
1536 const SimdInt4 ret = {
1537 -static_cast<int>(_a.x <= _b.x), -static_cast<int>(_a.y <= _b.y),
1538 -static_cast<int>(_a.z <= _b.z), -static_cast<int>(_a.w <= _b.w)};
1539 return ret;
1540}
1541
1542OZZ_INLINE SimdInt4 CmpGt(_SimdInt4 _a, _SimdInt4 _b) {
1543 const SimdInt4 ret = {
1544 -static_cast<int>(_a.x > _b.x), -static_cast<int>(_a.y > _b.y),
1545 -static_cast<int>(_a.z > _b.z), -static_cast<int>(_a.w > _b.w)};
1546 return ret;
1547}
1548
1549OZZ_INLINE SimdInt4 CmpGe(_SimdInt4 _a, _SimdInt4 _b) {
1550 const SimdInt4 ret = {
1551 -static_cast<int>(_a.x >= _b.x), -static_cast<int>(_a.y >= _b.y),
1552 -static_cast<int>(_a.z >= _b.z), -static_cast<int>(_a.w >= _b.w)};
1553 return ret;
1554}
1555
1556OZZ_INLINE Float4x4 Float4x4::identity() {
1557 const Float4x4 ret = {{{1.f, 0.f, 0.f, 0.f},
1558 {0.f, 1.f, 0.f, 0.f},
1559 {0.f, 0.f, 1.f, 0.f},
1560 {0.f, 0.f, 0.f, 1.f}}};
1561 return ret;
1562}
1563
1564OZZ_INLINE Float4x4 Transpose(const Float4x4& _m) {
1565 const Float4x4 ret = {
1566 {{_m.cols[0].x, _m.cols[1].x, _m.cols[2].x, _m.cols[3].x},
1567 {_m.cols[0].y, _m.cols[1].y, _m.cols[2].y, _m.cols[3].y},
1568 {_m.cols[0].z, _m.cols[1].z, _m.cols[2].z, _m.cols[3].z},
1569 {_m.cols[0].w, _m.cols[1].w, _m.cols[2].w, _m.cols[3].w}}};
1570 return ret;
1571}
1572
1573OZZ_INLINE Float4x4 Invert(const Float4x4& _m, SimdInt4* _invertible) {
1574 const SimdFloat4* cols = _m.cols;
1575 const float a00 = cols[2].z * cols[3].w - cols[3].z * cols[2].w;
1576 const float a01 = cols[2].y * cols[3].w - cols[3].y * cols[2].w;
1577 const float a02 = cols[2].y * cols[3].z - cols[3].y * cols[2].z;
1578 const float a03 = cols[2].x * cols[3].w - cols[3].x * cols[2].w;
1579 const float a04 = cols[2].x * cols[3].z - cols[3].x * cols[2].z;
1580 const float a05 = cols[2].x * cols[3].y - cols[3].x * cols[2].y;
1581 const float a06 = cols[1].z * cols[3].w - cols[3].z * cols[1].w;
1582 const float a07 = cols[1].y * cols[3].w - cols[3].y * cols[1].w;
1583 const float a08 = cols[1].y * cols[3].z - cols[3].y * cols[1].z;
1584 const float a09 = cols[1].x * cols[3].w - cols[3].x * cols[1].w;
1585 const float a10 = cols[1].x * cols[3].z - cols[3].x * cols[1].z;
1586 const float a11 = cols[1].y * cols[3].w - cols[3].y * cols[1].w;
1587 const float a12 = cols[1].x * cols[3].y - cols[3].x * cols[1].y;
1588 const float a13 = cols[1].z * cols[2].w - cols[2].z * cols[1].w;
1589 const float a14 = cols[1].y * cols[2].w - cols[2].y * cols[1].w;
1590 const float a15 = cols[1].y * cols[2].z - cols[2].y * cols[1].z;
1591 const float a16 = cols[1].x * cols[2].w - cols[2].x * cols[1].w;
1592 const float a17 = cols[1].x * cols[2].z - cols[2].x * cols[1].z;
1593 const float a18 = cols[1].x * cols[2].y - cols[2].x * cols[1].y;
1594
1595 const float b0x = cols[1].y * a00 - cols[1].z * a01 + cols[1].w * a02;
1596 const float b1x = -cols[1].x * a00 + cols[1].z * a03 - cols[1].w * a04;
1597 const float b2x = cols[1].x * a01 - cols[1].y * a03 + cols[1].w * a05;
1598 const float b3x = -cols[1].x * a02 + cols[1].y * a04 - cols[1].z * a05;
1599
1600 const float b0y = -cols[0].y * a00 + cols[0].z * a01 - cols[0].w * a02;
1601 const float b1y = cols[0].x * a00 - cols[0].z * a03 + cols[0].w * a04;
1602 const float b2y = -cols[0].x * a01 + cols[0].y * a03 - cols[0].w * a05;
1603 const float b3y = cols[0].x * a02 - cols[0].y * a04 + cols[0].z * a05;
1604
1605 const float b0z = cols[0].y * a06 - cols[0].z * a07 + cols[0].w * a08;
1606 const float b1z = -cols[0].x * a06 + cols[0].z * a09 - cols[0].w * a10;
1607 const float b2z = cols[0].x * a11 - cols[0].y * a09 + cols[0].w * a12;
1608 const float b3z = -cols[0].x * a08 + cols[0].y * a10 - cols[0].z * a12;
1609
1610 const float b0w = -cols[0].y * a13 + cols[0].z * a14 - cols[0].w * a15;
1611 const float b1w = cols[0].x * a13 - cols[0].z * a16 + cols[0].w * a17;
1612 const float b2w = -cols[0].x * a14 + cols[0].y * a16 - cols[0].w * a18;
1613 const float b3w = cols[0].x * a15 - cols[0].y * a17 + cols[0].z * a18;
1614
1615 const float det =
1616 cols[0].x * b0x + cols[0].y * b1x + cols[0].z * b2x + cols[0].w * b3x;
1617 const bool invertible = det != 0.f;
1618 assert((_invertible || invertible) && "Matrix is not invertible");
1619 if (_invertible != nullptr) {
1620 *_invertible = simd_int4::LoadX(invertible);
1621 }
1622 const float inv_det = invertible ? 1.f / det : 0.f;
1623
1624 const Float4x4 ret = {
1625 {{b0x * inv_det, b0y * inv_det, b0z * inv_det, b0w * inv_det},
1626 {b1x * inv_det, b1y * inv_det, b1z * inv_det, b1w * inv_det},
1627 {b2x * inv_det, b2y * inv_det, b2z * inv_det, b2w * inv_det},
1628 {b3x * inv_det, b3y * inv_det, b3z * inv_det, b3w * inv_det}}};
1629 return ret;
1630}
1631
1632Float4x4 Float4x4::Scaling(_SimdFloat4 _v) {
1633 const Float4x4 ret = {{{_v.x, 0.f, 0.f, 0.f},
1634 {0.f, _v.y, 0.f, 0.f},
1635 {0.f, 0.f, _v.z, 0.f},
1636 {0.f, 0.f, 0.f, 1.f}}};
1637 return ret;
1638}
1639
1640Float4x4 Float4x4::Translation(_SimdFloat4 _v) {
1641 const Float4x4 ret = {{{1.f, 0.f, 0.f, 0.f},
1642 {0.f, 1.f, 0.f, 0.f},
1643 {0.f, 0.f, 1.f, 0.f},
1644 {_v.x, _v.y, _v.z, 1.f}}};
1645 return ret;
1646}
1647
1648OZZ_INLINE Float4x4 Translate(const Float4x4& _m, _SimdFloat4 _v) {
1649 const Float4x4 ret = {{_m.cols[0],
1650 _m.cols[1],
1651 _m.cols[2],
1652 {_m.cols[0].x * _v.x + _m.cols[1].x * _v.y +
1653 _m.cols[2].x * _v.z + _m.cols[3].x,
1654 _m.cols[0].y * _v.x + _m.cols[1].y * _v.y +
1655 _m.cols[2].y * _v.z + _m.cols[3].y,
1656 _m.cols[0].z * _v.x + _m.cols[1].z * _v.y +
1657 _m.cols[2].z * _v.z + _m.cols[3].z,
1658 _m.cols[0].w * _v.x + _m.cols[1].w * _v.y +
1659 _m.cols[2].w * _v.z + _m.cols[3].w}}};
1660 return ret;
1661}
1662
1663OZZ_INLINE Float4x4 Scale(const Float4x4& _m, _SimdFloat4 _v) {
1664 const Float4x4 ret = {{{_m.cols[0].x * _v.x, _m.cols[0].y * _v.x,
1665 _m.cols[0].z * _v.x, _m.cols[0].w * _v.x},
1666 {_m.cols[1].x * _v.y, _m.cols[1].y * _v.y,
1667 _m.cols[1].z * _v.y, _m.cols[1].w * _v.y},
1668 {_m.cols[2].x * _v.z, _m.cols[2].y * _v.z,
1669 _m.cols[2].z * _v.z, _m.cols[2].w * _v.z},
1670 _m.cols[3]}};
1671 return ret;
1672}
1673
1674OZZ_INLINE Float4x4 ColumnMultiply(const Float4x4& _m, _SimdFloat4 _v) {
1675 const Float4x4 ret = {{{_m.cols[0].x * _v.x, _m.cols[0].y * _v.y,
1676 _m.cols[0].z * _v.z, _m.cols[0].w * _v.w},
1677 {_m.cols[1].x * _v.x, _m.cols[1].y * _v.y,
1678 _m.cols[1].z * _v.z, _m.cols[1].w * _v.w},
1679 {_m.cols[2].x * _v.x, _m.cols[2].y * _v.y,
1680 _m.cols[2].z * _v.z, _m.cols[2].w * _v.w},
1681 {_m.cols[3].x * _v.x, _m.cols[3].y * _v.y,
1682 _m.cols[3].z * _v.z, _m.cols[3].w * _v.w}}};
1683 return ret;
1684}
1685
1686OZZ_INLINE SimdInt4 IsNormalized(const Float4x4& _m) {
1687 const SimdInt4 ret = {IsNormalized3(_m.cols[0]).x,
1688 IsNormalized3(_m.cols[1]).x,
1689 IsNormalized3(_m.cols[2]).x, 0};
1690 return ret;
1691}
1692
1693OZZ_INLINE SimdInt4 IsNormalizedEst(const Float4x4& _m) {
1694 const SimdInt4 ret = {IsNormalizedEst3(_m.cols[0]).x,
1695 IsNormalizedEst3(_m.cols[1]).x,
1696 IsNormalizedEst3(_m.cols[2]).x, 0};
1697 return ret;
1698}
1699
1700OZZ_INLINE SimdInt4 IsOrthogonal(const Float4x4& _m) {
1701 // Use simd_float4::zero() if one of the normalization fails. _m will then be
1702 // considered not orthogonal.
1703 const SimdFloat4 cross =
1704 NormalizeSafe3(Cross3(_m.cols[0], _m.cols[1]), simd_float4::zero());
1705 const SimdFloat4 at = NormalizeSafe3(_m.cols[2], simd_float4::zero());
1706
1707 const float sq_len = cross.x * at.x + cross.y * at.y + cross.z * at.z;
1708 const bool same = std::abs(sq_len - 1.f) < kNormalizationToleranceSq;
1709 const SimdInt4 ret = {-static_cast<int>(same), 0, 0, 0};
1710 return ret;
1711}
1712
1713OZZ_INLINE SimdFloat4 ToQuaternion(const Float4x4& _m) {
1714 assert(AreAllTrue3(IsNormalized(_m)));
1715 assert(AreAllTrue1(IsOrthogonal(_m)));
1716 // Cf From Quaternion to Matrix and Back, J.M.P. van Waveren 2005.
1717 SimdFloat4 ret;
1718 if (_m.cols[0].x + _m.cols[1].y + _m.cols[2].z > .0f) {
1719 const float t = _m.cols[0].x + _m.cols[1].y + _m.cols[2].z + 1.0f;
1720 const float s = (1.f / std::sqrt(t)) * .5f;
1721 ret.x = (_m.cols[1].z - _m.cols[2].y) * s;
1722 ret.y = (_m.cols[2].x - _m.cols[0].z) * s;
1723 ret.z = (_m.cols[0].y - _m.cols[1].x) * s;
1724 ret.w = s * t;
1725 } else if (_m.cols[0].x > _m.cols[1].y && _m.cols[0].x > _m.cols[2].z) {
1726 const float t = _m.cols[0].x - _m.cols[1].y - _m.cols[2].z + 1.0f;
1727 const float s = (1.f / std::sqrt(t)) * .5f;
1728 ret.x = s * t;
1729 ret.y = (_m.cols[0].y + _m.cols[1].x) * s;
1730 ret.z = (_m.cols[2].x + _m.cols[0].z) * s;
1731 ret.w = (_m.cols[1].z - _m.cols[2].y) * s;
1732 } else if (_m.cols[1].y > _m.cols[2].z) {
1733 const float t = -_m.cols[0].x + _m.cols[1].y - _m.cols[2].z + 1.0f;
1734 const float s = (1.f / std::sqrt(t)) * .5f;
1735 ret.x = (_m.cols[0].y + _m.cols[1].x) * s;
1736 ret.y = s * t;
1737 ret.z = (_m.cols[1].z + _m.cols[2].y) * s;
1738 ret.w = (_m.cols[2].x - _m.cols[0].z) * s;
1739 } else {
1740 const float t = -_m.cols[0].x - _m.cols[1].y + _m.cols[2].z + 1.0f;
1741 const float s = (1.f / std::sqrt(t)) * .5f;
1742 ret.x = (_m.cols[2].x + _m.cols[0].z) * s;
1743 ret.y = (_m.cols[1].z + _m.cols[2].y) * s;
1744 ret.z = s * t;
1745 ret.w = (_m.cols[0].y - _m.cols[1].x) * s;
1746 }
1747 assert(AreAllTrue1(IsNormalizedEst4(ret)));
1748 return ret;
1749}
1750
1751OZZ_INLINE bool ToAffine(const Float4x4& _m, SimdFloat4* _translation,
1752 SimdFloat4* _quaternion, SimdFloat4* _scale) {
1753 _translation->x = _m.cols[3].x;
1754 _translation->y = _m.cols[3].y;
1755 _translation->z = _m.cols[3].z;
1756 _translation->w = 1.f;
1757
1758 // Extracts scale.
1759 const float sq_scale_x = Length3Sqr(_m.cols[0]).x;
1760 const float scale_x = std::sqrt(sq_scale_x);
1761 const float sq_scale_y = Length3Sqr(_m.cols[1]).x;
1762 const float scale_y = std::sqrt(sq_scale_y);
1763 const float sq_scale_z = Length3Sqr(_m.cols[2]).x;
1764 const float scale_z = std::sqrt(sq_scale_z);
1765
1766 // Builds an orthonormal matrix in order to support quaternion extraction.
1767 const bool x_zero = std::abs(sq_scale_x) < kOrthogonalisationToleranceSq;
1768 const bool y_zero = std::abs(sq_scale_y) < kOrthogonalisationToleranceSq;
1769 const bool z_zero = std::abs(sq_scale_z) < kOrthogonalisationToleranceSq;
1770
1771 Float4x4 orthonormal;
1772 if (x_zero) {
1773 if (y_zero || z_zero) {
1774 return false;
1775 }
1776 orthonormal.cols[1].x = _m.cols[1].x / scale_y;
1777 orthonormal.cols[1].y = _m.cols[1].y / scale_y;
1778 orthonormal.cols[1].z = _m.cols[1].z / scale_y;
1779 orthonormal.cols[1].w = 0.f;
1780 orthonormal.cols[0] = Normalize3(Cross3(orthonormal.cols[1], _m.cols[2]));
1781 orthonormal.cols[2] =
1782 Normalize3(Cross3(orthonormal.cols[0], orthonormal.cols[1]));
1783 } else if (z_zero) {
1784 if (x_zero || y_zero) {
1785 return false;
1786 }
1787 orthonormal.cols[0].x = _m.cols[0].x / scale_x;
1788 orthonormal.cols[0].y = _m.cols[0].y / scale_x;
1789 orthonormal.cols[0].z = _m.cols[0].z / scale_x;
1790 orthonormal.cols[0].w = 0.f;
1791 orthonormal.cols[2] = Normalize3(Cross3(orthonormal.cols[0], _m.cols[1]));
1792 orthonormal.cols[1] =
1793 Normalize3(Cross3(orthonormal.cols[2], orthonormal.cols[0]));
1794 } else { // Favor z axis in the default case
1795 if (x_zero || z_zero) {
1796 return false;
1797 }
1798 orthonormal.cols[2].x = _m.cols[2].x / scale_z;
1799 orthonormal.cols[2].y = _m.cols[2].y / scale_z;
1800 orthonormal.cols[2].z = _m.cols[2].z / scale_z;
1801 orthonormal.cols[2].w = 0.f;
1802 orthonormal.cols[1] = Normalize3(Cross3(orthonormal.cols[2], _m.cols[0]));
1803 orthonormal.cols[0] =
1804 Normalize3(Cross3(orthonormal.cols[1], orthonormal.cols[2]));
1805 }
1806
1807 // orthonormal.cols[3] = simd_float4::w_axis(); Not used by ToQuaternion.
1808
1809 // Get back scale signs in case of reflexions
1810 _scale->x =
1811 Dot3(orthonormal.cols[0], _m.cols[0]).x > 0.f ? scale_x : -scale_x;
1812 _scale->y =
1813 Dot3(orthonormal.cols[1], _m.cols[1]).x > 0.f ? scale_y : -scale_y;
1814 _scale->z =
1815 Dot3(orthonormal.cols[2], _m.cols[2]).x > 0.f ? scale_z : -scale_z;
1816 _scale->w = 1.f;
1817
1818 // Extracts quaternion.
1819 *_quaternion = ToQuaternion(orthonormal);
1820 return true;
1821}
1822
1823OZZ_INLINE Float4x4 Float4x4::FromEuler(_SimdFloat4 _v) {
1824 return Float4x4::FromAxisAngle(simd_float4::y_axis(), SplatX(_v)) *
1825 Float4x4::FromAxisAngle(simd_float4::x_axis(), SplatY(_v)) *
1826 Float4x4::FromAxisAngle(simd_float4::z_axis(), SplatZ(_v));
1827}
1828
1829OZZ_INLINE Float4x4 Float4x4::FromAxisAngle(_SimdFloat4 _axis,
1830 _SimdFloat4 _angle) {
1831 assert(AreAllTrue1(IsNormalizedEst3(_axis)));
1832
1833 const float cos = std::cos(_angle.x);
1834 const float sin = std::sin(_angle.x);
1835 const float t = 1.f - cos;
1836
1837 const float a = _axis.x * _axis.y * t;
1838 const float b = _axis.z * sin;
1839 const float c = _axis.x * _axis.z * t;
1840 const float d = _axis.y * sin;
1841 const float e = _axis.y * _axis.z * t;
1842 const float f = _axis.x * sin;
1843
1844 const Float4x4 ret = {{{cos + _axis.x * _axis.x * t, a + b, c - d, 0.f},
1845 {a - b, cos + _axis.y * _axis.y * t, e + f, 0.f},
1846 {c + d, e - f, cos + _axis.z * _axis.z * t, 0.f},
1847 {0.f, 0.f, 0.f, 1.f}}};
1848 return ret;
1849}
1850
1851OZZ_INLINE Float4x4 Float4x4::FromQuaternion(_SimdFloat4 _v) {
1852 assert(AreAllTrue1(IsNormalizedEst4(_v)));
1853
1854 const float xx = _v.x * _v.x;
1855 const float xy = _v.x * _v.y;
1856 const float xz = _v.x * _v.z;
1857 const float xw = _v.x * _v.w;
1858 const float yy = _v.y * _v.y;
1859 const float yz = _v.y * _v.z;
1860 const float yw = _v.y * _v.w;
1861 const float zz = _v.z * _v.z;
1862 const float zw = _v.z * _v.w;
1863
1864 const Float4x4 ret = {
1865 {{1.f - 2.f * (yy + zz), 2.f * (xy + zw), 2.f * (xz - yw), 0.f},
1866 {2.f * (xy - zw), 1.f - 2.f * (xx + zz), 2.f * (yz + xw), 0.f},
1867 {2.f * (xz + yw), 2.f * (yz - xw), 1.f - 2.f * (xx + yy), 0.f},
1868 {0.f, 0.f, 0.f, 1.f}}};
1869 return ret;
1870}
1871
1872OZZ_INLINE Float4x4 Float4x4::FromAffine(_SimdFloat4 _translation,
1873 _SimdFloat4 _quaternion,
1874 _SimdFloat4 _scale) {
1875 assert(AreAllTrue1(IsNormalizedEst4(_quaternion)));
1876
1877 const float xx = _quaternion.x * _quaternion.x;
1878 const float xy = _quaternion.x * _quaternion.y;
1879 const float xz = _quaternion.x * _quaternion.z;
1880 const float xw = _quaternion.x * _quaternion.w;
1881 const float yy = _quaternion.y * _quaternion.y;
1882 const float yz = _quaternion.y * _quaternion.z;
1883 const float yw = _quaternion.y * _quaternion.w;
1884 const float zz = _quaternion.z * _quaternion.z;
1885 const float zw = _quaternion.z * _quaternion.w;
1886
1887 const Float4x4 ret = {
1888 {{_scale.x * (1.f - 2.f * (yy + zz)), _scale.x * 2.f * (xy + zw),
1889 _scale.x * 2.f * (xz - yw), 0.f},
1890 {_scale.y * 2.f * (xy - zw), _scale.y * (1.f - 2.f * (xx + zz)),
1891 _scale.y * (2.f * (yz + xw)), 0.f},
1892 {_scale.z * 2.f * (xz + yw), _scale.z * 2.f * (yz - xw),
1893 _scale.z * (1.f - 2.f * (xx + yy)), 0.f},
1894 {_translation.x, _translation.y, _translation.z, 1.f}}};
1895 return ret;
1896}
1897
1898OZZ_INLINE ozz::math::SimdFloat4 TransformPoint(const ozz::math::Float4x4& _m,
1900 const ozz::math::SimdFloat4 ret = {_m.cols[0].x * _v.x + _m.cols[1].x * _v.y +
1901 _m.cols[2].x * _v.z + _m.cols[3].x,
1902 _m.cols[0].y * _v.x + _m.cols[1].y * _v.y +
1903 _m.cols[2].y * _v.z + _m.cols[3].y,
1904 _m.cols[0].z * _v.x + _m.cols[1].z * _v.y +
1905 _m.cols[2].z * _v.z + _m.cols[3].z,
1906 _m.cols[0].w * _v.x + _m.cols[1].w * _v.y +
1907 _m.cols[2].w * _v.z + _m.cols[3].w};
1908 return ret;
1909}
1910
1911OZZ_INLINE ozz::math::SimdFloat4 TransformVector(const ozz::math::Float4x4& _m,
1913 const ozz::math::SimdFloat4 ret = {
1914 _m.cols[0].x * _v.x + _m.cols[1].x * _v.y + _m.cols[2].x * _v.z,
1915 _m.cols[0].y * _v.x + _m.cols[1].y * _v.y + _m.cols[2].y * _v.z,
1916 _m.cols[0].z * _v.x + _m.cols[1].z * _v.y + _m.cols[2].z * _v.z,
1917 _m.cols[0].w * _v.x + _m.cols[1].w * _v.y + _m.cols[2].w * _v.z};
1918 return ret;
1919}
1920
1921OZZ_INLINE ozz::math::SimdFloat4 operator*(const ozz::math::Float4x4& _m,
1923 const ozz::math::SimdFloat4 ret = {
1924 _m.cols[0].x * _v.x + _m.cols[1].x * _v.y + _m.cols[2].x * _v.z +
1925 _m.cols[3].x * _v.w,
1926 _m.cols[0].y * _v.x + _m.cols[1].y * _v.y + _m.cols[2].y * _v.z +
1927 _m.cols[3].y * _v.w,
1928 _m.cols[0].z * _v.x + _m.cols[1].z * _v.y + _m.cols[2].z * _v.z +
1929 _m.cols[3].z * _v.w,
1930 _m.cols[0].w * _v.x + _m.cols[1].w * _v.y + _m.cols[2].w * _v.z +
1931 _m.cols[3].w * _v.w};
1932 return ret;
1933}
1934
1935OZZ_INLINE ozz::math::Float4x4 operator*(const ozz::math::Float4x4& _a,
1936 const ozz::math::Float4x4& _b) {
1937 const ozz::math::Float4x4 ret = {
1938 {_a * _b.cols[0], _a * _b.cols[1], _a * _b.cols[2], _a * _b.cols[3]}};
1939 return ret;
1940}
1941
1942OZZ_INLINE ozz::math::Float4x4 operator+(const ozz::math::Float4x4& _a,
1943 const ozz::math::Float4x4& _b) {
1944 const ozz::math::Float4x4 ret = {
1945 {{_a.cols[0].x + _b.cols[0].x, _a.cols[0].y + _b.cols[0].y,
1946 _a.cols[0].z + _b.cols[0].z, _a.cols[0].w + _b.cols[0].w},
1947 {_a.cols[1].x + _b.cols[1].x, _a.cols[1].y + _b.cols[1].y,
1948 _a.cols[1].z + _b.cols[1].z, _a.cols[1].w + _b.cols[1].w},
1949 {_a.cols[2].x + _b.cols[2].x, _a.cols[2].y + _b.cols[2].y,
1950 _a.cols[2].z + _b.cols[2].z, _a.cols[2].w + _b.cols[2].w},
1951 {_a.cols[3].x + _b.cols[3].x, _a.cols[3].y + _b.cols[3].y,
1952 _a.cols[3].z + _b.cols[3].z, _a.cols[3].w + _b.cols[3].w}}};
1953 return ret;
1954}
1955
1956OZZ_INLINE ozz::math::Float4x4 operator-(const ozz::math::Float4x4& _a,
1957 const ozz::math::Float4x4& _b) {
1958 const ozz::math::Float4x4 ret = {
1959 {{_a.cols[0].x - _b.cols[0].x, _a.cols[0].y - _b.cols[0].y,
1960 _a.cols[0].z - _b.cols[0].z, _a.cols[0].w - _b.cols[0].w},
1961 {_a.cols[1].x - _b.cols[1].x, _a.cols[1].y - _b.cols[1].y,
1962 _a.cols[1].z - _b.cols[1].z, _a.cols[1].w - _b.cols[1].w},
1963 {_a.cols[2].x - _b.cols[2].x, _a.cols[2].y - _b.cols[2].y,
1964 _a.cols[2].z - _b.cols[2].z, _a.cols[2].w - _b.cols[2].w},
1965 {_a.cols[3].x - _b.cols[3].x, _a.cols[3].y - _b.cols[3].y,
1966 _a.cols[3].z - _b.cols[3].z, _a.cols[3].w - _b.cols[3].w}}};
1967 return ret;
1968}
1969} // namespace math
1970} // namespace ozz
1971
1972OZZ_INLINE ozz::math::SimdFloat4 operator+(ozz::math::_SimdFloat4 _a,
1974 const ozz::math::SimdFloat4 ret = {_a.x + _b.x, _a.y + _b.y, _a.z + _b.z,
1975 _a.w + _b.w};
1976 return ret;
1977}
1978
1979OZZ_INLINE ozz::math::SimdFloat4 operator-(ozz::math::_SimdFloat4 _a,
1981 const ozz::math::SimdFloat4 ret = {_a.x - _b.x, _a.y - _b.y, _a.z - _b.z,
1982 _a.w - _b.w};
1983 return ret;
1984}
1985
1986OZZ_INLINE ozz::math::SimdFloat4 operator-(ozz::math::_SimdFloat4 _v) {
1987 const ozz::math::SimdFloat4 ret = {-_v.x, -_v.y, -_v.z, -_v.w};
1988 return ret;
1989}
1990
1991OZZ_INLINE ozz::math::SimdFloat4 operator*(ozz::math::_SimdFloat4 _a,
1993 const ozz::math::SimdFloat4 ret = {_a.x * _b.x, _a.y * _b.y, _a.z * _b.z,
1994 _a.w * _b.w};
1995 return ret;
1996}
1997
1998OZZ_INLINE ozz::math::SimdFloat4 operator/(ozz::math::_SimdFloat4 _a,
2000 const ozz::math::SimdFloat4 ret = {_a.x / _b.x, _a.y / _b.y, _a.z / _b.z,
2001 _a.w / _b.w};
2002 return ret;
2003}
2004
2005namespace ozz {
2006namespace math {
2007// Half <-> Float implementation is based on:
2008// http://fgiesen.wordpress.com/2012/03/28/half-to-float-done-quic/.
2009OZZ_INLINE uint16_t FloatToHalf(float _f) {
2010 const uint32_t f32infty = 255 << 23;
2011 const uint32_t f16infty = 31 << 23;
2012 const union {
2013 uint32_t u;
2014 float f;
2015 } magic = {15 << 23};
2016 const uint32_t sign_mask = 0x80000000u;
2017 const uint32_t round_mask = ~0x00000fffu;
2018
2019 const union {
2020 float f;
2021 uint32_t u;
2022 } f = {_f};
2023 const uint32_t sign = f.u & sign_mask;
2024 const uint32_t f_nosign = f.u & ~sign_mask;
2025
2026 if (f_nosign >= f32infty) { // Inf or NaN (all exponent bits set)
2027 // NaN->qNaN and Inf->Inf
2028 const uint32_t result =
2029 ((f_nosign > f32infty) ? 0x7e00 : 0x7c00) | (sign >> 16);
2030 return static_cast<uint16_t>(result);
2031 } else { // (De)normalized number or zero
2032 const union {
2033 uint32_t u;
2034 float f;
2035 } rounded = {f_nosign & round_mask};
2036 const union {
2037 float f;
2038 uint32_t u;
2039 } exp = {rounded.f * magic.f};
2040 const uint32_t re_rounded = exp.u - round_mask;
2041 // Clamp to signed infinity if overflowed
2042 const uint32_t result =
2043 ((re_rounded > f16infty ? f16infty : re_rounded) >> 13) | (sign >> 16);
2044 return static_cast<uint16_t>(result);
2045 }
2046}
2047
2048OZZ_INLINE float HalfToFloat(uint16_t _h) {
2049 const union {
2050 uint32_t u;
2051 float f;
2052 } magic = {(254 - 15) << 23};
2053 const union {
2054 uint32_t u;
2055 float f;
2056 } infnan = {(127 + 16) << 23};
2057
2058 const uint32_t sign = _h & 0x8000;
2059 const union {
2060 int32_t u;
2061 float f;
2062 } exp_mant = {(_h & 0x7fff) << 13};
2063 const union {
2064 float f;
2065 uint32_t u;
2066 } adjust = {exp_mant.f * magic.f};
2067 // Make sure Inf/NaN survive
2068 const union {
2069 uint32_t u;
2070 float f;
2071 } result = {(adjust.f >= infnan.f ? (adjust.u | 255 << 23) : adjust.u) |
2072 (sign << 16)};
2073 return result.f;
2074}
2075
2076OZZ_INLINE SimdInt4 FloatToHalf(_SimdFloat4 _f) {
2077 const ozz::math::SimdInt4 ret = {FloatToHalf(_f.x), FloatToHalf(_f.y),
2078 FloatToHalf(_f.z), FloatToHalf(_f.w)};
2079 return ret;
2080}
2081
2082OZZ_INLINE SimdFloat4 HalfToFloat(_SimdInt4 _h) {
2083 const ozz::math::SimdFloat4 ret = {
2084 HalfToFloat(_h.x & 0x0000ffff), HalfToFloat(_h.y & 0x0000ffff),
2085 HalfToFloat(_h.z & 0x0000ffff), HalfToFloat(_h.w & 0x0000ffff)};
2086 return ret;
2087}
2088} // namespace math
2089} // namespace ozz
2090
2091#undef OZZ_RCP_EST
2092#undef OZZ_RSQRT_EST
2093#endif // OZZ_OZZ_BASE_MATHS_INTERNAL_SIMD_MATH_REF_INL_H_
GLM_FUNC_QUALIFIER vec< L, T, Q > exp(vec< L, T, Q > const &x)
Definition func_exponential.inl:80
GLM_FUNC_QUALIFIER vec< 3, T, Q > cross(vec< 3, T, Q > const &x, vec< 3, T, Q > const &y)
Definition func_geometric.inl:175
GLM_FUNC_QUALIFIER vec< L, T, Q > sin(vec< L, T, Q > const &v)
Definition func_trigonometric.inl:41
GLM_FUNC_QUALIFIER vec< L, T, Q > cos(vec< L, T, Q > const &v)
Definition func_trigonometric.inl:50
GLM_FUNC_DECL GLM_CONSTEXPR genType pi()
Return the pi constant for floating point types.
Definition scalar_constants.inl:13
GLM_FUNC_DECL GLM_CONSTEXPR genType zero()
Definition constants.inl:6
uint32 uint32_t
Definition fwd.hpp:131
int32 int32_t
Definition fwd.hpp:71
uint16 uint16_t
Definition fwd.hpp:117
Definition simd_math_config.h:121
Definition simd_math_config.h:129
Definition simd_math.h:1066
Definition simd_math_ref-inl.h:47
Definition simd_math_ref-inl.h:51