RavEngine
Loading...
Searching...
No Matches
simd_math.h
1//----------------------------------------------------------------------------//
2// //
3// ozz-animation is hosted at http://github.com/guillaumeblanc/ozz-animation //
4// and distributed under the MIT License (MIT). //
5// //
6// Copyright (c) Guillaume Blanc //
7// //
8// Permission is hereby granted, free of charge, to any person obtaining a //
9// copy of this software and associated documentation files (the "Software"), //
10// to deal in the Software without restriction, including without limitation //
11// the rights to use, copy, modify, merge, publish, distribute, sublicense, //
12// and/or sell copies of the Software, and to permit persons to whom the //
13// Software is furnished to do so, subject to the following conditions: //
14// //
15// The above copyright notice and this permission notice shall be included in //
16// all copies or substantial portions of the Software. //
17// //
18// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR //
19// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, //
20// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL //
21// THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER //
22// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING //
23// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER //
24// DEALINGS IN THE SOFTWARE. //
25// //
26//----------------------------------------------------------------------------//
27
28#ifndef OZZ_OZZ_BASE_MATHS_SIMD_MATH_H_
29#define OZZ_OZZ_BASE_MATHS_SIMD_MATH_H_
30
31#include "ozz/base/maths/internal/simd_math_config.h"
32#include "ozz/base/platform.h"
33
34namespace ozz {
35namespace math {
36
37// Returns SIMDimplementation name has decided at library build time.
38OZZ_BASE_DLL const char* SimdImplementationName();
39
40namespace simd_float4 {
41// Returns a SimdFloat4 vector with all components set to 0.
42OZZ_INLINE SimdFloat4 zero();
43
44// Returns a SimdFloat4 vector with all components set to 1.
45OZZ_INLINE SimdFloat4 one();
46
47// Returns a SimdFloat4 vector with the x component set to 1 and all the others
48// to 0.
49OZZ_INLINE SimdFloat4 x_axis();
50
51// Returns a SimdFloat4 vector with the y component set to 1 and all the others
52// to 0.
53OZZ_INLINE SimdFloat4 y_axis();
54
55// Returns a SimdFloat4 vector with the z component set to 1 and all the others
56// to 0.
57OZZ_INLINE SimdFloat4 z_axis();
58
59// Returns a SimdFloat4 vector with the w component set to 1 and all the others
60// to 0.
61OZZ_INLINE SimdFloat4 w_axis();
62
63// Loads _x, _y, _z, _w to the returned vector.
64// r.x = _x
65// r.y = _y
66// r.z = _z
67// r.w = _w
68OZZ_INLINE SimdFloat4 Load(float _x, float _y, float _z, float _w);
69
70// Loads _x to the x component of the returned vector, and sets y, z and w to 0.
71// r.x = _x
72// r.y = 0
73// r.z = 0
74// r.w = 0
75OZZ_INLINE SimdFloat4 LoadX(float _x);
76
77// Loads _x to the all the components of the returned vector.
78// r.x = _x
79// r.y = _x
80// r.z = _x
81// r.w = _x
82OZZ_INLINE SimdFloat4 Load1(float _x);
83
84// Loads the 4 values of _f to the returned vector.
85// _f must be aligned to 16 bytes.
86// r.x = _f[0]
87// r.y = _f[1]
88// r.z = _f[2]
89// r.w = _f[3]
90OZZ_INLINE SimdFloat4 LoadPtr(const float* _f);
91
92// Loads the 4 values of _f to the returned vector.
93// _f must be aligned to 4 bytes.
94// r.x = _f[0]
95// r.y = _f[1]
96// r.z = _f[2]
97// r.w = _f[3]
98OZZ_INLINE SimdFloat4 LoadPtrU(const float* _f);
99
100// Loads _f[0] to the x component of the returned vector, and sets y, z and w
101// to 0.
102// _f must be aligned to 4 bytes.
103// r.x = _f[0]
104// r.y = 0
105// r.z = 0
106// r.w = 0
107OZZ_INLINE SimdFloat4 LoadXPtrU(const float* _f);
108
109// Loads _f[0] to all the components of the returned vector.
110// _f must be aligned to 4 bytes.
111// r.x = _f[0]
112// r.y = _f[0]
113// r.z = _f[0]
114// r.w = _f[0]
115OZZ_INLINE SimdFloat4 Load1PtrU(const float* _f);
116
117// Loads the 2 first value of _f to the x and y components of the returned
118// vector. The remaining components are set to 0.
119// _f must be aligned to 4 bytes.
120// r.x = _f[0]
121// r.y = _f[1]
122// r.z = 0
123// r.w = 0
124OZZ_INLINE SimdFloat4 Load2PtrU(const float* _f);
125
126// Loads the 3 first value of _f to the x, y and z components of the returned
127// vector. The remaining components are set to 0.
128// _f must be aligned to 4 bytes.
129// r.x = _f[0]
130// r.y = _f[1]
131// r.z = _f[2]
132// r.w = 0
133OZZ_INLINE SimdFloat4 Load3PtrU(const float* _f);
134
135// Convert from integer to float.
136OZZ_INLINE SimdFloat4 FromInt(_SimdInt4 _i);
137} // namespace simd_float4
138
139// Returns the x component of _v as a float.
140OZZ_INLINE float GetX(_SimdFloat4 _v);
141
142// Returns the y component of _v as a float.
143OZZ_INLINE float GetY(_SimdFloat4 _v);
144
145// Returns the z component of _v as a float.
146OZZ_INLINE float GetZ(_SimdFloat4 _v);
147
148// Returns the w component of _v as a float.
149OZZ_INLINE float GetW(_SimdFloat4 _v);
150
151// Returns _v with the x component set to x component of _f.
152OZZ_INLINE SimdFloat4 SetX(_SimdFloat4 _v, _SimdFloat4 _f);
153
154// Returns _v with the y component set to x component of _f.
155OZZ_INLINE SimdFloat4 SetY(_SimdFloat4 _v, _SimdFloat4 _f);
156
157// Returns _v with the z component set to x component of _f.
158OZZ_INLINE SimdFloat4 SetZ(_SimdFloat4 _v, _SimdFloat4 _f);
159
160// Returns _v with the w component set to x component of _f.
161OZZ_INLINE SimdFloat4 SetW(_SimdFloat4 _v, _SimdFloat4 _f);
162
163// Returns _v with the _i th component set to _f.
164// _i must be in range [0,3]
165OZZ_INLINE SimdFloat4 SetI(_SimdFloat4 _v, _SimdFloat4 _f, int _i);
166
167// Stores the 4 components of _v to the four first floats of _f.
168// _f must be aligned to 16 bytes.
169// _f[0] = _v.x
170// _f[1] = _v.y
171// _f[2] = _v.z
172// _f[3] = _v.w
173OZZ_INLINE void StorePtr(_SimdFloat4 _v, float* _f);
174
175// Stores the x component of _v to the first float of _f.
176// _f must be aligned to 16 bytes.
177// _f[0] = _v.x
178OZZ_INLINE void Store1Ptr(_SimdFloat4 _v, float* _f);
179
180// Stores x and y components of _v to the two first floats of _f.
181// _f must be aligned to 16 bytes.
182// _f[0] = _v.x
183// _f[1] = _v.y
184OZZ_INLINE void Store2Ptr(_SimdFloat4 _v, float* _f);
185
186// Stores x, y and z components of _v to the three first floats of _f.
187// _f must be aligned to 16 bytes.
188// _f[0] = _v.x
189// _f[1] = _v.y
190// _f[2] = _v.z
191OZZ_INLINE void Store3Ptr(_SimdFloat4 _v, float* _f);
192
193// Stores the 4 components of _v to the four first floats of _f.
194// _f must be aligned to 4 bytes.
195// _f[0] = _v.x
196// _f[1] = _v.y
197// _f[2] = _v.z
198// _f[3] = _v.w
199OZZ_INLINE void StorePtrU(_SimdFloat4 _v, float* _f);
200
201// Stores the x component of _v to the first float of _f.
202// _f must be aligned to 4 bytes.
203// _f[0] = _v.x
204OZZ_INLINE void Store1PtrU(_SimdFloat4 _v, float* _f);
205
206// Stores x and y components of _v to the two first floats of _f.
207// _f must be aligned to 4 bytes.
208// _f[0] = _v.x
209// _f[1] = _v.y
210OZZ_INLINE void Store2PtrU(_SimdFloat4 _v, float* _f);
211
212// Stores x, y and z components of _v to the three first floats of _f.
213// _f must be aligned to 4 bytes.
214// _f[0] = _v.x
215// _f[1] = _v.y
216// _f[2] = _v.z
217OZZ_INLINE void Store3PtrU(_SimdFloat4 _v, float* _f);
218
219// Replicates x of _a to all the components of the returned vector.
220OZZ_INLINE SimdFloat4 SplatX(_SimdFloat4 _v);
221
222// Replicates y of _a to all the components of the returned vector.
223OZZ_INLINE SimdFloat4 SplatY(_SimdFloat4 _v);
224
225// Replicates z of _a to all the components of the returned vector.
226OZZ_INLINE SimdFloat4 SplatZ(_SimdFloat4 _v);
227
228// Replicates w of _a to all the components of the returned vector.
229OZZ_INLINE SimdFloat4 SplatW(_SimdFloat4 _v);
230
231// Swizzle x, y, z and w components based on compile time arguments _X, _Y, _Z
232// and _W. Arguments can vary from 0 (x), to 3 (w).
233template <size_t _X, size_t _Y, size_t _Z, size_t _W>
234OZZ_INLINE SimdFloat4 Swizzle(_SimdFloat4 _v);
235
236// Transposes the x components of the 4 SimdFloat4 of _in into the 1
237// SimdFloat4 of _out.
238OZZ_INLINE void Transpose4x1(const SimdFloat4 _in[4], SimdFloat4 _out[1]);
239
240// Transposes x, y, z and w components of _in to the x components of _out.
241// Remaining y, z and w are set to 0.
242OZZ_INLINE void Transpose1x4(const SimdFloat4 _in[1], SimdFloat4 _out[4]);
243
244// Transposes the x and y components of the 4 SimdFloat4 of _in into the 2
245// SimdFloat4 of _out.
246OZZ_INLINE void Transpose4x2(const SimdFloat4 _in[4], SimdFloat4 _out[2]);
247
248// Transposes the 2 SimdFloat4 of _in into the x and y components of the 4
249// SimdFloat4 of _out. Remaining z and w are set to 0.
250OZZ_INLINE void Transpose2x4(const SimdFloat4 _in[2], SimdFloat4 _out[4]);
251
252// Transposes the x, y and z components of the 4 SimdFloat4 of _in into the 3
253// SimdFloat4 of _out.
254OZZ_INLINE void Transpose4x3(const SimdFloat4 _in[4], SimdFloat4 _out[3]);
255
256// Transposes the 3 SimdFloat4 of _in into the x, y and z components of the 4
257// SimdFloat4 of _out. Remaining w are set to 0.
258OZZ_INLINE void Transpose3x4(const SimdFloat4 _in[3], SimdFloat4 _out[4]);
259
260// Transposes the 4 SimdFloat4 of _in into the 4 SimdFloat4 of _out.
261OZZ_INLINE void Transpose4x4(const SimdFloat4 _in[4], SimdFloat4 _out[4]);
262
263// Transposes the 16 SimdFloat4 of _in into the 16 SimdFloat4 of _out.
264OZZ_INLINE void Transpose16x16(const SimdFloat4 _in[16], SimdFloat4 _out[16]);
265
266// Multiplies _a and _b, then adds _c.
267// v = (_a * _b) + _c
268OZZ_INLINE SimdFloat4 MAdd(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c);
269
270// Multiplies _a and _b, then subs _c.
271// v = (_a * _b) + _c
272OZZ_INLINE SimdFloat4 MSub(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c);
273
274// Multiplies _a and _b, negate it, then adds _c.
275// v = -(_a * _b) + _c
276OZZ_INLINE SimdFloat4 NMAdd(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c);
277
278// Multiplies _a and _b, negate it, then subs _c.
279// v = -(_a * _b) + _c
280OZZ_INLINE SimdFloat4 NMSub(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _c);
281
282// Divides the x component of _a by the _x component of _b and stores it in the
283// x component of the returned vector. y, z, w of the returned vector are the
284// same as _a respective components.
285// r.x = _a.x / _b.x
286// r.y = _a.y
287// r.z = _a.z
288// r.w = _a.w
289OZZ_INLINE SimdFloat4 DivX(_SimdFloat4 _a, _SimdFloat4 _b);
290
291// Computes the (horizontal) addition of x and y components of _v. The result is
292// stored in the x component of the returned value. y, z, w of the returned
293// vector are the same as their respective components in _v.
294// r.x = _a.x + _a.y
295// r.y = _a.y
296// r.z = _a.z
297// r.w = _a.w
298OZZ_INLINE SimdFloat4 HAdd2(_SimdFloat4 _v);
299
300// Computes the (horizontal) addition of x, y and z components of _v. The result
301// is stored in the x component of the returned value. y, z, w of the returned
302// vector are the same as their respective components in _v.
303// r.x = _a.x + _a.y + _a.z
304// r.y = _a.y
305// r.z = _a.z
306// r.w = _a.w
307OZZ_INLINE SimdFloat4 HAdd3(_SimdFloat4 _v);
308
309// Computes the (horizontal) addition of x and y components of _v. The result is
310// stored in the x component of the returned value. y, z, w of the returned
311// vector are the same as their respective components in _v.
312// r.x = _a.x + _a.y + _a.z + _a.w
313// r.y = _a.y
314// r.z = _a.z
315// r.w = _a.w
316OZZ_INLINE SimdFloat4 HAdd4(_SimdFloat4 _v);
317
318// Computes the dot product of x and y components of _v. The result is
319// stored in the x component of the returned value. y, z, w of the returned
320// vector are undefined.
321// r.x = _a.x * _a.x + _a.y * _a.y
322// r.y = ?
323// r.z = ?
324// r.w = ?
325OZZ_INLINE SimdFloat4 Dot2(_SimdFloat4 _a, _SimdFloat4 _b);
326
327// Computes the dot product of x, y and z components of _v. The result is
328// stored in the x component of the returned value. y, z, w of the returned
329// vector are undefined.
330// r.x = _a.x * _a.x + _a.y * _a.y + _a.z * _a.z
331// r.y = ?
332// r.z = ?
333// r.w = ?
334OZZ_INLINE SimdFloat4 Dot3(_SimdFloat4 _a, _SimdFloat4 _b);
335
336// Computes the dot product of x, y, z and w components of _v. The result is
337// stored in the x component of the returned value. y, z, w of the returned
338// vector are undefined.
339// r.x = _a.x * _a.x + _a.y * _a.y + _a.z * _a.z + _a.w * _a.w
340// r.y = ?
341// r.z = ?
342// r.w = ?
343OZZ_INLINE SimdFloat4 Dot4(_SimdFloat4 _a, _SimdFloat4 _b);
344
345// Computes the cross product of x, y and z components of _v. The result is
346// stored in the x, y and z components of the returned value. w of the returned
347// vector is undefined.
348// r.x = _a.y * _b.z - _a.z * _b.y
349// r.y = _a.z * _b.x - _a.x * _b.z
350// r.z = _a.x * _b.y - _a.y * _b.x
351// r.w = ?
352OZZ_INLINE SimdFloat4 Cross3(_SimdFloat4 _a, _SimdFloat4 _b);
353
354// Returns the per component estimated reciprocal of _v.
355OZZ_INLINE SimdFloat4 RcpEst(_SimdFloat4 _v);
356
357// Returns the per component estimated reciprocal of _v, where approximation is
358// improved with one more new Newton-Raphson step.
359OZZ_INLINE SimdFloat4 RcpEstNR(_SimdFloat4 _v);
360
361// Returns the estimated reciprocal of the x component of _v and stores it in
362// the x component of the returned vector. y, z, w of the returned vector are
363// the same as their respective components in _v.
364OZZ_INLINE SimdFloat4 RcpEstX(_SimdFloat4 _v);
365
366// Returns the estimated reciprocal of the x component of _v, where
367// approximation is improved with one more new Newton-Raphson step. y, z, w of
368// the returned vector are undefined.
369OZZ_INLINE SimdFloat4 RcpEstXNR(_SimdFloat4 _v);
370
371// Returns the per component square root of _v.
372OZZ_INLINE SimdFloat4 Sqrt(_SimdFloat4 _v);
373
374// Returns the square root of the x component of _v and stores it in the x
375// component of the returned vector. y, z, w of the returned vector are the
376// same as their respective components in _v.
377OZZ_INLINE SimdFloat4 SqrtX(_SimdFloat4 _v);
378
379// Returns the per component estimated reciprocal square root of _v.
380OZZ_INLINE SimdFloat4 RSqrtEst(_SimdFloat4 _v);
381
382// Returns the per component estimated reciprocal square root of _v, where
383// approximation is improved with one more new Newton-Raphson step.
384OZZ_INLINE SimdFloat4 RSqrtEstNR(_SimdFloat4 _v);
385
386// Returns the estimated reciprocal square root of the x component of _v and
387// stores it in the x component of the returned vector. y, z, w of the returned
388// vector are the same as their respective components in _v.
389OZZ_INLINE SimdFloat4 RSqrtEstX(_SimdFloat4 _v);
390
391// Returns the estimated reciprocal square root of the x component of _v, where
392// approximation is improved with one more new Newton-Raphson step. y, z, w of
393// the returned vector are undefined.
394OZZ_INLINE SimdFloat4 RSqrtEstXNR(_SimdFloat4 _v);
395
396// Returns the per element absolute value of _v.
397OZZ_INLINE SimdFloat4 Abs(_SimdFloat4 _v);
398
399// Returns the sign bit of _v.
400OZZ_INLINE SimdInt4 Sign(_SimdFloat4 _v);
401
402// Returns the per component minimum of _a and _b.
403OZZ_INLINE SimdFloat4 Min(_SimdFloat4 _a, _SimdFloat4 _b);
404
405// Returns the per component maximum of _a and _b.
406OZZ_INLINE SimdFloat4 Max(_SimdFloat4 _a, _SimdFloat4 _b);
407
408// Returns the per component minimum of _v and 0.
409OZZ_INLINE SimdFloat4 Min(_SimdFloat4 _v);
410
411// Returns the per component maximum of _v and 0.
412OZZ_INLINE SimdFloat4 Max0(_SimdFloat4 _v);
413
414// Clamps each element of _x between _a and _b.
415// Result is unknown if _a is not less or equal to _b.
416OZZ_INLINE SimdFloat4 Clamp(_SimdFloat4 _a, _SimdFloat4 _v, _SimdFloat4 _b);
417
418// Computes the length of the components x and y of _v, and stores it in the x
419// component of the returned vector. y, z, w of the returned vector are
420// undefined.
421OZZ_INLINE SimdFloat4 Length2(_SimdFloat4 _v);
422
423// Computes the length of the components x, y and z of _v, and stores it in the
424// x component of the returned vector. undefined.
425OZZ_INLINE SimdFloat4 Length3(_SimdFloat4 _v);
426
427// Computes the length of _v, and stores it in the x component of the returned
428// vector. y, z, w of the returned vector are undefined.
429OZZ_INLINE SimdFloat4 Length4(_SimdFloat4 _v);
430
431// Computes the square length of the components x and y of _v, and stores it
432// in the x component of the returned vector. y, z, w of the returned vector are
433// undefined.
434OZZ_INLINE SimdFloat4 Length2Sqr(_SimdFloat4 _v);
435
436// Computes the square length of the components x, y and z of _v, and stores it
437// in the x component of the returned vector. y, z, w of the returned vector are
438// undefined.
439OZZ_INLINE SimdFloat4 Length3Sqr(_SimdFloat4 _v);
440
441// Computes the square length of the components x, y, z and w of _v, and stores
442// it in the x component of the returned vector. y, z, w of the returned vector
443// undefined.
444OZZ_INLINE SimdFloat4 Length4Sqr(_SimdFloat4 _v);
445
446// Returns the normalized vector of the components x and y of _v, and stores
447// it in the x and y components of the returned vector. z and w of the returned
448// vector are the same as their respective components in _v.
449OZZ_INLINE SimdFloat4 Normalize2(_SimdFloat4 _v);
450
451// Returns the normalized vector of the components x, y and z of _v, and stores
452// it in the x, y and z components of the returned vector. w of the returned
453// vector is the same as its respective component in _v.
454OZZ_INLINE SimdFloat4 Normalize3(_SimdFloat4 _v);
455
456// Returns the normalized vector _v.
457OZZ_INLINE SimdFloat4 Normalize4(_SimdFloat4 _v);
458
459// Returns the estimated normalized vector of the components x and y of _v, and
460// stores it in the x and y components of the returned vector. z and w of the
461// returned vector are the same as their respective components in _v.
462OZZ_INLINE SimdFloat4 NormalizeEst2(_SimdFloat4 _v);
463
464// Returns the estimated normalized vector of the components x, y and z of _v,
465// and stores it in the x, y and z components of the returned vector. w of the
466// returned vector is the same as its respective component in _v.
467OZZ_INLINE SimdFloat4 NormalizeEst3(_SimdFloat4 _v);
468
469// Returns the estimated normalized vector _v.
470OZZ_INLINE SimdFloat4 NormalizeEst4(_SimdFloat4 _v);
471
472// Tests if the components x and y of _v forms a normalized vector.
473// Returns the result in the x component of the returned vector. y, z and w are
474// set to 0.
475OZZ_INLINE SimdInt4 IsNormalized2(_SimdFloat4 _v);
476
477// Tests if the components x, y and z of _v forms a normalized vector.
478// Returns the result in the x component of the returned vector. y, z and w are
479// set to 0.
480OZZ_INLINE SimdInt4 IsNormalized3(_SimdFloat4 _v);
481
482// Tests if the _v is a normalized vector.
483// Returns the result in the x component of the returned vector. y, z and w are
484// set to 0.
485OZZ_INLINE SimdInt4 IsNormalized4(_SimdFloat4 _v);
486
487// Tests if the components x and y of _v forms a normalized vector.
488// Uses the estimated normalization coefficient, that matches estimated math
489// functions (RecpEst, MormalizeEst...).
490// Returns the result in the x component of the returned vector. y, z and w are
491// set to 0.
492OZZ_INLINE SimdInt4 IsNormalizedEst2(_SimdFloat4 _v);
493
494// Tests if the components x, y and z of _v forms a normalized vector.
495// Uses the estimated normalization coefficient, that matches estimated math
496// functions (RecpEst, MormalizeEst...).
497// Returns the result in the x component of the returned vector. y, z and w are
498// set to 0.
499OZZ_INLINE SimdInt4 IsNormalizedEst3(_SimdFloat4 _v);
500
501// Tests if the _v is a normalized vector.
502// Uses the estimated normalization coefficient, that matches estimated math
503// functions (RecpEst, MormalizeEst...).
504// Returns the result in the x component of the returned vector. y, z and w are
505// set to 0.
506OZZ_INLINE SimdInt4 IsNormalizedEst4(_SimdFloat4 _v);
507
508// Returns the normalized vector of the components x and y of _v if it is
509// normalizable, otherwise returns _safe. z and w of the returned vector are
510// the same as their respective components in _v.
511OZZ_INLINE SimdFloat4 NormalizeSafe2(_SimdFloat4 _v, _SimdFloat4 _safe);
512
513// Returns the normalized vector of the components x, y, z and w of _v if it is
514// normalizable, otherwise returns _safe. w of the returned vector is the same
515// as its respective components in _v.
516OZZ_INLINE SimdFloat4 NormalizeSafe3(_SimdFloat4 _v, _SimdFloat4 _safe);
517
518// Returns the normalized vector _v if it is normalizable, otherwise returns
519// _safe.
520OZZ_INLINE SimdFloat4 NormalizeSafe4(_SimdFloat4 _v, _SimdFloat4 _safe);
521
522// Returns the estimated normalized vector of the components x and y of _v if it
523// is normalizable, otherwise returns _safe. z and w of the returned vector are
524// the same as their respective components in _v.
525OZZ_INLINE SimdFloat4 NormalizeSafeEst2(_SimdFloat4 _v, _SimdFloat4 _safe);
526
527// Returns the estimated normalized vector of the components x, y, z and w of _v
528// if it is normalizable, otherwise returns _safe. w of the returned vector is
529// the same as its respective components in _v.
530OZZ_INLINE SimdFloat4 NormalizeSafeEst3(_SimdFloat4 _v, _SimdFloat4 _safe);
531
532// Returns the estimated normalized vector _v if it is normalizable, otherwise
533// returns _safe.
534OZZ_INLINE SimdFloat4 NormalizeSafeEst4(_SimdFloat4 _v, _SimdFloat4 _safe);
535
536// Computes the per element linear interpolation of _a and _b, where _alpha is
537// not bound to range [0,1].
538OZZ_INLINE SimdFloat4 Lerp(_SimdFloat4 _a, _SimdFloat4 _b, _SimdFloat4 _alpha);
539
540// Computes the per element cosine of _v.
541OZZ_INLINE SimdFloat4 Cos(_SimdFloat4 _v);
542
543// Computes the cosine of the x component of _v and stores it in the x
544// component of the returned vector. y, z and w of the returned vector are the
545// same as their respective components in _v.
546OZZ_INLINE SimdFloat4 CosX(_SimdFloat4 _v);
547
548// Computes the per element arccosine of _v.
549OZZ_INLINE SimdFloat4 ACos(_SimdFloat4 _v);
550
551// Computes the arccosine of the x component of _v and stores it in the x
552// component of the returned vector. y, z and w of the returned vector are the
553// same as their respective components in _v.
554OZZ_INLINE SimdFloat4 ACosX(_SimdFloat4 _v);
555
556// Computes the per element sines of _v.
557OZZ_INLINE SimdFloat4 Sin(_SimdFloat4 _v);
558
559// Computes the sines of the x component of _v and stores it in the x
560// component of the returned vector. y, z and w of the returned vector are the
561// same as their respective components in _v.
562OZZ_INLINE SimdFloat4 SinX(_SimdFloat4 _v);
563
564// Computes the per element arcsine of _v.
565OZZ_INLINE SimdFloat4 ASin(_SimdFloat4 _v);
566
567// Computes the arcsine of the x component of _v and stores it in the x
568// component of the returned vector. y, z and w of the returned vector are the
569// same as their respective components in _v.
570OZZ_INLINE SimdFloat4 ASinX(_SimdFloat4 _v);
571
572// Computes the per element tangent of _v.
573OZZ_INLINE SimdFloat4 Tan(_SimdFloat4 _v);
574
575// Computes the tangent of the x component of _v and stores it in the x
576// component of the returned vector. y, z and w of the returned vector are the
577// same as their respective components in _v.
578OZZ_INLINE SimdFloat4 TanX(_SimdFloat4 _v);
579
580// Computes the per element arctangent of _v.
581OZZ_INLINE SimdFloat4 ATan(_SimdFloat4 _v);
582
583// Computes the arctangent of the x component of _v and stores it in the x
584// component of the returned vector. y, z and w of the returned vector are the
585// same as their respective components in _v.
586OZZ_INLINE SimdFloat4 ATanX(_SimdFloat4 _v);
587
588// Returns boolean selection of vectors _true and _false according to condition
589// _b. All bits a each component of _b must have the same value (O or
590// 0xffffffff) to ensure portability.
591OZZ_INLINE SimdFloat4 Select(_SimdInt4 _b, _SimdFloat4 _true,
592 _SimdFloat4 _false);
593
594// Per element "equal" comparison of _a and _b.
595OZZ_INLINE SimdInt4 CmpEq(_SimdFloat4 _a, _SimdFloat4 _b);
596
597// Per element "not equal" comparison of _a and _b.
598OZZ_INLINE SimdInt4 CmpNe(_SimdFloat4 _a, _SimdFloat4 _b);
599
600// Per element "less than" comparison of _a and _b.
601OZZ_INLINE SimdInt4 CmpLt(_SimdFloat4 _a, _SimdFloat4 _b);
602
603// Per element "less than or equal" comparison of _a and _b.
604OZZ_INLINE SimdInt4 CmpLe(_SimdFloat4 _a, _SimdFloat4 _b);
605
606// Per element "greater than" comparison of _a and _b.
607OZZ_INLINE SimdInt4 CmpGt(_SimdFloat4 _a, _SimdFloat4 _b);
608
609// Per element "greater than or equal" comparison of _a and _b.
610OZZ_INLINE SimdInt4 CmpGe(_SimdFloat4 _a, _SimdFloat4 _b);
611
612// Returns per element binary and operation of _a and _b.
613// _v[0...127] = _a[0...127] & _b[0...127]
614OZZ_INLINE SimdFloat4 And(_SimdFloat4 _a, _SimdFloat4 _b);
615
616// Returns per element binary or operation of _a and _b.
617// _v[0...127] = _a[0...127] | _b[0...127]
618OZZ_INLINE SimdFloat4 Or(_SimdFloat4 _a, _SimdFloat4 _b);
619
620// Returns per element binary logical xor operation of _a and _b.
621// _v[0...127] = _a[0...127] ^ _b[0...127]
622OZZ_INLINE SimdFloat4 Xor(_SimdFloat4 _a, _SimdFloat4 _b);
623
624// Returns per element binary and operation of _a and _b.
625// _v[0...127] = _a[0...127] & _b[0...127]
626OZZ_INLINE SimdFloat4 And(_SimdFloat4 _a, _SimdInt4 _b);
627
628// Returns per element binary and operation of _a and ~_b.
629// _v[0...127] = _a[0...127] & ~_b[0...127]
630OZZ_INLINE SimdFloat4 AndNot(_SimdFloat4 _a, _SimdInt4 _b);
631
632// Returns per element binary or operation of _a and _b.
633// _v[0...127] = _a[0...127] | _b[0...127]
634OZZ_INLINE SimdFloat4 Or(_SimdFloat4 _a, _SimdInt4 _b);
635
636// Returns per element binary logical xor operation of _a and _b.
637// _v[0...127] = _a[0...127] ^ _b[0...127]
638OZZ_INLINE SimdFloat4 Xor(_SimdFloat4 _a, _SimdInt4 _b);
639
640namespace simd_int4 {
641// Returns a SimdInt4 vector with all components set to 0.
642OZZ_INLINE SimdInt4 zero();
643
644// Returns a SimdInt4 vector with all components set to 1.
645OZZ_INLINE SimdInt4 one();
646
647// Returns a SimdInt4 vector with the x component set to 1 and all the others
648// to 0.
649OZZ_INLINE SimdInt4 x_axis();
650
651// Returns a SimdInt4 vector with the y component set to 1 and all the others
652// to 0.
653OZZ_INLINE SimdInt4 y_axis();
654
655// Returns a SimdInt4 vector with the z component set to 1 and all the others
656// to 0.
657OZZ_INLINE SimdInt4 z_axis();
658
659// Returns a SimdInt4 vector with the w component set to 1 and all the others
660// to 0.
661OZZ_INLINE SimdInt4 w_axis();
662
663// Returns a SimdInt4 vector with all components set to true (0xffffffff).
664OZZ_INLINE SimdInt4 all_true();
665
666// Returns a SimdInt4 vector with all components set to false (0).
667OZZ_INLINE SimdInt4 all_false();
668
669// Returns a SimdInt4 vector with sign bits set to 1.
670OZZ_INLINE SimdInt4 mask_sign();
671
672// Returns a SimdInt4 vector with all bits set to 1 except sign.
673OZZ_INLINE SimdInt4 mask_not_sign();
674
675// Returns a SimdInt4 vector with sign bits of x, y and z components set to 1.
676OZZ_INLINE SimdInt4 mask_sign_xyz();
677
678// Returns a SimdInt4 vector with sign bits of w component set to 1.
679OZZ_INLINE SimdInt4 mask_sign_w();
680
681// Returns a SimdInt4 vector with all bits set to 1.
682OZZ_INLINE SimdInt4 mask_ffff();
683
684// Returns a SimdInt4 vector with all bits set to 0.
685OZZ_INLINE SimdInt4 mask_0000();
686
687// Returns a SimdInt4 vector with all the bits of the x, y, z components set to
688// 1, while z is set to 0.
689OZZ_INLINE SimdInt4 mask_fff0();
690
691// Returns a SimdInt4 vector with all the bits of the x component set to 1,
692// while the others are set to 0.
693OZZ_INLINE SimdInt4 mask_f000();
694
695// Returns a SimdInt4 vector with all the bits of the y component set to 1,
696// while the others are set to 0.
697OZZ_INLINE SimdInt4 mask_0f00();
698
699// Returns a SimdInt4 vector with all the bits of the z component set to 1,
700// while the others are set to 0.
701OZZ_INLINE SimdInt4 mask_00f0();
702
703// Returns a SimdInt4 vector with all the bits of the w component set to 1,
704// while the others are set to 0.
705OZZ_INLINE SimdInt4 mask_000f();
706
707// Loads _x, _y, _z, _w to the returned vector.
708// r.x = _x
709// r.y = _y
710// r.z = _z
711// r.w = _w
712OZZ_INLINE SimdInt4 Load(int _x, int _y, int _z, int _w);
713
714// Loads _x, _y, _z, _w to the returned vector using the following conversion
715// rule.
716// r.x = _x ? 0xffffffff:0
717// r.y = _y ? 0xffffffff:0
718// r.z = _z ? 0xffffffff:0
719// r.w = _w ? 0xffffffff:0
720OZZ_INLINE SimdInt4 Load(bool _x, bool _y, bool _z, bool _w);
721
722// Loads _x to the x component of the returned vector using the following
723// conversion rule, and sets y, z and w to 0.
724// r.x = _x ? 0xffffffff:0
725// r.y = 0
726// r.z = 0
727// r.w = 0
728OZZ_INLINE SimdInt4 LoadX(bool _x);
729
730// Loads _x to the all the components of the returned vector using the following
731// conversion rule.
732// r.x = _x ? 0xffffffff:0
733// r.y = _x ? 0xffffffff:0
734// r.z = _x ? 0xffffffff:0
735// r.w = _x ? 0xffffffff:0
736OZZ_INLINE SimdInt4 Load1(bool _x);
737
738// Loads the 4 values of _f to the returned vector.
739// _i must be aligned to 16 bytes.
740// r.x = _i[0]
741// r.y = _i[1]
742// r.z = _i[2]
743// r.w = _i[3]
744OZZ_INLINE SimdInt4 LoadPtr(const int* _i);
745
746// Loads _i[0] to the x component of the returned vector, and sets y, z and w
747// to 0.
748// _i must be aligned to 16 bytes.
749// r.x = _i[0]
750// r.y = 0
751// r.z = 0
752// r.w = 0
753OZZ_INLINE SimdInt4 LoadXPtr(const int* _i);
754
755// Loads _i[0] to all the components of the returned vector.
756// _i must be aligned to 16 bytes.
757// r.x = _i[0]
758// r.y = _i[0]
759// r.z = _i[0]
760// r.w = _i[0]
761OZZ_INLINE SimdInt4 Load1Ptr(const int* _i);
762
763// Loads the 2 first value of _i to the x and y components of the returned
764// vector. The remaining components are set to 0.
765// _f must be aligned to 4 bytes.
766// r.x = _i[0]
767// r.y = _i[1]
768// r.z = 0
769// r.w = 0
770OZZ_INLINE SimdInt4 Load2Ptr(const int* _i);
771
772// Loads the 3 first value of _i to the x, y and z components of the returned
773// vector. The remaining components are set to 0.
774// _f must be aligned to 16 bytes.
775// r.x = _i[0]
776// r.y = _i[1]
777// r.z = _i[2]
778// r.w = 0
779OZZ_INLINE SimdInt4 Load3Ptr(const int* _i);
780
781// Loads the 4 values of _f to the returned vector.
782// _i must be aligned to 16 bytes.
783// r.x = _i[0]
784// r.y = _i[1]
785// r.z = _i[2]
786// r.w = _i[3]
787OZZ_INLINE SimdInt4 LoadPtrU(const int* _i);
788
789// Loads _i[0] to the x component of the returned vector, and sets y, z and w
790// to 0.
791// _f must be aligned to 4 bytes.
792// r.x = _i[0]
793// r.y = 0
794// r.z = 0
795// r.w = 0
796OZZ_INLINE SimdInt4 LoadXPtrU(const int* _i);
797
798// Loads the 4 values of _i to the returned vector.
799// _i must be aligned to 4 bytes.
800// r.x = _i[0]
801// r.y = _i[0]
802// r.z = _i[0]
803// r.w = _i[0]
804OZZ_INLINE SimdInt4 Load1PtrU(const int* _i);
805
806// Loads the 2 first value of _i to the x and y components of the returned
807// vector. The remaining components are set to 0.
808// _f must be aligned to 4 bytes.
809// r.x = _i[0]
810// r.y = _i[1]
811// r.z = 0
812// r.w = 0
813OZZ_INLINE SimdInt4 Load2PtrU(const int* _i);
814
815// Loads the 3 first value of _i to the x, y and z components of the returned
816// vector. The remaining components are set to 0.
817// _f must be aligned to 4 bytes.
818// r.x = _i[0]
819// r.y = _i[1]
820// r.z = _i[2]
821// r.w = 0
822OZZ_INLINE SimdInt4 Load3PtrU(const int* _i);
823
824// Convert from float to integer by rounding the nearest value.
825OZZ_INLINE SimdInt4 FromFloatRound(_SimdFloat4 _f);
826
827// Convert from float to integer by truncating.
828OZZ_INLINE SimdInt4 FromFloatTrunc(_SimdFloat4 _f);
829} // namespace simd_int4
830
831// Returns the x component of _v as an integer.
832OZZ_INLINE int GetX(_SimdInt4 _v);
833
834// Returns the y component of _v as a integer.
835OZZ_INLINE int GetY(_SimdInt4 _v);
836
837// Returns the z component of _v as a integer.
838OZZ_INLINE int GetZ(_SimdInt4 _v);
839
840// Returns the w component of _v as a integer.
841OZZ_INLINE int GetW(_SimdInt4 _v);
842
843// Returns _v with the x component set to x component of _i.
844OZZ_INLINE SimdInt4 SetX(_SimdInt4 _v, _SimdInt4 _i);
845
846// Returns _v with the y component set to x component of _i.
847OZZ_INLINE SimdInt4 SetY(_SimdInt4 _v, _SimdInt4 _i);
848
849// Returns _v with the z component set to x component of _i.
850OZZ_INLINE SimdInt4 SetZ(_SimdInt4 _v, _SimdInt4 _i);
851
852// Returns _v with the w component set to x component of _i.
853OZZ_INLINE SimdInt4 SetW(_SimdInt4 _v, _SimdInt4 _i);
854
855// Returns _v with the _ith component set to _i.
856// _i must be in range [0,3]
857OZZ_INLINE SimdInt4 SetI(_SimdInt4 _v, _SimdInt4 _i, int _ith);
858
859// Stores the 4 components of _v to the four first integers of _i.
860// _i must be aligned to 16 bytes.
861// _i[0] = _v.x
862// _i[1] = _v.y
863// _i[2] = _v.z
864// _i[3] = _v.w
865OZZ_INLINE void StorePtr(_SimdInt4 _v, int* _i);
866
867// Stores the x component of _v to the first integers of _i.
868// _i must be aligned to 16 bytes.
869// _i[0] = _v.x
870OZZ_INLINE void Store1Ptr(_SimdInt4 _v, int* _i);
871
872// Stores x and y components of _v to the two first integers of _i.
873// _i must be aligned to 16 bytes.
874// _i[0] = _v.x
875// _i[1] = _v.y
876OZZ_INLINE void Store2Ptr(_SimdInt4 _v, int* _i);
877
878// Stores x, y and z components of _v to the three first integers of _i.
879// _i must be aligned to 16 bytes.
880// _i[0] = _v.x
881// _i[1] = _v.y
882// _i[2] = _v.z
883OZZ_INLINE void Store3Ptr(_SimdInt4 _v, int* _i);
884
885// Stores the 4 components of _v to the four first integers of _i.
886// _i must be aligned to 4 bytes.
887// _i[0] = _v.x
888// _i[1] = _v.y
889// _i[2] = _v.z
890// _i[3] = _v.w
891OZZ_INLINE void StorePtrU(_SimdInt4 _v, int* _i);
892
893// Stores the x component of _v to the first float of _i.
894// _i must be aligned to 4 bytes.
895// _i[0] = _v.x
896OZZ_INLINE void Store1PtrU(_SimdInt4 _v, int* _i);
897
898// Stores x and y components of _v to the two first integers of _i.
899// _i must be aligned to 4 bytes.
900// _i[0] = _v.x
901// _i[1] = _v.y
902OZZ_INLINE void Store2PtrU(_SimdInt4 _v, int* _i);
903
904// Stores x, y and z components of _v to the three first integers of _i.
905// _i must be aligned to 4 bytes.
906// _i[0] = _v.x
907// _i[1] = _v.y
908// _i[2] = _v.z
909OZZ_INLINE void Store3PtrU(_SimdInt4 _v, int* _i);
910
911// Replicates x of _a to all the components of the returned vector.
912OZZ_INLINE SimdInt4 SplatX(_SimdInt4 _v);
913
914// Replicates y of _a to all the components of the returned vector.
915OZZ_INLINE SimdInt4 SplatY(_SimdInt4 _v);
916
917// Replicates z of _a to all the components of the returned vector.
918OZZ_INLINE SimdInt4 SplatZ(_SimdInt4 _v);
919
920// Replicates w of _a to all the components of the returned vector.
921OZZ_INLINE SimdInt4 SplatW(_SimdInt4 _v);
922
923// Swizzle x, y, z and w components based on compile time arguments _X, _Y, _Z
924// and _W. Arguments can vary from 0 (x), to 3 (w).
925template <size_t _X, size_t _Y, size_t _Z, size_t _W>
926OZZ_INLINE SimdInt4 Swizzle(_SimdInt4 _v);
927
928// Creates a 4-bit mask from the most significant bits of each component of _v.
929// i := sign(a3)<<3 | sign(a2)<<2 | sign(a1)<<1 | sign(a0)
930OZZ_INLINE int MoveMask(_SimdInt4 _v);
931
932// Returns true if all the components of _v are not 0.
933OZZ_INLINE bool AreAllTrue(_SimdInt4 _v);
934
935// Returns true if x, y and z components of _v are not 0.
936OZZ_INLINE bool AreAllTrue3(_SimdInt4 _v);
937
938// Returns true if x and y components of _v are not 0.
939OZZ_INLINE bool AreAllTrue2(_SimdInt4 _v);
940
941// Returns true if x component of _v is not 0.
942OZZ_INLINE bool AreAllTrue1(_SimdInt4 _v);
943
944// Returns true if all the components of _v are 0.
945OZZ_INLINE bool AreAllFalse(_SimdInt4 _v);
946
947// Returns true if x, y and z components of _v are 0.
948OZZ_INLINE bool AreAllFalse3(_SimdInt4 _v);
949
950// Returns true if x and y components of _v are 0.
951OZZ_INLINE bool AreAllFalse2(_SimdInt4 _v);
952
953// Returns true if x component of _v is 0.
954OZZ_INLINE bool AreAllFalse1(_SimdInt4 _v);
955
956// Computes the (horizontal) addition of x and y components of _v. The result is
957// stored in the x component of the returned value. y, z, w of the returned
958// vector are the same as their respective components in _v.
959// r.x = _a.x + _a.y
960// r.y = _a.y
961// r.z = _a.z
962// r.w = _a.w
963OZZ_INLINE SimdInt4 HAdd2(_SimdInt4 _v);
964
965// Computes the (horizontal) addition of x, y and z components of _v. The result
966// is stored in the x component of the returned value. y, z, w of the returned
967// vector are the same as their respective components in _v.
968// r.x = _a.x + _a.y + _a.z
969// r.y = _a.y
970// r.z = _a.z
971// r.w = _a.w
972OZZ_INLINE SimdInt4 HAdd3(_SimdInt4 _v);
973
974// Computes the (horizontal) addition of x and y components of _v. The result is
975// stored in the x component of the returned value. y, z, w of the returned
976// vector are the same as their respective components in _v.
977// r.x = _a.x + _a.y + _a.z + _a.w
978// r.y = _a.y
979// r.z = _a.z
980// r.w = _a.w
981OZZ_INLINE SimdInt4 HAdd4(_SimdInt4 _v);
982
983// Returns the per element absolute value of _v.
984OZZ_INLINE SimdInt4 Abs(_SimdInt4 _v);
985
986// Returns the sign bit of _v.
987OZZ_INLINE SimdInt4 Sign(_SimdInt4 _v);
988
989// Returns the per component minimum of _a and _b.
990OZZ_INLINE SimdInt4 Min(_SimdInt4 _a, _SimdInt4 _b);
991
992// Returns the per component maximum of _a and _b.
993OZZ_INLINE SimdInt4 Max(_SimdInt4 _a, _SimdInt4 _b);
994
995// Returns the per component minimum of _v and 0.
996OZZ_INLINE SimdInt4 Min0(_SimdInt4 _v);
997
998// Returns the per component maximum of _v and 0.
999OZZ_INLINE SimdInt4 Max0(_SimdInt4 _v);
1000
1001// Clamps each element of _x between _a and _b.
1002// Result is unknown if _a is not less or equal to _b.
1003OZZ_INLINE SimdInt4 Clamp(_SimdInt4 _a, _SimdInt4 _v, _SimdInt4 _b);
1004
1005// Returns boolean selection of vectors _true and _false according to consition
1006// _b. All bits a each component of _b must have the same value (O or
1007// 0xffffffff) to ensure portability.
1008OZZ_INLINE SimdInt4 Select(_SimdInt4 _b, _SimdInt4 _true, _SimdInt4 _false);
1009
1010// Returns per element binary and operation of _a and _b.
1011// _v[0...127] = _a[0...127] & _b[0...127]
1012OZZ_INLINE SimdInt4 And(_SimdInt4 _a, _SimdInt4 _b);
1013
1014// Returns per element binary and operation of _a and ~_b.
1015// _v[0...127] = _a[0...127] & ~_b[0...127]
1016OZZ_INLINE SimdInt4 AndNot(_SimdInt4 _a, _SimdInt4 _b);
1017
1018// Returns per element binary or operation of _a and _b.
1019// _v[0...127] = _a[0...127] | _b[0...127]
1020OZZ_INLINE SimdInt4 Or(_SimdInt4 _a, _SimdInt4 _b);
1021
1022// Returns per element binary logical xor operation of _a and _b.
1023// _v[0...127] = _a[0...127] ^ _b[0...127]
1024OZZ_INLINE SimdInt4 Xor(_SimdInt4 _a, _SimdInt4 _b);
1025
1026// Returns per element binary complement of _v.
1027// _v[0...127] = ~_b[0...127]
1028OZZ_INLINE SimdInt4 Not(_SimdInt4 _v);
1029
1030// Shifts the 4 signed or unsigned 32-bit integers in a left by count _bits
1031// while shifting in zeros.
1032OZZ_INLINE SimdInt4 ShiftL(_SimdInt4 _v, int _bits);
1033
1034// Shifts the 4 signed 32-bit integers in a right by count bits while shifting
1035// in the sign bit.
1036OZZ_INLINE SimdInt4 ShiftR(_SimdInt4 _v, int _bits);
1037
1038// Shifts the 4 signed or unsigned 32-bit integers in a right by count bits
1039// while shifting in zeros.
1040OZZ_INLINE SimdInt4 ShiftRu(_SimdInt4 _v, int _bits);
1041
1042// Per element "equal" comparison of _a and _b.
1043OZZ_INLINE SimdInt4 CmpEq(_SimdInt4 _a, _SimdInt4 _b);
1044
1045// Per element "not equal" comparison of _a and _b.
1046OZZ_INLINE SimdInt4 CmpNe(_SimdInt4 _a, _SimdInt4 _b);
1047
1048// Per element "less than" comparison of _a and _b.
1049OZZ_INLINE SimdInt4 CmpLt(_SimdInt4 _a, _SimdInt4 _b);
1050
1051// Per element "less than or equal" comparison of _a and _b.
1052OZZ_INLINE SimdInt4 CmpLe(_SimdInt4 _a, _SimdInt4 _b);
1053
1054// Per element "greater than" comparison of _a and _b.
1055OZZ_INLINE SimdInt4 CmpGt(_SimdInt4 _a, _SimdInt4 _b);
1056
1057// Per element "greater than or equal" comparison of _a and _b.
1058OZZ_INLINE SimdInt4 CmpGe(_SimdInt4 _a, _SimdInt4 _b);
1059
1060// Declare the 4x4 matrix type. Uses the column major convention where the
1061// matrix-times-vector is written v'=Mv:
1062// [ m.cols[0].x m.cols[1].x m.cols[2].x m.cols[3].x ] {v.x}
1063// | m.cols[0].y m.cols[1].y m.cols[2].y m.cols[3].y | * {v.y}
1064// | m.cols[0].z m.cols[1].y m.cols[2].y m.cols[3].y | {v.z}
1065// [ m.cols[0].w m.cols[1].w m.cols[2].w m.cols[3].w ] {v.1}
1066struct Float4x4 {
1067 // Matrix columns.
1068 SimdFloat4 cols[4];
1069
1070 // Returns the identity matrix.
1071 static OZZ_INLINE Float4x4 identity();
1072
1073 // Returns a translation matrix.
1074 // _v.w is ignored.
1075 static OZZ_INLINE Float4x4 Translation(_SimdFloat4 _v);
1076
1077 // Returns a scaling matrix that scales along _v.
1078 // _v.w is ignored.
1079 static OZZ_INLINE Float4x4 Scaling(_SimdFloat4 _v);
1080
1081 // Returns the rotation matrix built from Euler angles defined by x, y and z
1082 // components of _v. Euler angles are ordered Heading, Elevation and Bank, or
1083 // Yaw, Pitch and Roll. _v.w is ignored.
1084 static OZZ_INLINE Float4x4 FromEuler(_SimdFloat4 _v);
1085
1086 // Returns the rotation matrix built from axis defined by _axis.xyz and
1087 // _angle.x
1088 static OZZ_INLINE Float4x4 FromAxisAngle(_SimdFloat4 _axis,
1089 _SimdFloat4 _angle);
1090
1091 // Returns the rotation matrix built from quaternion defined by x, y, z and w
1092 // components of _v.
1093 static OZZ_INLINE Float4x4 FromQuaternion(_SimdFloat4 _v);
1094
1095 // Returns the affine transformation matrix built from split translation,
1096 // rotation (quaternion) and scale.
1097 static OZZ_INLINE Float4x4 FromAffine(_SimdFloat4 _translation,
1098 _SimdFloat4 _quaternion,
1099 _SimdFloat4 _scale);
1100};
1101
1102// Returns the transpose of matrix _m.
1103OZZ_INLINE Float4x4 Transpose(const Float4x4& _m);
1104
1105// Returns the inverse of matrix _m.
1106// If _invertible is not nullptr, its x component will be set to true if matrix is
1107// invertible. If _invertible is nullptr, then an assert is triggered in case the
1108// matrix isn't invertible.
1109OZZ_INLINE Float4x4 Invert(const Float4x4& _m, SimdInt4* _invertible = nullptr);
1110
1111// Translates matrix _m along the axis defined by _v components.
1112// _v.w is ignored.
1113OZZ_INLINE Float4x4 Translate(const Float4x4& _m, _SimdFloat4 _v);
1114
1115// Scales matrix _m along each axis with x, y, z components of _v.
1116// _v.w is ignored.
1117OZZ_INLINE Float4x4 Scale(const Float4x4& _m, _SimdFloat4 _v);
1118
1119// Multiply each column of matrix _m with vector _v.
1120OZZ_INLINE Float4x4 ColumnMultiply(const Float4x4& _m, _SimdFloat4 _v);
1121
1122// Tests if each 3 column of upper 3x3 matrix of _m is a normal matrix.
1123// Returns the result in the x, y and z component of the returned vector. w is
1124// set to 0.
1125OZZ_INLINE SimdInt4 IsNormalized(const Float4x4& _m);
1126
1127// Tests if each 3 column of upper 3x3 matrix of _m is a normal matrix.
1128// Uses the estimated tolerance
1129// Returns the result in the x, y and z component of the returned vector. w is
1130// set to 0.
1131OZZ_INLINE SimdInt4 IsNormalizedEst(const Float4x4& _m);
1132
1133// Tests if the upper 3x3 matrix of _m is an orthogonal matrix.
1134// A matrix that contains a reflexion cannot be considered orthogonal.
1135// Returns the result in the x component of the returned vector. y, z and w are
1136// set to 0.
1137OZZ_INLINE SimdInt4 IsOrthogonal(const Float4x4& _m);
1138
1139// Returns the quaternion that represent the rotation of matrix _m.
1140// _m must be normalized and orthogonal.
1141// the return quaternion is normalized.
1142OZZ_INLINE SimdFloat4 ToQuaternion(const Float4x4& _m);
1143
1144// Decompose a general 3D transformation matrix _m into its scalar, rotational
1145// and translational components.
1146// Returns false if it was not possible to decompose the matrix. This would be
1147// because more than 1 of the 3 first column of _m are scaled to 0.
1148OZZ_INLINE bool ToAffine(const Float4x4& _m, SimdFloat4* _translation,
1149 SimdFloat4* _quaternion, SimdFloat4* _scale);
1150
1151// Computes the transformation of a Float4x4 matrix and a point _p.
1152// This is equivalent to multiplying a matrix by a SimdFloat4 with a w component
1153// of 1.
1154OZZ_INLINE ozz::math::SimdFloat4 TransformPoint(const ozz::math::Float4x4& _m,
1156
1157// Computes the transformation of a Float4x4 matrix and a vector _v.
1158// This is equivalent to multiplying a matrix by a SimdFloat4 with a w component
1159// of 0.
1160OZZ_INLINE ozz::math::SimdFloat4 TransformVector(const ozz::math::Float4x4& _m,
1162
1163// Computes the multiplication of matrix Float4x4 and vector _v.
1164OZZ_INLINE ozz::math::SimdFloat4 operator*(const ozz::math::Float4x4& _m,
1166
1167// Computes the multiplication of two matrices _a and _b.
1168OZZ_INLINE ozz::math::Float4x4 operator*(const ozz::math::Float4x4& _a,
1169 const ozz::math::Float4x4& _b);
1170
1171// Computes the per element addition of two matrices _a and _b.
1172OZZ_INLINE ozz::math::Float4x4 operator+(const ozz::math::Float4x4& _a,
1173 const ozz::math::Float4x4& _b);
1174
1175// Computes the per element subtraction of two matrices _a and _b.
1176OZZ_INLINE ozz::math::Float4x4 operator-(const ozz::math::Float4x4& _a,
1177 const ozz::math::Float4x4& _b);
1178} // namespace math
1179} // namespace ozz
1180
1181#if !defined(OZZ_DISABLE_SSE_NATIVE_OPERATORS)
1182// Returns per element addition of _a and _b.
1183OZZ_INLINE ozz::math::SimdFloat4 operator+(ozz::math::_SimdFloat4 _a,
1185
1186// Returns per element subtraction of _a and _b.
1187OZZ_INLINE ozz::math::SimdFloat4 operator-(ozz::math::_SimdFloat4 _a,
1189
1190// Returns per element negation of _v.
1191OZZ_INLINE ozz::math::SimdFloat4 operator-(ozz::math::_SimdFloat4 _v);
1192
1193// Returns per element multiplication of _a and _b.
1194OZZ_INLINE ozz::math::SimdFloat4 operator*(ozz::math::_SimdFloat4 _a,
1196
1197// Returns per element division of _a and _b.
1198OZZ_INLINE ozz::math::SimdFloat4 operator/(ozz::math::_SimdFloat4 _a,
1200#endif // !defined(OZZ_DISABLE_SSE_NATIVE_OPERATORS)
1201
1202// Implement format conversions.
1203namespace ozz {
1204namespace math {
1205// Converts from a float to a half.
1206OZZ_INLINE uint16_t FloatToHalf(float _f);
1207
1208// Converts from a half to a float.
1209OZZ_INLINE float HalfToFloat(uint16_t _h);
1210
1211// Converts from a float to a half.
1212OZZ_INLINE SimdInt4 FloatToHalf(_SimdFloat4 _f);
1213
1214// Converts from a half to a float.
1215OZZ_INLINE SimdFloat4 HalfToFloat(_SimdInt4 _h);
1216} // namespace math
1217} // namespace ozz
1218
1219#if defined(OZZ_SIMD_SSEx)
1220#include "ozz/base/maths/internal/simd_math_sse-inl.h"
1221#elif defined(OZZ_SIMD_REF)
1222#include "ozz/base/maths/internal/simd_math_ref-inl.h"
1223#else
1224#error No simd_math implementation detected
1225#endif
1226#endif // OZZ_OZZ_BASE_MATHS_SIMD_MATH_H_
GLM_FUNC_DECL GLM_CONSTEXPR genType zero()
Definition constants.inl:6
Definition simd_math_config.h:121
Definition simd_math_config.h:129
Definition simd_math.h:1066