RavEngine
Loading...
Searching...
No Matches
PxVecMath.h
1// Redistribution and use in source and binary forms, with or without
2// modification, are permitted provided that the following conditions
3// are met:
4// * Redistributions of source code must retain the above copyright
5// notice, this list of conditions and the following disclaimer.
6// * Redistributions in binary form must reproduce the above copyright
7// notice, this list of conditions and the following disclaimer in the
8// documentation and/or other materials provided with the distribution.
9// * Neither the name of NVIDIA CORPORATION nor the names of its
10// contributors may be used to endorse or promote products derived
11// from this software without specific prior written permission.
12//
13// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ''AS IS'' AND ANY
14// EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
15// IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
16// PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
17// CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
18// EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
19// PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
20// PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
21// OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
22// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
23// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
24//
25// Copyright (c) 2008-2022 NVIDIA Corporation. All rights reserved.
26// Copyright (c) 2004-2008 AGEIA Technologies, Inc. All rights reserved.
27// Copyright (c) 2001-2004 NovodeX AG. All rights reserved.
28
29#ifndef PX_VEC_MATH_H
30#define PX_VEC_MATH_H
31
32#include "foundation/Px.h"
33#include "foundation/PxIntrinsics.h"
34#include "foundation/PxVec3.h"
35#include "foundation/PxVec4.h"
36#include "foundation/PxMat33.h"
37#include "foundation/PxUnionCast.h"
38
39// We can opt to use the scalar version of vectorised functions.
40// This can catch type safety issues and might even work out more optimal on pc.
41// It will also be useful for benchmarking and testing.
42// NEVER submit with vector intrinsics deactivated without good reason.
43// AM: deactivating SIMD for debug win64 just so autobuild will also exercise
44// non-SIMD path, until a dedicated non-SIMD platform sich as Arm comes online.
45// TODO: dima: reference all platforms with SIMD support here,
46// all unknown/experimental cases should better default to NO SIMD.
47
48// enable/disable SIMD
49#if !defined(PX_SIMD_DISABLED)
50#if PX_INTEL_FAMILY && (!defined(__EMSCRIPTEN__) || defined(__SSE2__))
51 #define COMPILE_VECTOR_INTRINSICS 1
52#elif PX_SWITCH
53 #define COMPILE_VECTOR_INTRINSICS 1
54#else
55 #define COMPILE_VECTOR_INTRINSICS 0
56#endif
57#else
58 #define COMPILE_VECTOR_INTRINSICS 0
59#endif
60
61#if COMPILE_VECTOR_INTRINSICS && PX_INTEL_FAMILY && PX_UNIX_FAMILY
62// only SSE2 compatible platforms should reach this
63#if PX_EMSCRIPTEN
64 #include <emmintrin.h>
65#endif
66 #include <xmmintrin.h>
67#endif
68
69#if COMPILE_VECTOR_INTRINSICS
70 #include "PxAoS.h"
71#else
72 #include "PxVecMathAoSScalar.h"
73#endif
74
75#if !PX_DOXYGEN
76namespace physx
77{
78#endif
79namespace aos
80{
81
82// Basic AoS types are
83// FloatV - 16-byte aligned representation of float.
84// Vec3V - 16-byte aligned representation of PxVec3 stored as (x y z 0).
85// Vec4V - 16-byte aligned representation of vector of 4 floats stored as (x y z w).
86// BoolV - 16-byte aligned representation of vector of 4 bools stored as (x y z w).
87// VecU32V - 16-byte aligned representation of 4 unsigned ints stored as (x y z w).
88// VecI32V - 16-byte aligned representation of 4 signed ints stored as (x y z w).
89// Mat33V - 16-byte aligned representation of any 3x3 matrix.
90// Mat34V - 16-byte aligned representation of transformation matrix (rotation in col1,col2,col3 and translation in
91// col4).
92// Mat44V - 16-byte aligned representation of any 4x4 matrix.
93
95// Construct a simd type from a scalar type
97
98// FloatV
99//(f,f,f,f)
100PX_FORCE_INLINE FloatV FLoad(const PxF32 f);
101
102// Vec3V
103//(f,f,f,0)
104PX_FORCE_INLINE Vec3V V3Load(const PxF32 f);
105//(f.x,f.y,f.z,0)
106PX_FORCE_INLINE Vec3V V3LoadU(const PxVec3& f);
107//(f.x,f.y,f.z,0), f must be 16-byte aligned
108PX_FORCE_INLINE Vec3V V3LoadA(const PxVec3& f);
109//(f.x,f.y,f.z,w_undefined), f must be 16-byte aligned
110PX_FORCE_INLINE Vec3V V3LoadUnsafeA(const PxVec3& f);
111//(f.x,f.y,f.z,0)
112PX_FORCE_INLINE Vec3V V3LoadU(const PxF32* f);
113//(f.x,f.y,f.z,0), f must be 16-byte aligned
114PX_FORCE_INLINE Vec3V V3LoadA(const PxF32* f);
115
116// Vec4V
117//(f,f,f,f)
118PX_FORCE_INLINE Vec4V V4Load(const PxF32 f);
119//(f[0],f[1],f[2],f[3])
120PX_FORCE_INLINE Vec4V V4LoadU(const PxF32* const f);
121//(f[0],f[1],f[2],f[3]), f must be 16-byte aligned
122PX_FORCE_INLINE Vec4V V4LoadA(const PxF32* const f);
123//(x,y,z,w)
124PX_FORCE_INLINE Vec4V V4LoadXYZW(const PxF32& x, const PxF32& y, const PxF32& z, const PxF32& w);
125
126// BoolV
127//(f,f,f,f)
128PX_FORCE_INLINE BoolV BLoad(const bool f);
129//(f[0],f[1],f[2],f[3])
130PX_FORCE_INLINE BoolV BLoad(const bool* const f);
131
132// VecU32V
133//(f,f,f,f)
134PX_FORCE_INLINE VecU32V U4Load(const PxU32 f);
135//(f[0],f[1],f[2],f[3])
136PX_FORCE_INLINE VecU32V U4LoadU(const PxU32* f);
137//(f[0],f[1],f[2],f[3]), f must be 16-byte aligned
138PX_FORCE_INLINE VecU32V U4LoadA(const PxU32* f);
139//((U32)x, (U32)y, (U32)z, (U32)w)
140PX_FORCE_INLINE VecU32V U4LoadXYZW(PxU32 x, PxU32 y, PxU32 z, PxU32 w);
141
142// VecI32V
143//(i,i,i,i)
144PX_FORCE_INLINE VecI32V I4Load(const PxI32 i);
145//(i,i,i,i)
146PX_FORCE_INLINE VecI32V I4LoadU(const PxI32* i);
147//(i,i,i,i)
148PX_FORCE_INLINE VecI32V I4LoadA(const PxI32* i);
149
150// QuatV
151//(x = v[0], y=v[1], z=v[2], w=v3[3]) and array don't need to aligned
152PX_FORCE_INLINE QuatV QuatVLoadU(const PxF32* v);
153//(x = v[0], y=v[1], z=v[2], w=v3[3]) and array need to aligned, fast load
154PX_FORCE_INLINE QuatV QuatVLoadA(const PxF32* v);
155//(x, y, z, w)
156PX_FORCE_INLINE QuatV QuatVLoadXYZW(const PxF32 x, const PxF32 y, const PxF32 z, const PxF32 w);
157
158// not added to public api
159Vec4V Vec4V_From_PxVec3_WUndefined(const PxVec3& v);
160
162// Construct a simd type from a different simd type
164
165// Vec3V
166//(v.x,v.y,v.z,0)
167PX_FORCE_INLINE Vec3V Vec3V_From_Vec4V(Vec4V v);
168//(v.x,v.y,v.z,undefined) - be very careful with w!=0 because many functions require w==0 for correct operation eg V3Dot, V3Length, V3Cross etc etc.
169PX_FORCE_INLINE Vec3V Vec3V_From_Vec4V_WUndefined(const Vec4V v);
170
171// Vec4V
172//(f.x,f.y,f.z,f.w)
173PX_FORCE_INLINE Vec4V Vec4V_From_Vec3V(Vec3V f);
174//((PxF32)f.x, (PxF32)f.y, (PxF32)f.z, (PxF32)f.w)
175PX_FORCE_INLINE Vec4V Vec4V_From_VecU32V(VecU32V a);
176//((PxF32)f.x, (PxF32)f.y, (PxF32)f.z, (PxF32)f.w)
177PX_FORCE_INLINE Vec4V Vec4V_From_VecI32V(VecI32V a);
178//(*(reinterpret_cast<PxF32*>(&f.x), (reinterpret_cast<PxF32*>(&f.y), (reinterpret_cast<PxF32*>(&f.z),
179//(reinterpret_cast<PxF32*>(&f.w))
180PX_FORCE_INLINE Vec4V Vec4V_ReinterpretFrom_VecU32V(VecU32V a);
181//(*(reinterpret_cast<PxF32*>(&f.x), (reinterpret_cast<PxF32*>(&f.y), (reinterpret_cast<PxF32*>(&f.z),
182//(reinterpret_cast<PxF32*>(&f.w))
183PX_FORCE_INLINE Vec4V Vec4V_ReinterpretFrom_VecI32V(VecI32V a);
184
185// VecU32V
186//(*(reinterpret_cast<PxU32*>(&f.x), (reinterpret_cast<PxU32*>(&f.y), (reinterpret_cast<PxU32*>(&f.z),
187//(reinterpret_cast<PxU32*>(&f.w))
188PX_FORCE_INLINE VecU32V VecU32V_ReinterpretFrom_Vec4V(Vec4V a);
189//(b[0], b[1], b[2], b[3])
190PX_FORCE_INLINE VecU32V VecU32V_From_BoolV(const BoolVArg b);
191
192// VecI32V
193//(*(reinterpret_cast<PxI32*>(&f.x), (reinterpret_cast<PxI32*>(&f.y), (reinterpret_cast<PxI32*>(&f.z),
194//(reinterpret_cast<PxI32*>(&f.w))
195PX_FORCE_INLINE VecI32V VecI32V_ReinterpretFrom_Vec4V(Vec4V a);
196//((I32)a.x, (I32)a.y, (I32)a.z, (I32)a.w)
197PX_FORCE_INLINE VecI32V VecI32V_From_Vec4V(Vec4V a);
198//((I32)b.x, (I32)b.y, (I32)b.z, (I32)b.w)
199PX_FORCE_INLINE VecI32V VecI32V_From_BoolV(const BoolVArg b);
200
202// Convert from a simd type back to a scalar type
204
205// FloatV
206// a.x
207PX_FORCE_INLINE void FStore(const FloatV a, PxF32* PX_RESTRICT f);
208
209// Vec3V
210//(a.x,a.y,a.z)
211PX_FORCE_INLINE void V3StoreA(const Vec3V a, PxVec3& f);
212//(a.x,a.y,a.z)
213PX_FORCE_INLINE void V3StoreU(const Vec3V a, PxVec3& f);
214
215// Vec4V
216PX_FORCE_INLINE void V4StoreA(const Vec4V a, PxF32* f);
217PX_FORCE_INLINE void V4StoreU(const Vec4V a, PxF32* f);
218
219// BoolV
220PX_FORCE_INLINE void BStoreA(const BoolV b, PxU32* f);
221
222// VecU32V
223PX_FORCE_INLINE void U4StoreA(const VecU32V uv, PxU32* u);
224
225// VecI32V
226PX_FORCE_INLINE void I4StoreA(const VecI32V iv, PxI32* i);
227
229// Test that simd types have elements in the floating point range
231
232// check for each component is valid ie in floating point range
233PX_FORCE_INLINE bool isFiniteFloatV(const FloatV a);
234// check for each component is valid ie in floating point range
235PX_FORCE_INLINE bool isFiniteVec3V(const Vec3V a);
236// check for each component is valid ie in floating point range
237PX_FORCE_INLINE bool isFiniteVec4V(const Vec4V a);
238
239// Check that w-component is zero.
240PX_FORCE_INLINE bool isValidVec3V(const Vec3V a);
241
243// Tests that all elements of two 16-byte types are completely equivalent.
244// Use these tests for unit testing and asserts only.
246
247namespace vecMathTests
248{
249PX_FORCE_INLINE Vec3V getInvalidVec3V();
250PX_FORCE_INLINE bool allElementsEqualFloatV(const FloatV a, const FloatV b);
251PX_FORCE_INLINE bool allElementsEqualVec3V(const Vec3V a, const Vec3V b);
252PX_FORCE_INLINE bool allElementsEqualVec4V(const Vec4V a, const Vec4V b);
253PX_FORCE_INLINE bool allElementsEqualBoolV(const BoolV a, const BoolV b);
254PX_FORCE_INLINE bool allElementsEqualVecU32V(const VecU32V a, const VecU32V b);
255PX_FORCE_INLINE bool allElementsEqualVecI32V(const VecI32V a, const VecI32V b);
256
257PX_FORCE_INLINE bool allElementsEqualMat33V(const Mat33V& a, const Mat33V& b)
258{
259 return (allElementsEqualVec3V(a.col0, b.col0) && allElementsEqualVec3V(a.col1, b.col1) &&
260 allElementsEqualVec3V(a.col2, b.col2));
261}
262PX_FORCE_INLINE bool allElementsEqualMat34V(const Mat34V& a, const Mat34V& b)
263{
264 return (allElementsEqualVec3V(a.col0, b.col0) && allElementsEqualVec3V(a.col1, b.col1) &&
265 allElementsEqualVec3V(a.col2, b.col2) && allElementsEqualVec3V(a.col3, b.col3));
266}
267PX_FORCE_INLINE bool allElementsEqualMat44V(const Mat44V& a, const Mat44V& b)
268{
269 return (allElementsEqualVec4V(a.col0, b.col0) && allElementsEqualVec4V(a.col1, b.col1) &&
270 allElementsEqualVec4V(a.col2, b.col2) && allElementsEqualVec4V(a.col3, b.col3));
271}
272
273PX_FORCE_INLINE bool allElementsNearEqualFloatV(const FloatV a, const FloatV b);
274PX_FORCE_INLINE bool allElementsNearEqualVec3V(const Vec3V a, const Vec3V b);
275PX_FORCE_INLINE bool allElementsNearEqualVec4V(const Vec4V a, const Vec4V b);
276PX_FORCE_INLINE bool allElementsNearEqualMat33V(const Mat33V& a, const Mat33V& b)
277{
278 return (allElementsNearEqualVec3V(a.col0, b.col0) && allElementsNearEqualVec3V(a.col1, b.col1) &&
279 allElementsNearEqualVec3V(a.col2, b.col2));
280}
281PX_FORCE_INLINE bool allElementsNearEqualMat34V(const Mat34V& a, const Mat34V& b)
282{
283 return (allElementsNearEqualVec3V(a.col0, b.col0) && allElementsNearEqualVec3V(a.col1, b.col1) &&
284 allElementsNearEqualVec3V(a.col2, b.col2) && allElementsNearEqualVec3V(a.col3, b.col3));
285}
286PX_FORCE_INLINE bool allElementsNearEqualMat44V(const Mat44V& a, const Mat44V& b)
287{
288 return (allElementsNearEqualVec4V(a.col0, b.col0) && allElementsNearEqualVec4V(a.col1, b.col1) &&
289 allElementsNearEqualVec4V(a.col2, b.col2) && allElementsNearEqualVec4V(a.col3, b.col3));
290}
291}
292
294// Math operations on FloatV
296
297//(0,0,0,0)
298PX_FORCE_INLINE FloatV FZero();
299//(1,1,1,1)
300PX_FORCE_INLINE FloatV FOne();
301//(0.5,0.5,0.5,0.5)
302PX_FORCE_INLINE FloatV FHalf();
303//(PX_EPS_REAL,PX_EPS_REAL,PX_EPS_REAL,PX_EPS_REAL)
304PX_FORCE_INLINE FloatV FEps();
306//(PX_MAX_REAL, PX_MAX_REAL, PX_MAX_REAL PX_MAX_REAL)
307PX_FORCE_INLINE FloatV FMax();
309//(-PX_MAX_REAL, -PX_MAX_REAL, -PX_MAX_REAL -PX_MAX_REAL)
310PX_FORCE_INLINE FloatV FNegMax();
311//(1e-6f, 1e-6f, 1e-6f, 1e-6f)
312PX_FORCE_INLINE FloatV FEps6();
313//((PxF32*)&1, (PxF32*)&1, (PxF32*)&1, (PxF32*)&1)
314
315//-f (per component)
316PX_FORCE_INLINE FloatV FNeg(const FloatV f);
317// a+b (per component)
318PX_FORCE_INLINE FloatV FAdd(const FloatV a, const FloatV b);
319// a-b (per component)
320PX_FORCE_INLINE FloatV FSub(const FloatV a, const FloatV b);
321// a*b (per component)
322PX_FORCE_INLINE FloatV FMul(const FloatV a, const FloatV b);
323// a/b (per component)
324PX_FORCE_INLINE FloatV FDiv(const FloatV a, const FloatV b);
325// a/b (per component)
326PX_FORCE_INLINE FloatV FDivFast(const FloatV a, const FloatV b);
327// 1.0f/a
328PX_FORCE_INLINE FloatV FRecip(const FloatV a);
329// 1.0f/a
330PX_FORCE_INLINE FloatV FRecipFast(const FloatV a);
331// 1.0f/sqrt(a)
332PX_FORCE_INLINE FloatV FRsqrt(const FloatV a);
333// 1.0f/sqrt(a)
334PX_FORCE_INLINE FloatV FRsqrtFast(const FloatV a);
335// sqrt(a)
336PX_FORCE_INLINE FloatV FSqrt(const FloatV a);
337// a*b+c
338PX_FORCE_INLINE FloatV FScaleAdd(const FloatV a, const FloatV b, const FloatV c);
339// c-a*b
340PX_FORCE_INLINE FloatV FNegScaleSub(const FloatV a, const FloatV b, const FloatV c);
341// fabs(a)
342PX_FORCE_INLINE FloatV FAbs(const FloatV a);
343// c ? a : b (per component)
344PX_FORCE_INLINE FloatV FSel(const BoolV c, const FloatV a, const FloatV b);
345// a>b (per component)
346PX_FORCE_INLINE BoolV FIsGrtr(const FloatV a, const FloatV b);
347// a>=b (per component)
348PX_FORCE_INLINE BoolV FIsGrtrOrEq(const FloatV a, const FloatV b);
349// a==b (per component)
350PX_FORCE_INLINE BoolV FIsEq(const FloatV a, const FloatV b);
351// Max(a,b) (per component)
352PX_FORCE_INLINE FloatV FMax(const FloatV a, const FloatV b);
353// Min(a,b) (per component)
354PX_FORCE_INLINE FloatV FMin(const FloatV a, const FloatV b);
355// Clamp(a,b) (per component)
356PX_FORCE_INLINE FloatV FClamp(const FloatV a, const FloatV minV, const FloatV maxV);
357
358// a.x>b.x
359PX_FORCE_INLINE PxU32 FAllGrtr(const FloatV a, const FloatV b);
360// a.x>=b.x
361PX_FORCE_INLINE PxU32 FAllGrtrOrEq(const FloatV a, const FloatV b);
362// a.x==b.x
363PX_FORCE_INLINE PxU32 FAllEq(const FloatV a, const FloatV b);
364// a<min || a>max
365PX_FORCE_INLINE PxU32 FOutOfBounds(const FloatV a, const FloatV min, const FloatV max);
366// a>=min && a<=max
367PX_FORCE_INLINE PxU32 FInBounds(const FloatV a, const FloatV min, const FloatV max);
368// a<-bounds || a>bounds
369PX_FORCE_INLINE PxU32 FOutOfBounds(const FloatV a, const FloatV bounds);
370// a>=-bounds && a<=bounds
371PX_FORCE_INLINE PxU32 FInBounds(const FloatV a, const FloatV bounds);
372
373// round float a to the near int
374PX_FORCE_INLINE FloatV FRound(const FloatV a);
375// calculate the sin of float a
376PX_FORCE_INLINE FloatV FSin(const FloatV a);
377// calculate the cos of float b
378PX_FORCE_INLINE FloatV FCos(const FloatV a);
379
381// Math operations on Vec3V
383
384//(f,f,f,f)
385PX_FORCE_INLINE Vec3V V3Splat(const FloatV f);
386
387//(x,y,z)
388PX_FORCE_INLINE Vec3V V3Merge(const FloatVArg x, const FloatVArg y, const FloatVArg z);
389
390//(1,0,0,0)
391PX_FORCE_INLINE Vec3V V3UnitX();
392//(0,1,0,0)
393PX_FORCE_INLINE Vec3V V3UnitY();
394//(0,0,1,0)
395PX_FORCE_INLINE Vec3V V3UnitZ();
396
397//(f.x,f.x,f.x,f.x)
398PX_FORCE_INLINE FloatV V3GetX(const Vec3V f);
399//(f.y,f.y,f.y,f.y)
400PX_FORCE_INLINE FloatV V3GetY(const Vec3V f);
401//(f.z,f.z,f.z,f.z)
402PX_FORCE_INLINE FloatV V3GetZ(const Vec3V f);
403
404//(f,v.y,v.z,v.w)
405PX_FORCE_INLINE Vec3V V3SetX(const Vec3V v, const FloatV f);
406//(v.x,f,v.z,v.w)
407PX_FORCE_INLINE Vec3V V3SetY(const Vec3V v, const FloatV f);
408//(v.x,v.y,f,v.w)
409PX_FORCE_INLINE Vec3V V3SetZ(const Vec3V v, const FloatV f);
410
411// v.x=f
412PX_FORCE_INLINE void V3WriteX(Vec3V& v, const PxF32 f);
413// v.y=f
414PX_FORCE_INLINE void V3WriteY(Vec3V& v, const PxF32 f);
415// v.z=f
416PX_FORCE_INLINE void V3WriteZ(Vec3V& v, const PxF32 f);
417// v.x=f.x, v.y=f.y, v.z=f.z
418PX_FORCE_INLINE void V3WriteXYZ(Vec3V& v, const PxVec3& f);
419// return v.x
420PX_FORCE_INLINE PxF32 V3ReadX(const Vec3V& v);
421// return v.y
422PX_FORCE_INLINE PxF32 V3ReadY(const Vec3V& v);
423// return v.y
424PX_FORCE_INLINE PxF32 V3ReadZ(const Vec3V& v);
425// return (v.x,v.y,v.z)
426PX_FORCE_INLINE const PxVec3& V3ReadXYZ(const Vec3V& v);
427
428//(a.x, b.x, c.x)
429PX_FORCE_INLINE Vec3V V3ColX(const Vec3V a, const Vec3V b, const Vec3V c);
430//(a.y, b.y, c.y)
431PX_FORCE_INLINE Vec3V V3ColY(const Vec3V a, const Vec3V b, const Vec3V c);
432//(a.z, b.z, c.z)
433PX_FORCE_INLINE Vec3V V3ColZ(const Vec3V a, const Vec3V b, const Vec3V c);
434
435//(0,0,0,0)
436PX_FORCE_INLINE Vec3V V3Zero();
437//(1,1,1,1)
438PX_FORCE_INLINE Vec3V V3One();
439//(PX_EPS_REAL,PX_EPS_REAL,PX_EPS_REAL,PX_EPS_REAL)
440PX_FORCE_INLINE Vec3V V3Eps();
441//-c (per component)
442PX_FORCE_INLINE Vec3V V3Neg(const Vec3V c);
443// a+b (per component)
444PX_FORCE_INLINE Vec3V V3Add(const Vec3V a, const Vec3V b);
445// a-b (per component)
446PX_FORCE_INLINE Vec3V V3Sub(const Vec3V a, const Vec3V b);
447// a*b (per component)
448PX_FORCE_INLINE Vec3V V3Scale(const Vec3V a, const FloatV b);
449// a*b (per component)
450PX_FORCE_INLINE Vec3V V3Mul(const Vec3V a, const Vec3V b);
451// a/b (per component)
452PX_FORCE_INLINE Vec3V V3ScaleInv(const Vec3V a, const FloatV b);
453// a/b (per component)
454PX_FORCE_INLINE Vec3V V3Div(const Vec3V a, const Vec3V b);
455// a/b (per component)
456PX_FORCE_INLINE Vec3V V3ScaleInvFast(const Vec3V a, const FloatV b);
457// a/b (per component)
458PX_FORCE_INLINE Vec3V V3DivFast(const Vec3V a, const Vec3V b);
459// 1.0f/a
460PX_FORCE_INLINE Vec3V V3Recip(const Vec3V a);
461// 1.0f/a
462PX_FORCE_INLINE Vec3V V3RecipFast(const Vec3V a);
463// 1.0f/sqrt(a)
464PX_FORCE_INLINE Vec3V V3Rsqrt(const Vec3V a);
465// 1.0f/sqrt(a)
466PX_FORCE_INLINE Vec3V V3RsqrtFast(const Vec3V a);
467// a*b+c
468PX_FORCE_INLINE Vec3V V3ScaleAdd(const Vec3V a, const FloatV b, const Vec3V c);
469// c-a*b
470PX_FORCE_INLINE Vec3V V3NegScaleSub(const Vec3V a, const FloatV b, const Vec3V c);
471// a*b+c
472PX_FORCE_INLINE Vec3V V3MulAdd(const Vec3V a, const Vec3V b, const Vec3V c);
473// c-a*b
474PX_FORCE_INLINE Vec3V V3NegMulSub(const Vec3V a, const Vec3V b, const Vec3V c);
475// fabs(a)
476PX_FORCE_INLINE Vec3V V3Abs(const Vec3V a);
477
478// a.b
479// Note: a.w and b.w must have value zero
480PX_FORCE_INLINE FloatV V3Dot(const Vec3V a, const Vec3V b);
481// aXb
482// Note: a.w and b.w must have value zero
483PX_FORCE_INLINE Vec3V V3Cross(const Vec3V a, const Vec3V b);
484// |a.a|^1/2
485// Note: a.w must have value zero
486PX_FORCE_INLINE FloatV V3Length(const Vec3V a);
487// a.a
488// Note: a.w must have value zero
489PX_FORCE_INLINE FloatV V3LengthSq(const Vec3V a);
490// a*|a.a|^-1/2
491// Note: a.w must have value zero
492PX_FORCE_INLINE Vec3V V3Normalize(const Vec3V a);
493// a.a>0 ? a*|a.a|^-1/2 : (0,0,0,0)
494// Note: a.w must have value zero
495PX_FORCE_INLINE FloatV V3Length(const Vec3V a);
496// a.a>0 ? a*|a.a|^-1/2 : unsafeReturnValue
497// Note: a.w must have value zero
498PX_FORCE_INLINE Vec3V V3NormalizeSafe(const Vec3V a, const Vec3V unsafeReturnValue);
499// a.x + a.y + a.z
500// Note: a.w must have value zero
501PX_FORCE_INLINE FloatV V3SumElems(const Vec3V a);
502
503// c ? a : b (per component)
504PX_FORCE_INLINE Vec3V V3Sel(const BoolV c, const Vec3V a, const Vec3V b);
505// a>b (per component)
506PX_FORCE_INLINE BoolV V3IsGrtr(const Vec3V a, const Vec3V b);
507// a>=b (per component)
508PX_FORCE_INLINE BoolV V3IsGrtrOrEq(const Vec3V a, const Vec3V b);
509// a==b (per component)
510PX_FORCE_INLINE BoolV V3IsEq(const Vec3V a, const Vec3V b);
511// Max(a,b) (per component)
512PX_FORCE_INLINE Vec3V V3Max(const Vec3V a, const Vec3V b);
513// Min(a,b) (per component)
514PX_FORCE_INLINE Vec3V V3Min(const Vec3V a, const Vec3V b);
515
516// Extract the maximum value from a
517// Note: a.w must have value zero
518PX_FORCE_INLINE FloatV V3ExtractMax(const Vec3V a);
519
520// Extract the minimum value from a
521// Note: a.w must have value zero
522PX_FORCE_INLINE FloatV V3ExtractMin(const Vec3V a);
523
524// Clamp(a,b) (per component)
525PX_FORCE_INLINE Vec3V V3Clamp(const Vec3V a, const Vec3V minV, const Vec3V maxV);
526
527// Extract the sign for each component
528PX_FORCE_INLINE Vec3V V3Sign(const Vec3V a);
529
530// Test all components.
531// (a.x>b.x && a.y>b.y && a.z>b.z)
532// Note: a.w and b.w must have value zero
533PX_FORCE_INLINE PxU32 V3AllGrtr(const Vec3V a, const Vec3V b);
534// (a.x>=b.x && a.y>=b.y && a.z>=b.z)
535// Note: a.w and b.w must have value zero
536PX_FORCE_INLINE PxU32 V3AllGrtrOrEq(const Vec3V a, const Vec3V b);
537// (a.x==b.x && a.y==b.y && a.z==b.z)
538// Note: a.w and b.w must have value zero
539PX_FORCE_INLINE PxU32 V3AllEq(const Vec3V a, const Vec3V b);
540// a.x<min.x || a.y<min.y || a.z<min.z || a.x>max.x || a.y>max.y || a.z>max.z
541// Note: a.w and min.w and max.w must have value zero
542PX_FORCE_INLINE PxU32 V3OutOfBounds(const Vec3V a, const Vec3V min, const Vec3V max);
543// a.x>=min.x && a.y>=min.y && a.z>=min.z && a.x<=max.x && a.y<=max.y && a.z<=max.z
544// Note: a.w and min.w and max.w must have value zero
545PX_FORCE_INLINE PxU32 V3InBounds(const Vec3V a, const Vec3V min, const Vec3V max);
546// a.x<-bounds.x || a.y<=-bounds.y || a.z<bounds.z || a.x>bounds.x || a.y>bounds.y || a.z>bounds.z
547// Note: a.w and bounds.w must have value zero
548PX_FORCE_INLINE PxU32 V3OutOfBounds(const Vec3V a, const Vec3V bounds);
549// a.x>=-bounds.x && a.y>=-bounds.y && a.z>=-bounds.z && a.x<=bounds.x && a.y<=bounds.y && a.z<=bounds.z
550// Note: a.w and bounds.w must have value zero
551PX_FORCE_INLINE PxU32 V3InBounds(const Vec3V a, const Vec3V bounds);
552
553//(floor(a.x + 0.5f), floor(a.y + 0.5f), floor(a.z + 0.5f))
554PX_FORCE_INLINE Vec3V V3Round(const Vec3V a);
555
556//(sinf(a.x), sinf(a.y), sinf(a.z))
557PX_FORCE_INLINE Vec3V V3Sin(const Vec3V a);
558//(cosf(a.x), cosf(a.y), cosf(a.z))
559PX_FORCE_INLINE Vec3V V3Cos(const Vec3V a);
560
561//(a.y,a.z,a.z)
562PX_FORCE_INLINE Vec3V V3PermYZZ(const Vec3V a);
563//(a.x,a.y,a.x)
564PX_FORCE_INLINE Vec3V V3PermXYX(const Vec3V a);
565//(a.y,a.z,a.x)
566PX_FORCE_INLINE Vec3V V3PermYZX(const Vec3V a);
567//(a.z, a.x, a.y)
568PX_FORCE_INLINE Vec3V V3PermZXY(const Vec3V a);
569//(a.z,a.z,a.y)
570PX_FORCE_INLINE Vec3V V3PermZZY(const Vec3V a);
571//(a.y,a.x,a.x)
572PX_FORCE_INLINE Vec3V V3PermYXX(const Vec3V a);
573//(0, v1.z, v0.y)
574PX_FORCE_INLINE Vec3V V3Perm_Zero_1Z_0Y(const Vec3V v0, const Vec3V v1);
575//(v0.z, 0, v1.x)
576PX_FORCE_INLINE Vec3V V3Perm_0Z_Zero_1X(const Vec3V v0, const Vec3V v1);
577//(v1.y, v0.x, 0)
578PX_FORCE_INLINE Vec3V V3Perm_1Y_0X_Zero(const Vec3V v0, const Vec3V v1);
579
580// Transpose 3 Vec3Vs inplace. Sets the w component to zero
581// [ x0, y0, z0, w0] [ x1, y1, z1, w1] [ x2, y2, z2, w2] -> [x0 x1 x2 0] [y0 y1 y2 0] [z0 z1 z2 0]
582PX_FORCE_INLINE void V3Transpose(Vec3V& col0, Vec3V& col1, Vec3V& col2);
583
585// Math operations on Vec4V
587
588//(f,f,f,f)
589PX_FORCE_INLINE Vec4V V4Splat(const FloatV f);
590
591//(f[0],f[1],f[2],f[3])
592PX_FORCE_INLINE Vec4V V4Merge(const FloatV* const f);
593//(x,y,z,w)
594PX_FORCE_INLINE Vec4V V4Merge(const FloatVArg x, const FloatVArg y, const FloatVArg z, const FloatVArg w);
595//(x.w, y.w, z.w, w.w)
596PX_FORCE_INLINE Vec4V V4MergeW(const Vec4VArg x, const Vec4VArg y, const Vec4VArg z, const Vec4VArg w);
597//(x.z, y.z, z.z, w.z)
598PX_FORCE_INLINE Vec4V V4MergeZ(const Vec4VArg x, const Vec4VArg y, const Vec4VArg z, const Vec4VArg w);
599//(x.y, y.y, z.y, w.y)
600PX_FORCE_INLINE Vec4V V4MergeY(const Vec4VArg x, const Vec4VArg y, const Vec4VArg z, const Vec4VArg w);
601//(x.x, y.x, z.x, w.x)
602PX_FORCE_INLINE Vec4V V4MergeX(const Vec4VArg x, const Vec4VArg y, const Vec4VArg z, const Vec4VArg w);
603
604//(a.x, b.x, a.y, b.y)
605PX_FORCE_INLINE Vec4V V4UnpackXY(const Vec4VArg a, const Vec4VArg b);
606//(a.z, b.z, a.w, b.w)
607PX_FORCE_INLINE Vec4V V4UnpackZW(const Vec4VArg a, const Vec4VArg b);
608
609//(1,0,0,0)
610PX_FORCE_INLINE Vec4V V4UnitW();
611//(0,1,0,0)
612PX_FORCE_INLINE Vec4V V4UnitY();
613//(0,0,1,0)
614PX_FORCE_INLINE Vec4V V4UnitZ();
615//(0,0,0,1)
616PX_FORCE_INLINE Vec4V V4UnitW();
617
618//(f.x,f.x,f.x,f.x)
619PX_FORCE_INLINE FloatV V4GetX(const Vec4V f);
620//(f.y,f.y,f.y,f.y)
621PX_FORCE_INLINE FloatV V4GetY(const Vec4V f);
622//(f.z,f.z,f.z,f.z)
623PX_FORCE_INLINE FloatV V4GetZ(const Vec4V f);
624//(f.w,f.w,f.w,f.w)
625PX_FORCE_INLINE FloatV V4GetW(const Vec4V f);
626
627//(f,v.y,v.z,v.w)
628PX_FORCE_INLINE Vec4V V4SetX(const Vec4V v, const FloatV f);
629//(v.x,f,v.z,v.w)
630PX_FORCE_INLINE Vec4V V4SetY(const Vec4V v, const FloatV f);
631//(v.x,v.y,f,v.w)
632PX_FORCE_INLINE Vec4V V4SetZ(const Vec4V v, const FloatV f);
633//(v.x,v.y,v.z,f)
634PX_FORCE_INLINE Vec4V V4SetW(const Vec4V v, const FloatV f);
635
636//(v.x,v.y,v.z,0)
637PX_FORCE_INLINE Vec4V V4ClearW(const Vec4V v);
638
639//(a[elementIndex], a[elementIndex], a[elementIndex], a[elementIndex])
640template <int elementIndex>
641PX_FORCE_INLINE Vec4V V4SplatElement(Vec4V a);
642
643// v.x=f
644PX_FORCE_INLINE void V4WriteX(Vec4V& v, const PxF32 f);
645// v.y=f
646PX_FORCE_INLINE void V4WriteY(Vec4V& v, const PxF32 f);
647// v.z=f
648PX_FORCE_INLINE void V4WriteZ(Vec4V& v, const PxF32 f);
649// v.w=f
650PX_FORCE_INLINE void V4WriteW(Vec4V& v, const PxF32 f);
651// v.x=f.x, v.y=f.y, v.z=f.z
652PX_FORCE_INLINE void V4WriteXYZ(Vec4V& v, const PxVec3& f);
653// return v.x
654PX_FORCE_INLINE PxF32 V4ReadX(const Vec4V& v);
655// return v.y
656PX_FORCE_INLINE PxF32 V4ReadY(const Vec4V& v);
657// return v.z
658PX_FORCE_INLINE PxF32 V4ReadZ(const Vec4V& v);
659// return v.w
660PX_FORCE_INLINE PxF32 V4ReadW(const Vec4V& v);
661// return (v.x,v.y,v.z)
662PX_FORCE_INLINE const PxVec3& V4ReadXYZ(const Vec4V& v);
663
664//(0,0,0,0)
665PX_FORCE_INLINE Vec4V V4Zero();
666//(1,1,1,1)
667PX_FORCE_INLINE Vec4V V4One();
668//(PX_EPS_REAL,PX_EPS_REAL,PX_EPS_REAL,PX_EPS_REAL)
669PX_FORCE_INLINE Vec4V V4Eps();
670
671//-c (per component)
672PX_FORCE_INLINE Vec4V V4Neg(const Vec4V c);
673// a+b (per component)
674PX_FORCE_INLINE Vec4V V4Add(const Vec4V a, const Vec4V b);
675// a-b (per component)
676PX_FORCE_INLINE Vec4V V4Sub(const Vec4V a, const Vec4V b);
677// a*b (per component)
678PX_FORCE_INLINE Vec4V V4Scale(const Vec4V a, const FloatV b);
679// a*b (per component)
680PX_FORCE_INLINE Vec4V V4Mul(const Vec4V a, const Vec4V b);
681// a/b (per component)
682PX_FORCE_INLINE Vec4V V4ScaleInv(const Vec4V a, const FloatV b);
683// a/b (per component)
684PX_FORCE_INLINE Vec4V V4Div(const Vec4V a, const Vec4V b);
685// a/b (per component)
686PX_FORCE_INLINE Vec4V V4ScaleInvFast(const Vec4V a, const FloatV b);
687// a/b (per component)
688PX_FORCE_INLINE Vec4V V4DivFast(const Vec4V a, const Vec4V b);
689// 1.0f/a
690PX_FORCE_INLINE Vec4V V4Recip(const Vec4V a);
691// 1.0f/a
692PX_FORCE_INLINE Vec4V V4RecipFast(const Vec4V a);
693// 1.0f/sqrt(a)
694PX_FORCE_INLINE Vec4V V4Rsqrt(const Vec4V a);
695// 1.0f/sqrt(a)
696PX_FORCE_INLINE Vec4V V4RsqrtFast(const Vec4V a);
697// a*b+c
698PX_FORCE_INLINE Vec4V V4ScaleAdd(const Vec4V a, const FloatV b, const Vec4V c);
699// c-a*b
700PX_FORCE_INLINE Vec4V V4NegScaleSub(const Vec4V a, const FloatV b, const Vec4V c);
701// a*b+c
702PX_FORCE_INLINE Vec4V V4MulAdd(const Vec4V a, const Vec4V b, const Vec4V c);
703// c-a*b
704PX_FORCE_INLINE Vec4V V4NegMulSub(const Vec4V a, const Vec4V b, const Vec4V c);
705
706// fabs(a)
707PX_FORCE_INLINE Vec4V V4Abs(const Vec4V a);
708// bitwise a & ~b
709PX_FORCE_INLINE Vec4V V4Andc(const Vec4V a, const VecU32V b);
710
711// a.b (W is taken into account)
712PX_FORCE_INLINE FloatV V4Dot(const Vec4V a, const Vec4V b);
713// a.b (same computation as V3Dot. W is ignored in input)
714PX_FORCE_INLINE FloatV V4Dot3(const Vec4V a, const Vec4V b);
715// aXb (same computation as V3Cross. W is ignored in input and undefined in output)
716PX_FORCE_INLINE Vec4V V4Cross(const Vec4V a, const Vec4V b);
717
718//|a.a|^1/2
719PX_FORCE_INLINE FloatV V4Length(const Vec4V a);
720// a.a
721PX_FORCE_INLINE FloatV V4LengthSq(const Vec4V a);
722
723// a*|a.a|^-1/2
724PX_FORCE_INLINE Vec4V V4Normalize(const Vec4V a);
725// a.a>0 ? a*|a.a|^-1/2 : unsafeReturnValue
726PX_FORCE_INLINE Vec4V V4NormalizeSafe(const Vec4V a, const Vec4V unsafeReturnValue);
727// a*|a.a|^-1/2
728PX_FORCE_INLINE Vec4V V4NormalizeFast(const Vec4V a);
729
730// c ? a : b (per component)
731PX_FORCE_INLINE Vec4V V4Sel(const BoolV c, const Vec4V a, const Vec4V b);
732// a>b (per component)
733PX_FORCE_INLINE BoolV V4IsGrtr(const Vec4V a, const Vec4V b);
734// a>=b (per component)
735PX_FORCE_INLINE BoolV V4IsGrtrOrEq(const Vec4V a, const Vec4V b);
736// a==b (per component)
737PX_FORCE_INLINE BoolV V4IsEq(const Vec4V a, const Vec4V b);
738// Max(a,b) (per component)
739PX_FORCE_INLINE Vec4V V4Max(const Vec4V a, const Vec4V b);
740// Min(a,b) (per component)
741PX_FORCE_INLINE Vec4V V4Min(const Vec4V a, const Vec4V b);
742// Get the maximum component from a
743PX_FORCE_INLINE FloatV V4ExtractMax(const Vec4V a);
744// Get the minimum component from a
745PX_FORCE_INLINE FloatV V4ExtractMin(const Vec4V a);
746
747// Clamp(a,b) (per component)
748PX_FORCE_INLINE Vec4V V4Clamp(const Vec4V a, const Vec4V minV, const Vec4V maxV);
749
750// return 1 if all components of a are greater than all components of b.
751PX_FORCE_INLINE PxU32 V4AllGrtr(const Vec4V a, const Vec4V b);
752// return 1 if all components of a are greater than or equal to all components of b
753PX_FORCE_INLINE PxU32 V4AllGrtrOrEq(const Vec4V a, const Vec4V b);
754// return 1 if XYZ components of a are greater than or equal to XYZ components of b. W is ignored.
755PX_FORCE_INLINE PxU32 V4AllGrtrOrEq3(const Vec4V a, const Vec4V b);
756// return 1 if all components of a are equal to all components of b
757PX_FORCE_INLINE PxU32 V4AllEq(const Vec4V a, const Vec4V b);
758// return 1 if any XYZ component of a is greater than the corresponding component of b. W is ignored.
759PX_FORCE_INLINE PxU32 V4AnyGrtr3(const Vec4V a, const Vec4V b);
760
761// round(a)(per component)
762PX_FORCE_INLINE Vec4V V4Round(const Vec4V a);
763// sin(a) (per component)
764PX_FORCE_INLINE Vec4V V4Sin(const Vec4V a);
765// cos(a) (per component)
766PX_FORCE_INLINE Vec4V V4Cos(const Vec4V a);
767
768// Permute v into a new vec4v with YXWZ format
769PX_FORCE_INLINE Vec4V V4PermYXWZ(const Vec4V v);
770// Permute v into a new vec4v with XZXZ format
771PX_FORCE_INLINE Vec4V V4PermXZXZ(const Vec4V v);
772// Permute v into a new vec4v with YWYW format
773PX_FORCE_INLINE Vec4V V4PermYWYW(const Vec4V v);
774// Permute v into a new vec4v with YZXW format
775PX_FORCE_INLINE Vec4V V4PermYZXW(const Vec4V v);
776// Permute v into a new vec4v with ZWXY format - equivalent to a swap of the two 64bit parts of the vector
777PX_FORCE_INLINE Vec4V V4PermZWXY(const Vec4V a);
778
779// Permute v into a new vec4v with format {a[x], a[y], a[z], a[w]}
780// V4Perm<1,3,1,3> is equal to V4PermYWYW
781// V4Perm<0,2,0,2> is equal to V4PermXZXZ
782// V3Perm<1,0,3,2> is equal to V4PermYXWZ
783template <PxU8 x, PxU8 y, PxU8 z, PxU8 w>
784PX_FORCE_INLINE Vec4V V4Perm(const Vec4V a);
785
786// Transpose 4 Vec4Vs inplace.
787// [ x0, y0, z0, w0] [ x1, y1, z1, w1] [ x2, y2, z2, w2] [ x3, y3, z3, w3] ->
788// [ x0, x1, x2, x3] [ y0, y1, y2, y3] [ z0, z1, z2, z3] [ w0, w1, w2, w3]
789PX_FORCE_INLINE void V3Transpose(Vec3V& col0, Vec3V& col1, Vec3V& col2);
790
791// q = cos(a/2) + u*sin(a/2)
792PX_FORCE_INLINE QuatV QuatV_From_RotationAxisAngle(const Vec3V u, const FloatV a);
793// convert q to a unit quaternion
794PX_FORCE_INLINE QuatV QuatNormalize(const QuatV q);
795//|q.q|^1/2
796PX_FORCE_INLINE FloatV QuatLength(const QuatV q);
797// q.q
798PX_FORCE_INLINE FloatV QuatLengthSq(const QuatV q);
799// a.b
800PX_FORCE_INLINE FloatV QuatDot(const QuatV a, const QuatV b);
801//(-q.x, -q.y, -q.z, q.w)
802PX_FORCE_INLINE QuatV QuatConjugate(const QuatV q);
803//(q.x, q.y, q.z)
804PX_FORCE_INLINE Vec3V QuatGetImaginaryPart(const QuatV q);
805// convert quaternion to matrix 33
806PX_FORCE_INLINE Mat33V QuatGetMat33V(const QuatVArg q);
807// convert quaternion to matrix 33
808PX_FORCE_INLINE void QuatGetMat33V(const QuatVArg q, Vec3V& column0, Vec3V& column1, Vec3V& column2);
809// convert matrix 33 to quaternion
810PX_FORCE_INLINE QuatV Mat33GetQuatV(const Mat33V& a);
811// brief computes rotation of x-axis
812PX_FORCE_INLINE Vec3V QuatGetBasisVector0(const QuatV q);
813// brief computes rotation of y-axis
814PX_FORCE_INLINE Vec3V QuatGetBasisVector1(const QuatV q);
815// brief computes rotation of z-axis
816PX_FORCE_INLINE Vec3V QuatGetBasisVector2(const QuatV q);
817// calculate the rotation vector from q and v
818PX_FORCE_INLINE Vec3V QuatRotate(const QuatV q, const Vec3V v);
819// calculate the rotation vector from the conjugate quaternion and v
820PX_FORCE_INLINE Vec3V QuatRotateInv(const QuatV q, const Vec3V v);
821// quaternion multiplication
822PX_FORCE_INLINE QuatV QuatMul(const QuatV a, const QuatV b);
823// quaternion add
824PX_FORCE_INLINE QuatV QuatAdd(const QuatV a, const QuatV b);
825// (-q.x, -q.y, -q.z, -q.w)
826PX_FORCE_INLINE QuatV QuatNeg(const QuatV q);
827// (a.x - b.x, a.y-b.y, a.z-b.z, a.w-b.w )
828PX_FORCE_INLINE QuatV QuatSub(const QuatV a, const QuatV b);
829// (a.x*b, a.y*b, a.z*b, a.w*b)
830PX_FORCE_INLINE QuatV QuatScale(const QuatV a, const FloatV b);
831// (x = v[0], y = v[1], z = v[2], w =v[3])
832PX_FORCE_INLINE QuatV QuatMerge(const FloatV* const v);
833// (x = v[0], y = v[1], z = v[2], w =v[3])
834PX_FORCE_INLINE QuatV QuatMerge(const FloatVArg x, const FloatVArg y, const FloatVArg z, const FloatVArg w);
835// (x = 0.f, y = 0.f, z = 0.f, w = 1.f)
836PX_FORCE_INLINE QuatV QuatIdentity();
837// check for each component is valid
838PX_FORCE_INLINE bool isFiniteQuatV(const QuatV q);
839// check for each component is valid
840PX_FORCE_INLINE bool isValidQuatV(const QuatV q);
841// check for each component is valid
842PX_FORCE_INLINE bool isSaneQuatV(const QuatV q);
843
844// Math operations on 16-byte aligned booleans.
845// x=false y=false z=false w=false
846PX_FORCE_INLINE BoolV BFFFF();
847// x=false y=false z=false w=true
848PX_FORCE_INLINE BoolV BFFFT();
849// x=false y=false z=true w=false
850PX_FORCE_INLINE BoolV BFFTF();
851// x=false y=false z=true w=true
852PX_FORCE_INLINE BoolV BFFTT();
853// x=false y=true z=false w=false
854PX_FORCE_INLINE BoolV BFTFF();
855// x=false y=true z=false w=true
856PX_FORCE_INLINE BoolV BFTFT();
857// x=false y=true z=true w=false
858PX_FORCE_INLINE BoolV BFTTF();
859// x=false y=true z=true w=true
860PX_FORCE_INLINE BoolV BFTTT();
861// x=true y=false z=false w=false
862PX_FORCE_INLINE BoolV BTFFF();
863// x=true y=false z=false w=true
864PX_FORCE_INLINE BoolV BTFFT();
865// x=true y=false z=true w=false
866PX_FORCE_INLINE BoolV BTFTF();
867// x=true y=false z=true w=true
868PX_FORCE_INLINE BoolV BTFTT();
869// x=true y=true z=false w=false
870PX_FORCE_INLINE BoolV BTTFF();
871// x=true y=true z=false w=true
872PX_FORCE_INLINE BoolV BTTFT();
873// x=true y=true z=true w=false
874PX_FORCE_INLINE BoolV BTTTF();
875// x=true y=true z=true w=true
876PX_FORCE_INLINE BoolV BTTTT();
877
878// x=false y=false z=false w=true
879PX_FORCE_INLINE BoolV BWMask();
880// x=true y=false z=false w=false
881PX_FORCE_INLINE BoolV BXMask();
882// x=false y=true z=false w=false
883PX_FORCE_INLINE BoolV BYMask();
884// x=false y=false z=true w=false
885PX_FORCE_INLINE BoolV BZMask();
886
887// get x component
888PX_FORCE_INLINE BoolV BGetX(const BoolV f);
889// get y component
890PX_FORCE_INLINE BoolV BGetY(const BoolV f);
891// get z component
892PX_FORCE_INLINE BoolV BGetZ(const BoolV f);
893// get w component
894PX_FORCE_INLINE BoolV BGetW(const BoolV f);
895
896// Use elementIndex to splat xxxx or yyyy or zzzz or wwww
897template <int elementIndex>
898PX_FORCE_INLINE BoolV BSplatElement(Vec4V a);
899
900// component-wise && (AND)
901PX_FORCE_INLINE BoolV BAnd(const BoolV a, const BoolV b);
902// component-wise || (OR)
903PX_FORCE_INLINE BoolV BOr(const BoolV a, const BoolV b);
904// component-wise not
905PX_FORCE_INLINE BoolV BNot(const BoolV a);
906
907// if all four components are true, return true, otherwise return false
908PX_FORCE_INLINE BoolV BAllTrue4(const BoolV a);
909
910// if any four components is true, return true, otherwise return false
911PX_FORCE_INLINE BoolV BAnyTrue4(const BoolV a);
912
913// if all three(0, 1, 2) components are true, return true, otherwise return false
914PX_FORCE_INLINE BoolV BAllTrue3(const BoolV a);
915
916// if any three (0, 1, 2) components is true, return true, otherwise return false
917PX_FORCE_INLINE BoolV BAnyTrue3(const BoolV a);
918
919// Return 1 if all components equal, zero otherwise.
920PX_FORCE_INLINE PxU32 BAllEq(const BoolV a, const BoolV b);
921
922// Specialized/faster BAllEq function for b==TTTT
923PX_FORCE_INLINE PxU32 BAllEqTTTT(const BoolV a);
924// Specialized/faster BAllEq function for b==FFFF
925PX_FORCE_INLINE PxU32 BAllEqFFFF(const BoolV a);
926
932PX_FORCE_INLINE PxU32 BGetBitMask(const BoolV a);
933
934// VecI32V stuff
935
936PX_FORCE_INLINE VecI32V VecI32V_Zero();
937
938PX_FORCE_INLINE VecI32V VecI32V_One();
939
940PX_FORCE_INLINE VecI32V VecI32V_Two();
941
942PX_FORCE_INLINE VecI32V VecI32V_MinusOne();
943
944// Compute a shift parameter for VecI32V_LeftShift and VecI32V_RightShift
945// Each element of shift must be identical ie the vector must have form {count, count, count, count} with count>=0
946PX_FORCE_INLINE VecShiftV VecI32V_PrepareShift(const VecI32VArg shift);
947
948// Shift each element of a leftwards by the same amount
949// Compute shift with VecI32V_PrepareShift
950//{a.x<<shift[0], a.y<<shift[0], a.z<<shift[0], a.w<<shift[0]}
951PX_FORCE_INLINE VecI32V VecI32V_LeftShift(const VecI32VArg a, const VecShiftVArg shift);
952
953// Shift each element of a rightwards by the same amount
954// Compute shift with VecI32V_PrepareShift
955//{a.x>>shift[0], a.y>>shift[0], a.z>>shift[0], a.w>>shift[0]}
956PX_FORCE_INLINE VecI32V VecI32V_RightShift(const VecI32VArg a, const VecShiftVArg shift);
957
958PX_FORCE_INLINE VecI32V VecI32V_Add(const VecI32VArg a, const VecI32VArg b);
959
960PX_FORCE_INLINE VecI32V VecI32V_Or(const VecI32VArg a, const VecI32VArg b);
961
962PX_FORCE_INLINE VecI32V VecI32V_GetX(const VecI32VArg a);
963
964PX_FORCE_INLINE VecI32V VecI32V_GetY(const VecI32VArg a);
965
966PX_FORCE_INLINE VecI32V VecI32V_GetZ(const VecI32VArg a);
967
968PX_FORCE_INLINE VecI32V VecI32V_GetW(const VecI32VArg a);
969
970PX_FORCE_INLINE VecI32V VecI32V_Sub(const VecI32VArg a, const VecI32VArg b);
971
972PX_FORCE_INLINE BoolV VecI32V_IsGrtr(const VecI32VArg a, const VecI32VArg b);
973
974PX_FORCE_INLINE BoolV VecI32V_IsEq(const VecI32VArg a, const VecI32VArg b);
975
976PX_FORCE_INLINE VecI32V V4I32Sel(const BoolV c, const VecI32V a, const VecI32V b);
977
978// VecU32V stuff
979
980PX_FORCE_INLINE VecU32V U4Zero();
981
982PX_FORCE_INLINE VecU32V U4One();
983
984PX_FORCE_INLINE VecU32V U4Two();
985
986PX_FORCE_INLINE BoolV V4IsEqU32(const VecU32V a, const VecU32V b);
987
988PX_FORCE_INLINE VecU32V V4U32Sel(const BoolV c, const VecU32V a, const VecU32V b);
989
990PX_FORCE_INLINE VecU32V V4U32or(VecU32V a, VecU32V b);
991
992PX_FORCE_INLINE VecU32V V4U32xor(VecU32V a, VecU32V b);
993
994PX_FORCE_INLINE VecU32V V4U32and(VecU32V a, VecU32V b);
995
996PX_FORCE_INLINE VecU32V V4U32Andc(VecU32V a, VecU32V b);
997
998// VecU32 - why does this not return a bool?
999PX_FORCE_INLINE VecU32V V4IsGrtrV32u(const Vec4V a, const Vec4V b);
1000
1001// Math operations on 16-byte aligned Mat33s (represents any 3x3 matrix)
1002PX_FORCE_INLINE Mat33V M33Load(const PxMat33& m)
1003{
1004 return Mat33V(Vec3V_From_Vec4V(V4LoadU(&m.column0.x)),
1005 Vec3V_From_Vec4V(V4LoadU(&m.column1.x)), V3LoadU(m.column2));
1006}
1007// a*b
1008PX_FORCE_INLINE Vec3V M33MulV3(const Mat33V& a, const Vec3V b);
1009// A*x + b
1010PX_FORCE_INLINE Vec3V M33MulV3AddV3(const Mat33V& A, const Vec3V b, const Vec3V c);
1011// transpose(a) * b
1012PX_FORCE_INLINE Vec3V M33TrnspsMulV3(const Mat33V& a, const Vec3V b);
1013// a*b
1014PX_FORCE_INLINE Mat33V M33MulM33(const Mat33V& a, const Mat33V& b);
1015// a+b
1016PX_FORCE_INLINE Mat33V M33Add(const Mat33V& a, const Mat33V& b);
1017// a+b
1018PX_FORCE_INLINE Mat33V M33Sub(const Mat33V& a, const Mat33V& b);
1019//-a
1020PX_FORCE_INLINE Mat33V M33Neg(const Mat33V& a);
1021// absolute value of the matrix
1022PX_FORCE_INLINE Mat33V M33Abs(const Mat33V& a);
1023// inverse mat
1024PX_FORCE_INLINE Mat33V M33Inverse(const Mat33V& a);
1025// transpose(a)
1026PX_FORCE_INLINE Mat33V M33Trnsps(const Mat33V& a);
1027// create an identity matrix
1028PX_FORCE_INLINE Mat33V M33Identity();
1029
1030// create a vec3 to store the diagonal element of the M33
1031PX_FORCE_INLINE Mat33V M33Diagonal(const Vec3VArg);
1032
1033// Not implemented
1034// return 1 if all components of a are equal to all components of b
1035// PX_FORCE_INLINE PxU32 V4U32AllEq(const VecU32V a, const VecU32V b);
1036// v.w=f
1037// PX_FORCE_INLINE void V3WriteW(Vec3V& v, const PxF32 f);
1038// PX_FORCE_INLINE PxF32 V3ReadW(const Vec3V& v);
1039
1040// Not used
1041// PX_FORCE_INLINE Vec4V V4LoadAligned(Vec4V* addr);
1042// PX_FORCE_INLINE Vec4V V4LoadUnaligned(Vec4V* addr);
1043// floor(a)(per component)
1044// PX_FORCE_INLINE Vec4V V4Floor(Vec4V a);
1045// ceil(a) (per component)
1046// PX_FORCE_INLINE Vec4V V4Ceil(Vec4V a);
1047// PX_FORCE_INLINE VecU32V V4ConvertToU32VSaturate(const Vec4V a, PxU32 power);
1048
1049// Math operations on 16-byte aligned Mat34s (represents transformation matrix - rotation and translation).
1050// namespace _Mat34V
1051//{
1052// //a*b
1053// PX_FORCE_INLINE Vec3V multiplyV(const Mat34V& a, const Vec3V b);
1054// //a_rotation * b
1055// PX_FORCE_INLINE Vec3V multiply3X3V(const Mat34V& a, const Vec3V b);
1056// //transpose(a_rotation)*b
1057// PX_FORCE_INLINE Vec3V multiplyTranspose3X3V(const Mat34V& a, const Vec3V b);
1058// //a*b
1059// PX_FORCE_INLINE Mat34V multiplyV(const Mat34V& a, const Mat34V& b);
1060// //a_rotation*b
1061// PX_FORCE_INLINE Mat33V multiply3X3V(const Mat34V& a, const Mat33V& b);
1062// //a_rotation*b_rotation
1063// PX_FORCE_INLINE Mat33V multiply3X3V(const Mat34V& a, const Mat34V& b);
1064// //a+b
1065// PX_FORCE_INLINE Mat34V addV(const Mat34V& a, const Mat34V& b);
1066// //a^-1
1067// PX_FORCE_INLINE Mat34V getInverseV(const Mat34V& a);
1068// //transpose(a_rotation)
1069// PX_FORCE_INLINE Mat33V getTranspose3X3(const Mat34V& a);
1070//}; //namespace _Mat34V
1071
1072// a*b
1073//#define M34MulV3(a,b) (M34MulV3(a,b))
1075//#define M34Mul33V3(a,b) (M34Mul33V3(a,b))
1077//#define M34TrnspsMul33V3(a,b) (M34TrnspsMul33V3(a,b))
1079//#define M34MulM34(a,b) (_Mat34V::multiplyV(a,b))
1080// a_rotation*b
1081//#define M34MulM33(a,b) (M34MulM33(a,b))
1082// a_rotation*b_rotation
1083//#define M34Mul33MM34(a,b) (M34MulM33(a,b))
1084// a+b
1085//#define M34Add(a,b) (M34Add(a,b))
1087//#define M34Inverse(a,b) (M34Inverse(a))
1088// transpose(a_rotation)
1089//#define M34Trnsps33(a) (M33Trnsps3X3(a))
1090
1091// Math operations on 16-byte aligned Mat44s (represents any 4x4 matrix)
1092// namespace _Mat44V
1093//{
1094// //a*b
1095// PX_FORCE_INLINE Vec4V multiplyV(const Mat44V& a, const Vec4V b);
1096// //transpose(a)*b
1097// PX_FORCE_INLINE Vec4V multiplyTransposeV(const Mat44V& a, const Vec4V b);
1098// //a*b
1099// PX_FORCE_INLINE Mat44V multiplyV(const Mat44V& a, const Mat44V& b);
1100// //a+b
1101// PX_FORCE_INLINE Mat44V addV(const Mat44V& a, const Mat44V& b);
1102// //a&-1
1103// PX_FORCE_INLINE Mat44V getInverseV(const Mat44V& a);
1104// //transpose(a)
1105// PX_FORCE_INLINE Mat44V getTransposeV(const Mat44V& a);
1106//}; //namespace _Mat44V
1107
1108// namespace _VecU32V
1109//{
1110// // pack 8 U32s to 8 U16s with saturation
1111// PX_FORCE_INLINE VecU16V pack2U32VToU16VSaturate(VecU32V a, VecU32V b);
1112// PX_FORCE_INLINE VecU32V orV(VecU32V a, VecU32V b);
1113// PX_FORCE_INLINE VecU32V andV(VecU32V a, VecU32V b);
1114// PX_FORCE_INLINE VecU32V andcV(VecU32V a, VecU32V b);
1115// // conversion from integer to float
1116// PX_FORCE_INLINE Vec4V convertToVec4V(VecU32V a);
1117// // splat a[elementIndex] into all fields of a
1118// template<int elementIndex>
1119// PX_FORCE_INLINE VecU32V splatElement(VecU32V a);
1120// PX_FORCE_INLINE void storeAligned(VecU32V a, VecU32V* address);
1121//};
1122
1123// namespace _VecI32V
1124//{
1125// template<int a> PX_FORCE_INLINE VecI32V splatI32();
1126//};
1127//
1128// namespace _VecU16V
1129//{
1130// PX_FORCE_INLINE VecU16V orV(VecU16V a, VecU16V b);
1131// PX_FORCE_INLINE VecU16V andV(VecU16V a, VecU16V b);
1132// PX_FORCE_INLINE VecU16V andcV(VecU16V a, VecU16V b);
1133// PX_FORCE_INLINE void storeAligned(VecU16V val, VecU16V *address);
1134// PX_FORCE_INLINE VecU16V loadAligned(VecU16V* addr);
1135// PX_FORCE_INLINE VecU16V loadUnaligned(VecU16V* addr);
1136// PX_FORCE_INLINE VecU16V compareGt(VecU16V a, VecU16V b);
1137// template<int elementIndex>
1138// PX_FORCE_INLINE VecU16V splatElement(VecU16V a);
1139// PX_FORCE_INLINE VecU16V subtractModulo(VecU16V a, VecU16V b);
1140// PX_FORCE_INLINE VecU16V addModulo(VecU16V a, VecU16V b);
1141// PX_FORCE_INLINE VecU32V getLo16(VecU16V a); // [0,2,4,6] 16-bit values to [0,1,2,3] 32-bit vector
1142// PX_FORCE_INLINE VecU32V getHi16(VecU16V a); // [1,3,5,7] 16-bit values to [0,1,2,3] 32-bit vector
1143//};
1144//
1145// namespace _VecI16V
1146//{
1147// template <int val> PX_FORCE_INLINE VecI16V splatImmediate();
1148//};
1149//
1150// namespace _VecU8V
1151//{
1152//};
1153
1154// a*b
1155//#define M44MulV4(a,b) (M44MulV4(a,b))
1157//#define M44TrnspsMulV4(a,b) (M44TrnspsMulV4(a,b))
1159//#define M44MulM44(a,b) (M44MulM44(a,b))
1161//#define M44Add(a,b) (M44Add(a,b))
1163//#define M44Inverse(a) (M44Inverse(a))
1165//#define M44Trnsps(a) (M44Trnsps(a))
1166
1167// dsequeira: these used to be assert'd out in SIMD builds, but they're necessary if
1168// we want to be able to write some scalar functions which run using SIMD data structures
1169
1170PX_FORCE_INLINE void V3WriteX(Vec3V& v, const PxF32 f)
1171{
1172 reinterpret_cast<PxVec3&>(v).x = f;
1173}
1174
1175PX_FORCE_INLINE void V3WriteY(Vec3V& v, const PxF32 f)
1176{
1177 reinterpret_cast<PxVec3&>(v).y = f;
1178}
1179
1180PX_FORCE_INLINE void V3WriteZ(Vec3V& v, const PxF32 f)
1181{
1182 reinterpret_cast<PxVec3&>(v).z = f;
1183}
1184
1185PX_FORCE_INLINE void V3WriteXYZ(Vec3V& v, const PxVec3& f)
1186{
1187 reinterpret_cast<PxVec3&>(v) = f;
1188}
1189
1190PX_FORCE_INLINE PxF32 V3ReadX(const Vec3V& v)
1191{
1192 return reinterpret_cast<const PxVec3&>(v).x;
1193}
1194
1195PX_FORCE_INLINE PxF32 V3ReadY(const Vec3V& v)
1196{
1197 return reinterpret_cast<const PxVec3&>(v).y;
1198}
1199
1200PX_FORCE_INLINE PxF32 V3ReadZ(const Vec3V& v)
1201{
1202 return reinterpret_cast<const PxVec3&>(v).z;
1203}
1204
1205PX_FORCE_INLINE const PxVec3& V3ReadXYZ(const Vec3V& v)
1206{
1207 return reinterpret_cast<const PxVec3&>(v);
1208}
1209
1210PX_FORCE_INLINE void V4WriteX(Vec4V& v, const PxF32 f)
1211{
1212 reinterpret_cast<PxVec4&>(v).x = f;
1213}
1214
1215PX_FORCE_INLINE void V4WriteY(Vec4V& v, const PxF32 f)
1216{
1217 reinterpret_cast<PxVec4&>(v).y = f;
1218}
1219
1220PX_FORCE_INLINE void V4WriteZ(Vec4V& v, const PxF32 f)
1221{
1222 reinterpret_cast<PxVec4&>(v).z = f;
1223}
1224
1225PX_FORCE_INLINE void V4WriteW(Vec4V& v, const PxF32 f)
1226{
1227 reinterpret_cast<PxVec4&>(v).w = f;
1228}
1229
1230PX_FORCE_INLINE void V4WriteXYZ(Vec4V& v, const PxVec3& f)
1231{
1232 reinterpret_cast<PxVec3&>(v) = f;
1233}
1234
1235PX_FORCE_INLINE PxF32 V4ReadX(const Vec4V& v)
1236{
1237 return reinterpret_cast<const PxVec4&>(v).x;
1238}
1239
1240PX_FORCE_INLINE PxF32 V4ReadY(const Vec4V& v)
1241{
1242 return reinterpret_cast<const PxVec4&>(v).y;
1243}
1244
1245PX_FORCE_INLINE PxF32 V4ReadZ(const Vec4V& v)
1246{
1247 return reinterpret_cast<const PxVec4&>(v).z;
1248}
1249
1250PX_FORCE_INLINE PxF32 V4ReadW(const Vec4V& v)
1251{
1252 return reinterpret_cast<const PxVec4&>(v).w;
1253}
1254
1255PX_FORCE_INLINE const PxVec3& V4ReadXYZ(const Vec4V& v)
1256{
1257 return reinterpret_cast<const PxVec3&>(v);
1258}
1259
1260// this macro transposes 4 Vec4V into 3 Vec4V (assuming that the W component can be ignored
1261#define PX_TRANSPOSE_44_34(inA, inB, inC, inD, outA, outB, outC) \
1262outA = V4UnpackXY(inA, inC); \
1263inA = V4UnpackZW(inA, inC); \
1264inC = V4UnpackXY(inB, inD); \
1265inB = V4UnpackZW(inB, inD); \
1266outB = V4UnpackZW(outA, inC); \
1267outA = V4UnpackXY(outA, inC); \
1268outC = V4UnpackXY(inA, inB);
1269
1270// this macro transposes 3 Vec4V into 4 Vec4V (with W components as garbage!)
1271#define PX_TRANSPOSE_34_44(inA, inB, inC, outA, outB, outC, outD) \
1272 outA = V4UnpackXY(inA, inC); \
1273 inA = V4UnpackZW(inA, inC); \
1274 outC = V4UnpackXY(inB, inB); \
1275 inC = V4UnpackZW(inB, inB); \
1276 outB = V4UnpackZW(outA, outC); \
1277 outA = V4UnpackXY(outA, outC); \
1278 outC = V4UnpackXY(inA, inC); \
1279 outD = V4UnpackZW(inA, inC);
1280
1281#define PX_TRANSPOSE_44(inA, inB, inC, inD, outA, outB, outC, outD) \
1282 outA = V4UnpackXY(inA, inC); \
1283 inA = V4UnpackZW(inA, inC); \
1284 inC = V4UnpackXY(inB, inD); \
1285 inB = V4UnpackZW(inB, inD); \
1286 outB = V4UnpackZW(outA, inC); \
1287 outA = V4UnpackXY(outA, inC); \
1288 outC = V4UnpackXY(inA, inB); \
1289 outD = V4UnpackZW(inA, inB);
1290
1291// This function returns a Vec4V, where each element is the dot product of one pair of Vec3Vs. On PC, each element in
1292// the result should be identical to the results if V3Dot was performed
1293// for each pair of Vec3V.
1294// However, on other platforms, the result might diverge by some small margin due to differences in FP rounding, e.g. if
1295// _mm_dp_ps was used or some other approximate dot product or fused madd operations
1296// were used.
1297// Where there does not exist a hw-accelerated dot-product operation, this approach should be the fastest way to compute
1298// the dot product of 4 vectors.
1299PX_FORCE_INLINE Vec4V V3Dot4(const Vec3VArg a0, const Vec3VArg b0, const Vec3VArg a1, const Vec3VArg b1,
1300 const Vec3VArg a2, const Vec3VArg b2, const Vec3VArg a3, const Vec3VArg b3)
1301{
1302 Vec4V a0b0 = Vec4V_From_Vec3V(V3Mul(a0, b0));
1303 Vec4V a1b1 = Vec4V_From_Vec3V(V3Mul(a1, b1));
1304 Vec4V a2b2 = Vec4V_From_Vec3V(V3Mul(a2, b2));
1305 Vec4V a3b3 = Vec4V_From_Vec3V(V3Mul(a3, b3));
1306
1307 Vec4V aTrnsps, bTrnsps, cTrnsps;
1308
1309 PX_TRANSPOSE_44_34(a0b0, a1b1, a2b2, a3b3, aTrnsps, bTrnsps, cTrnsps);
1310
1311 return V4Add(V4Add(aTrnsps, bTrnsps), cTrnsps);
1312}
1313
1314//(f.x,f.y,f.z,0) - Alternative/faster V3LoadU implementation when it is safe to read "W", i.e. the 32bits after the PxVec3.
1315PX_FORCE_INLINE Vec3V V3LoadU_SafeReadW(const PxVec3& f)
1316{
1317 return Vec3V_From_Vec4V(V4LoadU(&f.x));
1318}
1319
1320} // namespace aos
1321#if !PX_DOXYGEN
1322} // namespace physx
1323#endif
1324
1325// Now for the cross-platform implementations of the 16-byte aligned maths functions (win32/360/ppu/spu etc).
1326#if COMPILE_VECTOR_INTRINSICS
1327#include "PxInlineAoS.h"
1328#else // #if COMPILE_VECTOR_INTRINSICS
1329#include "PxVecMathAoSScalarInline.h"
1330#endif // #if !COMPILE_VECTOR_INTRINSICS
1331#include "PxVecQuat.h"
1332
1333#endif
1334
#define PX_RESTRICT
Definition PxPreprocessor.h:355
#define PX_FORCE_INLINE
Definition PxPreprocessor.h:335
Sorts an array of objects in ascending order, assuming that the predicate implements the < operator:
Definition PxBoxController.h:39