RavEngine
Loading...
Searching...
No Matches
GuBV4_BoxBoxOverlapTest.h
1// Redistribution and use in source and binary forms, with or without
2// modification, are permitted provided that the following conditions
3// are met:
4// * Redistributions of source code must retain the above copyright
5// notice, this list of conditions and the following disclaimer.
6// * Redistributions in binary form must reproduce the above copyright
7// notice, this list of conditions and the following disclaimer in the
8// documentation and/or other materials provided with the distribution.
9// * Neither the name of NVIDIA CORPORATION nor the names of its
10// contributors may be used to endorse or promote products derived
11// from this software without specific prior written permission.
12//
13// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ''AS IS'' AND ANY
14// EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
15// IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
16// PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
17// CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
18// EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
19// PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
20// PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
21// OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
22// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
23// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
24//
25// Copyright (c) 2008-2022 NVIDIA Corporation. All rights reserved.
26// Copyright (c) 2004-2008 AGEIA Technologies, Inc. All rights reserved.
27// Copyright (c) 2001-2004 NovodeX AG. All rights reserved.
28
29#ifndef GU_BV4_BOX_BOX_OVERLAP_TEST_H
30#define GU_BV4_BOX_BOX_OVERLAP_TEST_H
31
32#ifndef GU_BV4_USE_SLABS
33 PX_FORCE_INLINE PxIntBool BV4_BoxBoxOverlap(const PxVec3& extents, const PxVec3& center, const OBBTestParams* PX_RESTRICT params)
34 {
35 const Vec4V extentsV = V4LoadU(&extents.x);
36
37 const Vec4V TV = V4Sub(V4LoadA_Safe(&params->mTBoxToModel_PaddedAligned.x), V4LoadU(&center.x));
38 {
39 const Vec4V absTV = V4Abs(TV);
40 const BoolV resTV = V4IsGrtr(absTV, V4Add(extentsV, V4LoadA_Safe(&params->mBB_PaddedAligned.x)));
41 const PxU32 test = BGetBitMask(resTV);
42 if(test&7)
43 return 0;
44 }
45
46 Vec4V tV;
47 {
48 const Vec4V T_YZX_V = V4Perm<1, 2, 0, 3>(TV);
49 const Vec4V T_ZXY_V = V4Perm<2, 0, 1, 3>(TV);
50
51 tV = V4Mul(TV, V4LoadA_Safe(&params->mPreca0_PaddedAligned.x));
52 tV = V4Add(tV, V4Mul(T_YZX_V, V4LoadA_Safe(&params->mPreca1_PaddedAligned.x)));
53 tV = V4Add(tV, V4Mul(T_ZXY_V, V4LoadA_Safe(&params->mPreca2_PaddedAligned.x)));
54 }
55
56 Vec4V t2V;
57 {
58 const Vec4V extents_YZX_V = V4Perm<1, 2, 0, 3>(extentsV);
59 const Vec4V extents_ZXY_V = V4Perm<2, 0, 1, 3>(extentsV);
60
61 t2V = V4Mul(extentsV, V4LoadA_Safe(&params->mPreca0b_PaddedAligned.x));
62 t2V = V4Add(t2V, V4Mul(extents_YZX_V, V4LoadA_Safe(&params->mPreca1b_PaddedAligned.x)));
63 t2V = V4Add(t2V, V4Mul(extents_ZXY_V, V4LoadA_Safe(&params->mPreca2b_PaddedAligned.x)));
64 t2V = V4Add(t2V, V4LoadA_Safe(&params->mBoxExtents_PaddedAligned.x));
65 }
66
67 {
68 const Vec4V abstV = V4Abs(tV);
69 const BoolV resB = V4IsGrtr(abstV, t2V);
70 const PxU32 test = BGetBitMask(resB);
71 if(test&7)
72 return 0;
73 }
74 return 1;
75 }
76
77#ifdef GU_BV4_QUANTIZED_TREE
78 template<class T>
79 PX_FORCE_INLINE PxIntBool BV4_BoxBoxOverlap(const T* PX_RESTRICT node, const OBBTestParams* PX_RESTRICT params)
80 {
81// A.B. enable new version only for intel non simd path
82#if PX_INTEL_FAMILY && !defined(PX_SIMD_DISABLED)
83// #define NEW_VERSION
84#endif
85#ifdef NEW_VERSION
86 SSE_CONST4(maskV, 0x7fffffff);
87 SSE_CONST4(maskQV, 0x0000ffff);
88#endif
89
90#ifdef NEW_VERSION
91 Vec4V centerV = V4LoadA((float*)node->mAABB.mData);
92 __m128 extentsV = _mm_castsi128_ps(_mm_and_si128(_mm_castps_si128(centerV), SSE_CONST(maskQV)));
93 extentsV = V4Mul(_mm_cvtepi32_ps(_mm_castps_si128(extentsV)), V4LoadA_Safe(&params->mExtentsOrMaxCoeff_PaddedAligned.x));
94 centerV = _mm_castsi128_ps(_mm_srai_epi32(_mm_castps_si128(centerV), 16));
95 centerV = V4Mul(_mm_cvtepi32_ps(_mm_castps_si128(centerV)), V4LoadA_Safe(&params->mCenterOrMinCoeff_PaddedAligned.x));
96#else
97 const VecI32V centerVI = I4LoadA((PxI32*)node->mAABB.mData);
98 const VecI32V extentsVI = VecI32V_And(centerVI, I4Load(0x0000ffff));
99 const Vec4V extentsV = V4Mul(Vec4V_From_VecI32V(extentsVI), V4LoadA_Safe(&params->mExtentsOrMaxCoeff_PaddedAligned.x));
100 const VecI32V centerVShift = VecI32V_RightShift(centerVI, 16);
101 const Vec4V centerV = V4Mul(Vec4V_From_VecI32V(centerVShift), V4LoadA_Safe(&params->mCenterOrMinCoeff_PaddedAligned.x));
102#endif
103
104 const Vec4V TV = V4Sub(V4LoadA_Safe(&params->mTBoxToModel_PaddedAligned.x), centerV);
105 {
106#ifdef NEW_VERSION
107 const __m128 absTV = _mm_and_ps(TV, SSE_CONSTF(maskV));
108#else
109 const Vec4V absTV = V4Abs(TV);
110#endif
111 const BoolV resTV = V4IsGrtr(absTV, V4Add(extentsV, V4LoadA_Safe(&params->mBB_PaddedAligned.x)));
112 const PxU32 test = BGetBitMask(resTV);
113 if(test&7)
114 return 0;
115 }
116
117 Vec4V tV;
118 {
119 const Vec4V T_YZX_V = V4Perm<1, 2, 0, 3>(TV);
120 const Vec4V T_ZXY_V = V4Perm<2, 0, 1, 3>(TV);
121
122 tV = V4Mul(TV, V4LoadA_Safe(&params->mPreca0_PaddedAligned.x));
123 tV = V4Add(tV, V4Mul(T_YZX_V, V4LoadA_Safe(&params->mPreca1_PaddedAligned.x)));
124 tV = V4Add(tV, V4Mul(T_ZXY_V, V4LoadA_Safe(&params->mPreca2_PaddedAligned.x)));
125 }
126
127 Vec4V t2V;
128 {
129 const Vec4V extents_YZX_V = V4Perm<1, 2, 0, 3>(extentsV);
130 const Vec4V extents_ZXY_V = V4Perm<2, 0, 1, 3>(extentsV);
131
132 t2V = V4Mul(extentsV, V4LoadA_Safe(&params->mPreca0b_PaddedAligned.x));
133 t2V = V4Add(t2V, V4Mul(extents_YZX_V, V4LoadA_Safe(&params->mPreca1b_PaddedAligned.x)));
134 t2V = V4Add(t2V, V4Mul(extents_ZXY_V, V4LoadA_Safe(&params->mPreca2b_PaddedAligned.x)));
135 t2V = V4Add(t2V, V4LoadA_Safe(&params->mBoxExtents_PaddedAligned.x));
136 }
137
138 {
139#ifdef NEW_VERSION
140 const __m128 abstV = _mm_and_ps(tV, SSE_CONSTF(maskV));
141#else
142 const Vec4V abstV = V4Abs(tV);
143#endif
144 const BoolV resB = V4IsGrtr(abstV, t2V);
145 const PxU32 test = BGetBitMask(resB);
146 if(test&7)
147 return 0;
148 }
149 return 1;
150 }
151#endif // GU_BV4_QUANTIZED_TREE
152#endif // GU_BV4_USE_SLABS
153
154#ifdef GU_BV4_USE_SLABS
155 PX_FORCE_INLINE PxIntBool BV4_BoxBoxOverlap(const Vec4V boxCenter, const Vec4V extentsV, const OBBTestParams* PX_RESTRICT params)
156 {
157 const Vec4V TV = V4Sub(V4LoadA_Safe(&params->mTBoxToModel_PaddedAligned.x), boxCenter);
158 {
159 const Vec4V absTV = V4Abs(TV);
160 const BoolV res = V4IsGrtr(absTV, V4Add(extentsV, V4LoadA_Safe(&params->mBB_PaddedAligned.x)));
161 const PxU32 test = BGetBitMask(res);
162 if(test&7)
163 return 0;
164 }
165
166 Vec4V tV;
167 {
168 const Vec4V T_YZX_V = V4Perm<1, 2, 0, 3>(TV);
169 const Vec4V T_ZXY_V = V4Perm<2, 0, 1, 3>(TV);
170
171 tV = V4Mul(TV, V4LoadA_Safe(&params->mPreca0_PaddedAligned.x));
172 tV = V4Add(tV, V4Mul(T_YZX_V, V4LoadA_Safe(&params->mPreca1_PaddedAligned.x)));
173 tV = V4Add(tV, V4Mul(T_ZXY_V, V4LoadA_Safe(&params->mPreca2_PaddedAligned.x)));
174 }
175
176 Vec4V t2V;
177 {
178 const Vec4V extents_YZX_V = V4Perm<1, 2, 0, 3>(extentsV);
179 const Vec4V extents_ZXY_V = V4Perm<2, 0, 1, 3>(extentsV);
180
181 t2V = V4Mul(extentsV, V4LoadA_Safe(&params->mPreca0b_PaddedAligned.x));
182 t2V = V4Add(t2V, V4Mul(extents_YZX_V, V4LoadA_Safe(&params->mPreca1b_PaddedAligned.x)));
183 t2V = V4Add(t2V, V4Mul(extents_ZXY_V, V4LoadA_Safe(&params->mPreca2b_PaddedAligned.x)));
184 t2V = V4Add(t2V, V4LoadA_Safe(&params->mBoxExtents_PaddedAligned.x));
185 }
186
187 {
188 const Vec4V abstV = V4Abs(tV);
189 const BoolV resB = V4IsGrtr(abstV, t2V);
190 const PxU32 test = BGetBitMask(resB);
191 if(test&7)
192 return 0;
193 }
194 return 1;
195 }
196#endif // GU_BV4_USE_SLABS
197
198#endif // GU_BV4_BOX_BOX_OVERLAP_TEST_H
#define PX_RESTRICT
Definition PxPreprocessor.h:355
#define PX_FORCE_INLINE
Definition PxPreprocessor.h:335
Definition GuBV4_BoxOverlap_Internal.h:79