RavEngine
Loading...
Searching...
No Matches
GuBV4_Internal.h
1// Redistribution and use in source and binary forms, with or without
2// modification, are permitted provided that the following conditions
3// are met:
4// * Redistributions of source code must retain the above copyright
5// notice, this list of conditions and the following disclaimer.
6// * Redistributions in binary form must reproduce the above copyright
7// notice, this list of conditions and the following disclaimer in the
8// documentation and/or other materials provided with the distribution.
9// * Neither the name of NVIDIA CORPORATION nor the names of its
10// contributors may be used to endorse or promote products derived
11// from this software without specific prior written permission.
12//
13// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ''AS IS'' AND ANY
14// EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
15// IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
16// PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
17// CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
18// EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
19// PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
20// PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
21// OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
22// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
23// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
24//
25// Copyright (c) 2008-2022 NVIDIA Corporation. All rights reserved.
26// Copyright (c) 2004-2008 AGEIA Technologies, Inc. All rights reserved.
27// Copyright (c) 2001-2004 NovodeX AG. All rights reserved.
28
29#ifndef GU_BV4_INTERNAL_H
30#define GU_BV4_INTERNAL_H
31
32#include "foundation/PxFPU.h"
33
34 // PT: the general structure is that there is a root "process stream" function which is the entry point for the query.
35 // It then calls "process node" functions for each traversed node, except for the Slabs-based raycast versions that deal
36 // with 4 nodes at a time within the "process stream" function itself. When a leaf is found, "doLeafTest" functors
37 // passed to the "process stream" entry point are called.
38#ifdef GU_BV4_USE_SLABS
39 // PT: Linux tries to compile templates even when they're not used so I had to wrap them all with defines to avoid build errors. Blame the only platform that does this.
40 #ifdef GU_BV4_PROCESS_STREAM_NO_ORDER
41 template<class LeafTestT, class ParamsT>
42 PX_FORCE_INLINE PxIntBool processStreamNoOrder(const BV4Tree& tree, ParamsT* PX_RESTRICT params)
43 {
44 if(tree.mQuantized)
45 return BV4_ProcessStreamSwizzledNoOrderQ<LeafTestT, ParamsT>(reinterpret_cast<const BVDataPackedQ*>(tree.mNodes), tree.mInitData, params);
46 else
47 return BV4_ProcessStreamSwizzledNoOrderNQ<LeafTestT, ParamsT>(reinterpret_cast<const BVDataPackedNQ*>(tree.mNodes), tree.mInitData, params);
48 }
49 #endif
50
51 #ifdef GU_BV4_PROCESS_STREAM_ORDERED
52 template<class LeafTestT, class ParamsT>
53 PX_FORCE_INLINE void processStreamOrdered(const BV4Tree& tree, ParamsT* PX_RESTRICT params)
54 {
55 if(tree.mQuantized)
56 BV4_ProcessStreamSwizzledOrderedQ<LeafTestT, ParamsT>(reinterpret_cast<const BVDataPackedQ*>(tree.mNodes), tree.mInitData, params);
57 else
58 BV4_ProcessStreamSwizzledOrderedNQ<LeafTestT, ParamsT>(reinterpret_cast<const BVDataPackedNQ*>(tree.mNodes), tree.mInitData, params);
59 }
60 #endif
61
62 #ifdef GU_BV4_PROCESS_STREAM_RAY_NO_ORDER
63 template<int inflateT, class LeafTestT, class ParamsT>
64 PX_FORCE_INLINE PxIntBool processStreamRayNoOrder(const BV4Tree& tree, ParamsT* PX_RESTRICT params)
65 {
66 if(tree.mQuantized)
67 return BV4_ProcessStreamKajiyaNoOrderQ<inflateT, LeafTestT, ParamsT>(reinterpret_cast<const BVDataPackedQ*>(tree.mNodes), tree.mInitData, params);
68 else
69 return BV4_ProcessStreamKajiyaNoOrderNQ<inflateT, LeafTestT, ParamsT>(reinterpret_cast<const BVDataPackedNQ*>(tree.mNodes), tree.mInitData, params);
70 }
71 #endif
72
73 #ifdef GU_BV4_PROCESS_STREAM_RAY_ORDERED
74 template<int inflateT, class LeafTestT, class ParamsT>
75 PX_FORCE_INLINE void processStreamRayOrdered(const BV4Tree& tree, ParamsT* PX_RESTRICT params)
76 {
77 if(tree.mQuantized)
78 BV4_ProcessStreamKajiyaOrderedQ<inflateT, LeafTestT, ParamsT>(reinterpret_cast<const BVDataPackedQ*>(tree.mNodes), tree.mInitData, params);
79 else
80 BV4_ProcessStreamKajiyaOrderedNQ<inflateT, LeafTestT, ParamsT>(reinterpret_cast<const BVDataPackedNQ*>(tree.mNodes), tree.mInitData, params);
81 }
82 #endif
83#else
84 #define processStreamNoOrder BV4_ProcessStreamNoOrder
85 #define processStreamOrdered BV4_ProcessStreamOrdered2
86 #define processStreamRayNoOrder(a, b) BV4_ProcessStreamNoOrder<b>
87 #define processStreamRayOrdered(a, b) BV4_ProcessStreamOrdered2<b>
88#endif
89
90#ifndef GU_BV4_USE_SLABS
91#ifdef GU_BV4_PRECOMPUTED_NODE_SORT
92 // PT: see http://www.codercorner.com/blog/?p=734
93
94 // PT: TODO: refactor with dup in bucket pruner (TA34704)
95 PX_FORCE_INLINE PxU32 computeDirMask(const PxVec3& dir)
96 {
97 // XYZ
98 // ---
99 // --+
100 // -+-
101 // -++
102 // +--
103 // +-+
104 // ++-
105 // +++
106
107 const PxU32 X = PX_IR(dir.x)>>31;
108 const PxU32 Y = PX_IR(dir.y)>>31;
109 const PxU32 Z = PX_IR(dir.z)>>31;
110 const PxU32 bitIndex = Z|(Y<<1)|(X<<2);
111 return 1u<<bitIndex;
112 }
113
114 // 0 0 0 PP PN NP NN 0 1 2 3
115 // 0 0 1 PP PN NN NP 0 1 3 2
116 // 0 1 0 PN PP NP NN 1 0 2 3
117 // 0 1 1 PN PP NN NP 1 0 3 2
118 // 1 0 0 NP NN PP PN 2 3 0 1
119 // 1 0 1 NN NP PP PN 3 2 0 1
120 // 1 1 0 NP NN PN PP 2 3 1 0
121 // 1 1 1 NN NP PN PP 3 2 1 0
122 static const PxU8 order[] = {
123 0,1,2,3,
124 0,1,3,2,
125 1,0,2,3,
126 1,0,3,2,
127 2,3,0,1,
128 3,2,0,1,
129 2,3,1,0,
130 3,2,1,0,
131 };
132
133 PX_FORCE_INLINE PxU32 decodePNS(const BVDataPacked* PX_RESTRICT node, const PxU32 dirMask)
134 {
135 const PxU32 bit0 = (node[0].decodePNSNoShift() & dirMask) ? 1u : 0;
136 const PxU32 bit1 = (node[1].decodePNSNoShift() & dirMask) ? 1u : 0;
137 const PxU32 bit2 = (node[2].decodePNSNoShift() & dirMask) ? 1u : 0; //### potentially reads past the end of the stream here!
138 return bit2|(bit1<<1)|(bit0<<2);
139 }
140#endif // GU_BV4_PRECOMPUTED_NODE_SORT
141
142 #define PNS_BLOCK(i, a, b, c, d) \
143 case i: \
144 { \
145 if(code & (1<<a)) { stack[nb++] = node[a].getChildData(); } \
146 if(code & (1<<b)) { stack[nb++] = node[b].getChildData(); } \
147 if(code & (1<<c)) { stack[nb++] = node[c].getChildData(); } \
148 if(code & (1<<d)) { stack[nb++] = node[d].getChildData(); } \
149 }break;
150
151 #define PNS_BLOCK1(i, a, b, c, d) \
152 case i: \
153 { \
154 stack[nb] = node[a].getChildData(); nb += (code & (1<<a))?1:0; \
155 stack[nb] = node[b].getChildData(); nb += (code & (1<<b))?1:0; \
156 stack[nb] = node[c].getChildData(); nb += (code & (1<<c))?1:0; \
157 stack[nb] = node[d].getChildData(); nb += (code & (1<<d))?1:0; \
158 }break;
159
160 #define PNS_BLOCK2(a, b, c, d) { \
161 if(code & (1<<a)) { stack[nb++] = node[a].getChildData(); } \
162 if(code & (1<<b)) { stack[nb++] = node[b].getChildData(); } \
163 if(code & (1<<c)) { stack[nb++] = node[c].getChildData(); } \
164 if(code & (1<<d)) { stack[nb++] = node[d].getChildData(); } } \
165
166 template<class LeafTestT, class ParamsT>
167 static PxIntBool BV4_ProcessStreamNoOrder(const BVDataPacked* PX_RESTRICT node, PxU32 initData, ParamsT* PX_RESTRICT params)
168 {
169 const BVDataPacked* root = node;
170
171 PxU32 nb=1;
172 PxU32 stack[GU_BV4_STACK_SIZE];
173 stack[0] = initData;
174
175 do
176 {
177 const PxU32 childData = stack[--nb];
178 node = root + getChildOffset(childData);
179 const PxU32 nodeType = getChildType(childData);
180
181 if(nodeType>1 && BV4_ProcessNodeNoOrder<LeafTestT, 3>(stack, nb, node, params))
182 return 1;
183 if(nodeType>0 && BV4_ProcessNodeNoOrder<LeafTestT, 2>(stack, nb, node, params))
184 return 1;
185 if(BV4_ProcessNodeNoOrder<LeafTestT, 1>(stack, nb, node, params))
186 return 1;
187 if(BV4_ProcessNodeNoOrder<LeafTestT, 0>(stack, nb, node, params))
188 return 1;
189
190 }while(nb);
191
192 return 0;
193 }
194
195 template<class LeafTestT, class ParamsT>
196 static void BV4_ProcessStreamOrdered(const BVDataPacked* PX_RESTRICT node, PxU32 initData, ParamsT* PX_RESTRICT params)
197 {
198 const BVDataPacked* root = node;
199
200 PxU32 nb=1;
201 PxU32 stack[GU_BV4_STACK_SIZE];
202 stack[0] = initData;
203
204 const PxU32 dirMask = computeDirMask(params->mLocalDir)<<3;
205
206 do
207 {
208 const PxU32 childData = stack[--nb];
209 node = root + getChildOffset(childData);
210
211 const PxU8* PX_RESTRICT ord = order + decodePNS(node, dirMask)*4;
212 const PxU32 limit = 2 + getChildType(childData);
213
214 BV4_ProcessNodeOrdered<LeafTestT>(stack, nb, node, params, ord[0], limit);
215 BV4_ProcessNodeOrdered<LeafTestT>(stack, nb, node, params, ord[1], limit);
216 BV4_ProcessNodeOrdered<LeafTestT>(stack, nb, node, params, ord[2], limit);
217 BV4_ProcessNodeOrdered<LeafTestT>(stack, nb, node, params, ord[3], limit);
218 }while(Nb);
219 }
220
221 // Alternative, experimental version using PNS
222 template<class LeafTestT, class ParamsT>
223 static void BV4_ProcessStreamOrdered2(const BVDataPacked* PX_RESTRICT node, PxU32 initData, ParamsT* PX_RESTRICT params)
224 {
225 const BVDataPacked* root = node;
226
227 PxU32 nb=1;
228 PxU32 stack[GU_BV4_STACK_SIZE];
229 stack[0] = initData;
230
231 const PxU32 X = PX_IR(params->mLocalDir_Padded.x)>>31;
232 const PxU32 Y = PX_IR(params->mLocalDir_Padded.y)>>31;
233 const PxU32 Z = PX_IR(params->mLocalDir_Padded.z)>>31;
234 const PxU32 bitIndex = 3+(Z|(Y<<1)|(X<<2));
235 const PxU32 dirMask = 1u<<bitIndex;
236
237 do
238 {
239 const PxU32 childData = stack[--nb];
240 node = root + getChildOffset(childData);
241 const PxU32 nodeType = getChildType(childData);
242
243 PxU32 code = 0;
244 BV4_ProcessNodeOrdered2<LeafTestT, 0>(code, node, params);
245 BV4_ProcessNodeOrdered2<LeafTestT, 1>(code, node, params);
246 if(nodeType>0)
247 BV4_ProcessNodeOrdered2<LeafTestT, 2>(code, node, params);
248 if(nodeType>1)
249 BV4_ProcessNodeOrdered2<LeafTestT, 3>(code, node, params);
250
251 if(code)
252 {
253 // PT: TODO: check which implementation is best on each platform (TA34704)
254#define FOURTH_TEST // Version avoids computing the PNS index, and also avoids all non-constant shifts. Full of branches though. Fastest on Win32.
255#ifdef FOURTH_TEST
256 {
257 if(node[0].decodePNSNoShift() & dirMask) // Bit2
258 {
259 if(node[1].decodePNSNoShift() & dirMask) // Bit1
260 {
261 if(node[2].decodePNSNoShift() & dirMask) // Bit0
262 PNS_BLOCK2(3,2,1,0) // 7
263 else
264 PNS_BLOCK2(2,3,1,0) // 6
265 }
266 else
267 {
268 if(node[2].decodePNSNoShift() & dirMask) // Bit0
269 PNS_BLOCK2(3,2,0,1) // 5
270 else
271 PNS_BLOCK2(2,3,0,1) // 4
272 }
273 }
274 else
275 {
276 if(node[1].decodePNSNoShift() & dirMask) // Bit1
277 {
278 if(node[2].decodePNSNoShift() & dirMask) // Bit0
279 PNS_BLOCK2(1,0,3,2) // 3
280 else
281 PNS_BLOCK2(1,0,2,3) // 2
282 }
283 else
284 {
285 if(node[2].decodePNSNoShift() & dirMask) // Bit0
286 PNS_BLOCK2(0,1,3,2) // 1
287 else
288 PNS_BLOCK2(0,1,2,3) // 0
289 }
290 }
291 }
292#endif
293 }
294 }while(nb);
295 }
296#endif // GU_BV4_USE_SLABS
297
298#endif // GU_BV4_INTERNAL_H
#define PX_RESTRICT
Definition PxPreprocessor.h:355
#define PX_FORCE_INLINE
Definition PxPreprocessor.h:335