RavEngine
Loading...
Searching...
No Matches
GuBV4_Slabs.h
1// Redistribution and use in source and binary forms, with or without
2// modification, are permitted provided that the following conditions
3// are met:
4// * Redistributions of source code must retain the above copyright
5// notice, this list of conditions and the following disclaimer.
6// * Redistributions in binary form must reproduce the above copyright
7// notice, this list of conditions and the following disclaimer in the
8// documentation and/or other materials provided with the distribution.
9// * Neither the name of NVIDIA CORPORATION nor the names of its
10// contributors may be used to endorse or promote products derived
11// from this software without specific prior written permission.
12//
13// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ''AS IS'' AND ANY
14// EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
15// IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
16// PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
17// CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
18// EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
19// PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
20// PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
21// OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
22// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
23// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
24//
25// Copyright (c) 2008-2022 NVIDIA Corporation. All rights reserved.
26// Copyright (c) 2004-2008 AGEIA Technologies, Inc. All rights reserved.
27// Copyright (c) 2001-2004 NovodeX AG. All rights reserved.
28
29#ifndef GU_BV4_SLABS_H
30#define GU_BV4_SLABS_H
31
32#include "foundation/PxFPU.h"
33#include "GuBV4_Common.h"
34
35#ifdef GU_BV4_USE_SLABS
36
37 // PT: contains code for tree-traversal using the swizzled format.
38 // PT: ray traversal based on Kay & Kajiya's slab intersection code, but using SIMD to do 4 ray-vs-AABB tests at a time.
39 // PT: other (ordered or unordered) traversals just process one node at a time, similar to the non-swizzled format.
40
41 #define BV4_SLABS_FIX
42 #define BV4_SLABS_SORT
43
44 #define PNS_BLOCK3(a, b, c, d) { \
45 if(code2 & (1<<a)) { stack[nb++] = tn->getChildData(a); } \
46 if(code2 & (1<<b)) { stack[nb++] = tn->getChildData(b); } \
47 if(code2 & (1<<c)) { stack[nb++] = tn->getChildData(c); } \
48 if(code2 & (1<<d)) { stack[nb++] = tn->getChildData(d); } } \
49
50 #define OPC_SLABS_GET_MIN_MAX(i) \
51 const VecI32V minVi = I4LoadXYZW(node->mX[i].mMin, node->mY[i].mMin, node->mZ[i].mMin, 0); \
52 const Vec4V minCoeffV = V4LoadA_Safe(&params->mCenterOrMinCoeff_PaddedAligned.x); \
53 Vec4V minV = V4Mul(Vec4V_From_VecI32V(minVi), minCoeffV); \
54 const VecI32V maxVi = I4LoadXYZW(node->mX[i].mMax, node->mY[i].mMax, node->mZ[i].mMax, 0); \
55 const Vec4V maxCoeffV = V4LoadA_Safe(&params->mExtentsOrMaxCoeff_PaddedAligned.x); \
56 Vec4V maxV = V4Mul(Vec4V_From_VecI32V(maxVi), maxCoeffV); \
57
58 #define OPC_SLABS_GET_CEQ(i) \
59 OPC_SLABS_GET_MIN_MAX(i) \
60 const FloatV HalfV = FLoad(0.5f); \
61 const Vec4V centerV = V4Scale(V4Add(maxV, minV), HalfV); \
62 const Vec4V extentsV = V4Scale(V4Sub(maxV, minV), HalfV);
63
64 #define OPC_SLABS_GET_CE2Q(i) \
65 OPC_SLABS_GET_MIN_MAX(i) \
66 const Vec4V centerV = V4Add(maxV, minV); \
67 const Vec4V extentsV = V4Sub(maxV, minV);
68
69 #define OPC_SLABS_GET_CENQ(i) \
70 const FloatV HalfV = FLoad(0.5f); \
71 const Vec4V minV = V4LoadXYZW(node->mMinX[i], node->mMinY[i], node->mMinZ[i], 0.0f); \
72 const Vec4V maxV = V4LoadXYZW(node->mMaxX[i], node->mMaxY[i], node->mMaxZ[i], 0.0f); \
73 const Vec4V centerV = V4Scale(V4Add(maxV, minV), HalfV); \
74 const Vec4V extentsV = V4Scale(V4Sub(maxV, minV), HalfV);
75
76 #define OPC_SLABS_GET_CE2NQ(i) \
77 const Vec4V minV = V4LoadXYZW(node->mMinX[i], node->mMinY[i], node->mMinZ[i], 0.0f); \
78 const Vec4V maxV = V4LoadXYZW(node->mMaxX[i], node->mMaxY[i], node->mMaxZ[i], 0.0f); \
79 const Vec4V centerV = V4Add(maxV, minV); \
80 const Vec4V extentsV = V4Sub(maxV, minV);
81
82#define OPC_DEQ4(part2xV, part1xV, mMember, minCoeff, maxCoeff) \
83{ \
84 part2xV = V4LoadA(reinterpret_cast<const float*>(tn->mMember)); \
85 part1xV = Vec4V_ReinterpretFrom_VecI32V(VecI32V_And(VecI32V_ReinterpretFrom_Vec4V(part2xV), I4Load(0x0000ffff))); \
86 part1xV = Vec4V_ReinterpretFrom_VecI32V(VecI32V_RightShift(VecI32V_LeftShift(VecI32V_ReinterpretFrom_Vec4V(part1xV),16), 16)); \
87 part1xV = V4Mul(Vec4V_From_VecI32V(VecI32V_ReinterpretFrom_Vec4V(part1xV)), minCoeff); \
88 part2xV = Vec4V_ReinterpretFrom_VecI32V(VecI32V_RightShift(VecI32V_ReinterpretFrom_Vec4V(part2xV), 16)); \
89 part2xV = V4Mul(Vec4V_From_VecI32V(VecI32V_ReinterpretFrom_Vec4V(part2xV)), maxCoeff); \
90}
91
92#define SLABS_INIT\
93 Vec4V maxT4 = V4Load(params->mStabbedFace.mDistance);\
94 const Vec4V rayP = V4LoadU_Safe(&params->mOrigin_Padded.x);\
95 Vec4V rayD = V4LoadU_Safe(&params->mLocalDir_Padded.x);\
96 const VecU32V raySign = V4U32and(VecU32V_ReinterpretFrom_Vec4V(rayD), signMask);\
97 const Vec4V rayDAbs = V4Abs(rayD);\
98 Vec4V rayInvD = Vec4V_ReinterpretFrom_VecU32V(V4U32or(raySign, VecU32V_ReinterpretFrom_Vec4V(V4Max(rayDAbs, epsFloat4))));\
99 rayD = rayInvD;\
100 rayInvD = V4RecipFast(rayInvD);\
101 rayInvD = V4Mul(rayInvD, V4NegMulSub(rayD, rayInvD, twos));\
102 const Vec4V rayPinvD = V4NegMulSub(rayInvD, rayP, zeroes);\
103 const Vec4V rayInvDsplatX = V4SplatElement<0>(rayInvD);\
104 const Vec4V rayInvDsplatY = V4SplatElement<1>(rayInvD);\
105 const Vec4V rayInvDsplatZ = V4SplatElement<2>(rayInvD);\
106 const Vec4V rayPinvDsplatX = V4SplatElement<0>(rayPinvD);\
107 const Vec4V rayPinvDsplatY = V4SplatElement<1>(rayPinvD);\
108 const Vec4V rayPinvDsplatZ = V4SplatElement<2>(rayPinvD);
109
110#define SLABS_TEST\
111 const Vec4V tminxa0 = V4MulAdd(minx4a, rayInvDsplatX, rayPinvDsplatX);\
112 const Vec4V tminya0 = V4MulAdd(miny4a, rayInvDsplatY, rayPinvDsplatY);\
113 const Vec4V tminza0 = V4MulAdd(minz4a, rayInvDsplatZ, rayPinvDsplatZ);\
114 const Vec4V tmaxxa0 = V4MulAdd(maxx4a, rayInvDsplatX, rayPinvDsplatX);\
115 const Vec4V tmaxya0 = V4MulAdd(maxy4a, rayInvDsplatY, rayPinvDsplatY);\
116 const Vec4V tmaxza0 = V4MulAdd(maxz4a, rayInvDsplatZ, rayPinvDsplatZ);\
117 const Vec4V tminxa = V4Min(tminxa0, tmaxxa0);\
118 const Vec4V tmaxxa = V4Max(tminxa0, tmaxxa0);\
119 const Vec4V tminya = V4Min(tminya0, tmaxya0);\
120 const Vec4V tmaxya = V4Max(tminya0, tmaxya0);\
121 const Vec4V tminza = V4Min(tminza0, tmaxza0);\
122 const Vec4V tmaxza = V4Max(tminza0, tmaxza0);\
123 const Vec4V maxOfNeasa = V4Max(V4Max(tminxa, tminya), tminza);\
124 const Vec4V minOfFarsa = V4Min(V4Min(tmaxxa, tmaxya), tmaxza);\
125
126 #define SLABS_TEST2\
127 BoolV ignore4a = V4IsGrtr(epsFloat4, minOfFarsa); /* if tfar is negative, ignore since its a ray, not a line */\
128 ignore4a = BOr(ignore4a, V4IsGrtr(maxOfNeasa, maxT4)); /* if tnear is over maxT, ignore this result */\
129 BoolV resa4 = V4IsGrtr(maxOfNeasa, minOfFarsa); /* if 1 => fail */\
130 resa4 = BOr(resa4, ignore4a);\
131 const PxU32 code = BGetBitMask(resa4);\
132 if(code==15)\
133 continue;
134
135#define SLABS_PNS \
136 if(code2) \
137 { \
138 if(tn->decodePNSNoShift(0) & dirMask) \
139 { \
140 if(tn->decodePNSNoShift(1) & dirMask) \
141 { \
142 if(tn->decodePNSNoShift(2) & dirMask) \
143 PNS_BLOCK3(3,2,1,0) \
144 else \
145 PNS_BLOCK3(2,3,1,0) \
146 } \
147 else \
148 { \
149 if(tn->decodePNSNoShift(2) & dirMask) \
150 PNS_BLOCK3(3,2,0,1) \
151 else \
152 PNS_BLOCK3(2,3,0,1) \
153 } \
154 } \
155 else \
156 { \
157 if(tn->decodePNSNoShift(1) & dirMask) \
158 { \
159 if(tn->decodePNSNoShift(2) & dirMask) \
160 PNS_BLOCK3(1,0,3,2) \
161 else \
162 PNS_BLOCK3(1,0,2,3) \
163 } \
164 else \
165 { \
166 if(tn->decodePNSNoShift(2) & dirMask) \
167 PNS_BLOCK3(0,1,3,2) \
168 else \
169 PNS_BLOCK3(0,1,2,3) \
170 } \
171 } \
172 }
173
174#endif // GU_BV4_USE_SLABS
175
176#endif // GU_BV4_SLABS_H