RavEngine
Loading...
Searching...
No Matches
PxCudaContextManager.h
1// Redistribution and use in source and binary forms, with or without
2// modification, are permitted provided that the following conditions
3// are met:
4// * Redistributions of source code must retain the above copyright
5// notice, this list of conditions and the following disclaimer.
6// * Redistributions in binary form must reproduce the above copyright
7// notice, this list of conditions and the following disclaimer in the
8// documentation and/or other materials provided with the distribution.
9// * Neither the name of NVIDIA CORPORATION nor the names of its
10// contributors may be used to endorse or promote products derived
11// from this software without specific prior written permission.
12//
13// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ''AS IS'' AND ANY
14// EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
15// IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
16// PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
17// CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
18// EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
19// PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
20// PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
21// OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
22// (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
23// OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
24//
25// Copyright (c) 2008-2022 NVIDIA Corporation. All rights reserved.
26
27#ifndef PX_CUDA_CONTEXT_MANAGER_H
28#define PX_CUDA_CONTEXT_MANAGER_H
29
30#include "foundation/PxPreprocessor.h"
31
32#if PX_SUPPORT_GPU_PHYSX
33
34#include "foundation/PxSimpleTypes.h"
35#include "foundation/PxErrorCallback.h"
36#include "foundation/PxFlags.h"
37
38#include "PxCudaTypes.h"
39
40#if !PX_DOXYGEN
41namespace physx
42{
43#endif
44
45class PxCudaContext;
46
48struct PxCudaInteropMode
49{
53 enum Enum
54 {
55 NO_INTEROP = 0,
56 D3D10_INTEROP,
57 D3D11_INTEROP,
58 OGL_INTEROP,
59
60 COUNT
61 };
62};
63
64struct PxCudaInteropRegisterFlag
65{
66 enum Enum
67 {
68 eNONE = 0x00,
69 eREAD_ONLY = 0x01,
70 eWRITE_DISCARD = 0x02,
71 eSURFACE_LDST = 0x04,
72 eTEXTURE_GATHER = 0x08
73 };
74};
75
79class PxDeviceAllocatorCallback
80{
81public:
82
89 virtual bool memAlloc(void** ptr, size_t size) = 0;
90
96 virtual bool memFree(void* ptr) = 0;
97
98protected:
99 virtual ~PxDeviceAllocatorCallback() {}
100};
106typedef PxFlags<PxCudaInteropRegisterFlag::Enum, uint32_t> PxCudaInteropRegisterFlags;
107PX_FLAGS_OPERATORS(PxCudaInteropRegisterFlag::Enum, uint32_t)
108
109
110class PxCudaContextManagerDesc
111{
112public:
130 CUcontext* ctx;
131
139 void* graphicsDevice;
140
148 const char* appGUID;
149
157 PxDeviceAllocatorCallback* deviceAllocator;
158
166 PxCudaInteropMode::Enum interopMode;
167
168 PX_INLINE PxCudaContextManagerDesc() :
169 ctx (NULL),
170 graphicsDevice (NULL),
171 appGUID (NULL),
172 deviceAllocator (NULL),
173 interopMode (PxCudaInteropMode::NO_INTEROP)
174 {
175 }
176};
177
181struct PxKernelIndex
182{
183 PxU32 moduleIndex;
184 const char* functionName;
185};
186
197class PxCudaContextManager
198{
199public:
205 template<typename T>
206 void clearDeviceBufferAsync(T* deviceBuffer, PxU32 numElements, CUstream stream, PxI32 value = 0)
207 {
208 clearDeviceBufferAsyncInternal(deviceBuffer, numElements * sizeof(T), stream, value);
209 }
210
216 template<typename T>
217 void copyDToH(T* hostBuffer, const T* deviceBuffer, PxU32 numElements)
218 {
219 copyDToHInternal(hostBuffer, deviceBuffer, numElements * sizeof(T));
220 }
221
227 template<typename T>
228 void copyHToD(T* deviceBuffer, const T* hostBuffer, PxU32 numElements)
229 {
230 copyHToDInternal(deviceBuffer, hostBuffer, numElements * sizeof(T));
231 }
232
238 template<typename T>
239 void copyDToHAsync(T* hostBuffer, const T* deviceBuffer, PxU32 numElements, CUstream stream)
240 {
241 copyDToHAsyncInternal(hostBuffer, deviceBuffer, numElements * sizeof(T), stream);
242 }
243
249 template<typename T>
250 void copyHToDAsync(T* deviceBuffer, const T* hostBuffer, PxU32 numElements, CUstream stream)
251 {
252 copyHToDAsyncInternal(deviceBuffer, hostBuffer, numElements * sizeof(T), stream);
253 }
254
260 template<typename T>
261 void copyDToDAsync(T* dstDeviceBuffer, const T* srcDeviceBuffer, PxU32 numElements, CUstream stream)
262 {
263 copyDToDAsyncInternal(dstDeviceBuffer, srcDeviceBuffer, numElements * sizeof(T), stream);
264 }
265
271 template<typename T>
272 void allocDeviceBuffer(T*& deviceBuffer, PxU32 numElements, const char* filename = __FILE__, PxI32 line = __LINE__)
273 {
274 void* ptr = allocDeviceBufferInternal(numElements * sizeof(T), filename, line);
275 deviceBuffer = reinterpret_cast<T*>(ptr);
276 }
277
283 template<typename T>
284 T* allocDeviceBuffer(PxU32 numElements, const char* filename = __FILE__, PxI32 line = __LINE__)
285 {
286 void* ptr = allocDeviceBufferInternal(numElements * sizeof(T), filename, line);
287 return reinterpret_cast<T*>(ptr);
288 }
289
295 template<typename T>
296 void freeDeviceBuffer(T*& deviceBuffer)
297 {
298 freeDeviceBufferInternal(deviceBuffer);
299 deviceBuffer = NULL;
300 }
301
309 template<typename T>
310 void allocPinnedHostBuffer(T*& pinnedHostBuffer, PxU32 numElements, const char* filename = __FILE__, PxI32 line = __LINE__)
311 {
312 void* ptr = allocPinnedHostBufferInternal(numElements * sizeof(T), filename, line);
313 pinnedHostBuffer = reinterpret_cast<T*>(ptr);
314 }
315
323 template<typename T>
324 T* allocPinnedHostBuffer(PxU32 numElements, const char* filename = __FILE__, PxI32 line = __LINE__)
325 {
326 void* ptr = allocPinnedHostBufferInternal(numElements * sizeof(T), filename, line);
327 return reinterpret_cast<T*>(ptr);
328 }
329
335 template<typename T>
336 void freePinnedHostBuffer(T*& pinnedHostBuffer)
337 {
338 freePinnedHostBufferInternal(pinnedHostBuffer);
339 pinnedHostBuffer = NULL;
340 }
341
349 virtual CUdeviceptr getMappedDevicePtr(void* pinnedHostBuffer) = 0;
350
360 virtual void acquireContext() = 0;
361
368 virtual void releaseContext() = 0;
369
373 virtual CUcontext getContext() = 0;
374
378 virtual PxCudaContext* getCudaContext() = 0;
379
387 virtual bool contextIsValid() const = 0;
388
389 /* Query CUDA context and device properties, without acquiring context */
390
391 virtual bool supportsArchSM10() const = 0;
392 virtual bool supportsArchSM11() const = 0;
393 virtual bool supportsArchSM12() const = 0;
394 virtual bool supportsArchSM13() const = 0;
395 virtual bool supportsArchSM20() const = 0;
396 virtual bool supportsArchSM30() const = 0;
397 virtual bool supportsArchSM35() const = 0;
398 virtual bool supportsArchSM50() const = 0;
399 virtual bool supportsArchSM52() const = 0;
400 virtual bool supportsArchSM60() const = 0;
401 virtual bool isIntegrated() const = 0;
402 virtual bool canMapHostMemory() const = 0;
403 virtual int getDriverVersion() const = 0;
404 virtual size_t getDeviceTotalMemBytes() const = 0;
405 virtual int getMultiprocessorCount() const = 0;
406 virtual unsigned int getClockRate() const = 0;
407 virtual int getSharedMemPerBlock() const = 0;
408 virtual int getSharedMemPerMultiprocessor() const = 0;
409 virtual unsigned int getMaxThreadsPerBlock() const = 0;
410 virtual const char *getDeviceName() const = 0;
411 virtual CUdevice getDevice() const = 0;
412 virtual PxCudaInteropMode::Enum getInteropMode() const = 0;
413
414 virtual void setUsingConcurrentStreams(bool) = 0;
415 virtual bool getUsingConcurrentStreams() const = 0;
416 /* End query methods that don't require context to be acquired */
417
438 virtual bool registerResourceInCudaGL(CUgraphicsResource& resource, uint32_t buffer, PxCudaInteropRegisterFlags flags = PxCudaInteropRegisterFlags()) = 0;
439
460 virtual bool registerResourceInCudaD3D(CUgraphicsResource& resource, void* resourcePointer, PxCudaInteropRegisterFlags flags = PxCudaInteropRegisterFlags()) = 0;
461
469 virtual bool unregisterResourceInCuda(CUgraphicsResource resource) = 0;
470
478 virtual int usingDedicatedGPU() const = 0;
479
484 virtual CUmodule* getCuModules() = 0;
485
496 virtual void release() = 0;
497
498protected:
499
503 virtual ~PxCudaContextManager() {}
504
505 virtual void* allocDeviceBufferInternal(PxU32 numBytes, const char* filename = NULL, PxI32 line = -1) = 0;
506 virtual void* allocPinnedHostBufferInternal(PxU32 numBytes, const char* filename = NULL, PxI32 line = -1) = 0;
507
508 virtual void freeDeviceBufferInternal(void* deviceBuffer) = 0;
509 virtual void freePinnedHostBufferInternal(void* pinnedHostBuffer) = 0;
510
511 virtual void clearDeviceBufferAsyncInternal(void* deviceBuffer, PxU32 numBytes, CUstream stream, PxI32 value) = 0;
512
513 virtual void copyDToHAsyncInternal(void* hostBuffer, const void* deviceBuffer, PxU32 numBytes, CUstream stream) = 0;
514 virtual void copyHToDAsyncInternal(void* deviceBuffer, const void* hostBuffer, PxU32 numBytes, CUstream stream) = 0;
515 virtual void copyDToDAsyncInternal(void* dstDeviceBuffer, const void* srcDeviceBuffer, PxU32 numBytes, CUstream stream) = 0;
516
517 virtual void copyDToHInternal(void* hostBuffer, const void* deviceBuffer, PxU32 numBytes) = 0;
518 virtual void copyHToDInternal(void* deviceBuffer, const void* hostBuffer, PxU32 numBytes) = 0;
519};
520
521#define PX_DEVICE_ALLOC(cudaContextManager, deviceBuffer, numElements) cudaContextManager->allocDeviceBuffer(deviceBuffer, numElements, __FILE__, __LINE__)
522#define PX_DEVICE_ALLOC_T(T, cudaContextManager, numElements) cudaContextManager->allocDeviceBuffer<T>(numElements, __FILE__, __LINE__)
523#define PX_DEVICE_FREE(cudaContextManager, deviceBuffer) cudaContextManager->freeDeviceBuffer(deviceBuffer);
524
525#define PX_PINNED_HOST_ALLOC(cudaContextManager, pinnedHostBuffer, numElements) cudaContextManager->allocPinnedHostBuffer(pinnedHostBuffer, numElements, __FILE__, __LINE__)
526#define PX_PINNED_HOST_ALLOC_T(T, cudaContextManager, numElements) cudaContextManager->allocPinnedHostBuffer<T>(numElements, __FILE__, __LINE__)
527#define PX_PINNED_HOST_FREE(cudaContextManager, pinnedHostBuffer) cudaContextManager->freePinnedHostBuffer(pinnedHostBuffer);
528
532class PxScopedCudaLock
533{
534public:
538 PxScopedCudaLock(PxCudaContextManager& ctx) : mCtx(&ctx)
539 {
540 mCtx->acquireContext();
541 }
542
546 ~PxScopedCudaLock()
547 {
548 mCtx->releaseContext();
549 }
550
551protected:
552
556 PxCudaContextManager* mCtx;
557};
558
559#if !PX_DOXYGEN
560} // namespace physx
561#endif
562
563#endif // PX_SUPPORT_GPU_PHYSX
564#endif
#define PX_INLINE
Definition PxPreprocessor.h:320
Sorts an array of objects in ascending order, assuming that the predicate implements the < operator:
Definition PxBoxController.h:39