28#ifdef NV_FLOW_CPU_SHADER
29#undef NV_FLOW_SHADER_TYPES_H
30#undef NV_FLOW_SHADER_HLSLI
31#undef NV_FLOW_RAY_MARCH_PARAMS_H
32#undef NV_FLOW_RAY_MARCH_HLSLI
33#undef NV_FLOW_RAY_MARCH_COMMON_HLSLI
37#define NV_FLOW_CPU_SHADER_DISABLE
39#ifndef NV_FLOW_RESOURCE_CPU_H
40#define NV_FLOW_RESOURCE_CPU_H
42#include "NvFlowContext.h"
47typedef NvFlowUint NvFlowCPU_Uint;
62NV_FLOW_INLINE
int NvFlowCPU_max(
int a,
int b)
67NV_FLOW_INLINE
int NvFlowCPU_min(
int a,
int b)
72NV_FLOW_INLINE
float NvFlowCPU_round(
float v)
77NV_FLOW_INLINE
float NvFlowCPU_abs(
float v)
82NV_FLOW_INLINE
float NvFlowCPU_floor(
float v)
87NV_FLOW_INLINE
int NvFlowCPU_abs(
int v)
89 return v < 0 ? -v : v;
92NV_FLOW_INLINE
float NvFlowCPU_sqrt(
float v)
97NV_FLOW_INLINE
float NvFlowCPU_exp(
float v)
102NV_FLOW_INLINE
float NvFlowCPU_pow(
float a,
float b)
107NV_FLOW_INLINE
float NvFlowCPU_log2(
float v)
112NV_FLOW_INLINE
float NvFlowCPU_min(
float a,
float b)
115 return a < b ? a : b;
118NV_FLOW_INLINE
float NvFlowCPU_max(
float a,
float b)
121 return a > b ? a : b;
124NV_FLOW_INLINE
float NvFlowCPU_clamp(
float v,
float min,
float max)
126 return NvFlowCPU_max(min, NvFlowCPU_min(v, max));
153 NvFlowCPU_Float2& operator+=(
const float& rhs) { x += rhs; y += rhs;
return *
this; }
154 NvFlowCPU_Float2& operator-=(
const float& rhs) { x -= rhs; y -= rhs;
return *
this; }
155 NvFlowCPU_Float2& operator*=(
const float& rhs) { x *= rhs; y *= rhs;
return *
this; }
156 NvFlowCPU_Float2& operator/=(
const float& rhs) { x /= rhs; y /= rhs;
return *
this; }
195 NvFlowCPU_Float3& operator+=(
const float& rhs) { x += rhs; y += rhs; z += rhs;
return *
this; }
196 NvFlowCPU_Float3& operator-=(
const float& rhs) { x -= rhs; y -= rhs; z -= rhs;
return *
this; }
197 NvFlowCPU_Float3& operator*=(
const float& rhs) { x *= rhs; y *= rhs; z *= rhs;
return *
this; }
198 NvFlowCPU_Float3& operator/=(
const float& rhs) { x /= rhs; y /= rhs; z /= rhs;
return *
this; }
218 return sqrtf(v.x * v.x + v.y * v.y + v.z * v.z);
223 return NvFlowCPU_Float3(NvFlowCPU_max(a.x, b.x), NvFlowCPU_max(a.y, b.y), NvFlowCPU_max(a.z, b.z));
228 return NvFlowCPU_Float3(NvFlowCPU_min(a.x, b.x), NvFlowCPU_min(a.y, b.y), NvFlowCPU_min(a.z, b.z));
233 return a.x * b.x + a.y * b.y + a.z * b.z;
238 float length = NvFlowCPU_length(v);
251 NvFlowCPU_Float4(
float x,
float y,
float z,
float w) : x(x), y(y), z(z), w(w) {}
256 float& r() {
return *((
float*)
this); }
261 const float& r()
const {
return *((
const float*)
this); }
279 NvFlowCPU_Float4& operator*=(
const float& rhs) { x *= rhs; y *= rhs; z *= rhs; w *= rhs;
return *
this; }
286 return a.x * b.x + a.y * b.y + a.z * b.z + a.w * b.w;
291 return NvFlowCPU_Float4(NvFlowCPU_max(a.x, b.x), NvFlowCPU_max(a.y, b.y), NvFlowCPU_max(a.z, b.z), NvFlowCPU_max(a.w, b.w));
296 return NvFlowCPU_Float4(NvFlowCPU_min(a.x, b.x), NvFlowCPU_min(a.y, b.y), NvFlowCPU_min(a.z, b.z), NvFlowCPU_min(a.w, b.w));
302 v.x == 0.f ? 0.f : (v.x < 0.f ? -1.f : +1.f),
303 v.y == 0.f ? 0.f : (v.y < 0.f ? -1.f : +1.f),
304 v.z == 0.f ? 0.f : (v.z < 0.f ? -1.f : +1.f),
305 v.w == 0.f ? 0.f : (v.w < 0.f ? -1.f : +1.f)
325 { A.x.x * x.x + A.x.y * x.y + A.x.z * x.z + A.x.w * x.w },
326 { A.y.x * x.x + A.y.y * x.y + A.y.z * x.z + A.y.w * x.w },
327 { A.z.x * x.x + A.z.y * x.y + A.z.z * x.z + A.z.w * x.w },
328 { A.w.x * x.x + A.w.y * x.y + A.w.z * x.z + A.w.w * x.w }
352NV_FLOW_INLINE NvFlowCPU_Float2::NvFlowCPU_Float2(
const NvFlowCPU_Int2& rhs) : x(float(rhs.x)), y(float(rhs.y)) {}
356 return NvFlowCPU_Int2(NvFlowCPU_max(a.x, b.x), NvFlowCPU_max(a.y, b.y));
361 return NvFlowCPU_Int2(NvFlowCPU_min(a.x, b.x), NvFlowCPU_min(a.y, b.y));
374 int& r() {
return *((
int*)
this); }
402 return NvFlowCPU_Int3(NvFlowCPU_max(a.x, b.x), NvFlowCPU_max(a.y, b.y), NvFlowCPU_max(a.z, b.z));
407 return NvFlowCPU_Int3(NvFlowCPU_min(a.x, b.x), NvFlowCPU_min(a.y, b.y), NvFlowCPU_min(a.z, b.z));
415 NvFlowCPU_Int4(
int x,
int y,
int z,
int w) : x(x), y(y), z(z), w(w) {}
422 int& r() {
return *((
int*)
this); }
427 const int& r()
const {
return *((
const int*)
this); }
442NV_FLOW_INLINE NvFlowCPU_Int2::NvFlowCPU_Int2(
const NvFlowCPU_Uint2& rhs) : x(rhs.x), y(rhs.y) {}
449 NvFlowCPU_Uint3(NvFlowUint x, NvFlowUint y, NvFlowUint z) : x(x), y(y), z(z) {}
453 NvFlowCPU_Uint& r() {
return *((NvFlowCPU_Uint*)
this); }
481 NvFlowUint x, y, z, w;
484 NvFlowCPU_Uint4(NvFlowUint x, NvFlowUint y, NvFlowUint z, NvFlowUint w) : x(x), y(y), z(z), w(w) {}
500NV_FLOW_INLINE NvFlowCPU_Float3::NvFlowCPU_Float3(
const NvFlowCPU_Int3& v) : x(float(v.x)), y(float(v.y)), z(float(v.z)) {}
502NV_FLOW_INLINE NvFlowCPU_Int3::NvFlowCPU_Int3(
const NvFlowCPU_Uint3& v) : x(int(v.x)), y(int(v.y)), z(int(v.z)) {}
503NV_FLOW_INLINE NvFlowCPU_Int3::NvFlowCPU_Int3(
const NvFlowCPU_Float3& v) : x(int(v.x)), y(int(v.y)), z(int(v.z)) {}
505NV_FLOW_INLINE NvFlowCPU_Uint3::NvFlowCPU_Uint3(
const NvFlowCPU_Int3& v) : x(int(v.x)), y(int(v.y)), z(int(v.z)) {}
507NV_FLOW_INLINE NvFlowCPU_Int4::NvFlowCPU_Int4(
const NvFlowCPU_Uint4& rhs) : x(int(rhs.x)), y(int(rhs.y)), z(int(rhs.z)), w(int(rhs.w)) {}
512NV_FLOW_INLINE
float NvFlowCPU_asfloat(NvFlowUint v) {
return *((
float*)&v);}
517NV_FLOW_INLINE
float NvFlowCPU_asfloat(
int v) {
return *((
float*)&v);}
522NV_FLOW_INLINE NvFlowUint NvFlowCPU_asuint(
float v) {
return *((NvFlowUint*)&v);}
527NV_FLOW_INLINE
int NvFlowCPU_asint(
float v) {
return *((
int*)&v);}
532 NvFlowUint64 sizeInBytes;
533 NvFlowUint elementSizeInBytes;
534 NvFlowUint elementCount;
549 data = (
const T*)resource->data;
561 data = (
const T*)resource->data;
562 count = resource->elementCount;
565 const T& operator[](
int index) {
566 if (index < 0 || index >=
int(count)) index = 0;
579 data = (T*)resource->data;
580 count = resource->elementCount;
583 T& operator[](
int index) {
584 if (index < 0 || index >=
int(count)) index = 0;
595 desc = resource->samplerDesc;
605 T out_of_bounds = {};
609 data = (
const T*)resource->data;
610 format = resource->format;
611 width = resource->width;
612 memset(&out_of_bounds, 0,
sizeof(out_of_bounds));
619 if (index < 0 || index >=
int(tex.width))
621 return tex.out_of_bounds;
623 return tex.data[index];
629 float posf(
float(tex.width) * pos);
632 if (posf < 0.5f) posf = 0.5f;
633 if (posf >
float(tex.width) - 0.5f) posf = float(tex.width) - 0.5f;
635 int pos0 = int(NvFlowCPU_floor(posf - 0.5f));
636 float f = posf - 0.5f - float(pos0);
639 T sum = of * NvFlowCPU_textureRead(tex, pos0 + 0);
640 sum += f * NvFlowCPU_textureRead(tex, pos0 + 1);
654 data = (T*)resource->data;
655 format = resource->format;
656 width = resource->width;
663 return tex.data[index];
669 tex.data[index] = value;
677 NvFlowCPU_textureWrite(tex, index, value);
692 data = (
const T*)resource->data;
693 format = resource->format;
694 width = resource->width;
695 height = resource->height;
696 memset(&out_of_bounds, 0,
sizeof(out_of_bounds));
703 if (index.x < 0 || index.x >=
int(tex.width) ||
704 index.y < 0 || index.y >=
int(tex.height))
706 return tex.out_of_bounds;
708 return tex.data[index.y * tex.width + index.x];
717 if (posf.x < 0.5f) posf.x = 0.5f;
718 if (posf.x >
float(tex.width) - 0.5f) posf.x = float(tex.width) - 0.5f;
719 if (posf.y < 0.5f) posf.y = 0.5f;
720 if (posf.y >
float(tex.height) - 0.5f) posf.y = float(tex.height) - 0.5f;
726 T sum = of.x * of.y * NvFlowCPU_textureRead(tex, pos00 +
NvFlowCPU_Int2(0, 0));
727 sum += f.x * of.y * NvFlowCPU_textureRead(tex, pos00 +
NvFlowCPU_Int2(1, 0));
728 sum += of.x * f.y * NvFlowCPU_textureRead(tex, pos00 +
NvFlowCPU_Int2(0, 1));
729 sum += f.x * f.y * NvFlowCPU_textureRead(tex, pos00 +
NvFlowCPU_Int2(1, 1));
744 data = (T*)resource->data;
745 format = resource->format;
746 width = resource->width;
747 height = resource->height;
754 return tex.data[index.y * tex.width + index.x];
760 tex.data[index.y * tex.width + index.x] = value;
768 NvFlowCPU_textureWrite(tex, index, value);
787 data = (
const T*)resource->data;
788 format = resource->format;
789 width = resource->width;
790 height = resource->height;
791 depth = resource->depth;
792 memset(&out_of_bounds, 0,
sizeof(out_of_bounds));
802 if (index.x < 0 || index.x >=
int(tex.width) ||
803 index.y < 0 || index.y >=
int(tex.height) ||
804 index.z < 0 || index.z >=
int(tex.depth))
806 return tex.out_of_bounds;
808 return tex.data[(index.z * tex.height + index.y) * tex.width + index.x];
817 if (posf.x < 0.5f) posf.x = 0.5f;
818 if (posf.x >
float(tex.width) - 0.5f) posf.x = float(tex.width) - 0.5f;
819 if (posf.y < 0.5f) posf.y = 0.5f;
820 if (posf.y >
float(tex.height) - 0.5f) posf.y = float(tex.height) - 0.5f;
821 if (posf.z < 0.5f) posf.z = 0.5f;
822 if (posf.z >
float(tex.depth) - 0.5f) posf.z = float(tex.depth) - 0.5f;
842 if (pos000.x >= 0 && pos000.y >= 0 && pos000.z >= 0 &&
843 pos000.x <=
int(tex.width - 2) && pos000.y <=
int(tex.height - 2) && pos000.z <=
int(tex.depth - 2))
845 NvFlowUint idx000 = pos000.z * tex.wh + pos000.y * tex.width + pos000.x;
846 sum = wl.x * tex.data[idx000];
847 sum += wl.y * tex.data[idx000 + 1u];
848 sum += wl.z * tex.data[idx000 + tex.width];
849 sum += wl.w * tex.data[idx000 + 1u + tex.width];
850 sum += wh.x * tex.data[idx000 + tex.wh];
851 sum += wh.y * tex.data[idx000 + 1u + tex.wh];
852 sum += wh.z * tex.data[idx000 + tex.width + tex.wh];
853 sum += wh.w * tex.data[idx000 + 1u + tex.width + tex.wh];
857 sum = wl.x * NvFlowCPU_textureRead(tex, pos000.rgb() +
NvFlowCPU_Int3(0, 0, 0));
858 sum += wl.y * NvFlowCPU_textureRead(tex, pos000.rgb() +
NvFlowCPU_Int3(1, 0, 0));
859 sum += wl.z * NvFlowCPU_textureRead(tex, pos000.rgb() +
NvFlowCPU_Int3(0, 1, 0));
860 sum += wl.w * NvFlowCPU_textureRead(tex, pos000.rgb() +
NvFlowCPU_Int3(1, 1, 0));
861 sum += wh.x * NvFlowCPU_textureRead(tex, pos000.rgb() +
NvFlowCPU_Int3(0, 0, 1));
862 sum += wh.y * NvFlowCPU_textureRead(tex, pos000.rgb() +
NvFlowCPU_Int3(1, 0, 1));
863 sum += wh.z * NvFlowCPU_textureRead(tex, pos000.rgb() +
NvFlowCPU_Int3(0, 1, 1));
864 sum += wh.w * NvFlowCPU_textureRead(tex, pos000.rgb() +
NvFlowCPU_Int3(1, 1, 1));
880 data = (T*)resource->data;
881 format = resource->format;
882 width = resource->width;
883 height = resource->height;
884 depth = resource->depth;
891 return tex.data[(index.z * tex.height + index.y) * tex.width + index.x];
897 tex.data[(index.z * tex.height + index.y) * tex.width + index.x] = value;
905 NvFlowCPU_textureWrite(tex, index, value);
912 ((std::atomic<T>*)&tex.data[(index.z * tex.height + index.y) * tex.width + index.x])->fetch_add(value);
918 ((std::atomic<T>*)&tex.data[(index.z * tex.height + index.y) * tex.width + index.x])->fetch_min(value);
924 ((std::atomic<T>*)&tex.data[(index.z * tex.height + index.y) * tex.width + index.x])->fetch_or(value);
930 ((std::atomic<T>*)&tex.data[(index.z * tex.height + index.y) * tex.width + index.x])->fetch_and(value);
940NV_FLOW_FORCE_INLINE
void NvFlowCPU_swrite(
int _groupshared_pass,
int _groupshared_sync_count,
NvFlowCPU_Groupshared<T>& g,
const T& value)
942 if (_groupshared_pass == _groupshared_sync_count)
954template <
class T,
unsigned int arraySize>
960template <
class T,
unsigned int arraySize>
963 if (_groupshared_pass == _groupshared_sync_count)
965 g.data[index] = value;
969template <
class T,
unsigned int arraySize>
972 return g.data[index];
Definition NvFlowResourceCPU.h:544
Definition NvFlowResourceCPU.h:130
Definition NvFlowResourceCPU.h:173
Definition NvFlowResourceCPU.h:247
Definition NvFlowResourceCPU.h:315
Definition NvFlowResourceCPU.h:956
Definition NvFlowResourceCPU.h:935
Definition NvFlowResourceCPU.h:333
Definition NvFlowResourceCPU.h:365
Definition NvFlowResourceCPU.h:411
Definition NvFlowResourceCPU.h:573
Definition NvFlowResourceCPU.h:647
Definition NvFlowResourceCPU.h:736
Definition NvFlowResourceCPU.h:871
Definition NvFlowResourceCPU.h:530
Definition NvFlowResourceCPU.h:590
Definition NvFlowResourceCPU.h:555
Definition NvFlowResourceCPU.h:601
Definition NvFlowResourceCPU.h:683
Definition NvFlowResourceCPU.h:774
Definition NvFlowResourceCPU.h:434
Definition NvFlowResourceCPU.h:445
Definition NvFlowResourceCPU.h:480
Definition NvFlowContext.h:169