diff --git a/WickedEngine/ShaderInterop_BVH.h b/WickedEngine/ShaderInterop_BVH.h index 55b53e057..c67896717 100644 --- a/WickedEngine/ShaderInterop_BVH.h +++ b/WickedEngine/ShaderInterop_BVH.h @@ -2,7 +2,7 @@ #define _SHADERINTEROP_BVH_H_ #include "ShaderInterop.h" -#define BVH_BUILDER_GROUPSIZE 64 +static const uint BVH_BUILDER_GROUPSIZE = 64; CBUFFER(BVHCB, CBSLOT_RENDERER_BVH) { diff --git a/WickedEngine/ShaderInterop_Raytracing.h b/WickedEngine/ShaderInterop_Raytracing.h index fd0d586f2..d68cc4754 100644 --- a/WickedEngine/ShaderInterop_Raytracing.h +++ b/WickedEngine/ShaderInterop_Raytracing.h @@ -25,7 +25,7 @@ struct RaytracingStoredRay uint3 direction_energy; // packed half3 direction | half3 energy uint primitiveID; float2 bary; - uint2 userdata; // vulkan complains about 16-byte padding here so might as well add userdata here and not pack barycentric coords + uint2 color; // packed rgba16 }; diff --git a/WickedEngine/WickedEngine_SHADERS.vcxproj b/WickedEngine/WickedEngine_SHADERS.vcxproj index 90fd4147a..a44d1cac7 100644 --- a/WickedEngine/WickedEngine_SHADERS.vcxproj +++ b/WickedEngine/WickedEngine_SHADERS.vcxproj @@ -641,10 +641,6 @@ Compute 5.0 - - Compute - 5.0 - Compute 5.0 diff --git a/WickedEngine/WickedEngine_SHADERS.vcxproj.filters b/WickedEngine/WickedEngine_SHADERS.vcxproj.filters index 8d8436031..bfc236de3 100644 --- a/WickedEngine/WickedEngine_SHADERS.vcxproj.filters +++ b/WickedEngine/WickedEngine_SHADERS.vcxproj.filters @@ -747,9 +747,6 @@ CS - - CS - CS diff --git a/WickedEngine/globals.hlsli b/WickedEngine/globals.hlsli index 1242da61b..4be96e385 100644 --- a/WickedEngine/globals.hlsli +++ b/WickedEngine/globals.hlsli @@ -425,6 +425,21 @@ inline float4 unpack_rgba(in uint value) return retVal; } +inline uint2 pack_half3(in float3 value) +{ + uint2 retVal = 0; + retVal.x = f32tof16(value.x) | (f32tof16(value.y) << 16); + retVal.y = f32tof16(value.z); + return retVal; +} +inline float3 unpack_half3(in uint2 value) +{ + float3 retVal; + retVal.x = f16tof32(value.x); + retVal.y = f16tof32(value.x >> 16); + retVal.z = f16tof32(value.y); + return retVal; +} inline uint2 pack_half4(in float4 value) { uint2 retVal = 0; diff --git a/WickedEngine/raytrace_accumulateCS.hlsl b/WickedEngine/raytrace_accumulateCS.hlsl deleted file mode 100644 index afc332b84..000000000 --- a/WickedEngine/raytrace_accumulateCS.hlsl +++ /dev/null @@ -1,19 +0,0 @@ -#include "globals.hlsli" -#include "ShaderInterop_Raytracing.h" - -TEXTURE2D(sourceTexture, float4, TEXSLOT_ONDEMAND0); - -RWTEXTURE2D(resultTexture, float4, 0); - -[numthreads(RAYTRACING_ACCUMULATE_BLOCKSIZE, RAYTRACING_ACCUMULATE_BLOCKSIZE, 1)] -void main( uint3 DTid : SV_DispatchThreadID ) -{ - if (xTraceAccumulationFactor == 1.0f) - { - resultTexture[DTid.xy] = sourceTexture[DTid.xy]; // naturally the lerp solution below would be enough, but if the result texture is not initialized, it can contain nan that doesn't work well with lerp! - } - else - { - resultTexture[DTid.xy] = lerp(resultTexture[DTid.xy], sourceTexture[DTid.xy], xTraceAccumulationFactor); - } -} diff --git a/WickedEngine/raytrace_launchCS.hlsl b/WickedEngine/raytrace_launchCS.hlsl index 8c763d36b..ca24f268c 100644 --- a/WickedEngine/raytrace_launchCS.hlsl +++ b/WickedEngine/raytrace_launchCS.hlsl @@ -13,15 +13,13 @@ void main( uint3 DTid : SV_DispatchThreadID ) // Compute screen coordinates: float2 uv = float2((DTid.xy + xTracePixelOffset) * xTraceResolution_rcp.xy * 2.0f - 1.0f) * float2(1, -1); - // Target pixel: - uint pixelID = flatten2D(DTid.xy, xTraceResolution.xy); - // Create starting ray: Ray ray = CreateCameraRay(uv); + ray.pixelID = flatten2D(DTid.xy, xTraceResolution.xy); // The launch writes each ray to the pixel location: - rayIndexBuffer[pixelID] = pixelID; - raySortBuffer[pixelID] = CreateRaySortCode(ray); - rayBuffer[pixelID] = CreateStoredRay(ray, pixelID); + rayIndexBuffer[ray.pixelID] = ray.pixelID; + raySortBuffer[ray.pixelID] = CreateRaySortCode(ray); + rayBuffer[ray.pixelID] = CreateStoredRay(ray); } } diff --git a/WickedEngine/raytrace_lightsamplingCS.hlsl b/WickedEngine/raytrace_lightsamplingCS.hlsl index 7c76bb9a9..20ced72a6 100644 --- a/WickedEngine/raytrace_lightsamplingCS.hlsl +++ b/WickedEngine/raytrace_lightsamplingCS.hlsl @@ -4,24 +4,23 @@ RAWBUFFER(counterBuffer_READ, TEXSLOT_ONDEMAND7); STRUCTUREDBUFFER(rayIndexBuffer_READ, uint, TEXSLOT_ONDEMAND8); -STRUCTUREDBUFFER(rayBuffer_READ, RaytracingStoredRay, TEXSLOT_ONDEMAND9); -RWTEXTURE2D(resultTexture, float4, 0); +RWSTRUCTUREDBUFFER(rayBuffer, RaytracingStoredRay, 0); [numthreads(RAYTRACING_TRACE_GROUPSIZE, 1, 1)] void main( uint3 DTid : SV_DispatchThreadID, uint groupIndex : SV_GroupIndex) { // Initialize ray and pixel ID as non-contributing: Ray ray = (Ray)0; - uint pixelID = 0xFFFFFFFF; if (DTid.x < counterBuffer_READ.Load(0)) { // Load the current ray: - LoadRay(rayBuffer_READ[rayIndexBuffer_READ[DTid.x]], ray, pixelID); + const uint rayIndex = rayIndexBuffer_READ[DTid.x]; + LoadRay(rayBuffer[rayIndex], ray); // Compute real pixel coords from flattened: - uint2 coords2D = unflatten2D(pixelID, xTraceResolution.xy); + uint2 coords2D = unflatten2D(ray.pixelID, xTraceResolution.xy); // Compute screen coordinates: float2 uv = float2((coords2D + xTracePixelOffset) * xTraceResolution_rcp.xy * 2.0f - 1.0f) * float2(1, -1); @@ -219,11 +218,9 @@ void main( uint3 DTid : SV_DispatchThreadID, uint groupIndex : SV_GroupIndex) } } - finalResult *= ray.energy; - - resultTexture[coords2D] += float4(max(0, finalResult), 0); + ray.color += max(0, ray.energy * finalResult); + // Store the current ray color: + rayBuffer[rayIndex].color = pack_half3(ray.color); } - - // This shader doesn't export any rays! } \ No newline at end of file diff --git a/WickedEngine/raytrace_primaryCS.hlsl b/WickedEngine/raytrace_primaryCS.hlsl index fde750cef..915508d19 100644 --- a/WickedEngine/raytrace_primaryCS.hlsl +++ b/WickedEngine/raytrace_primaryCS.hlsl @@ -46,15 +46,21 @@ void main( uint3 DTid : SV_DispatchThreadID, uint groupIndex : SV_GroupIndex ) // Initialize ray and pixel ID as non-contributing: Ray ray = (Ray)0; - uint pixelID = 0xFFFFFFFF; + bool ray_active = false; if (DTid.x < counterBuffer_READ.Load(0)) { // Load the current ray: - LoadRay(rayBuffer_READ[rayIndexBuffer_READ[DTid.x]], ray, pixelID); + LoadRay(rayBuffer_READ[rayIndexBuffer_READ[DTid.x]], ray); // Compute real pixel coords from flattened: - uint2 coords2D = unflatten2D(pixelID, xTraceResolution.xy); + uint2 coords2D = unflatten2D(ray.pixelID, xTraceResolution.xy); + + // Pre-clear result texture for first bounce and first accumulation sample: + if (xTraceUserData.x == 1) + { + resultTexture[coords2D] = 0; + } // Compute screen coordinates: float2 uv = float2((coords2D + xTracePixelOffset) * xTraceResolution_rcp.xy * 2.0f - 1.0f) * float2(1, -1); @@ -62,27 +68,25 @@ void main( uint3 DTid : SV_DispatchThreadID, uint groupIndex : SV_GroupIndex ) float seed = xTraceRandomSeed; RayHit hit = TraceScene(ray); - float4 result = float4(max(0, ray.energy * Shade(ray, hit, seed, uv)), 0); + ShadeRay(ray, hit, seed, uv); - // Write pixel color: - if (xTraceUserData.x == 0) // first bounce clears texture + ray_active = any(ray.energy); + + // If the ray is killed or last bounce, we write to accumulation texture: + if (!ray_active || xTraceUserData.y == 1) { - resultTexture[coords2D] = result; - } - else // other bounces accumulate to texture - { - resultTexture[coords2D] += result; + resultTexture[coords2D] = lerp(resultTexture[coords2D], float4(ray.color, 1), xTraceAccumulationFactor); } #ifndef ADVANCED_ALLOCATION - if (any(ray.energy)) + if (ray_active) { // Naive strategy to allocate active rays. Global memory atomics will be performed for every thread: uint dest; counterBuffer_WRITE.InterlockedAdd(0, 1, dest); rayIndexBuffer_WRITE[dest] = dest; raySortBuffer_WRITE[dest] = CreateRaySortCode(ray); - rayBuffer_WRITE[dest] = CreateStoredRay(ray, pixelID); + rayBuffer_WRITE[dest] = CreateStoredRay(ray); } #endif // ADVANCED_ALLOCATION @@ -91,13 +95,12 @@ void main( uint3 DTid : SV_DispatchThreadID, uint groupIndex : SV_GroupIndex ) #ifdef ADVANCED_ALLOCATION - const bool active = any(ray.energy); // does this thread append? const uint bucket = groupIndex / 32; // which bitfield bucket does this thread belong to? const uint threadIndexInBucket = groupIndex % 32; // thread bit offset from bucket start const uint threadMask = 1 << threadIndexInBucket; // thread bit mask in current bucket // Count rays that are still active with a bitmask insertion: - if (active) + if (ray_active) { InterlockedOr(GroupActiveRayMask[bucket], threadMask); } @@ -118,7 +121,7 @@ void main( uint3 DTid : SV_DispatchThreadID, uint groupIndex : SV_GroupIndex ) GroupMemoryBarrierWithGroupSync(); // Finally, write all active rays into global memory: - if (active) + if (ray_active) { // Need to compute prefix-sum of just the active ray count before this thread uint activePrefixSum = 0; @@ -137,10 +140,10 @@ void main( uint3 DTid : SV_DispatchThreadID, uint groupIndex : SV_GroupIndex ) activePrefixSum += countbits(GroupActiveRayMask[i] & prefixMask); } - const uint dest = GroupRayWriteOffset + activePrefixSum - 1; + const uint dest = GroupRayWriteOffset + activePrefixSum - 1; // -1 because activePrefixSum includes current thread, but arrays start from 0! rayIndexBuffer_WRITE[dest] = dest; raySortBuffer_WRITE[dest] = CreateRaySortCode(ray); - rayBuffer_WRITE[dest] = CreateStoredRay(ray, pixelID); // -1 because activePrefixSum includes current thread, but arrays start from 0! + rayBuffer_WRITE[dest] = CreateStoredRay(ray); } #endif // ADVANCED_ALLOCATION } diff --git a/WickedEngine/raytracingHF.hlsli b/WickedEngine/raytracingHF.hlsli index f642c0264..60695407f 100644 --- a/WickedEngine/raytracingHF.hlsli +++ b/WickedEngine/raytracingHF.hlsli @@ -17,27 +17,18 @@ static const float EPSILON = 0.0001f; inline float3 trace_bias_position(in float3 P, in float3 N) { return P + N * EPSILON; // this is the original version - //return P + sign(N) * abs(P * 0.0000002); // this is from https://ndotl.wordpress.com/2018/08/29/baking-artifact-free-lightmaps/ } - -//struct Sphere -//{ -// float3 position; -// float radius; -// float3 albedo; -// float3 specular; -// float emission; -//}; - struct Ray { + uint pixelID; float3 origin; float3 direction; float3 direction_rcp; float3 energy; uint primitiveID; float2 bary; + float3 color; inline void Update() { @@ -65,31 +56,28 @@ inline float CreateRaySortCode(in Ray ray) //return (float)hash; } -inline RaytracingStoredRay CreateStoredRay(in Ray ray, in uint pixelID) +inline RaytracingStoredRay CreateStoredRay(in Ray ray) { RaytracingStoredRay storedray; storedray.origin = ray.origin; - storedray.pixelID = pixelID; + storedray.pixelID = ray.pixelID; storedray.direction_energy = f32tof16(ray.direction) | (f32tof16(ray.energy) << 16); storedray.primitiveID = ray.primitiveID; - //storedray.bary = f32tof16(ray.bary.x) | (f32tof16(ray.bary.y) << 16); storedray.bary = ray.bary; - storedray.userdata = 0; // free to use for something + storedray.color = pack_half3(ray.color); return storedray; } -inline void LoadRay(in RaytracingStoredRay storedray, out Ray ray, out uint pixelID) +inline void LoadRay(in RaytracingStoredRay storedray, out Ray ray) { - pixelID = storedray.pixelID; - + ray.pixelID = storedray.pixelID; ray.origin = storedray.origin; ray.direction = asfloat(f16tof32(storedray.direction_energy)); ray.energy = asfloat(f16tof32(storedray.direction_energy >> 16)); ray.primitiveID = storedray.primitiveID; ray.bary = storedray.bary; - //ray.bary.x = f16tof32(storedray.bary); - //ray.bary.y = f16tof32(storedray.bary >> 16); + ray.color = unpack_half3(storedray.color); ray.Update(); } @@ -99,8 +87,10 @@ inline Ray CreateRay(float3 origin, float3 direction) ray.origin = origin; ray.direction = direction; ray.energy = float3(1, 1, 1); + ray.pixelID = 0xFFFFFFFF; ray.primitiveID = 0xFFFFFFFF; ray.bary = 0; + ray.color = 0; ray.Update(); return ray; } @@ -148,37 +138,6 @@ inline RayHit CreateRayHit() return hit; } -//inline void IntersectGroundPlane(Ray ray, inout RayHit bestHit) -//{ -// // Calculate distance along the ray where the ground plane is intersected -// float t = -ray.origin.y / ray.direction.y; -// if (t > 0 && t < bestHit.distance) -// { -// bestHit.distance = t; -// bestHit.position = ray.origin + t * ray.direction; -// bestHit.normal = float3(0.0f, 1.0f, 0.0f); -// } -//} -// -//inline void IntersectSphere(Ray ray, inout RayHit bestHit, Sphere sphere) -//{ -// // Calculate distance along the ray where the sphere is intersected -// float3 d = ray.origin - sphere.position; -// float p1 = -dot(ray.direction, d); -// float p2sqr = p1 * p1 - dot(d, d) + sphere.radius * sphere.radius; -// if (p2sqr < 0) -// return; -// float p2 = sqrt(p2sqr); -// float t = p1 - p2 > 0 ? p1 - p2 : p1 + p2; -// if (t > 0 && t < bestHit.distance) -// { -// bestHit.distance = t; -// bestHit.position = ray.origin + t * ray.direction; -// bestHit.normal = normalize(bestHit.position - sphere.position); -// } -//} - - struct TriangleData { float3 n0, n1, n2; // normals @@ -314,11 +273,6 @@ inline bool IntersectNode(in Ray ray, in BVHNode box, in float primitive_best_di } inline bool IntersectNode(in Ray ray, in BVHNode box) { - //if (ray.origin.x >= box.min.x && ray.origin.x <= box.max.x && - // ray.origin.y >= box.min.y && ray.origin.y <= box.max.y && - // ray.origin.z >= box.min.z && ray.origin.z <= box.max.z) - // return true; - float t[6]; t[0] = (box.min.x - ray.origin.x) * ray.direction_rcp.x; t[1] = (box.max.x - ray.origin.x) * ray.direction_rcp.x; @@ -519,7 +473,7 @@ inline uint TraceBVH(Ray ray) // Also fill the final params of rayHit, such as normal, uv, materialIndex // seed should be > 0 // pixel should be normalized uv coordinates of the ray start position (used to randomize) -inline float3 Shade(inout Ray ray, inout RayHit hit, inout float seed, in float2 pixel) +inline void ShadeRay(inout Ray ray, inout RayHit hit, inout float seed, in float2 pixel) { if (hit.distance < INFINITE_RAYHIT) { @@ -582,6 +536,8 @@ inline float3 Shade(inout Ray ray, inout RayHit hit, inout float seed, in float2 emissiveColor *= emissiveMap; } + ray.color += max(0, ray.energy * emissiveColor.rgb * emissiveColor.a); + [branch] if (material.uvset_normalMap >= 0) { @@ -594,8 +550,6 @@ inline float3 Shade(inout Ray ray, inout RayHit hit, inout float seed, in float2 hit.N = normalize(lerp(N, mul(normalMap, TBN), material.normalMapStrength)); } - - // Calculate chances of reflection types: const float refractChance = 1 - baseColor.a; @@ -641,13 +595,9 @@ inline float3 Shade(inout Ray ray, inout RayHit hit, inout float seed, in float2 ray.primitiveID = hit.primitiveID; ray.bary = hit.bary; ray.Update(); - - return emissiveColor.rgb * emissiveColor.a; } else { - // Erase the ray's energy - the sky doesn't reflect anything - ray.energy = 0.0f; float3 envColor; [branch] @@ -660,7 +610,10 @@ inline float3 Shade(inout Ray ray, inout RayHit hit, inout float seed, in float2 { envColor = GetDynamicSkyColor(ray.direction); } - return envColor; + ray.color += max(0, ray.energy * envColor); + + // Erase the ray's energy - the sky doesn't reflect anything + ray.energy = 0.0f; } } diff --git a/WickedEngine/renderlightmapPS_indirect.hlsl b/WickedEngine/renderlightmapPS_indirect.hlsl index 945eb15b8..6306048e7 100644 --- a/WickedEngine/renderlightmapPS_indirect.hlsl +++ b/WickedEngine/renderlightmapPS_indirect.hlsl @@ -18,14 +18,13 @@ float4 main(Input input) : SV_TARGET float seed = xTraceRandomSeed; float3 direction = SampleHemisphere_uniform(N, seed, uv); // uniform because we care about only diffuse here Ray ray = CreateRay(trace_bias_position(P, N), direction); - float3 finalResult = 0; const uint bounces = xTraceUserData.x; for (uint i = 0; (i < bounces) && any(ray.energy); ++i) { // Sample primary ray (scene materials, sky, etc): RayHit hit = TraceScene(ray); - finalResult += ray.energy * Shade(ray, hit, seed, uv); + ShadeRay(ray, hit, seed, uv); // We sample explicit lights for every bounce, but only diffuse part. Specular will not be baked here. // Also, because we do it after the primary ray was bounced off, we only get the indirect part. @@ -143,10 +142,10 @@ float4 main(Input input) : SV_TARGET newRay.direction_rcp = rcp(newRay.direction); newRay.energy = 0; bool hit = TraceSceneANY(newRay, dist); - finalResult += ray.energy * (hit ? 0 : NdotL) * (lighting.direct.diffuse); + ray.color += ray.energy * (hit ? 0 : NdotL) * (lighting.direct.diffuse); } } } - return max(0, float4(finalResult, xTraceAccumulationFactor)); + return max(0, float4(ray.color, xTraceAccumulationFactor)); } diff --git a/WickedEngine/wiEnums.h b/WickedEngine/wiEnums.h index 814b20bf7..453d4f8ca 100644 --- a/WickedEngine/wiEnums.h +++ b/WickedEngine/wiEnums.h @@ -289,7 +289,6 @@ enum CSTYPES CSTYPE_RAYTRACE_KICKJOBS, CSTYPE_RAYTRACE_PRIMARY, CSTYPE_RAYTRACE_LIGHTSAMPLING, - CSTYPE_RAYTRACE_ACCUMULATE, CSTYPE_POSTPROCESS_BLUR_GAUSSIAN_FLOAT1, CSTYPE_POSTPROCESS_BLUR_GAUSSIAN_FLOAT4, CSTYPE_POSTPROCESS_BLUR_GAUSSIAN_UNORM1, diff --git a/WickedEngine/wiRenderer.cpp b/WickedEngine/wiRenderer.cpp index 2e098ac35..3d8404c51 100644 --- a/WickedEngine/wiRenderer.cpp +++ b/WickedEngine/wiRenderer.cpp @@ -2149,7 +2149,6 @@ void LoadShaders() computeShaders[CSTYPE_RAYTRACE_KICKJOBS] = static_cast(wiResourceManager::GetShaderManager().add(SHADERPATH + "raytrace_kickjobsCS.cso", wiResourceManager::COMPUTESHADER)); computeShaders[CSTYPE_RAYTRACE_PRIMARY] = static_cast(wiResourceManager::GetShaderManager().add(SHADERPATH + "raytrace_primaryCS.cso", wiResourceManager::COMPUTESHADER)); computeShaders[CSTYPE_RAYTRACE_LIGHTSAMPLING] = static_cast(wiResourceManager::GetShaderManager().add(SHADERPATH + "raytrace_lightsamplingCS.cso", wiResourceManager::COMPUTESHADER)); - computeShaders[CSTYPE_RAYTRACE_ACCUMULATE] = static_cast(wiResourceManager::GetShaderManager().add(SHADERPATH + "raytrace_accumulateCS.cso", wiResourceManager::COMPUTESHADER)); computeShaders[CSTYPE_POSTPROCESS_BLUR_GAUSSIAN_FLOAT1] = static_cast(wiResourceManager::GetShaderManager().add(SHADERPATH + "blur_gaussian_float1CS.cso", wiResourceManager::COMPUTESHADER)); computeShaders[CSTYPE_POSTPROCESS_BLUR_GAUSSIAN_FLOAT4] = static_cast(wiResourceManager::GetShaderManager().add(SHADERPATH + "blur_gaussian_float4CS.cso", wiResourceManager::COMPUTESHADER)); @@ -7534,17 +7533,6 @@ void RayTraceScene(const RayBuffers* rayBuffers, const Texture2D* result, int ac } const TextureDesc& result_desc = result->GetDesc(); - static TextureDesc temp_desc; - static Texture2D temp_texture; - if (temp_desc.Width < result_desc.Width || temp_desc.Height < result_desc.Height) - { - temp_desc.Width = std::max(temp_desc.Width, result_desc.Width); - temp_desc.Height = std::max(temp_desc.Height, result_desc.Height); - temp_desc.Format = FORMAT_R16G16B16A16_FLOAT; - temp_desc.BindFlags = BIND_UNORDERED_ACCESS | BIND_SHADER_RESOURCE; - device->CreateTexture2D(&temp_desc, nullptr, &temp_texture); - device->SetName(&temp_texture, "raytrace_temp_texture"); - } // Begin raytrace @@ -7575,7 +7563,8 @@ void RayTraceScene(const RayBuffers* rayBuffers, const Texture2D* result, int ac uint32_t __readBufferID = bounce % 2; uint32_t __writeBufferID = (bounce + 1) % 2; - cb.xTraceUserData.x = bounce; + cb.xTraceUserData.x = (bounce == 0 && accumulation_sample == 0) ? 1 : 0; // pre-clear result texture? + cb.xTraceUserData.y = bounce == raytraceBounceCount ? 1 : 0; // accumulation step? cb.xTraceRandomSeed = renderTime + (float)bounce; device->UpdateBuffer(&constantBuffers[CBTYPE_RAYTRACE], &cb, cmd); device->BindConstantBuffer(CS, &constantBuffers[CBTYPE_RAYTRACE], CB_GETBINDSLOT(RaytracingCB), cmd); @@ -7609,7 +7598,50 @@ void RayTraceScene(const RayBuffers* rayBuffers, const Texture2D* result, int ac } device->EventEnd(cmd); - // 1.) Compute Primary Trace (closest hit) + // Sorting and light sampling only after first bounce: + if (bounce > 0) + { + // Sort rays to achieve more coherency: + device->EventBegin("Ray Sorting", cmd); + wiGPUSortLib::Sort(rayBuffers->rayCapacity, rayBuffers->raySortBuffer, counterBuffer[__readBufferID], 0, rayBuffers->rayIndexBuffer[__readBufferID], cmd); + device->EventEnd(cmd); + + // Light sampling (any hit) + { + device->EventBegin("Light Sampling Rays", cmd); + + wiProfiler::range_id range; + if (bounce == 1) + { + range = wiProfiler::BeginRangeGPU("RayTrace - First Light Sampling", cmd); + } + + device->BindComputeShader(computeShaders[CSTYPE_RAYTRACE_LIGHTSAMPLING], cmd); + + const GPUResource* res[] = { + &counterBuffer[__readBufferID], + &rayBuffers->rayIndexBuffer[__readBufferID], + }; + device->BindResources(CS, res, TEXSLOT_ONDEMAND7, ARRAYSIZE(res), cmd); + const GPUResource* uavs[] = { + &rayBuffers->rayBuffer[__readBufferID], + }; + device->BindUAVs(CS, uavs, 0, ARRAYSIZE(uavs), cmd); + + device->DispatchIndirect(&indirectBuffer, 0, cmd); + + device->UAVBarrier(uavs, ARRAYSIZE(uavs), cmd); + device->UnbindUAVs(0, ARRAYSIZE(uavs), cmd); + + if (bounce == 1) + { + wiProfiler::EndRange(range); // RayTrace - First Light Sampling + } + device->EventEnd(cmd); + } + } + + // Compute Primary Trace (closest hit) { device->EventBegin("Primary Rays", cmd); @@ -7636,7 +7668,7 @@ void RayTraceScene(const RayBuffers* rayBuffers, const Texture2D* result, int ac &rayBuffers->rayIndexBuffer[__writeBufferID], &rayBuffers->raySortBuffer, &rayBuffers->rayBuffer[__writeBufferID], - &temp_texture, + result, }; device->BindUAVs(CS, uavs, 0, ARRAYSIZE(uavs), cmd); @@ -7651,80 +7683,8 @@ void RayTraceScene(const RayBuffers* rayBuffers, const Texture2D* result, int ac } device->EventEnd(cmd); } - - // Primary trace has written new alive ray buffer, so light sampling will use that: - std::swap(__readBufferID, __writeBufferID); - - // 2.) Sort rays to achieve more coherency: - device->EventBegin("Ray Sorting", cmd); - wiGPUSortLib::Sort(rayBuffers->rayCapacity, rayBuffers->raySortBuffer, counterBuffer[__readBufferID], 0, rayBuffers->rayIndexBuffer[__readBufferID], cmd); - device->EventEnd(cmd); - - - // 3.) Light sampling (any hit) <- only after first bounce has occured - { - device->EventBegin("Light Sampling Rays", cmd); - - wiProfiler::range_id range; - if (bounce == 1) - { - range = wiProfiler::BeginRangeGPU("RayTrace - First Light Sampling", cmd); - } - - device->BindComputeShader(computeShaders[CSTYPE_RAYTRACE_LIGHTSAMPLING], cmd); - - const GPUResource* res[] = { - &counterBuffer[__readBufferID], - &rayBuffers->rayIndexBuffer[__readBufferID], - &rayBuffers->rayBuffer[__readBufferID], - }; - device->BindResources(CS, res, TEXSLOT_ONDEMAND7, ARRAYSIZE(res), cmd); - const GPUResource* uavs[] = { - &temp_texture, - }; - device->BindUAVs(CS, uavs, 0, ARRAYSIZE(uavs), cmd); - - device->DispatchIndirect(&indirectBuffer, 0, cmd); - - device->UAVBarrier(uavs, ARRAYSIZE(uavs), cmd); - device->UnbindUAVs(0, ARRAYSIZE(uavs), cmd); - - if (bounce == 1) - { - wiProfiler::EndRange(range); // RayTrace - First Light Sampling - } - device->EventEnd(cmd); - } - } - device->EventBegin("Accumulate", cmd); - { - device->BindComputeShader(computeShaders[CSTYPE_RAYTRACE_ACCUMULATE], cmd); - - device->BindConstantBuffer(CS, &constantBuffers[CBTYPE_RAYTRACE], CB_GETBINDSLOT(RaytracingCB), cmd); - - const GPUResource* res[] = { - &temp_texture - }; - device->BindResources(CS, res, TEXSLOT_ONDEMAND0, ARRAYSIZE(res), cmd); - const GPUResource* uavs[] = { - result, - }; - device->BindUAVs(CS, uavs, 0, ARRAYSIZE(uavs), cmd); - - device->Dispatch( - (result_desc.Width + RAYTRACING_ACCUMULATE_BLOCKSIZE - 1) / RAYTRACING_ACCUMULATE_BLOCKSIZE, - (result_desc.Height + RAYTRACING_ACCUMULATE_BLOCKSIZE - 1) / RAYTRACING_ACCUMULATE_BLOCKSIZE, - 1, - cmd); - - device->UAVBarrier(uavs, ARRAYSIZE(uavs), cmd); - - device->UnbindUAVs(0, ARRAYSIZE(uavs), cmd); - } - device->EventEnd(cmd); - wiProfiler::EndRange(range); // RayTrace - ALL diff --git a/WickedEngine/wiVersion.cpp b/WickedEngine/wiVersion.cpp index 768c3657f..2a44f12e5 100644 --- a/WickedEngine/wiVersion.cpp +++ b/WickedEngine/wiVersion.cpp @@ -9,7 +9,7 @@ namespace wiVersion // minor features, major updates const int minor = 28; // minor bug fixes, alterations, refactors, updates - const int revision = 10; + const int revision = 11; long GetVersion()