diff --git a/Content/Shaders/Utils/Culling.flax b/Content/Shaders/Utils/Culling.flax
index b24424f3c..5789b00b1 100644
--- a/Content/Shaders/Utils/Culling.flax
+++ b/Content/Shaders/Utils/Culling.flax
@@ -1,3 +1,3 @@
version https://git-lfs.github.com/spec/v1
-oid sha256:9ed1450e1f11317da6235e802c0c19e9e6128251dc6545a60b730fcf128722db
-size 883
+oid sha256:23c5c44dab35c72e364f7d4a51694759267bde8aa3aea7ef3aecd061f926ac5f
+size 4322
diff --git a/Content/Shaders/Utils/MultiScaler.flax b/Content/Shaders/Utils/MultiScaler.flax
index 3d85d9076..ede322b5e 100644
--- a/Content/Shaders/Utils/MultiScaler.flax
+++ b/Content/Shaders/Utils/MultiScaler.flax
@@ -1,3 +1,3 @@
version https://git-lfs.github.com/spec/v1
-oid sha256:54e9dc715c0778cb12dde21b3c6d30e47bb5bcdb9a321f8c2e7aeb766d7e986d
-size 9694
+oid sha256:d671350d8c4e689c697b169e9878bc388aaef89d31d839702370d66d52cdc1c5
+size 9629
diff --git a/Source/Engine/Graphics/GPUContext.h b/Source/Engine/Graphics/GPUContext.h
index f479ef82b..de3b3ee42 100644
--- a/Source/Engine/Graphics/GPUContext.h
+++ b/Source/Engine/Graphics/GPUContext.h
@@ -467,7 +467,7 @@ public:
/// The number of groups dispatched in the x direction.
/// The number of groups dispatched in the y direction.
/// The number of groups dispatched in the z direction.
- API_FUNCTION() virtual void Dispatch(GPUShaderProgramCS* shader, uint32 threadGroupCountX, uint32 threadGroupCountY, uint32 threadGroupCountZ) = 0;
+ API_FUNCTION() virtual void Dispatch(GPUShaderProgramCS* shader, uint32 threadGroupCountX, uint32 threadGroupCountY = 1, uint32 threadGroupCountZ = 1) = 0;
///
/// Executes a command list from a thread group. Buffer must contain GPUDispatchIndirectArgs.
diff --git a/Source/Engine/Graphics/RenderBuffers.cpp b/Source/Engine/Graphics/RenderBuffers.cpp
index b5fb094a8..0a6519ef0 100644
--- a/Source/Engine/Graphics/RenderBuffers.cpp
+++ b/Source/Engine/Graphics/RenderBuffers.cpp
@@ -69,7 +69,8 @@ void RenderBuffers::ReleaseUnusedMemory()
UPDATE_LAZY_KEEP_RT(TemporalSSR);
UPDATE_LAZY_KEEP_RT(TemporalAA);
UPDATE_LAZY_KEEP_RT(HalfResDepth);
- UPDATE_LAZY_KEEP_RT(HiZ);
+ UPDATE_LAZY_KEEP_RT(HiZ[0]);
+ UPDATE_LAZY_KEEP_RT(HiZ[1]);
UPDATE_LAZY_KEEP_RT(LuminanceMap);
#undef UPDATE_LAZY_KEEP_RT
for (int32 i = CustomBuffers.Count() - 1; i >= 0; i--)
@@ -118,45 +119,51 @@ GPUTexture* RenderBuffers::RequestHalfResDepth(GPUContext* context)
return HalfResDepth;
}
-GPUTexture* RenderBuffers::RequestHiZ(GPUContext* context, bool fullRes, int32 mipLevels)
+GPUTexture* RenderBuffers::RequestHiZ(GPUContext* context, bool fullRes, int32 mipLevels, bool closest, bool powerOfTwo)
{
// Skip if already done in the current frame
const auto currentFrame = Engine::FrameCount;
- if (LastFrameHiZ == currentFrame)
- return HiZ;
+ int32 idx = closest ? 0 : 1;
+ if (LastFrameHiZ[idx] == currentFrame)
+ return HiZ[idx];
if (!MultiScaler::Instance()->IsReady())
return nullptr;
- LastFrameHiZ = currentFrame;
+ LastFrameHiZ[idx] = currentFrame;
// Allocate or resize buffer (with full mip-chain)
auto format = PixelFormat::R32_Float;
auto width = fullRes ? _width : Math::Max(_width >> 1, 1);
auto height = fullRes ? _height : Math::Max(_height >> 1, 1);
+ if (powerOfTwo)
+ {
+ width = Math::RoundUpToPowerOf2(width);
+ height = Math::RoundUpToPowerOf2(height);
+ }
auto desc = GPUTextureDescription::New2D(width, height, mipLevels, format, GPUTextureFlags::ShaderResource);
bool useCompute = false; // TODO: impl Compute Shader for downscaling depth to HiZ with a single dispatch (eg. FidelityFX Single Pass Downsampler)
if (useCompute)
desc.Flags |= GPUTextureFlags::UnorderedAccess;
else
desc.Flags |= GPUTextureFlags::RenderTarget | GPUTextureFlags::PerMipViews;
- if (HiZ && HiZ->GetDescription() != desc)
+ if (HiZ[idx] && HiZ[idx]->GetDescription() != desc)
{
- RenderTargetPool::Release(HiZ);
- HiZ = nullptr;
+ RenderTargetPool::Release(HiZ[idx]);
+ HiZ[idx] = nullptr;
}
- if (HiZ == nullptr)
+ if (HiZ[idx] == nullptr)
{
- HiZ = RenderTargetPool::Get(desc);
- RENDER_TARGET_POOL_SET_NAME(HiZ, "HiZ");
+ HiZ[idx] = RenderTargetPool::Get(desc);
+ RENDER_TARGET_POOL_SET_NAME(HiZ[idx], "HiZ");
#if PLATFORM_WEB
// Hack to fix WebGPU limitation that requires to specify different sampler type manually to load 32-bit float texture
- SetWebGPUTextureViewSampler(HiZ->View(), GPU_WEBGPU_SAMPLER_TYPE_UNFILTERABLE_FLOAT);
+ SetWebGPUTextureViewSampler(HiZ[idx]->View(), GPU_WEBGPU_SAMPLER_TYPE_UNFILTERABLE_FLOAT);
#endif
}
// Downscale
- MultiScaler::Instance()->BuildHiZ(context, DepthBuffer, HiZ);
+ MultiScaler::Instance()->BuildHiZ(context, DepthBuffer, HiZ[idx], closest);
- return HiZ;
+ return HiZ[idx];
}
PixelFormat RenderBuffers::GetOutputFormat() const
@@ -298,7 +305,8 @@ void RenderBuffers::Release()
UPDATE_LAZY_KEEP_RT(TemporalSSR);
UPDATE_LAZY_KEEP_RT(TemporalAA);
UPDATE_LAZY_KEEP_RT(HalfResDepth);
- UPDATE_LAZY_KEEP_RT(HiZ);
+ UPDATE_LAZY_KEEP_RT(HiZ[0]);
+ UPDATE_LAZY_KEEP_RT(HiZ[1]);
UPDATE_LAZY_KEEP_RT(LuminanceMap);
#undef UPDATE_LAZY_KEEP_RT
CustomBuffers.ClearDelete();
diff --git a/Source/Engine/Graphics/RenderBuffers.h b/Source/Engine/Graphics/RenderBuffers.h
index 6cb168ac8..f1c239e18 100644
--- a/Source/Engine/Graphics/RenderBuffers.h
+++ b/Source/Engine/Graphics/RenderBuffers.h
@@ -62,9 +62,9 @@ API_CLASS() class FLAXENGINE_API RenderBuffers : public ScriptingObject, private
private:
GPUTexture* HalfResDepth = nullptr;
- GPUTexture* HiZ = nullptr;
+ GPUTexture* HiZ[2] = {};
uint64 LastFrameHalfResDepth = 0;
- uint64 LastFrameHiZ = 0;
+ uint64 LastFrameHiZ[2] = {};
// Scene drawing cache with the per-object state (eg. LOD transitions, motion-vectors movement)
struct SceneData
@@ -155,13 +155,15 @@ public:
GPUTexture* RequestHalfResDepth(GPUContext* context);
///
- /// Requests the Hierarchical Z-Buffer (closest) to be prepared for the current frame.
+ /// Requests the Hierarchical Z-Buffer (closest or furthest) to be prepared for the current frame.
///
/// The context.
/// Generates the full-resolution buffer, otherwise HiZ starts at half-res of the original Depth Buffer.
/// Maximum amount of mip levels to generate. Value 0 generates a full mip chain down to 1x1.
+ /// True if generate the closest depth values, otherwise will use the furthest depths filter.
+ /// True if the buffer dimensions should be powers of two.
/// The HiZ depth buffer.
- GPUTexture* RequestHiZ(GPUContext* context, bool fullRes = false, int32 mipLevels = 0);
+ GPUTexture* RequestHiZ(GPUContext* context, bool fullRes = false, int32 mipLevels = 0, bool closest = true, bool powerOfTwo = false);
public:
///
diff --git a/Source/Engine/GraphicsDevice/DirectX/DX11/GPUBufferDX11.cpp b/Source/Engine/GraphicsDevice/DirectX/DX11/GPUBufferDX11.cpp
index b345f0754..f324d5bdc 100644
--- a/Source/Engine/GraphicsDevice/DirectX/DX11/GPUBufferDX11.cpp
+++ b/Source/Engine/GraphicsDevice/DirectX/DX11/GPUBufferDX11.cpp
@@ -20,6 +20,7 @@ GPUBufferView* GPUBufferDX11::View() const
void* GPUBufferDX11::Map(GPUResourceMapMode mode)
{
+ CHECK_RETURN(_resource, nullptr);
const bool isMainThread = IsInMainThread();
if (!isMainThread)
_device->Locker.Lock();
diff --git a/Source/Engine/GraphicsDevice/Vulkan/GPUSwapChainVulkan.cpp b/Source/Engine/GraphicsDevice/Vulkan/GPUSwapChainVulkan.cpp
index 1bf1ec859..65dffc255 100644
--- a/Source/Engine/GraphicsDevice/Vulkan/GPUSwapChainVulkan.cpp
+++ b/Source/Engine/GraphicsDevice/Vulkan/GPUSwapChainVulkan.cpp
@@ -642,7 +642,8 @@ void GPUSwapChainVulkan::Present(bool vsync)
// Cache a command buffer to wait on its fence before drawing to this backbuffer again
auto& acquiredBackBuffer = _backBuffers[_acquiredImageIndex];
- ASSERT(acquiredBackBuffer.SubmitCmdBuffer == nullptr || acquiredBackBuffer.SubmitCmdBuffer->IsSubmitted());
+ // TODO: fix rare issues with the assert below
+ //ASSERT(acquiredBackBuffer.SubmitCmdBuffer == nullptr || acquiredBackBuffer.SubmitCmdBuffer->IsSubmitted());
acquiredBackBuffer.SubmitCmdBuffer = context->GetCmdBufferManager()->GetActiveCmdBuffer();
context->GetCmdBufferManager()->SubmitActiveCmdBuffer(_backBuffers[_acquiredImageIndex].RenderingDoneSemaphore);
diff --git a/Source/Engine/GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp b/Source/Engine/GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp
index 5e1bd8359..ebefb2839 100644
--- a/Source/Engine/GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp
+++ b/Source/Engine/GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp
@@ -85,10 +85,10 @@ bool GPUBufferWebGPU::OnInit()
bufferDesc.usage |= WGPUBufferUsage_MapWrite | WGPUBufferUsage_CopySrc;
break;
case GPUResourceUsage::StagingReadback:
- bufferDesc.usage |= WGPUBufferUsage_MapRead;
+ bufferDesc.usage |= WGPUBufferUsage_MapRead | WGPUBufferUsage_CopyDst;
break;
case GPUResourceUsage::Staging:
- bufferDesc.usage |= WGPUBufferUsage_MapRead | WGPUBufferUsage_MapWrite | WGPUBufferUsage_CopySrc;
+ bufferDesc.usage |= WGPUBufferUsage_MapRead | WGPUBufferUsage_MapWrite | WGPUBufferUsage_CopySrc | WGPUBufferUsage_CopyDst;
break;
}
bufferDesc.size = (_desc.Size + 3) & ~0x3; // Align up to the multiple of 4 bytes
diff --git a/Source/Engine/Renderer/Culling/HZBOcclusionCulling.cpp b/Source/Engine/Renderer/Culling/HZBOcclusionCulling.cpp
new file mode 100644
index 000000000..f2d1f3e2e
--- /dev/null
+++ b/Source/Engine/Renderer/Culling/HZBOcclusionCulling.cpp
@@ -0,0 +1,283 @@
+// Copyright (c) Wojciech Figat. All rights reserved.
+
+#include "HZBOcclusionCulling.h"
+#include "Engine/Renderer/DrawCall.h"
+#include "Engine/Content/Content.h"
+#include "Engine/Content/Assets/Shader.h"
+#include "Engine/Graphics/GPUContext.h"
+#include "Engine/Graphics/GPUDevice.h"
+#include "Engine/Graphics/RenderTask.h"
+#include "Engine/Graphics/RenderBuffers.h"
+#include "Engine/Graphics/RenderContext.h"
+#include "Engine/Graphics/Shaders/GPUShader.h"
+#include "Engine/Profiler/ProfilerCPU.h"
+#include "Engine/Profiler/ProfilerGPU.h"
+#include "Engine/Core/Config/GraphicsSettings.h"
+#include "Engine/Engine/Engine.h"
+
+#define ResultValue uint32
+#define ResultType PixelFormat::R32_UInt
+
+HZBOcclusionCulling::HZBOcclusionCulling(const SpawnParams& params)
+ : ScriptingObject(params)
+ , _shader(Content::LoadAsyncInternal(TEXT("Shaders/Utils/Culling")))
+ , _boundsBuffer(0, PixelFormat::R32G32B32_Float, false, TEXT("HZB.Bounds"))
+{
+ _boundsBuffer.Usage = GPUResourceUsage::Dynamic;
+}
+
+HZBOcclusionCulling::~HZBOcclusionCulling()
+{
+ SAFE_DELETE_GPU_RESOURCE(_resultsBuffer);
+ SAFE_DELETE_GPU_RESOURCES(_readbackBuffers);
+}
+
+bool HZBOcclusionCulling::IsSupported()
+{
+ const GPULimits& limits = GPUDevice::Instance->Limits;
+ return limits.HasCompute;
+}
+
+void HZBOcclusionCulling::BeginFrame(const RenderContext& renderContext)
+{
+ PROFILE_CPU();
+
+ // Read settings
+ auto settings = GraphicsSettings::Get();
+ auto framesCount = Math::Clamp(settings->OcclusionBufferedFrames, 1, MaxFrames);
+ if (_framesCount != framesCount)
+ {
+ // Reset state
+ for (int32 i = framesCount; i < _framesCount; i++)
+ SAFE_DELETE_GPU_RESOURCE(_readbackBuffers[i]);
+ for (int32 i = _framesCount; i < framesCount; i++)
+ _readbackBuffers[i] = GPUDevice::Instance->CreateBuffer(TEXT("HZB.Readback"));
+ if (!_resultsBuffer)
+ _resultsBuffer = GPUDevice::Instance->CreateBuffer(TEXT("HZB.Results"));
+ else if (_resultsBuffer->IsAllocated())
+ {
+ auto desc = _resultsBuffer->GetDescription().ToStagingReadback();
+ for (int32 i = _framesCount; i < framesCount; i++)
+ _readbackBuffers[i]->Init(desc);
+ }
+ _framesCount = framesCount;
+ for (auto& item : _items)
+ {
+ item.Occluded = false;
+ Platform::MemoryClear(item.Frames, sizeof(item.Frames));
+ }
+ Platform::MemoryClear(_readbackFrames, sizeof(_readbackFrames));
+ }
+ _boundsScale = Math::Max(settings->OcclusionBoundsScale, 1.01f);
+
+ // Handle origin-relative rendering
+ _forceUpdateBounds = renderContext.View.Origin != _origin;
+ _origin = renderContext.View.Origin;
+ _viewPos = renderContext.View.WorldPosition;
+
+ // Skip reading occlusion results on camera cuts
+ bool forceVisible = renderContext.Task->IsCameraCut;
+
+ // Skip reading when view was not rendered for some time
+ uint64 engineFrame = Engine::FrameCount;
+ forceVisible |= (int32)(engineFrame - _lastEngineFrameUsed) >= framesCount;
+ _lastEngineFrameUsed = engineFrame;
+
+ int32 frame = _frameCounter, bufferedFrame = _frameCounter % framesCount;
+ if (forceVisible || _submitFailed)
+ {
+ // Reset
+ _submitFailed = false;
+ for (auto& item : _items)
+ item.Occluded = false;
+ }
+ else if (_readbackFrames[bufferedFrame])
+ {
+ // Read results from the last frame
+ int32 itemsCount = _items.Count(), itemsUsed = 0;
+ auto* readback = (const ResultValue*)_readbackBuffers[bufferedFrame]->Map(GPUResourceMapMode::Read);
+ if (readback)
+ {
+ auto* items = _items.Get();
+ int32 frameItemsCount = Math::Min(_readbackCounts[bufferedFrame], itemsCount);
+ ASSERT(_readbackBuffers[bufferedFrame]->GetSize() >= frameItemsCount * sizeof(ResultValue));
+ for (int32 i = 0; i < frameItemsCount; i++)
+ {
+ auto& item = items[i];
+ int32 itemFrame = item.Frames[bufferedFrame];
+ int32 lag = frame - itemFrame;
+ if (itemFrame && lag <= framesCount)
+ {
+ // Clear frame
+ item.Frames[bufferedFrame] = 0;
+
+ // Read result
+ ResultValue result = readback[i];
+ item.Occluded = result == 0;
+ itemsUsed++;
+ }
+ }
+ _readbackBuffers[bufferedFrame]->Unmap();
+ }
+ else
+ {
+ for (auto& item : _items)
+ item.Occluded = false;
+ }
+ ZoneValue(itemsUsed);
+ _readbackFrames[bufferedFrame] = 0;
+ }
+
+ _items.BeginFrame();
+
+ // Resize buffers to store items culling results
+ int32 itemsCapacity = Math::RoundUpToPowerOf2(_items.Count());
+ if (_resultsBuffer->GetSize() < (uint32)itemsCapacity * sizeof(ResultValue))
+ {
+ itemsCapacity = Math::Max(itemsCapacity, 512);
+ auto desc = GPUBufferDescription::Buffer(itemsCapacity * sizeof(ResultValue), GPUBufferFlags::UnorderedAccess, ResultType, nullptr, sizeof(ResultValue));
+ _resultsBuffer->Init(desc);
+ desc = desc.ToStagingReadback();
+ for (int32 i = 0; i < _framesCount; i++)
+ _readbackBuffers[i]->Init(desc);
+ }
+
+ // Prepare buffer to write geometry bounds in async during drawing
+ bool init = _boundsBuffer.Data.IsEmpty();
+ _boundsBuffer.Data.Resize(_items.Count() * 2 * sizeof(Float3), true);
+ if (init && _boundsBuffer.Data.HasItems())
+ Platform::MemoryClear(_boundsBuffer.Data.Get(), sizeof(Float3) * 2); // Clear first unused item
+ _dirtyBounds = 0;
+ _submitFailed = false;
+}
+
+void HZBOcclusionCulling::EndFrame(const RenderContext& renderContext)
+{
+ // Move to the next frame
+ _frameCounter++;
+}
+
+void HZBOcclusionCulling::Submit(const RenderContext& renderContext)
+{
+ if (_items.IsEmpty() || !_shader || !_shader->IsLoaded())
+ return;
+ PROFILE_CPU();
+ PROFILE_GPU("HZB Occlusion Culling");
+ GPUContext* context = GPUDevice::Instance->GetMainContext();
+
+ // Update objects to cull
+ auto frustum = renderContext.View.Frustum;
+ auto* items = _items.Get();
+ int32 frame = _frameCounter, bufferedFrame = _frameCounter % _framesCount, itemsCount = _items.Count(), itemsEnd = 0;
+ for (int32 i = 0; i < itemsCount; i++)
+ {
+ auto& item = items[i];
+ if (item.LastUsedFrame != frame || frustum.Contains(item.Bounds) == ContainmentType::Disjoint)
+ continue;
+ itemsEnd = i + 1;
+
+ // Mark as used in this HZB frame (to read results later)
+ item.Frames[bufferedFrame] = frame;
+ }
+ ZoneValue(itemsEnd);
+ _readbackCounts[bufferedFrame] = itemsEnd;
+ if (itemsEnd == 0)
+ return;
+
+ // Build HZB with furthest depths (full mip chain) for the current frame
+ context->ResetRenderTarget();
+ GPUTexture* hzb = renderContext.Buffers->RequestHiZ(context, false, 0, false, true);
+ if (!hzb)
+ {
+ _submitFailed = true;
+ return;
+ }
+
+ // Upload object bounds data
+ if (_dirtyBounds)
+ _boundsBuffer.Flush(context);
+ ASSERT(_boundsBuffer.GetBuffer()->GetSize() >= sizeof(Float3) * 2 * itemsCount);
+ ASSERT(_resultsBuffer->GetSize() >= itemsCount * sizeof(ResultValue));
+ ASSERT(_readbackBuffers[bufferedFrame]->GetSize() >= itemsCount * sizeof(ResultValue));
+
+ // Test all object bounds against current frame HZB
+ auto cs = _shader->GPU->GetCS("CS_HZBCull");
+ auto cb = _shader->GPU->GetCB(0);
+ {
+ OcclusionCullingData data;
+ Matrix::Transpose(renderContext.View.ViewProjection(), data.ViewProjectionMatrix);
+ data.RTSizeX = (float)hzb->Width();
+ data.RTSizeY = (float)hzb->Height();
+ data.MaxMipLevel = (float)hzb->MipLevels();
+ data.CullCount = itemsEnd;
+ context->UpdateCB(cb, &data);
+ }
+ context->BindCB(0, cb);
+ context->BindUA(0, _resultsBuffer->View());
+ context->BindSR(0, _boundsBuffer.GetBuffer()->View());
+ context->BindSR(1, hzb->View());
+ context->Dispatch(cs, Math::DivideAndRoundUp(itemsEnd, 64));
+
+ // Copy results to the readback buffer
+ context->CopyBuffer(_readbackBuffers[bufferedFrame], _resultsBuffer, itemsEnd * sizeof(ResultValue));
+
+ // Mark the readback buffer has a valid frame data
+ _readbackFrames[bufferedFrame] = frame;
+
+ // Restore state
+ context->ResetSR();
+ context->SetViewportAndScissors(renderContext.Buffers->GetViewport());
+}
+
+bool HZBOcclusionCulling::IsVisible(const BoundingBox& bounds, uint32& cullingId)
+{
+ return IsVisible(bounds, cullingId, nullptr);
+}
+
+bool HZBOcclusionCulling::IsVisible(const BoundingBox& bounds, GeometryDrawState& drawState)
+{
+ return IsVisible(bounds, drawState.CullingId, &drawState);
+}
+
+bool HZBOcclusionCulling::IsVisible(BoundingBox bounds, uint32& cullingId, GeometryDrawState* drawState)
+{
+ // Enlarge bounds to reduce popping
+ bounds = BoundingBox::MakeScaled(bounds, _boundsScale);
+ // TODO: use camera motion to enlarge bounds
+ // TODO: use object motion (from prev frame world matrix) to enlarge bounds in the direction of movement to reduce popping
+
+ // Assume visible when view is right inside the bounds
+ if (bounds.Contains(_viewPos) == ContainmentType::Contains)
+ return true;
+
+ // Check id
+ if (_items.GetCullingId(cullingId))
+ return true;
+
+ // Update item bounds
+ auto& item = _items.Get()[cullingId];
+ if (item.Bounds != bounds || _forceUpdateBounds)
+ {
+ // Update bounds
+ item.Bounds = bounds;
+
+ // Write to the bounds buffer
+ Float3 boxMin = bounds.Minimum - _origin;
+ Float3 boxMax = bounds.Maximum - _origin;
+ ASSERT_LOW_LAYER(_boundsBuffer.Data.Count() >= (2 * sizeof(Float3)) * (cullingId + 1));
+ Float3* boundsPtr = (Float3*)(_boundsBuffer.Data.Get() + (2 * sizeof(Float3)) * cullingId);
+ boundsPtr[0] = boxMin;
+ boundsPtr[1] = boxMax;
+ Platform::InterlockedIncrement(&_dirtyBounds);
+ }
+
+ // Force visible when object was not rendered last frame (eg. outside the frustum)
+ if (_frameCounter - item.LastUsedFrame > 1)
+ {
+ item.Occluded = false;
+ }
+ item.LastUsedFrame = _frameCounter;
+
+ // Read occlusion result
+ return !item.Occluded;
+}
diff --git a/Source/Engine/Renderer/Culling/HZBOcclusionCulling.h b/Source/Engine/Renderer/Culling/HZBOcclusionCulling.h
new file mode 100644
index 000000000..00583dbde
--- /dev/null
+++ b/Source/Engine/Renderer/Culling/HZBOcclusionCulling.h
@@ -0,0 +1,61 @@
+// Copyright (c) Wojciech Figat. All rights reserved.
+
+#pragma once
+
+#include "IOcclusionCulling.h"
+#include "OcclusionCullingTools.h"
+#include "Engine/Core/Collections/Array.h"
+#include "Engine/Graphics/DynamicBuffer.h"
+#include "Engine/Scripting/ScriptingObject.h"
+#include "Engine/Content/AssetReference.h"
+
+///
+/// Occlusion culling system based on Hierarchical Z-Buffer visibility test with a readback.
+/// It builds a mipmap chain from the depth buffer and runs a compute shader to test object bounding boxes against it to skip rendering unseen geometry.
+/// Results are readback with a few frames latency (objects can pop in fast motion).
+///
+API_CLASS(Sealed) class FLAXENGINE_API HZBOcclusionCulling : public ScriptingObject, public IOcclusionCulling
+{
+ DECLARE_SCRIPTING_TYPE(HZBOcclusionCulling);
+ ~HZBOcclusionCulling();
+
+ // Maximum number of frames to delay the visibility results readback from GPU (to avoid stalls). The higher value the more latency but less GPU stalls.
+ constexpr static int32 MaxFrames = 4;
+
+private:
+ struct alignas(sizeof(uint64)) Item
+ {
+ BoundingBox Bounds; // Object bounds (world-space)
+ int32 Frames[MaxFrames]; // Frame counter for each test (0 if not performed)
+ int32 LastUsedFrame; // Last frame object was drawn (incl. not frustum-culled or hidden)
+ bool Occluded; // Result from the last frame
+ };
+
+ int32 _framesCount = 0, _frameCounter = 0;
+ volatile int64 _dirtyBounds = 0;
+ uint64 _lastEngineFrameUsed = 0;
+ Vector3 _origin = Vector3::Zero;
+ Vector3 _viewPos = Vector3::Zero;
+ AssetReference _shader;
+ OcclusionCullingItems- _items;
+ DynamicTypedBuffer _boundsBuffer;
+ GPUBuffer* _resultsBuffer = nullptr;
+ GPUBuffer* _readbackBuffers[MaxFrames] = {};
+ int32 _readbackFrames[MaxFrames] = {};
+ int32 _readbackCounts[MaxFrames] = {};
+ float _boundsScale = 1.0f;
+ bool _forceUpdateBounds = true;
+ bool _submitFailed = false;
+
+public:
+ // [IOcclusionCulling]
+ bool IsSupported() override;
+ void BeginFrame(const RenderContext& renderContext) override;
+ void EndFrame(const RenderContext& renderContext) override;
+ void Submit(const RenderContext& renderContext) override;
+ bool IsVisible(const BoundingBox& bounds, uint32& cullingId) override;
+ bool IsVisible(const BoundingBox& bounds, GeometryDrawState& drawState) override;
+
+private:
+ bool IsVisible(BoundingBox bounds, uint32& cullingId, GeometryDrawState* drawState);
+};
diff --git a/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.cpp b/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.cpp
index 4ac40f6f7..2aa270589 100644
--- a/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.cpp
+++ b/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.cpp
@@ -28,6 +28,10 @@ HardwareOcclusionCulling::HardwareOcclusionCulling(const SpawnParams& params)
HardwareOcclusionCulling::~HardwareOcclusionCulling()
{
SAFE_DELETE_GPU_RESOURCE(_indexBuffer);
+#if COMPILE_WITH_DEV_ENV
+ if (_shader)
+ _shader->Reloading.Unbind(this);
+#endif
}
void HardwareOcclusionCulling::BeginFrame(const RenderContext& renderContext)
@@ -100,42 +104,7 @@ void HardwareOcclusionCulling::BeginFrame(const RenderContext& renderContext)
ZoneValue(itemsUsed);
}
- // Remove used free items
- _freeItems.Resize(Math::Max((int32)_freeItemsCount, 0));
-
-#if 0 // TODO: find a different way as there might be some invisible object with CullingId assigned and drawing it later will overlap with reused IDs
- // Trim history
- constexpr int32 frameTTL = 20;
- if (_frameCounter % 10 == 0 && _frameCounter > frameTTL)
- {
- const int32 lastFrame = _frameCounter - frameTTL;
- for (int32 i = 0; i < _items.Count(); i++)
- {
- auto& item = _items.Get()[i];
- if (item.LastUsedFrame && item.LastUsedFrame < lastFrame)
- {
- Platform::MemoryClear(&item, sizeof(item));
- _freeItems.Add(i);
- }
- }
- }
-#endif
-
- // Allocate new items (as requested during the previous frame)
- if (_newItemsCount > 0)
- {
- int32 itemsStart = _items.Count(), count = (int32)_newItemsCount, freeStart = _freeItems.Count();
- if (itemsStart == 0)
- count++; // 0 is invalid for cullingId
- _items.AddZeroed(count);
- _freeItems.AddUninitialized(count);
- for (int32 i = 0; i < count; i++)
- _freeItems.Get()[freeStart + i] = itemsStart + i;
- if (itemsStart == 0)
- _freeItems.RemoveAt(0); // 0 is invalid for cullingId
- _newItemsCount = 0;
- }
- _freeItemsCount = _freeItems.Count();
+ _items.BeginFrame();
// Prepare vertex buffer to build geometry bound meshes in async during drawing
_vertexBuffer.Data.Resize(_items.Count() * 8 * sizeof(Float3), true);
@@ -203,9 +172,9 @@ void HardwareOcclusionCulling::Submit(const RenderContext& renderContext)
}
auto cb = _shader->GPU->GetCB(0);
{
- Matrix viewProjectionMatrix;
- Matrix::Transpose(renderContext.View.ViewProjection(), viewProjectionMatrix);
- context->UpdateCB(cb, &viewProjectionMatrix);
+ OcclusionCullingData data;
+ Matrix::Transpose(renderContext.View.ViewProjection(), data.ViewProjectionMatrix);
+ context->UpdateCB(cb, &data);
}
context->BindCB(0, cb);
context->BindIB(_indexBuffer);
@@ -266,23 +235,9 @@ bool HardwareOcclusionCulling::IsVisible(BoundingBox bounds, uint32& cullingId,
if (bounds.Contains(_viewPos) == ContainmentType::Contains)
return true;
- // Check if object doesn't have ID assigned yet
- if (cullingId == 0 || cullingId >= (uint32)_items.Count())
- {
- int64 freeIndex = Platform::InterlockedDecrement(&_freeItemsCount);
- if (freeIndex >= 0)
- {
- // Use the ID from the free list
- ASSERT_LOW_LAYER(freeIndex < _freeItems.Count());
- cullingId = _freeItems.Get()[freeIndex];
- }
- else
- {
- // Count space needed to contain all objects (for the next frame)
- Platform::InterlockedIncrement(&_newItemsCount);
- return true;
- }
- }
+ // Check id
+ if (_items.GetCullingId(cullingId))
+ return true;
// Update item bounds
auto& item = _items.Get()[cullingId];
diff --git a/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.h b/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.h
index fe0b26ea7..451c770e8 100644
--- a/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.h
+++ b/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.h
@@ -3,6 +3,7 @@
#pragma once
#include "IOcclusionCulling.h"
+#include "OcclusionCullingTools.h"
#include "Engine/Core/Collections/Array.h"
#include "Engine/Graphics/DynamicBuffer.h"
#include "Engine/Scripting/ScriptingObject.h"
@@ -34,20 +35,17 @@ private:
};
int32 _framesCount = 2, _frameCounter = 0;
- volatile int64 _freeItemsCount = 0;
- volatile int64 _newItemsCount = 0;
volatile int64 _dirtyBounds = 0;
uint64 _lastEngineFrameUsed = 0;
- float _boundsScale = 1.0f;
- Array
- _items;
- Array _freeItems;
GPUPipelineState* _pso = nullptr;
DynamicVertexBuffer _vertexBuffer;
GPUBuffer* _indexBuffer = nullptr;
Vector3 _origin = Vector3::Zero;
Vector3 _viewPos = Vector3::Zero;
- bool _forceUpdateBounds = true;
AssetReference _shader;
+ OcclusionCullingItems
- _items;
+ float _boundsScale = 1.0f;
+ bool _forceUpdateBounds = true;
public:
// [IOcclusionCulling]
diff --git a/Source/Engine/Renderer/Culling/OcclusionCullingTools.h b/Source/Engine/Renderer/Culling/OcclusionCullingTools.h
new file mode 100644
index 000000000..5384653cf
--- /dev/null
+++ b/Source/Engine/Renderer/Culling/OcclusionCullingTools.h
@@ -0,0 +1,91 @@
+// Copyright (c) Wojciech Figat. All rights reserved.
+
+#pragma once
+
+#include "Engine/Core/Math/Matrix.h"
+#include "Engine/Core/Collections/Array.h"
+#include "Engine/Graphics/Config.h"
+
+GPU_CB_STRUCT(OcclusionCullingData {
+ Matrix ViewProjectionMatrix;
+ float RTSizeX;
+ float RTSizeY;
+ float MaxMipLevel;
+ uint32 CullCount;
+ });
+
+///
+/// Utility for occlusion culling implementations to manage stable CullingId for objects with state tracking (over multiple frames).
+///
+template
+class OcclusionCullingItems : public Array
-
+{
+private:
+ volatile int64 _freeItemsCount = 0;
+ volatile int64 _newItemsCount = 0;
+ Array _freeItems;
+
+public:
+ void BeginFrame()
+ {
+ // Remove used free items
+ _freeItems.Resize(Math::Max((int32)_freeItemsCount, 0));
+
+#if 0 // TODO: find a different way as there might be some invisible object with CullingId assigned and drawing it later will overlap with reused IDs
+ // Trim history
+ constexpr int32 frameTTL = 20;
+ if (_frameCounter % 10 == 0 && _frameCounter > frameTTL)
+ {
+ const int32 lastFrame = _frameCounter - frameTTL;
+ for (int32 i = 0; i < this->Count(); i++)
+ {
+ auto& item = this->Get()[i];
+ if (item.LastUsedFrame && item.LastUsedFrame < lastFrame)
+ {
+ Platform::MemoryClear(&item, sizeof(item));
+ _freeItems.Add(i);
+ }
+ }
+ }
+#endif
+
+ // Allocate new items (as requested during the previous frame)
+ if (_newItemsCount > 0)
+ {
+ int32 itemsStart = this->Count(), count = (int32)_newItemsCount, freeStart = _freeItems.Count();
+ if (itemsStart == 0)
+ count++; // 0 is invalid for cullingId
+ this->AddZeroed(count);
+ _freeItems.AddUninitialized(count);
+ for (int32 i = 0; i < count; i++)
+ _freeItems.Get()[freeStart + i] = itemsStart + i;
+ if (itemsStart == 0)
+ _freeItems.RemoveAt(0); // 0 is invalid for cullingId
+ _newItemsCount = 0;
+ }
+ _freeItemsCount = _freeItems.Count();
+ }
+
+ bool GetCullingId(uint32& cullingId)
+ {
+ // Check if object doesn't have ID assigned yet
+ if (cullingId == 0 || cullingId >= (uint32)this->Count())
+ {
+ int64 freeIndex = Platform::InterlockedDecrement(&_freeItemsCount);
+ if (freeIndex >= 0)
+ {
+ // Use the ID from the free list
+ ASSERT_LOW_LAYER(freeIndex < _freeItems.Count());
+ cullingId = _freeItems.Get()[freeIndex];
+ }
+ else
+ {
+ // Count space needed to contain all objects (for the next frame)
+ Platform::InterlockedIncrement(&_newItemsCount);
+ return true;
+ }
+ }
+
+ return false;
+ }
+};
diff --git a/Source/Engine/Renderer/Utils/MultiScaler.cpp b/Source/Engine/Renderer/Utils/MultiScaler.cpp
index 0bb5f848c..80813db6b 100644
--- a/Source/Engine/Renderer/Utils/MultiScaler.cpp
+++ b/Source/Engine/Renderer/Utils/MultiScaler.cpp
@@ -102,6 +102,23 @@ void MultiScaler::Dispose()
_shader = nullptr;
}
+#if COMPILE_WITH_DEV_ENV
+
+void MultiScaler::OnShaderReloading(Asset* obj)
+{
+ for (const auto& e : _psBilateralUpscale)
+ e.Value->ReleaseGPU();
+ _psBilateralUpscale.ClearDelete();
+ _psUpscale->ReleaseGPU();
+ _psBlur5.Release();
+ _psBlur9.Release();
+ _psBlur13.Release();
+ _psHalfDepth.Release();
+ invalidateResources();
+}
+
+#endif
+
void MultiScaler::Filter(FilterMode mode, GPUContext* context, int32 width, int32 height, GPUTextureView* src, GPUTextureView* dst, GPUTextureView* tmp)
{
PROFILE_GPU_CPU("MultiScaler Filter");
@@ -260,30 +277,34 @@ void MultiScaler::DownscaleDepth(GPUContext* context, int32 dstWidth, int32 dstH
context->UnBindCB(0);
}
-void MultiScaler::BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture* dstHiZ)
+void MultiScaler::BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture* dstHiZ, bool closest)
{
PROFILE_GPU_CPU("Build HiZ");
int32 dstWidth = dstHiZ->Width();
int32 dstHeight = dstHiZ->Height();
+ GPUPipelineState* state = _psHalfDepth[closest ? 2 : 0]; // See PS_HalfDepth permutations
// Copy mip0
if (srcDepth->Size() == dstHiZ->Size() && srcDepth->Format() == dstHiZ->Format())
{
+ // Size and format match
context->CopySubresource(dstHiZ, 0, srcDepth, 0);
}
else if (srcDepth->Size() == dstHiZ->Size())
{
+ // Size match
context->Draw(dstHiZ, srcDepth);
}
else
{
+ // Downscale
auto rt = dstHiZ->View();
auto rtAction = GPUDrawPassAction::Store;
GPUDrawPass drawPass(context, ToSpan(&rt, 1), ToSpan(&rtAction, 1));
context->SetViewportAndScissors((float)dstWidth, (float)dstHeight);
context->BindSR(0, srcDepth);
- context->SetState(_psHalfDepth[2]);
+ context->SetState(state);
context->DrawFullscreenTriangle();
}
@@ -299,7 +320,7 @@ void MultiScaler::BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture
GPUDrawPass drawPass(context, ToSpan(&rt, 1), ToSpan(&rtAction, 1));
context->SetViewportAndScissors((float)mipWidth, (float)mipHeight);
context->BindSR(0, dstHiZ->View(0, mip - 1));
- context->SetState(_psHalfDepth[2]);
+ context->SetState(state);
context->DrawFullscreenTriangle();
}
diff --git a/Source/Engine/Renderer/Utils/MultiScaler.h b/Source/Engine/Renderer/Utils/MultiScaler.h
index ce21e7731..bc0bab5a9 100644
--- a/Source/Engine/Renderer/Utils/MultiScaler.h
+++ b/Source/Engine/Renderer/Utils/MultiScaler.h
@@ -67,7 +67,7 @@ public:
void Filter(FilterMode mode, GPUContext* context, int32 width, int32 height, GPUTextureView* srcDst, GPUTextureView* tmp);
///
- /// Downscales the depth buffer (to half resolution). Uses `min` operator (`max` for inverted depth) to output the furthest depths for conservative usage.
+ /// Downscales the depth buffer (to half resolution). Uses `max` operator (`min` for inverted depth) to output the furthest depths for conservative usage.
///
/// The context.
/// The width of the destination texture (in pixels).
@@ -77,12 +77,13 @@ public:
void DownscaleDepth(GPUContext* context, int32 dstWidth, int32 dstHeight, GPUTexture* src, GPUTextureView* dst);
///
- /// Generates the Hierarchical Z-Buffer (HiZ). Uses `min` operator (`max` for inverted depth) to output the furthest depths for conservative usage.
+ /// Generates the Hierarchical Z-Buffer (HiZ) for a given depth buffer with a mip chain.
///
/// The context.
/// The source depth buffer texture (has to have ShaderResource flag).
/// The destination HiZ texture (has to have DepthStencil or RenderTarget flag).
- void BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture* dstHiZ);
+ /// True if generate the closest depth values, otherwise will use the furthest depths filter.
+ void BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture* dstHiZ, bool closest = true);
///
/// Upscales the texture using Catmull-Rom filtering with 9-taps.
@@ -111,18 +112,7 @@ public:
bool Init() override;
void Dispose() override;
#if COMPILE_WITH_DEV_ENV
- void OnShaderReloading(Asset* obj)
- {
- for (const auto& e : _psBilateralUpscale)
- e.Value->ReleaseGPU();
- _psBilateralUpscale.ClearDelete();
- _psUpscale->ReleaseGPU();
- _psBlur5.Release();
- _psBlur9.Release();
- _psBlur13.Release();
- _psHalfDepth.Release();
- invalidateResources();
- }
+ void OnShaderReloading(Asset* obj);
#endif
protected:
diff --git a/Source/Shaders/Common.hlsl b/Source/Shaders/Common.hlsl
index 2a92d4088..00e48c90a 100644
--- a/Source/Shaders/Common.hlsl
+++ b/Source/Shaders/Common.hlsl
@@ -173,9 +173,17 @@ float4 LoadTextureWGSL(Texture2D tex, float2 uv)
tex.GetDimensions(size.x, size.y);
return tex.Load(uint3(size * uv, 0));
}
+float4 LoadTextureWGSL(Texture2D tex, float2 uv, float level)
+{
+ uint2 size, levels;
+ tex.GetDimensions(level, size.x, size.y, levels);
+ return tex.Load(uint3(size * uv, level));
+}
#define SAMPLE_RT_DEPTH(rt, texCoord) LoadTextureWGSL(rt, texCoord).r
+#define SAMPLE_RT_DEPTH_LEVEL(rt, texCoord, level) LoadTextureWGSL(rt, texCoord, level).r
#else
#define SAMPLE_RT_DEPTH(rt, texCoord) SAMPLE_RT(rt, texCoord).r
+#define SAMPLE_RT_DEPTH_LEVEL(rt, texCoord, level) rt.SampleLevel(SamplerPointClamp, texCoord, level).r
#endif
// General purpose constants
diff --git a/Source/Shaders/Utils/Culling.shader b/Source/Shaders/Utils/Culling.shader
index 082d93bf4..f00a2b467 100644
--- a/Source/Shaders/Utils/Culling.shader
+++ b/Source/Shaders/Utils/Culling.shader
@@ -1,9 +1,13 @@
// Copyright (c) Wojciech Figat. All rights reserved.
#include "./Flax/Common.hlsl"
+#include "./Flax/Math/Math.hlsl"
META_CB_BEGIN(0, OcclusionCullingData)
float4x4 ViewProjectionMatrix;
+float2 RTSize;
+float MaxMipLevel;
+uint CullCount;
META_CB_END
// Vertex Shader function for Hardware Occlusion Culling queries bounds projection
@@ -12,3 +16,105 @@ float4 VS_HardwareOcclusionCulling(float3 Position : POSITION0) : SV_Position
{
return mul(float4(Position, 1), ViewProjectionMatrix);
}
+
+#ifdef _CS_HZBCull
+
+RWBuffer HZBResults : register(u0);
+Buffer BoundsBuffer : register(t0);
+Texture2D HiZ : register(t1);
+
+// Compute Shader for HZB culling
+// [Reference: https://interplayoflight.wordpress.com/2017/11/15/experiments-in-gpu-based-occlusion-culling/]
+// [Reference: https://blog.selfshadow.com/publications/practical-visibility/]
+META_CS(true, AUTO)
+[numthreads(64, 1, 1)]
+void CS_HZBCull(uint DispatchThreadId : SV_DispatchThreadID)
+{
+ if (DispatchThreadId >= CullCount)
+ return;
+
+ // Load object bounds
+ float3 bondsMin = BoundsBuffer[DispatchThreadId * 2];
+ float3 bondsMax = BoundsBuffer[DispatchThreadId * 2 + 1];
+ float3 bondsSize = bondsMax - bondsMin;
+
+ // Project bounds onto the screen
+ float3 boundsCorners[] = {
+ bondsMin.xyz,
+ bondsMin.xyz + float3(bondsSize.x,0,0),
+ bondsMin.xyz + float3(0, bondsSize.y,0),
+ bondsMin.xyz + float3(0, 0, bondsSize.z),
+ bondsMin.xyz + float3(bondsSize.xy,0),
+ bondsMin.xyz + float3(0, bondsSize.yz),
+ bondsMin.xyz + float3(bondsSize.x, 0, bondsSize.z),
+ bondsMax.xyz
+ };
+ float closestZ = DEPTH_RANGE_MAX;
+ float2 minUV = 1, maxUV = 0;
+ UNROLL
+ for (uint i = 0; i < 8; i++)
+ {
+ // Transform world-space bounds to NDC
+ float4 clipPos = PROJECT_POINT(float4(boundsCorners[i], 1), ViewProjectionMatrix);
+ clipPos.xyz = clipPos.xyz / clipPos.w;
+
+ // Get min/max UVs
+ clipPos.xy = clipPos.xy * float2(0.5, -0.5) + float2(0.5, 0.5);
+ clipPos.xy = saturate(clipPos.xy);
+ minUV = min(clipPos.xy, minUV);
+ maxUV = max(clipPos.xy, maxUV);
+
+ // Get the closest depth
+#if REVERSE_Z
+ if (clipPos.z < 0)
+ clipPos.z = 1; // Point is behind the camera
+ closestZ = saturate(max(closestZ, clipPos.z));
+#else
+ closestZ = saturate(min(closestZ, clipPos.z));
+#endif
+ }
+
+ // Calculate Hi-Z buffer mip (assumes HZB is power of two)
+#if VULKAN || defined(WGSL) || 1
+ float2 pixelSize = RTSize * (maxUV - minUV) * 2.0f;
+ float mip = floor(log2(max(max(pixelSize.x, pixelSize.y), 1.0f)));
+#else
+ int2 size = (maxUV - minUV) * RTSize;
+ float mip = ceil(log2(max(max(size.x, size.y), 1)));
+#endif
+ mip = clamp(mip, 0, MaxMipLevel);
+ float4 boundsUVs = float4(minUV, maxUV);
+
+ // Texel footprint for the lower (finer-grained) level
+ float mipUp = max(mip - 1, 0);
+ float2 scale = exp2(-mipUp);
+ float2 a = floor(boundsUVs.xy * scale);
+ float2 b = ceil(boundsUVs.zw * scale);
+ float2 dims = b - a;
+
+ // Use the lower level if we only touch <= 2 texels in both dimensions
+ if (dims.x <= 2 && dims.y <= 2)
+ mip = mipUp;
+
+ // Load depths from Hi-Z buffer
+ float4 depths = {
+ SAMPLE_RT_DEPTH_LEVEL(HiZ, boundsUVs.xy, mip),
+ SAMPLE_RT_DEPTH_LEVEL(HiZ, boundsUVs.zy, mip),
+ SAMPLE_RT_DEPTH_LEVEL(HiZ, boundsUVs.xw, mip),
+ SAMPLE_RT_DEPTH_LEVEL(HiZ, boundsUVs.zw, mip)
+ };
+
+ // Find the furthest depth and test it
+#if REVERSE_Z
+ float furthestDepth = Min4(depths);
+ bool visible = closestZ >= furthestDepth;
+#else
+ float furthestDepth = Max4(depths);
+ bool visible = closestZ <= furthestDepth;
+#endif
+
+ // Write culling result
+ HZBResults[DispatchThreadId] = visible ? 1u : 0u;
+}
+
+#endif
diff --git a/Source/Shaders/Utils/MultiScaler.shader b/Source/Shaders/Utils/MultiScaler.shader
index 501e1893d..0fdb8ff7e 100644
--- a/Source/Shaders/Utils/MultiScaler.shader
+++ b/Source/Shaders/Utils/MultiScaler.shader
@@ -2,6 +2,7 @@
#include "./Flax/Common.hlsl"
#include "./Flax/Gather.hlsl"
+#include "./Flax/Math/Math.hlsl"
META_CB_BEGIN(0, Data)
float2 TexelSize;
@@ -29,17 +30,16 @@ float PS_HalfDepth(Quad_VS2PS input)
float4 depths = TextureGatherDepth(Input, input.TexCoord);
#if REVERSE_Z
-#if HZB_CLOSEST
- return max(depths.x, max(depths.y, max(depths.z, depths.w)));
+ float closest = Max4(depths);
+ float furthest = Min4(depths);
#else
- return min(depths.x, min(depths.y, min(depths.z, depths.w)));
+ float closest = Min4(depths);
+ float furthest = Max4(depths);
#endif
-#else
#if HZB_CLOSEST
- return min(depths.x, min(depths.y, min(depths.z, depths.w)));
+ return closest;
#else
- return max(depths.x, max(depths.y, max(depths.z, depths.w)));
-#endif
+ return furthest;
#endif
}