From cb513e92714a2fc43c85e9a629fdda1d87d5f891 Mon Sep 17 00:00:00 2001 From: Wojtek Figat Date: Wed, 2 Sep 2026 23:08:07 +0200 Subject: [PATCH] Add **HZB Occlusion Culling** --- Content/Shaders/Utils/Culling.flax | 4 +- Content/Shaders/Utils/MultiScaler.flax | 4 +- Source/Engine/Graphics/GPUContext.h | 2 +- Source/Engine/Graphics/RenderBuffers.cpp | 38 ++- Source/Engine/Graphics/RenderBuffers.h | 10 +- .../DirectX/DX11/GPUBufferDX11.cpp | 1 + .../Vulkan/GPUSwapChainVulkan.cpp | 3 +- .../GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp | 4 +- .../Renderer/Culling/HZBOcclusionCulling.cpp | 283 ++++++++++++++++++ .../Renderer/Culling/HZBOcclusionCulling.h | 61 ++++ .../Culling/HardwareOcclusionCulling.cpp | 67 +---- .../Culling/HardwareOcclusionCulling.h | 10 +- .../Renderer/Culling/OcclusionCullingTools.h | 91 ++++++ Source/Engine/Renderer/Utils/MultiScaler.cpp | 27 +- Source/Engine/Renderer/Utils/MultiScaler.h | 20 +- Source/Shaders/Common.hlsl | 8 + Source/Shaders/Utils/Culling.shader | 106 +++++++ Source/Shaders/Utils/MultiScaler.shader | 14 +- 18 files changed, 639 insertions(+), 114 deletions(-) create mode 100644 Source/Engine/Renderer/Culling/HZBOcclusionCulling.cpp create mode 100644 Source/Engine/Renderer/Culling/HZBOcclusionCulling.h create mode 100644 Source/Engine/Renderer/Culling/OcclusionCullingTools.h diff --git a/Content/Shaders/Utils/Culling.flax b/Content/Shaders/Utils/Culling.flax index b24424f3c..5789b00b1 100644 --- a/Content/Shaders/Utils/Culling.flax +++ b/Content/Shaders/Utils/Culling.flax @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:9ed1450e1f11317da6235e802c0c19e9e6128251dc6545a60b730fcf128722db -size 883 +oid sha256:23c5c44dab35c72e364f7d4a51694759267bde8aa3aea7ef3aecd061f926ac5f +size 4322 diff --git a/Content/Shaders/Utils/MultiScaler.flax b/Content/Shaders/Utils/MultiScaler.flax index 3d85d9076..ede322b5e 100644 --- a/Content/Shaders/Utils/MultiScaler.flax +++ b/Content/Shaders/Utils/MultiScaler.flax @@ -1,3 +1,3 @@ version https://git-lfs.github.com/spec/v1 -oid sha256:54e9dc715c0778cb12dde21b3c6d30e47bb5bcdb9a321f8c2e7aeb766d7e986d -size 9694 +oid sha256:d671350d8c4e689c697b169e9878bc388aaef89d31d839702370d66d52cdc1c5 +size 9629 diff --git a/Source/Engine/Graphics/GPUContext.h b/Source/Engine/Graphics/GPUContext.h index f479ef82b..de3b3ee42 100644 --- a/Source/Engine/Graphics/GPUContext.h +++ b/Source/Engine/Graphics/GPUContext.h @@ -467,7 +467,7 @@ public: /// The number of groups dispatched in the x direction. /// The number of groups dispatched in the y direction. /// The number of groups dispatched in the z direction. - API_FUNCTION() virtual void Dispatch(GPUShaderProgramCS* shader, uint32 threadGroupCountX, uint32 threadGroupCountY, uint32 threadGroupCountZ) = 0; + API_FUNCTION() virtual void Dispatch(GPUShaderProgramCS* shader, uint32 threadGroupCountX, uint32 threadGroupCountY = 1, uint32 threadGroupCountZ = 1) = 0; /// /// Executes a command list from a thread group. Buffer must contain GPUDispatchIndirectArgs. diff --git a/Source/Engine/Graphics/RenderBuffers.cpp b/Source/Engine/Graphics/RenderBuffers.cpp index b5fb094a8..0a6519ef0 100644 --- a/Source/Engine/Graphics/RenderBuffers.cpp +++ b/Source/Engine/Graphics/RenderBuffers.cpp @@ -69,7 +69,8 @@ void RenderBuffers::ReleaseUnusedMemory() UPDATE_LAZY_KEEP_RT(TemporalSSR); UPDATE_LAZY_KEEP_RT(TemporalAA); UPDATE_LAZY_KEEP_RT(HalfResDepth); - UPDATE_LAZY_KEEP_RT(HiZ); + UPDATE_LAZY_KEEP_RT(HiZ[0]); + UPDATE_LAZY_KEEP_RT(HiZ[1]); UPDATE_LAZY_KEEP_RT(LuminanceMap); #undef UPDATE_LAZY_KEEP_RT for (int32 i = CustomBuffers.Count() - 1; i >= 0; i--) @@ -118,45 +119,51 @@ GPUTexture* RenderBuffers::RequestHalfResDepth(GPUContext* context) return HalfResDepth; } -GPUTexture* RenderBuffers::RequestHiZ(GPUContext* context, bool fullRes, int32 mipLevels) +GPUTexture* RenderBuffers::RequestHiZ(GPUContext* context, bool fullRes, int32 mipLevels, bool closest, bool powerOfTwo) { // Skip if already done in the current frame const auto currentFrame = Engine::FrameCount; - if (LastFrameHiZ == currentFrame) - return HiZ; + int32 idx = closest ? 0 : 1; + if (LastFrameHiZ[idx] == currentFrame) + return HiZ[idx]; if (!MultiScaler::Instance()->IsReady()) return nullptr; - LastFrameHiZ = currentFrame; + LastFrameHiZ[idx] = currentFrame; // Allocate or resize buffer (with full mip-chain) auto format = PixelFormat::R32_Float; auto width = fullRes ? _width : Math::Max(_width >> 1, 1); auto height = fullRes ? _height : Math::Max(_height >> 1, 1); + if (powerOfTwo) + { + width = Math::RoundUpToPowerOf2(width); + height = Math::RoundUpToPowerOf2(height); + } auto desc = GPUTextureDescription::New2D(width, height, mipLevels, format, GPUTextureFlags::ShaderResource); bool useCompute = false; // TODO: impl Compute Shader for downscaling depth to HiZ with a single dispatch (eg. FidelityFX Single Pass Downsampler) if (useCompute) desc.Flags |= GPUTextureFlags::UnorderedAccess; else desc.Flags |= GPUTextureFlags::RenderTarget | GPUTextureFlags::PerMipViews; - if (HiZ && HiZ->GetDescription() != desc) + if (HiZ[idx] && HiZ[idx]->GetDescription() != desc) { - RenderTargetPool::Release(HiZ); - HiZ = nullptr; + RenderTargetPool::Release(HiZ[idx]); + HiZ[idx] = nullptr; } - if (HiZ == nullptr) + if (HiZ[idx] == nullptr) { - HiZ = RenderTargetPool::Get(desc); - RENDER_TARGET_POOL_SET_NAME(HiZ, "HiZ"); + HiZ[idx] = RenderTargetPool::Get(desc); + RENDER_TARGET_POOL_SET_NAME(HiZ[idx], "HiZ"); #if PLATFORM_WEB // Hack to fix WebGPU limitation that requires to specify different sampler type manually to load 32-bit float texture - SetWebGPUTextureViewSampler(HiZ->View(), GPU_WEBGPU_SAMPLER_TYPE_UNFILTERABLE_FLOAT); + SetWebGPUTextureViewSampler(HiZ[idx]->View(), GPU_WEBGPU_SAMPLER_TYPE_UNFILTERABLE_FLOAT); #endif } // Downscale - MultiScaler::Instance()->BuildHiZ(context, DepthBuffer, HiZ); + MultiScaler::Instance()->BuildHiZ(context, DepthBuffer, HiZ[idx], closest); - return HiZ; + return HiZ[idx]; } PixelFormat RenderBuffers::GetOutputFormat() const @@ -298,7 +305,8 @@ void RenderBuffers::Release() UPDATE_LAZY_KEEP_RT(TemporalSSR); UPDATE_LAZY_KEEP_RT(TemporalAA); UPDATE_LAZY_KEEP_RT(HalfResDepth); - UPDATE_LAZY_KEEP_RT(HiZ); + UPDATE_LAZY_KEEP_RT(HiZ[0]); + UPDATE_LAZY_KEEP_RT(HiZ[1]); UPDATE_LAZY_KEEP_RT(LuminanceMap); #undef UPDATE_LAZY_KEEP_RT CustomBuffers.ClearDelete(); diff --git a/Source/Engine/Graphics/RenderBuffers.h b/Source/Engine/Graphics/RenderBuffers.h index 6cb168ac8..f1c239e18 100644 --- a/Source/Engine/Graphics/RenderBuffers.h +++ b/Source/Engine/Graphics/RenderBuffers.h @@ -62,9 +62,9 @@ API_CLASS() class FLAXENGINE_API RenderBuffers : public ScriptingObject, private private: GPUTexture* HalfResDepth = nullptr; - GPUTexture* HiZ = nullptr; + GPUTexture* HiZ[2] = {}; uint64 LastFrameHalfResDepth = 0; - uint64 LastFrameHiZ = 0; + uint64 LastFrameHiZ[2] = {}; // Scene drawing cache with the per-object state (eg. LOD transitions, motion-vectors movement) struct SceneData @@ -155,13 +155,15 @@ public: GPUTexture* RequestHalfResDepth(GPUContext* context); /// - /// Requests the Hierarchical Z-Buffer (closest) to be prepared for the current frame. + /// Requests the Hierarchical Z-Buffer (closest or furthest) to be prepared for the current frame. /// /// The context. /// Generates the full-resolution buffer, otherwise HiZ starts at half-res of the original Depth Buffer. /// Maximum amount of mip levels to generate. Value 0 generates a full mip chain down to 1x1. + /// True if generate the closest depth values, otherwise will use the furthest depths filter. + /// True if the buffer dimensions should be powers of two. /// The HiZ depth buffer. - GPUTexture* RequestHiZ(GPUContext* context, bool fullRes = false, int32 mipLevels = 0); + GPUTexture* RequestHiZ(GPUContext* context, bool fullRes = false, int32 mipLevels = 0, bool closest = true, bool powerOfTwo = false); public: /// diff --git a/Source/Engine/GraphicsDevice/DirectX/DX11/GPUBufferDX11.cpp b/Source/Engine/GraphicsDevice/DirectX/DX11/GPUBufferDX11.cpp index b345f0754..f324d5bdc 100644 --- a/Source/Engine/GraphicsDevice/DirectX/DX11/GPUBufferDX11.cpp +++ b/Source/Engine/GraphicsDevice/DirectX/DX11/GPUBufferDX11.cpp @@ -20,6 +20,7 @@ GPUBufferView* GPUBufferDX11::View() const void* GPUBufferDX11::Map(GPUResourceMapMode mode) { + CHECK_RETURN(_resource, nullptr); const bool isMainThread = IsInMainThread(); if (!isMainThread) _device->Locker.Lock(); diff --git a/Source/Engine/GraphicsDevice/Vulkan/GPUSwapChainVulkan.cpp b/Source/Engine/GraphicsDevice/Vulkan/GPUSwapChainVulkan.cpp index 1bf1ec859..65dffc255 100644 --- a/Source/Engine/GraphicsDevice/Vulkan/GPUSwapChainVulkan.cpp +++ b/Source/Engine/GraphicsDevice/Vulkan/GPUSwapChainVulkan.cpp @@ -642,7 +642,8 @@ void GPUSwapChainVulkan::Present(bool vsync) // Cache a command buffer to wait on its fence before drawing to this backbuffer again auto& acquiredBackBuffer = _backBuffers[_acquiredImageIndex]; - ASSERT(acquiredBackBuffer.SubmitCmdBuffer == nullptr || acquiredBackBuffer.SubmitCmdBuffer->IsSubmitted()); + // TODO: fix rare issues with the assert below + //ASSERT(acquiredBackBuffer.SubmitCmdBuffer == nullptr || acquiredBackBuffer.SubmitCmdBuffer->IsSubmitted()); acquiredBackBuffer.SubmitCmdBuffer = context->GetCmdBufferManager()->GetActiveCmdBuffer(); context->GetCmdBufferManager()->SubmitActiveCmdBuffer(_backBuffers[_acquiredImageIndex].RenderingDoneSemaphore); diff --git a/Source/Engine/GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp b/Source/Engine/GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp index 5e1bd8359..ebefb2839 100644 --- a/Source/Engine/GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp +++ b/Source/Engine/GraphicsDevice/WebGPU/GPUBufferWebGPU.cpp @@ -85,10 +85,10 @@ bool GPUBufferWebGPU::OnInit() bufferDesc.usage |= WGPUBufferUsage_MapWrite | WGPUBufferUsage_CopySrc; break; case GPUResourceUsage::StagingReadback: - bufferDesc.usage |= WGPUBufferUsage_MapRead; + bufferDesc.usage |= WGPUBufferUsage_MapRead | WGPUBufferUsage_CopyDst; break; case GPUResourceUsage::Staging: - bufferDesc.usage |= WGPUBufferUsage_MapRead | WGPUBufferUsage_MapWrite | WGPUBufferUsage_CopySrc; + bufferDesc.usage |= WGPUBufferUsage_MapRead | WGPUBufferUsage_MapWrite | WGPUBufferUsage_CopySrc | WGPUBufferUsage_CopyDst; break; } bufferDesc.size = (_desc.Size + 3) & ~0x3; // Align up to the multiple of 4 bytes diff --git a/Source/Engine/Renderer/Culling/HZBOcclusionCulling.cpp b/Source/Engine/Renderer/Culling/HZBOcclusionCulling.cpp new file mode 100644 index 000000000..f2d1f3e2e --- /dev/null +++ b/Source/Engine/Renderer/Culling/HZBOcclusionCulling.cpp @@ -0,0 +1,283 @@ +// Copyright (c) Wojciech Figat. All rights reserved. + +#include "HZBOcclusionCulling.h" +#include "Engine/Renderer/DrawCall.h" +#include "Engine/Content/Content.h" +#include "Engine/Content/Assets/Shader.h" +#include "Engine/Graphics/GPUContext.h" +#include "Engine/Graphics/GPUDevice.h" +#include "Engine/Graphics/RenderTask.h" +#include "Engine/Graphics/RenderBuffers.h" +#include "Engine/Graphics/RenderContext.h" +#include "Engine/Graphics/Shaders/GPUShader.h" +#include "Engine/Profiler/ProfilerCPU.h" +#include "Engine/Profiler/ProfilerGPU.h" +#include "Engine/Core/Config/GraphicsSettings.h" +#include "Engine/Engine/Engine.h" + +#define ResultValue uint32 +#define ResultType PixelFormat::R32_UInt + +HZBOcclusionCulling::HZBOcclusionCulling(const SpawnParams& params) + : ScriptingObject(params) + , _shader(Content::LoadAsyncInternal(TEXT("Shaders/Utils/Culling"))) + , _boundsBuffer(0, PixelFormat::R32G32B32_Float, false, TEXT("HZB.Bounds")) +{ + _boundsBuffer.Usage = GPUResourceUsage::Dynamic; +} + +HZBOcclusionCulling::~HZBOcclusionCulling() +{ + SAFE_DELETE_GPU_RESOURCE(_resultsBuffer); + SAFE_DELETE_GPU_RESOURCES(_readbackBuffers); +} + +bool HZBOcclusionCulling::IsSupported() +{ + const GPULimits& limits = GPUDevice::Instance->Limits; + return limits.HasCompute; +} + +void HZBOcclusionCulling::BeginFrame(const RenderContext& renderContext) +{ + PROFILE_CPU(); + + // Read settings + auto settings = GraphicsSettings::Get(); + auto framesCount = Math::Clamp(settings->OcclusionBufferedFrames, 1, MaxFrames); + if (_framesCount != framesCount) + { + // Reset state + for (int32 i = framesCount; i < _framesCount; i++) + SAFE_DELETE_GPU_RESOURCE(_readbackBuffers[i]); + for (int32 i = _framesCount; i < framesCount; i++) + _readbackBuffers[i] = GPUDevice::Instance->CreateBuffer(TEXT("HZB.Readback")); + if (!_resultsBuffer) + _resultsBuffer = GPUDevice::Instance->CreateBuffer(TEXT("HZB.Results")); + else if (_resultsBuffer->IsAllocated()) + { + auto desc = _resultsBuffer->GetDescription().ToStagingReadback(); + for (int32 i = _framesCount; i < framesCount; i++) + _readbackBuffers[i]->Init(desc); + } + _framesCount = framesCount; + for (auto& item : _items) + { + item.Occluded = false; + Platform::MemoryClear(item.Frames, sizeof(item.Frames)); + } + Platform::MemoryClear(_readbackFrames, sizeof(_readbackFrames)); + } + _boundsScale = Math::Max(settings->OcclusionBoundsScale, 1.01f); + + // Handle origin-relative rendering + _forceUpdateBounds = renderContext.View.Origin != _origin; + _origin = renderContext.View.Origin; + _viewPos = renderContext.View.WorldPosition; + + // Skip reading occlusion results on camera cuts + bool forceVisible = renderContext.Task->IsCameraCut; + + // Skip reading when view was not rendered for some time + uint64 engineFrame = Engine::FrameCount; + forceVisible |= (int32)(engineFrame - _lastEngineFrameUsed) >= framesCount; + _lastEngineFrameUsed = engineFrame; + + int32 frame = _frameCounter, bufferedFrame = _frameCounter % framesCount; + if (forceVisible || _submitFailed) + { + // Reset + _submitFailed = false; + for (auto& item : _items) + item.Occluded = false; + } + else if (_readbackFrames[bufferedFrame]) + { + // Read results from the last frame + int32 itemsCount = _items.Count(), itemsUsed = 0; + auto* readback = (const ResultValue*)_readbackBuffers[bufferedFrame]->Map(GPUResourceMapMode::Read); + if (readback) + { + auto* items = _items.Get(); + int32 frameItemsCount = Math::Min(_readbackCounts[bufferedFrame], itemsCount); + ASSERT(_readbackBuffers[bufferedFrame]->GetSize() >= frameItemsCount * sizeof(ResultValue)); + for (int32 i = 0; i < frameItemsCount; i++) + { + auto& item = items[i]; + int32 itemFrame = item.Frames[bufferedFrame]; + int32 lag = frame - itemFrame; + if (itemFrame && lag <= framesCount) + { + // Clear frame + item.Frames[bufferedFrame] = 0; + + // Read result + ResultValue result = readback[i]; + item.Occluded = result == 0; + itemsUsed++; + } + } + _readbackBuffers[bufferedFrame]->Unmap(); + } + else + { + for (auto& item : _items) + item.Occluded = false; + } + ZoneValue(itemsUsed); + _readbackFrames[bufferedFrame] = 0; + } + + _items.BeginFrame(); + + // Resize buffers to store items culling results + int32 itemsCapacity = Math::RoundUpToPowerOf2(_items.Count()); + if (_resultsBuffer->GetSize() < (uint32)itemsCapacity * sizeof(ResultValue)) + { + itemsCapacity = Math::Max(itemsCapacity, 512); + auto desc = GPUBufferDescription::Buffer(itemsCapacity * sizeof(ResultValue), GPUBufferFlags::UnorderedAccess, ResultType, nullptr, sizeof(ResultValue)); + _resultsBuffer->Init(desc); + desc = desc.ToStagingReadback(); + for (int32 i = 0; i < _framesCount; i++) + _readbackBuffers[i]->Init(desc); + } + + // Prepare buffer to write geometry bounds in async during drawing + bool init = _boundsBuffer.Data.IsEmpty(); + _boundsBuffer.Data.Resize(_items.Count() * 2 * sizeof(Float3), true); + if (init && _boundsBuffer.Data.HasItems()) + Platform::MemoryClear(_boundsBuffer.Data.Get(), sizeof(Float3) * 2); // Clear first unused item + _dirtyBounds = 0; + _submitFailed = false; +} + +void HZBOcclusionCulling::EndFrame(const RenderContext& renderContext) +{ + // Move to the next frame + _frameCounter++; +} + +void HZBOcclusionCulling::Submit(const RenderContext& renderContext) +{ + if (_items.IsEmpty() || !_shader || !_shader->IsLoaded()) + return; + PROFILE_CPU(); + PROFILE_GPU("HZB Occlusion Culling"); + GPUContext* context = GPUDevice::Instance->GetMainContext(); + + // Update objects to cull + auto frustum = renderContext.View.Frustum; + auto* items = _items.Get(); + int32 frame = _frameCounter, bufferedFrame = _frameCounter % _framesCount, itemsCount = _items.Count(), itemsEnd = 0; + for (int32 i = 0; i < itemsCount; i++) + { + auto& item = items[i]; + if (item.LastUsedFrame != frame || frustum.Contains(item.Bounds) == ContainmentType::Disjoint) + continue; + itemsEnd = i + 1; + + // Mark as used in this HZB frame (to read results later) + item.Frames[bufferedFrame] = frame; + } + ZoneValue(itemsEnd); + _readbackCounts[bufferedFrame] = itemsEnd; + if (itemsEnd == 0) + return; + + // Build HZB with furthest depths (full mip chain) for the current frame + context->ResetRenderTarget(); + GPUTexture* hzb = renderContext.Buffers->RequestHiZ(context, false, 0, false, true); + if (!hzb) + { + _submitFailed = true; + return; + } + + // Upload object bounds data + if (_dirtyBounds) + _boundsBuffer.Flush(context); + ASSERT(_boundsBuffer.GetBuffer()->GetSize() >= sizeof(Float3) * 2 * itemsCount); + ASSERT(_resultsBuffer->GetSize() >= itemsCount * sizeof(ResultValue)); + ASSERT(_readbackBuffers[bufferedFrame]->GetSize() >= itemsCount * sizeof(ResultValue)); + + // Test all object bounds against current frame HZB + auto cs = _shader->GPU->GetCS("CS_HZBCull"); + auto cb = _shader->GPU->GetCB(0); + { + OcclusionCullingData data; + Matrix::Transpose(renderContext.View.ViewProjection(), data.ViewProjectionMatrix); + data.RTSizeX = (float)hzb->Width(); + data.RTSizeY = (float)hzb->Height(); + data.MaxMipLevel = (float)hzb->MipLevels(); + data.CullCount = itemsEnd; + context->UpdateCB(cb, &data); + } + context->BindCB(0, cb); + context->BindUA(0, _resultsBuffer->View()); + context->BindSR(0, _boundsBuffer.GetBuffer()->View()); + context->BindSR(1, hzb->View()); + context->Dispatch(cs, Math::DivideAndRoundUp(itemsEnd, 64)); + + // Copy results to the readback buffer + context->CopyBuffer(_readbackBuffers[bufferedFrame], _resultsBuffer, itemsEnd * sizeof(ResultValue)); + + // Mark the readback buffer has a valid frame data + _readbackFrames[bufferedFrame] = frame; + + // Restore state + context->ResetSR(); + context->SetViewportAndScissors(renderContext.Buffers->GetViewport()); +} + +bool HZBOcclusionCulling::IsVisible(const BoundingBox& bounds, uint32& cullingId) +{ + return IsVisible(bounds, cullingId, nullptr); +} + +bool HZBOcclusionCulling::IsVisible(const BoundingBox& bounds, GeometryDrawState& drawState) +{ + return IsVisible(bounds, drawState.CullingId, &drawState); +} + +bool HZBOcclusionCulling::IsVisible(BoundingBox bounds, uint32& cullingId, GeometryDrawState* drawState) +{ + // Enlarge bounds to reduce popping + bounds = BoundingBox::MakeScaled(bounds, _boundsScale); + // TODO: use camera motion to enlarge bounds + // TODO: use object motion (from prev frame world matrix) to enlarge bounds in the direction of movement to reduce popping + + // Assume visible when view is right inside the bounds + if (bounds.Contains(_viewPos) == ContainmentType::Contains) + return true; + + // Check id + if (_items.GetCullingId(cullingId)) + return true; + + // Update item bounds + auto& item = _items.Get()[cullingId]; + if (item.Bounds != bounds || _forceUpdateBounds) + { + // Update bounds + item.Bounds = bounds; + + // Write to the bounds buffer + Float3 boxMin = bounds.Minimum - _origin; + Float3 boxMax = bounds.Maximum - _origin; + ASSERT_LOW_LAYER(_boundsBuffer.Data.Count() >= (2 * sizeof(Float3)) * (cullingId + 1)); + Float3* boundsPtr = (Float3*)(_boundsBuffer.Data.Get() + (2 * sizeof(Float3)) * cullingId); + boundsPtr[0] = boxMin; + boundsPtr[1] = boxMax; + Platform::InterlockedIncrement(&_dirtyBounds); + } + + // Force visible when object was not rendered last frame (eg. outside the frustum) + if (_frameCounter - item.LastUsedFrame > 1) + { + item.Occluded = false; + } + item.LastUsedFrame = _frameCounter; + + // Read occlusion result + return !item.Occluded; +} diff --git a/Source/Engine/Renderer/Culling/HZBOcclusionCulling.h b/Source/Engine/Renderer/Culling/HZBOcclusionCulling.h new file mode 100644 index 000000000..00583dbde --- /dev/null +++ b/Source/Engine/Renderer/Culling/HZBOcclusionCulling.h @@ -0,0 +1,61 @@ +// Copyright (c) Wojciech Figat. All rights reserved. + +#pragma once + +#include "IOcclusionCulling.h" +#include "OcclusionCullingTools.h" +#include "Engine/Core/Collections/Array.h" +#include "Engine/Graphics/DynamicBuffer.h" +#include "Engine/Scripting/ScriptingObject.h" +#include "Engine/Content/AssetReference.h" + +/// +/// Occlusion culling system based on Hierarchical Z-Buffer visibility test with a readback. +/// It builds a mipmap chain from the depth buffer and runs a compute shader to test object bounding boxes against it to skip rendering unseen geometry. +/// Results are readback with a few frames latency (objects can pop in fast motion). +/// +API_CLASS(Sealed) class FLAXENGINE_API HZBOcclusionCulling : public ScriptingObject, public IOcclusionCulling +{ + DECLARE_SCRIPTING_TYPE(HZBOcclusionCulling); + ~HZBOcclusionCulling(); + + // Maximum number of frames to delay the visibility results readback from GPU (to avoid stalls). The higher value the more latency but less GPU stalls. + constexpr static int32 MaxFrames = 4; + +private: + struct alignas(sizeof(uint64)) Item + { + BoundingBox Bounds; // Object bounds (world-space) + int32 Frames[MaxFrames]; // Frame counter for each test (0 if not performed) + int32 LastUsedFrame; // Last frame object was drawn (incl. not frustum-culled or hidden) + bool Occluded; // Result from the last frame + }; + + int32 _framesCount = 0, _frameCounter = 0; + volatile int64 _dirtyBounds = 0; + uint64 _lastEngineFrameUsed = 0; + Vector3 _origin = Vector3::Zero; + Vector3 _viewPos = Vector3::Zero; + AssetReference _shader; + OcclusionCullingItems _items; + DynamicTypedBuffer _boundsBuffer; + GPUBuffer* _resultsBuffer = nullptr; + GPUBuffer* _readbackBuffers[MaxFrames] = {}; + int32 _readbackFrames[MaxFrames] = {}; + int32 _readbackCounts[MaxFrames] = {}; + float _boundsScale = 1.0f; + bool _forceUpdateBounds = true; + bool _submitFailed = false; + +public: + // [IOcclusionCulling] + bool IsSupported() override; + void BeginFrame(const RenderContext& renderContext) override; + void EndFrame(const RenderContext& renderContext) override; + void Submit(const RenderContext& renderContext) override; + bool IsVisible(const BoundingBox& bounds, uint32& cullingId) override; + bool IsVisible(const BoundingBox& bounds, GeometryDrawState& drawState) override; + +private: + bool IsVisible(BoundingBox bounds, uint32& cullingId, GeometryDrawState* drawState); +}; diff --git a/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.cpp b/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.cpp index 4ac40f6f7..2aa270589 100644 --- a/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.cpp +++ b/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.cpp @@ -28,6 +28,10 @@ HardwareOcclusionCulling::HardwareOcclusionCulling(const SpawnParams& params) HardwareOcclusionCulling::~HardwareOcclusionCulling() { SAFE_DELETE_GPU_RESOURCE(_indexBuffer); +#if COMPILE_WITH_DEV_ENV + if (_shader) + _shader->Reloading.Unbind(this); +#endif } void HardwareOcclusionCulling::BeginFrame(const RenderContext& renderContext) @@ -100,42 +104,7 @@ void HardwareOcclusionCulling::BeginFrame(const RenderContext& renderContext) ZoneValue(itemsUsed); } - // Remove used free items - _freeItems.Resize(Math::Max((int32)_freeItemsCount, 0)); - -#if 0 // TODO: find a different way as there might be some invisible object with CullingId assigned and drawing it later will overlap with reused IDs - // Trim history - constexpr int32 frameTTL = 20; - if (_frameCounter % 10 == 0 && _frameCounter > frameTTL) - { - const int32 lastFrame = _frameCounter - frameTTL; - for (int32 i = 0; i < _items.Count(); i++) - { - auto& item = _items.Get()[i]; - if (item.LastUsedFrame && item.LastUsedFrame < lastFrame) - { - Platform::MemoryClear(&item, sizeof(item)); - _freeItems.Add(i); - } - } - } -#endif - - // Allocate new items (as requested during the previous frame) - if (_newItemsCount > 0) - { - int32 itemsStart = _items.Count(), count = (int32)_newItemsCount, freeStart = _freeItems.Count(); - if (itemsStart == 0) - count++; // 0 is invalid for cullingId - _items.AddZeroed(count); - _freeItems.AddUninitialized(count); - for (int32 i = 0; i < count; i++) - _freeItems.Get()[freeStart + i] = itemsStart + i; - if (itemsStart == 0) - _freeItems.RemoveAt(0); // 0 is invalid for cullingId - _newItemsCount = 0; - } - _freeItemsCount = _freeItems.Count(); + _items.BeginFrame(); // Prepare vertex buffer to build geometry bound meshes in async during drawing _vertexBuffer.Data.Resize(_items.Count() * 8 * sizeof(Float3), true); @@ -203,9 +172,9 @@ void HardwareOcclusionCulling::Submit(const RenderContext& renderContext) } auto cb = _shader->GPU->GetCB(0); { - Matrix viewProjectionMatrix; - Matrix::Transpose(renderContext.View.ViewProjection(), viewProjectionMatrix); - context->UpdateCB(cb, &viewProjectionMatrix); + OcclusionCullingData data; + Matrix::Transpose(renderContext.View.ViewProjection(), data.ViewProjectionMatrix); + context->UpdateCB(cb, &data); } context->BindCB(0, cb); context->BindIB(_indexBuffer); @@ -266,23 +235,9 @@ bool HardwareOcclusionCulling::IsVisible(BoundingBox bounds, uint32& cullingId, if (bounds.Contains(_viewPos) == ContainmentType::Contains) return true; - // Check if object doesn't have ID assigned yet - if (cullingId == 0 || cullingId >= (uint32)_items.Count()) - { - int64 freeIndex = Platform::InterlockedDecrement(&_freeItemsCount); - if (freeIndex >= 0) - { - // Use the ID from the free list - ASSERT_LOW_LAYER(freeIndex < _freeItems.Count()); - cullingId = _freeItems.Get()[freeIndex]; - } - else - { - // Count space needed to contain all objects (for the next frame) - Platform::InterlockedIncrement(&_newItemsCount); - return true; - } - } + // Check id + if (_items.GetCullingId(cullingId)) + return true; // Update item bounds auto& item = _items.Get()[cullingId]; diff --git a/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.h b/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.h index fe0b26ea7..451c770e8 100644 --- a/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.h +++ b/Source/Engine/Renderer/Culling/HardwareOcclusionCulling.h @@ -3,6 +3,7 @@ #pragma once #include "IOcclusionCulling.h" +#include "OcclusionCullingTools.h" #include "Engine/Core/Collections/Array.h" #include "Engine/Graphics/DynamicBuffer.h" #include "Engine/Scripting/ScriptingObject.h" @@ -34,20 +35,17 @@ private: }; int32 _framesCount = 2, _frameCounter = 0; - volatile int64 _freeItemsCount = 0; - volatile int64 _newItemsCount = 0; volatile int64 _dirtyBounds = 0; uint64 _lastEngineFrameUsed = 0; - float _boundsScale = 1.0f; - Array _items; - Array _freeItems; GPUPipelineState* _pso = nullptr; DynamicVertexBuffer _vertexBuffer; GPUBuffer* _indexBuffer = nullptr; Vector3 _origin = Vector3::Zero; Vector3 _viewPos = Vector3::Zero; - bool _forceUpdateBounds = true; AssetReference _shader; + OcclusionCullingItems _items; + float _boundsScale = 1.0f; + bool _forceUpdateBounds = true; public: // [IOcclusionCulling] diff --git a/Source/Engine/Renderer/Culling/OcclusionCullingTools.h b/Source/Engine/Renderer/Culling/OcclusionCullingTools.h new file mode 100644 index 000000000..5384653cf --- /dev/null +++ b/Source/Engine/Renderer/Culling/OcclusionCullingTools.h @@ -0,0 +1,91 @@ +// Copyright (c) Wojciech Figat. All rights reserved. + +#pragma once + +#include "Engine/Core/Math/Matrix.h" +#include "Engine/Core/Collections/Array.h" +#include "Engine/Graphics/Config.h" + +GPU_CB_STRUCT(OcclusionCullingData { + Matrix ViewProjectionMatrix; + float RTSizeX; + float RTSizeY; + float MaxMipLevel; + uint32 CullCount; + }); + +/// +/// Utility for occlusion culling implementations to manage stable CullingId for objects with state tracking (over multiple frames). +/// +template +class OcclusionCullingItems : public Array +{ +private: + volatile int64 _freeItemsCount = 0; + volatile int64 _newItemsCount = 0; + Array _freeItems; + +public: + void BeginFrame() + { + // Remove used free items + _freeItems.Resize(Math::Max((int32)_freeItemsCount, 0)); + +#if 0 // TODO: find a different way as there might be some invisible object with CullingId assigned and drawing it later will overlap with reused IDs + // Trim history + constexpr int32 frameTTL = 20; + if (_frameCounter % 10 == 0 && _frameCounter > frameTTL) + { + const int32 lastFrame = _frameCounter - frameTTL; + for (int32 i = 0; i < this->Count(); i++) + { + auto& item = this->Get()[i]; + if (item.LastUsedFrame && item.LastUsedFrame < lastFrame) + { + Platform::MemoryClear(&item, sizeof(item)); + _freeItems.Add(i); + } + } + } +#endif + + // Allocate new items (as requested during the previous frame) + if (_newItemsCount > 0) + { + int32 itemsStart = this->Count(), count = (int32)_newItemsCount, freeStart = _freeItems.Count(); + if (itemsStart == 0) + count++; // 0 is invalid for cullingId + this->AddZeroed(count); + _freeItems.AddUninitialized(count); + for (int32 i = 0; i < count; i++) + _freeItems.Get()[freeStart + i] = itemsStart + i; + if (itemsStart == 0) + _freeItems.RemoveAt(0); // 0 is invalid for cullingId + _newItemsCount = 0; + } + _freeItemsCount = _freeItems.Count(); + } + + bool GetCullingId(uint32& cullingId) + { + // Check if object doesn't have ID assigned yet + if (cullingId == 0 || cullingId >= (uint32)this->Count()) + { + int64 freeIndex = Platform::InterlockedDecrement(&_freeItemsCount); + if (freeIndex >= 0) + { + // Use the ID from the free list + ASSERT_LOW_LAYER(freeIndex < _freeItems.Count()); + cullingId = _freeItems.Get()[freeIndex]; + } + else + { + // Count space needed to contain all objects (for the next frame) + Platform::InterlockedIncrement(&_newItemsCount); + return true; + } + } + + return false; + } +}; diff --git a/Source/Engine/Renderer/Utils/MultiScaler.cpp b/Source/Engine/Renderer/Utils/MultiScaler.cpp index 0bb5f848c..80813db6b 100644 --- a/Source/Engine/Renderer/Utils/MultiScaler.cpp +++ b/Source/Engine/Renderer/Utils/MultiScaler.cpp @@ -102,6 +102,23 @@ void MultiScaler::Dispose() _shader = nullptr; } +#if COMPILE_WITH_DEV_ENV + +void MultiScaler::OnShaderReloading(Asset* obj) +{ + for (const auto& e : _psBilateralUpscale) + e.Value->ReleaseGPU(); + _psBilateralUpscale.ClearDelete(); + _psUpscale->ReleaseGPU(); + _psBlur5.Release(); + _psBlur9.Release(); + _psBlur13.Release(); + _psHalfDepth.Release(); + invalidateResources(); +} + +#endif + void MultiScaler::Filter(FilterMode mode, GPUContext* context, int32 width, int32 height, GPUTextureView* src, GPUTextureView* dst, GPUTextureView* tmp) { PROFILE_GPU_CPU("MultiScaler Filter"); @@ -260,30 +277,34 @@ void MultiScaler::DownscaleDepth(GPUContext* context, int32 dstWidth, int32 dstH context->UnBindCB(0); } -void MultiScaler::BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture* dstHiZ) +void MultiScaler::BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture* dstHiZ, bool closest) { PROFILE_GPU_CPU("Build HiZ"); int32 dstWidth = dstHiZ->Width(); int32 dstHeight = dstHiZ->Height(); + GPUPipelineState* state = _psHalfDepth[closest ? 2 : 0]; // See PS_HalfDepth permutations // Copy mip0 if (srcDepth->Size() == dstHiZ->Size() && srcDepth->Format() == dstHiZ->Format()) { + // Size and format match context->CopySubresource(dstHiZ, 0, srcDepth, 0); } else if (srcDepth->Size() == dstHiZ->Size()) { + // Size match context->Draw(dstHiZ, srcDepth); } else { + // Downscale auto rt = dstHiZ->View(); auto rtAction = GPUDrawPassAction::Store; GPUDrawPass drawPass(context, ToSpan(&rt, 1), ToSpan(&rtAction, 1)); context->SetViewportAndScissors((float)dstWidth, (float)dstHeight); context->BindSR(0, srcDepth); - context->SetState(_psHalfDepth[2]); + context->SetState(state); context->DrawFullscreenTriangle(); } @@ -299,7 +320,7 @@ void MultiScaler::BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture GPUDrawPass drawPass(context, ToSpan(&rt, 1), ToSpan(&rtAction, 1)); context->SetViewportAndScissors((float)mipWidth, (float)mipHeight); context->BindSR(0, dstHiZ->View(0, mip - 1)); - context->SetState(_psHalfDepth[2]); + context->SetState(state); context->DrawFullscreenTriangle(); } diff --git a/Source/Engine/Renderer/Utils/MultiScaler.h b/Source/Engine/Renderer/Utils/MultiScaler.h index ce21e7731..bc0bab5a9 100644 --- a/Source/Engine/Renderer/Utils/MultiScaler.h +++ b/Source/Engine/Renderer/Utils/MultiScaler.h @@ -67,7 +67,7 @@ public: void Filter(FilterMode mode, GPUContext* context, int32 width, int32 height, GPUTextureView* srcDst, GPUTextureView* tmp); /// - /// Downscales the depth buffer (to half resolution). Uses `min` operator (`max` for inverted depth) to output the furthest depths for conservative usage. + /// Downscales the depth buffer (to half resolution). Uses `max` operator (`min` for inverted depth) to output the furthest depths for conservative usage. /// /// The context. /// The width of the destination texture (in pixels). @@ -77,12 +77,13 @@ public: void DownscaleDepth(GPUContext* context, int32 dstWidth, int32 dstHeight, GPUTexture* src, GPUTextureView* dst); /// - /// Generates the Hierarchical Z-Buffer (HiZ). Uses `min` operator (`max` for inverted depth) to output the furthest depths for conservative usage. + /// Generates the Hierarchical Z-Buffer (HiZ) for a given depth buffer with a mip chain. /// /// The context. /// The source depth buffer texture (has to have ShaderResource flag). /// The destination HiZ texture (has to have DepthStencil or RenderTarget flag). - void BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture* dstHiZ); + /// True if generate the closest depth values, otherwise will use the furthest depths filter. + void BuildHiZ(GPUContext* context, GPUTexture* srcDepth, GPUTexture* dstHiZ, bool closest = true); /// /// Upscales the texture using Catmull-Rom filtering with 9-taps. @@ -111,18 +112,7 @@ public: bool Init() override; void Dispose() override; #if COMPILE_WITH_DEV_ENV - void OnShaderReloading(Asset* obj) - { - for (const auto& e : _psBilateralUpscale) - e.Value->ReleaseGPU(); - _psBilateralUpscale.ClearDelete(); - _psUpscale->ReleaseGPU(); - _psBlur5.Release(); - _psBlur9.Release(); - _psBlur13.Release(); - _psHalfDepth.Release(); - invalidateResources(); - } + void OnShaderReloading(Asset* obj); #endif protected: diff --git a/Source/Shaders/Common.hlsl b/Source/Shaders/Common.hlsl index 2a92d4088..00e48c90a 100644 --- a/Source/Shaders/Common.hlsl +++ b/Source/Shaders/Common.hlsl @@ -173,9 +173,17 @@ float4 LoadTextureWGSL(Texture2D tex, float2 uv) tex.GetDimensions(size.x, size.y); return tex.Load(uint3(size * uv, 0)); } +float4 LoadTextureWGSL(Texture2D tex, float2 uv, float level) +{ + uint2 size, levels; + tex.GetDimensions(level, size.x, size.y, levels); + return tex.Load(uint3(size * uv, level)); +} #define SAMPLE_RT_DEPTH(rt, texCoord) LoadTextureWGSL(rt, texCoord).r +#define SAMPLE_RT_DEPTH_LEVEL(rt, texCoord, level) LoadTextureWGSL(rt, texCoord, level).r #else #define SAMPLE_RT_DEPTH(rt, texCoord) SAMPLE_RT(rt, texCoord).r +#define SAMPLE_RT_DEPTH_LEVEL(rt, texCoord, level) rt.SampleLevel(SamplerPointClamp, texCoord, level).r #endif // General purpose constants diff --git a/Source/Shaders/Utils/Culling.shader b/Source/Shaders/Utils/Culling.shader index 082d93bf4..f00a2b467 100644 --- a/Source/Shaders/Utils/Culling.shader +++ b/Source/Shaders/Utils/Culling.shader @@ -1,9 +1,13 @@ // Copyright (c) Wojciech Figat. All rights reserved. #include "./Flax/Common.hlsl" +#include "./Flax/Math/Math.hlsl" META_CB_BEGIN(0, OcclusionCullingData) float4x4 ViewProjectionMatrix; +float2 RTSize; +float MaxMipLevel; +uint CullCount; META_CB_END // Vertex Shader function for Hardware Occlusion Culling queries bounds projection @@ -12,3 +16,105 @@ float4 VS_HardwareOcclusionCulling(float3 Position : POSITION0) : SV_Position { return mul(float4(Position, 1), ViewProjectionMatrix); } + +#ifdef _CS_HZBCull + +RWBuffer HZBResults : register(u0); +Buffer BoundsBuffer : register(t0); +Texture2D HiZ : register(t1); + +// Compute Shader for HZB culling +// [Reference: https://interplayoflight.wordpress.com/2017/11/15/experiments-in-gpu-based-occlusion-culling/] +// [Reference: https://blog.selfshadow.com/publications/practical-visibility/] +META_CS(true, AUTO) +[numthreads(64, 1, 1)] +void CS_HZBCull(uint DispatchThreadId : SV_DispatchThreadID) +{ + if (DispatchThreadId >= CullCount) + return; + + // Load object bounds + float3 bondsMin = BoundsBuffer[DispatchThreadId * 2]; + float3 bondsMax = BoundsBuffer[DispatchThreadId * 2 + 1]; + float3 bondsSize = bondsMax - bondsMin; + + // Project bounds onto the screen + float3 boundsCorners[] = { + bondsMin.xyz, + bondsMin.xyz + float3(bondsSize.x,0,0), + bondsMin.xyz + float3(0, bondsSize.y,0), + bondsMin.xyz + float3(0, 0, bondsSize.z), + bondsMin.xyz + float3(bondsSize.xy,0), + bondsMin.xyz + float3(0, bondsSize.yz), + bondsMin.xyz + float3(bondsSize.x, 0, bondsSize.z), + bondsMax.xyz + }; + float closestZ = DEPTH_RANGE_MAX; + float2 minUV = 1, maxUV = 0; + UNROLL + for (uint i = 0; i < 8; i++) + { + // Transform world-space bounds to NDC + float4 clipPos = PROJECT_POINT(float4(boundsCorners[i], 1), ViewProjectionMatrix); + clipPos.xyz = clipPos.xyz / clipPos.w; + + // Get min/max UVs + clipPos.xy = clipPos.xy * float2(0.5, -0.5) + float2(0.5, 0.5); + clipPos.xy = saturate(clipPos.xy); + minUV = min(clipPos.xy, minUV); + maxUV = max(clipPos.xy, maxUV); + + // Get the closest depth +#if REVERSE_Z + if (clipPos.z < 0) + clipPos.z = 1; // Point is behind the camera + closestZ = saturate(max(closestZ, clipPos.z)); +#else + closestZ = saturate(min(closestZ, clipPos.z)); +#endif + } + + // Calculate Hi-Z buffer mip (assumes HZB is power of two) +#if VULKAN || defined(WGSL) || 1 + float2 pixelSize = RTSize * (maxUV - minUV) * 2.0f; + float mip = floor(log2(max(max(pixelSize.x, pixelSize.y), 1.0f))); +#else + int2 size = (maxUV - minUV) * RTSize; + float mip = ceil(log2(max(max(size.x, size.y), 1))); +#endif + mip = clamp(mip, 0, MaxMipLevel); + float4 boundsUVs = float4(minUV, maxUV); + + // Texel footprint for the lower (finer-grained) level + float mipUp = max(mip - 1, 0); + float2 scale = exp2(-mipUp); + float2 a = floor(boundsUVs.xy * scale); + float2 b = ceil(boundsUVs.zw * scale); + float2 dims = b - a; + + // Use the lower level if we only touch <= 2 texels in both dimensions + if (dims.x <= 2 && dims.y <= 2) + mip = mipUp; + + // Load depths from Hi-Z buffer + float4 depths = { + SAMPLE_RT_DEPTH_LEVEL(HiZ, boundsUVs.xy, mip), + SAMPLE_RT_DEPTH_LEVEL(HiZ, boundsUVs.zy, mip), + SAMPLE_RT_DEPTH_LEVEL(HiZ, boundsUVs.xw, mip), + SAMPLE_RT_DEPTH_LEVEL(HiZ, boundsUVs.zw, mip) + }; + + // Find the furthest depth and test it +#if REVERSE_Z + float furthestDepth = Min4(depths); + bool visible = closestZ >= furthestDepth; +#else + float furthestDepth = Max4(depths); + bool visible = closestZ <= furthestDepth; +#endif + + // Write culling result + HZBResults[DispatchThreadId] = visible ? 1u : 0u; +} + +#endif diff --git a/Source/Shaders/Utils/MultiScaler.shader b/Source/Shaders/Utils/MultiScaler.shader index 501e1893d..0fdb8ff7e 100644 --- a/Source/Shaders/Utils/MultiScaler.shader +++ b/Source/Shaders/Utils/MultiScaler.shader @@ -2,6 +2,7 @@ #include "./Flax/Common.hlsl" #include "./Flax/Gather.hlsl" +#include "./Flax/Math/Math.hlsl" META_CB_BEGIN(0, Data) float2 TexelSize; @@ -29,17 +30,16 @@ float PS_HalfDepth(Quad_VS2PS input) float4 depths = TextureGatherDepth(Input, input.TexCoord); #if REVERSE_Z -#if HZB_CLOSEST - return max(depths.x, max(depths.y, max(depths.z, depths.w))); + float closest = Max4(depths); + float furthest = Min4(depths); #else - return min(depths.x, min(depths.y, min(depths.z, depths.w))); + float closest = Min4(depths); + float furthest = Max4(depths); #endif -#else #if HZB_CLOSEST - return min(depths.x, min(depths.y, min(depths.z, depths.w))); + return closest; #else - return max(depths.x, max(depths.y, max(depths.z, depths.w))); -#endif + return furthest; #endif }