Merge pull request 'perf(tlas): dirty-track the per-frame TLAS instance+metadata upload (#118)' (#140) from claude/issue-118 into master

This commit is contained in:
catbot 2026-06-18 15:07:44 +02:00
commit 2783e47674
5 changed files with 437 additions and 12 deletions

View file

@ -139,9 +139,37 @@ void RenderingElement3D::BuildTLAS(VkCommandBuffer cmd, std::uint32_t index, RTB
tlas.instanceBuffer.Resize(VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_ACCELERATION_STRUCTURE_BUILD_INPUT_READ_ONLY_BIT_KHR | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT, primitiveCount, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
tlas.metadataBuffer.Resize(VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT, primitiveCount, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
}
// A topology change reshuffles which element occupies each slot (a count
// change, and Remove's swap-and-pop within it) and may have reallocated
// the host buffers just above — so every slot's prior contents are
// stale. Clear the per-slot version record (sized to the live count) so
// the copy loop below re-uploads all primitiveCount slots this frame.
tlas.uploadedVersion.assign(primitiveCount, 0);
}
// Copy only the slots whose host-authored data changed since this frame's
// buffers last saw them (#118), tracking the first/last dirty index so the
// flush below covers just that [first, last] envelope rather than the whole
// high-water capacity (the envelope may span a few clean slots between
// sparse dirty ones — flushing a couple of extra entries is far cheaper than
// the per-slot vkFlushMappedMemoryRanges calls true sub-ranges would need).
// A slot is dirty when its element advanced past the version
// recorded for it; an element left at version 0 (never MarkHostDataDirty'd)
// reads dirty every frame, preserving the pre-#118 copy-every-frame
// behaviour for callers that don't opt into dirty tracking. Globally-unique
// versions make this correct under relocation: a slot now holding a
// different element never matches the version recorded for its previous
// occupant. The GPU-owned transform is never copied here regardless — the
// application's compute shader writes it in place earlier in this submission.
bool anyDirty = false;
std::uint32_t dirtyFirst = 0;
std::uint32_t dirtyLast = 0;
for(std::uint32_t i = 0; i < primitiveCount; i++) {
const std::uint64_t version = elements[i]->hostDataVersion;
if (version != 0 && version == tlas.uploadedVersion[i]) {
continue; // slot already holds this element's current host data
}
if (elements[i]->transformOwnedByGpu) {
// Skip the transform field — the application's compute shader
// writes it earlier in this submission. Copy everything else.
@ -156,21 +184,43 @@ void RenderingElement3D::BuildTLAS(VkCommandBuffer cmd, std::uint32_t index, RTB
tlas.instanceBuffer.value[i] = elements[i]->instance;
}
tlas.metadataBuffer.value[i] = elements[i]->userMetadata;
tlas.uploadedVersion[i] = version;
if (!anyDirty) { dirtyFirst = i; anyDirty = true; }
dirtyLast = i;
}
tlas.instanceBuffer.FlushDevice(cmd, VK_ACCESS_MEMORY_READ_BIT, VK_PIPELINE_STAGE_ACCELERATION_STRUCTURE_BUILD_BIT_KHR);
if (anyDirty) {
// Flush (and, for the instance buffer, barrier) only the [dirtyFirst,
// dirtyLast] span actually rewritten this frame. When nothing is dirty
// there were no host writes at all, so both flushes — and the
// instanceBuffer HOST->build barrier — are skipped: the AS build reads
// memory already made visible by the submit that last wrote it, and any
// GPU-owned transform write is ordered by the application's own
// compute->build barrier, not this host barrier.
const std::uint32_t count = dirtyLast - dirtyFirst + 1;
constexpr VkDeviceSize instStride = sizeof(VkAccelerationStructureInstanceKHR);
constexpr VkDeviceSize metaStride = sizeof(std::uint32_t);
// Make the per-frame metadata host writes available before the consuming
// submit. metadataBuffer is not an AS build input (the build never reads it)
// and has no GPU co-write — only the ray shaders read it, in a later pass —
// so it needs neither the build-stage barrier instanceBuffer takes above nor
// a HOST->shader command barrier: the queue submit already orders host
// writes made available before it against every command in the submission,
// exactly as it did when this buffer was HOST_COHERENT. This plain host
// flush is what now makes those writes available on a non-coherent
// (BAR/VRAM) type; it self-gates to a no-op on a coherent type (#60),
// preserving the previous behaviour where the type ends up coherent.
tlas.metadataBuffer.FlushDevice();
tlas.instanceBuffer.FlushDevice(
cmd, VK_ACCESS_MEMORY_READ_BIT, VK_PIPELINE_STAGE_ACCELERATION_STRUCTURE_BUILD_BIT_KHR,
static_cast<VkDeviceSize>(dirtyFirst) * instStride,
static_cast<VkDeviceSize>(count) * instStride);
// Make the per-frame metadata host writes available before the consuming
// submit. metadataBuffer is not an AS build input (the build never reads
// it) and has no GPU co-write — only the ray shaders read it, in a later
// pass — so it needs neither the build-stage barrier instanceBuffer takes
// above nor a HOST->shader command barrier: the queue submit already
// orders host writes made available before it against every command in
// the submission, exactly as it did when this buffer was HOST_COHERENT.
// This ranged host flush is what now makes those writes available on a
// non-coherent (BAR/VRAM) type; it self-gates to a no-op on a coherent
// type (#60), preserving the previous behaviour where the type ends up
// coherent.
tlas.metadataBuffer.FlushDevice(
static_cast<VkDeviceSize>(dirtyFirst) * metaStride,
static_cast<VkDeviceSize>(count) * metaStride);
}
VkAccelerationStructureGeometryInstancesDataKHR instancesData {
.sType = VK_STRUCTURE_TYPE_ACCELERATION_STRUCTURE_GEOMETRY_INSTANCES_DATA_KHR,