2026-01-28 18:51:11 +01:00
|
|
|
/*
|
|
|
|
|
Crafter®.Graphics
|
|
|
|
|
Copyright (C) 2026 Catcrafts®
|
|
|
|
|
Catcrafts.net
|
|
|
|
|
|
|
|
|
|
This library is free software; you can redistribute it and/or
|
|
|
|
|
modify it under the terms of the GNU Lesser General Public
|
|
|
|
|
License as published by the Free Software Foundation; either
|
|
|
|
|
version 3.0 of the License, or (at your option) any later version.
|
|
|
|
|
|
|
|
|
|
This library is distributed in the hope that it will be useful,
|
|
|
|
|
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
|
|
|
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
|
|
|
|
Lesser General Public License for more details.
|
|
|
|
|
|
|
|
|
|
You should have received a copy of the GNU Lesser General Public
|
|
|
|
|
License along with this library; if not, write to the Free Software
|
|
|
|
|
Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
|
|
|
|
*/
|
|
|
|
|
module;
|
|
|
|
|
#include <vulkan/vulkan_core.h>
|
2026-03-09 20:10:19 +01:00
|
|
|
module Crafter.Graphics:RenderingElement3D_impl;
|
|
|
|
|
import :RenderingElement3D;
|
2026-01-28 18:51:11 +01:00
|
|
|
import std;
|
|
|
|
|
|
|
|
|
|
using namespace Crafter;
|
|
|
|
|
|
2026-03-09 21:50:24 +01:00
|
|
|
|
2026-03-09 20:10:19 +01:00
|
|
|
std::vector<RenderingElement3D*> RenderingElement3D::elements;
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
void RenderingElement3D::Add(RenderingElement3D* e) {
|
|
|
|
|
e->indexInElements = static_cast<std::uint32_t>(elements.size());
|
|
|
|
|
elements.push_back(e);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
void RenderingElement3D::Remove(RenderingElement3D* e) {
|
|
|
|
|
// Idempotent: callers like Builder ghost flow toggle elements in/out
|
|
|
|
|
// and may try to remove an already-removed element.
|
|
|
|
|
std::uint32_t idx = e->indexInElements;
|
|
|
|
|
if (idx == std::numeric_limits<std::uint32_t>::max()) return;
|
|
|
|
|
std::uint32_t last = static_cast<std::uint32_t>(elements.size() - 1);
|
|
|
|
|
if (idx != last) {
|
|
|
|
|
elements[idx] = elements[last];
|
|
|
|
|
elements[idx]->indexInElements = idx;
|
|
|
|
|
}
|
|
|
|
|
elements.pop_back();
|
|
|
|
|
e->indexInElements = std::numeric_limits<std::uint32_t>::max();
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-16 15:49:17 +02:00
|
|
|
void RenderingElement3D::BuildTLAS(VkCommandBuffer cmd, std::uint32_t index, RTBuildPreference preference) {
|
2026-05-05 23:49:29 +02:00
|
|
|
auto& tlas = tlases[index];
|
|
|
|
|
const std::uint32_t primitiveCount = static_cast<std::uint32_t>(elements.size());
|
|
|
|
|
|
2026-06-16 15:49:17 +02:00
|
|
|
// ALLOW_UPDATE is always set — BuildTLAS refits in place on later
|
|
|
|
|
// frames. The FAST_TRACE/FAST_BUILD preference is layered on top; the
|
|
|
|
|
// two preference bits are mutually exclusive, so exactly one is chosen.
|
|
|
|
|
// Both bits are baked into the AS at BUILD time and must be identical on
|
|
|
|
|
// every subsequent UPDATE, so a change in `preference` is treated as a
|
|
|
|
|
// topology change below to force a fresh rebuild with the new flags.
|
|
|
|
|
const VkBuildAccelerationStructureFlagsKHR buildFlags =
|
|
|
|
|
VK_BUILD_ACCELERATION_STRUCTURE_ALLOW_UPDATE_BIT_KHR
|
|
|
|
|
| (preference == RTBuildPreference::FastBuild
|
|
|
|
|
? VK_BUILD_ACCELERATION_STRUCTURE_PREFER_FAST_BUILD_BIT_KHR
|
|
|
|
|
: VK_BUILD_ACCELERATION_STRUCTURE_PREFER_FAST_TRACE_BIT_KHR);
|
|
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
// Refit (UPDATE) is allowed when the count matches the count this AS
|
2026-06-16 15:49:17 +02:00
|
|
|
// was last built for and the build flags are unchanged. A change forces
|
|
|
|
|
// a full rebuild because the AS storage and instance buffer were sized
|
|
|
|
|
// for the old count (and a refit must inherit the original build flags).
|
|
|
|
|
// Refit is dramatically cheaper at scale (millions of instances) — it
|
|
|
|
|
// walks the existing BVH and updates AABBs rather than reconstructing
|
|
|
|
|
// topology.
|
2026-05-05 23:49:29 +02:00
|
|
|
const bool topologyChanged =
|
|
|
|
|
tlas.accelerationStructure == VK_NULL_HANDLE
|
2026-06-16 15:49:17 +02:00
|
|
|
|| primitiveCount != tlas.builtInstanceCount
|
|
|
|
|
|| buildFlags != tlas.builtFlags;
|
2026-05-05 23:49:29 +02:00
|
|
|
|
2026-04-30 23:15:43 +02:00
|
|
|
{
|
|
|
|
|
VkMemoryBarrier asBarrier {
|
|
|
|
|
.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER,
|
|
|
|
|
.srcAccessMask = VK_ACCESS_ACCELERATION_STRUCTURE_WRITE_BIT_KHR,
|
|
|
|
|
.dstAccessMask = VK_ACCESS_ACCELERATION_STRUCTURE_READ_BIT_KHR | VK_ACCESS_ACCELERATION_STRUCTURE_WRITE_BIT_KHR
|
|
|
|
|
};
|
|
|
|
|
vkCmdPipelineBarrier(cmd,
|
|
|
|
|
VK_PIPELINE_STAGE_ACCELERATION_STRUCTURE_BUILD_BIT_KHR,
|
|
|
|
|
VK_PIPELINE_STAGE_ACCELERATION_STRUCTURE_BUILD_BIT_KHR,
|
|
|
|
|
0, 1, &asBarrier, 0, nullptr, 0, nullptr);
|
|
|
|
|
}
|
|
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
if (topologyChanged) {
|
2026-06-16 18:24:09 +00:00
|
|
|
// Grow the host-visible inputs on a high-water mark rather than
|
|
|
|
|
// resizing to the exact count every topology change. These buffers
|
|
|
|
|
// only need to hold *at least* primitiveCount entries — the AS build
|
|
|
|
|
// reads exactly primitiveCount of them via tlasRangeInfo, and the
|
|
|
|
|
// copy loop below writes [0, primitiveCount) — so an allocation left
|
|
|
|
|
// over from a larger earlier frame is reused as-is. This drops the
|
|
|
|
|
// two host-buffer reallocations on a count decrease (and on an
|
|
|
|
|
// increase that still fits the previous high-water capacity),
|
|
|
|
|
// leaving only the AS storage + scratch rebuild below, which is tied
|
|
|
|
|
// to the AS itself and dominates this path regardless.
|
|
|
|
|
//
|
|
|
|
|
// instanceBuffer and metadataBuffer grow in lockstep (always resized
|
|
|
|
|
// together to the same entry count), so one capacity check covers
|
|
|
|
|
// both. size is only meaningful once buffer is non-null, so the
|
|
|
|
|
// null check must short-circuit before the division.
|
2026-05-05 23:49:29 +02:00
|
|
|
// STORAGE_BUFFER_BIT is required because the application's compute
|
2026-06-16 18:24:09 +00:00
|
|
|
// shaders bind these buffers as storage SSBOs (e.g. to write
|
2026-05-05 23:49:29 +02:00
|
|
|
// per-instance transforms directly into the TLAS instance data).
|
perf(rt): allocate TLAS instance buffer in BAR/VRAM, not system RAM (#65)
The TLAS instance buffer is rebuilt/refit every frame and is co-written by
both the CPU (host-authored fields) and the GPU (the compute shader writes
the transform field in place for transformOwnedByGpu elements). Allocated
HOST_VISIBLE | HOST_COHERENT (system RAM), both GPU accesses crossed PCIe:
the compute write of the transform, and the AS build's read of the instances.
Upgrade the request to HOST_VISIBLE with a best-effort DEVICE_LOCAL preference
(BAR/VRAM), keeping the single shared, persistently-mapped buffer — no staging,
no copy, no extra barrier. A whole-buffer staging copy was rejected: it would
clobber the GPU-written transforms in this in-place co-write design. Now the
compute write and AS build stay in local VRAM; the CPU's write-only,
sequential, never-read-back field writes are the ideal write-combined BAR load.
Correctness:
- DEVICE_LOCAL is a preference only (depends on #59): GetMemoryType falls back
to plain HOST_VISIBLE (the previous behaviour) on GPUs without resizable BAR.
- HOST_COHERENT is dropped from the required set — the combined type may not be
coherent. The existing FlushDevice(cmd, ...) already establishes host->build
visibility and gates its flush-skip on the *chosen* type's flags (#60), so a
non-coherent BAR type still flushes correctly. The barrier is never removed.
metadataBuffer gets the same upgrade separately (#75).
Extends the TLASHighWaterMark hardware test to assert the chosen memory type
matches the HOST_VISIBLE | prefer-DEVICE_LOCAL request and lands on a
DEVICE_LOCAL type where one is reachable; the existing zero-validation-errors
check covers the dropped HOST_COHERENT path.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-17 17:37:39 +00:00
|
|
|
//
|
|
|
|
|
// instanceBuffer prefers HOST_VISIBLE | DEVICE_LOCAL (BAR/VRAM) so the
|
|
|
|
|
// two per-frame GPU accesses stay on-chip instead of crossing PCIe: the
|
|
|
|
|
// compute shader writes the transform field in place, and the AS build
|
|
|
|
|
// reads the whole instance. The CPU still writes the host-authored
|
|
|
|
|
// fields below — those are write-only, sequential, never-read-back
|
|
|
|
|
// writes, the ideal write-combined BAR workload. DEVICE_LOCAL is a
|
|
|
|
|
// best-effort preference: GetMemoryType falls back to plain
|
|
|
|
|
// HOST_VISIBLE (the previous behaviour) when no combined type exists
|
|
|
|
|
// (no resizable BAR). HOST_COHERENT is intentionally dropped from the
|
|
|
|
|
// required set — the combined type may not be coherent, and the
|
|
|
|
|
// FlushDevice(cmd, ...) below already establishes host->build
|
|
|
|
|
// visibility, gating its flush-skip on the *chosen* type's flags. The
|
|
|
|
|
// single shared buffer is preserved: a whole-buffer copy would clobber
|
|
|
|
|
// the GPU-written transforms, so staging is deliberately avoided.
|
perf(rt): allocate TLAS metadata buffer in BAR/VRAM, not system RAM (#75)
The TLAS metadata buffer is CPU-written every frame (one userMetadata word
per element in BuildTLAS's copy loop) and read by the ray shaders as a
STORAGE_BUFFER via device address every frame. Allocated
HOST_VISIBLE | HOST_COHERENT (system RAM), every per-frame shader read
traversed PCIe — the same cost the sibling instance buffer paid before #65.
Upgrade the request to HOST_VISIBLE with a best-effort DEVICE_LOCAL preference
(BAR/VRAM), keeping the single persistently-mapped buffer. The CPU's writes are
write-only, sequential, never-read-back — the ideal write-combined BAR load —
and the shader reads now hit local VRAM.
Unlike the instance buffer there is no GPU co-write, so the direct upgrade is
unconditional. Staging is still deliberately avoided: this buffer is rewritten
by the CPU every frame, so a per-frame stage+copy+barrier would defeat the
purpose (#75 caveat 2).
Correctness:
- DEVICE_LOCAL is a preference only (depends on #59): GetMemoryType falls back
to plain HOST_VISIBLE (the previous behaviour) on GPUs without resizable BAR.
- HOST_COHERENT is dropped from the required set — the combined type may not be
coherent. A FlushDevice() is added after the copy loop to make the host writes
available on a non-coherent type; it self-gates to a no-op on a coherent type
(#60), so the previous behaviour is preserved wherever the type ends up
coherent. The metadata buffer is not an AS build input and has no GPU
co-write, so it needs neither the build-stage barrier the instance buffer
takes nor a HOST->shader command barrier: the queue submit already orders host
writes made available before it against every command in the submission,
exactly as it did when this buffer was HOST_COHERENT.
Extends the TLASHighWaterMark hardware test with a metadata-buffer assertion
block mirroring #65's instance-buffer check: the chosen memory type must equal
what GetMemoryType resolves for the HOST_VISIBLE | prefer-DEVICE_LOCAL request,
and must be DEVICE_LOCAL wherever a host-visible device-local type is reachable.
The zero-validation-errors check covers the dropped HOST_COHERENT path.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-17 17:50:01 +00:00
|
|
|
//
|
|
|
|
|
// metadataBuffer (#75) gets the same HOST_VISIBLE | prefer-DEVICE_LOCAL
|
|
|
|
|
// upgrade: it is CPU-written every frame (the copy loop below) and read
|
|
|
|
|
// by the ray shaders as a STORAGE_BUFFER every frame, so it has the same
|
|
|
|
|
// per-frame-shader-read-over-PCIe cost. Unlike instanceBuffer there is
|
|
|
|
|
// no GPU co-write, so the direct upgrade is unconditional — but staging
|
|
|
|
|
// is still wrong here: a per-frame-rewritten buffer would pay a stage+
|
|
|
|
|
// copy+barrier every frame, defeating the point. HOST_COHERENT is
|
|
|
|
|
// likewise dropped; the FlushDevice() after the copy loop makes the host
|
|
|
|
|
// writes available before the consuming submit on a non-coherent type
|
|
|
|
|
// (no-op when the chosen type is coherent — the previous behaviour),
|
|
|
|
|
// and the queue submit carries the host-write -> shader-read ordering as
|
|
|
|
|
// it always has.
|
2026-06-16 18:24:09 +00:00
|
|
|
if (tlas.instanceBuffer.buffer == VK_NULL_HANDLE
|
|
|
|
|
|| primitiveCount > tlas.instanceBuffer.size / sizeof(VkAccelerationStructureInstanceKHR)) {
|
perf(rt): allocate TLAS instance buffer in BAR/VRAM, not system RAM (#65)
The TLAS instance buffer is rebuilt/refit every frame and is co-written by
both the CPU (host-authored fields) and the GPU (the compute shader writes
the transform field in place for transformOwnedByGpu elements). Allocated
HOST_VISIBLE | HOST_COHERENT (system RAM), both GPU accesses crossed PCIe:
the compute write of the transform, and the AS build's read of the instances.
Upgrade the request to HOST_VISIBLE with a best-effort DEVICE_LOCAL preference
(BAR/VRAM), keeping the single shared, persistently-mapped buffer — no staging,
no copy, no extra barrier. A whole-buffer staging copy was rejected: it would
clobber the GPU-written transforms in this in-place co-write design. Now the
compute write and AS build stay in local VRAM; the CPU's write-only,
sequential, never-read-back field writes are the ideal write-combined BAR load.
Correctness:
- DEVICE_LOCAL is a preference only (depends on #59): GetMemoryType falls back
to plain HOST_VISIBLE (the previous behaviour) on GPUs without resizable BAR.
- HOST_COHERENT is dropped from the required set — the combined type may not be
coherent. The existing FlushDevice(cmd, ...) already establishes host->build
visibility and gates its flush-skip on the *chosen* type's flags (#60), so a
non-coherent BAR type still flushes correctly. The barrier is never removed.
metadataBuffer gets the same upgrade separately (#75).
Extends the TLASHighWaterMark hardware test to assert the chosen memory type
matches the HOST_VISIBLE | prefer-DEVICE_LOCAL request and lands on a
DEVICE_LOCAL type where one is reachable; the existing zero-validation-errors
check covers the dropped HOST_COHERENT path.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-17 17:37:39 +00:00
|
|
|
tlas.instanceBuffer.Resize(VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_ACCELERATION_STRUCTURE_BUILD_INPUT_READ_ONLY_BIT_KHR | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT, primitiveCount, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
perf(rt): allocate TLAS metadata buffer in BAR/VRAM, not system RAM (#75)
The TLAS metadata buffer is CPU-written every frame (one userMetadata word
per element in BuildTLAS's copy loop) and read by the ray shaders as a
STORAGE_BUFFER via device address every frame. Allocated
HOST_VISIBLE | HOST_COHERENT (system RAM), every per-frame shader read
traversed PCIe — the same cost the sibling instance buffer paid before #65.
Upgrade the request to HOST_VISIBLE with a best-effort DEVICE_LOCAL preference
(BAR/VRAM), keeping the single persistently-mapped buffer. The CPU's writes are
write-only, sequential, never-read-back — the ideal write-combined BAR load —
and the shader reads now hit local VRAM.
Unlike the instance buffer there is no GPU co-write, so the direct upgrade is
unconditional. Staging is still deliberately avoided: this buffer is rewritten
by the CPU every frame, so a per-frame stage+copy+barrier would defeat the
purpose (#75 caveat 2).
Correctness:
- DEVICE_LOCAL is a preference only (depends on #59): GetMemoryType falls back
to plain HOST_VISIBLE (the previous behaviour) on GPUs without resizable BAR.
- HOST_COHERENT is dropped from the required set — the combined type may not be
coherent. A FlushDevice() is added after the copy loop to make the host writes
available on a non-coherent type; it self-gates to a no-op on a coherent type
(#60), so the previous behaviour is preserved wherever the type ends up
coherent. The metadata buffer is not an AS build input and has no GPU
co-write, so it needs neither the build-stage barrier the instance buffer
takes nor a HOST->shader command barrier: the queue submit already orders host
writes made available before it against every command in the submission,
exactly as it did when this buffer was HOST_COHERENT.
Extends the TLASHighWaterMark hardware test with a metadata-buffer assertion
block mirroring #65's instance-buffer check: the chosen memory type must equal
what GetMemoryType resolves for the HOST_VISIBLE | prefer-DEVICE_LOCAL request,
and must be DEVICE_LOCAL wherever a host-visible device-local type is reachable.
The zero-validation-errors check covers the dropped HOST_COHERENT path.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-17 17:50:01 +00:00
|
|
|
tlas.metadataBuffer.Resize(VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT, primitiveCount, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
2026-06-16 18:24:09 +00:00
|
|
|
}
|
2026-01-28 19:16:28 +01:00
|
|
|
}
|
2026-01-29 19:18:47 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
for(std::uint32_t i = 0; i < primitiveCount; i++) {
|
|
|
|
|
if (elements[i]->transformOwnedByGpu) {
|
|
|
|
|
// Skip the transform field — the application's compute shader
|
|
|
|
|
// writes it earlier in this submission. Copy everything else.
|
|
|
|
|
auto& dst = tlas.instanceBuffer.value[i];
|
|
|
|
|
const auto& src = elements[i]->instance;
|
|
|
|
|
dst.instanceCustomIndex = src.instanceCustomIndex;
|
|
|
|
|
dst.mask = src.mask;
|
|
|
|
|
dst.instanceShaderBindingTableRecordOffset = src.instanceShaderBindingTableRecordOffset;
|
|
|
|
|
dst.flags = src.flags;
|
|
|
|
|
dst.accelerationStructureReference = src.accelerationStructureReference;
|
|
|
|
|
} else {
|
|
|
|
|
tlas.instanceBuffer.value[i] = elements[i]->instance;
|
|
|
|
|
}
|
|
|
|
|
tlas.metadataBuffer.value[i] = elements[i]->userMetadata;
|
|
|
|
|
}
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
tlas.instanceBuffer.FlushDevice(cmd, VK_ACCESS_MEMORY_READ_BIT, VK_PIPELINE_STAGE_ACCELERATION_STRUCTURE_BUILD_BIT_KHR);
|
|
|
|
|
|
perf(rt): allocate TLAS metadata buffer in BAR/VRAM, not system RAM (#75)
The TLAS metadata buffer is CPU-written every frame (one userMetadata word
per element in BuildTLAS's copy loop) and read by the ray shaders as a
STORAGE_BUFFER via device address every frame. Allocated
HOST_VISIBLE | HOST_COHERENT (system RAM), every per-frame shader read
traversed PCIe — the same cost the sibling instance buffer paid before #65.
Upgrade the request to HOST_VISIBLE with a best-effort DEVICE_LOCAL preference
(BAR/VRAM), keeping the single persistently-mapped buffer. The CPU's writes are
write-only, sequential, never-read-back — the ideal write-combined BAR load —
and the shader reads now hit local VRAM.
Unlike the instance buffer there is no GPU co-write, so the direct upgrade is
unconditional. Staging is still deliberately avoided: this buffer is rewritten
by the CPU every frame, so a per-frame stage+copy+barrier would defeat the
purpose (#75 caveat 2).
Correctness:
- DEVICE_LOCAL is a preference only (depends on #59): GetMemoryType falls back
to plain HOST_VISIBLE (the previous behaviour) on GPUs without resizable BAR.
- HOST_COHERENT is dropped from the required set — the combined type may not be
coherent. A FlushDevice() is added after the copy loop to make the host writes
available on a non-coherent type; it self-gates to a no-op on a coherent type
(#60), so the previous behaviour is preserved wherever the type ends up
coherent. The metadata buffer is not an AS build input and has no GPU
co-write, so it needs neither the build-stage barrier the instance buffer
takes nor a HOST->shader command barrier: the queue submit already orders host
writes made available before it against every command in the submission,
exactly as it did when this buffer was HOST_COHERENT.
Extends the TLASHighWaterMark hardware test with a metadata-buffer assertion
block mirroring #65's instance-buffer check: the chosen memory type must equal
what GetMemoryType resolves for the HOST_VISIBLE | prefer-DEVICE_LOCAL request,
and must be DEVICE_LOCAL wherever a host-visible device-local type is reachable.
The zero-validation-errors check covers the dropped HOST_COHERENT path.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-17 17:50:01 +00:00
|
|
|
// Make the per-frame metadata host writes available before the consuming
|
|
|
|
|
// submit. metadataBuffer is not an AS build input (the build never reads it)
|
|
|
|
|
// and has no GPU co-write — only the ray shaders read it, in a later pass —
|
|
|
|
|
// so it needs neither the build-stage barrier instanceBuffer takes above nor
|
|
|
|
|
// a HOST->shader command barrier: the queue submit already orders host
|
|
|
|
|
// writes made available before it against every command in the submission,
|
|
|
|
|
// exactly as it did when this buffer was HOST_COHERENT. This plain host
|
|
|
|
|
// flush is what now makes those writes available on a non-coherent
|
|
|
|
|
// (BAR/VRAM) type; it self-gates to a no-op on a coherent type (#60),
|
|
|
|
|
// preserving the previous behaviour where the type ends up coherent.
|
|
|
|
|
tlas.metadataBuffer.FlushDevice();
|
|
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
VkAccelerationStructureGeometryInstancesDataKHR instancesData {
|
2026-01-28 19:16:28 +01:00
|
|
|
.sType = VK_STRUCTURE_TYPE_ACCELERATION_STRUCTURE_GEOMETRY_INSTANCES_DATA_KHR,
|
|
|
|
|
.arrayOfPointers = VK_FALSE,
|
2026-05-05 23:49:29 +02:00
|
|
|
.data = {tlas.instanceBuffer.address}
|
2026-01-28 19:16:28 +01:00
|
|
|
};
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-01-28 19:16:28 +01:00
|
|
|
VkAccelerationStructureGeometryDataKHR geometryData;
|
|
|
|
|
geometryData.instances = instancesData;
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-01-28 19:16:28 +01:00
|
|
|
VkAccelerationStructureGeometryKHR tlasGeometry {
|
|
|
|
|
.sType = VK_STRUCTURE_TYPE_ACCELERATION_STRUCTURE_GEOMETRY_KHR,
|
|
|
|
|
.geometryType = VK_GEOMETRY_TYPE_INSTANCES_KHR,
|
|
|
|
|
.geometry = geometryData
|
|
|
|
|
};
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
VkAccelerationStructureBuildGeometryInfoKHR tlasBuildGeometryInfo {
|
|
|
|
|
.sType = VK_STRUCTURE_TYPE_ACCELERATION_STRUCTURE_BUILD_GEOMETRY_INFO_KHR,
|
2026-01-28 19:16:28 +01:00
|
|
|
.type = VK_ACCELERATION_STRUCTURE_TYPE_TOP_LEVEL_KHR,
|
2026-06-16 15:49:17 +02:00
|
|
|
// ALLOW_UPDATE (required for any subsequent UPDATE-mode refit) plus
|
|
|
|
|
// the caller's FAST_TRACE/FAST_BUILD preference — see buildFlags above.
|
|
|
|
|
.flags = buildFlags,
|
2026-05-05 23:49:29 +02:00
|
|
|
.mode = topologyChanged
|
|
|
|
|
? VK_BUILD_ACCELERATION_STRUCTURE_MODE_BUILD_KHR
|
|
|
|
|
: VK_BUILD_ACCELERATION_STRUCTURE_MODE_UPDATE_KHR,
|
2026-01-28 19:16:28 +01:00
|
|
|
.geometryCount = 1,
|
|
|
|
|
.pGeometries = &tlasGeometry
|
|
|
|
|
};
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
if (topologyChanged) {
|
|
|
|
|
// Query sizes for the fresh build, allocate AS storage + scratch.
|
|
|
|
|
VkAccelerationStructureBuildSizesInfoKHR tlasBuildSizes {
|
|
|
|
|
.sType = VK_STRUCTURE_TYPE_ACCELERATION_STRUCTURE_BUILD_SIZES_INFO_KHR
|
|
|
|
|
};
|
|
|
|
|
Device::vkGetAccelerationStructureBuildSizesKHR(
|
|
|
|
|
Device::device,
|
|
|
|
|
VK_ACCELERATION_STRUCTURE_BUILD_TYPE_DEVICE_KHR,
|
|
|
|
|
&tlasBuildGeometryInfo,
|
|
|
|
|
&primitiveCount,
|
|
|
|
|
&tlasBuildSizes
|
|
|
|
|
);
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
// Scratch buffer must hold at least max(buildScratchSize, updateScratchSize).
|
|
|
|
|
// Sizing for buildScratchSize covers both — refit is always smaller.
|
|
|
|
|
tlas.scratchBuffer.Resize(VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, tlasBuildSizes.buildScratchSize);
|
|
|
|
|
tlas.buffer.Resize(VK_BUFFER_USAGE_ACCELERATION_STRUCTURE_STORAGE_BIT_KHR | VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT | VK_BUFFER_USAGE_ACCELERATION_STRUCTURE_BUILD_INPUT_READ_ONLY_BIT_KHR, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, tlasBuildSizes.accelerationStructureSize);
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
// Destroy the previous AS handle before creating a new one — the
|
|
|
|
|
// pre-refit path leaked here on every frame.
|
|
|
|
|
if (tlas.accelerationStructure != VK_NULL_HANDLE) {
|
|
|
|
|
Device::vkDestroyAccelerationStructureKHR(Device::device, tlas.accelerationStructure, nullptr);
|
|
|
|
|
tlas.accelerationStructure = VK_NULL_HANDLE;
|
|
|
|
|
}
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
VkAccelerationStructureCreateInfoKHR tlasCreateInfo {
|
|
|
|
|
.sType = VK_STRUCTURE_TYPE_ACCELERATION_STRUCTURE_CREATE_INFO_KHR,
|
|
|
|
|
.buffer = tlas.buffer.buffer,
|
|
|
|
|
.offset = 0,
|
|
|
|
|
.size = tlasBuildSizes.accelerationStructureSize,
|
|
|
|
|
.type = VK_ACCELERATION_STRUCTURE_TYPE_TOP_LEVEL_KHR,
|
|
|
|
|
};
|
|
|
|
|
Device::CheckVkResult(Device::vkCreateAccelerationStructureKHR(Device::device, &tlasCreateInfo, nullptr, &tlas.accelerationStructure));
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
VkAccelerationStructureDeviceAddressInfoKHR addrInfo {
|
|
|
|
|
.sType = VK_STRUCTURE_TYPE_ACCELERATION_STRUCTURE_DEVICE_ADDRESS_INFO_KHR,
|
|
|
|
|
.accelerationStructure = tlas.accelerationStructure
|
|
|
|
|
};
|
|
|
|
|
tlas.address = Device::vkGetAccelerationStructureDeviceAddressKHR(Device::device, &addrInfo);
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-05-05 23:49:29 +02:00
|
|
|
tlas.builtInstanceCount = primitiveCount;
|
2026-06-16 15:49:17 +02:00
|
|
|
tlas.builtFlags = buildFlags;
|
2026-05-05 23:49:29 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// For UPDATE mode, src == dst (in-place refit). For BUILD, src is
|
|
|
|
|
// VK_NULL_HANDLE and dst is the freshly-created handle.
|
|
|
|
|
tlasBuildGeometryInfo.scratchData.deviceAddress = tlas.scratchBuffer.address;
|
|
|
|
|
tlasBuildGeometryInfo.dstAccelerationStructure = tlas.accelerationStructure;
|
|
|
|
|
tlasBuildGeometryInfo.srcAccelerationStructure =
|
|
|
|
|
topologyChanged ? VK_NULL_HANDLE : tlas.accelerationStructure;
|
2026-01-28 18:51:11 +01:00
|
|
|
|
2026-01-28 19:16:28 +01:00
|
|
|
VkAccelerationStructureBuildRangeInfoKHR tlasRangeInfo {
|
|
|
|
|
.primitiveCount = primitiveCount,
|
|
|
|
|
.primitiveOffset = 0,
|
|
|
|
|
.firstVertex = 0,
|
|
|
|
|
.transformOffset = 0
|
|
|
|
|
};
|
|
|
|
|
VkAccelerationStructureBuildRangeInfoKHR* tlasRangeInfoPP = &tlasRangeInfo;
|
2026-03-09 20:10:19 +01:00
|
|
|
Device::vkCmdBuildAccelerationStructuresKHR(cmd, 1, &tlasBuildGeometryInfo, &tlasRangeInfoPP);
|
2026-01-30 00:09:37 +01:00
|
|
|
|
|
|
|
|
vkCmdPipelineBarrier(
|
|
|
|
|
cmd,
|
|
|
|
|
VK_PIPELINE_STAGE_ACCELERATION_STRUCTURE_BUILD_BIT_KHR,
|
2026-05-05 23:49:29 +02:00
|
|
|
VK_PIPELINE_STAGE_RAY_TRACING_SHADER_BIT_KHR | VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
|
2026-01-30 00:09:37 +01:00
|
|
|
0,
|
|
|
|
|
0, nullptr,
|
|
|
|
|
0, nullptr,
|
|
|
|
|
0, nullptr
|
|
|
|
|
);
|
2026-05-01 23:35:37 +02:00
|
|
|
}
|