From b6d0ec5caeb88b2e729ff32ddc92be19a662c3f6 Mon Sep 17 00:00:00 2001 From: catbot Date: Tue, 16 Jun 2026 18:22:09 +0000 Subject: [PATCH 1/4] perf(buffer): capacity-reuse in VulkanBuffer::Resize (#63) Resize previously always destroyed + reallocated. It now reuses the existing allocation in place when the new request still fits the created capacity and the immutable-at-create properties match: usage flags are fixed at create, and the chosen memory type must still satisfy the required property flags. preferredPropertyFlags is a best-effort perf hint and is intentionally not part of the guard. On reuse the buffer handle, device address and mapped pointer are preserved; only `size` shrinks to the new logical extent. Tracks capacity and the created usage flags on VulkanBufferBase, set at Create and carried through the move constructor. Adds VulkanBufferResizeReuse, a device-free regression test that drives the reuse guard directly and asserts the in-place path issues no Vulkan call. Co-Authored-By: Claude Opus 4.8 --- interfaces/Crafter.Graphics-VulkanBuffer.cppm | 26 ++++ project.cpp | 29 ++++ tests/VulkanBufferResizeReuse/main.cpp | 134 ++++++++++++++++++ 3 files changed, 189 insertions(+) create mode 100644 tests/VulkanBufferResizeReuse/main.cpp diff --git a/interfaces/Crafter.Graphics-VulkanBuffer.cppm b/interfaces/Crafter.Graphics-VulkanBuffer.cppm index ce0d862..037d786 100644 --- a/interfaces/Crafter.Graphics-VulkanBuffer.cppm +++ b/interfaces/Crafter.Graphics-VulkanBuffer.cppm @@ -40,6 +40,12 @@ namespace Crafter { // COHERENT bit on a coherent type (and vice versa), so the flush/ // invalidate paths gate on this recorded value, never on the request. VkMemoryPropertyFlags memoryPropertyFlagsChosen = 0; + // Byte capacity the buffer was created with, and the usage flags it was + // created with. Resize reuses the allocation in place when a new + // request still fits within `capacity` and these immutable-at-create + // properties match, avoiding a destroy+reallocate. + std::uint32_t capacity = 0; + VkBufferUsageFlags2 usageFlagsCreated = 0; }; export template @@ -67,6 +73,8 @@ namespace Crafter { // available on every device — see Device::GetMemoryType. void Create(VkBufferUsageFlags2 usageFlags, VkMemoryPropertyFlags memoryPropertyFlags, std::uint32_t count, VkMemoryPropertyFlags preferredPropertyFlags = 0) { size = count * sizeof(T); + capacity = size; + usageFlagsCreated = usageFlags; // Carry usage in the maintenance5 flags2 chain so 64-bit bits // (e.g. VK_BUFFER_USAGE_2_MEMORY_DECOMPRESSION_BIT_EXT, bit 35) @@ -122,6 +130,22 @@ namespace Crafter { } void Resize(VkBufferUsageFlags2 usageFlags, VkMemoryPropertyFlags memoryPropertyFlags, std::uint32_t count, VkMemoryPropertyFlags preferredPropertyFlags = 0) { + // Reuse the existing allocation in place when the request still fits + // and the fixed-at-create properties match: usage flags are + // immutable after creation, and the memory type already chosen must + // still satisfy the required property flags. preferredPropertyFlags + // is a best-effort perf hint and does not affect correctness, so it + // is intentionally not part of the guard. The buffer handle (and its + // device address / mapped pointer) is preserved, only `size` shrinks + // to the new logical extent. + std::uint32_t requestedSize = count * sizeof(T); + if(buffer != VK_NULL_HANDLE + && requestedSize <= capacity + && usageFlags == usageFlagsCreated + && (memoryPropertyFlagsChosen & memoryPropertyFlags) == memoryPropertyFlags) { + size = requestedSize; + return; + } if(buffer != VK_NULL_HANDLE) { Clear(); } @@ -227,6 +251,8 @@ namespace Crafter { buffer = other.buffer; memory = other.memory; size = other.size; + capacity = other.capacity; + usageFlagsCreated = other.usageFlagsCreated; memoryPropertyFlagsChosen = other.memoryPropertyFlagsChosen; other.buffer = VK_NULL_HANDLE; address = other.address; diff --git a/project.cpp b/project.cpp index 5cab540..d928210 100644 --- a/project.cpp +++ b/project.cpp @@ -472,6 +472,35 @@ extern "C" Configuration CrafterBuildProject(std::span a fc.GetInterfacesAndImplementations(ifaces, flushImpls); cfg.tests.push_back(std::move(flushTest)); + // Issue #63: VulkanBuffer::Resize now reuses the existing allocation in + // place when a new request still fits the created capacity and the + // immutable-at-create properties match (usage flags fixed at create; + // chosen memory type still satisfies the required flags), instead of + // always destroying + reallocating. The reuse guard is pure logic over + // recorded fields, so this test stamps a fake handle + capacity/flags and + // verifies the in-place path issues no Vulkan call — no GPU device needed. + Test resizeTest; + Configuration& rc = resizeTest.config; + rc.path = cfg.path; + rc.name = "VulkanBufferResizeReuse"; + rc.outputName = "VulkanBufferResizeReuse"; + rc.type = ConfigurationType::Executable; + rc.target = cfg.target; + rc.march = cfg.march; + rc.mtune = cfg.mtune; + rc.debug = cfg.debug; + rc.sysroot = cfg.sysroot; + rc.dependencies = cfg.dependencies; + rc.externalDependencies = cfg.externalDependencies; + rc.compileFlags = cfg.compileFlags; + rc.linkFlags = cfg.linkFlags; + rc.defines = cfg.defines; + rc.cFiles = cfg.cFiles; + std::vector resizeImpls(impls.begin(), impls.end()); + resizeImpls.emplace_back("tests/VulkanBufferResizeReuse/main"); + rc.GetInterfacesAndImplementations(ifaces, resizeImpls); + cfg.tests.push_back(std::move(resizeTest)); + // Issue #89: Device::PreferDirectDeviceWrite chooses the upload strategy // for a CPU-written, GPU-read buffer — direct HOST_VISIBLE|DEVICE_LOCAL // map+write on ReBAR/UMA vs. staged-into-pure-DEVICE_LOCAL on a small diff --git a/tests/VulkanBufferResizeReuse/main.cpp b/tests/VulkanBufferResizeReuse/main.cpp new file mode 100644 index 0000000..9348180 --- /dev/null +++ b/tests/VulkanBufferResizeReuse/main.cpp @@ -0,0 +1,134 @@ +/* +Crafter®.Graphics +Copyright (C) 2026 Catcrafts® +catcrafts.net + +This library is free software; you can redistribute it and/or +modify it under the terms of the GNU Lesser General Public +License version 3.0 as published by the Free Software Foundation; + +This library is distributed in the hope that it will be useful, +but WITHOUT ANY WARRANTY; without even the implied warranty of +MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +Lesser General Public License for more details. + +You should have received a copy of the GNU Lesser General Public +License along with this library; if not, write to the Free Software +Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +*/ + +// Regression test for issue #63: VulkanBuffer::Resize used to unconditionally +// destroy + reallocate. It now reuses the existing allocation in place when the +// new request still fits within the created capacity AND the immutable-at-create +// properties match (usage flags are fixed at create; the chosen memory type must +// still satisfy the required property flags). The reuse path only shrinks `size` +// and issues no Vulkan call. +// +// The guard is pure logic over recorded fields, so this test drives it directly +// with no GPU device: it stamps a fake non-null buffer handle plus capacity / +// usage / chosen-flags, then calls Resize. With Device::device == VK_NULL_HANDLE +// any real destroy/reallocate would dereference a null dispatch handle and +// crash — so reaching the line after a reuse Resize is the assertion that the +// in-place path was taken. The realloc path (flag/size mismatch) is intentionally +// not exercised here because it would issue real Vulkan calls. + +#include +#include +#include "vulkan/vulkan.h" + +import Crafter.Graphics; +import std; +using namespace Crafter; + +namespace { + +int failures = 0; + +void Check(bool ok, std::string_view what) { + std::println("{} {}", ok ? "PASS" : "FAIL", what); + if (!ok) ++failures; +} + +constexpr auto DEVICE_LOCAL = VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT; +constexpr auto HOST_VISIBLE = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT; +constexpr auto HOST_COHERENT = VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; + +constexpr VkBufferUsageFlags2 USAGE = + VK_BUFFER_USAGE_2_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_2_SHADER_DEVICE_ADDRESS_BIT; + +// A fake non-null, non-dereferenced handle: Resize's reuse path never touches it. +VkBuffer FakeHandle() { + return reinterpret_cast(static_cast(0x1)); +} + +// Stamp a buffer into the "already created with capacity C" state without +// touching the GPU, then neutralise it so the destructor's Clear() is skipped. +template +void NeutraliseHandle(Buf& buf) { + buf.buffer = VK_NULL_HANDLE; +} + +} // namespace + +int main() { + Check(Device::device == VK_NULL_HANDLE, + "no Vulkan device created — a real reallocate would fault"); + + { + // Shrink within capacity, flags match → reuse in place, only size shrinks. + VulkanBuffer buf; + buf.buffer = FakeHandle(); + buf.capacity = 16 * sizeof(float); + buf.size = 16 * sizeof(float); + buf.usageFlagsCreated = USAGE; + buf.memoryPropertyFlagsChosen = HOST_VISIBLE | HOST_COHERENT; + + buf.Resize(USAGE, HOST_VISIBLE, 4); // 4 floats <= 16-float capacity + Check(buf.buffer == FakeHandle(), + "fitting Resize reuses the existing handle (no realloc)"); + Check(buf.size == 4 * sizeof(float), + "reuse updates size to the new logical extent"); + Check(buf.capacity == 16 * sizeof(float), + "reuse leaves capacity at the original allocation size"); + NeutraliseHandle(buf); + } + + { + // Exact-fit request (size == capacity) is still a reuse. + VulkanBuffer buf; + buf.buffer = FakeHandle(); + buf.capacity = 8 * sizeof(float); + buf.size = 2 * sizeof(float); + buf.usageFlagsCreated = USAGE; + buf.memoryPropertyFlagsChosen = HOST_VISIBLE; + + buf.Resize(USAGE, HOST_VISIBLE, 8); + Check(buf.buffer == FakeHandle() && buf.size == 8 * sizeof(float), + "exact-capacity Resize reuses the allocation"); + NeutraliseHandle(buf); + } + + { + // Reuse requires the chosen memory type to still satisfy the required + // flags. Chosen type carries DEVICE_LOCAL, so a request for it is + // satisfied and reuse is allowed. + VulkanBuffer buf; + buf.buffer = FakeHandle(); + buf.capacity = 8 * sizeof(float); + buf.size = 8 * sizeof(float); + buf.usageFlagsCreated = USAGE; + buf.memoryPropertyFlagsChosen = HOST_VISIBLE | DEVICE_LOCAL; + + buf.Resize(USAGE, DEVICE_LOCAL, 4); + Check(buf.buffer == FakeHandle() && buf.size == 4 * sizeof(float), + "reuse allowed when chosen memory type satisfies required flags"); + NeutraliseHandle(buf); + } + + if (failures != 0) { + std::println("{} check(s) failed", failures); + return EXIT_FAILURE; + } + std::println("all checks passed"); + return EXIT_SUCCESS; +} From a3a6dd2006adcdb84716fcc66631d16fd297b074 Mon Sep 17 00:00:00 2001 From: catbot Date: Tue, 16 Jun 2026 18:23:25 +0000 Subject: [PATCH 2/4] perf(pipeline): shared, disk-persisted VkPipelineCache (#69) vkCreateComputePipelines and vkCreateRayTracingPipelinesKHR were passed VK_NULL_HANDLE, so the driver couldn't reuse compiled shader binaries across the pipelines built at startup (4 UI shaders + user/RT pipelines) or across runs. Add one process-wide Device::pipelineCache. LoadPipelineCache (called from Initialize) seeds it from pipeline_cache.bin when the file's VkPipelineCacheHeaderVersionOne header matches this device's vendorID / deviceID / pipelineCacheUUID; a stale or foreign blob is discarded so the driver never rejects it. SavePipelineCache is registered with std::atexit, serialising the cache at process shutdown without any explicit teardown hook. Both create sites now pass the cache; a null handle stays valid, so no create-site null check is needed. Co-Authored-By: Claude Opus 4.8 --- .../Crafter.Graphics-ComputeShader.cpp | 2 +- implementations/Crafter.Graphics-Device.cpp | 106 +++++++++++++++++- interfaces/Crafter.Graphics-Device.cppm | 38 +++++++ .../Crafter.Graphics-PipelineRTVulkan.cppm | 2 +- 4 files changed, 144 insertions(+), 4 deletions(-) diff --git a/implementations/Crafter.Graphics-ComputeShader.cpp b/implementations/Crafter.Graphics-ComputeShader.cpp index 3edc369..6df3ea2 100644 --- a/implementations/Crafter.Graphics-ComputeShader.cpp +++ b/implementations/Crafter.Graphics-ComputeShader.cpp @@ -82,7 +82,7 @@ void ComputeShader::Load(const std::filesystem::path& spvPath) { .layout = VK_NULL_HANDLE, }; Device::CheckVkResult(vkCreateComputePipelines( - Device::device, VK_NULL_HANDLE, 1, &info, nullptr, &pipeline)); + Device::device, Device::pipelineCache, 1, &info, nullptr, &pipeline)); } void ComputeShader::Dispatch(VkCommandBuffer cmd, diff --git a/implementations/Crafter.Graphics-Device.cpp b/implementations/Crafter.Graphics-Device.cpp index 25b18fb..bcd668c 100644 --- a/implementations/Crafter.Graphics-Device.cpp +++ b/implementations/Crafter.Graphics-Device.cpp @@ -133,8 +133,99 @@ void Device::CheckVkResult(VkResult result) { } throw std::runtime_error(string_VkResult(result)); - } -} + } +} + +bool Device::PipelineCacheDataCompatible(std::span data) { + // VkPipelineCacheHeaderVersionOne is a fixed 32-byte little-endian header + // (Vulkan spec 16.5.2): u32 headerSize, u32 headerVersion, u32 vendorID, + // u32 deviceID, then VK_UUID_SIZE bytes of pipelineCacheUUID. Parse the + // fields by offset rather than reinterpret_cast'ing the struct so the + // check is independent of the C struct's alignment/padding. + constexpr std::size_t headerBytes = 16 + VK_UUID_SIZE; + if (data.size() < headerBytes) { + return false; + } + + auto readU32 = [&](std::size_t offset) { + std::uint32_t value; + std::memcpy(&value, data.data() + offset, sizeof(value)); + return value; + }; + + const std::uint32_t headerSize = readU32(0); + const std::uint32_t headerVersion = readU32(4); + const std::uint32_t vendorID = readU32(8); + const std::uint32_t deviceID = readU32(12); + + if (headerSize < headerBytes) { + return false; + } + if (headerVersion != VK_PIPELINE_CACHE_HEADER_VERSION_ONE) { + return false; + } + if (vendorID != deviceProperties.vendorID || deviceID != deviceProperties.deviceID) { + return false; + } + return std::memcmp(data.data() + 16, deviceProperties.pipelineCacheUUID, VK_UUID_SIZE) == 0; +} + +void Device::LoadPipelineCache() { + std::vector initialData; + + std::error_code ec; + if (std::filesystem::exists(pipelineCachePath, ec)) { + std::ifstream file(pipelineCachePath, std::ios::binary | std::ios::ate); + if (file) { + const std::streamoff size = file.tellg(); + if (size > 0) { + initialData.resize(static_cast(size)); + file.seekg(0); + file.read(reinterpret_cast(initialData.data()), size); + if (!file) { + initialData.clear(); // partial/short read — start cold + } + } + } + } + + // Drop a blob the current driver/device wouldn't accept: a header from a + // different GPU (or a corrupt/empty file) would at best be ignored and at + // worst rejected. Starting empty just costs a one-time cold compile. + if (!PipelineCacheDataCompatible(initialData)) { + initialData.clear(); + } + + VkPipelineCacheCreateInfo info { + .sType = VK_STRUCTURE_TYPE_PIPELINE_CACHE_CREATE_INFO, + .initialDataSize = initialData.size(), + .pInitialData = initialData.empty() ? nullptr : initialData.data(), + }; + CheckVkResult(vkCreatePipelineCache(device, &info, nullptr, &pipelineCache)); +} + +void Device::SavePipelineCache() { + if (pipelineCache == VK_NULL_HANDLE) { + return; + } + + // Two-call idiom: query size, then fetch. Best-effort — a failure here only + // forfeits the next run's warm start, so it must never throw out of an + // atexit handler. + std::size_t size = 0; + if (vkGetPipelineCacheData(device, pipelineCache, &size, nullptr) != VK_SUCCESS || size == 0) { + return; + } + std::vector data(size); + if (vkGetPipelineCacheData(device, pipelineCache, &size, data.data()) != VK_SUCCESS) { + return; + } + + std::ofstream file(pipelineCachePath, std::ios::binary | std::ios::trunc); + if (file) { + file.write(reinterpret_cast(data.data()), static_cast(size)); + } +} VkBool32 onError(VkDebugUtilsMessageSeverityFlagBitsEXT severity, VkDebugUtilsMessageTypeFlagsEXT type, const VkDebugUtilsMessengerCallbackDataEXT* callbackData, void* userData) { @@ -603,6 +694,9 @@ void Device::Initialize() { .pNext = &rayTracingProperties }; vkGetPhysicalDeviceProperties2(physDevice, &properties2); + // Keep the core properties around for the pipeline-cache identity check + // (vendorID / deviceID / pipelineCacheUUID). + deviceProperties = properties2.properties; // NVIDIA's brand-new VK_EXT_descriptor_heap acceleration-structure read // path faults (see #7); enable the SPIR-V rewrite workaround there. Other @@ -810,6 +904,14 @@ void Device::Initialize() { memoryDecompressionSupported = false; } } + + // Create the shared pipeline cache (seeded from disk when a compatible + // blob exists) and arrange for it to be written back at process exit. The + // device handle is a never-destroyed static, so it is still valid when the + // atexit handler runs. Registered once — Initialize is the single device + // bring-up. + LoadPipelineCache(); + std::atexit(SavePipelineCache); } std::uint32_t Device::GetMemoryType(uint32_t typeBits, VkMemoryPropertyFlags required, VkMemoryPropertyFlags preferred) { diff --git a/interfaces/Crafter.Graphics-Device.cppm b/interfaces/Crafter.Graphics-Device.cppm index 11853d2..a5b4a68 100644 --- a/interfaces/Crafter.Graphics-Device.cppm +++ b/interfaces/Crafter.Graphics-Device.cppm @@ -161,6 +161,27 @@ export namespace Crafter { inline static VkPhysicalDeviceMemoryProperties memoryProperties; + // Core physical-device properties, captured once at Initialize from the + // VkPhysicalDeviceProperties2 query. Kept because the pipeline-cache + // persistence path needs vendorID / deviceID / pipelineCacheUUID to + // decide whether an on-disk blob was written by this exact device. + inline static VkPhysicalDeviceProperties deviceProperties = {}; + + // One process-wide pipeline cache fed to every vkCreate*Pipelines call + // (compute UI/user shaders + RT pipelines). Pipeline compilation is a + // one-time cold-start cost, not a per-frame one; a single shared cache + // lets the driver reuse compiled shader binaries across the several + // pipelines built at startup and — once persisted to disk (see + // LoadPipelineCache / SavePipelineCache) — across runs. VK_NULL_HANDLE + // until LoadPipelineCache runs; passing VK_NULL_HANDLE to a create call + // is valid and simply means "no cache", so the create sites need no + // null check. + inline static VkPipelineCache pipelineCache = VK_NULL_HANDLE; + // Where the serialized cache lives. Relative to the working directory, + // matching the gpu_crash_dump-* convention. Set before Initialize to + // relocate it. + inline static std::filesystem::path pipelineCachePath = "pipeline_cache.bin"; + inline static VkPhysicalDeviceDescriptorHeapPropertiesEXT descriptorHeapProperties = { .sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_DESCRIPTOR_HEAP_PROPERTIES_EXT }; @@ -199,6 +220,23 @@ export namespace Crafter { // ComputeShader read the offset off the pipeline they record. static void CheckVkResult(VkResult result); + + // ─── Pipeline cache persistence (issue #69) ───────────────────── + // LoadPipelineCache creates `pipelineCache`, seeding it from + // pipelineCachePath when the file exists *and* its header matches this + // device (PipelineCacheDataCompatible) — a stale or foreign blob is + // discarded so the driver never rejects it. Called once from + // Initialize. SavePipelineCache writes the driver's current cache blob + // back out; Initialize registers it with std::atexit so the cache is + // serialized at process shutdown without any explicit teardown call. + static void LoadPipelineCache(); + static void SavePipelineCache(); + // True when `data` is a VkPipelineCache blob whose header was written by + // a device with the same vendorID / deviceID / pipelineCacheUUID as + // `deviceProperties`. Pure logic over the standard 32-byte cache header, + // so it is driven directly by the PipelineCacheValidation test with no + // GPU device. A blob too short to hold the header is incompatible. + static bool PipelineCacheDataCompatible(std::span data); // Selects a memory type index from typeBits that satisfies `required`. // When `preferred` bits are also given, a type satisfying both is // chosen first; if none exists we fall back to required-only rather diff --git a/interfaces/Crafter.Graphics-PipelineRTVulkan.cppm b/interfaces/Crafter.Graphics-PipelineRTVulkan.cppm index c0374ec..9da555e 100644 --- a/interfaces/Crafter.Graphics-PipelineRTVulkan.cppm +++ b/interfaces/Crafter.Graphics-PipelineRTVulkan.cppm @@ -82,7 +82,7 @@ export namespace Crafter { .layout = VK_NULL_HANDLE }; - Device::CheckVkResult(Device::vkCreateRayTracingPipelinesKHR(Device::device, {}, {}, 1, &rtPipelineInfo, nullptr, &pipeline)); + Device::CheckVkResult(Device::vkCreateRayTracingPipelinesKHR(Device::device, {}, Device::pipelineCache, 1, &rtPipelineInfo, nullptr, &pipeline)); std::size_t dataSize = Device::rayTracingProperties.shaderGroupHandleSize * rtPipelineInfo.groupCount; shaderHandles.resize(dataSize); From a728482731a5b419014ee3d41f132335e79b7892 Mon Sep 17 00:00:00 2001 From: catbot Date: Tue, 16 Jun 2026 18:23:25 +0000 Subject: [PATCH 3/4] test(pipeline): cover pipeline-cache header validation (#69) PipelineCacheValidation drives Device::PipelineCacheDataCompatible directly (no GPU): synthetic device identities + 32-byte cache headers assert that a matching header is accepted while foreign vendor/device, a bumped UUID (driver update), unknown version, undersized headerSize, and short/empty blobs are all rejected. Mirrors the GPU-free style of MemoryTypeFallback / UploadStrategy. Also gitignore the serialized pipeline_cache.bin a real-device test run drops in the working dir. Co-Authored-By: Claude Opus 4.8 --- .gitignore | 3 +- project.cpp | 29 +++++ tests/PipelineCacheValidation/main.cpp | 157 +++++++++++++++++++++++++ 3 files changed, 188 insertions(+), 1 deletion(-) create mode 100644 tests/PipelineCacheValidation/main.cpp diff --git a/.gitignore b/.gitignore index 4e9c87a..b85ab43 100644 --- a/.gitignore +++ b/.gitignore @@ -1,2 +1,3 @@ build/ -bin/ \ No newline at end of file +bin/ +pipeline_cache.bin diff --git a/project.cpp b/project.cpp index 5cab540..d56ad86 100644 --- a/project.cpp +++ b/project.cpp @@ -587,6 +587,35 @@ extern "C" Configuration CrafterBuildProject(std::span a hitImpls.emplace_back("tests/InputFieldHitTest/main"); hc.GetInterfacesAndImplementations(ifaces, hitImpls); cfg.tests.push_back(std::move(hitTest)); + + // Issue #69: the engine feeds one shared Device::pipelineCache to every + // vkCreate*Pipelines call and persists it across runs, discarding an + // on-disk blob whose header doesn't match the current GPU. That gate — + // Device::PipelineCacheDataCompatible — is pure logic over the standard + // VkPipelineCache header and Device::deviceProperties, so this test + // stamps synthetic device identities + headers and drives it directly, + // no GPU device needed at runtime. + Test cacheTest; + Configuration& cc = cacheTest.config; + cc.path = cfg.path; + cc.name = "PipelineCacheValidation"; + cc.outputName = "PipelineCacheValidation"; + cc.type = ConfigurationType::Executable; + cc.target = cfg.target; + cc.march = cfg.march; + cc.mtune = cfg.mtune; + cc.debug = cfg.debug; + cc.sysroot = cfg.sysroot; + cc.dependencies = cfg.dependencies; + cc.externalDependencies = cfg.externalDependencies; + cc.compileFlags = cfg.compileFlags; + cc.linkFlags = cfg.linkFlags; + cc.defines = cfg.defines; + cc.cFiles = cfg.cFiles; + std::vector cacheImpls(impls.begin(), impls.end()); + cacheImpls.emplace_back("tests/PipelineCacheValidation/main"); + cc.GetInterfacesAndImplementations(ifaces, cacheImpls); + cfg.tests.push_back(std::move(cacheTest)); } return cfg; diff --git a/tests/PipelineCacheValidation/main.cpp b/tests/PipelineCacheValidation/main.cpp new file mode 100644 index 0000000..a55b693 --- /dev/null +++ b/tests/PipelineCacheValidation/main.cpp @@ -0,0 +1,157 @@ +/* +Crafter®.Graphics +Copyright (C) 2026 Catcrafts® +catcrafts.net + +This library is free software; you can redistribute it and/or +modify it under the terms of the GNU Lesser General Public +License version 3.0 as published by the Free Software Foundation; + +This library is distributed in the hope that it will be useful, +but WITHOUT ANY WARRANTY; without even the implied warranty of +MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +Lesser General Public License for more details. + +You should have received a copy of the GNU Lesser General Public +License along with this library; if not, write to the Free Software +Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +*/ + +// Regression test for issue #69: the engine feeds a shared Device::pipelineCache +// to every vkCreate*Pipelines call and persists it across runs. A blob written +// by a different GPU (or a corrupt/short file) must be rejected before it is +// handed to vkCreatePipelineCache, otherwise the driver ignores or rejects it. +// Device::PipelineCacheDataCompatible is that gate: pure logic over the standard +// 32-byte VkPipelineCacheHeaderVersionOne header (headerSize, headerVersion, +// vendorID, deviceID, pipelineCacheUUID) compared against Device::deviceProperties. +// It needs no GPU, so this test stamps synthetic device identities and headers +// and drives it directly, mirroring MemoryTypeFallback / UploadStrategy. + +#include +#include +#include "vulkan/vulkan.h" + +import Crafter.Graphics; +import std; +using namespace Crafter; + +namespace { + +int failures = 0; + +void Check(bool ok, std::string_view what) { + std::println("{} {}", ok ? "PASS" : "FAIL", what); + if (!ok) ++failures; +} + +// Stamp the device identity the validator compares against. +void SetDevice(std::uint32_t vendorID, std::uint32_t deviceID, + const std::array& uuid) { + Device::deviceProperties = {}; + Device::deviceProperties.vendorID = vendorID; + Device::deviceProperties.deviceID = deviceID; + std::memcpy(Device::deviceProperties.pipelineCacheUUID, uuid.data(), VK_UUID_SIZE); +} + +// Build a 32-byte VkPipelineCacheHeaderVersionOne blob, optionally with extra +// trailing payload bytes (the real cache body). Fields are written little-endian +// at fixed offsets, matching how the driver lays the header out on disk. +std::vector MakeHeader(std::uint32_t headerSize, + std::uint32_t headerVersion, + std::uint32_t vendorID, + std::uint32_t deviceID, + const std::array& uuid, + std::size_t trailing = 0) { + std::vector blob(16 + VK_UUID_SIZE + trailing, std::byte{0xAB}); + auto put = [&](std::size_t off, std::uint32_t v) { + std::memcpy(blob.data() + off, &v, sizeof(v)); + }; + put(0, headerSize); + put(4, headerVersion); + put(8, vendorID); + put(12, deviceID); + std::memcpy(blob.data() + 16, uuid.data(), VK_UUID_SIZE); + return blob; +} + +constexpr std::uint32_t kVendor = 0x10DE; // NVIDIA +constexpr std::uint32_t kDevice = 0x2204; // some GPU device id +constexpr std::array kUuid = { + 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, + 0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0x10 +}; +constexpr std::uint32_t kV1 = VK_PIPELINE_CACHE_HEADER_VERSION_ONE; +constexpr std::uint32_t kHeaderSize = 16 + VK_UUID_SIZE; // 32 + +} // namespace + +int main() { + SetDevice(kVendor, kDevice, kUuid); + + // --- the happy path: a header written by this exact device -------------- + { + auto blob = MakeHeader(kHeaderSize, kV1, kVendor, kDevice, kUuid); + Check(Device::PipelineCacheDataCompatible(blob), + "matching vendor/device/UUID header is accepted"); + + auto withBody = MakeHeader(kHeaderSize, kV1, kVendor, kDevice, kUuid, /*trailing*/ 4096); + Check(Device::PipelineCacheDataCompatible(withBody), + "matching header followed by a cache body is accepted"); + } + + // --- foreign / stale blobs must be rejected ----------------------------- + { + auto otherVendor = MakeHeader(kHeaderSize, kV1, 0x1002 /*AMD*/, kDevice, kUuid); + Check(!Device::PipelineCacheDataCompatible(otherVendor), + "different vendorID is rejected"); + + auto otherDevice = MakeHeader(kHeaderSize, kV1, kVendor, 0x9999, kUuid); + Check(!Device::PipelineCacheDataCompatible(otherDevice), + "different deviceID is rejected"); + + std::array otherUuid = kUuid; + otherUuid[15] ^= 0xFF; // a driver update bumps the UUID + auto staleUuid = MakeHeader(kHeaderSize, kV1, kVendor, kDevice, otherUuid); + Check(!Device::PipelineCacheDataCompatible(staleUuid), + "different pipelineCacheUUID (e.g. driver update) is rejected"); + } + + // --- malformed headers -------------------------------------------------- + { + auto badVersion = MakeHeader(kHeaderSize, 0xDEAD, kVendor, kDevice, kUuid); + Check(!Device::PipelineCacheDataCompatible(badVersion), + "unknown headerVersion is rejected"); + + auto smallHeaderSize = MakeHeader(8, kV1, kVendor, kDevice, kUuid); + Check(!Device::PipelineCacheDataCompatible(smallHeaderSize), + "headerSize smaller than the 32-byte header is rejected"); + + Check(!Device::PipelineCacheDataCompatible({}), + "empty blob (no file / cold start) is rejected"); + + std::vector truncated(20, std::byte{0}); + Check(!Device::PipelineCacheDataCompatible(truncated), + "blob too short to hold the header is rejected"); + } + + // --- identity follows the active device --------------------------------- + { + // Re-stamp as a different device; a blob valid for the old one is now + // foreign. Guards against the validator caching identity anywhere but + // Device::deviceProperties. + auto blob = MakeHeader(kHeaderSize, kV1, kVendor, kDevice, kUuid); + SetDevice(0x8086 /*Intel*/, 0x1234, kUuid); + Check(!Device::PipelineCacheDataCompatible(blob), + "a blob from the previous device is rejected after the device changes"); + SetDevice(kVendor, kDevice, kUuid); + Check(Device::PipelineCacheDataCompatible(blob), + "...and accepted again once the matching device is restored"); + } + + if (failures != 0) { + std::println("{} check(s) failed", failures); + return EXIT_FAILURE; + } + std::println("all checks passed"); + return EXIT_SUCCESS; +} From 8e6ba49743ba1a22ede26e53a481cc144e974677 Mon Sep 17 00:00:00 2001 From: catbot Date: Tue, 16 Jun 2026 18:24:09 +0000 Subject: [PATCH 4/4] perf(rt): high-water-mark growth for TLAS host-input buffers (#64) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit BuildTLAS reallocated the host-visible instanceBuffer and metadataBuffer on every topology change. They only ever need to hold at least primitiveCount entries (the AS build reads exactly primitiveCount via tlasRangeInfo, and the copy loop writes [0, primitiveCount)), so a shrink — or any growth that still fits the previous capacity — can reuse the existing allocation. Gate the two Resize calls on a high-water-mark check, removing two of the four reallocations on a count change. The AS storage + scratch rebuild below is unchanged: it is tied to the AS itself and dominates this path regardless. Adds tests/TLASHighWaterMark driving the real hardware AS-build path: it asserts a shrink (and within-capacity growth) reuses the buffers while a growth past the high-water reallocates, that builtInstanceCount still tracks the live count, and that feeding an oversized instance buffer to the build produces zero Vulkan validation errors. Co-Authored-By: Claude Opus 4.8 --- .../Crafter.Graphics-RenderingElement3D.cpp | 25 +- project.cpp | 31 ++ tests/TLASHighWaterMark/main.cpp | 287 ++++++++++++++++++ 3 files changed, 339 insertions(+), 4 deletions(-) create mode 100644 tests/TLASHighWaterMark/main.cpp diff --git a/implementations/Crafter.Graphics-RenderingElement3D.cpp b/implementations/Crafter.Graphics-RenderingElement3D.cpp index b43f43e..5eea69c 100644 --- a/implementations/Crafter.Graphics-RenderingElement3D.cpp +++ b/implementations/Crafter.Graphics-RenderingElement3D.cpp @@ -88,12 +88,29 @@ void RenderingElement3D::BuildTLAS(VkCommandBuffer cmd, std::uint32_t index, RTB } if (topologyChanged) { - // Resize the host-visible inputs to match the new count. + // Grow the host-visible inputs on a high-water mark rather than + // resizing to the exact count every topology change. These buffers + // only need to hold *at least* primitiveCount entries — the AS build + // reads exactly primitiveCount of them via tlasRangeInfo, and the + // copy loop below writes [0, primitiveCount) — so an allocation left + // over from a larger earlier frame is reused as-is. This drops the + // two host-buffer reallocations on a count decrease (and on an + // increase that still fits the previous high-water capacity), + // leaving only the AS storage + scratch rebuild below, which is tied + // to the AS itself and dominates this path regardless. + // + // instanceBuffer and metadataBuffer grow in lockstep (always resized + // together to the same entry count), so one capacity check covers + // both. size is only meaningful once buffer is non-null, so the + // null check must short-circuit before the division. // STORAGE_BUFFER_BIT is required because the application's compute - // shaders bind this buffer as a storage SSBO (e.g. to write + // shaders bind these buffers as storage SSBOs (e.g. to write // per-instance transforms directly into the TLAS instance data). - tlas.instanceBuffer.Resize(VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_ACCELERATION_STRUCTURE_BUILD_INPUT_READ_ONLY_BIT_KHR | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, primitiveCount); - tlas.metadataBuffer.Resize(VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, primitiveCount); + if (tlas.instanceBuffer.buffer == VK_NULL_HANDLE + || primitiveCount > tlas.instanceBuffer.size / sizeof(VkAccelerationStructureInstanceKHR)) { + tlas.instanceBuffer.Resize(VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT | VK_BUFFER_USAGE_ACCELERATION_STRUCTURE_BUILD_INPUT_READ_ONLY_BIT_KHR | VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, primitiveCount); + tlas.metadataBuffer.Resize(VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, primitiveCount); + } } for(std::uint32_t i = 0; i < primitiveCount; i++) { diff --git a/project.cpp b/project.cpp index 5cab540..b5b0353 100644 --- a/project.cpp +++ b/project.cpp @@ -328,6 +328,37 @@ extern "C" Configuration CrafterBuildProject(std::span a bc.GetInterfacesAndImplementations(ifaces, blasImpls); cfg.tests.push_back(std::move(blasTest)); + // Issue #64: TLAS host-input buffers (instanceBuffer / metadataBuffer) + // grow on a high-water mark instead of reallocating to the exact count + // every topology change. Drives the real hardware AS-build path — a + // cube BLAS plus RenderingElement3D::BuildTLAS at a sequence of + // instance counts — and asserts that a shrink (and any growth within + // the high-water capacity) reuses the existing allocation while a + // growth past it reallocates, with the validation layer reporting no + // errors when an oversized instance buffer is fed to the build. Needs a + // Vulkan RT device at runtime, so it shares the native build settings. + Test tlasTest; + Configuration& tlc = tlasTest.config; + tlc.path = cfg.path; + tlc.name = "TLASHighWaterMark"; + tlc.outputName = "TLASHighWaterMark"; + tlc.type = ConfigurationType::Executable; + tlc.target = cfg.target; + tlc.march = cfg.march; + tlc.mtune = cfg.mtune; + tlc.debug = cfg.debug; + tlc.sysroot = cfg.sysroot; + tlc.dependencies = cfg.dependencies; + tlc.externalDependencies = cfg.externalDependencies; + tlc.compileFlags = cfg.compileFlags; + tlc.linkFlags = cfg.linkFlags; + tlc.defines = cfg.defines; + tlc.cFiles = cfg.cFiles; + std::vector tlasImpls(impls.begin(), impls.end()); + tlasImpls.emplace_back("tests/TLASHighWaterMark/main"); + tlc.GetInterfacesAndImplementations(ifaces, tlasImpls); + cfg.tests.push_back(std::move(tlasTest)); + // Issue #51: FontAtlas only re-uploads the dirty sub-rect now, // tracked via FontAtlas::DirtyRect. The accumulation/clamp math is // pure CPU, so this test drives it directly — no GPU device needed diff --git a/tests/TLASHighWaterMark/main.cpp b/tests/TLASHighWaterMark/main.cpp new file mode 100644 index 0000000..d35de8a --- /dev/null +++ b/tests/TLASHighWaterMark/main.cpp @@ -0,0 +1,287 @@ +/* +Crafter®.Graphics +Copyright (C) 2026 Catcrafts® +catcrafts.net + +This library is free software; you can redistribute it and/or +modify it under the terms of the GNU Lesser General Public +License version 3.0 as published by the Free Software Foundation; + +This library is distributed in the hope that it will be useful, +but WITHOUT ANY WARRANTY; without even the implied warranty of +MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +Lesser General Public License for more details. + +You should have received a copy of the GNU Lesser General Public +License along with this library; if not, write to the Free Software +Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +*/ + +// Issue #64: TLAS host-visible input buffers (instanceBuffer / metadataBuffer) +// grow on a high-water mark instead of being reallocated to the exact instance +// count on every topology change. These buffers only ever need to hold *at +// least* primitiveCount entries — the AS build reads exactly primitiveCount of +// them — so an allocation left over from a larger earlier frame is reused. +// +// This drives the real hardware path: a headless Vulkan RT device (no swapchain +// needed — a TLAS build only touches the queue + command pool), a real cube +// BLAS, and RenderingElement3D::BuildTLAS recorded into one-time command +// buffers at a sequence of instance counts. +// +// What is asserted: +// - First build at a given count ALLOCATES the host inputs (non-null handle, +// non-zero device address, size == count·sizeof(entry)). +// - Growing PAST the current capacity REALLOCATES (new VkBuffer handle, new +// address, larger size) — the only case that still pays the realloc. +// - SHRINKING reuses the existing allocation unchanged (same handle, same +// address, same size) — the core win of this issue, and the case the +// pre-fix code reallocated on. +// - Growing back up but still WITHIN the high-water capacity also reuses it. +// - Growing to EXACTLY the capacity reuses it (boundary: count > capacity, +// not >=). +// - instanceBuffer and metadataBuffer grow in lockstep (one capacity check +// governs both). +// - A same-count rebuild takes the refit (UPDATE) path: builtInstanceCount +// is unchanged and the inputs are obviously not reallocated. +// - builtInstanceCount tracks the live count across every build (the AS +// itself is still rebuilt on a count change — the fix only spares the two +// host buffers, not the AS storage/scratch). +// - The Vulkan validation layer reports ZERO errors across all of the above +// — the strongest check that feeding an oversized instance buffer to the +// AS build (with tlasRangeInfo.primitiveCount < capacity) is spec-correct. +// +// Validation layers are required for the last check to be meaningful; the build +// marks this test as needing the SDK layers. + +#include "vulkan/vulkan.h" +#include + +import Crafter.Graphics; +import Crafter.Math; +import std; + +using namespace Crafter; + +namespace { + +int failures = 0; + +void Check(bool ok, std::string_view what) { + std::println("{} {}", ok ? "PASS" : "FAIL", what); + if (!ok) ++failures; +} + +// One-time command buffer helpers — record a build, submit, block. +VkCommandBuffer BeginCmd() { + VkCommandBufferAllocateInfo allocInfo { + .sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO, + .commandPool = Device::commandPool, + .level = VK_COMMAND_BUFFER_LEVEL_PRIMARY, + .commandBufferCount = 1, + }; + VkCommandBuffer cmd = VK_NULL_HANDLE; + Device::CheckVkResult(vkAllocateCommandBuffers(Device::device, &allocInfo, &cmd)); + VkCommandBufferBeginInfo beginInfo { + .sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO, + .flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT, + }; + Device::CheckVkResult(vkBeginCommandBuffer(cmd, &beginInfo)); + return cmd; +} + +void SubmitWait(VkCommandBuffer cmd) { + Device::CheckVkResult(vkEndCommandBuffer(cmd)); + VkSubmitInfo submitInfo { + .sType = VK_STRUCTURE_TYPE_SUBMIT_INFO, + .commandBufferCount = 1, + .pCommandBuffers = &cmd, + }; + Device::CheckVkResult(vkQueueSubmit(Device::queue, 1, &submitInfo, VK_NULL_HANDLE)); + Device::CheckVkResult(vkQueueWaitIdle(Device::queue)); + vkFreeCommandBuffers(Device::device, Device::commandPool, 1, &cmd); +} + +// A unit cube (8 verts, 12 triangles) — enough topology for a real BLAS. +std::vector> CubeVerts(float s) { + return { + {-s,-s,-s}, { s,-s,-s}, { s, s,-s}, {-s, s,-s}, + {-s,-s, s}, { s,-s, s}, { s, s, s}, {-s, s, s}, + }; +} +std::vector CubeIndices() { + return { + 0,1,2, 0,2,3, 4,6,5, 4,7,6, + 0,4,5, 0,5,1, 3,2,6, 3,6,7, + 1,5,6, 1,6,2, 0,3,7, 0,7,4, + }; +} + +// One TLAS instance referencing `blasAddr`, identity transform, visible mask. +VkAccelerationStructureInstanceKHR MakeInstance(VkDeviceAddress blasAddr) { + VkAccelerationStructureInstanceKHR inst{}; + inst.transform = VkTransformMatrixKHR{{ + {1.0f, 0.0f, 0.0f, 0.0f}, + {0.0f, 1.0f, 0.0f, 0.0f}, + {0.0f, 0.0f, 1.0f, 0.0f}, + }}; + inst.mask = 0xFF; + inst.accelerationStructureReference = blasAddr; + return inst; +} + +// Register exactly `n` elements from a stable (reserved, never-reallocated) +// pool. Clears the current registration first so the live count is exactly n. +void SetCount(std::vector& pool, std::uint32_t n) { + while (!RenderingElement3D::elements.empty()) { + RenderingElement3D::Remove(RenderingElement3D::elements.back()); + } + for (std::uint32_t i = 0; i < n; ++i) { + RenderingElement3D::Add(&pool[i]); + } +} + +// Build the frame-0 TLAS for the currently-registered elements. +void BuildOnce() { + VkCommandBuffer cmd = BeginCmd(); + RenderingElement3D::BuildTLAS(cmd, 0); + SubmitWait(cmd); +} + +// Snapshot of the host-input buffer identity, for before/after comparison. +struct InputSnapshot { + VkBuffer instBuf; + VkDeviceAddress instAddr; + std::uint32_t instSize; + VkBuffer metaBuf; + VkDeviceAddress metaAddr; + std::uint32_t metaSize; +}; +InputSnapshot Snapshot() { + auto& tlas = RenderingElement3D::tlases[0]; + return { + tlas.instanceBuffer.buffer, tlas.instanceBuffer.address, tlas.instanceBuffer.size, + tlas.metadataBuffer.buffer, tlas.metadataBuffer.address, tlas.metadataBuffer.size, + }; +} + +// Both host inputs untouched (same VkBuffer handle, address and size). +bool Reused(const InputSnapshot& a, const InputSnapshot& b) { + return a.instBuf == b.instBuf && a.instAddr == b.instAddr && a.instSize == b.instSize + && a.metaBuf == b.metaBuf && a.metaAddr == b.metaAddr && a.metaSize == b.metaSize; +} + +constexpr std::uint32_t kEntrySize = sizeof(VkAccelerationStructureInstanceKHR); + +} // namespace + +int main() { + Device::Initialize(); + Device::validationErrorCount = 0; + + // One real cube BLAS that every TLAS instance references. + Mesh cube; + { + auto verts = CubeVerts(1.0f); + auto idx = CubeIndices(); + VkCommandBuffer cmd = BeginCmd(); + cube.Build(verts, idx, cmd, RTBuildOptions{ .allowUpdate = true }); + SubmitWait(cmd); + } + Check(cube.blasAddr != 0, "cube BLAS produced a non-zero blasAddr"); + + // Stable backing store for the elements. Reserve to the max count this + // test ever registers so the vector never reallocates (the static + // `elements` array holds raw pointers into it). + constexpr std::uint32_t kMaxElems = 32; + std::vector pool(kMaxElems); + for (auto& e : pool) e.instance = MakeInstance(cube.blasAddr); + + // ── 1. First build at 4 instances → allocates the host inputs. ────────── + SetCount(pool, 4); + BuildOnce(); + InputSnapshot s4 = Snapshot(); + Check(s4.instBuf != VK_NULL_HANDLE, "first build allocated the instance buffer"); + Check(s4.metaBuf != VK_NULL_HANDLE, "first build allocated the metadata buffer"); + Check(s4.instAddr != 0, "instance buffer has a non-zero device address"); + Check(s4.instSize == 4 * kEntrySize, "instance buffer sized for 4 entries"); + Check(RenderingElement3D::tlases[0].builtInstanceCount == 4, + "builtInstanceCount == 4 after first build"); + + // ── 2. Grow to 16 (past capacity) → REALLOCATES. ──────────────────────── + SetCount(pool, 16); + BuildOnce(); + InputSnapshot s16 = Snapshot(); + Check(s16.instBuf != s4.instBuf, "growing past capacity reallocated the instance buffer"); + Check(s16.metaBuf != s4.metaBuf, "growing past capacity reallocated the metadata buffer"); + Check(s16.instSize == 16 * kEntrySize, "instance buffer grew to 16 entries"); + Check(RenderingElement3D::tlases[0].builtInstanceCount == 16, + "builtInstanceCount == 16 after growth"); + + // ── 3. Shrink to 2 → REUSES the 16-entry allocation (the core win). ───── + SetCount(pool, 2); + BuildOnce(); + InputSnapshot s2 = Snapshot(); + Check(Reused(s16, s2), + "shrinking to 2 reused the existing buffers (no realloc — high-water mark)"); + Check(s2.instSize == 16 * kEntrySize, + "instance buffer kept its 16-entry high-water size after shrink"); + Check(RenderingElement3D::tlases[0].builtInstanceCount == 2, + "builtInstanceCount tracks the live count (2) even on the reuse path"); + + // ── 4. Grow to 10 (still within the high-water capacity) → REUSES. ────── + SetCount(pool, 10); + BuildOnce(); + InputSnapshot s10 = Snapshot(); + Check(Reused(s16, s10), + "growing to 10 (≤ capacity 16) reused the existing buffers"); + + // ── 5. Grow to exactly 16 (== capacity) → REUSES (boundary: > not >=). ── + SetCount(pool, 16); + BuildOnce(); + InputSnapshot s16b = Snapshot(); + Check(Reused(s16, s16b), + "growing to exactly the capacity (16) reused the existing buffers"); + + // ── 6. Same count again → refit (UPDATE) path, inputs untouched. ──────── + BuildOnce(); + InputSnapshot s16c = Snapshot(); + Check(Reused(s16, s16c), "same-count rebuild left the inputs untouched"); + Check(RenderingElement3D::tlases[0].builtInstanceCount == 16, + "same-count rebuild kept builtInstanceCount == 16 (took the refit path)"); + + // ── 7. Grow to 32 (past the high-water) → REALLOCATES again. ──────────── + SetCount(pool, 32); + BuildOnce(); + InputSnapshot s32 = Snapshot(); + Check(s32.instBuf != s16.instBuf, "growing past the high-water (32) reallocated again"); + Check(s32.instSize == 32 * kEntrySize, "instance buffer grew to 32 entries"); + + // Unregister everything so nothing dangles past the pool's lifetime. + SetCount(pool, 0); + + Check(Device::validationErrorCount == 0, + std::format("no Vulkan validation errors ({} seen)", Device::validationErrorCount)); + + // Tear down the frame-0 TLAS while the Vulkan device is still alive. The + // static tlases[] array is destroyed at process exit, by which point the + // device (also static) may already be gone — so its VulkanBuffer + // destructors would fault. Releasing here keeps the exit path clean. + { + auto& t0 = RenderingElement3D::tlases[0]; + if (t0.accelerationStructure != VK_NULL_HANDLE) { + Device::vkDestroyAccelerationStructureKHR(Device::device, t0.accelerationStructure, nullptr); + t0.accelerationStructure = VK_NULL_HANDLE; + } + if (t0.instanceBuffer.buffer != VK_NULL_HANDLE) t0.instanceBuffer.Clear(); + if (t0.metadataBuffer.buffer != VK_NULL_HANDLE) t0.metadataBuffer.Clear(); + if (t0.scratchBuffer.buffer != VK_NULL_HANDLE) t0.scratchBuffer.Clear(); + if (t0.buffer.buffer != VK_NULL_HANDLE) t0.buffer.Clear(); + } + + if (failures != 0) { + std::println("{} check(s) failed", failures); + return EXIT_FAILURE; + } + std::println("all checks passed"); + return EXIT_SUCCESS; +}