// SponzaBench — headless benchmark around the native Sponza RT scene. // // Identical setup to examples/Sponza (native branch): load the Sponza // .cmesh + albedo .ctex, build BLAS/TLAS, wire an RTPass. Instead of // opening an interactive window and idling in StartSync(), it times two // things and exits: // // * setup — process start through the first command submission // (asset decompress + BLAS/TLAS build + pipeline + memory // placement + descriptor writes), which the bulk of the // post-#40 perf work targets (device-local placement, // staging release, pipeline cache, decompress, ...). // * frames — a warmup then a measured loop calling Window::Render(), // reporting per-frame wall-clock stats. Run with // CRAFTER_PRESENT_IMMEDIATE=1 to uncap from vblank. // // Tunables (env): BENCH_WARMUP (default 200), BENCH_FRAMES (default 2000). // Output lines are prefixed "BENCH " for easy scraping. #include "vulkan/vulkan.h" import Crafter.Graphics; import Crafter.Asset; import Crafter.Math; import Crafter.Event; import std; using namespace Crafter; namespace fs = std::filesystem; using Clock = std::chrono::steady_clock; namespace { struct RGBA8 { std::uint8_t r, g, b, a; }; void RequireAssets(const fs::path& mesh, const fs::path& tex) { if (fs::exists(mesh) && fs::exists(tex)) return; std::println(std::cerr, "[SponzaBench] missing asset(s): {} / {}", mesh.string(), tex.string()); std::abort(); } // Peak resident set size (high-water mark) in kB, from /proc/self/status. long PeakRSSkB() { std::ifstream st("/proc/self/status"); std::string line; while (std::getline(st, line)) { if (line.rfind("VmHWM:", 0) == 0) { long kb = 0; std::sscanf(line.c_str(), "VmHWM: %ld", &kb); return kb; } } return -1; } std::size_t EnvSize(const char* name, std::size_t fallback) { if (const char* v = std::getenv(name)) { char* end = nullptr; unsigned long parsed = std::strtoul(v, &end, 10); if (end != v) return static_cast(parsed); } return fallback; } double Ms(std::chrono::nanoseconds ns) { return std::chrono::duration(ns).count(); } } int main() { const std::size_t warmup = EnvSize("BENCH_WARMUP", 200); const std::size_t frames = EnvSize("BENCH_FRAMES", 2000); const auto tProcStart = Clock::now(); const fs::path texPath = "tex_0.ctex"; RequireAssets("mesh_0.cmesh", texPath); // scene.txt (from the asset bundle): line 1 albedoCount, line 2 // meshCount, then per-mesh albedo index. We render every mesh group — // unlike the interactive Sponza example, which is single-material on // native because of the hit-shader dynamic-heap-index driver fault. // SponzaBench's hit shader samples no texture, so it can build the // full multi-mesh atrium as one multi-instance TLAS. std::uint32_t albedoCount = 0, meshCount = 0; { std::ifstream manifest("scene.txt"); if (!manifest) { std::println(std::cerr, "[SponzaBench] missing scene.txt"); std::abort(); } manifest >> albedoCount >> meshCount; } // BENCH_MESHES caps the number of mesh groups loaded (default: all). // A small cap (e.g. 1) shrinks GPU work so per-frame CPU cost dominates; // the full scene is GPU (traversal) bound. meshCount = std::min(meshCount, EnvSize("BENCH_MESHES", meshCount)); CompressedTextureAsset loadedTex = LoadCompressedTexture(texPath); std::println("[SponzaBench] scene: {} meshes, {} albedos, {}x{} probe albedo", meshCount, albedoCount, loadedTex.sizeX, loadedTex.sizeY); Device::Initialize(); Window window(1280, 720, "SponzaBench"); VkCommandBuffer cmd = window.StartInit(); DescriptorHeapVulkan descriptorHeap; descriptorHeap.Initialize(/*images*/ 2, /*buffers*/ 1, /*samplers*/ 0); VkSpecializationMapEntry raygenEntry = { .constantID = 0, .offset = 0, .size = sizeof(std::uint16_t) }; VkSpecializationInfo raygenSpec = { .mapEntryCount = 1, .pMapEntries = &raygenEntry, .dataSize = sizeof(std::uint16_t), .pData = &descriptorHeap.bufferStartElement, }; auto imgSlots = descriptorHeap.AllocateImageSlots(1); auto bufSlots = descriptorHeap.AllocateBufferSlots(1); std::array shaders {{ { "raygen.spv", "main", VK_SHADER_STAGE_RAYGEN_BIT_KHR, &raygenSpec }, { "miss.spv", "main", VK_SHADER_STAGE_MISS_BIT_KHR, nullptr }, { "closesthit.spv", "main", VK_SHADER_STAGE_CLOSEST_HIT_BIT_KHR, nullptr }, }}; ShaderBindingTableVulkan shaderTable; shaderTable.Init(shaders); std::array raygenGroups {{ { .sType = VK_STRUCTURE_TYPE_RAY_TRACING_SHADER_GROUP_CREATE_INFO_KHR, .type = VK_RAY_TRACING_SHADER_GROUP_TYPE_GENERAL_KHR, .generalShader = 0, .closestHitShader = VK_SHADER_UNUSED_KHR, .anyHitShader = VK_SHADER_UNUSED_KHR, .intersectionShader = VK_SHADER_UNUSED_KHR, } }}; std::array missGroups {{ { .sType = VK_STRUCTURE_TYPE_RAY_TRACING_SHADER_GROUP_CREATE_INFO_KHR, .type = VK_RAY_TRACING_SHADER_GROUP_TYPE_GENERAL_KHR, .generalShader = 1, .closestHitShader = VK_SHADER_UNUSED_KHR, .anyHitShader = VK_SHADER_UNUSED_KHR, .intersectionShader = VK_SHADER_UNUSED_KHR, } }}; std::array hitGroups {{ { .sType = VK_STRUCTURE_TYPE_RAY_TRACING_SHADER_GROUP_CREATE_INFO_KHR, .type = VK_RAY_TRACING_SHADER_GROUP_TYPE_TRIANGLES_HIT_GROUP_KHR, .generalShader = VK_SHADER_UNUSED_KHR, .closestHitShader = 2, .anyHitShader = VK_SHADER_UNUSED_KHR, .intersectionShader = VK_SHADER_UNUSED_KHR, } }}; PipelineRTVulkan pipeline; pipeline.Init(cmd, raygenGroups, missGroups, hitGroups, shaderTable); // One Mesh + RenderingElement3D per mesh group → BLAS each, then a // single multi-instance TLAS. Pointers handed to RenderingElement3D // live in the static vectors, so reserve up-front: a reallocation // would dangle every registered element. static std::vector meshes; static std::vector renderers; meshes.reserve(meshCount); renderers.reserve(meshCount); std::uint64_t totalVerts = 0, totalIdx = 0; for (std::uint32_t i = 0; i < meshCount; ++i) { const fs::path mp = std::format("mesh_{}.cmesh", i); if (!fs::exists(mp)) continue; CompressedMeshAsset loaded = LoadCompressedMesh(mp); totalVerts += loaded.vertexCount; totalIdx += loaded.indexCount; meshes.emplace_back(); meshes.back().Build(loaded, cmd); renderers.emplace_back(); RenderingElement3D& r = renderers.back(); r.instance = { .transform = {}, .instanceCustomIndex = 0, .mask = 0xFF, .instanceShaderBindingTableRecordOffset = 0, .flags = VK_GEOMETRY_INSTANCE_FORCE_OPAQUE_BIT_KHR, .accelerationStructureReference = meshes.back().blasAddr, }; MatrixRowMajor::Identity() .Store(reinterpret_cast(r.instance.transform.matrix)); RenderingElement3D::Add(&r); } std::println("[SponzaBench] built {} BLAS, {} verts, {} idx total", renderers.size(), totalVerts, totalIdx); Image2D albedo; albedo.Create(loadedTex.sizeX, loadedTex.sizeY, /*mipLevels*/ 1, cmd, VK_FORMAT_R8G8B8A8_UNORM, VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_SAMPLED_BIT, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL); albedo.Update(loadedTex, cmd, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL); // The albedo is uploaded to keep the texture-decompress + staging // setup path in the measured "setup" window, but is not bound — the // closest-hit shades from barycentrics, so only the TLAS + per-frame // output image need descriptors. for (std::uint32_t f = 0; f < Window::numFrames; ++f) RenderingElement3D::BuildTLAS(cmd, f); window.FinishInit(); VkDeviceAddressRangeKHR tlasRanges[Window::numFrames]; VkImageDescriptorInfoEXT outImgInfos[Window::numFrames]; for (std::uint32_t f = 0; f < Window::numFrames; ++f) { tlasRanges[f] = { .address = RenderingElement3D::tlases[f].address }; outImgInfos[f] = { .sType = VK_STRUCTURE_TYPE_IMAGE_DESCRIPTOR_INFO_EXT, .pView = &window.imageViews[f], .layout = VK_IMAGE_LAYOUT_GENERAL, }; } std::vector resources; std::vector destinations; resources.reserve(Window::numFrames * 2); destinations.reserve(Window::numFrames * 2); for (std::uint32_t f = 0; f < Window::numFrames; ++f) { resources.push_back({ .sType = VK_STRUCTURE_TYPE_RESOURCE_DESCRIPTOR_INFO_EXT, .type = VK_DESCRIPTOR_TYPE_ACCELERATION_STRUCTURE_KHR, .data = { .pAddressRange = &tlasRanges[f] }, }); destinations.push_back({ .address = descriptorHeap.resourceHeap[f].value + descriptorHeap.BufferByteOffset(bufSlots.firstElement), .size = Device::descriptorHeapProperties.bufferDescriptorSize, }); resources.push_back({ .sType = VK_STRUCTURE_TYPE_RESOURCE_DESCRIPTOR_INFO_EXT, .type = VK_DESCRIPTOR_TYPE_STORAGE_IMAGE, .data = { .pImage = &outImgInfos[f] }, }); destinations.push_back({ .address = descriptorHeap.resourceHeap[f].value + descriptorHeap.ImageByteOffset(imgSlots.firstElement), .size = Device::descriptorHeapProperties.imageDescriptorSize, }); } Device::vkWriteResourceDescriptorsEXT(Device::device, static_cast(resources.size()), resources.data(), destinations.data()); for (std::uint32_t f = 0; f < Window::numFrames; ++f) { descriptorHeap.resourceHeap[f].FlushDevice(); } window.descriptorHeap = &descriptorHeap; RTPass rtPass(&pipeline); window.passes.push_back(&rtPass); // ── Setup complete. Everything above is the per-run, one-time cost. ── const auto tSetupDone = Clock::now(); const double setupMs = Ms(tSetupDone - tProcStart); // ── Warmup: let the pipeline fill (frames-in-flight), caches warm. ── for (std::size_t i = 0; i < warmup; ++i) window.Render(); // ── Measured loop: per-frame wall-clock around Render(). ──────────── std::vector frameMs; frameMs.reserve(frames); const auto tLoopStart = Clock::now(); for (std::size_t i = 0; i < frames; ++i) { const auto a = Clock::now(); window.Render(); const auto b = Clock::now(); frameMs.push_back(Ms(b - a)); } const auto tLoopEnd = Clock::now(); // Drain so the loop's wall-clock isn't charged the GPU tail of the // last few in-flight frames inconsistently across runs. vkQueueWaitIdle(Device::queue); const long peakRSS = PeakRSSkB(); // ── Stats. throughput uses the loop wall-clock (steady-state FPS); // per-frame percentiles use the sorted per-frame samples. ───────── const double loopWallMs = Ms(tLoopEnd - tLoopStart); std::sort(frameMs.begin(), frameMs.end()); auto pct = [&](double p) { if (frameMs.empty()) return 0.0; std::size_t idx = static_cast(p * (frameMs.size() - 1)); return frameMs[idx]; }; double sum = 0.0; for (double v : frameMs) sum += v; const double meanMs = frameMs.empty() ? 0.0 : sum / frameMs.size(); std::println(""); std::println("BENCH setup_ms {:.3f}", setupMs); std::println("BENCH frames {}", frames); std::println("BENCH loop_wall_ms {:.3f}", loopWallMs); std::println("BENCH fps {:.1f}", frames / (loopWallMs / 1000.0)); std::println("BENCH frame_mean_ms {:.4f}", meanMs); std::println("BENCH frame_min_ms {:.4f}", pct(0.0)); std::println("BENCH frame_p50_ms {:.4f}", pct(0.50)); std::println("BENCH frame_p99_ms {:.4f}", pct(0.99)); std::println("BENCH frame_max_ms {:.4f}", pct(1.0)); std::println("BENCH peak_rss_kb {}", peakRSS); // The interactive Sponza example never tears Device/Window down (it // loops in StartSync forever); doing so here races GPU-still-in-flight // resources at static-destruction time. The measurement is already // printed, so skip destructors entirely with a hard exit. // // BENCH_CLEAN_EXIT=1 instead exits via std::exit, which runs atexit // handlers (on builds that have the disk pipeline cache, #69, this is // where it is written) before the destructor teardown. Used once to // seed pipeline_cache.bin for a warm-cache setup measurement; the // subsequent destructor race is irrelevant since the data is flushed. std::cout.flush(); if (std::getenv("BENCH_CLEAN_EXIT")) std::exit(0); std::_Exit(0); }