perf(pipeline): shared, disk-persisted VkPipelineCache (#69) #97

Merged
catbot merged 2 commits from claude/issue-69 into master 2026-06-16 20:24:03 +02:00
7 changed files with 332 additions and 5 deletions

1
.gitignore vendored
View file

@ -1,2 +1,3 @@
build/
bin/
pipeline_cache.bin

View file

@ -82,7 +82,7 @@ void ComputeShader::Load(const std::filesystem::path& spvPath) {
.layout = VK_NULL_HANDLE,
};
Device::CheckVkResult(vkCreateComputePipelines(
Device::device, VK_NULL_HANDLE, 1, &info, nullptr, &pipeline));
Device::device, Device::pipelineCache, 1, &info, nullptr, &pipeline));
}
void ComputeShader::Dispatch(VkCommandBuffer cmd,

View file

@ -136,6 +136,97 @@ void Device::CheckVkResult(VkResult result) {
}
}
bool Device::PipelineCacheDataCompatible(std::span<const std::byte> data) {
// VkPipelineCacheHeaderVersionOne is a fixed 32-byte little-endian header
// (Vulkan spec 16.5.2): u32 headerSize, u32 headerVersion, u32 vendorID,
// u32 deviceID, then VK_UUID_SIZE bytes of pipelineCacheUUID. Parse the
// fields by offset rather than reinterpret_cast'ing the struct so the
// check is independent of the C struct's alignment/padding.
constexpr std::size_t headerBytes = 16 + VK_UUID_SIZE;
if (data.size() < headerBytes) {
return false;
}
auto readU32 = [&](std::size_t offset) {
std::uint32_t value;
std::memcpy(&value, data.data() + offset, sizeof(value));
return value;
};
const std::uint32_t headerSize = readU32(0);
const std::uint32_t headerVersion = readU32(4);
const std::uint32_t vendorID = readU32(8);
const std::uint32_t deviceID = readU32(12);
if (headerSize < headerBytes) {
return false;
}
if (headerVersion != VK_PIPELINE_CACHE_HEADER_VERSION_ONE) {
return false;
}
if (vendorID != deviceProperties.vendorID || deviceID != deviceProperties.deviceID) {
return false;
}
return std::memcmp(data.data() + 16, deviceProperties.pipelineCacheUUID, VK_UUID_SIZE) == 0;
}
void Device::LoadPipelineCache() {
std::vector<std::byte> initialData;
std::error_code ec;
if (std::filesystem::exists(pipelineCachePath, ec)) {
std::ifstream file(pipelineCachePath, std::ios::binary | std::ios::ate);
if (file) {
const std::streamoff size = file.tellg();
if (size > 0) {
initialData.resize(static_cast<std::size_t>(size));
file.seekg(0);
file.read(reinterpret_cast<char*>(initialData.data()), size);
if (!file) {
initialData.clear(); // partial/short read — start cold
}
}
}
}
// Drop a blob the current driver/device wouldn't accept: a header from a
// different GPU (or a corrupt/empty file) would at best be ignored and at
// worst rejected. Starting empty just costs a one-time cold compile.
if (!PipelineCacheDataCompatible(initialData)) {
initialData.clear();
}
VkPipelineCacheCreateInfo info {
.sType = VK_STRUCTURE_TYPE_PIPELINE_CACHE_CREATE_INFO,
.initialDataSize = initialData.size(),
.pInitialData = initialData.empty() ? nullptr : initialData.data(),
};
CheckVkResult(vkCreatePipelineCache(device, &info, nullptr, &pipelineCache));
}
void Device::SavePipelineCache() {
if (pipelineCache == VK_NULL_HANDLE) {
return;
}
// Two-call idiom: query size, then fetch. Best-effort — a failure here only
// forfeits the next run's warm start, so it must never throw out of an
// atexit handler.
std::size_t size = 0;
if (vkGetPipelineCacheData(device, pipelineCache, &size, nullptr) != VK_SUCCESS || size == 0) {
return;
}
std::vector<std::byte> data(size);
if (vkGetPipelineCacheData(device, pipelineCache, &size, data.data()) != VK_SUCCESS) {
return;
}
std::ofstream file(pipelineCachePath, std::ios::binary | std::ios::trunc);
if (file) {
file.write(reinterpret_cast<const char*>(data.data()), static_cast<std::streamsize>(size));
}
}
VkBool32 onError(VkDebugUtilsMessageSeverityFlagBitsEXT severity, VkDebugUtilsMessageTypeFlagsEXT type, const VkDebugUtilsMessengerCallbackDataEXT* callbackData, void* userData)
{
printf("Vulkan ");
@ -603,6 +694,9 @@ void Device::Initialize() {
.pNext = &rayTracingProperties
};
vkGetPhysicalDeviceProperties2(physDevice, &properties2);
// Keep the core properties around for the pipeline-cache identity check
// (vendorID / deviceID / pipelineCacheUUID).
deviceProperties = properties2.properties;
// NVIDIA's brand-new VK_EXT_descriptor_heap acceleration-structure read
// path faults (see #7); enable the SPIR-V rewrite workaround there. Other
@ -810,6 +904,14 @@ void Device::Initialize() {
memoryDecompressionSupported = false;
}
}
// Create the shared pipeline cache (seeded from disk when a compatible
// blob exists) and arrange for it to be written back at process exit. The
// device handle is a never-destroyed static, so it is still valid when the
// atexit handler runs. Registered once — Initialize is the single device
// bring-up.
LoadPipelineCache();
std::atexit(SavePipelineCache);
}
std::uint32_t Device::GetMemoryType(uint32_t typeBits, VkMemoryPropertyFlags required, VkMemoryPropertyFlags preferred) {

View file

@ -161,6 +161,27 @@ export namespace Crafter {
inline static VkPhysicalDeviceMemoryProperties memoryProperties;
// Core physical-device properties, captured once at Initialize from the
// VkPhysicalDeviceProperties2 query. Kept because the pipeline-cache
// persistence path needs vendorID / deviceID / pipelineCacheUUID to
// decide whether an on-disk blob was written by this exact device.
inline static VkPhysicalDeviceProperties deviceProperties = {};
// One process-wide pipeline cache fed to every vkCreate*Pipelines call
// (compute UI/user shaders + RT pipelines). Pipeline compilation is a
// one-time cold-start cost, not a per-frame one; a single shared cache
// lets the driver reuse compiled shader binaries across the several
// pipelines built at startup and — once persisted to disk (see
// LoadPipelineCache / SavePipelineCache) — across runs. VK_NULL_HANDLE
// until LoadPipelineCache runs; passing VK_NULL_HANDLE to a create call
// is valid and simply means "no cache", so the create sites need no
// null check.
inline static VkPipelineCache pipelineCache = VK_NULL_HANDLE;
// Where the serialized cache lives. Relative to the working directory,
// matching the gpu_crash_dump-* convention. Set before Initialize to
// relocate it.
inline static std::filesystem::path pipelineCachePath = "pipeline_cache.bin";
inline static VkPhysicalDeviceDescriptorHeapPropertiesEXT descriptorHeapProperties = {
.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_DESCRIPTOR_HEAP_PROPERTIES_EXT
};
@ -199,6 +220,23 @@ export namespace Crafter {
// ComputeShader read the offset off the pipeline they record.
static void CheckVkResult(VkResult result);
// ─── Pipeline cache persistence (issue #69) ─────────────────────
// LoadPipelineCache creates `pipelineCache`, seeding it from
// pipelineCachePath when the file exists *and* its header matches this
// device (PipelineCacheDataCompatible) — a stale or foreign blob is
// discarded so the driver never rejects it. Called once from
// Initialize. SavePipelineCache writes the driver's current cache blob
// back out; Initialize registers it with std::atexit so the cache is
// serialized at process shutdown without any explicit teardown call.
static void LoadPipelineCache();
static void SavePipelineCache();
// True when `data` is a VkPipelineCache blob whose header was written by
// a device with the same vendorID / deviceID / pipelineCacheUUID as
// `deviceProperties`. Pure logic over the standard 32-byte cache header,
// so it is driven directly by the PipelineCacheValidation test with no
// GPU device. A blob too short to hold the header is incompatible.
static bool PipelineCacheDataCompatible(std::span<const std::byte> data);
// Selects a memory type index from typeBits that satisfies `required`.
// When `preferred` bits are also given, a type satisfying both is
// chosen first; if none exists we fall back to required-only rather

View file

@ -82,7 +82,7 @@ export namespace Crafter {
.layout = VK_NULL_HANDLE
};
Device::CheckVkResult(Device::vkCreateRayTracingPipelinesKHR(Device::device, {}, {}, 1, &rtPipelineInfo, nullptr, &pipeline));
Device::CheckVkResult(Device::vkCreateRayTracingPipelinesKHR(Device::device, {}, Device::pipelineCache, 1, &rtPipelineInfo, nullptr, &pipeline));
std::size_t dataSize = Device::rayTracingProperties.shaderGroupHandleSize * rtPipelineInfo.groupCount;
shaderHandles.resize(dataSize);

View file

@ -587,6 +587,35 @@ extern "C" Configuration CrafterBuildProject(std::span<const std::string_view> a
hitImpls.emplace_back("tests/InputFieldHitTest/main");
hc.GetInterfacesAndImplementations(ifaces, hitImpls);
cfg.tests.push_back(std::move(hitTest));
// Issue #69: the engine feeds one shared Device::pipelineCache to every
// vkCreate*Pipelines call and persists it across runs, discarding an
// on-disk blob whose header doesn't match the current GPU. That gate —
// Device::PipelineCacheDataCompatible — is pure logic over the standard
// VkPipelineCache header and Device::deviceProperties, so this test
// stamps synthetic device identities + headers and drives it directly,
// no GPU device needed at runtime.
Test cacheTest;
Configuration& cc = cacheTest.config;
cc.path = cfg.path;
cc.name = "PipelineCacheValidation";
cc.outputName = "PipelineCacheValidation";
cc.type = ConfigurationType::Executable;
cc.target = cfg.target;
cc.march = cfg.march;
cc.mtune = cfg.mtune;
cc.debug = cfg.debug;
cc.sysroot = cfg.sysroot;
cc.dependencies = cfg.dependencies;
cc.externalDependencies = cfg.externalDependencies;
cc.compileFlags = cfg.compileFlags;
cc.linkFlags = cfg.linkFlags;
cc.defines = cfg.defines;
cc.cFiles = cfg.cFiles;
std::vector<fs::path> cacheImpls(impls.begin(), impls.end());
cacheImpls.emplace_back("tests/PipelineCacheValidation/main");
cc.GetInterfacesAndImplementations(ifaces, cacheImpls);
cfg.tests.push_back(std::move(cacheTest));
}
return cfg;

View file

@ -0,0 +1,157 @@
/*
Crafter®.Graphics
Copyright (C) 2026 Catcrafts®
catcrafts.net
This library is free software; you can redistribute it and/or
modify it under the terms of the GNU Lesser General Public
License version 3.0 as published by the Free Software Foundation;
This library is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
Lesser General Public License for more details.
You should have received a copy of the GNU Lesser General Public
License along with this library; if not, write to the Free Software
Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
*/
// Regression test for issue #69: the engine feeds a shared Device::pipelineCache
// to every vkCreate*Pipelines call and persists it across runs. A blob written
// by a different GPU (or a corrupt/short file) must be rejected before it is
// handed to vkCreatePipelineCache, otherwise the driver ignores or rejects it.
// Device::PipelineCacheDataCompatible is that gate: pure logic over the standard
// 32-byte VkPipelineCacheHeaderVersionOne header (headerSize, headerVersion,
// vendorID, deviceID, pipelineCacheUUID) compared against Device::deviceProperties.
// It needs no GPU, so this test stamps synthetic device identities and headers
// and drives it directly, mirroring MemoryTypeFallback / UploadStrategy.
#include <cstdlib>
#include <cstring>
#include "vulkan/vulkan.h"
import Crafter.Graphics;
import std;
using namespace Crafter;
namespace {
int failures = 0;
void Check(bool ok, std::string_view what) {
std::println("{} {}", ok ? "PASS" : "FAIL", what);
if (!ok) ++failures;
}
// Stamp the device identity the validator compares against.
void SetDevice(std::uint32_t vendorID, std::uint32_t deviceID,
const std::array<std::uint8_t, VK_UUID_SIZE>& uuid) {
Device::deviceProperties = {};
Device::deviceProperties.vendorID = vendorID;
Device::deviceProperties.deviceID = deviceID;
std::memcpy(Device::deviceProperties.pipelineCacheUUID, uuid.data(), VK_UUID_SIZE);
}
// Build a 32-byte VkPipelineCacheHeaderVersionOne blob, optionally with extra
// trailing payload bytes (the real cache body). Fields are written little-endian
// at fixed offsets, matching how the driver lays the header out on disk.
std::vector<std::byte> MakeHeader(std::uint32_t headerSize,
std::uint32_t headerVersion,
std::uint32_t vendorID,
std::uint32_t deviceID,
const std::array<std::uint8_t, VK_UUID_SIZE>& uuid,
std::size_t trailing = 0) {
std::vector<std::byte> blob(16 + VK_UUID_SIZE + trailing, std::byte{0xAB});
auto put = [&](std::size_t off, std::uint32_t v) {
std::memcpy(blob.data() + off, &v, sizeof(v));
};
put(0, headerSize);
put(4, headerVersion);
put(8, vendorID);
put(12, deviceID);
std::memcpy(blob.data() + 16, uuid.data(), VK_UUID_SIZE);
return blob;
}
constexpr std::uint32_t kVendor = 0x10DE; // NVIDIA
constexpr std::uint32_t kDevice = 0x2204; // some GPU device id
constexpr std::array<std::uint8_t, VK_UUID_SIZE> kUuid = {
0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08,
0x09, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0x10
};
constexpr std::uint32_t kV1 = VK_PIPELINE_CACHE_HEADER_VERSION_ONE;
constexpr std::uint32_t kHeaderSize = 16 + VK_UUID_SIZE; // 32
} // namespace
int main() {
SetDevice(kVendor, kDevice, kUuid);
// --- the happy path: a header written by this exact device --------------
{
auto blob = MakeHeader(kHeaderSize, kV1, kVendor, kDevice, kUuid);
Check(Device::PipelineCacheDataCompatible(blob),
"matching vendor/device/UUID header is accepted");
auto withBody = MakeHeader(kHeaderSize, kV1, kVendor, kDevice, kUuid, /*trailing*/ 4096);
Check(Device::PipelineCacheDataCompatible(withBody),
"matching header followed by a cache body is accepted");
}
// --- foreign / stale blobs must be rejected -----------------------------
{
auto otherVendor = MakeHeader(kHeaderSize, kV1, 0x1002 /*AMD*/, kDevice, kUuid);
Check(!Device::PipelineCacheDataCompatible(otherVendor),
"different vendorID is rejected");
auto otherDevice = MakeHeader(kHeaderSize, kV1, kVendor, 0x9999, kUuid);
Check(!Device::PipelineCacheDataCompatible(otherDevice),
"different deviceID is rejected");
std::array<std::uint8_t, VK_UUID_SIZE> otherUuid = kUuid;
otherUuid[15] ^= 0xFF; // a driver update bumps the UUID
auto staleUuid = MakeHeader(kHeaderSize, kV1, kVendor, kDevice, otherUuid);
Check(!Device::PipelineCacheDataCompatible(staleUuid),
"different pipelineCacheUUID (e.g. driver update) is rejected");
}
// --- malformed headers --------------------------------------------------
{
auto badVersion = MakeHeader(kHeaderSize, 0xDEAD, kVendor, kDevice, kUuid);
Check(!Device::PipelineCacheDataCompatible(badVersion),
"unknown headerVersion is rejected");
auto smallHeaderSize = MakeHeader(8, kV1, kVendor, kDevice, kUuid);
Check(!Device::PipelineCacheDataCompatible(smallHeaderSize),
"headerSize smaller than the 32-byte header is rejected");
Check(!Device::PipelineCacheDataCompatible({}),
"empty blob (no file / cold start) is rejected");
std::vector<std::byte> truncated(20, std::byte{0});
Check(!Device::PipelineCacheDataCompatible(truncated),
"blob too short to hold the header is rejected");
}
// --- identity follows the active device ---------------------------------
{
// Re-stamp as a different device; a blob valid for the old one is now
// foreign. Guards against the validator caching identity anywhere but
// Device::deviceProperties.
auto blob = MakeHeader(kHeaderSize, kV1, kVendor, kDevice, kUuid);
SetDevice(0x8086 /*Intel*/, 0x1234, kUuid);
Check(!Device::PipelineCacheDataCompatible(blob),
"a blob from the previous device is rejected after the device changes");
SetDevice(kVendor, kDevice, kUuid);
Check(Device::PipelineCacheDataCompatible(blob),
"...and accepted again once the matching device is restored");
}
if (failures != 0) {
std::println("{} check(s) failed", failures);
return EXIT_FAILURE;
}
std::println("all checks passed");
return EXIT_SUCCESS;
}