539 lines
28 KiB
C++
539 lines
28 KiB
C++
/*
|
|
Crafter®.Graphics
|
|
Copyright (C) 2026 Catcrafts®
|
|
catcrafts.net
|
|
|
|
This library is free software; you can redistribute it and/or
|
|
modify it under the terms of the GNU Lesser General Public
|
|
License version 3.0 as published by the Free Software Foundation;
|
|
|
|
This library is distributed in the hope that it will be useful,
|
|
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
|
Lesser General Public License for more details.
|
|
|
|
You should have received a copy of the GNU Lesser General Public
|
|
License along with this library; if not, write to the Free Software
|
|
Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
|
*/
|
|
|
|
module;
|
|
|
|
#ifndef CRAFTER_GRAPHICS_WINDOW_DOM
|
|
#include "vulkan/vulkan.h"
|
|
|
|
#endif // !CRAFTER_GRAPHICS_WINDOW_DOM
|
|
export module Crafter.Graphics:VulkanBuffer;
|
|
#ifndef CRAFTER_GRAPHICS_WINDOW_DOM
|
|
import std;
|
|
import :Device;
|
|
|
|
namespace Crafter {
|
|
// Round a host-write flush range outward to nonCoherentAtomSize boundaries —
|
|
// the alignment vkFlushMappedMemoryRanges demands for a sub-buffer range
|
|
// (whole-buffer VK_WHOLE_SIZE calls sidestep it). `atom` is
|
|
// VkPhysicalDeviceLimits::nonCoherentAtomSize (a power of two ≥ 1) and
|
|
// `mappingSize` is the byte size the memory was mapped with (the allocation
|
|
// size). The start rounds down and the end rounds up to atom multiples; the
|
|
// end is then clamped to `mappingSize`. Clamping is what keeps the range
|
|
// valid even when `mappingSize` itself is not atom-aligned: the spec permits
|
|
// a non-atom-multiple size only when offset+size equals the mapping size, and
|
|
// the clamp lands exactly on that exception. Pure math, so it is unit-tested
|
|
// without a device.
|
|
export struct MappedFlushRange { VkDeviceSize offset; VkDeviceSize size; };
|
|
export constexpr MappedFlushRange AlignMappedFlushRange(
|
|
VkDeviceSize offset, VkDeviceSize bytes,
|
|
VkDeviceSize atom, VkDeviceSize mappingSize) {
|
|
VkDeviceSize begin = (offset / atom) * atom;
|
|
VkDeviceSize end = ((offset + bytes + atom - 1) / atom) * atom;
|
|
if (end > mappingSize) end = mappingSize;
|
|
return { begin, end - begin };
|
|
}
|
|
|
|
export class VulkanBufferBase {
|
|
public:
|
|
VkDeviceAddress address;
|
|
std::uint32_t size;
|
|
// Byte size the memory was mapped with — equal to the allocation's
|
|
// VkMemoryRequirements::size, which is ≥ `size` and what ranged flushes
|
|
// clamp their upper bound to. Only meaningful for mapped buffers.
|
|
VkDeviceSize mappedSize = 0;
|
|
VkBuffer buffer = VK_NULL_HANDLE;
|
|
VkDeviceMemory memory;
|
|
// Property flags of the memory type actually chosen by GetMemoryType —
|
|
// not the requested flags. GetMemoryType may land a request without the
|
|
// COHERENT bit on a coherent type (and vice versa), so the flush/
|
|
// invalidate paths gate on this recorded value, never on the request.
|
|
VkMemoryPropertyFlags memoryPropertyFlagsChosen = 0;
|
|
// Byte capacity the buffer was created with, and the usage flags it was
|
|
// created with. Resize reuses the allocation in place when a new
|
|
// request still fits within `capacity` and these immutable-at-create
|
|
// properties match, avoiding a destroy+reallocate.
|
|
std::uint32_t capacity = 0;
|
|
VkBufferUsageFlags2 usageFlagsCreated = 0;
|
|
};
|
|
|
|
export template<typename T>
|
|
class VulkanBufferMapped {
|
|
public:
|
|
T* value;
|
|
};
|
|
export class VulkanBufferMappedEmpty {};
|
|
template<typename T, bool Mapped>
|
|
using VulkanBufferMappedConditional =
|
|
std::conditional_t<
|
|
Mapped,
|
|
VulkanBufferMapped<T>,
|
|
VulkanBufferMappedEmpty
|
|
>;
|
|
|
|
export template <typename T, bool Mapped>
|
|
class VulkanBuffer : public VulkanBufferBase, public VulkanBufferMappedConditional<T, Mapped> {
|
|
public:
|
|
VulkanBuffer() = default;
|
|
// `preferredPropertyFlags` are best-effort: the allocation prefers a
|
|
// memory type satisfying them on top of `memoryPropertyFlags`, but
|
|
// falls back to the required flags alone rather than failing. Use it
|
|
// for perf hints (e.g. DEVICE_LOCAL on a mapped buffer) that aren't
|
|
// available on every device — see Device::GetMemoryType.
|
|
void Create(VkBufferUsageFlags2 usageFlags, VkMemoryPropertyFlags memoryPropertyFlags, std::uint32_t count, VkMemoryPropertyFlags preferredPropertyFlags = 0) {
|
|
size = count * sizeof(T);
|
|
capacity = size;
|
|
usageFlagsCreated = usageFlags;
|
|
|
|
// Carry usage in the maintenance5 flags2 chain so 64-bit bits
|
|
// (e.g. VK_BUFFER_USAGE_2_MEMORY_DECOMPRESSION_BIT_EXT, bit 35)
|
|
// are not truncated.
|
|
VkBufferUsageFlags2CreateInfo usageFlags2 {
|
|
.sType = VK_STRUCTURE_TYPE_BUFFER_USAGE_FLAGS_2_CREATE_INFO,
|
|
.usage = usageFlags,
|
|
};
|
|
VkBufferCreateInfo bufferCreateInfo {};
|
|
bufferCreateInfo.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO;
|
|
bufferCreateInfo.pNext = &usageFlags2;
|
|
bufferCreateInfo.usage = 0;
|
|
bufferCreateInfo.size = sizeof(T)*count;
|
|
Device::CheckVkResult(vkCreateBuffer(Device::device, &bufferCreateInfo, nullptr, &buffer));
|
|
|
|
VkMemoryRequirements memReqs;
|
|
vkGetBufferMemoryRequirements(Device::device, buffer, &memReqs);
|
|
std::uint32_t memoryTypeIndex = Device::GetMemoryType(memReqs.memoryTypeBits, memoryPropertyFlags, preferredPropertyFlags);
|
|
memoryPropertyFlagsChosen = Device::memoryProperties.memoryTypes[memoryTypeIndex].propertyFlags;
|
|
VkMemoryAllocateInfo memAlloc {
|
|
.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO,
|
|
.allocationSize = memReqs.size,
|
|
.memoryTypeIndex = memoryTypeIndex
|
|
};
|
|
|
|
VkMemoryAllocateFlagsInfoKHR allocFlagsInfo {
|
|
.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_FLAGS_INFO_KHR,
|
|
.flags = VK_MEMORY_ALLOCATE_DEVICE_ADDRESS_BIT_KHR,
|
|
};
|
|
memAlloc.pNext = &allocFlagsInfo;
|
|
Device::CheckVkResult(vkAllocateMemory(Device::device, &memAlloc, nullptr, &memory));
|
|
|
|
Device::CheckVkResult(vkBindBufferMemory(Device::device, buffer, memory, 0));
|
|
|
|
VkBufferDeviceAddressInfo addressInfo = {
|
|
.sType = VK_STRUCTURE_TYPE_BUFFER_DEVICE_ADDRESS_INFO,
|
|
.buffer = buffer
|
|
};
|
|
address = vkGetBufferDeviceAddress(Device::device, &addressInfo);
|
|
|
|
// Record the allocation's byte size (≥ `size`) for every buffer,
|
|
// not just mapped ones: UploadDeviceLocalRange's direct path flushes
|
|
// only the dirty sub-range and clamps its rounded-up upper bound to
|
|
// this (the nonCoherentAtomSize-aligned-end exception, see
|
|
// AlignMappedFlushRange). A non-mapped buffer never maps persistently,
|
|
// but its memory is still the allocation a transient map writes into.
|
|
mappedSize = memReqs.size;
|
|
if constexpr(Mapped) {
|
|
Device::CheckVkResult(vkMapMemory(Device::device, memory, 0, memReqs.size, 0, reinterpret_cast<void**>(&(VulkanBufferMappedConditional<T, true>::value))));
|
|
}
|
|
}
|
|
|
|
void Clear() {
|
|
if constexpr(Mapped) {
|
|
vkUnmapMemory(Device::device, memory);
|
|
}
|
|
vkDestroyBuffer(Device::device, buffer, nullptr);
|
|
vkFreeMemory(Device::device, memory, nullptr);
|
|
buffer = VK_NULL_HANDLE;
|
|
}
|
|
|
|
// Like Clear(), but hands the destroy+free to Device's fence-keyed
|
|
// deletion queue instead of doing it immediately. Use when the buffer
|
|
// may still be read by an in-flight frame's GPU work — destroying it
|
|
// now would be a use-after-free, since frames are pipelined up to
|
|
// framesInFlight deep (issue #101). The handle is nulled immediately so
|
|
// this VulkanBuffer no longer owns it; vkFreeMemory (deferred) implicitly
|
|
// unmaps mapped memory, so no explicit vkUnmapMemory is needed here.
|
|
void DeferredClear() {
|
|
Device::EnqueueDeletion(buffer, memory);
|
|
buffer = VK_NULL_HANDLE;
|
|
}
|
|
|
|
void Resize(VkBufferUsageFlags2 usageFlags, VkMemoryPropertyFlags memoryPropertyFlags, std::uint32_t count, VkMemoryPropertyFlags preferredPropertyFlags = 0) {
|
|
// Reuse the existing allocation in place when the request still fits
|
|
// and the fixed-at-create properties match: usage flags are
|
|
// immutable after creation, and the memory type already chosen must
|
|
// still satisfy the required property flags. preferredPropertyFlags
|
|
// is a best-effort perf hint and does not affect correctness, so it
|
|
// is intentionally not part of the guard. The buffer handle (and its
|
|
// device address / mapped pointer) is preserved, only `size` shrinks
|
|
// to the new logical extent.
|
|
std::uint32_t requestedSize = count * sizeof(T);
|
|
if(buffer != VK_NULL_HANDLE
|
|
&& requestedSize <= capacity
|
|
&& usageFlags == usageFlagsCreated
|
|
&& (memoryPropertyFlagsChosen & memoryPropertyFlags) == memoryPropertyFlags) {
|
|
size = requestedSize;
|
|
return;
|
|
}
|
|
// Defer the old allocation's destruction: an in-flight frame may
|
|
// still reference it (the #63 hazard, no longer masked by a
|
|
// per-frame wait-idle since #40). DeferredClear nulls the handle,
|
|
// so the Create below starts from a clean slate.
|
|
if(buffer != VK_NULL_HANDLE) {
|
|
DeferredClear();
|
|
}
|
|
Create(usageFlags, memoryPropertyFlags, count, preferredPropertyFlags);
|
|
}
|
|
|
|
void Copy(VkCommandBuffer cmd, VulkanBuffer& dst) {
|
|
VkBufferCopy copyRegion = {
|
|
.srcOffset = 0,
|
|
.dstOffset = 0,
|
|
.size = size
|
|
};
|
|
|
|
vkCmdCopyBuffer(
|
|
cmd,
|
|
buffer,
|
|
dst.buffer,
|
|
1,
|
|
©Region
|
|
);
|
|
}
|
|
|
|
void Copy(VkCommandBuffer cmd, VulkanBuffer& dst, VkAccessFlags srcAccessMask, VkAccessFlags dstAccessMask, VkPipelineStageFlags srcStageMask, VkPipelineStageFlags dstStageMask) {
|
|
Copy(cmd, dst);
|
|
|
|
VkBufferMemoryBarrier barrier = {
|
|
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
|
.srcAccessMask = srcAccessMask,
|
|
.dstAccessMask = dstAccessMask,
|
|
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
|
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
|
.buffer = buffer,
|
|
.offset = 0,
|
|
.size = VK_WHOLE_SIZE
|
|
};
|
|
|
|
vkCmdPipelineBarrier(
|
|
cmd,
|
|
srcStageMask,
|
|
dstStageMask,
|
|
0,
|
|
0, NULL,
|
|
1, &barrier,
|
|
0, NULL
|
|
);
|
|
}
|
|
|
|
// Upload `count` host elements from `src` into this buffer as
|
|
// device-local, GPU-read geometry, choosing placement at runtime via
|
|
// the #89 upload strategy (Device::PreferDirectDeviceWrite) and then
|
|
// recording a barrier from the upload to (dstStageMask, dstAccessMask)
|
|
// on `cmd`:
|
|
// ReBAR / UMA (direct) — allocate HOST_VISIBLE with DEVICE_LOCAL as a
|
|
// best-effort preference (GetMemoryType lands the host-visible
|
|
// device-local type that the strategy already proved exists), map
|
|
// transiently, memcpy, flush if the chosen type isn't coherent,
|
|
// unmap. No staging buffer: it would be pure overhead on a bar where
|
|
// the device-local heap is itself host-writable.
|
|
// No / small BAR (staged) — allocate pure DEVICE_LOCAL (+ TRANSFER_DST),
|
|
// fill a transient HOST_VISIBLE staging buffer, vkCmdCopyBuffer into
|
|
// this buffer, then hand the staging buffer to the fence-keyed
|
|
// deferred-deletion queue (#101/#102) so it outlives the copy submit.
|
|
// Requires !Mapped: the destination must be free to be device-local-only
|
|
// (a persistent map would force HOST_VISIBLE), so the direct path maps
|
|
// just long enough to write. A same-size re-upload reuses the allocation
|
|
// (Resize), so the device address stays stable across an in-place AS
|
|
// UPDATE refit that re-calls this.
|
|
void UploadDeviceLocal(VkBufferUsageFlags2 usageFlags, const T* src, std::uint32_t count, VkCommandBuffer cmd, VkAccessFlags dstAccessMask, VkPipelineStageFlags dstStageMask) requires(!Mapped) {
|
|
VkDeviceSize bytes = static_cast<VkDeviceSize>(count) * sizeof(T);
|
|
VkAccessFlags srcAccessMask;
|
|
VkPipelineStageFlags srcStageMask;
|
|
if (Device::PreferDirectDeviceWrite(bytes)) {
|
|
Resize(usageFlags, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT, count, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT);
|
|
void* mapped = nullptr;
|
|
Device::CheckVkResult(vkMapMemory(Device::device, memory, 0, VK_WHOLE_SIZE, 0, &mapped));
|
|
std::memcpy(mapped, src, bytes);
|
|
// Match FlushDevice()'s gate: a non-coherent type needs an
|
|
// explicit flush before the device reads the written range.
|
|
if (!(memoryPropertyFlagsChosen & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT)) {
|
|
VkMappedMemoryRange range {
|
|
.sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE,
|
|
.memory = memory,
|
|
.offset = 0,
|
|
.size = VK_WHOLE_SIZE
|
|
};
|
|
vkFlushMappedMemoryRanges(Device::device, 1, &range);
|
|
}
|
|
vkUnmapMemory(Device::device, memory);
|
|
srcAccessMask = VK_ACCESS_HOST_WRITE_BIT;
|
|
srcStageMask = VK_PIPELINE_STAGE_HOST_BIT;
|
|
} else {
|
|
Resize(usageFlags | VK_BUFFER_USAGE_TRANSFER_DST_BIT, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, count);
|
|
// Persistent staging instead of a Create+DeferredClear per call
|
|
// (issue #120): on no-/small-BAR hardware Mesh::Refit /
|
|
// RecordProceduralBuild hit this branch every frame for deforming
|
|
// meshes, so a fresh alloc+map+free cycle per upload was pure
|
|
// churn on exactly the hardware this path targets. Keep a small
|
|
// ring of staging buffers, each grown to a high-water mark and
|
|
// reused — reallocated (by Resize, which defers the outgrown
|
|
// allocation) only when a larger upload arrives.
|
|
//
|
|
// The ring is grown lazily, on first entry into this staged
|
|
// branch, so ReBAR/UMA hardware (which always takes the direct
|
|
// `if` branch above) never constructs a staging allocation it
|
|
// does not use.
|
|
//
|
|
// Why a ring and not one shared buffer: the vkCmdCopyBuffer below
|
|
// still reads the staging buffer after this call returns — the
|
|
// reason the old code used DeferredClear. With frames pipelined
|
|
// framesInFlight deep, overwriting one shared staging buffer next
|
|
// frame would clobber data the previous frame's copy is still
|
|
// reading. Indexing by frameCounter % framesInFlight gives each
|
|
// in-flight frame its own slot; a slot is rewritten only after
|
|
// framesInFlight frames have elapsed — the exact window after
|
|
// which single-queue submission order (the same guarantee the
|
|
// #101 deletion queue relies on) ensures that copy has completed.
|
|
const std::uint32_t ringSize =
|
|
Device::framesInFlight ? Device::framesInFlight : 1;
|
|
if (stagingRing.size() < ringSize) {
|
|
stagingRing.resize(ringSize);
|
|
}
|
|
VulkanBuffer<T, true>& staging =
|
|
stagingRing[Device::frameCounter % ringSize];
|
|
// SHADER_DEVICE_ADDRESS: Create always queries the buffer device
|
|
// address (and allocates with the device-address bit); the
|
|
// staging buffer's own address is otherwise unused. Resize reuses
|
|
// the existing allocation (and its mapped pointer) when the
|
|
// upload still fits the slot's high-water capacity.
|
|
staging.Resize(VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT, count);
|
|
std::memcpy(staging.value, src, bytes);
|
|
staging.FlushDevice();
|
|
VkBufferCopy region { .srcOffset = 0, .dstOffset = 0, .size = bytes };
|
|
vkCmdCopyBuffer(cmd, staging.buffer, buffer, 1, ®ion);
|
|
srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
|
srcStageMask = VK_PIPELINE_STAGE_TRANSFER_BIT;
|
|
}
|
|
|
|
VkBufferMemoryBarrier barrier = {
|
|
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
|
.srcAccessMask = srcAccessMask,
|
|
.dstAccessMask = dstAccessMask,
|
|
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
|
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
|
.buffer = buffer,
|
|
.offset = 0,
|
|
.size = VK_WHOLE_SIZE
|
|
};
|
|
vkCmdPipelineBarrier(cmd, srcStageMask, dstStageMask, 0, 0, NULL, 1, &barrier, 0, NULL);
|
|
}
|
|
|
|
// Re-upload only the half-open sub-range [offset, offset+count) of an
|
|
// already-allocated device-local buffer — the dirty-range counterpart of
|
|
// UploadDeviceLocal. `src` points at the first changed element (not the
|
|
// start of the whole array); `offset` is that element's index in this
|
|
// buffer. The buffer must already exist at its full size (a prior
|
|
// UploadDeviceLocal / Build sized it): this never Resizes, so the device
|
|
// address stays stable and the untouched elements keep their last
|
|
// contents. Only the dirty bytes are written + flushed (direct path) or
|
|
// staged + copied (staged path), and the post-upload barrier covers only
|
|
// that sub-range — so a deforming-mesh refit that nudges a few vertices
|
|
// pays for those vertices, not a full-array re-upload + flush + copy (#119).
|
|
//
|
|
// The direct-vs-staged choice is read from the memory type the buffer was
|
|
// actually allocated with (HOST_VISIBLE → map + write in place;
|
|
// device-local-only → stage + GPU copy), NOT from the sub-range size: the
|
|
// allocation is fixed, so a small dirty range must not be mis-routed to a
|
|
// map of a non-host-visible buffer the way UploadDeviceLocal's size-based
|
|
// PreferDirectDeviceWrite check would. The staged path's GPU copy needs
|
|
// VK_BUFFER_USAGE_TRANSFER_DST_BIT on the destination — already present,
|
|
// since UploadDeviceLocal only allocates device-local-only memory on the
|
|
// staged path, where it adds that bit.
|
|
void UploadDeviceLocalRange(const T* src, std::uint32_t offset, std::uint32_t count, VkCommandBuffer cmd, VkAccessFlags dstAccessMask, VkPipelineStageFlags dstStageMask) requires(!Mapped) {
|
|
VkDeviceSize byteOffset = static_cast<VkDeviceSize>(offset) * sizeof(T);
|
|
VkDeviceSize bytes = static_cast<VkDeviceSize>(count) * sizeof(T);
|
|
VkAccessFlags srcAccessMask;
|
|
VkPipelineStageFlags srcStageMask;
|
|
if (memoryPropertyFlagsChosen & VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT) {
|
|
void* mapped = nullptr;
|
|
Device::CheckVkResult(vkMapMemory(Device::device, memory, 0, VK_WHOLE_SIZE, 0, &mapped));
|
|
std::memcpy(static_cast<std::byte*>(mapped) + byteOffset, src, bytes);
|
|
// Non-coherent memory needs an explicit flush — but only of the
|
|
// sub-range actually written, rounded outward to
|
|
// nonCoherentAtomSize (and clamped to the allocation size).
|
|
if (!(memoryPropertyFlagsChosen & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT)) {
|
|
MappedFlushRange r = AlignMappedFlushRange(
|
|
byteOffset, bytes, Device::nonCoherentAtomSize, mappedSize);
|
|
VkMappedMemoryRange range {
|
|
.sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE,
|
|
.memory = memory,
|
|
.offset = r.offset,
|
|
.size = r.size
|
|
};
|
|
vkFlushMappedMemoryRanges(Device::device, 1, &range);
|
|
}
|
|
vkUnmapMemory(Device::device, memory);
|
|
srcAccessMask = VK_ACCESS_HOST_WRITE_BIT;
|
|
srcStageMask = VK_PIPELINE_STAGE_HOST_BIT;
|
|
} else {
|
|
// Device-local-only: stage just the dirty elements and copy them
|
|
// into place at byteOffset. The staging buffer outlives the queued
|
|
// copy via the fence-keyed deletion queue (#101/#102), exactly as
|
|
// the full-buffer staged path does.
|
|
VulkanBuffer<T, true> staging;
|
|
staging.Create(VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_SHADER_DEVICE_ADDRESS_BIT, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT, count);
|
|
std::memcpy(staging.value, src, bytes);
|
|
staging.FlushDevice();
|
|
VkBufferCopy region { .srcOffset = 0, .dstOffset = byteOffset, .size = bytes };
|
|
vkCmdCopyBuffer(cmd, staging.buffer, buffer, 1, ®ion);
|
|
staging.DeferredClear();
|
|
srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT;
|
|
srcStageMask = VK_PIPELINE_STAGE_TRANSFER_BIT;
|
|
}
|
|
|
|
// Order only the written sub-range before the consumer reads it. The
|
|
// untouched bytes were made visible by their own prior upload's
|
|
// barrier and are not written in this submit, so they need no fresh
|
|
// dependency even though the build reads the whole buffer.
|
|
VkBufferMemoryBarrier barrier = {
|
|
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
|
.srcAccessMask = srcAccessMask,
|
|
.dstAccessMask = dstAccessMask,
|
|
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
|
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
|
.buffer = buffer,
|
|
.offset = byteOffset,
|
|
.size = bytes
|
|
};
|
|
vkCmdPipelineBarrier(cmd, srcStageMask, dstStageMask, 0, 0, NULL, 1, &barrier, 0, NULL);
|
|
}
|
|
|
|
void FlushDevice() requires(Mapped) {
|
|
// Coherent memory needs no explicit flush — host writes are
|
|
// automatically visible to the device.
|
|
if (memoryPropertyFlagsChosen & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) {
|
|
return;
|
|
}
|
|
VkMappedMemoryRange range = {
|
|
.sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE,
|
|
.memory = memory,
|
|
.offset = 0,
|
|
.size = VK_WHOLE_SIZE
|
|
};
|
|
vkFlushMappedMemoryRanges(Device::device, 1, &range);
|
|
}
|
|
|
|
// Flush only the host writes in [offset, offset+bytes) to the device,
|
|
// instead of the whole buffer. Use after touching a small sub-range
|
|
// (e.g. one descriptor in a multi-KB descriptor heap) so cache
|
|
// maintenance scales with the bytes actually written. The range is
|
|
// rounded outward to nonCoherentAtomSize as the Vulkan spec requires.
|
|
// No-op on coherent memory, same gate as the whole-buffer FlushDevice().
|
|
void FlushDevice(VkDeviceSize offset, VkDeviceSize bytes) requires(Mapped) {
|
|
if (memoryPropertyFlagsChosen & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) {
|
|
return;
|
|
}
|
|
MappedFlushRange r = AlignMappedFlushRange(
|
|
offset, bytes, Device::nonCoherentAtomSize, mappedSize);
|
|
VkMappedMemoryRange range = {
|
|
.sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE,
|
|
.memory = memory,
|
|
.offset = r.offset,
|
|
.size = r.size
|
|
};
|
|
vkFlushMappedMemoryRanges(Device::device, 1, &range);
|
|
}
|
|
|
|
void FlushDevice(VkCommandBuffer cmd, VkAccessFlags dstAccessMask, VkPipelineStageFlags dstStageMask) requires(Mapped) {
|
|
FlushDevice();
|
|
VkBufferMemoryBarrier barrier = {
|
|
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
|
.srcAccessMask = VK_ACCESS_HOST_WRITE_BIT,
|
|
.dstAccessMask = dstAccessMask,
|
|
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
|
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
|
.buffer = buffer,
|
|
.offset = 0,
|
|
.size = VK_WHOLE_SIZE
|
|
};
|
|
|
|
vkCmdPipelineBarrier(
|
|
cmd,
|
|
VK_PIPELINE_STAGE_HOST_BIT,
|
|
dstStageMask,
|
|
0,
|
|
0, NULL,
|
|
1, &barrier,
|
|
0, NULL
|
|
);
|
|
}
|
|
|
|
void FlushHost() requires(Mapped) {
|
|
// Coherent memory needs no explicit invalidate — device writes are
|
|
// automatically visible to the host.
|
|
if (memoryPropertyFlagsChosen & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) {
|
|
return;
|
|
}
|
|
VkMappedMemoryRange range = {
|
|
.sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE,
|
|
.memory = memory,
|
|
.offset = 0,
|
|
.size = VK_WHOLE_SIZE
|
|
};
|
|
vkInvalidateMappedMemoryRanges(Device::device, 1, &range);
|
|
}
|
|
|
|
// Persistent per-frame-in-flight staging ring for the staged
|
|
// UploadDeviceLocal path (issue #120). Empty — and so holding zero
|
|
// host-visible allocations — until the first upload that actually
|
|
// stages, which only happens on no-/small-BAR hardware (ReBAR/UMA always
|
|
// takes the direct-write branch and never touches this, so the lazy ring
|
|
// keeps the change a no-op there). Sized to Device::framesInFlight so
|
|
// each in-flight frame copies from its own slot; each slot grows to a
|
|
// high-water mark via Resize and is reused across frames rather than
|
|
// reallocated every call — mirroring the TLAS instance/metadata reuse.
|
|
std::vector<VulkanBuffer<T, true>> stagingRing;
|
|
|
|
VulkanBuffer(VulkanBuffer&& other) {
|
|
buffer = other.buffer;
|
|
memory = other.memory;
|
|
size = other.size;
|
|
mappedSize = other.mappedSize;
|
|
capacity = other.capacity;
|
|
usageFlagsCreated = other.usageFlagsCreated;
|
|
memoryPropertyFlagsChosen = other.memoryPropertyFlagsChosen;
|
|
other.buffer = VK_NULL_HANDLE;
|
|
address = other.address;
|
|
stagingRing = std::move(other.stagingRing);
|
|
if constexpr(Mapped) {
|
|
VulkanBufferMappedConditional<T, true>::value = other.VulkanBufferMappedConditional<T, true>::value;
|
|
}
|
|
};
|
|
|
|
~VulkanBuffer() {
|
|
if(buffer != VK_NULL_HANDLE) {
|
|
Clear();
|
|
}
|
|
}
|
|
|
|
VulkanBuffer(VulkanBuffer&) = delete;
|
|
VulkanBuffer& operator=(const VulkanBuffer&) = delete;
|
|
};
|
|
}
|
|
#endif // !CRAFTER_GRAPHICS_WINDOW_DOM
|