perf(font): upload only the dirty sub-rect of the glyph atlas (#51)
FontAtlas::Update re-uploaded the entire 1024x1024 R8 atlas (1 MiB) whenever a single glyph was rasterized, so warm-up / language-switch / size-churn cost roughly one full-atlas copy per frame that added a codepoint. Accumulate a dirty bounding box (FontAtlas::DirtyRect) in MarkDirty as glyphs are placed, and copy only that sub-rect. On Vulkan this is a new ImageVulkan::UpdateRegion: the staging buffer keeps the full atlas row stride, so bufferRowLength stays the atlas width and bufferOffset points at the rect origin while imageOffset/imageExtent carve out the rect. Layout transitions stay whole-image (cheap); only the copy extent shrinks. WebGPU passes the rect straight to wgpuWriteAtlasRegion, which already took x/y/w/h. The freshly-zeroed atlas marks its whole extent dirty in Initialize so the first Update clears the image once; thereafter every untouched texel stays the zero it was uploaded as, keeping the image byte-identical to the staging copy. Tested: new FontAtlasDirtyRect unit test covers the accumulation / clamp / lockstep-with-`dirty` math (no GPU needed); HelloUI renders text correctly on both Vulkan (NVIDIA, validation-clean) and WebGPU. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
parent
459f10afa8
commit
97c79c23ee
5 changed files with 276 additions and 6 deletions
|
|
@ -162,6 +162,34 @@ export namespace Crafter {
|
|||
}
|
||||
}
|
||||
|
||||
// Upload only the sub-rectangle [x, x+w) × [y, y+h) of the staging
|
||||
// buffer into the image, instead of the whole extent. The staging
|
||||
// buffer keeps the full image row stride, so bufferRowLength stays
|
||||
// `width` and bufferOffset just points at the rect's first texel —
|
||||
// imageOffset/imageExtent then carve out the rect on the GPU side.
|
||||
// Layout transitions stay whole-image (cheap); only the copy extent
|
||||
// shrinks. Single mip level only: the partial-upload path is for
|
||||
// CPU-streamed atlases (FontAtlas) which carry no mips, so there is
|
||||
// no blit chain to regenerate.
|
||||
void UpdateRegion(VkCommandBuffer cmd, VkImageLayout layout, std::uint32_t x, std::uint32_t y, std::uint32_t w, std::uint32_t h) {
|
||||
buffer.FlushDevice(cmd, VK_ACCESS_MEMORY_READ_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT);
|
||||
TransitionImageLayout(cmd, image, layout, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_PIPELINE_STAGE_RAY_TRACING_SHADER_BIT_KHR, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_ACCESS_SHADER_READ_BIT, VK_ACCESS_TRANSFER_WRITE_BIT, 0, mipLevels);
|
||||
|
||||
VkBufferImageCopy region{};
|
||||
region.bufferOffset = (static_cast<VkDeviceSize>(y) * width + x) * sizeof(PixelType);
|
||||
region.bufferRowLength = width;
|
||||
region.bufferImageHeight = 0;
|
||||
region.imageSubresource.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
|
||||
region.imageSubresource.mipLevel = 0;
|
||||
region.imageSubresource.baseArrayLayer = 0;
|
||||
region.imageSubresource.layerCount = 1;
|
||||
region.imageOffset = { static_cast<std::int32_t>(x), static_cast<std::int32_t>(y), 0 };
|
||||
region.imageExtent = { w, h, 1 };
|
||||
vkCmdCopyBufferToImage(cmd, buffer.buffer, image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, ®ion);
|
||||
|
||||
TransitionImageLayout(cmd, image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, layout, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_RAY_TRACING_SHADER_BIT_KHR, VK_ACCESS_TRANSFER_WRITE_BIT, VK_ACCESS_SHADER_READ_BIT, 0, mipLevels);
|
||||
}
|
||||
|
||||
// GPU compressed-asset Update: stage compressed bytes, decompress
|
||||
// into `buffer` via VK_EXT_memory_decompression, then copy buffer→image
|
||||
// and transition to `layout`. Falls back to CPU decode + the existing
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue