Crafter.Graphics/shaders/ui-images.comp.glsl
catbot 225556532a perf(ui): skip the destination read-modify-write on empty tiles (#127)
The standalone per-category UI shaders (quads/circles/images/text)
unconditionally imageLoad the destination pixel up front and imageStore
it at the end, paying a full read-modify-write even on tiles no item
touches — the common case for a sparse UI. The fused uber-kernel already
amortizes a single load/store across all four categories; the standalone
Dispatch* path has no such umbrella.

Defer the imageLoad to just before the first surviving blend (`loaded`
flag) and skip the imageStore unless something was actually blended.
Output is bit-identical: before the first blend `dst` is unused, and a
skipped store leaves the pixel exactly as a no-op load+store would have
rewritten it.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-18 13:26:32 +00:00

98 lines
3.5 KiB
GLSL
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#version 460
#extension GL_GOOGLE_include_directive : enable
#include "ui-shared.glsl"
// One workgroup per 8×8 screen tile. The workgroup cooperatively streams the
// ImageItem list in chunks of 64 (see ui-shared.glsl), culling each chunk
// against the tile and compacting survivors — in buffer order — into shared
// memory; every thread then accumulates over only those survivors.
layout(push_constant) uniform PC {
UIDispatchHeader hdr;
} pc;
layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in;
shared vec4 s_rect[UI_CHUNK];
shared vec4 s_uv[UI_CHUNK];
shared vec4 s_tint[UI_CHUNK];
shared uvec4 s_slots[UI_CHUNK];
shared uint s_keep[UI_CHUNK];
shared uint s_order[UI_CHUNK];
shared uint s_count;
void main() {
ivec2 screenPx;
bool valid = uiResolveScreenPixel(pc.hdr, screenPx);
// Defer the destination read-modify-write: a sparse UI leaves most tiles
// untouched, so only load the pixel when the first surviving item blends
// over it (`loaded`), and only store when something actually touched it.
// The fused kernel amortizes a single load/store across all categories; the
// standalone Dispatch* path has no such umbrella and otherwise pays a full
// read-modify-write per empty tile.
vec4 dst = vec4(0.0);
vec2 sp = vec2(0.0);
bool loaded = false;
if (valid) sp = vec2(screenPx) + 0.5;
vec2 tileMin, tileMax;
uiTileBounds(tileMin, tileMax);
uint lid = gl_LocalInvocationIndex;
for (uint base = 0u; base < pc.hdr.itemCount; base += UI_CHUNK) {
uint idx = base + lid;
bool keep = false;
if (idx < pc.hdr.itemCount) {
s_rect[lid] = uiImageHeap[pc.hdr.itemBuffer].items[idx].rect;
s_uv[lid] = uiImageHeap[pc.hdr.itemBuffer].items[idx].uv;
s_tint[lid] = uiImageHeap[pc.hdr.itemBuffer].items[idx].tint;
s_slots[lid] = uiImageHeap[pc.hdr.itemBuffer].items[idx].slots;
keep = uiAabbOverlapsTile(s_rect[lid].xy, s_rect[lid].xy + s_rect[lid].zw,
tileMin, tileMax);
}
s_keep[lid] = keep ? 1u : 0u;
barrier();
if (lid == 0u) {
uint n = 0u;
uint lim = min(UI_CHUNK, pc.hdr.itemCount - base);
for (uint k = 0u; k < lim; ++k)
if (s_keep[k] != 0u) s_order[n++] = k;
s_count = n;
}
barrier();
if (valid) {
for (uint j = 0u; j < s_count; ++j) {
uint c = s_order[j];
vec2 lo = s_rect[c].xy;
vec2 hi = s_rect[c].xy + s_rect[c].zw;
if (sp.x < lo.x || sp.y < lo.y) continue;
if (sp.x >= hi.x || sp.y >= hi.y) continue;
vec2 t = (sp - s_rect[c].xy) / s_rect[c].zw;
vec2 uv = mix(s_uv[c].xy, s_uv[c].zw, t);
uint texSlot = s_slots[c].x;
uint sampSlot = s_slots[c].y;
vec4 sampled = texture(
sampler2D(uiTextures[nonuniformEXT(texSlot)],
uiSamplers[nonuniformEXT(sampSlot)]),
uv
);
vec4 src = sampled * s_tint[c];
if (src.a <= 0.0) continue;
if (!loaded) {
dst = imageLoad(uiImages[pc.hdr.outImage], screenPx);
loaded = true;
}
dst = uiBlendOver(dst, src);
}
}
barrier();
}
if (loaded) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); // loaded ⇒ valid
}