Crafter.Graphics/shaders/ui-quads.comp.glsl
catbot 225556532a perf(ui): skip the destination read-modify-write on empty tiles (#127)
The standalone per-category UI shaders (quads/circles/images/text)
unconditionally imageLoad the destination pixel up front and imageStore
it at the end, paying a full read-modify-write even on tiles no item
touches — the common case for a sparse UI. The fused uber-kernel already
amortizes a single load/store across all four categories; the standalone
Dispatch* path has no such umbrella.

Defer the imageLoad to just before the first surviving blend (`loaded`
flag) and skip the imageStore unless something was actually blended.
Output is bit-identical: before the first blend `dst` is unused, and a
skipped store leaves the pixel exactly as a no-op load+store would have
rewritten it.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-18 13:26:32 +00:00

111 lines
4.4 KiB
GLSL
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#version 460
#extension GL_GOOGLE_include_directive : enable
#include "ui-shared.glsl"
// One workgroup per 8×8 screen tile. The workgroup cooperatively streams the
// QuadItem list in chunks of 64 (see ui-shared.glsl), culling each chunk
// against the tile and compacting survivors — in buffer order — into shared
// memory; every thread then runs the per-pixel accumulate over only those
// survivors. Item order in the buffer == draw order on screen.
layout(push_constant) uniform PC {
UIDispatchHeader hdr;
} pc;
layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in;
// Per-chunk cooperative-cull scratch (one slot per workgroup thread).
shared vec4 s_rect[UI_CHUNK];
shared vec4 s_color[UI_CHUNK];
shared vec4 s_corners[UI_CHUNK];
shared vec4 s_outline[UI_CHUNK];
shared uint s_keep[UI_CHUNK];
shared uint s_order[UI_CHUNK];
shared uint s_count;
void main() {
// NOTE: do not early-return — every thread must reach the barriers below.
ivec2 screenPx;
bool valid = uiResolveScreenPixel(pc.hdr, screenPx);
// Defer the destination read-modify-write: a sparse UI leaves most tiles
// untouched, so only load the pixel when the first surviving item blends
// over it (`loaded`), and only store when something actually touched it.
// The fused kernel amortizes a single load/store across all categories; the
// standalone Dispatch* path has no such umbrella and otherwise pays a full
// read-modify-write per empty tile.
vec4 dst = vec4(0.0);
vec2 sp = vec2(0.0);
bool loaded = false;
if (valid) sp = vec2(screenPx) + 0.5;
vec2 tileMin, tileMax;
uiTileBounds(tileMin, tileMax);
uint lid = gl_LocalInvocationIndex;
for (uint base = 0u; base < pc.hdr.itemCount; base += UI_CHUNK) {
// Cooperative load: each thread reads one item member-by-member
// (the NVIDIA descriptor-heap workaround) into shared and culls it.
uint idx = base + lid;
bool keep = false;
if (idx < pc.hdr.itemCount) {
s_rect[lid] = uiQuadHeap[pc.hdr.itemBuffer].items[idx].rect;
s_color[lid] = uiQuadHeap[pc.hdr.itemBuffer].items[idx].color;
s_corners[lid] = uiQuadHeap[pc.hdr.itemBuffer].items[idx].corners;
s_outline[lid] = uiQuadHeap[pc.hdr.itemBuffer].items[idx].outline;
keep = uiAabbOverlapsTile(s_rect[lid].xy, s_rect[lid].xy + s_rect[lid].zw,
tileMin, tileMax);
}
s_keep[lid] = keep ? 1u : 0u;
barrier();
// Stable in-order compaction of this chunk's survivors.
if (lid == 0u) {
uint n = 0u;
uint lim = min(UI_CHUNK, pc.hdr.itemCount - base);
for (uint k = 0u; k < lim; ++k)
if (s_keep[k] != 0u) s_order[n++] = k;
s_count = n;
}
barrier();
if (valid) {
for (uint j = 0u; j < s_count; ++j) {
uint c = s_order[j];
// Cheap pre-test against the item's axis-aligned rect.
vec2 lo = s_rect[c].xy;
vec2 hi = s_rect[c].xy + s_rect[c].zw;
if (sp.x < lo.x || sp.y < lo.y) continue;
if (sp.x >= hi.x || sp.y >= hi.y) continue;
vec2 halfSize = s_rect[c].zw * 0.5;
vec2 p = sp - (s_rect[c].xy + halfSize);
float d = uiSdRoundRect(p, halfSize, s_corners[c]);
vec4 outline = s_outline[c];
float bodyA = clamp(0.5 - d, 0.0, 1.0);
if (bodyA <= 0.0 && outline.x <= 0.0) continue;
vec4 col = s_color[c];
vec4 src = vec4(col.rgb, col.a * bodyA);
if (outline.x > 0.0) {
float t = abs(d + outline.x * 0.5) - outline.x * 0.5;
float outlineA = clamp(0.5 - t, 0.0, 1.0);
src.rgb = mix(src.rgb, outline.yzw, outlineA);
src.a = max(src.a, outlineA);
}
if (src.a <= 0.0) continue;
if (!loaded) {
dst = imageLoad(uiImages[pc.hdr.outImage], screenPx);
loaded = true;
}
dst = uiBlendOver(dst, src);
}
}
barrier(); // done reading shared for this chunk before it's overwritten
}
if (loaded) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); // loaded ⇒ valid
}