Crafter.Graphics/shaders/ui-circles.comp.glsl
catbot 6eb11fbf71 perf(ui): cooperative shared-memory tile culling in UI compute shaders
Each 8×8-tile workgroup previously had all 64 threads walk the entire
item list, re-reading every item from the SSBO 64× and running the
per-pixel reject for items that never touch the tile — O(tiles × N)
item loads dominated by SSBO traffic.

The workgroup now streams the item list in chunks of 64: each thread
loads one item, tests its AABB against the whole tile, and the survivors
are compacted — in original buffer order — into shared memory. Every
thread then runs the unchanged per-pixel accumulate over only those
survivors. This drops SSBO traffic ~64× (each item read once per
workgroup instead of once per pixel) and shrinks the inner loop to the
items that actually overlap the tile.

Draw order is preserved exactly: chunks run in array order and the
in-chunk compaction is a stable in-order scan, so later items still
overdraw earlier ones (no atomic-append nondeterminism). The inner loop
body is byte-for-byte the original per-pixel logic, so output is
pixel-identical — the cull only decides which items reach it. Threads
no longer early-return so every thread reaches the workgroup barriers.

Applied to ui-quads, ui-circles, ui-images and ui-text; shared helpers
(uiTileBounds / uiAabbOverlapsTile) live in ui-shared.glsl.

Resolves #46

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-16 15:45:12 +00:00

100 lines
3.4 KiB
GLSL
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#version 460
#extension GL_GOOGLE_include_directive : enable
#include "ui-shared.glsl"
// One workgroup per 8×8 screen tile. The workgroup cooperatively streams the
// CircleItem list in chunks of 64 (see ui-shared.glsl), culling each chunk
// against the tile and compacting survivors — in buffer order — into shared
// memory; every thread then accumulates over only those survivors.
layout(push_constant) uniform PC {
UIDispatchHeader hdr;
} pc;
layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in;
shared vec4 s_centerRadius[UI_CHUNK];
shared vec4 s_color[UI_CHUNK];
shared vec4 s_outline[UI_CHUNK];
shared uint s_keep[UI_CHUNK];
shared uint s_order[UI_CHUNK];
shared uint s_count;
void main() {
ivec2 screenPx;
bool valid = uiResolveScreenPixel(pc.hdr, screenPx);
vec4 dst = vec4(0.0);
vec2 sp = vec2(0.0);
if (valid) {
dst = imageLoad(uiImages[pc.hdr.outImage], screenPx);
sp = vec2(screenPx) + 0.5;
}
vec2 tileMin, tileMax;
uiTileBounds(tileMin, tileMax);
uint lid = gl_LocalInvocationIndex;
for (uint base = 0u; base < pc.hdr.itemCount; base += UI_CHUNK) {
uint idx = base + lid;
bool keep = false;
if (idx < pc.hdr.itemCount) {
s_centerRadius[lid] = uiCircleHeap[pc.hdr.itemBuffer].items[idx].centerRadius;
s_color[lid] = uiCircleHeap[pc.hdr.itemBuffer].items[idx].color;
s_outline[lid] = uiCircleHeap[pc.hdr.itemBuffer].items[idx].outline;
// Cull box matches the in-loop bounding-box reject (radius + 1px).
float radius = s_centerRadius[lid].z;
if (radius > 0.0) {
vec2 c = s_centerRadius[lid].xy;
vec2 r = vec2(radius + 1.0);
keep = uiAabbOverlapsTile(c - r, c + r, tileMin, tileMax);
}
}
s_keep[lid] = keep ? 1u : 0u;
barrier();
if (lid == 0u) {
uint n = 0u;
uint lim = min(UI_CHUNK, pc.hdr.itemCount - base);
for (uint k = 0u; k < lim; ++k)
if (s_keep[k] != 0u) s_order[n++] = k;
s_count = n;
}
barrier();
if (valid) {
for (uint j = 0u; j < s_count; ++j) {
uint c = s_order[j];
vec2 center = s_centerRadius[c].xy;
float radius = s_centerRadius[c].z;
if (radius <= 0.0) continue;
// Cheap bounding-box reject.
if (abs(sp.x - center.x) > radius + 1.0) continue;
if (abs(sp.y - center.y) > radius + 1.0) continue;
float d = length(sp - center) - radius;
vec4 outline = s_outline[c];
float bodyA = clamp(0.5 - d, 0.0, 1.0);
if (bodyA <= 0.0 && outline.x <= 0.0) continue;
vec4 col = s_color[c];
vec4 src = vec4(col.rgb, col.a * bodyA);
if (outline.x > 0.0) {
float t = abs(d + outline.x * 0.5) - outline.x * 0.5;
float outlineA = clamp(0.5 - t, 0.0, 1.0);
src.rgb = mix(src.rgb, outline.yzw, outlineA);
src.a = max(src.a, outlineA);
}
if (src.a <= 0.0) continue;
dst = uiBlendOver(dst, src);
}
}
barrier();
}
if (valid) imageStore(uiImages[pc.hdr.outImage], screenPx, dst);
}