perf(ui): cooperative shared-memory tile culling in UI compute shaders

Each 8×8-tile workgroup previously had all 64 threads walk the entire
item list, re-reading every item from the SSBO 64× and running the
per-pixel reject for items that never touch the tile — O(tiles × N)
item loads dominated by SSBO traffic.

The workgroup now streams the item list in chunks of 64: each thread
loads one item, tests its AABB against the whole tile, and the survivors
are compacted — in original buffer order — into shared memory. Every
thread then runs the unchanged per-pixel accumulate over only those
survivors. This drops SSBO traffic ~64× (each item read once per
workgroup instead of once per pixel) and shrinks the inner loop to the
items that actually overlap the tile.

Draw order is preserved exactly: chunks run in array order and the
in-chunk compaction is a stable in-order scan, so later items still
overdraw earlier ones (no atomic-append nondeterminism). The inner loop
body is byte-for-byte the original per-pixel logic, so output is
pixel-identical — the cull only decides which items reach it. Threads
no longer early-return so every thread reaches the workgroup barriers.

Applied to ui-quads, ui-circles, ui-images and ui-text; shared helpers
(uiTileBounds / uiAabbOverlapsTile) live in ui-shared.glsl.

Resolves #46

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
catbot 2026-06-16 15:45:12 +00:00
commit 6eb11fbf71
5 changed files with 337 additions and 107 deletions

View file

@ -2,8 +2,10 @@
#extension GL_GOOGLE_include_directive : enable
#include "ui-shared.glsl"
// One workgroup per 8×8 screen tile. Iterates every glyph in order; each
// pixel keeps a local accumulator so order in the buffer == draw order.
// One workgroup per 8×8 screen tile. The workgroup cooperatively streams the
// GlyphItem list in chunks of 64 (see ui-shared.glsl), culling each chunk
// against the tile and compacting survivors — in buffer order — into shared
// memory; every thread then accumulates over only those survivors.
layout(push_constant) uniform PC {
UIDispatchHeader hdr;
uint fontTextureSlot;
@ -18,48 +20,90 @@ layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in;
const float ON_EDGE = 128.0 / 255.0;
const float DIST_SCALE = 32.0;
shared vec4 s_rect[UI_CHUNK];
shared vec4 s_uv[UI_CHUNK];
shared vec4 s_color[UI_CHUNK];
shared uint s_keep[UI_CHUNK];
shared uint s_order[UI_CHUNK];
shared uint s_count;
void main() {
ivec2 screenPx;
if (!uiResolveScreenPixel(pc.hdr, screenPx)) return;
bool valid = uiResolveScreenPixel(pc.hdr, screenPx);
vec4 dst = imageLoad(uiImages[pc.hdr.outImage], screenPx);
vec2 sp = vec2(screenPx) + 0.5;
for (uint i = 0u; i < pc.hdr.itemCount; ++i) {
GlyphItem it = LoadGlpyhtem(pc.hdr.itemBuffer, i);
vec2 lo = it.rect.xy;
vec2 hi = it.rect.xy + it.rect.zw;
if (sp.x < lo.x || sp.y < lo.y) continue;
if (sp.x >= hi.x || sp.y >= hi.y) continue;
vec2 t = (sp - it.rect.xy) / it.rect.zw;
vec2 uv = mix(it.uv.xy, it.uv.zw, t);
float sdf = texture(
sampler2D(uiTextures[nonuniformEXT(pc.fontTextureSlot)],
uiSamplers[nonuniformEXT(pc.fontSamplerSlot)]),
uv
).r;
// Distance in atlas-pixels (negative inside the glyph).
float dAtlas = (ON_EDGE - sdf) * DIST_SCALE;
// Atlas-px per screen-px along this glyph's transform — keeps AA crisp
// at any rendering size. uvSpan * atlasSize / screenSpan.
vec2 uvSpan = it.uv.zw - it.uv.xy;
// FontAtlas::kAtlasSize = 1024.
vec2 atlasPerScreen = (uvSpan * 1024.0) / it.rect.zw;
float scalePx = max(atlasPerScreen.x, atlasPerScreen.y);
// 1-screen-px AA band, expressed in atlas-pixel units of dAtlas.
float band = max(scalePx, 0.0001);
float a = clamp(0.5 - dAtlas / band, 0.0, 1.0);
if (a <= 0.0) continue;
vec4 src = vec4(it.color.rgb, it.color.a * a);
dst = uiBlendOver(dst, src);
vec4 dst = vec4(0.0);
vec2 sp = vec2(0.0);
if (valid) {
dst = imageLoad(uiImages[pc.hdr.outImage], screenPx);
sp = vec2(screenPx) + 0.5;
}
imageStore(uiImages[pc.hdr.outImage], screenPx, dst);
vec2 tileMin, tileMax;
uiTileBounds(tileMin, tileMax);
uint lid = gl_LocalInvocationIndex;
for (uint base = 0u; base < pc.hdr.itemCount; base += UI_CHUNK) {
uint idx = base + lid;
bool keep = false;
if (idx < pc.hdr.itemCount) {
s_rect[lid] = uiGlyphHeap[pc.hdr.itemBuffer].items[idx].rect;
s_uv[lid] = uiGlyphHeap[pc.hdr.itemBuffer].items[idx].uv;
s_color[lid] = uiGlyphHeap[pc.hdr.itemBuffer].items[idx].color;
keep = uiAabbOverlapsTile(s_rect[lid].xy, s_rect[lid].xy + s_rect[lid].zw,
tileMin, tileMax);
}
s_keep[lid] = keep ? 1u : 0u;
barrier();
if (lid == 0u) {
uint n = 0u;
uint lim = min(UI_CHUNK, pc.hdr.itemCount - base);
for (uint k = 0u; k < lim; ++k)
if (s_keep[k] != 0u) s_order[n++] = k;
s_count = n;
}
barrier();
if (valid) {
for (uint j = 0u; j < s_count; ++j) {
uint c = s_order[j];
vec2 lo = s_rect[c].xy;
vec2 hi = s_rect[c].xy + s_rect[c].zw;
if (sp.x < lo.x || sp.y < lo.y) continue;
if (sp.x >= hi.x || sp.y >= hi.y) continue;
vec2 t = (sp - s_rect[c].xy) / s_rect[c].zw;
vec2 uv = mix(s_uv[c].xy, s_uv[c].zw, t);
float sdf = texture(
sampler2D(uiTextures[nonuniformEXT(pc.fontTextureSlot)],
uiSamplers[nonuniformEXT(pc.fontSamplerSlot)]),
uv
).r;
// Distance in atlas-pixels (negative inside the glyph).
float dAtlas = (ON_EDGE - sdf) * DIST_SCALE;
// Atlas-px per screen-px along this glyph's transform — keeps AA crisp
// at any rendering size. uvSpan * atlasSize / screenSpan.
vec2 uvSpan = s_uv[c].zw - s_uv[c].xy;
// FontAtlas::kAtlasSize = 1024.
vec2 atlasPerScreen = (uvSpan * 1024.0) / s_rect[c].zw;
float scalePx = max(atlasPerScreen.x, atlasPerScreen.y);
// 1-screen-px AA band, expressed in atlas-pixel units of dAtlas.
float band = max(scalePx, 0.0001);
float a = clamp(0.5 - dAtlas / band, 0.0, 1.0);
if (a <= 0.0) continue;
vec4 col = s_color[c];
vec4 src = vec4(col.rgb, col.a * a);
dst = uiBlendOver(dst, src);
}
}
barrier();
}
if (valid) imageStore(uiImages[pc.hdr.outImage], screenPx, dst);
}