2026-05-02 21:08:20 +02:00
|
|
|
|
#version 460
|
|
|
|
|
|
#extension GL_GOOGLE_include_directive : enable
|
2026-06-18 13:26:37 +00:00
|
|
|
|
#extension GL_KHR_shader_subgroup_basic : enable
|
|
|
|
|
|
#extension GL_KHR_shader_subgroup_arithmetic : enable
|
2026-05-02 21:08:20 +02:00
|
|
|
|
#include "ui-shared.glsl"
|
|
|
|
|
|
|
2026-06-16 15:45:12 +00:00
|
|
|
|
// One workgroup per 8×8 screen tile. The workgroup cooperatively streams the
|
|
|
|
|
|
// QuadItem list in chunks of 64 (see ui-shared.glsl), culling each chunk
|
|
|
|
|
|
// against the tile and compacting survivors — in buffer order — into shared
|
|
|
|
|
|
// memory; every thread then runs the per-pixel accumulate over only those
|
|
|
|
|
|
// survivors. Item order in the buffer == draw order on screen.
|
2026-05-02 21:08:20 +02:00
|
|
|
|
layout(push_constant) uniform PC {
|
|
|
|
|
|
UIDispatchHeader hdr;
|
|
|
|
|
|
} pc;
|
|
|
|
|
|
|
|
|
|
|
|
layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in;
|
|
|
|
|
|
|
2026-06-16 15:45:12 +00:00
|
|
|
|
// Per-chunk cooperative-cull scratch (one slot per workgroup thread).
|
|
|
|
|
|
shared vec4 s_rect[UI_CHUNK];
|
|
|
|
|
|
shared vec4 s_color[UI_CHUNK];
|
|
|
|
|
|
shared vec4 s_corners[UI_CHUNK];
|
|
|
|
|
|
shared vec4 s_outline[UI_CHUNK];
|
|
|
|
|
|
shared uint s_keep[UI_CHUNK];
|
|
|
|
|
|
shared uint s_order[UI_CHUNK];
|
|
|
|
|
|
shared uint s_count;
|
2026-06-18 13:26:37 +00:00
|
|
|
|
shared uint s_subTotals[UI_CHUNK]; // per-subgroup survivor totals (carry step)
|
|
|
|
|
|
|
|
|
|
|
|
// Stable in-order compaction of one chunk's survivors via a per-subgroup
|
|
|
|
|
|
// exclusive prefix-sum plus a carry across subgroups, so a survivor's slot in
|
|
|
|
|
|
// s_order[] equals the number of survivors with a smaller local index — buffer
|
|
|
|
|
|
// (draw) order preserved. Replaces the old lane-0 serial scan. See
|
|
|
|
|
|
// ui-fused.comp.glsl for the full rationale; the body is identical.
|
|
|
|
|
|
void uiCompactChunk() {
|
|
|
|
|
|
uint lid = gl_LocalInvocationIndex;
|
|
|
|
|
|
uint keep = s_keep[lid];
|
|
|
|
|
|
uint subPrefix = subgroupExclusiveAdd(keep);
|
|
|
|
|
|
uint subTotal = subgroupAdd(keep);
|
|
|
|
|
|
if (subgroupElect())
|
|
|
|
|
|
s_subTotals[gl_SubgroupID] = subTotal;
|
|
|
|
|
|
barrier();
|
|
|
|
|
|
|
|
|
|
|
|
uint base = 0u;
|
|
|
|
|
|
uint total = 0u;
|
|
|
|
|
|
for (uint s = 0u; s < gl_NumSubgroups; ++s) {
|
|
|
|
|
|
uint t = s_subTotals[s];
|
|
|
|
|
|
if (s < gl_SubgroupID) base += t;
|
|
|
|
|
|
total += t;
|
|
|
|
|
|
}
|
|
|
|
|
|
if (lid == 0u) s_count = total;
|
|
|
|
|
|
|
|
|
|
|
|
if (keep != 0u)
|
|
|
|
|
|
s_order[base + subPrefix] = lid;
|
|
|
|
|
|
}
|
2026-06-16 15:45:12 +00:00
|
|
|
|
|
2026-05-02 21:08:20 +02:00
|
|
|
|
void main() {
|
2026-06-16 15:45:12 +00:00
|
|
|
|
// NOTE: do not early-return — every thread must reach the barriers below.
|
2026-05-02 21:08:20 +02:00
|
|
|
|
ivec2 screenPx;
|
2026-06-16 15:45:12 +00:00
|
|
|
|
bool valid = uiResolveScreenPixel(pc.hdr, screenPx);
|
2026-05-02 21:08:20 +02:00
|
|
|
|
|
2026-06-18 13:26:32 +00:00
|
|
|
|
// Defer the destination read-modify-write: a sparse UI leaves most tiles
|
|
|
|
|
|
// untouched, so only load the pixel when the first surviving item blends
|
|
|
|
|
|
// over it (`loaded`), and only store when something actually touched it.
|
|
|
|
|
|
// The fused kernel amortizes a single load/store across all categories; the
|
|
|
|
|
|
// standalone Dispatch* path has no such umbrella and otherwise pays a full
|
|
|
|
|
|
// read-modify-write per empty tile.
|
|
|
|
|
|
vec4 dst = vec4(0.0);
|
|
|
|
|
|
vec2 sp = vec2(0.0);
|
|
|
|
|
|
bool loaded = false;
|
|
|
|
|
|
if (valid) sp = vec2(screenPx) + 0.5;
|
2026-05-02 21:08:20 +02:00
|
|
|
|
|
2026-06-16 15:45:12 +00:00
|
|
|
|
vec2 tileMin, tileMax;
|
|
|
|
|
|
uiTileBounds(tileMin, tileMax);
|
|
|
|
|
|
|
|
|
|
|
|
uint lid = gl_LocalInvocationIndex;
|
|
|
|
|
|
for (uint base = 0u; base < pc.hdr.itemCount; base += UI_CHUNK) {
|
|
|
|
|
|
// Cooperative load: each thread reads one item member-by-member
|
|
|
|
|
|
// (the NVIDIA descriptor-heap workaround) into shared and culls it.
|
|
|
|
|
|
uint idx = base + lid;
|
|
|
|
|
|
bool keep = false;
|
|
|
|
|
|
if (idx < pc.hdr.itemCount) {
|
|
|
|
|
|
s_rect[lid] = uiQuadHeap[pc.hdr.itemBuffer].items[idx].rect;
|
|
|
|
|
|
s_color[lid] = uiQuadHeap[pc.hdr.itemBuffer].items[idx].color;
|
|
|
|
|
|
s_corners[lid] = uiQuadHeap[pc.hdr.itemBuffer].items[idx].corners;
|
|
|
|
|
|
s_outline[lid] = uiQuadHeap[pc.hdr.itemBuffer].items[idx].outline;
|
|
|
|
|
|
keep = uiAabbOverlapsTile(s_rect[lid].xy, s_rect[lid].xy + s_rect[lid].zw,
|
|
|
|
|
|
tileMin, tileMax);
|
|
|
|
|
|
}
|
|
|
|
|
|
s_keep[lid] = keep ? 1u : 0u;
|
|
|
|
|
|
barrier();
|
|
|
|
|
|
|
2026-06-18 13:26:37 +00:00
|
|
|
|
uiCompactChunk();
|
2026-06-16 15:45:12 +00:00
|
|
|
|
barrier();
|
|
|
|
|
|
|
|
|
|
|
|
if (valid) {
|
|
|
|
|
|
for (uint j = 0u; j < s_count; ++j) {
|
|
|
|
|
|
uint c = s_order[j];
|
|
|
|
|
|
|
|
|
|
|
|
// Cheap pre-test against the item's axis-aligned rect.
|
|
|
|
|
|
vec2 lo = s_rect[c].xy;
|
|
|
|
|
|
vec2 hi = s_rect[c].xy + s_rect[c].zw;
|
|
|
|
|
|
if (sp.x < lo.x || sp.y < lo.y) continue;
|
|
|
|
|
|
if (sp.x >= hi.x || sp.y >= hi.y) continue;
|
|
|
|
|
|
|
|
|
|
|
|
vec2 halfSize = s_rect[c].zw * 0.5;
|
|
|
|
|
|
vec2 p = sp - (s_rect[c].xy + halfSize);
|
|
|
|
|
|
float d = uiSdRoundRect(p, halfSize, s_corners[c]);
|
|
|
|
|
|
|
|
|
|
|
|
vec4 outline = s_outline[c];
|
|
|
|
|
|
float bodyA = clamp(0.5 - d, 0.0, 1.0);
|
|
|
|
|
|
if (bodyA <= 0.0 && outline.x <= 0.0) continue;
|
|
|
|
|
|
|
|
|
|
|
|
vec4 col = s_color[c];
|
|
|
|
|
|
vec4 src = vec4(col.rgb, col.a * bodyA);
|
|
|
|
|
|
|
|
|
|
|
|
if (outline.x > 0.0) {
|
|
|
|
|
|
float t = abs(d + outline.x * 0.5) - outline.x * 0.5;
|
|
|
|
|
|
float outlineA = clamp(0.5 - t, 0.0, 1.0);
|
|
|
|
|
|
src.rgb = mix(src.rgb, outline.yzw, outlineA);
|
|
|
|
|
|
src.a = max(src.a, outlineA);
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
if (src.a <= 0.0) continue;
|
2026-06-18 13:26:32 +00:00
|
|
|
|
if (!loaded) {
|
|
|
|
|
|
dst = imageLoad(uiImages[pc.hdr.outImage], screenPx);
|
|
|
|
|
|
loaded = true;
|
|
|
|
|
|
}
|
2026-06-16 15:45:12 +00:00
|
|
|
|
dst = uiBlendOver(dst, src);
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
barrier(); // done reading shared for this chunk before it's overwritten
|
|
|
|
|
|
}
|
|
|
|
|
|
|
2026-06-18 13:26:32 +00:00
|
|
|
|
if (loaded) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); // loaded ⇒ valid
|
2026-05-02 21:08:20 +02:00
|
|
|
|
}
|