diff --git a/shaders/ui-circles.comp.glsl b/shaders/ui-circles.comp.glsl index eea0c93..b5cb547 100644 --- a/shaders/ui-circles.comp.glsl +++ b/shaders/ui-circles.comp.glsl @@ -23,12 +23,16 @@ void main() { ivec2 screenPx; bool valid = uiResolveScreenPixel(pc.hdr, screenPx); - vec4 dst = vec4(0.0); - vec2 sp = vec2(0.0); - if (valid) { - dst = imageLoad(uiImages[pc.hdr.outImage], screenPx); - sp = vec2(screenPx) + 0.5; - } + // Defer the destination read-modify-write: a sparse UI leaves most tiles + // untouched, so only load the pixel when the first surviving item blends + // over it (`loaded`), and only store when something actually touched it. + // The fused kernel amortizes a single load/store across all categories; the + // standalone Dispatch* path has no such umbrella and otherwise pays a full + // read-modify-write per empty tile. + vec4 dst = vec4(0.0); + vec2 sp = vec2(0.0); + bool loaded = false; + if (valid) sp = vec2(screenPx) + 0.5; vec2 tileMin, tileMax; uiTileBounds(tileMin, tileMax); @@ -90,11 +94,15 @@ void main() { } if (src.a <= 0.0) continue; + if (!loaded) { + dst = imageLoad(uiImages[pc.hdr.outImage], screenPx); + loaded = true; + } dst = uiBlendOver(dst, src); } } barrier(); } - if (valid) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); + if (loaded) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); // loaded ⇒ valid } diff --git a/shaders/ui-images.comp.glsl b/shaders/ui-images.comp.glsl index a4424a6..bafb2f5 100644 --- a/shaders/ui-images.comp.glsl +++ b/shaders/ui-images.comp.glsl @@ -24,12 +24,16 @@ void main() { ivec2 screenPx; bool valid = uiResolveScreenPixel(pc.hdr, screenPx); - vec4 dst = vec4(0.0); - vec2 sp = vec2(0.0); - if (valid) { - dst = imageLoad(uiImages[pc.hdr.outImage], screenPx); - sp = vec2(screenPx) + 0.5; - } + // Defer the destination read-modify-write: a sparse UI leaves most tiles + // untouched, so only load the pixel when the first surviving item blends + // over it (`loaded`), and only store when something actually touched it. + // The fused kernel amortizes a single load/store across all categories; the + // standalone Dispatch* path has no such umbrella and otherwise pays a full + // read-modify-write per empty tile. + vec4 dst = vec4(0.0); + vec2 sp = vec2(0.0); + bool loaded = false; + if (valid) sp = vec2(screenPx) + 0.5; vec2 tileMin, tileMax; uiTileBounds(tileMin, tileMax); @@ -80,11 +84,15 @@ void main() { ); vec4 src = sampled * s_tint[c]; if (src.a <= 0.0) continue; + if (!loaded) { + dst = imageLoad(uiImages[pc.hdr.outImage], screenPx); + loaded = true; + } dst = uiBlendOver(dst, src); } } barrier(); } - if (valid) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); + if (loaded) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); // loaded ⇒ valid } diff --git a/shaders/ui-quads.comp.glsl b/shaders/ui-quads.comp.glsl index 24e7daf..6bba7c4 100644 --- a/shaders/ui-quads.comp.glsl +++ b/shaders/ui-quads.comp.glsl @@ -27,12 +27,16 @@ void main() { ivec2 screenPx; bool valid = uiResolveScreenPixel(pc.hdr, screenPx); - vec4 dst = vec4(0.0); - vec2 sp = vec2(0.0); - if (valid) { - dst = imageLoad(uiImages[pc.hdr.outImage], screenPx); - sp = vec2(screenPx) + 0.5; - } + // Defer the destination read-modify-write: a sparse UI leaves most tiles + // untouched, so only load the pixel when the first surviving item blends + // over it (`loaded`), and only store when something actually touched it. + // The fused kernel amortizes a single load/store across all categories; the + // standalone Dispatch* path has no such umbrella and otherwise pays a full + // read-modify-write per empty tile. + vec4 dst = vec4(0.0); + vec2 sp = vec2(0.0); + bool loaded = false; + if (valid) sp = vec2(screenPx) + 0.5; vec2 tileMin, tileMax; uiTileBounds(tileMin, tileMax); @@ -93,11 +97,15 @@ void main() { } if (src.a <= 0.0) continue; + if (!loaded) { + dst = imageLoad(uiImages[pc.hdr.outImage], screenPx); + loaded = true; + } dst = uiBlendOver(dst, src); } } barrier(); // done reading shared for this chunk before it's overwritten } - if (valid) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); + if (loaded) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); // loaded ⇒ valid } diff --git a/shaders/ui-text.comp.glsl b/shaders/ui-text.comp.glsl index a64abbf..b34a485 100644 --- a/shaders/ui-text.comp.glsl +++ b/shaders/ui-text.comp.glsl @@ -31,12 +31,16 @@ void main() { ivec2 screenPx; bool valid = uiResolveScreenPixel(pc.hdr, screenPx); - vec4 dst = vec4(0.0); - vec2 sp = vec2(0.0); - if (valid) { - dst = imageLoad(uiImages[pc.hdr.outImage], screenPx); - sp = vec2(screenPx) + 0.5; - } + // Defer the destination read-modify-write: a sparse UI leaves most tiles + // untouched, so only load the pixel when the first surviving glyph blends + // over it (`loaded`), and only store when something actually touched it. + // The fused kernel amortizes a single load/store across all categories; the + // standalone Dispatch* path has no such umbrella and otherwise pays a full + // read-modify-write per empty tile. + vec4 dst = vec4(0.0); + vec2 sp = vec2(0.0); + bool loaded = false; + if (valid) sp = vec2(screenPx) + 0.5; vec2 tileMin, tileMax; uiTileBounds(tileMin, tileMax); @@ -102,11 +106,15 @@ void main() { vec4 col = s_color[c]; vec4 src = vec4(col.rgb, col.a * a); + if (!loaded) { + dst = imageLoad(uiImages[pc.hdr.outImage], screenPx); + loaded = true; + } dst = uiBlendOver(dst, src); } } barrier(); } - if (valid) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); + if (loaded) imageStore(uiImages[pc.hdr.outImage], screenPx, dst); // loaded ⇒ valid }