feat(webgpu-rt): atomic pixel accumulator + raysPerPixel — >1 ray per pixel per bounce (#30) #31

Merged
catbot merged 2 commits from claude/issue-30 into master 2026-06-10 00:46:24 +02:00
7 changed files with 406 additions and 0 deletions
Showing only changes of commit f7fc441253 - Show all commits

test(webgpu-rt): RTMultiShadow example exercising N shadow rays/pixel/bounce (#30)

Five pillars, four colored point lights; closest-hit emits one shadow ray
per light from the same invocation (raysPerPixel = 4, maxDepth = 2), so up
to four rays per pixel rtAccumulate in a single SHADE pass — the contention
case #30 exists for. Each pillar casts four separable colored shadows; an
accumulator race or capacity drop shows as flickering dark noise or a
missing shadow color. Two frames a second apart diff at 2 px / 1.85 M
(last-ulp CAS ordering), confirming no lost updates.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
catbot 2026-06-09 22:45:33 +00:00

View file

@ -106,3 +106,15 @@ barrier WebGPU only provides between submits (or between passes), never
within a single compute pass. WebGPU/DOM only; the same chain is wireable
on Vulkan today via an offscreen HDR heap image + a composite `RenderPass`
(the present path records passes generically and barriers between them).
### [RTMultiShadow](RTMultiShadow/)
Multi-light shadowing through the wavefront RT pipeline (issue #30). Five
pillars on a checkered ground, four colored point lights; the closest-hit
emits one shadow ray **per light** from the same invocation, so up to four
rays per pixel resolve in a single SHADE pass. Exercises both halves of
the >1-ray-per-pixel-per-bounce widening: the atomic `rtAccumulate`
(per-channel f32 CAS — concurrent same-pixel adds don't race) and
`RTPass::raysPerPixel` (scales the ray/hit/payload buffers so the
per-light emits aren't capacity-dropped). Any regression shows up as
flickering dark noise or a missing shadow color in the overlap regions.
WebGPU/DOM only.

View file

@ -0,0 +1,82 @@
// RTMultiShadow closest-hit (runs in SHADE). The multi-light counterpart
// of RTStress: EVERY light emits its own shadow ray from this single
// invocation, so several rays for the same pixel resolve in the next SHADE
// pass exactly the contention the atomic rtAccumulate exists for (#30).
// Before the atomic accumulator their rtAccumulate calls raced (lost
// updates flickering dark noise); before RTPass::raysPerPixel the extra
// rays were silently dropped by the capacity guard. The host sets
// raysPerPixel = LIGHT_COUNT so every emit fits the bounce.
//
// Payload declared here so the assembler sees it before wfPayload / SHADE.
struct Payload {
color: vec3<f32>, // shadow ray: pending direct contribution
shadowRay: u32, // 0 primary, 1 shadow
};
// Point lights, color premultiplied with intensity; 1/d² falloff at shade
// time. Four distinct hues so each occluder casts four separable shadows
// any accumulator race or dropped shadow ray is immediately visible as
// noise / a missing color in the overlap regions.
const LIGHT_COUNT: u32 = 4u;
struct Light {
pos: vec3<f32>,
color: vec3<f32>,
};
var<private> LIGHTS: array<Light, 4> = array<Light, 4>(
Light(vec3<f32>( 14.0, 9.0, 2.0), vec3<f32>(250.0, 205.0, 140.0)), // warm white
Light(vec3<f32>(-13.0, 8.0, 7.0), vec3<f32>(235.0, 45.0, 30.0)), // red
Light(vec3<f32>( 3.0, 8.0, -14.0), vec3<f32>( 55.0, 225.0, 105.0)), // green
Light(vec3<f32>( -5.0, 10.0, 13.0), vec3<f32>( 65.0, 105.0, 250.0)), // blue
);
const AMBIENT_COLOR: vec3<f32> = vec3<f32>(0.030, 0.034, 0.045);
// Ground (customIndex 0) is a subtle checker so the colored shadows read;
// pillars hash their instance index like RTStress.
fn surfaceAlbedo(customIndex: u32, worldPos: vec3<f32>) -> vec3<f32> {
if (customIndex == 0u) {
let cx = u32(floor(worldPos.x * 0.25 + 100.0));
let cz = u32(floor(worldPos.z * 0.25 + 100.0));
return mix(vec3<f32>(0.60), vec3<f32>(0.76), f32((cx + cz) & 1u));
}
let h = customIndex * 2654435761u;
return vec3<f32>(
0.45 + 0.5 * f32((h >> 0u) & 255u) / 255.0,
0.45 + 0.5 * f32((h >> 8u) & 255u) / 255.0,
0.45 + 0.5 * f32((h >> 16u) & 255u) / 255.0);
}
fn closesthit_main(ray: RayDesc, hit: HitInfo, payload: ptr<function, Payload>) {
let meshRec = meshRecords[tlasEntries[hit.instanceId].blasMeshIdx];
let verts = _rtFetchTri(meshRec, hit.primitiveId);
let nObj = normalize(cross(verts[1] - verts[0], verts[2] - verts[0]));
let nWorld = normalize(vec3<f32>(
dot(hit.objectToWorldR0.xyz, nObj),
dot(hit.objectToWorldR1.xyz, nObj),
dot(hit.objectToWorldR2.xyz, nObj)));
let worldPos = ray.origin + ray.direction * hit.t;
let nFacing = select(-nWorld, nWorld, dot(nWorld, -ray.direction) > 0.0);
let albedo = surfaceAlbedo(hit.customIndex, worldPos);
rtAccumulate(albedo * AMBIENT_COLOR);
// One shadow ray PER LIGHT from this one closest-hit invocation. All of
// them carry the same pixel; the ones that miss (light visible) each
// rtAccumulate their light's contribution in the same SHADE pass.
let shadowOrigin = worldPos + nFacing * 0.05;
for (var i: u32 = 0u; i < LIGHT_COUNT; i = i + 1u) {
let toLight = LIGHTS[i].pos - shadowOrigin;
let dist = length(toLight);
let dir = toLight / dist;
let nDotL = dot(nFacing, dir);
if (nDotL <= 0.0) { continue; }
var sp: Payload;
sp.color = albedo * LIGHTS[i].color * (nDotL / (dist * dist));
sp.shadowRay = 1u;
// tMax stops at the light so geometry beyond it can't occlude.
rtEmitRay(shadowOrigin, 0.01, dir, dist,
RT_FLAG_SKIP_CLOSEST_HIT | RT_FLAG_TERMINATE_ON_FIRST_HIT,
0xFFu, 0u, 0u, sp);
}
}

View file

@ -0,0 +1,210 @@
// RTMultiShadow — multi-light shadowing through the wavefront RT pipeline
// (issue #30). Five pillars on a checkered ground, lit by four colored
// point lights; the closest-hit emits one shadow ray PER LIGHT from the
// same invocation, so up to four rays per pixel resolve in a single SHADE
// pass. That requires both halves of #30:
// - atomic rtAccumulate — the concurrent per-pixel adds don't race;
// - RTPass::raysPerPixel — the ray/hit/payload buffers hold 4·W·H rays
// per bounce, so none of the per-light emits get capacity-dropped.
// Each pillar casting four differently-colored shadows is the visual
// proof; any lost accumulate (race) or dropped ray (capacity) shows up as
// flickering dark noise / a missing shadow color.
//
// WebGPU/DOM only — the wavefront tracer is the WebGPU software RT path.
#ifndef CRAFTER_GRAPHICS_WINDOW_DOM
int main() { return 0; } // native path is hardware RT; out of scope here
#else
import Crafter.Graphics;
import Crafter.Math;
import Crafter.Event;
import std;
using namespace Crafter;
namespace fs = std::filesystem;
namespace {
// Must match LIGHT_COUNT in closesthit.wgsl — it is the raysPerPixel
// budget the RTPass is configured with.
constexpr std::uint32_t kLightCount = 4;
struct CameraGPU {
float origin[3]; float pad0;
float right[3]; float tanHalf;
float up[3]; float aspect;
float forward[3]; float pad1;
};
static_assert(sizeof(CameraGPU) == 64);
// Axis-aligned box: 8 corners between mn and mx, same winding as the
// RTStress unit cube.
std::array<Vector<float, 3, 3>, 8> BoxVerts(float mnx, float mny, float mnz,
float mxx, float mxy, float mxz) {
return {{
{mnx, mny, mnz}, {mxx, mny, mnz}, {mxx, mxy, mnz}, {mnx, mxy, mnz},
{mnx, mny, mxz}, {mxx, mny, mxz}, {mxx, mxy, mxz}, {mnx, mxy, mxz},
}};
}
// Mesh::Build takes mutable spans, so this can't be constexpr.
std::array<std::uint32_t, 36> kBoxIndices {{
0,1,2, 0,2,3, 5,4,7, 5,7,6, 4,0,3, 4,3,7,
1,5,6, 1,6,2, 4,5,1, 4,1,0, 3,2,6, 3,6,7,
}};
}
int main() {
std::println("[RTMultiShadow] {} lights, one shadow ray per light per pixel", kLightCount);
Device::Initialize();
static Window window(1280, 720, "RTMultiShadow");
auto cmd = window.StartInit();
DescriptorHeapWebGPU heap;
heap.Initialize(/*images*/ 1, /*buffers*/ 2, /*samplers*/ 1);
std::array<WebGPUShader, 4> shaders {{
WebGPUShader(fs::path("raygen.wgsl"), "raygen_main", WebGPURTStage::Raygen),
WebGPUShader(fs::path("miss.wgsl"), "miss_main", WebGPURTStage::Miss),
WebGPUShader(fs::path("closesthit.wgsl"), "closesthit_main", WebGPURTStage::ClosestHit),
WebGPUShader(fs::path("resolve.wgsl"), "resolve_main", WebGPURTStage::Resolve),
}};
ShaderBindingTableWebGPU sbt;
sbt.Init(shaders);
std::array<RTShaderGroup, 1> raygenGroups {{ { .type = RTShaderGroupType::General, .generalShader = 0 } }};
std::array<RTShaderGroup, 1> missGroups {{ { .type = RTShaderGroupType::General, .generalShader = 1 } }};
std::array<RTShaderGroup, 1> hitGroups {{ { .type = RTShaderGroupType::TrianglesHitGroup, .closestHitShader = 2 } }};
// One user binding: the camera storage buffer at @group(3).
std::array<UICustomBinding, 1> bindings {{
{ .group = 3, .binding = 0, .kind = UICustomBindingKind::Buffer, .pushOffset = 0 },
}};
PipelineRTWebGPU pipeline;
pipeline.Init(cmd, raygenGroups, missGroups, hitGroups, sbt, bindings);
// ── Meshes: a large ground slab and a pillar (origin at its base). ──
static auto groundVerts = BoxVerts(-30.0f, -1.0f, -30.0f, 30.0f, 0.0f, 30.0f);
static auto pillarVerts = BoxVerts(-0.8f, 0.0f, -0.8f, 0.8f, 6.0f, 0.8f);
static Mesh ground, pillar;
ground.Build(groundVerts, kBoxIndices, cmd);
pillar.Build(pillarVerts, kBoxIndices, cmd);
// ── Camera buffer + handle array. ─────────────────────────────────
WebGPUBuffer<CameraGPU, true> cameraBuf;
cameraBuf.Create(1);
static std::array<std::uint32_t, 1> userHandles { cameraBuf.handle };
// ── Instances: ground (customIndex 0) + five pillars. ─────────────
struct Placement { float x, z; };
static constexpr std::array<Placement, 5> kPillars {{
{ 0.0f, 0.0f }, { 5.0f, 5.0f }, { -5.0f, 5.0f }, { 5.0f, -5.0f }, { -5.0f, -5.0f },
}};
static std::vector<RenderingElement3D> renderers;
renderers.reserve(1 + kPillars.size());
auto addInstance = [&](std::uint64_t blasAddr, float x, float z) {
renderers.emplace_back();
RenderingElement3D& r = renderers.back();
auto& tx = r.instance.transform.matrix;
tx[0][0] = 1; tx[0][1] = 0; tx[0][2] = 0; tx[0][3] = x;
tx[1][0] = 0; tx[1][1] = 1; tx[1][2] = 0; tx[1][3] = 0;
tx[2][0] = 0; tx[2][1] = 0; tx[2][2] = 1; tx[2][3] = z;
r.instance.instanceCustomIndex = static_cast<std::uint32_t>(renderers.size() - 1);
r.instance.mask = 0xFF;
r.instance.instanceShaderBindingTableRecordOffset = 0;
r.instance.flags = kRTGeometryInstanceForceOpaque;
r.instance.accelerationStructureReference = blasAddr;
RenderingElement3D::Add(&r);
};
addInstance(ground.blasAddr, 0.0f, 0.0f);
for (const auto& p : kPillars) addInstance(pillar.blasAddr, p.x, p.z);
RenderingElement3D::BuildTLAS(cmd, 0);
window.descriptorHeap = &heap;
window.FinishInit();
RTPass rtPass(&pipeline);
rtPass.handlesPtr = userHandles.data();
rtPass.handlesCount = static_cast<std::uint32_t>(userHandles.size());
rtPass.maxDepth = 2; // primary + shadow
rtPass.raysPerPixel = kLightCount; // one shadow ray per light per pixel
window.passes.push_back(&rtPass);
// ── Free camera framing the pillars from above one corner. ────────
struct CamState {
Vector<float, 3, 4> position;
float yaw;
float pitch;
} cam {
Vector<float, 3, 4>{ 16.0f, 13.0f, 16.0f },
0.0f, 0.0f,
};
{
// Aim at the scene centre, slightly above the ground.
Vector<float, 3, 4> d { -cam.position.x, 2.0f - cam.position.y, -cam.position.z };
const float len = std::sqrt(d.x*d.x + d.y*d.y + d.z*d.z);
cam.yaw = std::atan2(d.z, d.x);
cam.pitch = std::asin(d.y / len);
}
Input::Map inputMap;
Input::Action& moveAct = inputMap.AddAction("Move", Input::ActionType::Vector2);
Input::Action& lookAct = inputMap.AddAction("Look", Input::ActionType::Vector2);
moveAct.bindings = { Input::WASDBind{
Key(CrafterKeys::W), Key(CrafterKeys::S), Key(CrafterKeys::A), Key(CrafterKeys::D) } };
lookAct.bindings = { Input::MouseDeltaBind{ 1.0f } };
inputMap.Attach(window);
const float kMoveSpeed = 14.0f;
const float kLookSens = 0.05f;
const float kDt = 1.0f / 60.0f;
static int frames = 0;
EventListener<void> camTick(&window.onBeforeUpdate, [&]() {
inputMap.Tick();
cam.yaw += lookAct.vector2.x * kLookSens;
cam.pitch -= lookAct.vector2.y * kLookSens;
cam.pitch = std::clamp(cam.pitch, -1.55f, 1.55f);
const float cp = std::cos(cam.pitch), sp = std::sin(cam.pitch);
const float cy = std::cos(cam.yaw), sy = std::sin(cam.yaw);
Vector<float, 3, 4> forward { cp * cy, sp, cp * sy };
Vector<float, 3, 4> worldUp { 0.0f, 1.0f, 0.0f };
Vector<float, 3, 4> right { forward.y*worldUp.z - forward.z*worldUp.y,
forward.z*worldUp.x - forward.x*worldUp.z,
forward.x*worldUp.y - forward.y*worldUp.x };
const float rLen = std::sqrt(right.x*right.x + right.y*right.y + right.z*right.z);
right.x /= rLen; right.y /= rLen; right.z /= rLen;
Vector<float, 3, 4> up { right.y*forward.z - right.z*forward.y,
right.z*forward.x - right.x*forward.z,
right.x*forward.y - right.y*forward.x };
const float dx = moveAct.vector2.x * kMoveSpeed * kDt;
const float dy = moveAct.vector2.y * kMoveSpeed * kDt;
cam.position.x += right.x*dx + forward.x*dy;
cam.position.y += right.y*dx + forward.y*dy;
cam.position.z += right.z*dx + forward.z*dy;
CameraGPU& g = cameraBuf.value[0];
g.origin[0]=cam.position.x; g.origin[1]=cam.position.y; g.origin[2]=cam.position.z; g.pad0=0;
g.right[0]=right.x; g.right[1]=right.y; g.right[2]=right.z;
g.up[0]=up.x; g.up[1]=up.y; g.up[2]=up.z;
g.forward[0]=forward.x; g.forward[1]=forward.y; g.forward[2]=forward.z;
g.aspect = float(window.width) / float(window.height);
g.tanHalf = std::tan(70.0f * 3.14159265f / 360.0f);
g.pad1 = 0;
cameraBuf.FlushDevice();
if (++frames >= 60) {
std::println("[RTMultiShadow] {} lights x {} pillars rendering", kLightCount, kPillars.size());
frames = 0;
}
});
window.Render();
window.StartUpdate();
window.StartSync();
return 0;
}
#endif

View file

@ -0,0 +1,14 @@
// RTMultiShadow miss (runs in SHADE). Shadow miss that light is visible
// from the surface, so add its pending contribution; up to LIGHT_COUNT of
// these resolve for the same pixel in one pass (atomic rtAccumulate, #30).
// Primary miss near-black night sky so the colored lighting carries the
// frame.
fn miss_main(ray: RayDesc, payload: ptr<function, Payload>) {
if ((*payload).shadowRay == 1u) {
rtAccumulate((*payload).color);
return;
}
let t = clamp(ray.direction.y * 0.5 + 0.5, 0.0, 1.0);
rtAccumulate(mix(vec3<f32>(0.010, 0.012, 0.022),
vec3<f32>(0.030, 0.040, 0.075), t));
}

View file

@ -0,0 +1,46 @@
import std;
import Crafter.Build;
namespace fs = std::filesystem;
using namespace Crafter;
extern "C" Configuration CrafterBuildProject(std::span<const std::string_view> args) {
bool isWasm = false;
for (std::string_view a : args) {
if (a.starts_with("--target=") && a.find("wasm") != std::string_view::npos) {
isWasm = true;
break;
}
}
std::vector<std::string> graphicsArgs(args.begin(), args.end());
Configuration* graphics = LocalProject({
.projectFile = "../../project.cpp",
.args = graphicsArgs,
});
Configuration cfg;
cfg.path = "./";
cfg.name = "RTMultiShadow";
cfg.outputName = "RTMultiShadow";
cfg.type = ConfigurationType::Executable;
if (isWasm) {
cfg.target = "wasm32-wasip1";
cfg.defines.push_back({"CRAFTER_GRAPHICS_WINDOW_DOM", ""});
cfg.compileFlags.push_back("-msimd128");
}
ApplyStandardArgs(cfg, args);
cfg.dependencies = { graphics };
std::array<fs::path, 0> ifaces = {};
std::array<fs::path, 1> impls = { "main" };
cfg.GetInterfacesAndImplementations(ifaces, impls);
if (isWasm) {
cfg.files.emplace_back(fs::path("raygen.wgsl"));
cfg.files.emplace_back(fs::path("closesthit.wgsl"));
cfg.files.emplace_back(fs::path("miss.wgsl"));
cfg.files.emplace_back(fs::path("resolve.wgsl"));
EnableWasiBrowserRuntime(cfg);
}
return cfg;
}

View file

@ -0,0 +1,35 @@
// RTMultiShadow raygen (runs in GENERATE). Host-driven pinhole camera at
// @group(3) (groups 0..2 are reserved by the wavefront pipeline:
// 0 = WfParams, 1 = data heaps, 2 = indirect args).
struct Camera {
origin: vec3<f32>,
pad0: f32,
right: vec3<f32>,
tanHalf: f32,
up: vec3<f32>,
aspect: f32,
forward: vec3<f32>,
pad1: f32,
};
@group(3) @binding(0) var<storage, read> camera : Camera;
fn raygen_main(gid: vec3<u32>) {
if (gid.x >= wfParams.surfaceW || gid.y >= wfParams.surfaceH) { return; }
let pixelf = vec2<f32>(f32(gid.x), f32(gid.y));
let res = vec2<f32>(f32(wfParams.surfaceW), f32(wfParams.surfaceH));
let uv = (pixelf + vec2<f32>(0.5)) / res;
let ndc = uv * 2.0 - vec2<f32>(1.0);
let direction = normalize(
camera.right * (ndc.x * camera.aspect * camera.tanHalf) +
camera.up * (-ndc.y * camera.tanHalf) +
camera.forward);
var p: Payload;
p.color = vec3<f32>(0.0);
p.shadowRay = 0u;
rtEmitPrimaryRay(camera.origin, 0.01, direction, 100000.0,
0u, 0xFFu, 0u, 0u, p);
}

View file

@ -0,0 +1,7 @@
// RTMultiShadow RESOLVE-stage tonemap: Reinhard + gamma 2.2 over the
// linear accumulator. Registered as a WebGPURTStage::Resolve shader.
fn resolve_main(coord: vec2<u32>, hdr: vec4<f32>) -> vec4<f32> {
let mapped = hdr.rgb / (hdr.rgb + vec3<f32>(1.0));
let g = pow(mapped, vec3<f32>(1.0 / 2.2));
return vec4<f32>(g, 1.0);
}