GPU-accelerated UI toolkit (Vulkan)
git clone https://git.lucas.co/cce-ui.git
src/web/rt.rs (20.9K)
1 //! The path tracer on WebGPU: the Vulkan `RtStage`'s compute tier, ported.
2 //! The same shaders (`draw::shaders::rt_bvh_source`, `RT_DENOISE`), the
3 //! same scene packing, BVH and parameter blocks (`draw::rt`), the same
4 //! rules for when the accumulation restarts — a camera, pane size, scene,
5 //! background or environment change — and the same frame: one sample a
6 //! frame added into the running mean, three à-trous denoise iterations over
7 //! it, the result put into the backdrop's pane region in place of the
8 //! raster scene, a pane that moved clearing the backdrop once first.
9 //!
10 //! What WebGPU makes different:
11 //!
12 //! - **The compute tier only.** The hardware ray-query tier needs
13 //! `VK_KHR_ray_query`; WebGPU has no ray tracing. A BVH traversed in
14 //! compute is the tier every Vulkan device without RT cores runs too.
15 //! - **The blit is a draw.** Vulkan blits the tracer's `rgba8unorm` image
16 //! into the sRGB backdrop, converting as it copies; WebGPU copies only
17 //! between formats that differ in sRGB-ness at most, so a small render
18 //! pass loads each texel and writes it through the backdrop's sRGB view —
19 //! the same conversion (the unorm value taken as linear and encoded).
20 //! - **No frames in flight to juggle.** One parameter buffer and one bind
21 //! group per dispatch, written before the frame's single submission.
22
23 use wasm_bindgen::JsValue;
24 use web_sys::{
25 gpu_buffer_usage as buffer_usage, gpu_shader_stage as shader_stage, gpu_texture_usage as texture_usage, GpuBindGroup,
26 GpuBindGroupDescriptor, GpuBindGroupEntry, GpuBindGroupLayout, GpuBindGroupLayoutDescriptor, GpuBindGroupLayoutEntry,
27 GpuBuffer, GpuBufferBinding, GpuBufferBindingLayout, GpuBufferBindingType, GpuColorTargetState, GpuCommandEncoder,
28 GpuComputePassDescriptor, GpuComputePipeline, GpuComputePipelineDescriptor, GpuDevice, GpuFragmentState, GpuLoadOp,
29 GpuPipelineLayoutDescriptor, GpuProgrammableStage, GpuQueue, GpuRenderPassColorAttachment, GpuRenderPassDescriptor,
30 GpuRenderPipeline, GpuRenderPipelineDescriptor, GpuSampler, GpuSamplerBindingLayout, GpuSamplerBindingType,
31 GpuStorageTextureAccess, GpuStorageTextureBindingLayout, GpuStoreOp, GpuTexture, GpuTextureBindingLayout,
32 GpuTextureFormat, GpuTextureSampleType, GpuTextureView, GpuTextureViewDimension, GpuVertexState,
33 };
34
35 use super::renderer::{shader_module, texture, whole_view};
36 use crate::draw::rt::{
37 denoise_params, rt_params, DenoiseParams, ParamImage, PreparedRtScene, RtCamera, RtEnvironment, RtImage,
38 RtParams, DENOISE_ITERATIONS, MAX_SAMPLES, WORKGROUP,
39 };
40 use crate::draw::shaders::{rt_bvh_source, RT_DENOISE};
41
42 /// Each denoise iteration's block sits at a multiple of this.
43 const STRIDE: u32 = 256;
44
45 /// Loads the tracer's image at the fragment's place in the pane and writes
46 /// it through the backdrop's sRGB view: Vulkan's UNORM-to-sRGB blit.
47 const BLIT: &str = "
48 struct Blit { origin: vec2<i32>, _pad: vec2<i32> }
49 @group(0) @binding(0) var src: texture_2d<f32>;
50 @group(0) @binding(1) var<uniform> blit: Blit;
51 @vertex fn vs_main(@builtin(vertex_index) i: u32) -> @builtin(position) vec4<f32> {
52 let p = vec2<f32>(f32((i << 1u) & 2u), f32(i & 2u));
53 return vec4<f32>(p * 2.0 - 1.0, 0.0, 1.0);
54 }
55 @fragment fn fs_main(@builtin(position) pos: vec4<f32>) -> @location(0) vec4<f32> {
56 return textureLoad(src, vec2<i32>(pos.xy) - blit.origin, 0);
57 }";
58
59 /// The pane-sized targets: the running sum, the denoiser's features and
60 /// ping-pong, and the image the result is written to.
61 struct Targets {
62 accum: GpuBuffer,
63 features: GpuBuffer,
64 ping: GpuBuffer,
65 pong: GpuBuffer,
66 output: GpuTexture,
67 output_view: GpuTextureView,
68 /// Denoise iterations 0 and 2 (src pong, dst ping), and 1 (src ping, dst pong).
69 denoise_b: GpuBindGroup,
70 denoise_a: GpuBindGroup,
71 blit: GpuBindGroup,
72 size: (u32, u32),
73 }
74
75 /// The scene the buffers hold, and the image standing in it.
76 struct Scene {
77 nodes: GpuBuffer,
78 tris: GpuBuffer,
79 materials: GpuBuffer,
80 tri_count: u32,
81 image: Option<RtImage>,
82 }
83
84 pub(crate) struct WebRt {
85 trace: GpuComputePipeline,
86 trace_layout: GpuBindGroupLayout,
87 denoise: GpuComputePipeline,
88 denoise_layout: GpuBindGroupLayout,
89 blit: GpuRenderPipeline,
90 blit_layout: GpuBindGroupLayout,
91 params: GpuBuffer,
92 denoise_params: GpuBuffer,
93 blit_params: GpuBuffer,
94 stand_in: GpuTextureView,
95 sampler: GpuSampler,
96 scene: Option<Scene>,
97 targets: Option<Targets>,
98
99 pane: (u32, u32, u32, u32),
100 pane_moved: bool,
101 camera: Option<RtCamera>,
102 sample_index: u32,
103 spp: u32,
104 staged: bool,
105 background: Option<[f32; 3]>,
106 environment: RtEnvironment,
107 }
108
109 fn buffer_entry(binding: u32, ty: GpuBufferBindingType, dynamic: bool) -> GpuBindGroupLayoutEntry {
110 let entry = GpuBindGroupLayoutEntry::new(binding, shader_stage::COMPUTE);
111 let layout = GpuBufferBindingLayout::new();
112 layout.set_type(ty);
113 layout.set_has_dynamic_offset(dynamic);
114 entry.set_buffer(&layout);
115 entry
116 }
117
118 fn storage_texture_entry(binding: u32) -> GpuBindGroupLayoutEntry {
119 let entry = GpuBindGroupLayoutEntry::new(binding, shader_stage::COMPUTE);
120 let layout = GpuStorageTextureBindingLayout::new(GpuTextureFormat::Rgba8unorm);
121 layout.set_access(GpuStorageTextureAccess::WriteOnly);
122 layout.set_view_dimension(GpuTextureViewDimension::N2d);
123 entry.set_storage_texture(&layout);
124 entry
125 }
126
127 fn texture_entry(binding: u32, visibility: u32, sample: GpuTextureSampleType) -> GpuBindGroupLayoutEntry {
128 let entry = GpuBindGroupLayoutEntry::new(binding, visibility);
129 let layout = GpuTextureBindingLayout::new();
130 layout.set_sample_type(sample);
131 layout.set_view_dimension(GpuTextureViewDimension::N2d);
132 entry.set_texture(&layout);
133 entry
134 }
135
136 fn whole(binding: u32, buffer: &GpuBuffer) -> GpuBindGroupEntry {
137 GpuBindGroupEntry::new_with_gpu_buffer_binding(binding, &GpuBufferBinding::new(buffer))
138 }
139
140 fn sized(binding: u32, buffer: &GpuBuffer, size: u32) -> GpuBindGroupEntry {
141 let b = GpuBufferBinding::new(buffer);
142 b.set_size(size);
143 GpuBindGroupEntry::new_with_gpu_buffer_binding(binding, &b)
144 }
145
146 fn buffer(device: &GpuDevice, size: u32, usage: u32, label: &str) -> Result<GpuBuffer, JsValue> {
147 let desc = web_sys::GpuBufferDescriptor::new(size.max(16).next_multiple_of(4), usage);
148 desc.set_label(label);
149 device.create_buffer(&desc)
150 }
151
152 fn compute_pipeline(device: &GpuDevice, layout: &GpuBindGroupLayout, code: &str, entry: &str, label: &str) -> GpuComputePipeline {
153 let pipeline_layout = device.create_pipeline_layout(&GpuPipelineLayoutDescriptor::new(&[js_sys::JsOption::wrap(layout.clone())]));
154 let stage = GpuProgrammableStage::new(&shader_module(device, code, label));
155 stage.set_entry_point(entry);
156 let desc = GpuComputePipelineDescriptor::new(&pipeline_layout, &stage);
157 desc.set_label(label);
158 device.create_compute_pipeline(&desc)
159 }
160
161 impl WebRt {
162 /// The tracer's pipelines; `backdrop_format` is what the blit writes.
163 pub(crate) fn new(device: &GpuDevice, backdrop_format: GpuTextureFormat) -> Result<Self, JsValue> {
164 use GpuBufferBindingType::{ReadOnlyStorage, Storage, Uniform};
165 let sampler_entry = {
166 let entry = GpuBindGroupLayoutEntry::new(8, shader_stage::COMPUTE);
167 let layout = GpuSamplerBindingLayout::new();
168 layout.set_type(GpuSamplerBindingType::Filtering);
169 entry.set_sampler(&layout);
170 entry
171 };
172 let trace_layout = device.create_bind_group_layout(&GpuBindGroupLayoutDescriptor::new(&[
173 buffer_entry(0, Uniform, false),
174 buffer_entry(1, ReadOnlyStorage, false),
175 buffer_entry(2, ReadOnlyStorage, false),
176 buffer_entry(3, ReadOnlyStorage, false),
177 buffer_entry(4, Storage, false),
178 storage_texture_entry(5),
179 buffer_entry(6, Storage, false),
180 texture_entry(7, shader_stage::COMPUTE, GpuTextureSampleType::Float),
181 sampler_entry,
182 ]))?;
183 let trace = compute_pipeline(device, &trace_layout, &rt_bvh_source(), "cs_main", "rt-trace");
184 let denoise_layout = device.create_bind_group_layout(&GpuBindGroupLayoutDescriptor::new(&[
185 buffer_entry(0, Uniform, true),
186 buffer_entry(1, ReadOnlyStorage, false),
187 buffer_entry(2, ReadOnlyStorage, false),
188 buffer_entry(3, ReadOnlyStorage, false),
189 buffer_entry(4, Storage, false),
190 storage_texture_entry(5),
191 ]))?;
192 let denoise = compute_pipeline(device, &denoise_layout, RT_DENOISE, "cs_denoise", "rt-denoise");
193
194 let blit_layout = device.create_bind_group_layout(&GpuBindGroupLayoutDescriptor::new(&[
195 texture_entry(0, shader_stage::FRAGMENT, GpuTextureSampleType::UnfilterableFloat),
196 {
197 let entry = GpuBindGroupLayoutEntry::new(1, shader_stage::FRAGMENT);
198 let layout = GpuBufferBindingLayout::new();
199 layout.set_type(GpuBufferBindingType::Uniform);
200 entry.set_buffer(&layout);
201 entry
202 },
203 ]))?;
204 let blit_module = shader_module(device, BLIT, "rt-blit");
205 let vertex = GpuVertexState::new(&blit_module);
206 vertex.set_entry_point("vs_main");
207 let targets = [js_sys::JsOption::wrap(GpuColorTargetState::new(backdrop_format))];
208 let fragment = GpuFragmentState::new(&blit_module, &targets);
209 fragment.set_entry_point("fs_main");
210 let desc = GpuRenderPipelineDescriptor::new(
211 &device.create_pipeline_layout(&GpuPipelineLayoutDescriptor::new(&[js_sys::JsOption::wrap(blit_layout.clone())])),
212 &vertex,
213 );
214 desc.set_fragment(&fragment);
215 desc.set_label("rt-blit");
216 let blit = device.create_render_pipeline(&desc)?;
217
218 let params = buffer(device, std::mem::size_of::<RtParams>() as u32, buffer_usage::UNIFORM | buffer_usage::COPY_DST, "rt-params")?;
219 let denoise_params =
220 buffer(device, STRIDE * DENOISE_ITERATIONS as u32, buffer_usage::UNIFORM | buffer_usage::COPY_DST, "rt-denoise-params")?;
221 let blit_params = buffer(device, 16, buffer_usage::UNIFORM | buffer_usage::COPY_DST, "rt-blit-params")?;
222 // The image binding's stand-in while the scene has none: one clear texel.
223 let stand_in = texture(device, GpuTextureFormat::Rgba8unormSrgb, 1, 1, texture_usage::TEXTURE_BINDING | texture_usage::COPY_DST, "rt-stand-in")?;
224 let sampler = {
225 let desc = web_sys::GpuSamplerDescriptor::new();
226 desc.set_mag_filter(web_sys::GpuFilterMode::Linear);
227 desc.set_min_filter(web_sys::GpuFilterMode::Linear);
228 desc.set_mipmap_filter(web_sys::GpuMipmapFilterMode::Linear);
229 device.create_sampler_with_descriptor(&desc)
230 };
231 Ok(Self {
232 trace,
233 trace_layout,
234 denoise,
235 denoise_layout,
236 blit,
237 blit_layout,
238 params,
239 denoise_params,
240 blit_params,
241 stand_in: whole_view(&stand_in)?,
242 sampler,
243 scene: None,
244 targets: None,
245 pane: (0, 0, 0, 0),
246 pane_moved: false,
247 camera: None,
248 sample_index: 0,
249 spp: 1,
250 staged: false,
251 background: None,
252 environment: RtEnvironment::default(),
253 })
254 }
255
256 /// Replace the scene with one prepared on the CPU, uploaded — its BVH
257 /// built here first if it was prepared without one.
258 pub(crate) fn set_scene(&mut self, device: &GpuDevice, queue: &GpuQueue, scene: &PreparedRtScene) -> Result<(), JsValue> {
259 let packed = scene.packed.with_bvh();
260 let image = scene.image;
261 let upload = |bytes: &[u8], label: &str| -> Result<GpuBuffer, JsValue> {
262 let b = buffer(device, bytes.len() as u32, buffer_usage::STORAGE | buffer_usage::COPY_DST, label)?;
263 if !bytes.is_empty() {
264 queue.write_buffer_with_u32_and_u8_slice(&b, 0, bytes)?;
265 }
266 Ok(b)
267 };
268 if let Some(old) = self.scene.take() {
269 for b in [old.nodes, old.tris, old.materials] {
270 b.destroy();
271 }
272 }
273 self.scene = Some(Scene {
274 nodes: upload(bytemuck::cast_slice(&packed.nodes), "rt-nodes")?,
275 tris: upload(bytemuck::cast_slice(&packed.tris), "rt-tris")?,
276 materials: upload(bytemuck::cast_slice(&packed.materials), "rt-materials")?,
277 tri_count: packed.tris.len() as u32,
278 image,
279 });
280 self.sample_index = 0;
281 Ok(())
282 }
283
284 pub(crate) fn set_background(&mut self, background: Option<[f32; 3]>) {
285 if self.background != background {
286 self.background = background;
287 self.sample_index = 0;
288 }
289 }
290
291 pub(crate) fn set_environment(&mut self, environment: RtEnvironment) {
292 if self.environment != environment {
293 self.environment = environment;
294 self.sample_index = 0;
295 }
296 }
297
298 /// Stage a frame for `pane` (physical px): the targets follow its size,
299 /// and a camera or size change restarts the accumulation; a pane that
300 /// moved clears the backdrop once before the next blit.
301 pub(crate) fn stage(&mut self, device: &GpuDevice, pane: (u32, u32, u32, u32), camera: RtCamera) -> Result<(), JsValue> {
302 let (_, _, w, h) = pane;
303 if w == 0 || h == 0 {
304 self.staged = false;
305 return Ok(());
306 }
307 if self.targets.as_ref().map(|t| t.size) != Some((w, h)) {
308 self.recreate_targets(device, w, h)?;
309 self.sample_index = 0;
310 }
311 if pane != self.pane && self.pane != (0, 0, 0, 0) {
312 self.pane_moved = true;
313 }
314 if self.camera != Some(camera) {
315 self.camera = Some(camera);
316 self.sample_index = 0;
317 }
318 self.pane = pane;
319 self.staged = true;
320 Ok(())
321 }
322
323 pub(crate) fn staged(&self) -> bool {
324 self.staged
325 }
326
327 /// True while another dispatch would still refine the image.
328 pub(crate) fn accumulating(&self) -> bool {
329 self.scene.as_ref().is_some_and(|s| s.tri_count > 0) && self.sample_index < MAX_SAMPLES
330 }
331
332 fn recreate_targets(&mut self, device: &GpuDevice, w: u32, h: u32) -> Result<(), JsValue> {
333 if let Some(old) = self.targets.take() {
334 for b in [old.accum, old.features, old.ping, old.pong] {
335 b.destroy();
336 }
337 old.output.destroy();
338 }
339 let px = w * h;
340 let storage = buffer_usage::STORAGE;
341 let accum = buffer(device, px * 16, storage, "rt-accum")?;
342 let features = buffer(device, px * 32, storage, "rt-features")?;
343 let ping = buffer(device, px * 16, storage, "rt-denoise-ping")?;
344 let pong = buffer(device, px * 16, storage, "rt-denoise-pong")?;
345 let output = texture(
346 device,
347 GpuTextureFormat::Rgba8unorm,
348 w,
349 h,
350 texture_usage::STORAGE_BINDING | texture_usage::TEXTURE_BINDING,
351 "rt-output",
352 )?;
353 let output_view = whole_view(&output)?;
354 let denoise_group = |src: &GpuBuffer, dst: &GpuBuffer| {
355 let entries = [
356 sized(0, &self.denoise_params, std::mem::size_of::<DenoiseParams>() as u32),
357 whole(1, &accum),
358 whole(2, &features),
359 whole(3, src),
360 whole(4, dst),
361 GpuBindGroupEntry::new_with_gpu_texture_view(5, &output_view),
362 ];
363 device.create_bind_group(&GpuBindGroupDescriptor::new(&entries, &self.denoise_layout))
364 };
365 let denoise_b = denoise_group(&pong, &ping);
366 let denoise_a = denoise_group(&ping, &pong);
367 let blit = device.create_bind_group(&GpuBindGroupDescriptor::new(
368 &[GpuBindGroupEntry::new_with_gpu_texture_view(0, &output_view), whole(1, &self.blit_params)],
369 &self.blit_layout,
370 ));
371 self.targets = Some(Targets { accum, features, ping, pong, output, output_view, denoise_b, denoise_a, blit, size: (w, h) });
372 Ok(())
373 }
374
375 /// Record one accumulation dispatch, the denoise and the blit into
376 /// `backdrop` (the scene's, `backdrop_size` px). False when there is
377 /// nothing to do (not staged, an empty scene, converged): the backdrop
378 /// then keeps what it has. `image_view` looks the scene's image up in
379 /// the 2D pass's table: its view and size, if it has landed.
380 pub(crate) fn record(
381 &mut self,
382 device: &GpuDevice,
383 queue: &GpuQueue,
384 encoder: &GpuCommandEncoder,
385 backdrop: &GpuTextureView,
386 backdrop_size: (u32, u32),
387 image_view: &dyn Fn(u32) -> Option<(GpuTextureView, u32, u32)>,
388 ) -> Result<bool, JsValue> {
389 let ready = self.staged && self.scene.as_ref().is_some_and(|s| s.tri_count > 0) && self.targets.is_some();
390 self.staged = false;
391 if !ready || self.sample_index >= MAX_SAMPLES {
392 return Ok(false);
393 }
394 let (Some(scene), Some(t), Some(camera)) = (&self.scene, &self.targets, self.camera) else { return Ok(false) };
395 let (w, h) = t.size;
396
397 // The parameter blocks: the scene's image as it is NOW (its upload
398 // may land after the scene was set), or none, which lets rays through.
399 let bound = scene.image.and_then(|i| image_view(i.image).map(|(view, iw, ih)| (view, i, iw, ih)));
400 let (view, param_image) = match bound {
401 Some((view, i, iw, ih)) => (view, Some(ParamImage { width: iw, height: ih, corners: i.corners, opacity: i.opacity })),
402 None => (self.stand_in.clone(), None),
403 };
404 let params = rt_params(camera, (w, h), self.sample_index, self.spp, param_image, self.background, &self.environment);
405 queue.write_buffer_with_u32_and_u8_slice(&self.params, 0, bytemuck::bytes_of(¶ms))?;
406 let mut blocks = vec![0u8; STRIDE as usize * DENOISE_ITERATIONS];
407 for (i, p) in denoise_params(w, h, self.sample_index + self.spp).iter().enumerate() {
408 let at = i * STRIDE as usize;
409 blocks[at..at + std::mem::size_of::<DenoiseParams>()].copy_from_slice(bytemuck::bytes_of(p));
410 }
411 queue.write_buffer_with_u32_and_u8_slice(&self.denoise_params, 0, &blocks)?;
412 let (px, py, _, _) = self.pane;
413 let dst_x = px.min(backdrop_size.0);
414 let dst_y = py.min(backdrop_size.1);
415 let (bw, bh) = (w.min(backdrop_size.0 - dst_x), h.min(backdrop_size.1 - dst_y));
416 queue.write_buffer_with_u32_and_u8_slice(&self.blit_params, 0, bytemuck::cast_slice(&[dst_x as i32, dst_y as i32, 0, 0]))?;
417
418 let trace_group = device.create_bind_group(&GpuBindGroupDescriptor::new(
419 &[
420 whole(0, &self.params),
421 whole(1, &scene.nodes),
422 whole(2, &scene.tris),
423 whole(3, &scene.materials),
424 whole(4, &t.accum),
425 GpuBindGroupEntry::new_with_gpu_texture_view(5, &t.output_view),
426 whole(6, &t.features),
427 GpuBindGroupEntry::new_with_gpu_texture_view(7, &view),
428 GpuBindGroupEntry::new(8, &self.sampler),
429 ],
430 &self.trace_layout,
431 ));
432 let groups = (w.div_ceil(WORKGROUP), h.div_ceil(WORKGROUP));
433 let pass = encoder.begin_compute_pass_with_descriptor(&GpuComputePassDescriptor::new());
434 pass.set_pipeline(&self.trace);
435 pass.set_bind_group(0, Some(&trace_group));
436 pass.dispatch_workgroups_with_workgroup_count_y(groups.0, groups.1);
437 // À-trous iterations: 0 and 2 write ping, 1 writes pong; the last
438 // rewrites the output image.
439 pass.set_pipeline(&self.denoise);
440 for i in 0..DENOISE_ITERATIONS {
441 let group = if i % 2 == 0 { &t.denoise_b } else { &t.denoise_a };
442 pass.set_bind_group_with_u32_slice_and_u32_and_dynamic_offsets_data_length(0, Some(group), &[i as u32 * STRIDE], 0, 1)?;
443 pass.dispatch_workgroups_with_workgroup_count_y(groups.0, groups.1);
444 }
445 pass.end();
446
447 // Into the backdrop's pane region (clearing the whole backdrop first
448 // when the pane moved: stale pixels sit outside the new region).
449 let load = if std::mem::take(&mut self.pane_moved) { GpuLoadOp::Clear } else { GpuLoadOp::Load };
450 let color = GpuRenderPassColorAttachment::new_with_gpu_texture_view(load, GpuStoreOp::Store, backdrop);
451 color.set_clear_value(&[0.0, 0.0, 0.0, 0.0].map(js_sys::Number::from));
452 let pass = encoder.begin_render_pass(&GpuRenderPassDescriptor::new(&[js_sys::JsOption::wrap(color)]))?;
453 if bw > 0 && bh > 0 {
454 pass.set_viewport(dst_x as f32, dst_y as f32, bw as f32, bh as f32, 0.0, 1.0);
455 pass.set_scissor_rect(dst_x, dst_y, bw, bh);
456 pass.set_pipeline(&self.blit);
457 pass.set_bind_group(0, Some(&t.blit));
458 pass.draw(3);
459 }
460 pass.end();
461 self.sample_index += self.spp;
462 Ok(true)
463 }
464 }