git.lucas.co / cce-ui
GPU-accelerated UI toolkit (Vulkan)
git clone https://git.lucas.co/cce-ui.git

src/web/rt.rs (20.9K)

  1 //! The path tracer on WebGPU: the Vulkan `RtStage`'s compute tier, ported.
  2 //! The same shaders (`draw::shaders::rt_bvh_source`, `RT_DENOISE`), the
  3 //! same scene packing, BVH and parameter blocks (`draw::rt`), the same
  4 //! rules for when the accumulation restarts — a camera, pane size, scene,
  5 //! background or environment change — and the same frame: one sample a
  6 //! frame added into the running mean, three à-trous denoise iterations over
  7 //! it, the result put into the backdrop's pane region in place of the
  8 //! raster scene, a pane that moved clearing the backdrop once first.
  9 //!
 10 //! What WebGPU makes different:
 11 //!
 12 //! - **The compute tier only.** The hardware ray-query tier needs
 13 //!   `VK_KHR_ray_query`; WebGPU has no ray tracing. A BVH traversed in
 14 //!   compute is the tier every Vulkan device without RT cores runs too.
 15 //! - **The blit is a draw.** Vulkan blits the tracer's `rgba8unorm` image
 16 //!   into the sRGB backdrop, converting as it copies; WebGPU copies only
 17 //!   between formats that differ in sRGB-ness at most, so a small render
 18 //!   pass loads each texel and writes it through the backdrop's sRGB view —
 19 //!   the same conversion (the unorm value taken as linear and encoded).
 20 //! - **No frames in flight to juggle.** One parameter buffer and one bind
 21 //!   group per dispatch, written before the frame's single submission.
 22 
 23 use wasm_bindgen::JsValue;
 24 use web_sys::{
 25     gpu_buffer_usage as buffer_usage, gpu_shader_stage as shader_stage, gpu_texture_usage as texture_usage, GpuBindGroup,
 26     GpuBindGroupDescriptor, GpuBindGroupEntry, GpuBindGroupLayout, GpuBindGroupLayoutDescriptor, GpuBindGroupLayoutEntry,
 27     GpuBuffer, GpuBufferBinding, GpuBufferBindingLayout, GpuBufferBindingType, GpuColorTargetState, GpuCommandEncoder,
 28     GpuComputePassDescriptor, GpuComputePipeline, GpuComputePipelineDescriptor, GpuDevice, GpuFragmentState, GpuLoadOp,
 29     GpuPipelineLayoutDescriptor, GpuProgrammableStage, GpuQueue, GpuRenderPassColorAttachment, GpuRenderPassDescriptor,
 30     GpuRenderPipeline, GpuRenderPipelineDescriptor, GpuSampler, GpuSamplerBindingLayout, GpuSamplerBindingType,
 31     GpuStorageTextureAccess, GpuStorageTextureBindingLayout, GpuStoreOp, GpuTexture, GpuTextureBindingLayout,
 32     GpuTextureFormat, GpuTextureSampleType, GpuTextureView, GpuTextureViewDimension, GpuVertexState,
 33 };
 34 
 35 use super::renderer::{shader_module, texture, whole_view};
 36 use crate::draw::rt::{
 37     denoise_params, rt_params, DenoiseParams, ParamImage, PreparedRtScene, RtCamera, RtEnvironment, RtImage,
 38     RtParams, DENOISE_ITERATIONS, MAX_SAMPLES, WORKGROUP,
 39 };
 40 use crate::draw::shaders::{rt_bvh_source, RT_DENOISE};
 41 
 42 /// Each denoise iteration's block sits at a multiple of this.
 43 const STRIDE: u32 = 256;
 44 
 45 /// Loads the tracer's image at the fragment's place in the pane and writes
 46 /// it through the backdrop's sRGB view: Vulkan's UNORM-to-sRGB blit.
 47 const BLIT: &str = "
 48 struct Blit { origin: vec2<i32>, _pad: vec2<i32> }
 49 @group(0) @binding(0) var src: texture_2d<f32>;
 50 @group(0) @binding(1) var<uniform> blit: Blit;
 51 @vertex fn vs_main(@builtin(vertex_index) i: u32) -> @builtin(position) vec4<f32> {
 52     let p = vec2<f32>(f32((i << 1u) & 2u), f32(i & 2u));
 53     return vec4<f32>(p * 2.0 - 1.0, 0.0, 1.0);
 54 }
 55 @fragment fn fs_main(@builtin(position) pos: vec4<f32>) -> @location(0) vec4<f32> {
 56     return textureLoad(src, vec2<i32>(pos.xy) - blit.origin, 0);
 57 }";
 58 
 59 /// The pane-sized targets: the running sum, the denoiser's features and
 60 /// ping-pong, and the image the result is written to.
 61 struct Targets {
 62     accum: GpuBuffer,
 63     features: GpuBuffer,
 64     ping: GpuBuffer,
 65     pong: GpuBuffer,
 66     output: GpuTexture,
 67     output_view: GpuTextureView,
 68     /// Denoise iterations 0 and 2 (src pong, dst ping), and 1 (src ping, dst pong).
 69     denoise_b: GpuBindGroup,
 70     denoise_a: GpuBindGroup,
 71     blit: GpuBindGroup,
 72     size: (u32, u32),
 73 }
 74 
 75 /// The scene the buffers hold, and the image standing in it.
 76 struct Scene {
 77     nodes: GpuBuffer,
 78     tris: GpuBuffer,
 79     materials: GpuBuffer,
 80     tri_count: u32,
 81     image: Option<RtImage>,
 82 }
 83 
 84 pub(crate) struct WebRt {
 85     trace: GpuComputePipeline,
 86     trace_layout: GpuBindGroupLayout,
 87     denoise: GpuComputePipeline,
 88     denoise_layout: GpuBindGroupLayout,
 89     blit: GpuRenderPipeline,
 90     blit_layout: GpuBindGroupLayout,
 91     params: GpuBuffer,
 92     denoise_params: GpuBuffer,
 93     blit_params: GpuBuffer,
 94     stand_in: GpuTextureView,
 95     sampler: GpuSampler,
 96     scene: Option<Scene>,
 97     targets: Option<Targets>,
 98 
 99     pane: (u32, u32, u32, u32),
100     pane_moved: bool,
101     camera: Option<RtCamera>,
102     sample_index: u32,
103     spp: u32,
104     staged: bool,
105     background: Option<[f32; 3]>,
106     environment: RtEnvironment,
107 }
108 
109 fn buffer_entry(binding: u32, ty: GpuBufferBindingType, dynamic: bool) -> GpuBindGroupLayoutEntry {
110     let entry = GpuBindGroupLayoutEntry::new(binding, shader_stage::COMPUTE);
111     let layout = GpuBufferBindingLayout::new();
112     layout.set_type(ty);
113     layout.set_has_dynamic_offset(dynamic);
114     entry.set_buffer(&layout);
115     entry
116 }
117 
118 fn storage_texture_entry(binding: u32) -> GpuBindGroupLayoutEntry {
119     let entry = GpuBindGroupLayoutEntry::new(binding, shader_stage::COMPUTE);
120     let layout = GpuStorageTextureBindingLayout::new(GpuTextureFormat::Rgba8unorm);
121     layout.set_access(GpuStorageTextureAccess::WriteOnly);
122     layout.set_view_dimension(GpuTextureViewDimension::N2d);
123     entry.set_storage_texture(&layout);
124     entry
125 }
126 
127 fn texture_entry(binding: u32, visibility: u32, sample: GpuTextureSampleType) -> GpuBindGroupLayoutEntry {
128     let entry = GpuBindGroupLayoutEntry::new(binding, visibility);
129     let layout = GpuTextureBindingLayout::new();
130     layout.set_sample_type(sample);
131     layout.set_view_dimension(GpuTextureViewDimension::N2d);
132     entry.set_texture(&layout);
133     entry
134 }
135 
136 fn whole(binding: u32, buffer: &GpuBuffer) -> GpuBindGroupEntry {
137     GpuBindGroupEntry::new_with_gpu_buffer_binding(binding, &GpuBufferBinding::new(buffer))
138 }
139 
140 fn sized(binding: u32, buffer: &GpuBuffer, size: u32) -> GpuBindGroupEntry {
141     let b = GpuBufferBinding::new(buffer);
142     b.set_size(size);
143     GpuBindGroupEntry::new_with_gpu_buffer_binding(binding, &b)
144 }
145 
146 fn buffer(device: &GpuDevice, size: u32, usage: u32, label: &str) -> Result<GpuBuffer, JsValue> {
147     let desc = web_sys::GpuBufferDescriptor::new(size.max(16).next_multiple_of(4), usage);
148     desc.set_label(label);
149     device.create_buffer(&desc)
150 }
151 
152 fn compute_pipeline(device: &GpuDevice, layout: &GpuBindGroupLayout, code: &str, entry: &str, label: &str) -> GpuComputePipeline {
153     let pipeline_layout = device.create_pipeline_layout(&GpuPipelineLayoutDescriptor::new(&[js_sys::JsOption::wrap(layout.clone())]));
154     let stage = GpuProgrammableStage::new(&shader_module(device, code, label));
155     stage.set_entry_point(entry);
156     let desc = GpuComputePipelineDescriptor::new(&pipeline_layout, &stage);
157     desc.set_label(label);
158     device.create_compute_pipeline(&desc)
159 }
160 
161 impl WebRt {
162     /// The tracer's pipelines; `backdrop_format` is what the blit writes.
163     pub(crate) fn new(device: &GpuDevice, backdrop_format: GpuTextureFormat) -> Result<Self, JsValue> {
164         use GpuBufferBindingType::{ReadOnlyStorage, Storage, Uniform};
165         let sampler_entry = {
166             let entry = GpuBindGroupLayoutEntry::new(8, shader_stage::COMPUTE);
167             let layout = GpuSamplerBindingLayout::new();
168             layout.set_type(GpuSamplerBindingType::Filtering);
169             entry.set_sampler(&layout);
170             entry
171         };
172         let trace_layout = device.create_bind_group_layout(&GpuBindGroupLayoutDescriptor::new(&[
173             buffer_entry(0, Uniform, false),
174             buffer_entry(1, ReadOnlyStorage, false),
175             buffer_entry(2, ReadOnlyStorage, false),
176             buffer_entry(3, ReadOnlyStorage, false),
177             buffer_entry(4, Storage, false),
178             storage_texture_entry(5),
179             buffer_entry(6, Storage, false),
180             texture_entry(7, shader_stage::COMPUTE, GpuTextureSampleType::Float),
181             sampler_entry,
182         ]))?;
183         let trace = compute_pipeline(device, &trace_layout, &rt_bvh_source(), "cs_main", "rt-trace");
184         let denoise_layout = device.create_bind_group_layout(&GpuBindGroupLayoutDescriptor::new(&[
185             buffer_entry(0, Uniform, true),
186             buffer_entry(1, ReadOnlyStorage, false),
187             buffer_entry(2, ReadOnlyStorage, false),
188             buffer_entry(3, ReadOnlyStorage, false),
189             buffer_entry(4, Storage, false),
190             storage_texture_entry(5),
191         ]))?;
192         let denoise = compute_pipeline(device, &denoise_layout, RT_DENOISE, "cs_denoise", "rt-denoise");
193 
194         let blit_layout = device.create_bind_group_layout(&GpuBindGroupLayoutDescriptor::new(&[
195             texture_entry(0, shader_stage::FRAGMENT, GpuTextureSampleType::UnfilterableFloat),
196             {
197                 let entry = GpuBindGroupLayoutEntry::new(1, shader_stage::FRAGMENT);
198                 let layout = GpuBufferBindingLayout::new();
199                 layout.set_type(GpuBufferBindingType::Uniform);
200                 entry.set_buffer(&layout);
201                 entry
202             },
203         ]))?;
204         let blit_module = shader_module(device, BLIT, "rt-blit");
205         let vertex = GpuVertexState::new(&blit_module);
206         vertex.set_entry_point("vs_main");
207         let targets = [js_sys::JsOption::wrap(GpuColorTargetState::new(backdrop_format))];
208         let fragment = GpuFragmentState::new(&blit_module, &targets);
209         fragment.set_entry_point("fs_main");
210         let desc = GpuRenderPipelineDescriptor::new(
211             &device.create_pipeline_layout(&GpuPipelineLayoutDescriptor::new(&[js_sys::JsOption::wrap(blit_layout.clone())])),
212             &vertex,
213         );
214         desc.set_fragment(&fragment);
215         desc.set_label("rt-blit");
216         let blit = device.create_render_pipeline(&desc)?;
217 
218         let params = buffer(device, std::mem::size_of::<RtParams>() as u32, buffer_usage::UNIFORM | buffer_usage::COPY_DST, "rt-params")?;
219         let denoise_params =
220             buffer(device, STRIDE * DENOISE_ITERATIONS as u32, buffer_usage::UNIFORM | buffer_usage::COPY_DST, "rt-denoise-params")?;
221         let blit_params = buffer(device, 16, buffer_usage::UNIFORM | buffer_usage::COPY_DST, "rt-blit-params")?;
222         // The image binding's stand-in while the scene has none: one clear texel.
223         let stand_in = texture(device, GpuTextureFormat::Rgba8unormSrgb, 1, 1, texture_usage::TEXTURE_BINDING | texture_usage::COPY_DST, "rt-stand-in")?;
224         let sampler = {
225             let desc = web_sys::GpuSamplerDescriptor::new();
226             desc.set_mag_filter(web_sys::GpuFilterMode::Linear);
227             desc.set_min_filter(web_sys::GpuFilterMode::Linear);
228             desc.set_mipmap_filter(web_sys::GpuMipmapFilterMode::Linear);
229             device.create_sampler_with_descriptor(&desc)
230         };
231         Ok(Self {
232             trace,
233             trace_layout,
234             denoise,
235             denoise_layout,
236             blit,
237             blit_layout,
238             params,
239             denoise_params,
240             blit_params,
241             stand_in: whole_view(&stand_in)?,
242             sampler,
243             scene: None,
244             targets: None,
245             pane: (0, 0, 0, 0),
246             pane_moved: false,
247             camera: None,
248             sample_index: 0,
249             spp: 1,
250             staged: false,
251             background: None,
252             environment: RtEnvironment::default(),
253         })
254     }
255 
256     /// Replace the scene with one prepared on the CPU, uploaded — its BVH
257     /// built here first if it was prepared without one.
258     pub(crate) fn set_scene(&mut self, device: &GpuDevice, queue: &GpuQueue, scene: &PreparedRtScene) -> Result<(), JsValue> {
259         let packed = scene.packed.with_bvh();
260         let image = scene.image;
261         let upload = |bytes: &[u8], label: &str| -> Result<GpuBuffer, JsValue> {
262             let b = buffer(device, bytes.len() as u32, buffer_usage::STORAGE | buffer_usage::COPY_DST, label)?;
263             if !bytes.is_empty() {
264                 queue.write_buffer_with_u32_and_u8_slice(&b, 0, bytes)?;
265             }
266             Ok(b)
267         };
268         if let Some(old) = self.scene.take() {
269             for b in [old.nodes, old.tris, old.materials] {
270                 b.destroy();
271             }
272         }
273         self.scene = Some(Scene {
274             nodes: upload(bytemuck::cast_slice(&packed.nodes), "rt-nodes")?,
275             tris: upload(bytemuck::cast_slice(&packed.tris), "rt-tris")?,
276             materials: upload(bytemuck::cast_slice(&packed.materials), "rt-materials")?,
277             tri_count: packed.tris.len() as u32,
278             image,
279         });
280         self.sample_index = 0;
281         Ok(())
282     }
283 
284     pub(crate) fn set_background(&mut self, background: Option<[f32; 3]>) {
285         if self.background != background {
286             self.background = background;
287             self.sample_index = 0;
288         }
289     }
290 
291     pub(crate) fn set_environment(&mut self, environment: RtEnvironment) {
292         if self.environment != environment {
293             self.environment = environment;
294             self.sample_index = 0;
295         }
296     }
297 
298     /// Stage a frame for `pane` (physical px): the targets follow its size,
299     /// and a camera or size change restarts the accumulation; a pane that
300     /// moved clears the backdrop once before the next blit.
301     pub(crate) fn stage(&mut self, device: &GpuDevice, pane: (u32, u32, u32, u32), camera: RtCamera) -> Result<(), JsValue> {
302         let (_, _, w, h) = pane;
303         if w == 0 || h == 0 {
304             self.staged = false;
305             return Ok(());
306         }
307         if self.targets.as_ref().map(|t| t.size) != Some((w, h)) {
308             self.recreate_targets(device, w, h)?;
309             self.sample_index = 0;
310         }
311         if pane != self.pane && self.pane != (0, 0, 0, 0) {
312             self.pane_moved = true;
313         }
314         if self.camera != Some(camera) {
315             self.camera = Some(camera);
316             self.sample_index = 0;
317         }
318         self.pane = pane;
319         self.staged = true;
320         Ok(())
321     }
322 
323     pub(crate) fn staged(&self) -> bool {
324         self.staged
325     }
326 
327     /// True while another dispatch would still refine the image.
328     pub(crate) fn accumulating(&self) -> bool {
329         self.scene.as_ref().is_some_and(|s| s.tri_count > 0) && self.sample_index < MAX_SAMPLES
330     }
331 
332     fn recreate_targets(&mut self, device: &GpuDevice, w: u32, h: u32) -> Result<(), JsValue> {
333         if let Some(old) = self.targets.take() {
334             for b in [old.accum, old.features, old.ping, old.pong] {
335                 b.destroy();
336             }
337             old.output.destroy();
338         }
339         let px = w * h;
340         let storage = buffer_usage::STORAGE;
341         let accum = buffer(device, px * 16, storage, "rt-accum")?;
342         let features = buffer(device, px * 32, storage, "rt-features")?;
343         let ping = buffer(device, px * 16, storage, "rt-denoise-ping")?;
344         let pong = buffer(device, px * 16, storage, "rt-denoise-pong")?;
345         let output = texture(
346             device,
347             GpuTextureFormat::Rgba8unorm,
348             w,
349             h,
350             texture_usage::STORAGE_BINDING | texture_usage::TEXTURE_BINDING,
351             "rt-output",
352         )?;
353         let output_view = whole_view(&output)?;
354         let denoise_group = |src: &GpuBuffer, dst: &GpuBuffer| {
355             let entries = [
356                 sized(0, &self.denoise_params, std::mem::size_of::<DenoiseParams>() as u32),
357                 whole(1, &accum),
358                 whole(2, &features),
359                 whole(3, src),
360                 whole(4, dst),
361                 GpuBindGroupEntry::new_with_gpu_texture_view(5, &output_view),
362             ];
363             device.create_bind_group(&GpuBindGroupDescriptor::new(&entries, &self.denoise_layout))
364         };
365         let denoise_b = denoise_group(&pong, &ping);
366         let denoise_a = denoise_group(&ping, &pong);
367         let blit = device.create_bind_group(&GpuBindGroupDescriptor::new(
368             &[GpuBindGroupEntry::new_with_gpu_texture_view(0, &output_view), whole(1, &self.blit_params)],
369             &self.blit_layout,
370         ));
371         self.targets = Some(Targets { accum, features, ping, pong, output, output_view, denoise_b, denoise_a, blit, size: (w, h) });
372         Ok(())
373     }
374 
375     /// Record one accumulation dispatch, the denoise and the blit into
376     /// `backdrop` (the scene's, `backdrop_size` px). False when there is
377     /// nothing to do (not staged, an empty scene, converged): the backdrop
378     /// then keeps what it has. `image_view` looks the scene's image up in
379     /// the 2D pass's table: its view and size, if it has landed.
380     pub(crate) fn record(
381         &mut self,
382         device: &GpuDevice,
383         queue: &GpuQueue,
384         encoder: &GpuCommandEncoder,
385         backdrop: &GpuTextureView,
386         backdrop_size: (u32, u32),
387         image_view: &dyn Fn(u32) -> Option<(GpuTextureView, u32, u32)>,
388     ) -> Result<bool, JsValue> {
389         let ready = self.staged && self.scene.as_ref().is_some_and(|s| s.tri_count > 0) && self.targets.is_some();
390         self.staged = false;
391         if !ready || self.sample_index >= MAX_SAMPLES {
392             return Ok(false);
393         }
394         let (Some(scene), Some(t), Some(camera)) = (&self.scene, &self.targets, self.camera) else { return Ok(false) };
395         let (w, h) = t.size;
396 
397         // The parameter blocks: the scene's image as it is NOW (its upload
398         // may land after the scene was set), or none, which lets rays through.
399         let bound = scene.image.and_then(|i| image_view(i.image).map(|(view, iw, ih)| (view, i, iw, ih)));
400         let (view, param_image) = match bound {
401             Some((view, i, iw, ih)) => (view, Some(ParamImage { width: iw, height: ih, corners: i.corners, opacity: i.opacity })),
402             None => (self.stand_in.clone(), None),
403         };
404         let params = rt_params(camera, (w, h), self.sample_index, self.spp, param_image, self.background, &self.environment);
405         queue.write_buffer_with_u32_and_u8_slice(&self.params, 0, bytemuck::bytes_of(&params))?;
406         let mut blocks = vec![0u8; STRIDE as usize * DENOISE_ITERATIONS];
407         for (i, p) in denoise_params(w, h, self.sample_index + self.spp).iter().enumerate() {
408             let at = i * STRIDE as usize;
409             blocks[at..at + std::mem::size_of::<DenoiseParams>()].copy_from_slice(bytemuck::bytes_of(p));
410         }
411         queue.write_buffer_with_u32_and_u8_slice(&self.denoise_params, 0, &blocks)?;
412         let (px, py, _, _) = self.pane;
413         let dst_x = px.min(backdrop_size.0);
414         let dst_y = py.min(backdrop_size.1);
415         let (bw, bh) = (w.min(backdrop_size.0 - dst_x), h.min(backdrop_size.1 - dst_y));
416         queue.write_buffer_with_u32_and_u8_slice(&self.blit_params, 0, bytemuck::cast_slice(&[dst_x as i32, dst_y as i32, 0, 0]))?;
417 
418         let trace_group = device.create_bind_group(&GpuBindGroupDescriptor::new(
419             &[
420                 whole(0, &self.params),
421                 whole(1, &scene.nodes),
422                 whole(2, &scene.tris),
423                 whole(3, &scene.materials),
424                 whole(4, &t.accum),
425                 GpuBindGroupEntry::new_with_gpu_texture_view(5, &t.output_view),
426                 whole(6, &t.features),
427                 GpuBindGroupEntry::new_with_gpu_texture_view(7, &view),
428                 GpuBindGroupEntry::new(8, &self.sampler),
429             ],
430             &self.trace_layout,
431         ));
432         let groups = (w.div_ceil(WORKGROUP), h.div_ceil(WORKGROUP));
433         let pass = encoder.begin_compute_pass_with_descriptor(&GpuComputePassDescriptor::new());
434         pass.set_pipeline(&self.trace);
435         pass.set_bind_group(0, Some(&trace_group));
436         pass.dispatch_workgroups_with_workgroup_count_y(groups.0, groups.1);
437         // À-trous iterations: 0 and 2 write ping, 1 writes pong; the last
438         // rewrites the output image.
439         pass.set_pipeline(&self.denoise);
440         for i in 0..DENOISE_ITERATIONS {
441             let group = if i % 2 == 0 { &t.denoise_b } else { &t.denoise_a };
442             pass.set_bind_group_with_u32_slice_and_u32_and_dynamic_offsets_data_length(0, Some(group), &[i as u32 * STRIDE], 0, 1)?;
443             pass.dispatch_workgroups_with_workgroup_count_y(groups.0, groups.1);
444         }
445         pass.end();
446 
447         // Into the backdrop's pane region (clearing the whole backdrop first
448         // when the pane moved: stale pixels sit outside the new region).
449         let load = if std::mem::take(&mut self.pane_moved) { GpuLoadOp::Clear } else { GpuLoadOp::Load };
450         let color = GpuRenderPassColorAttachment::new_with_gpu_texture_view(load, GpuStoreOp::Store, backdrop);
451         color.set_clear_value(&[0.0, 0.0, 0.0, 0.0].map(js_sys::Number::from));
452         let pass = encoder.begin_render_pass(&GpuRenderPassDescriptor::new(&[js_sys::JsOption::wrap(color)]))?;
453         if bw > 0 && bh > 0 {
454             pass.set_viewport(dst_x as f32, dst_y as f32, bw as f32, bh as f32, 0.0, 1.0);
455             pass.set_scissor_rect(dst_x, dst_y, bw, bh);
456             pass.set_pipeline(&self.blit);
457             pass.set_bind_group(0, Some(&t.blit));
458             pass.draw(3);
459         }
460         pass.end();
461         self.sample_index += self.spp;
462         Ok(true)
463     }
464 }