GPU-accelerated UI toolkit (Vulkan)
git clone https://git.lucas.co/cce-ui.git
feat(rt): PreparedRtScene, the tracer's CPU work off the UI thread
set_rt_scene packed the scene and built its binned-SAH BVH wherever it was
called, and an app only holds a Stage3D inside stage_3d, so a 5M-triangle
STL froze cce-model for 2.3 s. The CPU half is now PreparedRtScene::new
(packing + BVH, Send + Sync, for a worker) and the GPU half
Stage3D::set_rt_scene_prepared, which only uploads; set_rt_scene and
set_rt_scene_with_image wrap the two, so existing callers are unchanged.
Stage3D::rt_needs_bvh says whether to build the BVH: the Vulkan
ray-query tier builds its BLAS on the GPU from the packed triangles and
ignores one. A scene prepared without a BVH gets it at upload on a
compute-tier renderer (PackedScene::with_bvh), so a wrong answer costs
time, not correctness. The WebGPU stage takes the prepared scene the same
way (not built here: no wasm32 target).
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
CLAUDE.md | 13 +++++
src/draw/rt.rs | 142 ++++++++++++++++++++++++++++++++++++++++++++++++----
src/draw/scene.rs | 28 +++++++++--
src/engine.rs | 2 +-
src/vk/mod.rs | 2 +-
src/vk/renderer.rs | 38 ++++++++++++--
src/vk/rt.rs | 77 ++++++++++++++++++----------
src/web/renderer.rs | 6 +--
src/web/rt.rs | 19 +++----
9 files changed, 265 insertions(+), 62 deletions(-)
diff --git a/CLAUDE.md b/CLAUDE.md
index 685ddbf..3f2e81a 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -262,6 +262,19 @@ are compared at eight samples: 2026-10-05, Vulkan compute tier (`CCE_VK_RT=compu
lavapipe vs SwiftShader, the traced pane's mean differs by 0.10 of a level, every pixel
within 8, 47 channels in the frame past 8 — a few paths that diverged.
+**A large traced scene is prepared off the UI thread** (since 2026-10-07).
+`set_rt_scene` builds the BVH where it is called, and an app only has the stage inside
+`stage_3d`, so 5M triangles froze cce-model for 2.3 s. `PreparedRtScene::new(tris, mats,
+image, with_bvh)` is the CPU half — packing into the buffer layouts plus the BVH — and is
+`Send + Sync`, built on a worker; `Stage3D::set_rt_scene_prepared(&scene)` only uploads
+(and keeps it: a reconnect re-uploads the same one). `with_bvh` is `Stage3D::rt_needs_bvh()`,
+asked on the UI thread first: false on the Vulkan ray-query tier, which builds its BLAS on
+the GPU from the packed triangles and ignores a BVH; a scene prepared without one on a
+compute-tier renderer gets it built at upload (`PackedScene::with_bvh`), so a wrong answer
+costs time, never a wrong image. `set_rt_scene` / `set_rt_scene_with_image` are wrappers
+over the two halves. The GPU tests run per tier with `VK_DRIVER_FILES` pinned (Intel =
+compute, NVIDIA = ray-query, NVIDIA + `CCE_VK_RT=compute`); no lavapipe ICD is installed.
+
**The reference app runs on both, through one input script.** `examples/demo_web.rs` is
`src/main.rs`'s `DemoApp` (included by `#[path]`, hence `pub(crate)`) in a page;
`scripts/web-probe/demo <dir>` builds it, serves it with the machine's fonts and replays the
diff --git a/src/draw/rt.rs b/src/draw/rt.rs
index 4845067..ea38d04 100644
--- a/src/draw/rt.rs
+++ b/src/draw/rt.rs
@@ -354,12 +354,47 @@ pub(crate) struct DenoiseParams {
pub(crate) const DENOISE_ITERATIONS: usize = 3; // à-trous steps 1, 2, 4
/// A scene as the tracer's buffers hold it.
+#[derive(Clone)]
pub(crate) struct PackedScene {
pub tris: Vec<GpuTriangle>,
pub materials: Vec<GpuMaterial>,
/// Empty unless `with_bvh` was asked for (the ray-query tier builds
/// its own structure).
pub nodes: Vec<GpuBvhNode>,
+ /// The BVH was built (`nodes` may still be empty: an empty scene).
+ pub has_bvh: bool,
+}
+
+impl PackedScene {
+ /// This scene with its BVH: itself when it has one, else one built
+ /// now over a copy of the triangles — the compute tier handed a scene
+ /// prepared for the ray-query one.
+ pub(crate) fn with_bvh(&self) -> std::borrow::Cow<'_, PackedScene> {
+ if self.has_bvh {
+ return std::borrow::Cow::Borrowed(self);
+ }
+ let xyz = |p: [f32; 4]| [p[0], p[1], p[2]];
+ let mut tris: Vec<RtTriangle> = self
+ .tris
+ .iter()
+ .map(|t| RtTriangle { p0: xyz(t.p0), p1: xyz(t.p1), p2: xyz(t.p2), material: t.p0[3].to_bits() })
+ .collect();
+ let nodes = build_bvh(&mut tris);
+ std::borrow::Cow::Owned(PackedScene {
+ tris: tris.iter().map(gpu_triangle).collect(),
+ materials: self.materials.clone(),
+ nodes,
+ has_bvh: true,
+ })
+ }
+}
+
+fn gpu_triangle(t: &RtTriangle) -> GpuTriangle {
+ GpuTriangle {
+ p0: [t.p0[0], t.p0[1], t.p0[2], f32::from_bits(t.material)],
+ p1: [t.p1[0], t.p1[1], t.p1[2], 0.0],
+ p2: [t.p2[0], t.p2[1], t.p2[2], 0.0],
+ }
}
/// Lay a scene out for the tracer. An image joins it as a quad of two
@@ -369,26 +404,19 @@ pub(crate) struct PackedScene {
/// meets it as it meets any triangle, and a scene that is an image alone is
/// not an empty one. `with_bvh` builds the BVH, reordering the triangles.
pub(crate) fn pack_scene(
- triangles: &[RtTriangle],
+ mut tris: Vec<RtTriangle>,
materials: &[RtMaterial],
image_corners: Option<[[f32; 3]; 4]>,
with_bvh: bool,
) -> PackedScene {
- let mut tris: Vec<RtTriangle> = triangles.to_vec();
let image_material = materials.len().max(1) as u32;
if let Some([tl, tr, br, bl]) = image_corners {
tris.push(RtTriangle { p0: tl, p1: bl, p2: tr, material: image_material });
tris.push(RtTriangle { p0: tr, p1: bl, p2: br, material: image_material });
}
let nodes = if with_bvh { build_bvh(&mut tris) } else { Vec::new() };
- let gpu_tris = tris
- .iter()
- .map(|t| GpuTriangle {
- p0: [t.p0[0], t.p0[1], t.p0[2], f32::from_bits(t.material)],
- p1: [t.p1[0], t.p1[1], t.p1[2], 0.0],
- p2: [t.p2[0], t.p2[1], t.p2[2], 0.0],
- })
- .collect();
+ let gpu_tris = tris.iter().map(gpu_triangle).collect();
+ drop(tris);
let mut gpu_mats: Vec<GpuMaterial> = if materials.is_empty() {
vec![GpuMaterial { albedo: [0.8, 0.8, 0.8, 0.0], emission: [0.0; 4] }]
} else {
@@ -404,7 +432,64 @@ pub(crate) fn pack_scene(
// albedo.w marks it textured: the shader takes the colour from the image.
gpu_mats.push(GpuMaterial { albedo: [1.0, 1.0, 1.0, 1.0], emission: [0.0; 4] });
}
- PackedScene { tris: gpu_tris, materials: gpu_mats, nodes }
+ PackedScene { tris: gpu_tris, materials: gpu_mats, nodes, has_bvh: with_bvh }
+}
+
+/// A traced scene with the CPU's share of the work already done: the
+/// triangles packed into the tracer's buffer layouts and, for the compute
+/// tier, the BVH built over them. Building one is plain CPU work with no
+/// renderer in it — seconds for millions of triangles — so an app builds it
+/// on a worker thread and hands it to
+/// [`Stage3D::set_rt_scene_prepared`](crate::draw::scene::Stage3D::set_rt_scene_prepared)
+/// on the UI thread, which only uploads it. `set_rt_scene` does both in
+/// one call, on whatever thread calls it.
+///
+/// It is not consumed by the upload: keep it to upload again into a
+/// renderer rebuilt after a reconnect.
+#[derive(Clone)]
+pub struct PreparedRtScene {
+ pub(crate) packed: PackedScene,
+ pub(crate) image: Option<RtImage>,
+}
+
+impl PreparedRtScene {
+ /// Prepare `triangles` (taken, so millions of them are not copied) and
+ /// `materials`, with `image` standing in the scene as
+ /// `set_rt_scene_with_image` takes it. `with_bvh` builds the BVH: pass
+ /// the renderer's [`Stage3D::rt_needs_bvh`](crate::draw::scene::Stage3D::rt_needs_bvh),
+ /// asked on the UI thread before the work is sent off. A scene prepared
+ /// without one still traces on a renderer that needs it — the upload
+ /// builds it then, on the UI thread, as `set_rt_scene` would.
+ pub fn new(
+ triangles: Vec<RtTriangle>,
+ materials: &[RtMaterial],
+ image: Option<RtImage>,
+ with_bvh: bool,
+ ) -> Self {
+ PreparedRtScene { packed: pack_scene(triangles, materials, image.map(|i| i.corners), with_bvh), image }
+ }
+
+ /// Triangles the tracer holds, an image's quad included.
+ pub fn triangle_count(&self) -> usize {
+ self.packed.tris.len()
+ }
+
+ /// Whether the BVH was built (`with_bvh`).
+ pub fn has_bvh(&self) -> bool {
+ self.packed.has_bvh
+ }
+}
+
+/// A summary: the buffers are megabytes.
+impl std::fmt::Debug for PreparedRtScene {
+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+ f.debug_struct("PreparedRtScene")
+ .field("triangles", &self.packed.tris.len())
+ .field("materials", &self.packed.materials.len())
+ .field("bvh_nodes", &self.packed.has_bvh.then_some(self.packed.nodes.len()))
+ .field("image", &self.image)
+ .finish()
+ }
}
/// The image a frame traces, as its parameter block describes it: the
@@ -688,4 +773,39 @@ mod tests {
assert_eq!(hit, Some(0));
assert!((t - 3.0).abs() < 1e-4);
}
+
+ #[test]
+ fn test_prepared_scene_bvh_built_late_matches_built_early() {
+ // A scene prepared for the ray-query tier and uploaded to the
+ // compute one gets the BVH it would have been prepared with.
+ let tris = random_scene(400, 3);
+ let mats = [RtMaterial { albedo: [0.5; 3], emission: [0.0; 3] }; 5];
+ let image = RtImage {
+ image: 1,
+ corners: [[-1.0, 1.0, 0.0], [1.0, 1.0, 0.0], [1.0, -1.0, 0.0], [-1.0, -1.0, 0.0]],
+ opacity: 1.0,
+ };
+ let early = PreparedRtScene::new(tris.clone(), &mats, Some(image), true);
+ let late = PreparedRtScene::new(tris, &mats, Some(image), false);
+ assert!(early.has_bvh() && !late.has_bvh());
+ assert!(late.packed.nodes.is_empty());
+ assert_eq!(early.triangle_count(), 402, "the image's quad joins the scene");
+ let built = late.packed.with_bvh();
+ assert!(built.has_bvh);
+ let bytes = |s: &PackedScene| {
+ (bytemuck::cast_slice::<_, u8>(&s.tris).to_vec(), bytemuck::cast_slice::<_, u8>(&s.nodes).to_vec())
+ };
+ assert_eq!(bytes(&built), bytes(&early.packed));
+ assert!(matches!(early.packed.with_bvh(), std::borrow::Cow::Borrowed(_)));
+ }
+
+ #[test]
+ fn test_prepared_scene_crosses_threads() {
+ fn send_sync<T: Send + Sync>() {}
+ send_sync::<PreparedRtScene>();
+ let prepared = std::thread::spawn(|| PreparedRtScene::new(random_scene(50, 1), &[], None, true))
+ .join()
+ .unwrap();
+ assert_eq!(prepared.triangle_count(), 50);
+ }
}
diff --git a/src/draw/scene.rs b/src/draw/scene.rs
index c84bb08..942d599 100644
--- a/src/draw/scene.rs
+++ b/src/draw/scene.rs
@@ -160,7 +160,7 @@ pub(crate) struct SceneUniforms {
/// it with its INWARD derivative normal, said the right way round.
pub(crate) const DEFAULT_SCENE_LIGHT: [f32; 3] = [0.55, -0.45, -0.7];
-use super::rt::{RtCamera, RtEnvironment, RtImage, RtMaterial, RtTriangle};
+use super::rt::{PreparedRtScene, RtCamera, RtEnvironment, RtImage, RtMaterial, RtTriangle};
/// What an app stages a 3D scene through: the renderer's half of the
/// pass, the same on every renderer (`vk::VkRenderer`, `web::WebRenderer`).
@@ -182,14 +182,36 @@ pub trait Stage3D {
fn set_scene_light(&mut self, toward: [f32; 3]);
/// Replace the path tracer's scene (triangles in the space the camera's
- /// `inv_mvp` unprojects into); the BVH is built on the CPU. Rare: a
+ /// `inv_mvp` unprojects into); the BVH is built on the CPU, here, on the
+ /// UI thread — seconds for millions of triangles, during which the
+ /// window is frozen. Fine for small scenes; a large one is built as a
+ /// [`PreparedRtScene`] on a worker and handed to
+ /// [`set_rt_scene_prepared`](Self::set_rt_scene_prepared). Rare: a
/// geometry rebuild. Restarts the accumulation.
fn set_rt_scene(&mut self, triangles: &[RtTriangle], materials: &[RtMaterial]) {
self.set_rt_scene_with_image(triangles, materials, None);
}
/// [`set_rt_scene`](Self::set_rt_scene) with an uploaded image standing
/// in the scene (the picture the raster pass draws as a `SceneImage`).
- fn set_rt_scene_with_image(&mut self, triangles: &[RtTriangle], materials: &[RtMaterial], image: Option<RtImage>);
+ fn set_rt_scene_with_image(&mut self, triangles: &[RtTriangle], materials: &[RtMaterial], image: Option<RtImage>) {
+ let prepared = PreparedRtScene::new(triangles.to_vec(), materials, image, self.rt_needs_bvh());
+ self.set_rt_scene_prepared(&prepared);
+ }
+ /// Replace the path tracer's scene with one whose CPU work — packing,
+ /// and the BVH when [`rt_needs_bvh`](Self::rt_needs_bvh) — was done
+ /// ahead, on any thread: this only uploads it. Restarts the
+ /// accumulation. The scene is not consumed; keep it to upload again
+ /// after a reconnect (`Application::init_3d` runs again then).
+ fn set_rt_scene_prepared(&mut self, scene: &PreparedRtScene);
+ /// Whether this renderer's tracer traverses a CPU-built BVH — the
+ /// `with_bvh` to prepare its scenes with. False on the Vulkan
+ /// ray-query tier, which builds its own structure on the GPU from the
+ /// triangles; a scene prepared with a BVH still traces there (the BVH is
+ /// ignored), and one prepared without traces anywhere (the upload builds
+ /// it), so a wrong answer only costs time.
+ fn rt_needs_bvh(&self) -> bool {
+ true
+ }
/// The traced scene's sky and sun. A change restarts the accumulation.
fn set_rt_environment(&mut self, environment: RtEnvironment);
/// What a camera ray that meets nothing shows (linear RGB), or `None`
diff --git a/src/engine.rs b/src/engine.rs
index 2a4b648..c596e28 100644
--- a/src/engine.rs
+++ b/src/engine.rs
@@ -5,7 +5,7 @@
pub use crate::backend::app::{
Application, AppSender, LogicalPosition, LogicalSize, RenderContext, Stage3D, WindowAction, WindowSettings,
};
-pub use crate::draw::rt::{RtCamera, RtEnvironment, RtImage, RtMaterial, RtTriangle};
+pub use crate::draw::rt::{PreparedRtScene, RtCamera, RtEnvironment, RtImage, RtMaterial, RtTriangle};
pub use crate::draw::scene::{MeshId, SceneDraw, SceneImage, Vertex3D};
pub use crate::backend::driver::PressedKey;
pub use crate::backend::tessellate::{
diff --git a/src/vk/mod.rs b/src/vk/mod.rs
index 080bef2..d3a2e51 100644
--- a/src/vk/mod.rs
+++ b/src/vk/mod.rs
@@ -56,7 +56,7 @@ pub use image::{
};
pub use renderer::{Batch2D, Frame2D, PlatePush, VkRenderer, MAX_PLATE_FEATURES};
pub(crate) use renderer::present_debug;
-pub use rt::{RtCamera, RtEnvironment, RtImage, RtImagePixels, RtMaterial, RtOffscreen, RtTriangle};
+pub use rt::{PreparedRtScene, RtCamera, RtEnvironment, RtImage, RtImagePixels, RtMaterial, RtOffscreen, RtTriangle};
pub use scene::{MeshId, SceneDraw, SceneImage, Vertex3D};
pub use text::TextSpan;
diff --git a/src/vk/renderer.rs b/src/vk/renderer.rs
index 5d7dc3c..ed0eb88 100644
--- a/src/vk/renderer.rs
+++ b/src/vk/renderer.rs
@@ -21,7 +21,7 @@ use super::core::SurfaceLost;
use super::image::ImageStage;
pub use crate::draw::{Batch2D, Frame2D, PlatePush, MAX_PLATE_FEATURES};
pub(crate) use crate::draw::{batch_push_constants, PUSH_CONSTANT_FLOATS};
-use super::rt::{RtCamera, RtEnvironment, RtImage, RtImageSource, RtMaterial, RtStage, RtTriangle};
+use super::rt::{PreparedRtScene, RtCamera, RtEnvironment, RtImage, RtImageSource, RtMaterial, RtStage, RtTriangle};
use super::scene::{MeshId, SceneDraw, SceneImage, SceneStage, Vertex3D};
use super::text::{TextSpan, TextStage};
@@ -1787,7 +1787,9 @@ impl VkRenderer {
/// `inv_mvp` unprojects into). Builds the BVH on the CPU and uploads it;
/// waits for the GPU to go idle first — scene replacement is rare
/// (geometry rebuilds), matching `update_mesh`. The first call compiles
- /// the compute pipeline.
+ /// the compute pipeline. A large scene freezes the caller for the BVH
+ /// build: build a [`PreparedRtScene`] on a worker instead and hand it to
+ /// [`set_rt_scene_prepared`](Self::set_rt_scene_prepared).
pub fn set_rt_scene(&mut self, triangles: &[RtTriangle], materials: &[RtMaterial]) {
self.set_rt_scene_with_image(triangles, materials, None);
}
@@ -1802,6 +1804,16 @@ impl VkRenderer {
materials: &[RtMaterial],
image: Option<RtImage>,
) {
+ let prepared = PreparedRtScene::new(triangles.to_vec(), materials, image, self.rt_needs_bvh());
+ self.set_rt_scene_prepared(&prepared);
+ }
+
+ /// Replace the path tracer's scene with one prepared off this thread
+ /// ([`PreparedRtScene::new`]): only the upload — buffers written, and
+ /// on the ray-query tier the acceleration structures built on the GPU.
+ /// Waits for the GPU to go idle first, as `set_rt_scene` does. A scene
+ /// prepared without a BVH gets one built here if this renderer needs it.
+ pub fn set_rt_scene_prepared(&mut self, scene: &PreparedRtScene) {
unsafe {
let _ = self.core.device.device_wait_idle();
}
@@ -1824,12 +1836,22 @@ impl VkRenderer {
allocator,
core.queue,
core.command_pool,
- triangles,
- materials,
- image.map(|i| (RtImageSource::Shared(i.image), i.corners, i.opacity)),
+ &scene.packed,
+ scene.image.map(|i| (RtImageSource::Shared(i.image), i.corners, i.opacity)),
);
}
+ /// Whether this renderer's tracer traverses a CPU-built BVH (the compute
+ /// tier) rather than building driver acceleration structures (the
+ /// hardware ray-query tier): the `with_bvh` a [`PreparedRtScene`] for it
+ /// wants. Answered before the first scene from the device's features.
+ pub fn rt_needs_bvh(&self) -> bool {
+ match &self.rt {
+ Some(rt) => rt.needs_bvh(),
+ None => crate::vk::rt::needs_bvh(self.core.accel_loader.is_some()),
+ }
+ }
+
/// The direction TOWARD the 3D pass's light, in world space (any
/// length; zero is ignored). The flat shading reads it, and a host that
/// bakes smooth shading (`SceneDraw::prelit`) should light by the same
@@ -2743,6 +2765,12 @@ impl crate::draw::scene::Stage3D for VkRenderer {
fn set_rt_scene_with_image(&mut self, triangles: &[RtTriangle], materials: &[RtMaterial], image: Option<RtImage>) {
VkRenderer::set_rt_scene_with_image(self, triangles, materials, image)
}
+ fn set_rt_scene_prepared(&mut self, scene: &PreparedRtScene) {
+ VkRenderer::set_rt_scene_prepared(self, scene)
+ }
+ fn rt_needs_bvh(&self) -> bool {
+ VkRenderer::rt_needs_bvh(self)
+ }
fn set_rt_environment(&mut self, environment: RtEnvironment) {
VkRenderer::set_rt_environment(self, environment)
}
diff --git a/src/vk/rt.rs b/src/vk/rt.rs
index 540c597..ce1bea7 100644
--- a/src/vk/rt.rs
+++ b/src/vk/rt.rs
@@ -19,9 +19,9 @@ use gpu_allocator::MemoryLocation;
use super::renderer::{
compile_wgsl, compile_wgsl_ray_query, create_cpu_buffer, destroy_cpu_buffer, AllocatedBuffer,
};
-pub use crate::draw::rt::{RtCamera, RtEnvironment, RtImage, RtImagePixels, RtMaterial, RtTriangle};
+pub use crate::draw::rt::{PreparedRtScene, RtCamera, RtEnvironment, RtImage, RtImagePixels, RtMaterial, RtTriangle};
use crate::draw::rt::{
- denoise_params, pack_scene, rt_params, DenoiseParams, ParamImage, RtParams, DENOISE_ITERATIONS, MAX_SAMPLES,
+ denoise_params, pack_scene, rt_params, DenoiseParams, PackedScene, ParamImage, RtParams, DENOISE_ITERATIONS, MAX_SAMPLES,
WORKGROUP,
};
@@ -44,6 +44,27 @@ enum RtTier {
RayQuery,
}
+impl RtTier {
+ /// The tier a stage on a device with (`has_ray_query`) or without the
+ /// ray-query stack runs: tier 2 when it can, unless `CCE_VK_RT=compute`
+ /// forces the BVH tier.
+ fn choose(has_ray_query: bool) -> RtTier {
+ let force_compute = std::env::var("CCE_VK_RT").is_ok_and(|v| v == "compute");
+ if has_ray_query && !force_compute {
+ RtTier::RayQuery
+ } else {
+ RtTier::Compute
+ }
+ }
+}
+
+/// Whether a stage made on a device with (`has_ray_query`) or without the
+/// ray-query stack traverses a CPU-built BVH: what `Stage3D::rt_needs_bvh`
+/// answers before the stage exists.
+pub(crate) fn needs_bvh(has_ray_query: bool) -> bool {
+ RtTier::choose(has_ray_query) == RtTier::Compute
+}
+
/// Tier-2 GPU objects: one BLAS over the triangle buffer, a one-instance
/// TLAS over it. Rebuilt wholesale on every scene replacement.
struct Accel {
@@ -581,14 +602,9 @@ impl RtStage {
queue: vk::Queue,
command_pool: vk::CommandPool,
) -> Self {
- let force_compute = std::env::var("CCE_VK_RT").is_ok_and(|v| v == "compute");
let denoise_on = !std::env::var("CCE_VK_RT_DENOISE")
.is_ok_and(|v| v == "off" || v == "0" || v == "false");
- let tier = if accel_loader.is_some() && !force_compute {
- RtTier::RayQuery
- } else {
- RtTier::Compute
- };
+ let tier = RtTier::choose(accel_loader.is_some());
log::info!(
"RT stage: {} tier",
match tier {
@@ -873,28 +889,29 @@ impl RtStage {
}
}
- /// Replace the scene. Tier 1 builds the BVH on the CPU (reordering a copy
- /// of the triangles); tier 2 builds driver acceleration structures on the
- /// given queue instead. Caller must have the device idle.
- ///
- /// An image joins the scene as a quad of two triangles under a material
- /// of its own, marked textured — so both tiers meet it as they meet any
- /// triangle, and a scene that is an image alone is not an empty one.
- #[allow(clippy::too_many_arguments)]
+ /// Whether this stage traverses a CPU-built BVH (tier 1) rather than
+ /// building driver acceleration structures (tier 2).
+ pub(crate) fn needs_bvh(&self) -> bool {
+ self.tier == RtTier::Compute
+ }
+
+ /// Replace the scene with one already packed (`draw::rt::pack_scene`,
+ /// whose image quad must be the one `image` describes). Tier 1 uploads
+ /// its BVH, building it here only if it was packed without one; tier 2
+ /// builds driver acceleration structures on the given queue instead,
+ /// ignoring any BVH. Caller must have the device idle.
pub(crate) fn set_scene(
&mut self,
device: &ash::Device,
allocator: &mut Allocator,
queue: vk::Queue,
command_pool: vk::CommandPool,
- triangles: &[RtTriangle],
- materials: &[RtMaterial],
+ scene: &PackedScene,
image: Option<(RtImageSource, [[f32; 3]; 4], f32)>,
) {
if let Some(mut old) = self.image.take().and_then(|i| i.owned) {
old.destroy(device, allocator);
}
- let corners = image.as_ref().map(|(_, corners, _)| *corners);
self.image = image.map(|(source, corners, opacity)| match source {
RtImageSource::Shared(id) => {
StagedImage { shared: Some(id), owned: None, corners, opacity }
@@ -914,8 +931,11 @@ impl RtStage {
opacity,
},
});
- let packed = pack_scene(triangles, materials, corners, self.tier == RtTier::Compute);
- let (gpu_tris, gpu_mats, nodes) = (packed.tris, packed.materials, packed.nodes);
+ let scene = match self.tier {
+ RtTier::Compute => scene.with_bvh(),
+ RtTier::RayQuery => std::borrow::Cow::Borrowed(scene),
+ };
+ let (gpu_tris, gpu_mats, nodes) = (&scene.tris, &scene.materials, &scene.nodes);
self.destroy_accel(device, allocator);
for buf in [&mut self.nodes, &mut self.tris, &mut self.materials] {
@@ -949,10 +969,10 @@ impl RtStage {
| vk::BufferUsageFlags::ACCELERATION_STRUCTURE_BUILD_INPUT_READ_ONLY_KHR
}
};
- self.tris = upload(allocator, bytemuck::cast_slice(&gpu_tris), tri_usage, "rt-tris");
+ self.tris = upload(allocator, bytemuck::cast_slice(gpu_tris), tri_usage, "rt-tris");
self.materials = upload(
allocator,
- bytemuck::cast_slice(&gpu_mats),
+ bytemuck::cast_slice(gpu_mats),
vk::BufferUsageFlags::STORAGE_BUFFER,
"rt-materials",
);
@@ -964,7 +984,7 @@ impl RtStage {
RtTier::Compute => {
self.nodes = upload(
allocator,
- bytemuck::cast_slice(&nodes),
+ bytemuck::cast_slice(nodes),
vk::BufferUsageFlags::STORAGE_BUFFER,
"rt-nodes",
);
@@ -1854,6 +1874,12 @@ impl RtOffscreen {
materials: &[RtMaterial],
image: Option<RtImagePixels>,
) {
+ let packed = pack_scene(
+ triangles.to_vec(),
+ materials,
+ image.as_ref().map(|i| i.corners),
+ self.stage.needs_bvh(),
+ );
unsafe {
let _ = self.core.device.device_wait_idle();
}
@@ -1865,8 +1891,7 @@ impl RtOffscreen {
self.core.allocator.as_mut().unwrap(),
queue,
command_pool,
- triangles,
- materials,
+ &packed,
image.map(|i| {
(
RtImageSource::Pixels { pixels: i.pixels, width: i.width, height: i.height },
diff --git a/src/web/renderer.rs b/src/web/renderer.rs
index a1145c6..d0421a1 100644
--- a/src/web/renderer.rs
+++ b/src/web/renderer.rs
@@ -28,7 +28,7 @@ use std::collections::HashMap;
use super::rt::WebRt;
use super::scene::WebScene;
-use crate::draw::rt::{RtCamera, RtEnvironment, RtImage, RtMaterial, RtTriangle};
+use crate::draw::rt::{PreparedRtScene, RtCamera, RtEnvironment};
use crate::draw::scene::{MeshId, SceneDraw, SceneImage, Stage3D, Vertex3D};
use wasm_bindgen::{JsCast, JsValue};
@@ -918,7 +918,7 @@ impl Stage3D for WebRenderer {
self.scene.light = v.normalize().to_array();
}
}
- fn set_rt_scene_with_image(&mut self, triangles: &[RtTriangle], materials: &[RtMaterial], image: Option<RtImage>) {
+ fn set_rt_scene_prepared(&mut self, scene: &PreparedRtScene) {
if self.rt.is_none() {
match WebRt::new(&self.device, self.view_format) {
Ok(rt) => self.rt = Some(rt),
@@ -929,7 +929,7 @@ impl Stage3D for WebRenderer {
}
}
let rt = self.rt.as_mut().unwrap();
- if let Err(e) = rt.set_scene(&self.device, &self.queue, triangles, materials, image) {
+ if let Err(e) = rt.set_scene(&self.device, &self.queue, scene) {
web_sys::console::error_2(&"cce-ui: the traced scene was not uploaded:".into(), &e);
}
}
diff --git a/src/web/rt.rs b/src/web/rt.rs
index e0bf597..d0eab47 100644
--- a/src/web/rt.rs
+++ b/src/web/rt.rs
@@ -34,8 +34,8 @@ use web_sys::{
use super::renderer::{shader_module, texture, whole_view};
use crate::draw::rt::{
- denoise_params, pack_scene, rt_params, DenoiseParams, ParamImage, RtCamera, RtEnvironment, RtImage, RtMaterial,
- RtParams, RtTriangle, DENOISE_ITERATIONS, MAX_SAMPLES, WORKGROUP,
+ denoise_params, rt_params, DenoiseParams, ParamImage, PreparedRtScene, RtCamera, RtEnvironment, RtImage,
+ RtParams, DENOISE_ITERATIONS, MAX_SAMPLES, WORKGROUP,
};
use crate::draw::shaders::{rt_bvh_source, RT_DENOISE};
@@ -253,16 +253,11 @@ impl WebRt {
})
}
- /// Replace the scene: packed and its BVH built on the CPU, uploaded.
- pub(crate) fn set_scene(
- &mut self,
- device: &GpuDevice,
- queue: &GpuQueue,
- triangles: &[RtTriangle],
- materials: &[RtMaterial],
- image: Option<RtImage>,
- ) -> Result<(), JsValue> {
- let packed = pack_scene(triangles, materials, image.map(|i| i.corners), true);
+ /// Replace the scene with one prepared on the CPU, uploaded — its BVH
+ /// built here first if it was prepared without one.
+ pub(crate) fn set_scene(&mut self, device: &GpuDevice, queue: &GpuQueue, scene: &PreparedRtScene) -> Result<(), JsValue> {
+ let packed = scene.packed.with_bvh();
+ let image = scene.image;
let upload = |bytes: &[u8], label: &str| -> Result<GpuBuffer, JsValue> {
let b = buffer(device, bytes.len() as u32, buffer_usage::STORAGE | buffer_usage::COPY_DST, label)?;
if !bytes.is_empty() {