git.lucas.co / cce-ui
GPU-accelerated UI toolkit (Vulkan)
git clone https://git.lucas.co/cce-ui.git

commit958aa7500132c49ae1b3d07d39e4d9bf0db3ff7a
parentcabb234bdf
authorLucas Galante <lsgalante12@gmail.com>
date2026-10-07 22:44
perf(vk): a mesh update no longer waits for the device to go idle

update_mesh / update_lit_mesh write into a spare buffer no submitted frame
still reads (or a new one), and the replaced buffer becomes a spare tagged
with the frames submitted so far; after each fence wait the spares those
frames read are free again, two kept a mesh. A playing simulation updates
its meshes every frame and paid the wait each time: the designer's stage
pass at 57k points went from about 5.8 ms to 3.7.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

 CLAUDE.md          |  14 +++++
 src/vk/renderer.rs |  19 +++----
 src/vk/scene.rs    | 146 ++++++++++++++++++++++++++++++++++++-----------------
 3 files changed, 124 insertions(+), 55 deletions(-)

diff --git a/CLAUDE.md b/CLAUDE.md
index f1d9491..a64d330 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -212,6 +212,20 @@ image in the scene before the translucent draw, a host light, frost over the pan
 195 px differ by more than 8 levels, all on 1 px wires (where along its length a line
 steps a row is the rasterizer's), everything else within 2.
 
+**A mesh update does not wait for the GPU** (since 2026-10-07,
+`SceneStage::update_mesh` / `update_lit_mesh` through `Mesh::replace`). It
+called `device_wait_idle` first, reasoning that geometry updates are rare;
+a playing simulation updates several meshes every frame, and the wait made
+each frame's upload wait out the previous frame's GPU work. Now the new
+vertices go into another buffer — a spare of the mesh's that no submitted
+frame still reads, or a new one — and the replaced buffer becomes a spare
+tagged with the frames submitted so far (`SceneStage::submitted`, counted
+at each submit). After each frame-slot fence wait, `frame_waited` knows
+which frames have finished, releases spares too small for their mesh and
+keeps two of the rest, so a steady playback allocates nothing. Measured in
+the designer's replay at 57k points: the stage pass of a drawn frame 5.6–6.0 ms
+to 3.6–3.8.
+
 **A scene draw can be instanced** (since 2026-10-06, `SceneDraw::instances`). A draw
 naming an instance mesh draws its `mesh` once per vertex of that mesh: an instance is a
 `Vertex3D` read as an offset added to every vertex and a colour multiplying theirs, so a
diff --git a/src/vk/renderer.rs b/src/vk/renderer.rs
index 2500640..cd68810 100644
--- a/src/vk/renderer.rs
+++ b/src/vk/renderer.rs
@@ -1724,13 +1724,12 @@ impl VkRenderer {
             .create_mesh(&self.core.device, self.core.allocator.as_mut().unwrap(), verts)
     }
 
-    /// Replace a mesh's vertices. Waits for the GPU to go idle first — geometry
-    /// updates are rare (settings changes, graph rebuilds), matching the app.
-    #[allow(dead_code)] // cutover API: the app's rebuild_scene_geometry path
+    /// Replace a mesh's vertices, without waiting for the GPU: the frames
+    /// in flight keep the buffer they read, and the new vertices go into
+    /// another (`SceneStage::update_mesh`). It waited for the device to go
+    /// idle until 2026-10-07, which every frame of a playing simulation
+    /// paid.
     pub fn update_mesh(&mut self, id: MeshId, verts: &[Vertex3D]) {
-        unsafe {
-            let _ = self.core.device.device_wait_idle();
-        }
         self.scene
             .update_mesh(&self.core.device, self.core.allocator.as_mut().unwrap(), id, verts);
     }
@@ -1795,9 +1794,7 @@ impl VkRenderer {
 
     /// Replace a lit mesh's vertices; waits for the GPU first, as `update_mesh`.
     pub fn update_lit_mesh(&mut self, id: crate::draw::lit::LitMeshId, verts: &[crate::draw::lit::LitVertex]) {
-        unsafe {
-            let _ = self.core.device.device_wait_idle();
-        }
+        // No wait for the device: as `update_mesh`.
         self.scene.update_lit_mesh(&self.core.device, self.core.allocator.as_mut().unwrap(), id, verts);
     }
 
@@ -2137,6 +2134,9 @@ impl VkRenderer {
             self.core.device
                 .wait_for_fences(&[in_flight], true, u64::MAX)
                 .expect("Fence wait failed");
+            // What this slot last carried has finished: the mesh buffers an
+            // update replaced while those frames read them may be reused.
+            self.scene.frame_waited(&self.core.device, self.core.allocator.as_mut().unwrap(), FRAMES_IN_FLIGHT as u64);
 
             // The first frame with a blur plate that will copy the frame so
             // far allocates the snapshot it copies into. Decided the way the
@@ -2700,6 +2700,7 @@ impl VkRenderer {
             self.core.device
                 .queue_submit(self.core.queue, &[submit], in_flight)
                 .expect("Queue submit failed");
+            self.scene.submitted += 1;
 
             // The image now holds this frame; every other image fell behind
             // by this frame's damage.
diff --git a/src/vk/scene.rs b/src/vk/scene.rs
index 9ca2ed3..b155d5e 100644
--- a/src/vk/scene.rs
+++ b/src/vk/scene.rs
@@ -31,6 +31,77 @@ const SLOT_SIZE: vk::DeviceSize = if (LIT_UNIFORM_SIZE as vk::DeviceSize) > UNIF
 struct Mesh {
     buffer: AllocatedBuffer,
     count: u32,
+    /// Buffers this mesh held before, each tagged with the frames that may
+    /// still read it: every frame submitted before the tag. An update takes
+    /// one no frame still reads in place of waiting for the device to go
+    /// idle ([`Mesh::replace`]).
+    spare: Vec<(AllocatedBuffer, u64)>,
+}
+
+impl Mesh {
+    fn new(buffer: AllocatedBuffer, count: u32) -> Self {
+        Mesh { buffer, count, spare: Vec::new() }
+    }
+
+    /// Put `bytes` (`count` vertices) in place of what the mesh holds,
+    /// without waiting for the GPU. The frames already submitted may still
+    /// read the mesh's buffer, so the bytes go into another — a spare no
+    /// submitted frame still reads (`tag <= complete`) and big enough, or a
+    /// new one — and the buffer they replace becomes a spare tagged `tag`,
+    /// the frames submitted so far. Until 2026-10-07 an update waited for
+    /// the device to go idle, reasoning that geometry updates are rare; a
+    /// playing simulation updates its meshes every frame, and the wait put
+    /// the CPU's work and the GPU's end to end, so a frame took both.
+    fn replace(
+        &mut self,
+        device: &ash::Device,
+        allocator: &mut Allocator,
+        bytes: &[u8],
+        count: u32,
+        (tag, complete): (u64, u64),
+        label: &'static str,
+    ) {
+        let needed = (bytes.len() as vk::DeviceSize).max(64);
+        let free = self.spare.iter().position(|(b, t)| *t <= complete && b.size >= needed);
+        let fresh = match free {
+            Some(i) => self.spare.swap_remove(i).0,
+            None => create_cpu_buffer(device, allocator, needed.next_power_of_two(), vk::BufferUsageFlags::VERTEX_BUFFER, label),
+        };
+        let old = std::mem::replace(&mut self.buffer, fresh);
+        self.spare.push((old, tag));
+        if !bytes.is_empty() {
+            self.buffer.allocation.as_mut().unwrap().mapped_slice_mut().unwrap()[..bytes.len()].copy_from_slice(bytes);
+        }
+        self.count = count;
+    }
+
+    /// Release the spares no frame still reads that are too small for the
+    /// mesh as it stands, and keep two of the rest: a steady playback reuses
+    /// them and allocates nothing.
+    fn reclaim(&mut self, device: &ash::Device, allocator: &mut Allocator, complete: u64) {
+        let current = self.buffer.size;
+        let mut kept = 0;
+        let mut i = 0;
+        while i < self.spare.len() {
+            let (size, tag) = (self.spare[i].0.size, self.spare[i].1);
+            let free = tag <= complete;
+            if free && (size < current || kept >= 2) {
+                let (mut buffer, _) = self.spare.swap_remove(i);
+                destroy_cpu_buffer(device, allocator, &mut buffer);
+                continue;
+            }
+            kept += free as usize;
+            i += 1;
+        }
+    }
+
+    fn destroy(&mut self, device: &ash::Device, allocator: &mut Allocator) {
+        let mut buffer = std::mem::replace(&mut self.buffer, AllocatedBuffer::null());
+        destroy_cpu_buffer(device, allocator, &mut buffer);
+        for (mut spare, _) in self.spare.drain(..) {
+            destroy_cpu_buffer(device, allocator, &mut spare);
+        }
+    }
 }
 
 struct StagedScene {
@@ -96,6 +167,12 @@ pub(crate) struct SceneStage {
     framebuffer: vk::Framebuffer,
 
     meshes: Vec<Mesh>,
+    /// Frames submitted so far, counted by the renderer as it submits them.
+    pub(crate) submitted: u64,
+    /// Every frame numbered below this one has finished on the GPU, as the
+    /// renderer learns by waiting on a frame slot's fence
+    /// ([`SceneStage::frame_waited`]).
+    complete_before: u64,
     /// The one instance a draw without instances is drawn with
     /// ([`UNIT_INSTANCE`]), bound in binding 1 in place of an instance mesh.
     unit_instance: AllocatedBuffer,
@@ -672,6 +749,8 @@ impl SceneStage {
                 depth_allocation: None,
                 framebuffer: vk::Framebuffer::null(),
                 meshes: Vec::new(),
+                submitted: 0,
+                complete_before: 0,
                 unit_instance,
                 frames,
                 staged: None,
@@ -908,14 +987,12 @@ impl SceneStage {
             buffer.allocation.as_mut().unwrap().mapped_slice_mut().unwrap()[..bytes.len()]
                 .copy_from_slice(bytes);
         }
-        self.meshes.push(Mesh { buffer, count: verts.len() as u32 });
+        self.meshes.push(Mesh::new(buffer, verts.len() as u32));
         MeshId(self.meshes.len() - 1)
     }
 
-    /// Replace a mesh's vertices. Caller must have the device idle: meshes may be
-    /// referenced by in-flight frames (geometry updates are rare — settings
-    /// changes and graph rebuilds — so a wait is acceptable here).
-    #[allow(dead_code)] // cutover API: the app's rebuild_scene_geometry path
+    /// Replace a mesh's vertices without waiting for the GPU: the frames in
+    /// flight keep the buffer they read ([`Mesh::replace`]).
     pub(crate) fn update_mesh(
         &mut self,
         device: &ash::Device,
@@ -923,25 +1000,19 @@ impl SceneStage {
         id: MeshId,
         verts: &[Vertex3D],
     ) {
-        let mesh = &mut self.meshes[id.0];
-        let bytes: &[u8] = bytemuck::cast_slice(verts);
-        let needed = bytes.len() as vk::DeviceSize;
-        if needed > mesh.buffer.size {
-            let mut old = std::mem::replace(&mut mesh.buffer, AllocatedBuffer::null());
-            destroy_cpu_buffer(device, allocator, &mut old);
-            mesh.buffer = create_cpu_buffer(
-                device,
-                allocator,
-                needed.next_power_of_two(),
-                vk::BufferUsageFlags::VERTEX_BUFFER,
-                "mesh",
-            );
-        }
-        if !bytes.is_empty() {
-            mesh.buffer.allocation.as_mut().unwrap().mapped_slice_mut().unwrap()[..bytes.len()]
-                .copy_from_slice(bytes);
+        let frames = (self.submitted, self.complete_before);
+        self.meshes[id.0].replace(device, allocator, bytemuck::cast_slice(verts), verts.len() as u32, frames, "mesh");
+    }
+
+    /// After the renderer has waited on the fence of the slot the next frame
+    /// will use: the frame that slot last carried has finished, and every
+    /// frame before it, so the spares those frames read may be reused.
+    pub(crate) fn frame_waited(&mut self, device: &ash::Device, allocator: &mut Allocator, frames_in_flight: u64) {
+        self.complete_before = (self.submitted + 1).saturating_sub(frames_in_flight);
+        let complete = self.complete_before;
+        for mesh in self.meshes.iter_mut().chain(self.lit_meshes.iter_mut()) {
+            mesh.reclaim(device, allocator, complete);
         }
-        mesh.count = verts.len() as u32;
     }
 
     pub(crate) fn stage(&mut self, scissor: (u32, u32, u32, u32), draws: Vec<SceneDraw>) {
@@ -967,31 +1038,15 @@ impl SceneStage {
         if !bytes.is_empty() {
             buffer.allocation.as_mut().unwrap().mapped_slice_mut().unwrap()[..bytes.len()].copy_from_slice(bytes);
         }
-        self.lit_meshes.push(Mesh { buffer, count: verts.len() as u32 });
+        self.lit_meshes.push(Mesh::new(buffer, verts.len() as u32));
         LitMeshId(self.lit_meshes.len() - 1)
     }
 
-    /// Replace a lit mesh's vertices. Caller must have the device idle, as
-    /// for `update_mesh`.
+    /// Replace a lit mesh's vertices without waiting for the GPU, as
+    /// `update_mesh` does.
     pub(crate) fn update_lit_mesh(&mut self, device: &ash::Device, allocator: &mut Allocator, id: LitMeshId, verts: &[LitVertex]) {
-        let mesh = &mut self.lit_meshes[id.0];
-        let bytes: &[u8] = bytemuck::cast_slice(verts);
-        let needed = bytes.len() as vk::DeviceSize;
-        if needed > mesh.buffer.size {
-            let mut old = std::mem::replace(&mut mesh.buffer, AllocatedBuffer::null());
-            destroy_cpu_buffer(device, allocator, &mut old);
-            mesh.buffer = create_cpu_buffer(
-                device,
-                allocator,
-                needed.next_power_of_two(),
-                vk::BufferUsageFlags::VERTEX_BUFFER,
-                "lit-mesh",
-            );
-        }
-        if !bytes.is_empty() {
-            mesh.buffer.allocation.as_mut().unwrap().mapped_slice_mut().unwrap()[..bytes.len()].copy_from_slice(bytes);
-        }
-        mesh.count = verts.len() as u32;
+        let frames = (self.submitted, self.complete_before);
+        self.lit_meshes[id.0].replace(device, allocator, bytemuck::cast_slice(verts), verts.len() as u32, frames, "lit-mesh");
     }
 
     /// The staged scene's images; nothing when no scene is staged.
@@ -1282,8 +1337,7 @@ impl SceneStage {
                 destroy_cpu_buffer(device, allocator, &mut quads);
             }
             for mesh in self.meshes.iter_mut().chain(self.lit_meshes.iter_mut()) {
-                let mut buffer = std::mem::replace(&mut mesh.buffer, AllocatedBuffer::null());
-                destroy_cpu_buffer(device, allocator, &mut buffer);
+                mesh.destroy(device, allocator);
             }
             let mut unit = std::mem::replace(&mut self.unit_instance, AllocatedBuffer::null());
             destroy_cpu_buffer(device, allocator, &mut unit);