GPU-accelerated UI toolkit (Vulkan)
git clone https://git.lucas.co/cce-ui.git
src/vk/image.rs (49.4K)
1 //! User images in the 2D pass: upload RGBA pixels once, then draw them as
2 //! quads interleaved with the display list — 3D previews in graph nodes,
3 //! thumbnails in the files grid, any raster content in the UI.
4 //!
5 //! Upload is decoupled from the renderer because most apps never touch it
6 //! (the engine runner owns the frame): [`upload_rgba`] queues pixels from any
7 //! code and returns a stable id usable immediately in draws; the renderer
8 //! drains the queue at the next frame. [`free_image`] queues destruction the
9 //! same way.
10 //!
11 //! **Two shapes of caller.** Most upload an image once and draw it for the
12 //! rest of the process: a decoded PNG, a rasterized SVG, a thumbnail. One
13 //! uploads a *new* image every frame — cce-browser, whose whole page is a
14 //! readback of what the engine just painted. The one-shot path allocated a
15 //! fresh `VkImage` per upload and freed the old one behind a
16 //! `device_wait_idle`, which for a streaming caller meant an allocation, a
17 //! descriptor set and a full device stall per frame. [`update_pixels`] is the
18 //! streaming path: same id, same image, same descriptor, contents replaced in
19 //! place. [`recycle_buffer`] closes the loop on the CPU side by handing back
20 //! the pixel buffer the renderer has finished with, so a streaming caller
21 //! refills one buffer instead of allocating a frame-sized `Vec` per frame. Draw ordering comes from [`super::Frame2D::images`]: each
22 //! [`ImageQuad`] carries the vertex index it sorts before.
23 //!
24 //! **An image drawn much smaller than it is wants mipmaps**
25 //! ([`upload_rgba_mipmapped`]). The sampler filters between the four texels
26 //! nearest each pixel, which is every texel while the image is drawn near
27 //! its own size and one in twenty-five once it is drawn at a fifth of it: a
28 //! hairline is then on screen or not by where the sample happened to land,
29 //! and crawls as the image moves. It is asked for per image, because the
30 //! chain is a third more memory and is rebuilt on every update, and most
31 //! images — an icon, a thumbnail, a page read back at the size it is shown —
32 //! are drawn at their own size.
33
34 use std::collections::HashMap;
35
36 use ash::vk;
37 use gpu_allocator::vulkan::{Allocation, AllocationCreateDesc, AllocationScheme, Allocator};
38 use gpu_allocator::MemoryLocation;
39
40 use super::renderer::{create_cpu_buffer, destroy_cpu_buffer, AllocatedBuffer};
41 use crate::draw::images::{image_table_built, retire_buffer, take_pending, Pending};
42 pub use crate::draw::images::{
43 free_image, recycle_buffer, renderer_epoch, update_pixel_regions, update_pixels,
44 upload_pixels, upload_rgba, upload_rgba_mipmapped, ImageQuad, PixelFormat, Region,
45 };
46
47 impl PixelFormat {
48 fn vk(self) -> vk::Format {
49 // SRGB, not UNORM — see the note in `upload`.
50 match self {
51 Self::Rgba => vk::Format::R8G8B8A8_SRGB,
52 Self::Bgra => vk::Format::B8G8R8A8_SRGB,
53 }
54 }
55 }
56
57 /// How many levels a full mip chain of an image has: halved until the
58 /// longer side is one texel.
59 pub(crate) fn mip_level_count(width: u32, height: u32) -> u32 {
60 32 - width.max(height).max(1).leading_zeros()
61 }
62
63 /// An image quad's vertex: the glyph shader's, shared with the text stage.
64 type ImageVertex = crate::draw::glyphs::GlyphVertex;
65
66 struct GpuImage {
67 image: vk::Image,
68 view: vk::ImageView,
69 allocation: Option<Allocation>,
70 descriptor_set: vk::DescriptorSet,
71 /// What the image was created as, so an update can tell "same picture,
72 /// new contents" from "different image under the same id".
73 width: u32,
74 height: u32,
75 format: PixelFormat,
76 /// Levels in the image: 1 for one uploaded plain, the full chain for a
77 /// mipmapped one.
78 mip_levels: u32,
79 }
80
81 const MAX_IMAGES: u32 = 256;
82 /// Most staging one upload run holds; uploads queued back to back past it
83 /// go up as a further run. Keeps a burst of photographs from asking for a
84 /// staging buffer the size of all of them at once.
85 const RUN_STAGING_BYTES: usize = 64 << 20;
86
87 /// One image of an upload run (`ImageStage::upload_run`).
88 struct RunItem<'a> {
89 id: u32,
90 pixels: &'a [u8],
91 width: u32,
92 height: u32,
93 format: PixelFormat,
94 mips: bool,
95 }
96 /// Idle frames before the staging buffer is handed back — about two seconds
97 /// at 60 Hz. Long enough that a burst of uploads reuses one buffer, short
98 /// enough that a big one-shot upload does not hold its memory.
99 const STAGING_IDLE_FRAMES: u32 = 120;
100 /// Below this an idle staging buffer is simply kept; releasing and remaking a
101 /// small one costs more than it saves.
102 const STAGING_KEEP_BYTES: vk::DeviceSize = 1 << 20;
103
104 /// The bindings of a user image's descriptor set: the texture, then its
105 /// sampler. One definition, because the 3D scene pass binds these same sets
106 /// to a pipeline of its own (`SceneImage`), and a set is compatible with a
107 /// pipeline layout only where the two layouts are defined identically.
108 pub(crate) fn image_set_bindings() -> [vk::DescriptorSetLayoutBinding<'static>; 2] {
109 [
110 vk::DescriptorSetLayoutBinding::default()
111 .binding(0)
112 .descriptor_type(vk::DescriptorType::SAMPLED_IMAGE)
113 .descriptor_count(1)
114 .stage_flags(vk::ShaderStageFlags::FRAGMENT),
115 vk::DescriptorSetLayoutBinding::default()
116 .binding(1)
117 .descriptor_type(vk::DescriptorType::SAMPLER)
118 .descriptor_count(1)
119 .stage_flags(vk::ShaderStageFlags::FRAGMENT),
120 ]
121 }
122
123 pub(crate) struct ImageStage {
124 pipeline: vk::Pipeline,
125 pipeline_layout: vk::PipelineLayout,
126 descriptor_set_layout: vk::DescriptorSetLayout,
127 descriptor_pool: vk::DescriptorPool,
128 shader_module: vk::ShaderModule,
129 sampler: vk::Sampler,
130 /// Whether this device can build a mip chain by blitting.
131 mips_supported: bool,
132 images: HashMap<u32, GpuImage>,
133 /// One host-visible staging buffer, grown to the largest upload and kept.
134 /// Uploads are serialized against each other (each waits for its own copy
135 /// before returning), so one buffer serves them all — and a streaming
136 /// caller stops paying an allocation and a free per frame.
137 ///
138 /// Kept only while it is being used: a one-shot caller that uploads a
139 /// 96 MB photograph should not leave 96 MB of host memory mapped for the
140 /// life of the process, so an idle buffer is released (see
141 /// `STAGING_IDLE_FRAMES`). A streaming caller touches it every frame and
142 /// never reaches that.
143 staging: Option<AllocatedBuffer>,
144 /// Frames since the staging buffer was last used.
145 staging_idle: u32,
146 /// Per frame in flight: this frame's quad vertices (6 per ImageQuad).
147 frame_buffers: Vec<AllocatedBuffer>,
148 /// Whether this table takes the process-wide upload queue. True for a
149 /// window's renderer; false for a renderer that keeps images of its own
150 /// (the context menu's popup), which must not drain the queue — an
151 /// upload meant for the window, queued between the window's frame and
152 /// the popup's, would land in the popup's table and never be drawn.
153 pub(crate) shared_uploads: bool,
154 }
155
156 impl ImageStage {
157 pub(crate) fn new(
158 device: &ash::Device,
159 allocator: &mut Allocator,
160 render_pass: vk::RenderPass,
161 frames_in_flight: usize,
162 mips_supported: bool,
163 max_anisotropy: f32,
164 ) -> Self {
165 // One image table per renderer, so this is the renderer count — see
166 // `renderer_epoch`, which is what tells a cache of ids that its
167 // renderer is gone.
168 image_table_built();
169 unsafe {
170 let bindings = image_set_bindings();
171 let descriptor_set_layout = device
172 .create_descriptor_set_layout(
173 &vk::DescriptorSetLayoutCreateInfo::default().bindings(&bindings),
174 None,
175 )
176 .expect("Failed to create image descriptor set layout");
177 let set_layouts = [descriptor_set_layout];
178 let pipeline_layout = device
179 .create_pipeline_layout(
180 &vk::PipelineLayoutCreateInfo::default().set_layouts(&set_layouts),
181 None,
182 )
183 .expect("Failed to create image pipeline layout");
184
185 // Same shader as glyphs: sampled texel * vertex color (+ circle clip).
186 let spirv = super::renderer::glyph_spirv();
187 let shader_module = device
188 .create_shader_module(&vk::ShaderModuleCreateInfo::default().code(spirv), None)
189 .expect("Failed to create image shader module");
190 let stages = [
191 vk::PipelineShaderStageCreateInfo::default()
192 .stage(vk::ShaderStageFlags::VERTEX)
193 .module(shader_module)
194 .name(c"vs_main"),
195 vk::PipelineShaderStageCreateInfo::default()
196 .stage(vk::ShaderStageFlags::FRAGMENT)
197 .module(shader_module)
198 .name(c"fs_main"),
199 ];
200 let vertex_bindings = [vk::VertexInputBindingDescription::default()
201 .binding(0)
202 .stride(std::mem::size_of::<ImageVertex>() as u32)
203 .input_rate(vk::VertexInputRate::VERTEX)];
204 let vertex_attributes = [
205 vk::VertexInputAttributeDescription::default()
206 .location(0)
207 .binding(0)
208 .format(vk::Format::R32G32_SFLOAT)
209 .offset(0),
210 vk::VertexInputAttributeDescription::default()
211 .location(1)
212 .binding(0)
213 .format(vk::Format::R32G32_SFLOAT)
214 .offset(8),
215 vk::VertexInputAttributeDescription::default()
216 .location(2)
217 .binding(0)
218 .format(vk::Format::R32G32B32A32_SFLOAT)
219 .offset(16),
220 vk::VertexInputAttributeDescription::default()
221 .location(3)
222 .binding(0)
223 .format(vk::Format::R32G32B32_SFLOAT)
224 .offset(32),
225 // Location 4 is declared by the shared glyph shader; without this
226 // entry the pipeline is invalid and the attribute reads undefined.
227 vk::VertexInputAttributeDescription::default()
228 .location(4)
229 .binding(0)
230 .format(vk::Format::R32G32_SFLOAT)
231 .offset(44),
232 ];
233 let vertex_input = vk::PipelineVertexInputStateCreateInfo::default()
234 .vertex_binding_descriptions(&vertex_bindings)
235 .vertex_attribute_descriptions(&vertex_attributes);
236 let input_assembly = vk::PipelineInputAssemblyStateCreateInfo::default()
237 .topology(vk::PrimitiveTopology::TRIANGLE_LIST);
238 let viewport_state = vk::PipelineViewportStateCreateInfo::default()
239 .viewport_count(1)
240 .scissor_count(1);
241 let rasterization = vk::PipelineRasterizationStateCreateInfo::default()
242 .polygon_mode(vk::PolygonMode::FILL)
243 .cull_mode(vk::CullModeFlags::NONE)
244 .front_face(vk::FrontFace::COUNTER_CLOCKWISE)
245 .line_width(1.0);
246 let multisample = vk::PipelineMultisampleStateCreateInfo::default()
247 .rasterization_samples(vk::SampleCountFlags::TYPE_1);
248 let blend_attachments = [vk::PipelineColorBlendAttachmentState::default()
249 .blend_enable(true)
250 .src_color_blend_factor(vk::BlendFactor::SRC_ALPHA)
251 .dst_color_blend_factor(vk::BlendFactor::ONE_MINUS_SRC_ALPHA)
252 .color_blend_op(vk::BlendOp::ADD)
253 .src_alpha_blend_factor(vk::BlendFactor::ONE)
254 .dst_alpha_blend_factor(vk::BlendFactor::ONE_MINUS_SRC_ALPHA)
255 .alpha_blend_op(vk::BlendOp::ADD)
256 .color_write_mask(vk::ColorComponentFlags::RGBA)];
257 let color_blend = vk::PipelineColorBlendStateCreateInfo::default()
258 .attachments(&blend_attachments);
259 let dynamic_states = [vk::DynamicState::VIEWPORT, vk::DynamicState::SCISSOR];
260 let dynamic_state =
261 vk::PipelineDynamicStateCreateInfo::default().dynamic_states(&dynamic_states);
262 let pipeline = device
263 .create_graphics_pipelines(
264 vk::PipelineCache::null(),
265 &[vk::GraphicsPipelineCreateInfo::default()
266 .stages(&stages)
267 .vertex_input_state(&vertex_input)
268 .input_assembly_state(&input_assembly)
269 .viewport_state(&viewport_state)
270 .rasterization_state(&rasterization)
271 .multisample_state(&multisample)
272 .color_blend_state(&color_blend)
273 .dynamic_state(&dynamic_state)
274 .layout(pipeline_layout)
275 .render_pass(render_pass)
276 .subpass(0)],
277 None,
278 )
279 .expect("Failed to create image pipeline")[0];
280
281 // Linear filtering: thumbnails scale smoothly. Between mip
282 // levels too, and across every level an image has — which for
283 // an image uploaded plain is the one, so nothing changes for
284 // it. Anisotropy is for an image seen at a slant, whose long
285 // axis would otherwise be blurred to the level its short one
286 // asks for; an axis-aligned quad in the 2D pass has no slant and
287 // takes one sample as before.
288 let anisotropy = max_anisotropy.min(8.0);
289 let sampler = device
290 .create_sampler(
291 &vk::SamplerCreateInfo::default()
292 .mag_filter(vk::Filter::LINEAR)
293 .min_filter(vk::Filter::LINEAR)
294 .mipmap_mode(vk::SamplerMipmapMode::LINEAR)
295 .min_lod(0.0)
296 .max_lod(vk::LOD_CLAMP_NONE)
297 .anisotropy_enable(anisotropy > 1.0)
298 .max_anisotropy(anisotropy.max(1.0))
299 .address_mode_u(vk::SamplerAddressMode::CLAMP_TO_EDGE)
300 .address_mode_v(vk::SamplerAddressMode::CLAMP_TO_EDGE)
301 .address_mode_w(vk::SamplerAddressMode::CLAMP_TO_EDGE),
302 None,
303 )
304 .expect("Failed to create image sampler");
305
306 let pool_sizes = [
307 vk::DescriptorPoolSize::default()
308 .ty(vk::DescriptorType::SAMPLED_IMAGE)
309 .descriptor_count(MAX_IMAGES),
310 vk::DescriptorPoolSize::default()
311 .ty(vk::DescriptorType::SAMPLER)
312 .descriptor_count(MAX_IMAGES),
313 ];
314 let descriptor_pool = device
315 .create_descriptor_pool(
316 &vk::DescriptorPoolCreateInfo::default()
317 .flags(vk::DescriptorPoolCreateFlags::FREE_DESCRIPTOR_SET)
318 .max_sets(MAX_IMAGES)
319 .pool_sizes(&pool_sizes),
320 None,
321 )
322 .expect("Failed to create image descriptor pool");
323
324 let frame_buffers = (0..frames_in_flight)
325 .map(|_| {
326 create_cpu_buffer(
327 device,
328 allocator,
329 16 * 1024,
330 vk::BufferUsageFlags::VERTEX_BUFFER,
331 "image-quads",
332 )
333 })
334 .collect();
335
336 ImageStage {
337 pipeline,
338 pipeline_layout,
339 descriptor_set_layout,
340 descriptor_pool,
341 shader_module,
342 sampler,
343 mips_supported,
344 images: HashMap::new(),
345 staging: None,
346 staging_idle: 0,
347 frame_buffers,
348 shared_uploads: true,
349 }
350 }
351 }
352
353 /// Drain the global upload/free queue. Uploads are synchronous one-time
354 /// submits (rare: images load once); frees wait for device idle.
355 pub(crate) fn process_pending(
356 &mut self,
357 device: &ash::Device,
358 allocator: &mut Allocator,
359 queue: vk::Queue,
360 command_pool: vk::CommandPool,
361 ) {
362 if !self.shared_uploads {
363 return;
364 }
365 let pending: Vec<Pending> = take_pending();
366 if pending.is_empty() {
367 self.staging_idle = self.staging_idle.saturating_add(1);
368 if self.staging_idle > STAGING_IDLE_FRAMES {
369 if let Some(mut idle) = self
370 .staging
371 .take_if(|b| b.size > STAGING_KEEP_BYTES)
372 {
373 // Safe without a wait for the same reason `staging_for`
374 // needs none: every copy waits for itself before
375 // returning, so nothing is reading this buffer here.
376 destroy_cpu_buffer(device, allocator, &mut idle);
377 }
378 }
379 return;
380 }
381 self.staging_idle = 0;
382 let mut items = pending.into_iter().peekable();
383 while let Some(item) = items.next() {
384 match item {
385 Pending::Upload { id, pixels, width, height, format, mips } => {
386 // The uploads queued back to back go up together: one
387 // command buffer, one submit, one wait (`upload_run`).
388 let mut bytes = pixels.len();
389 let mut owned = vec![(id, pixels, width, height, format, mips)];
390 while let Some(Pending::Upload { pixels, .. }) = items.peek() {
391 if bytes + pixels.len() > RUN_STAGING_BYTES {
392 break;
393 }
394 bytes += pixels.len();
395 let Some(Pending::Upload { id, pixels, width, height, format, mips }) = items.next() else {
396 unreachable!()
397 };
398 owned.push((id, pixels, width, height, format, mips));
399 }
400 let run: Vec<RunItem<'_>> = owned
401 .iter()
402 .map(|(id, pixels, width, height, format, mips)| RunItem {
403 id: *id,
404 pixels,
405 width: *width,
406 height: *height,
407 format: *format,
408 mips: *mips,
409 })
410 .collect();
411 self.upload_run(device, allocator, queue, command_pool, &run);
412 drop(run);
413 for (_, pixels, ..) in owned {
414 retire_buffer(pixels);
415 }
416 }
417 Pending::Update { id, pixels, width, height, format } => {
418 // Same picture, new contents: copy into the image that is
419 // already there. Anything else about it changing (a window
420 // resize) falls back to building a fresh one under the
421 // same id.
422 let reusable = self.images.get(&id).is_some_and(|gpu| {
423 gpu.width == width && gpu.height == height && gpu.format == format
424 });
425 if reusable {
426 self.write_into(device, allocator, queue, command_pool, id, &pixels);
427 } else {
428 // Mipmapped if what it replaces was: the id is the
429 // same picture at another size.
430 let mips = self.images.get(&id).is_some_and(|gpu| gpu.mip_levels > 1);
431 self.destroy_image(device, allocator, id);
432 self.upload(
433 device, allocator, queue, command_pool, id, &pixels, width, height,
434 format, mips,
435 );
436 }
437 retire_buffer(pixels);
438 }
439 Pending::UpdateRegions { id, pixels, width, height, format, regions } => {
440 // Only into the picture these regions were cut from. A
441 // fresh image here would be blank outside them, so a
442 // mismatch writes nothing; see `update_pixel_regions`.
443 let matches = self.images.get(&id).is_some_and(|gpu| {
444 gpu.width == width && gpu.height == height && gpu.format == format
445 });
446 if matches {
447 self.write_regions(
448 device, allocator, queue, command_pool, id, &pixels, ®ions,
449 );
450 } else {
451 log::debug!("image {id}: region update for a {width}x{height} image it no longer matches, dropped");
452 }
453 retire_buffer(pixels);
454 }
455 Pending::Free { id } => {
456 // Frees queued together share one idle wait.
457 let mut ids = vec![id];
458 while let Some(Pending::Free { .. }) = items.peek() {
459 let Some(Pending::Free { id }) = items.next() else { unreachable!() };
460 ids.push(id);
461 }
462 self.destroy_images(device, allocator, &ids);
463 }
464 }
465 }
466 }
467
468 /// Tear one image down. Destroying something the GPU may still be reading
469 /// needs the device idle, which is why this is not on the per-frame path
470 /// any more: a streaming caller updates in place and never gets here until
471 /// it is finished with the image for good.
472 fn destroy_image(&mut self, device: &ash::Device, allocator: &mut Allocator, id: u32) {
473 self.destroy_images(device, allocator, &[id]);
474 }
475
476 /// Tear several images down behind ONE idle wait. Until 2026-10-06 each
477 /// free waited on its own, so leaving a folder of thumbnails ran a
478 /// device-idle per thumbnail.
479 fn destroy_images(&mut self, device: &ash::Device, allocator: &mut Allocator, ids: &[u32]) {
480 let gone: Vec<GpuImage> = ids.iter().filter_map(|id| self.images.remove(id)).collect();
481 if gone.is_empty() {
482 return;
483 }
484 unsafe {
485 let _ = device.device_wait_idle();
486 }
487 for mut gpu in gone {
488 unsafe {
489 device.destroy_image_view(gpu.view, None);
490 device.destroy_image(gpu.image, None);
491 let _ = device.free_descriptor_sets(self.descriptor_pool, &[gpu.descriptor_set]);
492 }
493 if let Some(a) = gpu.allocation.take() {
494 let _ = allocator.free(a);
495 }
496 }
497 }
498
499 /// The shared staging buffer, grown if this upload needs more room.
500 fn staging_for(
501 &mut self,
502 device: &ash::Device,
503 allocator: &mut Allocator,
504 bytes: usize,
505 ) -> &mut AllocatedBuffer {
506 let too_small = self
507 .staging
508 .as_ref()
509 .is_none_or(|b| (b.size as usize) < bytes);
510 if too_small {
511 if let Some(mut old) = self.staging.take() {
512 // No wait: every copy submitted from here is followed by
513 // `queue_wait_idle` before its caller returns, so no GPU work
514 // is referencing the old buffer by the time anything asks for
515 // a bigger one.
516 destroy_cpu_buffer(device, allocator, &mut old);
517 }
518 self.staging = Some(create_cpu_buffer(
519 device,
520 allocator,
521 bytes as vk::DeviceSize,
522 vk::BufferUsageFlags::TRANSFER_SRC,
523 "image-staging",
524 ));
525 }
526 self.staging.as_mut().expect("staging buffer")
527 }
528
529 #[allow(clippy::too_many_arguments)]
530 /// Upload RGBA8 pixels into THIS table now, rather than queueing them for
531 /// whichever renderer drains the shared queue next — for a renderer with
532 /// images of its own (see [`Self::shared_uploads`]). The id comes from
533 /// the process-wide counter, so it never collides with a queued one.
534 pub(crate) fn upload_now(
535 &mut self,
536 device: &ash::Device,
537 allocator: &mut Allocator,
538 queue: vk::Queue,
539 command_pool: vk::CommandPool,
540 pixels: &[u8],
541 width: u32,
542 height: u32,
543 ) -> u32 {
544 let id = crate::draw::images::next_image_id();
545 self.upload(device, allocator, queue, command_pool, id, pixels, width, height, PixelFormat::Rgba, false);
546 id
547 }
548
549 fn upload(
550 &mut self,
551 device: &ash::Device,
552 allocator: &mut Allocator,
553 queue: vk::Queue,
554 command_pool: vk::CommandPool,
555 id: u32,
556 pixels: &[u8],
557 width: u32,
558 height: u32,
559 format: PixelFormat,
560 mips: bool,
561 ) {
562 let one = [RunItem { id, pixels, width, height, format, mips }];
563 self.upload_run(device, allocator, queue, command_pool, &one);
564 }
565
566 /// Upload a run of images in one submission: every image's pixels
567 /// staged side by side in the shared staging buffer, every copy (and mip
568 /// chain) recorded into one command buffer, one submit, one wait. Until
569 /// 2026-10-06 each image was its own submit followed by a queue wait:
570 /// 200 thumbnails took ~70 ms of round trips in one frame.
571 fn upload_run(
572 &mut self,
573 device: &ash::Device,
574 allocator: &mut Allocator,
575 queue: vk::Queue,
576 command_pool: vk::CommandPool,
577 run: &[RunItem<'_>],
578 ) {
579 // Room in the registry, in order; what does not fit is dropped as before.
580 let room = (MAX_IMAGES as usize).saturating_sub(self.images.len());
581 for item in run.iter().skip(room) {
582 log::error!("image registry full ({MAX_IMAGES}); dropping upload {}", item.id);
583 }
584 let run = &run[..run.len().min(room)];
585 if run.is_empty() {
586 return;
587 }
588 // Each image's pixels at its own offset, 16-byte aligned (a copy's
589 // buffer offset must be a multiple of the texel size).
590 let mut offsets = Vec::with_capacity(run.len());
591 let mut total = 0usize;
592 for item in run {
593 offsets.push(total);
594 total += item.pixels.len().next_multiple_of(16);
595 }
596 let staging_buffer = {
597 let staging = self.staging_for(device, allocator, total);
598 let mapped = staging.allocation.as_mut().unwrap().mapped_slice_mut().unwrap();
599 for (item, &at) in run.iter().zip(&offsets) {
600 mapped[at..at + item.pixels.len()].copy_from_slice(item.pixels);
601 }
602 staging.buffer
603 };
604 unsafe {
605 let cmd = device
606 .allocate_command_buffers(
607 &vk::CommandBufferAllocateInfo::default()
608 .command_pool(command_pool)
609 .level(vk::CommandBufferLevel::PRIMARY)
610 .command_buffer_count(1),
611 )
612 .expect("Failed to allocate upload command buffer")[0];
613 device
614 .begin_command_buffer(
615 cmd,
616 &vk::CommandBufferBeginInfo::default()
617 .flags(vk::CommandBufferUsageFlags::ONE_TIME_SUBMIT),
618 )
619 .unwrap();
620 for (item, &at) in run.iter().zip(&offsets) {
621 self.record_upload(
622 device, allocator, cmd, staging_buffer, at as vk::DeviceSize, item.id, item.width,
623 item.height, item.format, item.mips,
624 );
625 }
626 device.end_command_buffer(cmd).unwrap();
627 let cmds = [cmd];
628 device
629 .queue_submit(queue, &[vk::SubmitInfo::default().command_buffers(&cmds)], vk::Fence::null())
630 .expect("Image upload submit failed");
631 device.queue_wait_idle(queue).expect("Image upload wait failed");
632 device.free_command_buffers(command_pool, &cmds);
633 }
634 }
635
636 /// Create one image and record its upload from `staging_buffer` at
637 /// `offset` into `cmd` (`upload_run` submits it). The view and the
638 /// descriptor set are made here too; nothing reads them before the run's
639 /// wait returns.
640 #[allow(clippy::too_many_arguments)]
641 unsafe fn record_upload(
642 &mut self,
643 device: &ash::Device,
644 allocator: &mut Allocator,
645 cmd: vk::CommandBuffer,
646 staging_buffer: vk::Buffer,
647 offset: vk::DeviceSize,
648 id: u32,
649 width: u32,
650 height: u32,
651 format: PixelFormat,
652 mips: bool,
653 ) {
654 let mip_levels =
655 if mips && self.mips_supported { mip_level_count(width, height) } else { 1 };
656 // Each level is blitted from the one above it, so a mipmapped image
657 // is a transfer's source as well as its destination.
658 let usage = if mip_levels > 1 {
659 vk::ImageUsageFlags::SAMPLED
660 | vk::ImageUsageFlags::TRANSFER_DST
661 | vk::ImageUsageFlags::TRANSFER_SRC
662 } else {
663 vk::ImageUsageFlags::SAMPLED | vk::ImageUsageFlags::TRANSFER_DST
664 };
665 unsafe {
666 let image = device
667 .create_image(
668 &vk::ImageCreateInfo::default()
669 .image_type(vk::ImageType::TYPE_2D)
670 // SRGB, not UNORM: uploaded pixels are sRGB-encoded
671 // (rasterized SVGs, decoded PNGs, Servo page readback),
672 // and the swapchain is an sRGB format, so the hardware
673 // encodes shader output on write. Sampling as UNORM
674 // fed those bytes through as if linear and encoded
675 // them a second time, lightening every midtone —
676 // a page's #101010 measured (71,71,71) on screen.
677 // Decoding on sample makes the round trip exact.
678 //
679 // Which channel comes first is the caller's business
680 // (`PixelFormat`): the sampler reads either order at
681 // no cost, so a BGRA source never needs a CPU swizzle.
682 .format(format.vk())
683 .extent(vk::Extent3D { width, height, depth: 1 })
684 .mip_levels(mip_levels)
685 .array_layers(1)
686 .samples(vk::SampleCountFlags::TYPE_1)
687 .tiling(vk::ImageTiling::OPTIMAL)
688 .usage(usage)
689 .initial_layout(vk::ImageLayout::UNDEFINED),
690 None,
691 )
692 .expect("Failed to create user image");
693 let requirements = device.get_image_memory_requirements(image);
694 let allocation = allocator
695 .allocate(&AllocationCreateDesc {
696 name: "user-image",
697 requirements,
698 location: MemoryLocation::GpuOnly,
699 linear: false,
700 allocation_scheme: AllocationScheme::GpuAllocatorManaged,
701 })
702 .expect("Failed to allocate user image memory");
703 device
704 .bind_image_memory(image, allocation.memory(), allocation.offset())
705 .expect("Failed to bind user image memory");
706
707 let range = vk::ImageSubresourceRange::default()
708 .aspect_mask(vk::ImageAspectFlags::COLOR)
709 .level_count(mip_levels)
710 .layer_count(1);
711 device.cmd_pipeline_barrier(
712 cmd,
713 vk::PipelineStageFlags::TOP_OF_PIPE,
714 vk::PipelineStageFlags::TRANSFER,
715 vk::DependencyFlags::empty(),
716 &[],
717 &[],
718 &[vk::ImageMemoryBarrier::default()
719 .src_access_mask(vk::AccessFlags::empty())
720 .dst_access_mask(vk::AccessFlags::TRANSFER_WRITE)
721 .old_layout(vk::ImageLayout::UNDEFINED)
722 .new_layout(vk::ImageLayout::TRANSFER_DST_OPTIMAL)
723 .src_queue_family_index(vk::QUEUE_FAMILY_IGNORED)
724 .dst_queue_family_index(vk::QUEUE_FAMILY_IGNORED)
725 .image(image)
726 .subresource_range(range)],
727 );
728 device.cmd_copy_buffer_to_image(
729 cmd,
730 staging_buffer,
731 image,
732 vk::ImageLayout::TRANSFER_DST_OPTIMAL,
733 &[vk::BufferImageCopy::default()
734 .buffer_offset(offset)
735 .buffer_row_length(width)
736 .buffer_image_height(height)
737 .image_subresource(
738 vk::ImageSubresourceLayers::default()
739 .aspect_mask(vk::ImageAspectFlags::COLOR)
740 .layer_count(1),
741 )
742 .image_extent(vk::Extent3D { width, height, depth: 1 })],
743 );
744 record_levels(device, cmd, image, width, height, mip_levels);
745
746 let view = device
747 .create_image_view(
748 &vk::ImageViewCreateInfo::default()
749 .image(image)
750 .view_type(vk::ImageViewType::TYPE_2D)
751 .format(format.vk())
752 .subresource_range(range),
753 None,
754 )
755 .expect("Failed to create user image view");
756
757 let set_layouts = [self.descriptor_set_layout];
758 let descriptor_set = device
759 .allocate_descriptor_sets(
760 &vk::DescriptorSetAllocateInfo::default()
761 .descriptor_pool(self.descriptor_pool)
762 .set_layouts(&set_layouts),
763 )
764 .expect("Failed to allocate image descriptor set")[0];
765 let image_infos = [vk::DescriptorImageInfo::default()
766 .image_view(view)
767 .image_layout(vk::ImageLayout::SHADER_READ_ONLY_OPTIMAL)];
768 let sampler_infos = [vk::DescriptorImageInfo::default().sampler(self.sampler)];
769 device.update_descriptor_sets(
770 &[
771 vk::WriteDescriptorSet::default()
772 .dst_set(descriptor_set)
773 .dst_binding(0)
774 .descriptor_type(vk::DescriptorType::SAMPLED_IMAGE)
775 .image_info(&image_infos),
776 vk::WriteDescriptorSet::default()
777 .dst_set(descriptor_set)
778 .dst_binding(1)
779 .descriptor_type(vk::DescriptorType::SAMPLER)
780 .image_info(&sampler_infos),
781 ],
782 &[],
783 );
784
785 self.images.insert(
786 id,
787 GpuImage {
788 image,
789 view,
790 allocation: Some(allocation),
791 descriptor_set,
792 width,
793 height,
794 format,
795 mip_levels,
796 },
797 );
798 }
799 }
800
801 /// Replace an existing image's contents in place.
802 ///
803 /// The whole streaming path. Against `upload` it skips creating an image,
804 /// allocating its memory, allocating and writing a descriptor set, and —
805 /// the expensive one — destroying last frame's image, which needs the
806 /// device idle and so waits for every frame still in flight.
807 ///
808 /// The copy is still its own submission followed by `queue_wait_idle`,
809 /// and that wait is doing real work: the image is one the *previous*
810 /// frame may still be sampling, and two submissions on one queue are not
811 /// ordered against each other by anything weaker. Lifting it means giving
812 /// each streaming image a second buffer to alternate between and
813 /// recording the copy into the frame's own command buffer, which is the
814 /// next step rather than this one.
815 fn write_into(
816 &mut self,
817 device: &ash::Device,
818 allocator: &mut Allocator,
819 queue: vk::Queue,
820 command_pool: vk::CommandPool,
821 id: u32,
822 pixels: &[u8],
823 ) {
824 let Some(&GpuImage { width, height, .. }) = self.images.get(&id) else {
825 return;
826 };
827 self.write_regions(device, allocator, queue, command_pool, id, pixels, &[(0, 0, width, height)]);
828 }
829
830 /// [`Self::write_into`] for part of the image: each region's texels,
831 /// packed one after another in `pixels`, copied into place in one
832 /// submission. Everything outside the regions keeps what it held.
833 #[allow(clippy::too_many_arguments)]
834 fn write_regions(
835 &mut self,
836 device: &ash::Device,
837 allocator: &mut Allocator,
838 queue: vk::Queue,
839 command_pool: vk::CommandPool,
840 id: u32,
841 pixels: &[u8],
842 regions: &[Region],
843 ) {
844 let Some(&GpuImage { image, width, height, mip_levels, .. }) = self.images.get(&id) else {
845 return;
846 };
847 let mut offset = 0u64;
848 let copies: Vec<vk::BufferImageCopy> = regions
849 .iter()
850 .map(|&(x, y, w, h)| {
851 let copy = vk::BufferImageCopy::default()
852 .buffer_offset(offset)
853 .buffer_row_length(w)
854 .buffer_image_height(h)
855 .image_subresource(
856 vk::ImageSubresourceLayers::default()
857 .aspect_mask(vk::ImageAspectFlags::COLOR)
858 .layer_count(1),
859 )
860 .image_offset(vk::Offset3D { x: x as i32, y: y as i32, z: 0 })
861 .image_extent(vk::Extent3D { width: w, height: h, depth: 1 });
862 offset += (w * h * 4) as u64;
863 copy
864 })
865 .collect();
866 unsafe {
867 let staging_buffer = {
868 let staging = self.staging_for(device, allocator, pixels.len());
869 staging.allocation.as_mut().unwrap().mapped_slice_mut().unwrap()[..pixels.len()]
870 .copy_from_slice(pixels);
871 staging.buffer
872 };
873 let range = vk::ImageSubresourceRange::default()
874 .aspect_mask(vk::ImageAspectFlags::COLOR)
875 .level_count(mip_levels)
876 .layer_count(1);
877 let cmd = device
878 .allocate_command_buffers(
879 &vk::CommandBufferAllocateInfo::default()
880 .command_pool(command_pool)
881 .level(vk::CommandBufferLevel::PRIMARY)
882 .command_buffer_count(1),
883 )
884 .expect("Failed to allocate update command buffer")[0];
885 device
886 .begin_command_buffer(
887 cmd,
888 &vk::CommandBufferBeginInfo::default()
889 .flags(vk::CommandBufferUsageFlags::ONE_TIME_SUBMIT),
890 )
891 .unwrap();
892 // Unlike a fresh upload this image holds a picture already, and
893 // it is in the layout the shader reads. Transitioning *from* that
894 // layout (not UNDEFINED) keeps the contents, which a region
895 // update depends on: everything outside its regions must survive.
896 device.cmd_pipeline_barrier(
897 cmd,
898 vk::PipelineStageFlags::FRAGMENT_SHADER,
899 vk::PipelineStageFlags::TRANSFER,
900 vk::DependencyFlags::empty(),
901 &[],
902 &[],
903 &[vk::ImageMemoryBarrier::default()
904 .src_access_mask(vk::AccessFlags::SHADER_READ)
905 .dst_access_mask(vk::AccessFlags::TRANSFER_WRITE)
906 .old_layout(vk::ImageLayout::SHADER_READ_ONLY_OPTIMAL)
907 .new_layout(vk::ImageLayout::TRANSFER_DST_OPTIMAL)
908 .src_queue_family_index(vk::QUEUE_FAMILY_IGNORED)
909 .dst_queue_family_index(vk::QUEUE_FAMILY_IGNORED)
910 .image(image)
911 .subresource_range(range)],
912 );
913 device.cmd_copy_buffer_to_image(
914 cmd,
915 staging_buffer,
916 image,
917 vk::ImageLayout::TRANSFER_DST_OPTIMAL,
918 &copies,
919 );
920 record_levels(device, cmd, image, width, height, mip_levels);
921 device.end_command_buffer(cmd).unwrap();
922 let cmds = [cmd];
923 device
924 .queue_submit(
925 queue,
926 &[vk::SubmitInfo::default().command_buffers(&cmds)],
927 vk::Fence::null(),
928 )
929 .expect("Image update submit failed");
930 device.queue_wait_idle(queue).expect("Image update wait failed");
931 device.free_command_buffers(command_pool, &cmds);
932 }
933 }
934
935 /// After the frame fence: build this frame's quad vertices (6 per image,
936 /// in `images` order).
937 pub(crate) fn write_frame_buffer(
938 &mut self,
939 device: &ash::Device,
940 allocator: &mut Allocator,
941 frame_index: usize,
942 images: &[ImageQuad],
943 extent: vk::Extent2D,
944 ) {
945 let verts = crate::draw::glyphs::image_quad_vertices(images, extent.width, extent.height);
946 let bytes: &[u8] = bytemuck::cast_slice(&verts);
947 let buf = &mut self.frame_buffers[frame_index];
948 if bytes.len() as vk::DeviceSize > buf.size {
949 let mut old = std::mem::replace(buf, AllocatedBuffer::null());
950 destroy_cpu_buffer(device, allocator, &mut old);
951 *buf = create_cpu_buffer(
952 device,
953 allocator,
954 (bytes.len() as vk::DeviceSize).next_power_of_two(),
955 vk::BufferUsageFlags::VERTEX_BUFFER,
956 "image-quads",
957 );
958 }
959 if !bytes.is_empty() {
960 buf.allocation.as_mut().unwrap().mapped_slice_mut().unwrap()[..bytes.len()]
961 .copy_from_slice(bytes);
962 }
963 }
964
965 /// The descriptor set an uploaded image is drawn with, or None while its
966 /// upload has not landed (or after it was freed).
967 pub(crate) fn descriptor_set(&self, image_id: u32) -> Option<vk::DescriptorSet> {
968 self.images.get(&image_id).map(|gpu| gpu.descriptor_set)
969 }
970
971 /// An uploaded image's view and size, for a pass that samples it under
972 /// bindings of its own (the path tracer). None while its upload has not
973 /// landed.
974 pub(crate) fn view_and_size(&self, image_id: u32) -> Option<(vk::ImageView, u32, u32)> {
975 self.images.get(&image_id).map(|gpu| (gpu.view, gpu.width, gpu.height))
976 }
977
978 /// Record one image quad (index `i` of this frame's list). The caller
979 /// restores its own pipeline/scissor state afterwards. Returns false if the
980 /// image hasn't finished uploading (draw skipped).
981 pub(crate) fn record_quad(
982 &self,
983 device: &ash::Device,
984 cmd: vk::CommandBuffer,
985 frame_index: usize,
986 i: usize,
987 image_id: u32,
988 ) -> bool {
989 let Some(gpu) = self.images.get(&image_id) else {
990 return false;
991 };
992 unsafe {
993 device.cmd_bind_pipeline(cmd, vk::PipelineBindPoint::GRAPHICS, self.pipeline);
994 device.cmd_bind_descriptor_sets(
995 cmd,
996 vk::PipelineBindPoint::GRAPHICS,
997 self.pipeline_layout,
998 0,
999 &[gpu.descriptor_set],
1000 &[],
1001 );
1002 device.cmd_bind_vertex_buffers(cmd, 0, &[self.frame_buffers[frame_index].buffer], &[0]);
1003 device.cmd_draw(cmd, 6, 1, (i * 6) as u32, 0);
1004 }
1005 true
1006 }
1007
1008 pub(crate) fn destroy(&mut self, device: &ash::Device, allocator: &mut Allocator) {
1009 unsafe {
1010 for (_, mut gpu) in self.images.drain() {
1011 device.destroy_image_view(gpu.view, None);
1012 device.destroy_image(gpu.image, None);
1013 if let Some(a) = gpu.allocation.take() {
1014 let _ = allocator.free(a);
1015 }
1016 }
1017 for buf in &mut self.frame_buffers {
1018 let mut b = std::mem::replace(buf, AllocatedBuffer::null());
1019 destroy_cpu_buffer(device, allocator, &mut b);
1020 }
1021 // The upload staging buffer, kept between uploads. Missed here
1022 // until 2026-10-05: gpu-allocator reported it leaked whenever a
1023 // renderer that had uploaded an image was dropped (a reconnect,
1024 // or a layer app hiding its surface).
1025 if let Some(mut staging) = self.staging.take() {
1026 destroy_cpu_buffer(device, allocator, &mut staging);
1027 }
1028 device.destroy_sampler(self.sampler, None);
1029 device.destroy_descriptor_pool(self.descriptor_pool, None);
1030 device.destroy_descriptor_set_layout(self.descriptor_set_layout, None);
1031 device.destroy_pipeline(self.pipeline, None);
1032 device.destroy_pipeline_layout(self.pipeline_layout, None);
1033 device.destroy_shader_module(self.shader_module, None);
1034 }
1035 }
1036 }
1037
1038 /// Finish an image whose level 0 has just been copied into and whose every
1039 /// level is in TRANSFER_DST: build each further level from the one above
1040 /// it, and leave them all in the layout the shader reads.
1041 ///
1042 /// A blit, filtered linearly, halves a level into the next: each texel of
1043 /// the smaller is the mean of the four it covers. The formats are sRGB, so
1044 /// the mean is taken of the light and not of its encoding — a level of a
1045 /// black and white check is the grey that looks half as bright, where a mean
1046 /// of the bytes is darker than that.
1047 ///
1048 /// With one level there is nothing to build and this is the transition the
1049 /// plain upload always ended on.
1050 unsafe fn record_levels(
1051 device: &ash::Device,
1052 cmd: vk::CommandBuffer,
1053 image: vk::Image,
1054 width: u32,
1055 height: u32,
1056 mip_levels: u32,
1057 ) {
1058 let level = |i: u32| {
1059 vk::ImageSubresourceRange::default()
1060 .aspect_mask(vk::ImageAspectFlags::COLOR)
1061 .base_mip_level(i)
1062 .level_count(1)
1063 .layer_count(1)
1064 };
1065 let barrier = |range, from_access, to_access, from_layout, to_layout, from_stage, to_stage| {
1066 device.cmd_pipeline_barrier(
1067 cmd,
1068 from_stage,
1069 to_stage,
1070 vk::DependencyFlags::empty(),
1071 &[],
1072 &[],
1073 &[vk::ImageMemoryBarrier::default()
1074 .src_access_mask(from_access)
1075 .dst_access_mask(to_access)
1076 .old_layout(from_layout)
1077 .new_layout(to_layout)
1078 .src_queue_family_index(vk::QUEUE_FAMILY_IGNORED)
1079 .dst_queue_family_index(vk::QUEUE_FAMILY_IGNORED)
1080 .image(image)
1081 .subresource_range(range)],
1082 );
1083 };
1084 let (mut w, mut h) = (width as i32, height as i32);
1085 for i in 1..mip_levels {
1086 let (next_w, next_h) = ((w / 2).max(1), (h / 2).max(1));
1087 barrier(
1088 level(i - 1),
1089 vk::AccessFlags::TRANSFER_WRITE,
1090 vk::AccessFlags::TRANSFER_READ,
1091 vk::ImageLayout::TRANSFER_DST_OPTIMAL,
1092 vk::ImageLayout::TRANSFER_SRC_OPTIMAL,
1093 vk::PipelineStageFlags::TRANSFER,
1094 vk::PipelineStageFlags::TRANSFER,
1095 );
1096 let layers = |i: u32| {
1097 vk::ImageSubresourceLayers::default()
1098 .aspect_mask(vk::ImageAspectFlags::COLOR)
1099 .mip_level(i)
1100 .layer_count(1)
1101 };
1102 device.cmd_blit_image(
1103 cmd,
1104 image,
1105 vk::ImageLayout::TRANSFER_SRC_OPTIMAL,
1106 image,
1107 vk::ImageLayout::TRANSFER_DST_OPTIMAL,
1108 &[vk::ImageBlit::default()
1109 .src_subresource(layers(i - 1))
1110 .src_offsets([vk::Offset3D::default(), vk::Offset3D { x: w, y: h, z: 1 }])
1111 .dst_subresource(layers(i))
1112 .dst_offsets([
1113 vk::Offset3D::default(),
1114 vk::Offset3D { x: next_w, y: next_h, z: 1 },
1115 ])],
1116 vk::Filter::LINEAR,
1117 );
1118 barrier(
1119 level(i - 1),
1120 vk::AccessFlags::TRANSFER_READ,
1121 vk::AccessFlags::SHADER_READ,
1122 vk::ImageLayout::TRANSFER_SRC_OPTIMAL,
1123 vk::ImageLayout::SHADER_READ_ONLY_OPTIMAL,
1124 vk::PipelineStageFlags::TRANSFER,
1125 vk::PipelineStageFlags::FRAGMENT_SHADER,
1126 );
1127 (w, h) = (next_w, next_h);
1128 }
1129 barrier(
1130 level(mip_levels - 1),
1131 vk::AccessFlags::TRANSFER_WRITE,
1132 vk::AccessFlags::SHADER_READ,
1133 vk::ImageLayout::TRANSFER_DST_OPTIMAL,
1134 vk::ImageLayout::SHADER_READ_ONLY_OPTIMAL,
1135 vk::PipelineStageFlags::TRANSFER,
1136 vk::PipelineStageFlags::FRAGMENT_SHADER,
1137 );
1138 }
1139
1140 #[cfg(test)]
1141 mod tests {
1142 use super::mip_level_count;
1143
1144 /// A full chain halves the longer side down to one texel.
1145 #[test]
1146 fn a_mip_chain_ends_at_one_texel() {
1147 assert_eq!(mip_level_count(1, 1), 1);
1148 assert_eq!(mip_level_count(2, 1), 2);
1149 assert_eq!(mip_level_count(256, 256), 9);
1150 assert_eq!(mip_level_count(257, 3), 9);
1151 assert_eq!(mip_level_count(2550, 3300), 12);
1152 assert_eq!(mip_level_count(0, 0), 1);
1153 }
1154 }