rlx_core/render/capture.rs
1//! Headless offscreen capture: draw into a texture with no window and read the
2//! pixels back as tight RGBA (Plan 0013). Dev/agent tooling over the native
3//! Rust API — no dependency, no present.
4//!
5//! **Not the hot path.** The readback blocks (`map_async` + `poll(Wait)`); it is
6//! only ever driven by capture/QA tooling, never wired into the live `render`
7//! loop (see CLAUDE.md real-time rules). The panic-denial pragma below is kept
8//! anyway so every file under `render/` satisfies the hygiene guard.
9
10#![deny(
11 clippy::unwrap_used,
12 clippy::expect_used,
13 clippy::indexing_slicing,
14 clippy::panic,
15 clippy::unreachable
16)]
17
18use super::RenderError;
19use crate::render::gpu;
20
21/// Bytes per pixel of [`HEADLESS_FORMAT`](super::context::HEADLESS_FORMAT).
22const BYTES_PER_PIXEL: u32 = 4;
23
24/// Bytes per pixel of [`COMPOSITE_FORMAT`](super::COMPOSITE_FORMAT) — four
25/// 16-bit halves. Only the linear readback below reads it, and that is test-only.
26#[cfg(test)]
27const LINEAR_BYTES_PER_PIXEL: u32 = 8;
28
29/// A captured frame: tight (row-unpadded) `Rgba8UnormSrgb` pixels, row-major
30/// top-to-bottom. `rgba.len() == width * height * 4`.
31#[derive(Clone)]
32pub struct CaptureImage {
33 /// Image width in pixels.
34 pub width: u32,
35 /// Image height in pixels.
36 pub height: u32,
37 /// `width * height * 4` bytes, RGBA8, no row padding.
38 pub rgba: Vec<u8>,
39}
40
41/// A `RENDER_ATTACHMENT | COPY_SRC` texture sized `width`×`height` plus a view,
42/// the offscreen draw target for one capture.
43pub(crate) fn create_target(
44 device: &wgpu::Device,
45 format: wgpu::TextureFormat,
46 width: u32,
47 height: u32,
48) -> (wgpu::Texture, wgpu::TextureView) {
49 let texture = device.create_texture(&wgpu::TextureDescriptor {
50 label: Some("rlx-capture-target"),
51 size: wgpu::Extent3d {
52 width,
53 height,
54 depth_or_array_layers: 1,
55 },
56 mip_level_count: 1,
57 sample_count: 1,
58 dimension: wgpu::TextureDimension::D2,
59 format,
60 // `COPY_DST` alongside the two it is drawn and read through, so this
61 // target can stand in for a swapchain image on the preview path — where
62 // the frame is drawn into an intermediate and reaches its destination by
63 // `copy_texture_to_texture`. Without it the capture paths could not
64 // exercise that copy at all, and the only instrument left for it would
65 // need a real window.
66 usage: wgpu::TextureUsages::RENDER_ATTACHMENT
67 | wgpu::TextureUsages::COPY_SRC
68 | wgpu::TextureUsages::COPY_DST,
69 view_formats: &[],
70 });
71 let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
72 (texture, view)
73}
74
75/// A `COPY_DST | MAP_READ` readback buffer sized for `height` rows padded to the
76/// 256-byte row alignment `copy_texture_to_buffer` requires; returns it with the
77/// padded bytes-per-row so [`read_back`] can strip the padding.
78pub(crate) fn create_readback(
79 device: &wgpu::Device,
80 width: u32,
81 height: u32,
82) -> (wgpu::Buffer, u32) {
83 let padded_bpr = padded_row_bytes(width);
84 let buffer = device.create_buffer(&wgpu::BufferDescriptor {
85 label: Some("rlx-capture-readback"),
86 size: padded_bpr as u64 * height as u64,
87 usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ,
88 mapped_at_creation: false,
89 });
90 (buffer, padded_bpr)
91}
92
93/// Clear the capture target to opaque black before the scene draws, so an empty
94/// or `Load`-op scene still yields defined, non-transparent pixels.
95pub(crate) fn record_clear(encoder: &mut wgpu::CommandEncoder, view: &wgpu::TextureView) {
96 gpu::color_pass(
97 encoder,
98 "rlx-capture-clear",
99 view,
100 wgpu::LoadOp::Clear(wgpu::Color::BLACK),
101 );
102}
103
104/// Record the texture→buffer copy honoring the padded row stride.
105pub(crate) fn record_copy(
106 encoder: &mut wgpu::CommandEncoder,
107 texture: &wgpu::Texture,
108 buffer: &wgpu::Buffer,
109 padded_bpr: u32,
110 width: u32,
111 height: u32,
112) {
113 encoder.copy_texture_to_buffer(
114 wgpu::TexelCopyTextureInfo {
115 texture,
116 mip_level: 0,
117 origin: wgpu::Origin3d::ZERO,
118 aspect: wgpu::TextureAspect::All,
119 },
120 wgpu::TexelCopyBufferInfo {
121 buffer,
122 layout: wgpu::TexelCopyBufferLayout {
123 offset: 0,
124 bytes_per_row: Some(padded_bpr),
125 rows_per_image: Some(height),
126 },
127 },
128 wgpu::Extent3d {
129 width,
130 height,
131 depth_or_array_layers: 1,
132 },
133 );
134}
135
136/// Map the readback buffer (blocking on `poll(Wait)`), strip the row padding,
137/// and return a tight [`CaptureImage`]. The caller must have already submitted
138/// the copy. Off the hot path by construction.
139pub(crate) fn read_back(
140 device: &wgpu::Device,
141 buffer: &wgpu::Buffer,
142 width: u32,
143 height: u32,
144 padded_bpr: u32,
145) -> Result<CaptureImage, RenderError> {
146 let slice = buffer.slice(..);
147 let (tx, rx) = std::sync::mpsc::channel();
148 slice.map_async(wgpu::MapMode::Read, move |res| {
149 let _ = tx.send(res);
150 });
151 device
152 .poll(wgpu::PollType::wait_indefinitely())
153 .map_err(|_| RenderError::CaptureReadback)?;
154 rx.recv()
155 .map_err(|_| RenderError::CaptureReadback)?
156 .map_err(|_| RenderError::CaptureReadback)?;
157
158 let rgba = {
159 let mapped = slice
160 .get_mapped_range()
161 .map_err(|_| RenderError::CaptureReadback)?;
162 unpad_rows(&mapped, width, height, padded_bpr)
163 };
164 buffer.unmap();
165
166 Ok(CaptureImage {
167 width,
168 height,
169 rgba,
170 })
171}
172
173/// `width * 4` rounded up to the 256-byte row alignment.
174fn padded_row_bytes(width: u32) -> u32 {
175 row_bytes(width, BYTES_PER_PIXEL)
176}
177
178/// `width * bytes_per_pixel` rounded up to the 256-byte row alignment
179/// `copy_texture_to_buffer` requires.
180fn row_bytes(width: u32, bytes_per_pixel: u32) -> u32 {
181 let unpadded = width * bytes_per_pixel;
182 let align = wgpu::COPY_BYTES_PER_ROW_ALIGNMENT;
183 unpadded.div_ceil(align) * align
184}
185
186// ---------------------------------------------------------------------------
187// Linear-light readback (Plan 0045 Phase 3)
188// ---------------------------------------------------------------------------
189
190/// A `COPY_DST | MAP_READ` buffer sized for a `Rgba16Float` texture of
191/// `width`×`height`; returns it with the padded bytes-per-row.
192#[cfg(test)]
193pub(crate) fn create_linear_readback(
194 device: &wgpu::Device,
195 width: u32,
196 height: u32,
197) -> (wgpu::Buffer, u32) {
198 let padded_bpr = row_bytes(width, LINEAR_BYTES_PER_PIXEL);
199 let buffer = device.create_buffer(&wgpu::BufferDescriptor {
200 label: Some("rlx-capture-linear-readback"),
201 size: padded_bpr as u64 * height as u64,
202 usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ,
203 mapped_at_creation: false,
204 });
205 (buffer, padded_bpr)
206}
207
208/// [`read_back`], for a `Rgba16Float` source: strips the row padding and decodes
209/// each half to `f32`, returning `width * height * 4` tight linear values
210/// row-major top-to-bottom.
211///
212/// The point is that these are **not clamped to 1.0** — this is the only way to
213/// observe the composite as light rather than as a picture, which is what Plan
214/// 0045 Phase 3's first done-when asks for. Test-only: nothing in the frame path
215/// reads a texture back (see the module docs on why the readback blocks).
216#[cfg(test)]
217pub(crate) fn read_back_linear(
218 device: &wgpu::Device,
219 buffer: &wgpu::Buffer,
220 width: u32,
221 height: u32,
222 padded_bpr: u32,
223) -> Result<Vec<f32>, RenderError> {
224 let slice = buffer.slice(..);
225 let (tx, rx) = std::sync::mpsc::channel();
226 slice.map_async(wgpu::MapMode::Read, move |res| {
227 let _ = tx.send(res);
228 });
229 device
230 .poll(wgpu::PollType::wait_indefinitely())
231 .map_err(|_| RenderError::CaptureReadback)?;
232 rx.recv()
233 .map_err(|_| RenderError::CaptureReadback)?
234 .map_err(|_| RenderError::CaptureReadback)?;
235
236 let rgba = {
237 let mapped = slice
238 .get_mapped_range()
239 .map_err(|_| RenderError::CaptureReadback)?;
240 let tight_bpr = (width * LINEAR_BYTES_PER_PIXEL) as usize;
241 let mut out = Vec::with_capacity(width as usize * height as usize * 4);
242 for row in mapped.chunks_exact(padded_bpr as usize) {
243 let Some(tight) = row.get(..tight_bpr) else {
244 continue; // a short final row — never expected
245 };
246 for half in tight.chunks_exact(2) {
247 let bits = u16::from_le_bytes([
248 half.first().copied().unwrap_or(0),
249 half.get(1).copied().unwrap_or(0),
250 ]);
251 out.push(f16_to_f32(bits));
252 }
253 }
254 out
255 };
256 buffer.unmap();
257
258 Ok(rgba)
259}
260
261/// Decode one IEEE-754 binary16 to `f32`. Twelve lines rather than a `half`
262/// dependency: it is test-only, and "every new crate is a cost" (CLAUDE.md).
263///
264/// Subnormals are handled by arithmetic (`mantissa * 2^-24`, exact in `f32`)
265/// rather than by renormalizing bit surgery, so there is no loop to bound.
266#[cfg(test)]
267fn f16_to_f32(bits: u16) -> f32 {
268 let sign = if bits & 0x8000 != 0 { -1.0f32 } else { 1.0 };
269 let exponent = u32::from((bits >> 10) & 0x1f);
270 let mantissa = u32::from(bits & 0x03ff);
271 match exponent {
272 // Zero or subnormal.
273 0 => sign * (mantissa as f32) * (1.0 / 16_777_216.0),
274 // Infinity or NaN.
275 0x1f => f32::from_bits(((bits as u32 & 0x8000) << 16) | 0x7f80_0000 | (mantissa << 13)),
276 // Normal: rebias the exponent from 15 to 127 and left-align the mantissa.
277 _ => f32::from_bits(
278 ((bits as u32 & 0x8000) << 16) | ((exponent + 127 - 15) << 23) | (mantissa << 13),
279 ),
280 }
281}
282
283/// Copy the tight `width*4` bytes out of each padded row into a contiguous
284/// buffer. A short final row (never expected) is skipped rather than panicking.
285pub(super) fn unpad_rows(padded: &[u8], width: u32, height: u32, padded_bpr: u32) -> Vec<u8> {
286 let tight_bpr = (width * BYTES_PER_PIXEL) as usize;
287 let mut out = Vec::with_capacity(tight_bpr * height as usize);
288 for row in padded.chunks_exact(padded_bpr as usize) {
289 if let Some(tight) = row.get(..tight_bpr) {
290 out.extend_from_slice(tight);
291 }
292 }
293 out
294}
295
296// ---------------------------------------------------------------------------
297// The sustained frame tap (Plan 0115 Phase 2)
298// ---------------------------------------------------------------------------
299
300/// A persistent offscreen target plus readback buffer, built once and reused for
301/// every frame of a sustained tap.
302///
303/// The difference from the rest of this file is lifetime, not stage. Every other
304/// capture entry point builds its target and buffer per call — right for QA
305/// tooling that takes one frame, and wrong for a source that takes 864,000 of
306/// them, where per-frame texture and buffer creation is GPU allocation inside
307/// the loop. `capture_stream` already reuses its pair, but only for the length
308/// of one fixed-`dt`, one-preset run it drives itself; this type hands that
309/// reuse to a caller who owns the loop.
310///
311/// # One frame in flight, and a wait only when the GPU is behind
312///
313/// The cycle is the preview readback's, three steps across two frames:
314/// **take** the previous submission's map, **record** this frame's copy into the
315/// freed buffer, and **arm** the map after the submission. So a caller sees
316/// frame *N* while frame *N+1* is being drawn, and the very first call yields
317/// nothing at all.
318///
319/// **The take waits when the previous map has not landed**
320/// (`take_previous`). That wait is the only backpressure
321/// a windowless loop has: the window is held to the GPU's pace by the
322/// swapchain, and nothing holds a tap to it but this. Without it, an adapter
323/// whose frame costs more GPU time than the caller's loop costs CPU time queues
324/// submissions without bound, the one map in flight lands only behind all of
325/// them, and nearly every frame drawn meanwhile is never copied out. On an
326/// adapter that keeps up, the map has landed by the next call and nothing
327/// waits.
328///
329/// **The buffer cannot be re-recorded while it is mapped**, which is why the
330/// take comes before the record: a second copy into a mapped buffer is a
331/// validation error, not a dropped frame.
332///
333/// **Sized at construction and never resized.** `record_copy`'s extent, the
334/// buffer's length and `padded_bpr` are all fixed against `width`×`height`, so a
335/// renderer that resizes underneath a live tap needs a new one — [`open_tap`]
336/// is the only thing that sets these.
337///
338/// [`open_tap`]: super::Renderer::open_tap
339pub struct FrameTap {
340 /// `RENDER_ATTACHMENT | COPY_SRC`, the offscreen the frame draws into.
341 pub(crate) texture: wgpu::Texture,
342 /// A view of `texture`, held rather than recreated per frame.
343 pub(crate) view: wgpu::TextureView,
344 /// `COPY_DST | MAP_READ`, sized `padded_bpr * height`.
345 pub(crate) buffer: wgpu::Buffer,
346 /// Row stride of `buffer`, padded to the 256-byte copy alignment.
347 pub(crate) padded_bpr: u32,
348 /// Pixel width the three resources above are sized against.
349 pub(crate) width: u32,
350 /// Pixel height the three resources above are sized against.
351 pub(crate) height: u32,
352 /// The per-pass GPU timer, `None` on an adapter with no timestamp queries.
353 ///
354 /// **This is the only thing in the engine that builds a query set**, which
355 /// is what makes "the window path encodes no timestamp writes" structural:
356 /// a window has no tap.
357 pub(crate) timer: Option<gpu::PassTimer>,
358 /// What the timer has measured since the last [`reset_pass_costs`].
359 ///
360 /// [`reset_pass_costs`]: FrameTap::reset_pass_costs
361 pub(crate) costs: PassCosts,
362 /// The armed map's result channel, `None` when [`buffer`](Self::buffer) is
363 /// free to record into.
364 armed: Option<std::sync::mpsc::Receiver<Result<(), wgpu::BufferAsyncError>>>,
365}
366
367impl FrameTap {
368 /// Build the target, its view, the readback buffer and — where the device
369 /// offers timestamp queries — the pass timer, in one step: the whole of the
370 /// tap's GPU allocation, paid here so the per-frame path pays none.
371 pub(crate) fn new(
372 device: &wgpu::Device,
373 queue: &wgpu::Queue,
374 format: wgpu::TextureFormat,
375 width: u32,
376 height: u32,
377 ) -> Self {
378 let (texture, view) = create_target(device, format, width, height);
379 let (buffer, padded_bpr) = create_readback(device, width, height);
380 Self {
381 texture,
382 view,
383 buffer,
384 padded_bpr,
385 width,
386 height,
387 timer: gpu::PassTimer::new(device, queue),
388 costs: PassCosts::default(),
389 armed: None,
390 }
391 }
392
393 /// The pixel size every frame this tap yields will carry.
394 pub fn size(&self) -> (u32, u32) {
395 (self.width, self.height)
396 }
397
398 /// Take the frame the **previous** submission's map produced, if it has
399 /// landed, and fold that frame's pass timings in with it.
400 ///
401 /// Polls without waiting. A map still in flight yields `None` and leaves the
402 /// buffer armed; the next call asks again. Call this before
403 /// [`record`](Self::record) within a frame — the buffer cannot be copied
404 /// into while it is mapped.
405 ///
406 /// The timings are collected **here**, before the timer is re-armed for the
407 /// frame about to be encoded, because the labels and the claim count it
408 /// still holds are the ones the landed timestamps belong to.
409 pub(crate) fn consume(&mut self, device: &wgpu::Device) -> Option<CaptureImage> {
410 use std::sync::mpsc::TryRecvError;
411
412 let armed = self.armed.as_ref()?;
413 // Non-blocking: a map that has landed is taken without paying for
414 // whatever else the GPU has queued. The wait, where one is owed, is
415 // `take_previous`'s.
416 let _ = device.poll(wgpu::PollType::Poll);
417 match armed.try_recv() {
418 Ok(Ok(())) => {}
419 // Still in flight: not an error and not a dropped frame. The buffer
420 // stays armed and the next call asks again.
421 Err(TryRecvError::Empty) => return None,
422 // Mapping failed, or the callback was dropped without firing. Either
423 // way this cycle is over: disarm so the next frame records afresh.
424 Ok(Err(_)) | Err(TryRecvError::Disconnected) => {
425 self.armed = None;
426 if let Some(timer) = self.timer.as_mut() {
427 timer.discard();
428 }
429 return None;
430 }
431 }
432 self.armed = None;
433 if let Some(timer) = self.timer.as_mut() {
434 timer.collect(&mut self.costs);
435 }
436 let slice = self.buffer.slice(..);
437 let image = slice.get_mapped_range().ok().map(|mapped| CaptureImage {
438 width: self.width,
439 height: self.height,
440 rgba: unpad_rows(&mapped, self.width, self.height, self.padded_bpr),
441 });
442 // Unmapped whether or not the range was readable: a buffer left mapped
443 // is one this tap can never record into again.
444 self.buffer.unmap();
445 image
446 }
447
448 /// Take the frame the previous submission carried, **waiting for it only if
449 /// its map has not landed**, so that on return the buffer is always free to
450 /// record into.
451 ///
452 /// This is what bounds the GPU queue to the frame about to be drawn: every
453 /// call after the first returns a frame, on any adapter, and a caller that
454 /// outruns the GPU is held to its rate instead of queueing draws it will
455 /// never read. `None` on the first call and after a map that failed.
456 pub(crate) fn take_previous(&mut self, device: &wgpu::Device) -> Option<CaptureImage> {
457 let image = self.consume(device);
458 if image.is_some() || self.armed.is_none() {
459 return image;
460 }
461 self.drain(device)
462 }
463
464 /// **Wait** for the frame still in flight and take it, drawing nothing.
465 ///
466 /// The blocking counterpart of [`consume`](Self::consume): a bounded run's
467 /// last frame, and the half of [`take_previous`](Self::take_previous) that
468 /// runs when the GPU is behind. `None` when nothing is in flight. This file
469 /// is one of the two the indefinite-wait allowlist admits, for this wait.
470 pub(crate) fn drain(&mut self, device: &wgpu::Device) -> Option<CaptureImage> {
471 self.armed.as_ref()?;
472 // The map was asked for on a submission that has already been made, so
473 // this returns as soon as the GPU retires it.
474 let _ = device.poll(wgpu::PollType::wait_indefinitely());
475 self.consume(device)
476 }
477
478 /// Ask for the mapping, after the submission that carried the copy.
479 ///
480 /// The callback only sends; every decision is taken on the caller's thread
481 /// when it next polls, so nothing wgpu calls back into does work.
482 pub(crate) fn arm(&mut self) {
483 let (tx, rx) = std::sync::mpsc::channel();
484 self.buffer
485 .slice(..)
486 .map_async(wgpu::MapMode::Read, move |res| {
487 let _ = tx.send(res);
488 });
489 self.armed = Some(rx);
490 }
491
492 /// Whether this tap can report per-pass GPU costs at all. `false` on an
493 /// adapter without `TIMESTAMP_QUERY` — the software rasterizers — where a
494 /// caller says so once rather than printing an empty table every window.
495 pub fn times_passes(&self) -> bool {
496 self.timer.is_some()
497 }
498
499 /// What every labelled pass has cost since the last reset.
500 pub fn pass_costs(&self) -> &PassCosts {
501 &self.costs
502 }
503
504 /// Start a fresh measurement window, so each report covers the interval
505 /// since the last one rather than the whole run to date.
506 pub fn reset_pass_costs(&mut self) {
507 self.costs.reset();
508 }
509}
510
511// ---------------------------------------------------------------------------
512// What the passes cost
513// ---------------------------------------------------------------------------
514
515/// GPU time per labelled render/compute pass, accumulated over a window of
516/// frames (ADR-0245).
517///
518/// **Summed per label, not per pass instance.** A frame encodes `bloom-blur-h`
519/// once per pyramid level, and what a reader wants to know is what the blur
520/// costs the frame — so the rows are what each *label* costs per frame, and a
521/// label that appears `N` times carries all `N`.
522#[derive(Debug, Clone, Default, PartialEq)]
523pub struct PassCosts {
524 /// `(label, total milliseconds)`, in first-seen order. A `Vec` rather than
525 /// a map because the roster is a few dozen entries walked once a frame: a
526 /// linear scan over that is cheaper than hashing, and it allocates nothing
527 /// once the labels have all been seen.
528 rows: Vec<(String, f64)>,
529 /// Frames these totals cover, so a row can be reported as a mean.
530 frames: u64,
531}
532
533impl PassCosts {
534 /// Add one pass's milliseconds to its label's running total.
535 pub(crate) fn add(&mut self, label: &str, ms: f64) {
536 if !ms.is_finite() {
537 return;
538 }
539 if let Some(row) = self.rows.iter_mut().find(|(name, _)| name == label) {
540 row.1 += ms;
541 return;
542 }
543 self.rows.push((label.to_owned(), ms));
544 }
545
546 /// Note that a frame's worth of passes has been added.
547 pub(crate) fn close_frame(&mut self) {
548 self.frames += 1;
549 }
550
551 /// Frames the accumulated totals cover.
552 pub fn frames(&self) -> u64 {
553 self.frames
554 }
555
556 /// Every label's **mean milliseconds per frame**, costliest first. Empty
557 /// while no frame has been measured.
558 ///
559 /// Sorted here rather than by the caller so every report orders the rows
560 /// the same way; ties keep first-seen order, which is roughly composite
561 /// order and reads as the frame's own sequence.
562 pub fn rows(&self) -> Vec<(&str, f64)> {
563 if self.frames == 0 {
564 return Vec::new();
565 }
566 let frames = self.frames as f64;
567 let mut rows: Vec<(&str, f64)> = self
568 .rows
569 .iter()
570 .map(|(label, total)| (label.as_str(), total / frames))
571 .collect();
572 rows.sort_by(|a, b| b.1.total_cmp(&a.1));
573 rows
574 }
575
576 /// Drop the accumulated totals and the frame count.
577 pub fn reset(&mut self) {
578 self.rows.clear();
579 self.frames = 0;
580 }
581}