rlx_core/render/capture_api.rs
1//! The capture API — `Renderer`'s headless, off-hot-path entry points
2//! (Plan 0013), carved out of `render/mod.rs` by Plan 0061 Phase 3.
3//!
4//! **Dev-tooling API, with one exception at the bottom of the file.** Every
5//! `capture_*` entry point here is driven by the `shot` example and
6//! `core/tests/`, never by the standalone's frame loop and never from behind the
7//! C ABI: each blocks on a GPU readback, so calling one from a *display* loop is
8//! a stutter by construction. The frame tap (Plan 0115) blocks on the same
9//! readback and is nonetheless a live path — a headless source has no present
10//! deadline to miss, only throughput to hold, and its readback is what bounds
11//! the run's memory. It lives here because it shares the offscreen machinery,
12//! not because it shares the caller.
13//!
14//! It is a second `impl Renderer` block rather than a separate type, because the
15//! methods are public API whose paths must not move — `Renderer::capture_preset`
16//! is spelled the same before and after this split.
17
18// Hot-path panic-denial pragma (Plan 0002 Phase 2; render/ is scanned by the
19// hygiene guard). These paths are off the frame loop, but they share
20// `Renderer`'s state and the pragma travels with the code, not with the file.
21#![deny(
22 clippy::unwrap_used,
23 clippy::expect_used,
24 clippy::indexing_slicing,
25 clippy::panic,
26 clippy::unreachable
27)]
28
29// A continuation of one `impl` block that was split across two files, so it
30// needs the same names `render/mod.rs` has in scope. Enumerating them would be a
31// list to keep in sync with a file whose whole purpose is to be the other half
32// of this one.
33use super::*;
34
35/// What one [`Renderer::capture_audio_after_warmup`] run produced.
36///
37/// The two fields beside the images exist so a caller can check *what the run
38/// did* rather than infer it from a stopwatch (Plan 0084 Phase 3): `analysis`
39/// is the analyzer state the run walked through, and `rendered` is how much of
40/// it reached a rasterizer.
41pub struct AudioCapture {
42 /// The requested frames, in `at_frames` order.
43 pub images: Vec<CaptureImage>,
44 /// One published [`AnalysisFrame`] per hop, in hop order. Independent of
45 /// whether the hop was rendered — which is the property that makes feeding
46 /// warm-up hops without pixels safe, and is asserted rather than argued
47 /// (`core/tests/suite/capture_advance.rs`).
48 pub analysis: Vec<AnalysisFrame>,
49 /// How many frames were rasterized. Zero when every hop is a warm-up hop.
50 pub rendered: usize,
51}
52
53impl Renderer {
54 /// Advance the scene clock one step and capture that single frame into an
55 /// offscreen texture, returning tight RGBA (Plan 0013). Off the hot path —
56 /// blocks on GPU readback; never call it from a live loop.
57 pub fn capture_frame(&mut self, frame: &AnalysisFrame) -> Result<CaptureImage, RenderError> {
58 self.time += scenes::FALLBACK_DT;
59 self.capture_at_clock(frame)
60 }
61
62 /// Draw the active preset for `frame` at the **current** clock into a fresh
63 /// offscreen texture and read it back. Does not advance the clock, so
64 /// callers that already stepped it share this. The whole path (clear → draw
65 /// → copy → map) is deterministic for a given `(preset, frame, clock)`.
66 fn capture_at_clock(&mut self, frame: &AnalysisFrame) -> Result<CaptureImage, RenderError> {
67 let (width, height) = (self.ctx.config.width, self.ctx.config.height);
68 let format = self.ctx.surface_format();
69 let (texture, view) = capture::create_target(&self.ctx.device, format, width, height);
70 let (buffer, padded_bpr) = capture::create_readback(&self.ctx.device, width, height);
71 let mut encoder = self
72 .ctx
73 .device
74 .create_command_encoder(&wgpu::CommandEncoderDescriptor {
75 label: Some("rlx-capture-frame"),
76 });
77 // The preview intermediate, when one is open, sits between the draw and
78 // the destination here exactly as it does on the present path — same
79 // clear, same draw, same `copy_texture_to_texture`. That is what lets
80 // the intermediate's one real claim — that a frame routed through it is
81 // byte-identical to one drawn straight at the target — be asserted with
82 // no window, in `core/tests/suite/console_preview.rs`.
83 let preview = self.preview.take_target();
84 capture::record_clear(&mut encoder, preview.as_ref().map_or(&view, |p| &p.view));
85 let _ = self.draw_frame(
86 frame,
87 &mut encoder,
88 preview.as_ref().map_or(&view, |p| &p.view),
89 (width, height),
90 scenes::FALLBACK_DT,
91 SaltMode::Pinned,
92 );
93 if let Some(p) = preview.as_ref() {
94 p.record_copy_to(&mut encoder, &texture);
95 }
96 self.preview.restore_target(preview);
97 // The preview readback advances on **every** frame drawn through the
98 // intermediate, which is this path as much as the present path: the two
99 // record the same clear, draw and copy, and stating the rule once is
100 // what lets the readback's own claims be asserted with no window
101 // (`core/tests/suite/console_preview.rs`). It changes nothing about the image
102 // returned below — it is an extra copy out of the intermediate, not a
103 // change to what was drawn into it.
104 let recorded = self.preview.step_readback(&self.ctx.device, &mut encoder);
105 capture::record_copy(&mut encoder, &texture, &buffer, padded_bpr, width, height);
106 self.ctx.queue.submit(std::iter::once(encoder.finish()));
107 if recorded {
108 self.preview.arm_readback();
109 }
110
111 #[cfg(feature = "text")]
112 self.text_layer.end_frame();
113
114 capture::read_back(&self.ctx.device, &buffer, width, height, padded_bpr)
115 }
116
117 /// Capture preset `name` after advancing it `frames` steps from a fixed
118 /// initial state, driven by a single constant `frame` (Plan 0013). A **pure
119 /// function** of `(name, frame, frames)`: the scenes are rebuilt so any
120 /// stateful system (e.g. the seeded swarm particles) starts from its
121 /// deterministic seed, and the scene clock resets to `0.0`, so the result is
122 /// independent of any earlier capture. Errors if `name` is not in the
123 /// roster. `frames` is treated as at least 1.
124 pub fn capture_preset(
125 &mut self,
126 name: &str,
127 frame: &AnalysisFrame,
128 frames: u32,
129 ) -> Result<CaptureImage, RenderError> {
130 self.reset_for_capture(name)?;
131
132 let (width, height) = (self.ctx.config.width, self.ctx.config.height);
133 let format = self.ctx.surface_format();
134 let (texture, view) = capture::create_target(&self.ctx.device, format, width, height);
135
136 // Warm the scene through the first frames-1 steps (state advances, pixels
137 // discarded); then capture the final frame.
138 let n = frames.max(1);
139 for _ in 1..n {
140 self.time += scenes::FALLBACK_DT;
141 self.step_offscreen(frame, &view, width, height, scenes::FALLBACK_DT);
142 }
143 self.time += scenes::FALLBACK_DT;
144
145 let (buffer, padded_bpr) = capture::create_readback(&self.ctx.device, width, height);
146 let mut encoder = self
147 .ctx
148 .device
149 .create_command_encoder(&wgpu::CommandEncoderDescriptor {
150 label: Some("rlx-capture-preset"),
151 });
152 capture::record_clear(&mut encoder, &view);
153 let _ = self.draw_frame(
154 frame,
155 &mut encoder,
156 &view,
157 (width, height),
158 scenes::FALLBACK_DT,
159 SaltMode::Pinned,
160 );
161 capture::record_copy(&mut encoder, &texture, &buffer, padded_bpr, width, height);
162 self.ctx.queue.submit(std::iter::once(encoder.finish()));
163
164 #[cfg(feature = "text")]
165 self.text_layer.end_frame();
166
167 capture::read_back(&self.ctx.device, &buffer, width, height, padded_bpr)
168 }
169
170 /// Capture preset `name` across a **time-varying** stimulus (Plan 0037):
171 /// one rendered frame per entry of `stimulus`, read back in order, so the
172 /// returned images are the response *while it changes* rather than after it
173 /// settles.
174 ///
175 /// This is the primitive [`capture_preset`](Self::capture_preset) cannot be:
176 /// holding one frame for every step converges every smoother before the
177 /// pixels are read, which makes the result identical for any `[smoothing]`
178 /// constant (ADR-0039). `capture_preset` is left exactly as it was — four
179 /// suites and `--report` consume it — and this is its sibling, sharing the
180 /// same `reset_for_capture` seed so both are pure
181 /// functions of their arguments.
182 ///
183 /// The clock advances one `FALLBACK_DT` per entry, so
184 /// index `i` is second `i * dt` of the response. An empty `stimulus` yields
185 /// no images. Errors if `name` is not in the roster.
186 ///
187 /// **Off the hot path, and more so than its sibling** — it blocks on a GPU
188 /// readback *per frame*, not once per call. The target and readback buffer
189 /// are allocated up front rather than per frame, because building GPU
190 /// resources mid-sequence perturbs what the feedback stages resolve to on the
191 /// DX12 software adapter.
192 pub fn capture_preset_over(
193 &mut self,
194 name: &str,
195 stimulus: &[AnalysisFrame],
196 ) -> Result<Vec<CaptureImage>, RenderError> {
197 self.reset_for_capture(name)?;
198
199 let (width, height) = (self.ctx.config.width, self.ctx.config.height);
200 let format = self.ctx.surface_format();
201 let (texture, view) = capture::create_target(&self.ctx.device, format, width, height);
202 let (buffer, padded_bpr) = capture::create_readback(&self.ctx.device, width, height);
203
204 let mut images = Vec::with_capacity(stimulus.len());
205 for frame in stimulus {
206 self.time += scenes::FALLBACK_DT;
207 let mut encoder =
208 self.ctx
209 .device
210 .create_command_encoder(&wgpu::CommandEncoderDescriptor {
211 label: Some("rlx-capture-over"),
212 });
213 capture::record_clear(&mut encoder, &view);
214 let _ = self.draw_frame(
215 frame,
216 &mut encoder,
217 &view,
218 (width, height),
219 scenes::FALLBACK_DT,
220 SaltMode::Pinned,
221 );
222 capture::record_copy(&mut encoder, &texture, &buffer, padded_bpr, width, height);
223 self.ctx.queue.submit(std::iter::once(encoder.finish()));
224
225 #[cfg(feature = "text")]
226 self.text_layer.end_frame();
227
228 images.push(capture::read_back(
229 &self.ctx.device,
230 &buffer,
231 width,
232 height,
233 padded_bpr,
234 )?);
235 }
236 Ok(images)
237 }
238
239 /// Advance preset `name` under a single constant `frame` and read back only
240 /// the frames named in `at_frames` (Plan 0085 Phase 1) — the **long-run**
241 /// primitive, and the one a horizon needs.
242 ///
243 /// Its two siblings cannot serve a run of tens of thousands of frames:
244 /// [`capture_preset`](Self::capture_preset) reseeds from scratch on every
245 /// call, so sampling *k* points costs `O(k·N)` renders, and
246 /// [`capture_preset_over`](Self::capture_preset_over) reads back *every*
247 /// frame, so a ten-minute run at 720p would materialize ~36,000 images. This
248 /// renders `N` frames once and holds `at_frames.len()` of them.
249 ///
250 /// Frame numbering matches [`capture_audio`](Self::capture_audio): frame 0
251 /// is the first advanced frame, so `at_frames = [n - 1]` returns exactly what
252 /// `capture_preset(name, frame, n)` returns — asserted in
253 /// `core/tests/suite/capture_advance.rs` rather than argued, because it is the
254 /// property that lets a horizon's rows be compared with every other capture
255 /// this repo takes.
256 ///
257 /// Deterministic on the same terms as its siblings: scenes are rebuilt to
258 /// their seed, the clock resets to `0.0`, and the step is a fixed
259 /// `FALLBACK_DT` — so a row at index *k* does not
260 /// depend on how far the run was asked to go. Images come back in
261 /// `at_frames` order; a repeated index yields the same frame twice rather
262 /// than rendering it twice. An empty `at_frames` renders nothing.
263 ///
264 /// **Off the hot path** — it blocks on a GPU readback per requested frame.
265 /// The readback buffer is built **once, at the first requested frame**, and
266 /// reused for every later one. Both halves of that matter on the DX12
267 /// software adapter, where building GPU resources mid-sequence perturbs what
268 /// the feedback stages resolve to (the hazard
269 /// [`capture_preset_over`](Self::capture_preset_over) documents, and a
270 /// horizon is precisely a long feedback sequence): reusing it means the
271 /// perturbation happens once rather than per sample, and doing it at the
272 /// first sample rather than up front is what puts the allocation at the same
273 /// point in the sequence [`capture_preset`](Self::capture_preset) puts it —
274 /// which is what makes the two agree pixel-for-pixel on WARP as well as on
275 /// hardware. It also stays independent of the horizon requested, since the
276 /// first sample sits at the same frame index however long the run is.
277 pub fn capture_preset_at(
278 &mut self,
279 name: &str,
280 frame: &AnalysisFrame,
281 at_frames: &[u32],
282 ) -> Result<Vec<CaptureImage>, RenderError> {
283 self.reset_for_capture(name)?;
284 let Some(&last) = at_frames.iter().max() else {
285 return Ok(Vec::new());
286 };
287
288 let (width, height) = (self.ctx.config.width, self.ctx.config.height);
289 let format = self.ctx.surface_format();
290 let (texture, view) = capture::create_target(&self.ctx.device, format, width, height);
291 let mut readback: Option<(wgpu::Buffer, u32)> = None;
292
293 let mut captured: Vec<(u32, CaptureImage)> = Vec::with_capacity(at_frames.len());
294 for index in 0..=last {
295 self.time += scenes::FALLBACK_DT;
296 if !at_frames.contains(&index) {
297 self.step_offscreen(frame, &view, width, height, scenes::FALLBACK_DT);
298 continue;
299 }
300 let slot = readback
301 .get_or_insert_with(|| capture::create_readback(&self.ctx.device, width, height));
302 let (buffer, padded_bpr) = (&slot.0, slot.1);
303 let mut encoder =
304 self.ctx
305 .device
306 .create_command_encoder(&wgpu::CommandEncoderDescriptor {
307 label: Some("rlx-capture-at"),
308 });
309 capture::record_clear(&mut encoder, &view);
310 let _ = self.draw_frame(
311 frame,
312 &mut encoder,
313 &view,
314 (width, height),
315 scenes::FALLBACK_DT,
316 SaltMode::Pinned,
317 );
318 capture::record_copy(&mut encoder, &texture, buffer, padded_bpr, width, height);
319 self.ctx.queue.submit(std::iter::once(encoder.finish()));
320
321 #[cfg(feature = "text")]
322 self.text_layer.end_frame();
323
324 let img = capture::read_back(&self.ctx.device, buffer, width, height, padded_bpr)?;
325 captured.push((index, img));
326 }
327
328 at_frames
329 .iter()
330 .map(|idx| {
331 captured
332 .iter()
333 .find(|(i, _)| i == idx)
334 .map(|(_, img)| img.clone())
335 .ok_or(RenderError::CaptureReadback)
336 })
337 .collect()
338 }
339
340 /// Render preset `name` for `frames` frames at an **injected** `dt`, handing
341 /// each frame to `sink` the moment it is read back (Plan 0101 / ADR-0114) —
342 /// the **streaming** primitive, and the one an offline video render needs.
343 ///
344 /// Its three siblings all return a `Vec<CaptureImage>`, which is exactly what
345 /// a video render cannot afford: a 1080p frame is 8.29 MB, so a four-minute
346 /// track at 60 fps is 119 GB of retained images. Nothing is retained here —
347 /// the frame is handed to `sink` and dropped, so the resident set of a
348 /// 14,400-frame render is the same as a 100-frame one.
349 ///
350 /// It is also the only capture entry point whose step is **not** the fixed
351 /// `FALLBACK_DT`. A render at `--fps 30` advances the
352 /// scene by 1/30 s a frame, or the visuals would run at half speed against
353 /// their own soundtrack; that `dt` is the caller's, exactly as it is for the
354 /// live frontend (ADR-0013). At 60 fps `dt` *is* `FALLBACK_DT`, which is what
355 /// makes a rendered frame comparable with every other capture this repo takes.
356 ///
357 /// `analysis` supplies the [`AnalysisFrame`] for each frame index. The audio
358 /// hop clock and the frame clock are different clocks and only the caller
359 /// knows the mapping between them, so this deliberately does not walk PCM —
360 /// unlike [`capture_audio`](Self::capture_audio), which welds one rendered
361 /// frame to one analysis hop.
362 ///
363 /// Deterministic on the same terms as its siblings: scenes rebuilt to their
364 /// seed, the clock reset to `0.0`, the salt pinned. Given a deterministic
365 /// `analysis` the whole run is a pure function of `(name, frames, dt)`.
366 ///
367 /// **Off the hot path** — it blocks on a GPU readback every frame, which is
368 /// also what bounds its memory: `read_back` polls, so each frame's submission
369 /// is retired before the next is encoded (the retention Plan 0099 measured).
370 /// The target and the readback buffer are built **once** and reused, so a
371 /// long run allocates no GPU resources mid-sequence.
372 ///
373 /// A `sink` error stops the run and comes back as
374 /// [`RenderError::Sink`] carrying the consumer's own
375 /// message.
376 ///
377 /// A `dt` that is not finite and positive is replaced by `FALLBACK_DT` for the
378 /// whole run (`sanitize_frame_dt`, ADR-0191).
379 pub fn capture_stream(
380 &mut self,
381 name: &str,
382 frames: u32,
383 dt: f32,
384 analysis: &mut dyn FnMut(u32) -> AnalysisFrame,
385 sink: &mut dyn FnMut(u32, &CaptureImage) -> Result<(), String>,
386 ) -> Result<(), RenderError> {
387 let dt = super::sanitize_frame_dt(dt);
388 self.reset_for_capture(name)?;
389
390 let (width, height) = (self.ctx.config.width, self.ctx.config.height);
391 let format = self.ctx.surface_format();
392 let (texture, view) = capture::create_target(&self.ctx.device, format, width, height);
393 let (buffer, padded_bpr) = capture::create_readback(&self.ctx.device, width, height);
394
395 for index in 0..frames {
396 let frame = analysis(index);
397 self.time += dt;
398 let mut encoder =
399 self.ctx
400 .device
401 .create_command_encoder(&wgpu::CommandEncoderDescriptor {
402 label: Some("rlx-capture-stream"),
403 });
404 capture::record_clear(&mut encoder, &view);
405 let _ = self.draw_frame(
406 &frame,
407 &mut encoder,
408 &view,
409 (width, height),
410 dt,
411 SaltMode::Pinned,
412 );
413 capture::record_copy(&mut encoder, &texture, &buffer, padded_bpr, width, height);
414 self.ctx.queue.submit(std::iter::once(encoder.finish()));
415
416 #[cfg(feature = "text")]
417 self.text_layer.end_frame();
418
419 let img = capture::read_back(&self.ctx.device, &buffer, width, height, padded_bpr)?;
420 sink(index, &img).map_err(RenderError::Sink)?;
421 }
422 Ok(())
423 }
424
425 /// Select `name` and reset every stateful system to its deterministic seed —
426 /// the shared preamble of [`capture_preset`](Self::capture_preset) and
427 /// [`capture_preset_over`](Self::capture_preset_over), so both are pure
428 /// functions of their arguments and neither inherits an earlier capture's
429 /// history.
430 fn reset_for_capture(&mut self, name: &str) -> Result<(), RenderError> {
431 if !self.select_preset_by_name_now(name) {
432 return Err(RenderError::UnknownPreset(name.to_string()));
433 }
434 self.scenes =
435 scenes::create_all(&self.ctx.device, COMPOSITE_FORMAT, &self.tier, self.budget);
436 self.cancel_transition();
437 self.side.reset_resources();
438 self.tonemap.reset_resources();
439 self.ink.reset_resources();
440 self.blend.reset_resources();
441 self.time = 0.0;
442 // The rebuilt scenes are fresh — re-apply the active preset's structural
443 // config (ADR-0007) so a line scene captures with its geometry built.
444 self.configure_active_scene();
445 Ok(())
446 }
447
448 /// Drive preset `name` with **real audio through the real analyzer** and
449 /// capture the frames at `at_frames` (Plan 0013). The PCM is fed hop-by-hop
450 /// into a fresh [`Analyzer`](crate::dsp::Analyzer) (format validated at the
451 /// intake boundary — the source-agnostic rule); each produced
452 /// [`AnalysisFrame`] drives one rendered frame, so `at_frames` indexes the
453 /// hop sequence (frame 0 is the first hop). Deterministic: scenes are rebuilt
454 /// to their seed and the clock resets to 0, exactly like
455 /// [`capture_preset`](Self::capture_preset).
456 ///
457 /// This is in-memory PCM only — no file, decoder, or OS audio-source code,
458 /// just like a frontend pushing samples. Returned images are in `at_frames`
459 /// order; an index past the audio length is an error.
460 pub fn capture_audio(
461 &mut self,
462 name: &str,
463 pcm: &[f32],
464 format: AudioFormat,
465 at_frames: &[u32],
466 ) -> Result<Vec<CaptureImage>, RenderError> {
467 Ok(self
468 .capture_audio_after_warmup(name, pcm, format, at_frames, 0)?
469 .images)
470 }
471
472 /// [`capture_audio`](Self::capture_audio), with the first `warmup_hops` hops
473 /// **advanced but not rasterized** (Plan 0084 Phase 3).
474 ///
475 /// A warm-up hop still pushes its samples, still publishes its
476 /// [`AnalysisFrame`], and still advances the scene clock by one
477 /// `FALLBACK_DT` — the hop happened, it just did not
478 /// draw. What it skips is the render pass, which is why a caller that only
479 /// needs the analyzer warm (`core/tests/reactivity.rs` drives
480 /// `WARMUP_HOPS` of them per capture, at silence, and reads none of them
481 /// back) stops paying a full rasterization per hop to reach a DSP state that
482 /// needs no pixels.
483 ///
484 /// **This does not warm GPU-side scene state.** Analysis is a pure function
485 /// of its window and the render pass never touches the analyzer, so the
486 /// published frames are bit-for-bit what they would have been — but a scene
487 /// that *integrates* on the GPU (particles, trails, reaction-diffusion) has
488 /// that many fewer steps behind it at the first rendered hop. Time-driven
489 /// scenes are unaffected, since the clock advances either way.
490 ///
491 /// An `at_frames` entry inside the warm-up span was never rendered and is an
492 /// error, the same one an index past the audio length gives.
493 pub fn capture_audio_after_warmup(
494 &mut self,
495 name: &str,
496 pcm: &[f32],
497 format: AudioFormat,
498 at_frames: &[u32],
499 warmup_hops: usize,
500 ) -> Result<AudioCapture, RenderError> {
501 if !self.select_preset_by_name_now(name) {
502 return Err(RenderError::UnknownPreset(name.to_string()));
503 }
504 let mut analyzer = crate::dsp::Analyzer::new(format).map_err(RenderError::AudioFormat)?;
505
506 self.scenes =
507 scenes::create_all(&self.ctx.device, COMPOSITE_FORMAT, &self.tier, self.budget);
508 self.cancel_transition();
509 self.side.reset_resources();
510 self.tonemap.reset_resources();
511 self.ink.reset_resources();
512 self.blend.reset_resources();
513 self.time = 0.0;
514 self.configure_active_scene();
515
516 let (width, height) = (self.ctx.config.width, self.ctx.config.height);
517 let target_format = self.ctx.surface_format();
518 let (texture, view) =
519 capture::create_target(&self.ctx.device, target_format, width, height);
520
521 let hop_samples = crate::dsp::HOP_SIZE * format.channels as usize;
522 let mut captured: Vec<(u32, CaptureImage)> = Vec::with_capacity(at_frames.len());
523 let mut published: Vec<AnalysisFrame> = Vec::new();
524 let mut rendered = 0usize;
525
526 for (index, hop) in pcm.chunks(hop_samples).enumerate() {
527 let frame_index = index as u32;
528 analyzer.push_interleaved(hop);
529 let analysis = analyzer.take_frame();
530 self.time += scenes::FALLBACK_DT;
531 published.push(analysis);
532
533 if index < warmup_hops {
534 continue;
535 }
536 rendered += 1;
537
538 let wanted = at_frames.contains(&frame_index)
539 && !captured.iter().any(|(i, _)| *i == frame_index);
540 if wanted {
541 let (buffer, padded_bpr) =
542 capture::create_readback(&self.ctx.device, width, height);
543 let mut encoder =
544 self.ctx
545 .device
546 .create_command_encoder(&wgpu::CommandEncoderDescriptor {
547 label: Some("rlx-capture-audio"),
548 });
549 capture::record_clear(&mut encoder, &view);
550 let _ = self.draw_frame(
551 &analysis,
552 &mut encoder,
553 &view,
554 (width, height),
555 scenes::FALLBACK_DT,
556 SaltMode::Pinned,
557 );
558 capture::record_copy(&mut encoder, &texture, &buffer, padded_bpr, width, height);
559 self.ctx.queue.submit(std::iter::once(encoder.finish()));
560 #[cfg(feature = "text")]
561 self.text_layer.end_frame();
562 let img = capture::read_back(&self.ctx.device, &buffer, width, height, padded_bpr)?;
563 captured.push((frame_index, img));
564 } else {
565 self.step_offscreen(&analysis, &view, width, height, scenes::FALLBACK_DT);
566 }
567 }
568
569 let images = at_frames
570 .iter()
571 .map(|idx| {
572 captured
573 .iter()
574 .find(|(i, _)| i == idx)
575 .map(|(_, img)| img.clone())
576 .ok_or(RenderError::CaptureReadback)
577 })
578 .collect::<Result<Vec<_>, _>>()?;
579
580 Ok(AudioCapture {
581 images,
582 analysis: published,
583 rendered,
584 })
585 }
586
587 /// Draw one frame into `view` and submit it — advancing scene state without
588 /// reading anything back. The warm-up step [`capture_preset`] uses to reach
589 /// frame `N`.
590 ///
591 /// **It polls, and that is what bounds the memory of a long run** (Plan
592 /// 0099). Nothing else here reads anything back, so before this line the
593 /// only `device.poll` in the whole capture path was
594 /// [`capture::read_back`]'s — meaning wgpu got no opportunity to retire a
595 /// completed submission between two *sampled* frames. A horizon at the
596 /// default 30 s interval is 1,800 consecutive unpolled submits, and the
597 /// retention is per **pass**, not per pixel: measured over one such stretch
598 /// on the Windows dev box (hardware adapter, debug, 96x96), a
599 /// reaction-diffusion world — 12 simulation sub-steps plus a present, 13
600 /// passes a frame — retained **950 KB per frame** against a captured frame
601 /// of 36 KB, while single-pass worlds retained ~30 KB. That is what made
602 /// the ceiling look like a property of the RD family: every world grew, RD
603 /// grew ~32x faster and hit the allocator first, at ~4.4 GB.
604 fn step_offscreen(
605 &mut self,
606 frame: &AnalysisFrame,
607 view: &wgpu::TextureView,
608 width: u32,
609 height: u32,
610 dt: f32,
611 ) {
612 let mut encoder = self
613 .ctx
614 .device
615 .create_command_encoder(&wgpu::CommandEncoderDescriptor {
616 label: Some("rlx-capture-step"),
617 });
618 capture::record_clear(&mut encoder, view);
619 let _ = self.draw_frame(
620 frame,
621 &mut encoder,
622 view,
623 (width, height),
624 dt,
625 SaltMode::Pinned,
626 );
627 self.ctx.queue.submit(std::iter::once(encoder.finish()));
628
629 // Retire this submission's resources. `wait_indefinitely` rather than a
630 // non-blocking `Poll`, and that was **measured, not assumed**: a
631 // `PollType::Poll` here took the same 3,600-frame stretch from
632 // 3,668 MB to 3,188 MB and no further, because a headless loop submits
633 // far faster than the GPU drains and a non-blocking poll finds almost
634 // nothing complete to retire. Waiting is what makes the retention
635 // per-frame instead of per-run.
636 //
637 // It is the same `poll(Wait)` `capture::read_back` already performs at
638 // every sampled frame, so this path pays what the sampled path always
639 // paid — and this whole file is off the hot path by construction (see
640 // the module docs); nothing in the frame loop or behind the C ABI
641 // reaches it.
642 //
643 // The result is discarded for the same reason `draw_frame`'s is above —
644 // this returns nothing, and a poll that fails means the device is gone,
645 // which the next `read_back` reports as `CaptureReadback` rather than
646 // letting the run pass silently.
647 let _ = self.ctx.device.poll(wgpu::PollType::wait_indefinitely());
648
649 #[cfg(feature = "text")]
650 self.text_layer.end_frame();
651 }
652}
653
654// ---------------------------------------------------------------------------
655// The sustained frame tap (Plan 0115 Phase 2)
656// ---------------------------------------------------------------------------
657
658impl Renderer {
659 /// Open a [`FrameTap`] sized to this renderer's configured target — the one
660 /// GPU allocation a tapped run makes, so the per-frame path makes none.
661 ///
662 /// The tap is fixed at the size it is built with. A [`resize`](Renderer::resize)
663 /// underneath a live tap leaves the two disagreeing, and the tap wins: it is
664 /// what [`render_tapped`](Self::render_tapped) draws and copies against.
665 /// Reopen after a resize to follow it.
666 ///
667 /// Infallible: `RenderContext` floors both dimensions at 1 where the size
668 /// enters, so there is nothing left to reject here.
669 pub fn open_tap(&self) -> FrameTap {
670 FrameTap::new(
671 &self.ctx.device,
672 &self.ctx.queue,
673 self.ctx.surface_format(),
674 self.ctx.config.width,
675 self.ctx.config.height,
676 )
677 }
678
679 /// Advance the scene clock by `dt` real seconds, draw the active preset for
680 /// `frame` through the same `draw_frame` the window presents through, and
681 /// hand back the **previous** call's frame out of `tap`.
682 ///
683 /// `dt` is **per call**, which is the difference between this and
684 /// [`capture_stream`](Self::capture_stream)'s one fixed step: a caller that
685 /// falls behind the wall clock yields fewer, correctly-timed frames rather
686 /// than a picture running slow against the music.
687 ///
688 /// Draws under `SaltMode::Live`, because a tap is a live render path and
689 /// not a capture: a preset declaring `seed = "random"` (ADR-0051) must vary
690 /// per launch here exactly as it does in the window. Both salts are equal for
691 /// every preset that declares anything else, which is why a tapped frame and
692 /// a [`capture_frame`](Self::capture_frame) of the same preset at the same
693 /// clock are still byte-identical (`core/tests/suite/frame_tap.rs`).
694 ///
695 /// # One frame in flight, and `Ok(None)` is the ordinary first answer
696 ///
697 /// This frame's submission carries the copy out of the tap's texture, the
698 /// *previous* frame's map is taken before it is drawn, and the caller sees
699 /// frame `N` while frame `N+1` is on the GPU. So the first call after
700 /// [`open_tap`](Self::open_tap) yields `Ok(None)`, and **every later call
701 /// yields a frame**.
702 ///
703 /// **It waits only when the GPU is behind.** A previous map that has landed
704 /// is taken without blocking; one that has not is waited for before this
705 /// frame is submitted, so at most one frame is ever queued and a caller
706 /// that outruns the GPU is held to the GPU's rate. [`FrameTap`] carries the
707 /// cycle and says why the wait is the loop's only backpressure. The same
708 /// bound keeps a long run's memory per-frame, the retention Plan 0099
709 /// measured.
710 ///
711 /// A `dt` that is not finite and positive is replaced by one nominal step
712 /// before the clock sees it (`sanitize_frame_dt`, ADR-0191).
713 ///
714 /// **Feeds the diagnostics frame clock**, as [`render`](Self::render) does,
715 /// because a tap is the whole of a windowless run's live output and
716 /// [`metrics`](Self::metrics) is what that run reports. Counted per frame
717 /// **drawn** rather than per frame handed back, so the rate is the tap's
718 /// own throughput and does not read as halved on the pipeline's first
719 /// call. No capture entry point feeds it: their frames are drawn offline
720 /// and would make a rate of nothing.
721 ///
722 /// **Times every pass it encodes**, where the tap has a timer — the query
723 /// set is armed around the draw, resolved into the same submission, and
724 /// read back on the same poll that takes the pixels, so a tapped run costs
725 /// one extra buffer copy and no extra wait (ADR-0245).
726 pub fn render_tapped(
727 &mut self,
728 tap: &mut FrameTap,
729 frame: &AnalysisFrame,
730 dt: f32,
731 ) -> Result<Option<CaptureImage>, RenderError> {
732 let dt = super::sanitize_frame_dt(dt);
733 let (width, height) = (tap.width, tap.height);
734 self.time += dt;
735 // Taken first: the buffer cannot be recorded into while it is mapped,
736 // and this is also where a caller ahead of the GPU waits for it. It
737 // collects the previous frame's pass timings too, while the timer
738 // still holds that frame's labels.
739 let image = tap.take_previous(&self.ctx.device);
740
741 let mut encoder = self
742 .ctx
743 .device
744 .create_command_encoder(&wgpu::CommandEncoderDescriptor {
745 label: Some("rlx-frame-tap"),
746 });
747 // Armed for the whole encode and taken back before the submit, so no
748 // pass encoded outside this window can write into the query set.
749 gpu::arm_pass_timer(tap.timer.take());
750 capture::record_clear(&mut encoder, &tap.view);
751 let draw_calls = self.draw_frame(
752 frame,
753 &mut encoder,
754 &tap.view,
755 (width, height),
756 dt,
757 SaltMode::Live,
758 );
759 capture::record_copy(
760 &mut encoder,
761 &tap.texture,
762 &tap.buffer,
763 tap.padded_bpr,
764 width,
765 height,
766 );
767 tap.timer = gpu::disarm_pass_timer(&mut encoder);
768 self.ctx.queue.submit(std::iter::once(encoder.finish()));
769 tap.arm();
770 if let Some(timer) = tap.timer.as_mut() {
771 timer.map();
772 }
773
774 #[cfg(feature = "text")]
775 self.text_layer.end_frame();
776
777 self.diag.set_draw_calls(draw_calls);
778 self.diag.record_frame();
779 Ok(image)
780 }
781
782 /// **Wait** for the frame `tap` still holds and hand it back, drawing
783 /// nothing.
784 ///
785 /// [`render_tapped`](Self::render_tapped) keeps one frame in flight, so a
786 /// run that stops asking for frames leaves one behind. This is how a
787 /// **bounded** run collects it, and how a test takes exactly one frame per
788 /// call: `render_tapped` then this, and the pipeline is empty again.
789 ///
790 /// `None` when nothing is in flight — an unused tap, or one already
791 /// drained.
792 ///
793 /// A live loop has no use for it: `render_tapped` already waits when the
794 /// GPU is behind, and calling this per frame would wait even when it is not.
795 pub fn drain_tap(&mut self, tap: &mut FrameTap) -> Option<CaptureImage> {
796 tap.drain(&self.ctx.device)
797 }
798}