rlx_core/render/tier.rs
1//! Quality tiers: the engine's capacity constants, resolved once (ADR-0045).
2//!
3//! NFR §1 specifies two quality levels — a reduced tier holding 60 fps
4//! at 1080p on the ~2015-iGPU baseline, and a richer presentation on
5//! capable hardware. This module is where both live.
6//!
7//! # What a tier is
8//!
9//! A [`TierConfig`] is a plain struct of **capacity** values: how many particles,
10//! how many segments, how large an internal grid may get. Nothing here changes
11//! *what* the engine draws, only how much of it — which is what makes
12//! [`Tier::Floor`] byte-identical to the pre-tier engine and what lets captures
13//! pin it (see below). A value that changes the *content* of a frame — the
14//! reaction-diffusion simulation grid, whose pattern scale moves with its
15//! resolution (ADR-0034) — deliberately does **not** live here.
16//!
17//! **That separation is a property of the consuming scene, not of this
18//! struct.** The attractor draws its particles with an *additive* blend
19//! into a linear accumulation, so
20//! [`attractor_particles`](TierConfig::attractor_particles) sets the
21//! total light in the frame as directly as it sets the sample count —
22//! `Rich` rendered every attractor preset three stops hot behind a
23//! green suite, because no capture pins `Rich`. What holds the claim up
24//! is [`deposit_scale`](super::scenes::particles::deposit_scale)
25//! dividing the deposit by the count (ADR-0065). The lesson
26//! generalizes: **a count feeding an accumulating pass is a look value
27//! until something normalizes it.** If a future field lands here for
28//! such a pass, that normalization is part of adding it.
29//!
30//! # Where the numbers come from
31//!
32//! [`TierConfig::FLOOR`] is the pre-tier engine, constant for constant: each
33//! value's former definition site now reads this struct, so no number exists
34//! twice. Its justifications came with it and are on the fields.
35//! [`TierConfig::RICH`] is calibrated against a midrange discrete GPU
36//! (RTX 3060 / RX 6600 class) on device — Plan 0044 Phase 4 — rather than
37//! asserted from a multiplier.
38//!
39//! # Resolution and the governor
40//!
41//! The tier resolves **once, at renderer construction**, from an optional pin
42//! ([`RendererOptions`](super::RendererOptions)); unpinned resolves [`Tier::Rich`].
43//! An unpinned renderer may then be demoted to [`Tier::Floor`] by the frame-time
44//! governor — one way, once per session, never silently. A pinned tier never
45//! moves.
46//!
47//! Headless capture is [`Tier::Floor`] **by construction**:
48//! [`Renderer::new_headless`](super::Renderer::new_headless) cannot produce any
49//! other tier, so every golden baseline stays byte-reproducible on the WARP
50//! software adapter and the suite's cost does not scale with the rich tier.
51//! [`Renderer::new_headless_tiered`](super::Renderer::new_headless_tiered) is the
52//! deliberate opt-in the `shot` CLI's `--tier` reaches.
53//!
54//! # The grid scale
55//!
56//! One value in a resolved [`TierConfig`] is not the tier's alone:
57//! [`grid_scale`](TierConfig::grid_scale), the fraction of the render target
58//! the internal grids are drawn at (ADR-0245). It comes from the tier **and**
59//! the adapter's class through `grid_scale_for`, or from an explicit pin, and
60//! `resolve_grid_scale` is the one place the three are weighed. It resolves
61//! when the tier does — at construction, on a tier change and on an adapter
62//! change — and a frame never reads it.
63//!
64//! Pure and GPU-free throughout — a tier is a set of numbers, so it is decided
65//! without a device and tested without one.
66
67// Hot-path panic-denial pragma (Plan 0002 Phase 2; render/ is scanned by the
68// hygiene guard). The governor runs once per displayed frame.
69#![deny(
70 clippy::unwrap_used,
71 clippy::expect_used,
72 clippy::indexing_slicing,
73 clippy::panic,
74 clippy::unreachable
75)]
76
77use super::context::AdapterClass;
78
79/// Which quality tier a renderer is running (ADR-0045).
80///
81/// Two named levels rather than a continuum: the output of a preset has to be
82/// predictable enough to baseline, document, and reproduce in a bug report, which
83/// a load-history-dependent feature-shedding scheme cannot deliver (ADR-0045
84/// Alternative B).
85#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)]
86pub enum Tier {
87 /// The NFR §1/§2 iGPU floor — the pre-tier engine's exact constants. The
88 /// default here because it is the safe answer: a `Tier` value that appeared
89 /// from nowhere should not raise anyone's budgets.
90 #[default]
91 Floor,
92 /// Calibrated for a midrange discrete GPU: higher particle, segment and
93 /// resolution budgets, same visual grammar.
94 Rich,
95}
96
97impl Tier {
98 /// The lowercase name the CLI, the env var and the config file all use.
99 pub fn as_str(self) -> &'static str {
100 match self {
101 Tier::Floor => "floor",
102 Tier::Rich => "rich",
103 }
104 }
105
106 /// The uppercase name the 5x7 diagnostics overlay paints. Separate from
107 /// [`as_str`](Self::as_str) only because that font has no lowercase glyphs;
108 /// both come off the same match so there is no second spelling to drift.
109 pub fn label(self) -> &'static str {
110 match self {
111 Tier::Floor => "FLOOR",
112 Tier::Rich => "RICH",
113 }
114 }
115
116 /// Parse a tier name, case-insensitively. `None` for anything else — callers
117 /// surface that as a usage error rather than guessing a tier.
118 pub fn from_name(name: &str) -> Option<Self> {
119 match name.trim().to_ascii_lowercase().as_str() {
120 "floor" => Some(Tier::Floor),
121 "rich" => Some(Tier::Rich),
122 _ => None,
123 }
124 }
125}
126
127/// The fraction of the render target every internal grid is drawn at — the post
128/// chain's grid and the attractor's trail field (ADR-0245).
129///
130/// A validated value: only [`new`](Self::new) and [`parse`](Self::parse) make
131/// one, and both refuse anything outside [`MIN`](Self::MIN)`..=`[`MAX`](Self::MAX),
132/// NaN included. That is what makes the `Eq` below honest — a `GridScale` is
133/// never NaN, so it equals itself.
134///
135/// A **capacity**, not a look value, in the sense the module docs use: it sets
136/// how many texels a field holds, and a grid is a resolution rather than a
137/// shape (ADR-0037), so the picture's geometry does not move with it. What does
138/// move is sharpness, which is the trade.
139#[derive(Clone, Copy, Debug, PartialEq, PartialOrd)]
140pub struct GridScale(f32);
141
142impl Eq for GridScale {}
143
144impl GridScale {
145 /// The smallest fraction accepted. Below a quarter a 1080p field is under
146 /// 480x270 and the grid floor (`grid::MIN_AXIS`) starts deciding the size
147 /// instead of the scale.
148 pub const MIN: f32 = 0.25;
149 /// The largest: the target's own resolution. A grid is never drawn above it.
150 pub const MAX: f32 = 1.0;
151 /// The whole target — every row of the table but Rich on an integrated
152 /// adapter, and what a headless renderer resolves unless it is told
153 /// otherwise.
154 pub const FULL: Self = Self(1.0);
155 /// Rich on an integrated adapter: the measured row of `grid_scale_for`.
156 pub(crate) const RICH_INTEGRATED: Self = Self(0.75);
157
158 /// `value` as a scale, or `None` when it is outside `MIN..=MAX` or not a
159 /// number.
160 pub fn new(value: f32) -> Option<Self> {
161 (Self::MIN..=Self::MAX)
162 .contains(&value)
163 .then_some(Self(value))
164 }
165
166 /// Parse a scale from its written form — the flag's value, the environment
167 /// variable's, or a settings value. The error names the accepted range,
168 /// because a caller surfaces it as a usage error verbatim.
169 pub fn parse(text: &str) -> Result<Self, String> {
170 let trimmed = text.trim();
171 trimmed
172 .parse::<f32>()
173 .ok()
174 .and_then(Self::new)
175 .ok_or_else(|| {
176 format!(
177 "`{trimmed}` is not a grid scale: expected a number from {} to {}",
178 Self::MIN,
179 Self::MAX
180 )
181 })
182 }
183
184 /// The fraction itself.
185 pub fn get(self) -> f32 {
186 self.0
187 }
188
189 /// `px` target pixels, as the texels a grid at this scale asks for before it
190 /// is quantized: `round(px * scale^2)`, saturating.
191 ///
192 /// **Exactly `px` at [`FULL`](Self::FULL)**, because `1.0 * 1.0` is exact and
193 /// every `u32` is exact in `f64` — so a count derived from this at scale 1.0
194 /// is the count derived from the target, byte for byte.
195 pub fn texels(self, px: u32) -> u32 {
196 let scaled = (f64::from(px) * f64::from(self.0) * f64::from(self.0)).round();
197 if scaled >= f64::from(u32::MAX) {
198 u32::MAX
199 } else {
200 scaled as u32
201 }
202 }
203}
204
205impl Default for GridScale {
206 fn default() -> Self {
207 Self::FULL
208 }
209}
210
211impl std::fmt::Display for GridScale {
212 /// Two decimals — `1.00`, `0.75` — the form the overlay and a log line print.
213 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
214 write!(f, "{:.2}", self.0)
215 }
216}
217
218/// **The grid-scale table**: the fraction a tier draws its internal grids at on
219/// an adapter of `class` (ADR-0245).
220///
221/// The two integrated rows are a reading, not a choice made here: the largest
222/// scale whose live 1080p window held a 60 fps median with no one-second sample
223/// under 60 across the ten bench presets on the reference laptop's integrated
224/// GPU (Plan 0223 Phase 6). Floor holds at 1.0; Rich needs 0.75. Neither is a
225/// promise above 1080p — at 2560x1440 no Rich scale held. The discrete,
226/// software and other rows are 1.0 **by decision** rather
227/// than by measurement: a discrete GPU holds the display rate
228/// at full resolution, and every golden baseline is taken on a software
229/// rasterizer, so a fraction there would move a picture no measurement asked to
230/// move.
231pub(crate) fn grid_scale_for(tier: Tier, class: AdapterClass) -> GridScale {
232 match (tier, class) {
233 (Tier::Floor, AdapterClass::Integrated) => GridScale::FULL,
234 (Tier::Rich, AdapterClass::Integrated) => GridScale::RICH_INTEGRATED,
235 (_, AdapterClass::Discrete | AdapterClass::Software | AdapterClass::Other) => {
236 GridScale::FULL
237 }
238 }
239}
240
241/// **The one grid-scale resolution in the engine**: an explicit pin wins, a
242/// renderer with no surface takes [`GridScale::FULL`], and a window takes the
243/// table's row.
244///
245/// The headless arm is the same guarantee [`Renderer::new_headless`] makes
246/// about the tier: a capture is a function of its inputs and not of the machine
247/// that took it, so a `shot` on an integrated laptop and one on a discrete
248/// desktop draw the same grid. A pin is how a headless run asks for less.
249///
250/// Pure, so all three arms are asserted without a device.
251///
252/// [`Renderer::new_headless`]: super::Renderer::new_headless
253pub(crate) fn resolve_grid_scale(
254 pin: Option<GridScale>,
255 tier: Tier,
256 class: AdapterClass,
257 has_surface: bool,
258) -> GridScale {
259 match pin {
260 Some(scale) => scale,
261 None if has_surface => grid_scale_for(tier, class),
262 None => GridScale::FULL,
263 }
264}
265
266/// The capacity values a tier sets. Resolved once at renderer construction and
267/// read at construction/reconfigure time only — never branched on per frame.
268#[derive(Clone, Copy, Debug, PartialEq, Eq)]
269pub struct TierConfig {
270 /// Which tier these values are, so a demotion and the overlay have one thing
271 /// to read rather than a parallel field to keep in step.
272 pub tier: Tier,
273
274 /// The fraction of the render target both internal grids are drawn at
275 /// ([`GridScale`], ADR-0245) — the one value here that is not the tier's
276 /// alone.
277 ///
278 /// The constants below carry [`GridScale::FULL`]; the renderer replaces it
279 /// with `resolve_grid_scale`'s answer for its own adapter and pin whenever
280 /// it takes a config, so a `TierConfig` a renderer holds names the scale its
281 /// grids were actually built at. It rides here rather than beside the struct
282 /// because both grids are built from a `TierConfig` already, and a rebuild
283 /// that took the tier and forgot the scale is then not writable.
284 pub grid_scale: GridScale,
285
286 /// Cap on a post stage's internal grid (ADR-0034), width then height.
287 ///
288 /// The floor value is NFR §12 memory arithmetic, **redone for the linear-light
289 /// composite** (Plan 0045 Phase 3 / ADR-0046). Every intermediate upstream of
290 /// the tonemap is now `COMPOSITE_FORMAT` — 8
291 /// bytes/texel, not 4 — so a stage offscreen costs twice what the surface
292 /// format would charge, while the trails accumulation
293 /// (`PingPongField`, two textures) was
294 /// already float and did not move.
295 ///
296 /// Per chain, both stages live, at this cap (1920x1080, 8 bytes/texel =
297 /// 16.6 MB a texture):
298 ///
299 /// | buffer | before | after |
300 /// |----------------------------|--------|-------|
301 /// | trails composited | 8.3 | 16.6 |
302 /// | trails accumulation (x2) | 33.2 | 33.2 |
303 /// | kaleidoscope source | 8.3 | 16.6 |
304 /// | **per chain** | **50** | **66** |
305 ///
306 /// Plan 0023's dual-live dissolve holds two whole `PostChain`s, so the peak is
307 /// ~133 MB rather than ~100. Outside the chains the frame carries one more
308 /// surface-sized float buffer beyond the chains — the tonemap's input, 16.6
309 /// MB, the one allocation ADR-0046 genuinely adds — plus the blend's
310 /// snapshot/live pair at 16.6 MB each while a dissolve runs (8.3 at 8-bit),
311 /// and ink's 8.3 MB input, which stays 8-bit because the tonemap hands it
312 /// display-referred pixels. Worst case — dual-live, both stages, ink on — is
313 /// ~191 MB against NFR §12's ~350 MB soft ceiling, which is mostly driver
314 /// floor already.
315 ///
316 /// At the rich cap (2560x1440) the same arithmetic is ~118 MB per chain and
317 /// ~236 MB dual-live, up from ~88 and ~177 — the trade ADR-0034 priced and
318 /// declined at floor budgets and the rich tier takes.
319 ///
320 /// **This cap is the relief lever** if the float chain misses NFR §1 on a
321 /// floor-tier iGPU: lower it rather than re-fixing the grids (ADR-0046), since
322 /// bandwidth roughly doubled with the format and the grid policy is shared.
323 ///
324 /// **Bloom adds to this only when a preset switches it on** (Plan 0045
325 /// Phase 4), and it is **two** allocations, not one. Its pyramid is two
326 /// textures per level, each level a quarter of the last, so the pyramid
327 /// converges to `2 * (1/4 + 1/16 + …) ≈ 2/3` of one grid-sized texture — ~11 MB
328 /// at this cap. On top of that the stage owns its own **grid-sized `bloom-src`
329 /// offscreen**, a full 16.6 MB at this cap, because a `PostStage` reads its
330 /// input from a texture it owns. So the stage costs **16.6 + ~11 ≈ 28 MB** on
331 /// top of the ~66 MB per chain above, and ~55 MB in the dual-live worst case —
332 /// which is what NFR §12's table charges and what the ~246 MB worst case there
333 /// is computed from. It is charged only against presets that bind
334 /// `bloom_amount`, since an inactive stage builds nothing.
335 pub post_cap: (u32, u32),
336
337 /// How many levels deep the bloom pyramid goes
338 /// (`Bloom`, ADR-0046).
339 ///
340 /// This is a **capacity**, not a look: each level doubles the halo's reach and
341 /// costs three passes at a quarter of the previous level's area, so the tail
342 /// is cheap in pixels and not free in passes. The floor runs four (a halo
343 /// reaching ~16 of the grid's texels at the default radius); rich runs six,
344 /// which is where the widest levels start to matter on a 1440p-class grid.
345 ///
346 /// `level_sizes` clamps this down on a small render target, so
347 /// a value here is an upper bound rather than a promise.
348 pub bloom_levels: u32,
349
350 /// The attractor's sample budget **at [`REFERENCE_PX`]** — the anchor of the
351 /// density law, not the count drawn (ADR-0140).
352 ///
353 /// [`attractor_budget`] scales this by `target_px / REFERENCE_PX` and clamps
354 /// it between this value and one of the two ceilings below, so a target at or
355 /// under the reference draws exactly this many and a larger one draws more.
356 /// What a *preset* then draws out of that budget is
357 /// `round(budget * density)` (ADR-0069), which is a different and smaller
358 /// number again.
359 ///
360 /// State is 48 bytes each ([`Particle`](super::scenes::particles)), and the
361 /// real ceiling is **additive-blend fill rate**, which is why the floor value
362 /// was described as the number to validate against the 60 fps @ 1080p floor
363 /// (ADR-0015 Risks).
364 ///
365 /// This is a sample count and **not** a brightness: the additive draw divides
366 /// its deposit by the *active* count
367 /// ([`deposit_scale`](super::scenes::particles::deposit_scale), ADR-0065), so
368 /// raising it buys a smoother figure rather than a brighter one. Changing it
369 /// changes shot noise and cost; it does not change exposure.
370 pub attractor_particles: u32,
371
372 /// The largest budget [`attractor_budget`] may resolve for a scene drawing
373 /// into a **live surface** — a window, or the plugin's host surface.
374 ///
375 /// Frame-time bound, and it is the number that keeps the law from spending a
376 /// display's whole budget on sample count. It is also the **allocation**: the
377 /// particle buffer is sized here at construction and never resized, so a
378 /// resize changes the active count and rebuilds no GPU resource. That costs
379 /// `ceiling * 48 B` of GPU storage plus the same again for the CPU seed
380 /// scatter the scene holds for re-upload, in **every** window, whether or not
381 /// the target is large and whether or not an attractor preset is loaded
382 /// (`create_all` builds every scene up front).
383 ///
384 /// # Where these numbers come from
385 ///
386 /// Measured, not chosen — Plan 0128 Phase 1, at 1920x1080 on
387 /// `attractor_leviathan`, counts interleaved in one process so a throttling
388 /// laptop GPU could not read as signal.
389 ///
390 /// **`Rich`: 600 000**, four times the anchor. On the midrange-discrete
391 /// reference NFR §1 calibrates `Rich` against, the rule was *the largest swept
392 /// count whose marginal p99 over today's anchor stays inside 10 % of the
393 /// 16.67 ms budget*: 600 000 reads +1.454 ms (8.7 %) and the next step up,
394 /// 1 200 000, reads +4.703 ms (28.2 %).
395 ///
396 /// **`Floor`: its own anchor**, so the law is a no-op there. On integrated
397 /// hardware — the baseline NFR §1's floor commitment is about — 1080p at
398 /// `Floor` already sits *on* the 16.67 ms budget at today's 50 000 (p99
399 /// 16.854 ms), and the law's own 1080p `Floor` value of 450 000 takes it to
400 /// 31.942 ms. NFR §1 promises `Floor` "values exactly the pre-tier engine's";
401 /// this is what keeps that true at every target size.
402 pub attractor_particles_live_ceiling: u32,
403
404 /// The same ceiling for a **headless render** — `shot --render`, where there
405 /// is no present deadline and no governor, and the only bound is memory.
406 ///
407 /// # Where these numbers come from
408 ///
409 /// The bound is the **device's storage-buffer binding limit**, not process
410 /// memory, because it is reached first: at 48 B a particle, wgpu's default
411 /// `max_storage_buffer_binding_size` of 134 217 728 B holds 2 796 202
412 /// particles, and 5 400 000 — the law's own unclamped 4K value — fails
413 /// outright with `Buffer binding 0 range 259200000 exceeds
414 /// max_*_buffer_binding_size limit 134217728` (Plan 0128 Phase 1).
415 ///
416 /// `Rich` takes **2 700 000**, the largest whole multiple of its anchor under
417 /// that wall — 18x, 129.6 MB, 3.4 % of headroom. `Floor` takes the **same
418 /// multiple** rather than the same number, so a tier still means something
419 /// offline: at one shared ceiling `--tier floor --render` and
420 /// `--tier rich --render` would draw an identical count at 4K.
421 pub attractor_particles_offline_ceiling: u32,
422
423 /// Upper bound on each axis of the attractor's trail accumulation grid.
424 ///
425 /// The floor is the ceiling Plan 0027/0029 chose for a high-DPI display while
426 /// keeping the worst case bounded on the iGPU: every frame pays a decay pass
427 /// plus the full additive instance draw over this grid, so cost scales with
428 /// its area. Rich lifts it to 4K so a 4K or ultrawide display sizes near 1:1
429 /// instead of degrading to a uniform upscale.
430 pub attractor_trail_cap: (u32, u32),
431
432 /// How many particles the swarm simulates.
433 ///
434 /// Plan 0043 left the floor value at an **unmeasured** iGPU cost (+0.5 ms per
435 /// frame of depth math on the dev box, and not fill rate), which is why the
436 /// plans index calls this a live tier candidate rather than a settled
437 /// constant. If the on-device floor check misses, this is the lever — on the
438 /// floor value, and that routes back through `architect`.
439 pub swarm_particles: usize,
440
441 /// How many objects the emitter's pool holds (ADR-0057).
442 ///
443 /// Unlike every other count here this is a **ceiling on a varying
444 /// population**, not the population: the emitter spawns and retires, so a
445 /// preset's `spawn_rate * lifetime` decides how many objects are actually
446 /// alive and this decides how many *can* be. Spawns past it are dropped
447 /// rather than queued or allocated for — that is the phase's whole real-time
448 /// hazard — so raising it does not brighten a preset that never reaches it,
449 /// and lowering it below one that does thins the shower rather than changing
450 /// its motion.
451 ///
452 /// Not an accumulating count in the sense the module docs warn about: each
453 /// object is one sprite drawn once per frame, so the light in the frame is
454 /// `population * brightness` and the population is a preset's own arithmetic.
455 /// The tier only says where that arithmetic is cut off.
456 ///
457 /// The floor holds a shipped preset's shower with room to spare (the emitter
458 /// family runs a few hundred objects live); rich triples it for the denser
459 /// looks a discrete GPU can carry. Cheap either way — see NFR §12: the pool
460 /// and its instance buffer are well under a megabyte at both tiers.
461 pub emitter_objects: usize,
462
463 /// Ceiling on the warp mesh's grid, in **cells**, width then height
464 /// (Plan 0100 Phase 1).
465 ///
466 /// A capacity in the strictest sense: the grid is a *resolution* for the
467 /// per-vertex program, not a shape (ADR-0037), so raising it refines the
468 /// warp's spatial detail and changes nothing about what the scene draws. The
469 /// vertex count is `(x + 1) * (y + 1)`, and every one of those vertices costs
470 /// one evaluation of each `[per_vertex]` binding **on the render thread**,
471 /// which is what this bounds.
472 ///
473 /// The upper bound either tier may name is the `.milk` format's own —
474 /// `meshx <= 128`, `meshy <= 96` — so a converted preset's requested grid is
475 /// representable at the top of the range and clamped below it.
476 /// [`warp_mesh::clamp_grid`](super::scenes::warp_mesh::clamp_grid) is the one
477 /// place the clamp happens, shared by the loader and the scene.
478 ///
479 /// # Where these numbers come from
480 ///
481 /// **Measured, not chosen** — Plan 0100 Phase 1's done-when. The rule it set
482 /// was: raise the grid until one frame of per-vertex evaluation costs more
483 /// than **1 ms** — 6 % of the 16.67 ms NFR §1 commits to at 1080p — and cap
484 /// the floor one step below.
485 ///
486 /// `mesh_cost_by_grid` in `scenes/warp_mesh/tests.rs` is the measurement and
487 /// prints the ladder on every run. Taken **2026-08-16 on the development box
488 /// (Windows 10, desktop CPU, `--release`)**, evaluating a four-binding
489 /// `[per_vertex]` program of the shape a real preset writes — two runs,
490 /// agreeing to about 1 %:
491 ///
492 /// ```text
493 /// grid vertices per frame share of 16.67 ms
494 /// 16x12 221 0.036 ms 0.2 %
495 /// 32x24 825 0.129 ms 0.8 %
496 /// 48x36 1 813 0.280 ms 1.7 %
497 /// 64x48 3 185 0.488 ms 2.9 % <- Floor
498 /// 72x54 4 015 0.616 ms 3.7 %
499 /// 80x60 4 941 0.755 ms 4.5 %
500 /// 88x66 5 963 0.909 ms 5.5 % <- Rich
501 /// 96x72 7 081 1.081 ms 6.5 % <- the bar is crossed here
502 /// 112x84 9 605 1.483 ms 8.9 %
503 /// 128x96 12 513 1.918 ms 11.5 %
504 /// ```
505 ///
506 /// **The bar is crossed between `88x66` and `96x72`**, so `88x66` is the
507 /// largest grid the rule admits and it is what `Rich` takes. The format's own
508 /// maximum is therefore **refused**: at 1.92 ms it is 11.5 % of the frame on
509 /// a desktop CPU, which is not a number any tier should spend on one
510 /// parameter surface. The grid is lowered because it did not measure clean,
511 /// which is the whole of the rule.
512 ///
513 /// **`Floor` sits a step further down than the rule alone would put it, and
514 /// deliberately.** The rig above is a desktop CPU; NFR §1's floor tier
515 /// targets a ~2015 iGPU-class machine whose single-thread performance this
516 /// box does not model, and this is CPU work on the render thread, so a
517 /// slower machine pays proportionally more of a budget it is already
518 /// struggling to hold. `64x48` is 2.9 % here and leaves room for that
519 /// machine to be several times slower before the surface is a problem.
520 ///
521 /// **When the floor tier is next exercised on real target hardware this is
522 /// the constant to re-measure**, and the ladder prints exactly what that
523 /// needs.
524 pub mesh_grid: (u32, u32),
525
526 /// The one capacity value here that a preset can *see*: past it geometry is
527 /// truncated, and ADR-0007 requires that be surfaced rather than silently cut.
528 /// So a preset whose mirror pushes over the floor cap reports an overflow at
529 /// the floor and not at rich — the message is the tier's most visible edge for
530 /// the content lane, which is why shipped presets are authored against the
531 /// floor.
532 pub max_segments: usize,
533
534 /// Cap on how many flat elements a `shape_collage` canvas may hold
535 /// (ADR-0123).
536 ///
537 /// **The one capacity here that bounds a per-pixel loop**, which is what
538 /// makes it load-bearing rather than a memory number. Every other count in
539 /// this struct bounds work paid once per particle, per vertex or per
540 /// segment; this one bounds work every *fragment* pays, so a frame costs
541 /// `elements x pixels` and ADR-0123 prices the bounding-box reject alone —
542 /// before anything is drawn — at roughly `6N` operations per pixel. The
543 /// buffer is irrelevant at any value either tier would take: 64 bytes an
544 /// element, so 128 elements is 8 KB.
545 ///
546 /// # Where the floor value comes from
547 ///
548 /// **Measured, then decided by a human** — Plan 0113 Phases 2 and 3.
549 /// `core/tests/collage_cost.rs` sweeps the count on hardware, prints
550 /// the ladder on every run, and its module docs own the readings and
551 /// the trap in quoting them. **Two tables are not interchangeable**:
552 /// the pre-roster ladder is what the Phase 3 gate read, and the
553 /// post-roster table is what a canvas costs today — the eight-kind
554 /// roster made the loop cheaper, because rings, sectors and checker
555 /// patches shade far less of their own bounding box than a quad
556 /// does.
557 ///
558 /// **40 is the reference set's own top, not a budget line.** The
559 /// gate was a look judgement and the cost was not the binding
560 /// constraint: the user's working density is **8 to 14 elements**,
561 /// which on the system as shipped costs **8.2 % of a 60 Hz frame at
562 /// eight and 10.7 % at sixteen**, and denser canvases were rejected
563 /// on sight long before they were rejected on cost. The ceiling is
564 /// the densest canvas in ADR-0123's roster, Kandinsky's *On White
565 /// II*, counted at just above 40 once its lines and arcs are
566 /// included — so this value sits **exactly on** that canvas, and a
567 /// `collage_onwhite` needing a forty-first element moves this
568 /// number rather than being quietly truncated.
569 ///
570 /// `Rich` is provisional in the sense every [`RICH`](TierConfig::RICH) value
571 /// is — see that constant's own note.
572 ///
573 /// # It clamps, and it does not yet say so
574 ///
575 /// `shape_collage::applied_count` holds a bound `count` to this value
576 /// **silently**, unlike [`max_segments`](Self::max_segments), which
577 /// ADR-0007 requires surface an overflow. That was harmless while the cap
578 /// sat far above any authored canvas and is not harmless now that it sits on
579 /// one. Recorded as a followup on Plan 0113 rather than fixed there: the
580 /// surfaced channel is [`CapOverflow`](super::scenes::lines::CapOverflow),
581 /// whose context enum is shared with the line scenes, so widening it is an
582 /// architect call.
583 pub collage_elements: usize,
584
585 /// The most iterations an `analytic_field` escape-time preset may follow an
586 /// orbit for (ADR-0045, ADR-0180 rule 3).
587 ///
588 /// **The one value here that can change a preset's picture**, which is
589 /// why it is a *cap on what the preset asks for* rather than a count the
590 /// tier chooses. An escape-time boundary at 48 iterations and at 512 is not
591 /// the same image, where a particle cloud at two densities is. So a preset
592 /// asks for `iterations`, this bounds it, and a preset asking within the
593 /// `Floor` value renders identically on both tiers. One that asks for more
594 /// is clamped and **says so**, through the same per-frame overflow the
595 /// segment cap surfaces through
596 /// ([`OverflowContext::Iterations`](super::scenes::OverflowContext::Iterations)),
597 /// which the standalone reports on the transition exactly as it reports a
598 /// tier demotion.
599 ///
600 /// # Where the numbers come from
601 ///
602 /// **Arithmetic, not a measurement.** It is a fullscreen per-pixel loop, so
603 /// the worst frame is one whose every pixel runs the whole budget: 1080p is
604 /// 2.07 M pixels, and one step of `z^2 + c` plus its escape test is about a
605 /// dozen floating-point operations, so 64 iterations is ~1.6 GFLOP a frame.
606 /// A ~2015 integrated GPU — the baseline NFR §1's floor is about — peaks at a
607 /// few hundred GFLOPS, which puts that frame at single-digit milliseconds at
608 /// peak and roughly two to three times that in practice: inside the 16.67 ms
609 /// budget with the composite still to pay for. 128 would sit on the budget.
610 /// `Rich` takes the top of `iterations`' own declared range.
611 ///
612 /// **When the floor tier is next exercised on real target hardware this is
613 /// a constant to measure**, on the heaviest escape-time frame: a zoom whose
614 /// whole view is the set's interior.
615 pub field_iterations: u32,
616
617 /// The widest neighbourhood a `cellular` `larger_than_life` preset may ask
618 /// for, as a `radius` in cells (ADR-0045, ADR-0180).
619 ///
620 /// Like [`field_iterations`](Self::field_iterations) this is a **cap on what
621 /// the preset asks for**, because a radius is content: the same rule at two
622 /// radii grows two different worlds, where a particle count at two densities
623 /// is one. A preset asking within the `Floor` value runs identically on both
624 /// tiers.
625 ///
626 /// # Where the numbers come from
627 ///
628 /// **Arithmetic, not a measurement.** The neighbourhood sum is separated — a
629 /// row pass and a column sum — so one generation costs `2 * (2r + 1) + 1`
630 /// texel reads a cell rather than `(2r + 1)^2`. At `Floor`'s 6 that is 27
631 /// reads; on a 512-cell grid at 60 generations a second, 425 M reads a
632 /// second, a few percent of what a ~2015 integrated GPU — the baseline NFR
633 /// §1's floor is about — fetches, and the grid and rate both have to sit at
634 /// their tops to reach it. `Rich` takes the top of `radius`' declared range,
635 /// 10: 43 reads, 2.7 G a second at a 1024 grid and the same rate. Unseparated,
636 /// radius 10 would be 441 reads a cell — ten times that.
637 ///
638 /// `Floor` is 6 rather than the whole range because the default rule (at
639 /// radius 5) fits under it with one to spare, and a floor
640 /// machine pays for every step past it at every generation.
641 ///
642 /// **When the floor tier is next exercised on real target hardware this is
643 /// a constant to measure**, at the tier's own grid cap and 60 generations a
644 /// second.
645 pub cellular_radius: u32,
646
647 /// The largest `[cellular] grid` a preset may run on, in cells a side
648 /// (ADR-0045).
649 ///
650 /// **A cap on content, like [`cellular_radius`](Self::cellular_radius),
651 /// and it is clamped and announced rather than silently reduced**: a
652 /// pattern is a fixed number of cells, so a grid clamped from 1024 to 512
653 /// draws every glider twice as large. A preset asking within the `Floor`
654 /// value runs identically on both tiers; one asking past it runs on the cap
655 /// and says so at load, through the cap overflow the standalone prints
656 /// ([`OverflowContext::Grid`](super::scenes::OverflowContext::Grid)).
657 ///
658 /// # Where the numbers come from
659 ///
660 /// **Arithmetic, not a measurement**, from the same read count
661 /// [`cellular_radius`](Self::cellular_radius) states. `Floor`'s 512 is the
662 /// grid that figure is quoted at — 262 144 cells, 425 M texel reads a
663 /// second at `Floor`'s radius and 60 generations a second — and its state
664 /// textures are 4 MB of the ping-pong pair and the row counts. `Rich` takes
665 /// the loader's own ceiling, 1024: four times the cells, 16 MB.
666 ///
667 /// **When the floor tier is next exercised on real target hardware this is
668 /// a constant to measure**, with `cellular_radius`, on one frame.
669 pub cellular_grid: u32,
670
671 /// The most points a `plexus` preset draws (ADR-0257), holding its
672 /// `[plexus] points` at load.
673 ///
674 /// **A cap on content, clamped and announced at load**
675 /// ([`OverflowContext::Points`](super::scenes::OverflowContext::Points)):
676 /// fewer points is a sparser network, not the same one cheaper.
677 ///
678 /// The graph is built by testing every pair, so the per-frame cost is
679 /// quadratic in this: `N (N - 1) / 2` distance checks. `Floor`'s 600 is
680 /// 180 thousand checks a frame and `Rich`'s 1500 is 1.1 million.
681 ///
682 /// # Where the numbers come from
683 ///
684 /// `Floor`'s three plexus caps were **measured together** (Plan 0235 Phase
685 /// 6): a preset at 600 points with links saturating the edge cap, nodes on
686 /// and an aperture past the blur cap costs under 4 ms a frame at 1080p on an
687 /// integrated GPU, which leaves the frame budget its composite. `Rich`'s are
688 /// the same proportions scaled, not measured.
689 pub plexus_points: u32,
690
691 /// The most edges a `plexus` frame draws (ADR-0257): the size of its
692 /// instance buffer, and so the fill the graph can cost.
693 ///
694 /// A frame whose graph links more pairs than this draws the first this many
695 /// in index order and announces the rest
696 /// ([`OverflowContext::Edges`](super::scenes::OverflowContext::Edges)).
697 /// Measured with [`plexus_points`](Self::plexus_points).
698 pub plexus_edges: u32,
699
700 /// The largest circle of confusion a 3D primitive is blurred by, as a
701 /// radius in pixels (ADR-0257).
702 ///
703 /// A blurred stroke costs fill in proportion to how far it is widened, so
704 /// this is the lever that keeps a wide-open aperture inside the frame
705 /// budget: past it a line stops spreading, and an operator on a lower tier
706 /// sees a shallower blur rather than a slower frame — and is told so
707 /// ([`OverflowContext::Blur`](super::scenes::OverflowContext::Blur)).
708 /// Measured with [`plexus_points`](Self::plexus_points).
709 pub max_coc_px: u32,
710}
711
712impl TierConfig {
713 /// The iGPU floor: the pre-tier engine's constants, unchanged.
714 pub const FLOOR: Self = Self {
715 tier: Tier::Floor,
716 grid_scale: GridScale::FULL,
717 post_cap: (1920, 1080),
718 bloom_levels: 4,
719 attractor_particles: 50_000,
720 attractor_particles_live_ceiling: 50_000,
721 attractor_particles_offline_ceiling: 900_000,
722 attractor_trail_cap: (2560, 1440),
723 swarm_particles: 10_000,
724 emitter_objects: 2_000,
725 mesh_grid: (64, 48),
726 max_segments: 20_000,
727 collage_elements: 40,
728 field_iterations: 64,
729 cellular_radius: 6,
730 cellular_grid: 512,
731 plexus_points: 600,
732 plexus_edges: 6_000,
733 max_coc_px: 12,
734 };
735
736 /// The midrange-discrete tier.
737 ///
738 /// **These are provisional multipliers, not measurements.** Plan 0044 Phase 4
739 /// runs the standalone pinned here on the target GPU at native fullscreen
740 /// across the heaviest preset of each family and records the frame times; the
741 /// values that ship are the ones that hold the display rate. A number that
742 /// misses gets lowered, and no number here is invented upward to look good.
743 /// Until that phase closes, treat every field below as a starting point.
744 pub const RICH: Self = Self {
745 tier: Tier::Rich,
746 grid_scale: GridScale::FULL,
747 post_cap: (2560, 1440),
748 bloom_levels: 6,
749 attractor_particles: 150_000,
750 attractor_particles_live_ceiling: 600_000,
751 attractor_particles_offline_ceiling: 2_700_000,
752 attractor_trail_cap: (3840, 2160),
753 swarm_particles: 30_000,
754 emitter_objects: 6_000,
755 mesh_grid: (88, 66),
756 max_segments: 60_000,
757 collage_elements: 96,
758 field_iterations: 512,
759 cellular_radius: 10,
760 cellular_grid: 1024,
761 plexus_points: 1_500,
762 plexus_edges: 20_000,
763 max_coc_px: 24,
764 };
765
766 /// The config for `tier`.
767 pub const fn for_tier(tier: Tier) -> Self {
768 match tier {
769 Tier::Floor => Self::FLOOR,
770 Tier::Rich => Self::RICH,
771 }
772 }
773
774 /// These values with their grids drawn at `scale`.
775 pub const fn with_grid_scale(self, scale: GridScale) -> Self {
776 Self {
777 grid_scale: scale,
778 ..self
779 }
780 }
781}
782
783impl Default for TierConfig {
784 fn default() -> Self {
785 Self::FLOOR
786 }
787}
788
789// ---------------------------------------------------------------------------
790// The attractor's sample budget (ADR-0140)
791// ---------------------------------------------------------------------------
792
793/// The render target the attractor's sample density is anchored to: 640x360,
794/// 230 400 pixels.
795///
796/// The attractor's trail grid is surface-sized, so its deposit spreads over
797/// whatever the target holds and a flat count therefore *falls* in density as the
798/// target grows — 0.651 particles per pixel per frame here, 0.072 at 1080p, which
799/// is the whole of the "it just looks like Leviathan upscaled" verdict. This is
800/// the size whose density is already accepted, so it is where
801/// [`attractor_budget`] resolves to exactly the tier's own anchor.
802///
803/// **The numerator is the grid's texels before quantization** —
804/// [`GridScale::texels`] of the target's pixels, which at [`GridScale::FULL`] is
805/// the target's pixel count exactly — so a field drawn at a fraction of the
806/// target is sampled at the density this reference names rather than
807/// over-sampled for texels it does not have (ADR-0245). What it still does not
808/// count is the grid's 128-texel rounding, a bounded difference either way —
809/// 1.07x here and at 720p, 0.95x at 1080p, 1.00x at a capped 4K — named
810/// because the deposit lands per texel.
811pub const REFERENCE_PX: u32 = 230_400;
812
813/// The attractor's drawn sample budget for a target of `target_px` pixels, before
814/// a preset's `[particles] density` narrows it further (ADR-0140).
815///
816/// `clamp(round(anchor * target_px / REFERENCE_PX), anchor, ceiling)`.
817///
818/// **The lower clamp is load-bearing.** The law can only ever *add* samples above
819/// [`REFERENCE_PX`], never remove them below it, so every existing capture — the
820/// 128x128 golden suite, the 96x96 sanity suite, every small `shot` still —
821/// resolves to exactly the count it resolved before this function existed and
822/// stays byte-identical. That is assertable on the value rather than inferred
823/// from pixels, which is the same shape of argument ADR-0065 used for
824/// `deposit_scale` being exactly `1.0` at `Floor`.
825///
826/// `f64` throughout: `anchor * target_px` reaches 41 bits at a 4K target, past
827/// `f32`'s 24-bit mantissa, so the product would be rounded before the divide.
828///
829/// A `ceiling` below `anchor` is raised to it rather than inverting the clamp —
830/// `u32::clamp` panics when `min > max`, and this runs on the resize path.
831pub fn attractor_budget(anchor: u32, target_px: u32, ceiling: u32) -> u32 {
832 let scaled = (f64::from(anchor) * f64::from(target_px) / f64::from(REFERENCE_PX)).round();
833 let scaled = if scaled >= f64::from(u32::MAX) {
834 u32::MAX
835 } else {
836 scaled as u32
837 };
838 scaled.clamp(anchor, ceiling.max(anchor))
839}
840
841impl TierConfig {
842 /// [`attractor_budget`] against this tier's **live** ceiling — what a window
843 /// and the plugin's host surface resolve.
844 pub fn attractor_budget_live(&self, target_px: u32) -> u32 {
845 attractor_budget(
846 self.attractor_particles,
847 target_px,
848 self.attractor_particles_live_ceiling,
849 )
850 }
851
852 /// [`attractor_budget`] against this tier's **offline** ceiling — what a
853 /// headless render resolves, where the bound is memory rather than frame time.
854 pub fn attractor_budget_offline(&self, target_px: u32) -> u32 {
855 attractor_budget(
856 self.attractor_particles,
857 target_px,
858 self.attractor_particles_offline_ceiling,
859 )
860 }
861}
862
863// ---------------------------------------------------------------------------
864// The frame-time governor (Plan 0044 Phase 2)
865// ---------------------------------------------------------------------------
866
867/// The display budget assumed when the frontend has not named a refresh rate —
868/// 60 Hz, the rate NFR §1's floor is quoted at.
869pub const DEFAULT_DISPLAY_HZ: f32 = 60.0;
870
871/// How far past the display budget a single frame must run to count as a miss.
872///
873/// Not 1.0. A frame landing a hair over the budget is the ordinary condition of a
874/// vsynced renderer — the measured interval *is* the refresh interval, plus
875/// scheduling noise — so a bare comparison would read a perfectly healthy 60 fps
876/// run as missing on half its frames. 1.25 is "missing the budget by a quarter",
877/// which at 60 Hz is 20.8 ms: past it the run is visibly not holding the rate.
878pub const MISS_FACTOR: f32 = 1.25;
879
880/// What fraction of the observed frames must be misses before the governor
881/// demotes. Three quarters: high enough that an intermittently-heavy passage
882/// rides through, low enough that a genuine overload does not have to be
883/// unanimous (a demotion is triggered by *sustained* pressure, and a real
884/// overload still has fast frames in it — a cheap preset in the rotation, a
885/// dissolve that ended).
886pub const MISS_FRACTION: f32 = 0.75;
887
888/// Frames of history required before the governor will demote at all.
889///
890/// This is the hysteresis, and it is a *count* rather than a smoothing constant
891/// on purpose: it makes "a single spike must not demote" true by arithmetic
892/// instead of by tuning. With 180 frames required and 75 % of them needing to
893/// miss, no run of fewer than 135 consecutive bad frames can demote — so a window
894/// drag, a driver hiccup, or a shader compile cannot, whatever their magnitude.
895/// At 60 Hz it is also a 3 s warm-up, which keeps the pathological first frames
896/// of a session (pipeline creation, first-use resource builds) out of the verdict.
897///
898/// **This number is only satisfiable because of a constant in another module.**
899/// The only series the renderer ever hands [`sustained_miss`] is
900/// [`FrameStats::samples`](crate::diag::FrameStats::samples), which yields at
901/// most `crate::diag::RING` items — so a `MIN_SAMPLES` above the ring's
902/// capacity makes the governor a permanent no-op. See the assertion below.
903pub const MIN_SAMPLES: usize = 180;
904
905/// The governor must be able to *reach* its own threshold from its real input.
906///
907/// A build failure rather than a test, because the failure it guards is silent:
908/// nothing observable happens when the governor stops demoting — a machine that
909/// cannot hold the rich budget simply stutters for the rest of the session, which
910/// is the exact NFR §1 outcome ADR-0045 built the governor to prevent. A runtime
911/// check could not fire (there is nothing to fire *on*), and a unit test over an
912/// injected series cannot see this at all: the series in the tests below are
913/// `Vec`s of any length we like, so they would keep passing while the real
914/// producer had gone too short to ever trigger a demotion.
915const _: () = assert!(
916 MIN_SAMPLES <= crate::diag::RING,
917 "the frame-time ring is shorter than the governor's minimum sample count, \
918 so the governor can never demote"
919);
920
921/// Whether a frame-time series shows a **sustained** miss of the display budget,
922/// which is the one condition that demotes [`Tier::Rich`] to [`Tier::Floor`]
923/// (ADR-0045).
924///
925/// Pure and total: a function of the series and the budget with no clock, no
926/// state and no allocation, so the policy is unit-testable against injected
927/// series (which is how the spike-versus-overload distinction above is checked
928/// rather than asserted). `frame_secs` is the rolling history in **seconds**, the
929/// unit [`FrameStats::samples`](crate::diag::FrameStats::samples) yields; order
930/// does not matter, only the counts.
931///
932/// Says `false` for a non-positive or non-finite budget, and for a series shorter
933/// than [`MIN_SAMPLES`] — the safe direction, since a wrong `true` costs the user
934/// the rich tier for the rest of the session and a wrong `false` costs nothing but
935/// another second of measurement.
936pub fn sustained_miss(frame_secs: impl Iterator<Item = f32>, budget_secs: f32) -> bool {
937 if !budget_secs.is_finite() || budget_secs <= 0.0 {
938 return false;
939 }
940 let threshold = budget_secs * MISS_FACTOR;
941 let mut total = 0usize;
942 let mut missed = 0usize;
943 for dt in frame_secs {
944 total += 1;
945 if dt.is_finite() && dt > threshold {
946 missed += 1;
947 }
948 }
949 total >= MIN_SAMPLES && missed as f32 >= total as f32 * MISS_FRACTION
950}
951
952/// **The governor's whole decision**: whether to demote `tier` right now.
953///
954/// Pure — every input is a value, so the three properties ADR-0045 asks of the
955/// governor are unit-testable together rather than one being a fact about
956/// `Renderer`'s field layout: a pin never demotes, an already-demoted session
957/// never demotes again (the one-way latch), and only a sustained miss demotes.
958///
959/// The caller owns the latch: it sets its "demoted" flag and rebuilds when this
960/// says `true`, and passes that flag back in on every later frame. Keeping the
961/// flag out here is what makes "exactly once" a property of the decision instead
962/// of a property of the call site.
963pub fn should_demote(
964 tier: Tier,
965 pinned: bool,
966 already_demoted: bool,
967 frame_secs: impl Iterator<Item = f32>,
968 budget_secs: f32,
969) -> bool {
970 // Ordered cheapest-first: three flag reads settle the steady state, and the
971 // series is only walked on a governed rich session that has not yet demoted.
972 if pinned || already_demoted || tier == Tier::Floor {
973 return false;
974 }
975 sustained_miss(frame_secs, budget_secs)
976}
977
978/// **Whether a runtime tier change is allowed at all** (ADR-0054): only on a
979/// context that has a surface.
980///
981/// A surface-less context is exactly the headless capture path, and ADR-0045's
982/// guarantee is that a capture is `Tier::Floor` **by construction** —
983/// `Renderer::new_headless` takes no tier argument, so no baseline can be blessed
984/// at another tier by forgetting a field. `Renderer::set_tier` is a public
985/// mutator on the very type the golden suite renders through, so it is the one
986/// hole that guarantee was shaped to exclude, and this predicate is what keeps it
987/// closed.
988///
989/// Pure, and separate from `set_tier`, deliberately. A `Renderer` **with** a
990/// surface cannot be constructed in CI — there is no window — so a test that only
991/// observed the headless no-op would pass equally well against a `set_tier` that
992/// did nothing at all. Expressed as a value-in/value-out function, both
993/// directions are assertable.
994pub fn tier_change_permitted(has_surface: bool) -> bool {
995 has_surface
996}
997
998/// The frame budget for a display running at `hz`, in seconds. Falls back to
999/// [`DEFAULT_DISPLAY_HZ`] for a value that is not a usable rate, so a frontend
1000/// that cannot read its monitor still gets a governed session rather than an
1001/// ungoverned one.
1002pub fn budget_secs(hz: f32) -> f32 {
1003 let hz = if hz.is_finite() && hz > 0.0 {
1004 hz
1005 } else {
1006 DEFAULT_DISPLAY_HZ
1007 };
1008 1.0 / hz
1009}
1010
1011#[cfg(test)]
1012mod tests;