From d8c497fcd7dfc2f10e00e2763efde6f0f97ef5e8 Mon Sep 17 00:00:00 2001 From: iris-ai <4+iris-ai@noreply.localhost> Date: Mon, 14 Sep 2026 03:31:33 -0400 Subject: [PATCH] Measure what the chain walk costs, and find that depth is the cost MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `tests/chain_cost.rs` times the pass on the GPU with timestamp queries, which this adapter supports, rather than by the clock. 200,000 two-pixel instances at 1024x1024, so vertex work dominates, best of eight batches: depth 1 77.9 us +0.0% depth 2 78.1 us +0.2% depth 4 79.0 us +1.3% depth 8 81.8 us +5.0% depth 16 111.1 us +42.6% depth 32 159.6 us +104.9% depth 64 250.3 us +221.3% Free to about depth 8 and then roughly 3 us per level. Each step is a storage load whose address is the previous load's result, so it is the chaining that costs rather than the arithmetic at each level -- which means the number would look the same for a slot carrying a whole region instead of a delta. That matters because every active widget owns a slot, so a primitive resolves through its full depth in the widget tree, and LAYOUT.md notes real trees have exceeded 16. At the couple of hundred primitives an example draws it is nothing; a transcript's glyphs are tens of thousands of primitives, which is the regime measured here. No behaviour change. Recorded rather than acted on: keeping the chain shallow means not giving every widget a slot, which is a design decision of LAYOUT.md §2 and §6 and the owner's to make. --- tests/chain_cost.rs | 224 ++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 224 insertions(+) create mode 100644 tests/chain_cost.rs diff --git a/tests/chain_cost.rs b/tests/chain_cost.rs new file mode 100644 index 0000000..6d6468d --- /dev/null +++ b/tests/chain_cost.rs @@ -0,0 +1,224 @@ +//! What the vertex shader's move-chain walk costs, against how deep the chain +//! is. Every active widget owns a slot, so the depth a primitive resolves +//! through is its depth in the widget tree. +//! +//! cargo test --release --test chain_cost -- --ignored --nocapture +//! +//! Timed on the GPU with timestamp queries rather than by the clock: wall time +//! here varied by 2x between runs of one unchanged binary. The pass is +//! submitted and waited on, so this is the GPU's cost and not the recording +//! loop's -- which is what `draw_cost.rs` measures instead. +//! +//! The instances are two pixels wide so that vertex work dominates; a chain +//! walk that does not show up against small quads will not show up against +//! anything. +//! +//! The instance is leaked deliberately, for the reason `draw_cost.rs` gives. + +use iris::prelude::*; +use iris_core::{ + MaskIdx, MoveIdx, PrimitiveInst, RectPrimitive, UiData, UiRegion, UiRenderNode, UiRenderState, + UiScalar, UiSpan, +}; +use wgpu::{Color as GpuColor, *}; + +const SIZE: u32 = 1024; +const INSTANCES: usize = 200_000; +const FRAMES: u32 = 20; +/// Reported as the best of this many batches, since the mean moves by more +/// than the thing being measured. +const BATCHES: u32 = 8; + +fn gpu() -> Option<(Device, Queue, f32)> { + let all = Instance::new(InstanceDescriptor::new_without_display_handle()); + let instance = match pollster::block_on(all.request_adapter(&RequestAdapterOptions::default())) + { + Ok(_) => all, + Err(_) => Instance::new(InstanceDescriptor { + backends: Backends::GL, + ..InstanceDescriptor::new_without_display_handle() + }), + }; + let instance: &'static Instance = Box::leak(Box::new(instance)); + let adapter = + pollster::block_on(instance.request_adapter(&RequestAdapterOptions::default())).ok()?; + if !adapter.features().contains(Features::TIMESTAMP_QUERY) { + println!("no timestamp queries on {:?}", adapter.get_info().name); + return None; + } + println!("adapter: {:?}", adapter.get_info().name); + let (device, queue) = pollster::block_on(adapter.request_device(&DeviceDescriptor { + required_features: Features::TIMESTAMP_QUERY, + ..Default::default() + })) + .ok()?; + let period = queue.get_timestamp_period(); + Some((device, queue, period)) +} + +fn config(format: TextureFormat) -> SurfaceConfiguration { + SurfaceConfiguration { + usage: TextureUsages::RENDER_ATTACHMENT, + format, + color_space: SurfaceColorSpace::Auto, + width: SIZE, + height: SIZE, + present_mode: PresentMode::Fifo, + desired_maximum_frame_latency: 2, + alpha_mode: CompositeAlphaMode::Auto, + view_formats: vec![], + } +} + +/// A chain `depth` slots long, and instances that all resolve through its end. +fn fill(ui: &mut UiData, render: &mut UiRenderState, depth: usize) { + let kind = ui.primitives.kind::(); + let id = ui.widgets.add_strong(Rect::new(UiColor::WHITE)).id(); + + let mut slot = MoveIdx::NONE; + for _ in 0..depth { + slot = render.moves.push(slot); + } + + let px = |v: f32| UiScalar { rel: 0.0, abs: v }; + for i in 0..INSTANCES { + let x = (i % (SIZE as usize / 2)) as f32 * 2.0; + let y = (i / (SIZE as usize / 2)) as f32; + render.layers.write( + 0, + PrimitiveInst { + kind, + id, + primitive: RectPrimitive::color(UiColor::WHITE), + region: UiRegion::new( + UiSpan::new(px(x), px(x + 2.0)), + UiSpan::new(px(y), px(y + 1.0)), + ), + mask_idx: MaskIdx::NONE, + move_idx: slot, + }, + ); + } +} + +/// Nanoseconds the pass took on the GPU, best of `BATCHES`. +fn pass_cost(device: &Device, queue: &Queue, period: f32, depth: usize) -> f64 { + let format = TextureFormat::Bgra8Unorm; + let mut node = UiRenderNode::new(device, &config(format)); + let mut ui = UiData::default(); + let mut render = UiRenderState::new(); + fill(&mut ui, &mut render, depth); + node.update(device, queue, &mut ui, &mut render); + + let target = device.create_texture(&TextureDescriptor { + label: Some("chain cost"), + size: Extent3d { + width: SIZE, + height: SIZE, + depth_or_array_layers: 1, + }, + mip_level_count: 1, + sample_count: 1, + dimension: TextureDimension::D2, + format, + usage: TextureUsages::RENDER_ATTACHMENT, + view_formats: &[], + }); + let view = target.create_view(&TextureViewDescriptor::default()); + + let queries = device.create_query_set(&QuerySetDescriptor { + label: Some("chain cost"), + ty: QueryType::Timestamp, + count: 2, + }); + let resolved = device.create_buffer(&BufferDescriptor { + label: Some("resolved"), + size: 16, + usage: BufferUsages::QUERY_RESOLVE | BufferUsages::COPY_SRC, + mapped_at_creation: false, + }); + let readback = device.create_buffer(&BufferDescriptor { + label: Some("readback"), + size: 16, + usage: BufferUsages::MAP_READ | BufferUsages::COPY_DST, + mapped_at_creation: false, + }); + + let frame = || { + let mut encoder = device.create_command_encoder(&CommandEncoderDescriptor::default()); + { + let pass = &mut encoder.begin_render_pass(&RenderPassDescriptor { + label: None, + color_attachments: &[Some(RenderPassColorAttachment { + view: &view, + resolve_target: None, + ops: Operations { + load: LoadOp::Clear(GpuColor::BLACK), + store: StoreOp::Store, + }, + depth_slice: None, + })], + depth_stencil_attachment: None, + timestamp_writes: Some(RenderPassTimestampWrites { + query_set: &queries, + beginning_of_pass_write_index: Some(0), + end_of_pass_write_index: Some(1), + }), + occlusion_query_set: None, + multiview_mask: None, + }); + node.draw(pass); + } + encoder.resolve_query_set(&queries, 0..2, &resolved, 0); + encoder.copy_buffer_to_buffer(&resolved, 0, &readback, 0, 16); + queue.submit(Some(encoder.finish())); + + let slice = readback.slice(..); + slice.map_async(MapMode::Read, |_| {}); + let _ = device.poll(PollType::Wait { + submission_index: None, + timeout: None, + }); + let ns = { + let view = slice.get_mapped_range().expect("timestamps did not map"); + let stamps: [u64; 2] = [ + u64::from_le_bytes(view[..8].try_into().unwrap()), + u64::from_le_bytes(view[8..16].try_into().unwrap()), + ]; + (stamps[1].saturating_sub(stamps[0])) as f64 * period as f64 + }; + readback.unmap(); + ns + }; + + frame(); + let mut best = f64::MAX; + for _ in 0..BATCHES { + let mut total = 0.0; + for _ in 0..FRAMES { + total += frame(); + } + best = best.min(total / FRAMES as f64); + } + best +} + +#[test] +#[ignore = "measurement, not a check"] +fn chain_cost_by_depth() { + let Some((device, queue, period)) = gpu() else { + println!("no gpu with timestamps; nothing measured"); + return; + }; + println!("{INSTANCES} instances, {SIZE}x{SIZE}, best of {BATCHES} batches"); + let mut base = None; + for depth in [1, 2, 4, 8, 16, 32, 64] { + let ns = pass_cost(&device, &queue, period, depth); + let base = *base.get_or_insert(ns); + println!( + "depth {depth:>3}: {:>9.1} us {:+6.1}% against depth 1", + ns / 1000.0, + (ns - base) / base * 100.0 + ); + } +}