The plumbing for O(1) subtree movement (LAYOUT.md §2), with every slot still at zero, so this changes no pixels and the next commit can change behaviour against a known-good picture. Every active widget owns a slot in `UiData::moves`: a translation in physical pixels and the slot it is relative to. A primitive instance and a mask each name one, and `prelude.wgsl` walks the chain and adds the accumulated delta. A mask resolves its own chain rather than the drawn primitive's, so a stationary viewport can clip content that moves inside it. `CHAIN_LIMIT` is stated on both sides; it bounds a malformed cycle rather than any real tree. A slot outlives any one `ActiveData`, because a redraw replaces that while the widget's children go on pointing at the slot, so it lives in `UiRenderState::moves` keyed by widget and is retired when the widget stops being drawn. `MoveIdx` is its own type rather than another `Id<u32>`: it sits beside `MaskIdx` in an instance and the two must not be swappable. `Vec2` is now `repr(align(8))`, which is WGSL's alignment for a `vec2<f32>`, so a GPU struct holding one is laid out the way its shader reads it without saying so itself -- `GlyphPrimitive` no longer states its own alignment, and `MoveOffset` never has to. Both keep a manual `unsafe impl Pod`, since the trailing padding that alignment introduces is what `derive(Pod)` refuses. `WindowUniform` holds the `Vec2` its shader has always called `dim` rather than two loose floats, which was the last place the two sides described the same bytes differently. Checked: fmt, clippy and 40 tests. `tabs` (with the image replay), `view` and `minimal` render byte-identical to `upstream/main`, and `text` is unchanged.
209 lines
7.4 KiB
Rust
209 lines
7.4 KiB
Rust
//! What one frame of `UiRenderNode::draw` costs on the CPU, against the number
|
|
//! of layers it walks. Recording only: the pass is built and dropped without
|
|
//! being submitted, so this is the loop's cost and not the GPU's.
|
|
//!
|
|
//! cargo test --release --test draw_cost -- --ignored --nocapture
|
|
//!
|
|
//! Wall time is the wrong number to read for anything under a few percent --
|
|
//! it varied by 2x between runs of one unchanged binary where instructions
|
|
//! retired varied by 0.1%. Count those instead:
|
|
//!
|
|
//! perf stat -e instructions:u target/release/.../draw_cost-* --ignored
|
|
//!
|
|
//! That is how `PrimitiveRender` was measured against a match in the renderer:
|
|
//! 6 instructions per list drawn, against the ~5,400 wgpu spends recording
|
|
//! one.
|
|
//!
|
|
//! The instance is leaked deliberately. A Vulkan loader may unload the driver
|
|
//! when the last one drops, which can fault as a thread that used it exits --
|
|
//! and every test runs on a spawned thread.
|
|
|
|
use std::time::Instant;
|
|
|
|
use iris::prelude::*;
|
|
use iris_core::{
|
|
GlyphPrimitive, MaskIdx, MoveIdx, PrimitiveInst, RectPrimitive, TextureHandle,
|
|
TexturePrimitive, UiData, UiRegion, UiRenderNode, UiRenderState,
|
|
};
|
|
use wgpu::{Color as GpuColor, *};
|
|
|
|
const SIZE: u32 = 1024;
|
|
const FRAMES: u32 = 200;
|
|
/// Reported as the best of this many batches, since the mean moves by more
|
|
/// than the thing being measured.
|
|
const BATCHES: u32 = 8;
|
|
|
|
fn gpu() -> Option<(Device, Queue)> {
|
|
// Probed rather than assumed: there may be no Vulkan adapter, and GL is
|
|
// what is left when there is not.
|
|
let all = Instance::new(InstanceDescriptor::new_without_display_handle());
|
|
let instance = match pollster::block_on(all.request_adapter(&RequestAdapterOptions::default()))
|
|
{
|
|
Ok(_) => all,
|
|
Err(_) => Instance::new(InstanceDescriptor {
|
|
backends: Backends::GL,
|
|
..InstanceDescriptor::new_without_display_handle()
|
|
}),
|
|
};
|
|
// Leaked rather than dropped: see the note at the top of the file.
|
|
let instance: &'static Instance = Box::leak(Box::new(instance));
|
|
let adapter =
|
|
pollster::block_on(instance.request_adapter(&RequestAdapterOptions::default())).ok()?;
|
|
println!("adapter: {:?}", adapter.get_info());
|
|
pollster::block_on(adapter.request_device(&DeviceDescriptor::default())).ok()
|
|
}
|
|
|
|
fn config(format: TextureFormat) -> SurfaceConfiguration {
|
|
SurfaceConfiguration {
|
|
usage: TextureUsages::RENDER_ATTACHMENT,
|
|
format,
|
|
color_space: SurfaceColorSpace::Auto,
|
|
width: SIZE,
|
|
height: SIZE,
|
|
present_mode: PresentMode::Fifo,
|
|
desired_maximum_frame_latency: 2,
|
|
alpha_mode: CompositeAlphaMode::Auto,
|
|
view_formats: vec![],
|
|
}
|
|
}
|
|
|
|
/// Every layer draws all three primitives, so the renderer takes a different
|
|
/// path for each list it walks -- which is the case a single-primitive layer
|
|
/// would never exercise. Images are bound per instance, so there are few.
|
|
fn fill(
|
|
ui: &mut UiData,
|
|
render: &mut UiRenderState,
|
|
layers: usize,
|
|
per_layer: usize,
|
|
) -> Vec<TextureHandle> {
|
|
let rect = ui.primitives.kind::<RectPrimitive>();
|
|
let glyph = ui.primitives.kind::<GlyphPrimitive>();
|
|
let texture = ui.primitives.kind::<TexturePrimitive>();
|
|
let id = ui.widgets.add_strong(Rect::new(UiColor::WHITE)).id();
|
|
let handles: Vec<_> = (0..4)
|
|
.map(|_| ui.textures.add(image::RgbaImage::new(4, 4)))
|
|
.collect();
|
|
|
|
let mut layer = 0;
|
|
for _ in 0..layers {
|
|
for _ in 0..per_layer {
|
|
render.layers.write(
|
|
layer,
|
|
PrimitiveInst {
|
|
kind: rect,
|
|
id,
|
|
primitive: RectPrimitive::color(UiColor::WHITE),
|
|
region: UiRegion::FULL,
|
|
mask_idx: MaskIdx::NONE,
|
|
move_idx: MoveIdx::NONE,
|
|
},
|
|
);
|
|
render.layers.write(
|
|
layer,
|
|
PrimitiveInst {
|
|
kind: glyph,
|
|
id,
|
|
primitive: GlyphPrimitive {
|
|
uv_min: vec2(0.0, 0.0),
|
|
uv_max: vec2(1.0, 1.0),
|
|
layer: 0,
|
|
color: UiColor::WHITE,
|
|
flags: 0,
|
|
},
|
|
region: UiRegion::FULL,
|
|
mask_idx: MaskIdx::NONE,
|
|
move_idx: MoveIdx::NONE,
|
|
},
|
|
);
|
|
}
|
|
for h in &handles[..2] {
|
|
render.layers.write(
|
|
layer,
|
|
PrimitiveInst {
|
|
kind: texture,
|
|
id,
|
|
primitive: TexturePrimitive::from(h),
|
|
region: UiRegion::FULL,
|
|
mask_idx: MaskIdx::NONE,
|
|
move_idx: MoveIdx::NONE,
|
|
},
|
|
);
|
|
}
|
|
layer = render.layers.next(layer);
|
|
}
|
|
handles
|
|
}
|
|
|
|
fn frame_cost(device: &Device, queue: &Queue, layers: usize, per_layer: usize) -> f64 {
|
|
let format = TextureFormat::Bgra8Unorm;
|
|
let mut node = UiRenderNode::new(device, &config(format));
|
|
let mut ui = UiData::default();
|
|
let mut render = UiRenderState::new();
|
|
let _handles = fill(&mut ui, &mut render, layers, per_layer);
|
|
node.update(device, queue, &mut ui, &mut render);
|
|
|
|
let target = device.create_texture(&TextureDescriptor {
|
|
label: Some("draw cost"),
|
|
size: Extent3d {
|
|
width: SIZE,
|
|
height: SIZE,
|
|
depth_or_array_layers: 1,
|
|
},
|
|
mip_level_count: 1,
|
|
sample_count: 1,
|
|
dimension: TextureDimension::D2,
|
|
format,
|
|
usage: TextureUsages::RENDER_ATTACHMENT,
|
|
view_formats: &[],
|
|
});
|
|
let view = target.create_view(&TextureViewDescriptor::default());
|
|
|
|
let record = |frames: u32| {
|
|
let start = Instant::now();
|
|
for _ in 0..frames {
|
|
let mut encoder = device.create_command_encoder(&CommandEncoderDescriptor::default());
|
|
{
|
|
let pass = &mut encoder.begin_render_pass(&RenderPassDescriptor {
|
|
color_attachments: &[Some(RenderPassColorAttachment {
|
|
view: &view,
|
|
resolve_target: None,
|
|
ops: Operations {
|
|
load: LoadOp::Clear(GpuColor::BLACK),
|
|
store: StoreOp::Store,
|
|
},
|
|
depth_slice: None,
|
|
})],
|
|
..Default::default()
|
|
});
|
|
node.draw(pass);
|
|
}
|
|
drop(encoder.finish());
|
|
}
|
|
start.elapsed().as_secs_f64() / frames as f64
|
|
};
|
|
record(FRAMES / 4);
|
|
(0..BATCHES)
|
|
.map(|_| record(FRAMES))
|
|
.fold(f64::MAX, f64::min)
|
|
}
|
|
|
|
#[test]
|
|
#[ignore = "measurement, not a check"]
|
|
fn draw_cost_by_layer_count() {
|
|
let Some((device, queue)) = gpu() else {
|
|
panic!("no wgpu device; see the this-machine-graphics notes");
|
|
};
|
|
println!(
|
|
"layers, each 8 rects + 8 glyphs + 2 images: us/frame (us per layer), best of {BATCHES}"
|
|
);
|
|
let base = frame_cost(&device, &queue, 1, 8) * 1e6;
|
|
for layers in [8, 64, 256, 1024] {
|
|
let per_frame = frame_cost(&device, &queue, layers, 8) * 1e6;
|
|
// Net of the empty pass, which is the same in any version of this.
|
|
println!(
|
|
"{layers:>5}: {per_frame:8.1} us ({:.3} us)",
|
|
(per_frame - base).max(0.0) / layers as f64
|
|
);
|
|
}
|
|
}
|