Watch
1
0
Fork
You've already forked smithay
0

Implement optional Tracy GPU profiling

Remove profiler glFlush() and most calls outside GlesFrame

Rename EnteredGpuTracepoint -> GpuSpan

Expose profiling methods on GlesFrame

Add GPU span to blit()

It exports a sync point so there's a flush we can piggyback off.

Gate GPU profiling behind new tracy_gpu_profiling feature flag

Add profiling scope to QueryPool::collect()

Implement deleting timestamp queries

Bump tracy-client

Handle missing timer query extension

Document methods

Add GlesFrame::with_gpu_span() instead of the error-prone manual API

Make ScopedGpuSpan use a shared borrow making it possible to nest them

The enter() in render texture actually had a mistake where early error
returns were possible without exit(), which would cause an assertion
failure.

Co-authored-by: Christian Meissl <meissl.christian@gmail.com>
This commit is contained in:
Ivan Molodetskikh 2026-01-05 15:08:31 +03:00 • committed by Victoria Brekenfeld
commit eda9f80c3f
5 changed files with 433 additions and 1 deletions

View file

@ -73,6 +73,7 @@ pixman = { version = "0.2.1", features = ["drm-fourcc", "sync"], optional = true
aliasable = { version = "0.1.3", optional = true }
atomic_float = "1.1.0"
sha2 = "0.10.9"
tracy-client = { version = "0.18.4", default-features = false, optional = true }
[dev-dependencies]
clap = { version = "4", features = ["derive"] }
@ -105,6 +106,7 @@ renderer_glow = ["renderer_gl", "glow"]
renderer_multi = ["backend_drm", "aliasable"]
renderer_pixman = ["pixman"]
renderer_test = []
tracy_gpu_profiling = ["renderer_gl", "tracy-client"]
use_system_lib = ["wayland_frontend", "wayland-backend/server_system", "wayland-sys", "gbm?/import-wayland"]
use_bindgen = ["drm-ffi/use_bindgen", "gbm/use_bindgen", "input/use_bindgen"]
wayland_frontend = ["wayland-server", "wayland-protocols", "wayland-protocols-wlr", "wayland-protocols-misc", "tempfile"]

View file

@ -66,6 +66,7 @@ fn gl_generate() {
"GL_EXT_texture_format_BGRA8888",
"GL_EXT_unpack_subimage",
"GL_OES_EGL_sync",
"GL_EXT_disjoint_timer_query",
],
)
.write_bindings(gl_generator::StructGenerator, &mut file)

View file

@ -22,6 +22,7 @@ use tracing::{debug, error, info, info_span, instrument, span, span::EnteredSpan
pub mod element;
mod error;
pub mod format;
pub mod profiler;
mod shaders;
mod texture;
mod uniform;
@ -33,6 +34,9 @@ pub use shaders::*;
pub use texture::*;
pub use uniform::*;
use crate::gpu_span_location;
use profiler::SpanLocation;
use self::version::GlVersion;
use super::{
@ -399,6 +403,9 @@ pub struct GlesRenderer {
// debug
span: tracing::Span,
gl_debug_span: Option<*mut tracing::Span>,
// profiling
profiler: profiler::GpuProfiler,
}
/// Handle to the currently rendered frame during [`GlesRenderer::render`](Renderer::render).
@ -418,6 +425,8 @@ pub struct GlesFrame<'frame, 'buffer> {
finished: AtomicBool,
span: EnteredSpan,
gpu_span: Option<profiler::GpuSpan>,
}
impl fmt::Debug for GlesFrame<'_, '_> {
@ -703,6 +712,8 @@ impl GlesRenderer {
drop(_guard);
let profiler = profiler::GpuProfiler::new(&gl, &exts);
let renderer = GlesRenderer {
gl,
egl: context,
@ -730,6 +741,8 @@ impl GlesRenderer {
_not_send: PhantomData,
span,
gl_debug_span,
profiler,
};
renderer.egl.unbind()?;
Ok(renderer)
@ -1766,6 +1779,9 @@ impl Blit for GlesRenderer {
},
}
self.profiler.collect(&self.gl);
let scope = self.profiler.scope(gpu_span_location!("blit"), &self.gl);
match &src_target.0 {
GlesTargetInternal::Image { ref buf, .. } => unsafe {
self.gl.BindFramebuffer(ffi::READ_FRAMEBUFFER, buf.0.fbo)
@ -1793,6 +1809,7 @@ impl Blit for GlesRenderer {
let status = unsafe { self.gl.CheckFramebufferStatus(ffi::FRAMEBUFFER) };
if status != ffi::FRAMEBUFFER_COMPLETE {
drop(scope);
let _ = self.unbind();
return Err(GlesError::FramebufferBindingError);
}
@ -1816,13 +1833,17 @@ impl Blit for GlesRenderer {
);
self.gl.GetError()
};
drop(scope);
if errno == ffi::INVALID_OPERATION {
Err(GlesError::BlitError)
} else {
if let Some(sync_point) = self.export_sync_point() {
// Sync after glFlush in export_sync_point() and right before returning.
self.profiler.sync_gpu(&self.gl);
return Ok(sync_point);
}
self.profiler.sync_gpu(&self.gl);
unsafe {
self.gl.Finish();
@ -1841,6 +1862,8 @@ impl Drop for GlesRenderer {
self.gl.DeleteProgram(self.solid_program.program);
self.gl.DeleteBuffers(self.vbos.len() as i32, self.vbos.as_ptr());
self.profiler.cleanup(Some(&self.gl));
if self.extensions.iter().any(|ext| ext == "GL_KHR_debug") {
self.gl.Disable(ffi::DEBUG_OUTPUT);
self.gl.DebugMessageCallback(None, ptr::null());
@ -1849,6 +1872,8 @@ impl Drop for GlesRenderer {
#[cfg(all(feature = "wayland_frontend", feature = "use_system_lib"))]
let _ = self.egl_reader.take();
let _ = self.egl.unbind();
} else {
self.profiler.cleanup(None);
}
if let Some(gl_debug_ptr) = self.gl_debug_span.take() {
@ -2075,6 +2100,28 @@ impl GlesFrame<'_, '_> {
{
Ok(func(&self.renderer.gl))
}
/// Run custom code in the GL context with GPU profiling.
///
/// Calls [`with_context()`](Self::with_context) inside
/// [`with_gpu_span()`](Self::with_gpu_span).
pub fn with_profiled_context<F, R>(&mut self, location: SpanLocation, func: F) -> Result<R, GlesError>
where
F: FnOnce(&ffi::Gles2) -> R,
{
self.with_gpu_span(location, |frame| frame.with_context(func))
}
/// Run a function in a GPU profiling span.
pub fn with_gpu_span<F, R>(&mut self, location: SpanLocation, func: F) -> R
where
F: FnOnce(&mut Self) -> R,
{
let span = self.renderer.profiler.enter(location, &self.renderer.gl);
let result = func(self);
self.renderer.profiler.exit(&self.renderer.gl, span);
result
}
}
impl RendererSuper for GlesRenderer {
@ -2125,6 +2172,11 @@ impl Renderer for GlesRenderer {
{
target.0.make_current(&self.gl, &self.egl)?;
// Collect last frame's timestamps.
self.profiler.collect(&self.gl);
let gpu_span = self.profiler.enter(gpu_span_location!("render"), &self.gl);
unsafe {
self.gl.Viewport(0, 0, output_size.w, output_size.h);
@ -2174,6 +2226,8 @@ impl Renderer for GlesRenderer {
finished: AtomicBool::new(false),
span,
gpu_span: Some(gpu_span),
})
}
@ -2292,6 +2346,10 @@ impl Frame for GlesFrame<'_, '_> {
return Ok(());
}
let scope = self
.renderer
.profiler
.enter(gpu_span_location!("clear"), &self.renderer.gl);
unsafe {
self.renderer.gl.Disable(ffi::BLEND);
}
@ -2302,6 +2360,8 @@ impl Frame for GlesFrame<'_, '_> {
self.renderer.gl.Enable(ffi::BLEND);
self.renderer.gl.BlendFunc(ffi::ONE, ffi::ONE_MINUS_SRC_ALPHA);
}
self.renderer.profiler.exit(&self.renderer.gl, scope);
res
}
@ -2319,6 +2379,10 @@ impl Frame for GlesFrame<'_, '_> {
let is_opaque = color.is_opaque();
let scope = self
.renderer
.profiler
.enter(gpu_span_location!("draw_solid"), &self.renderer.gl);
if is_opaque {
unsafe {
self.renderer.gl.Disable(ffi::BLEND);
@ -2333,6 +2397,7 @@ impl Frame for GlesFrame<'_, '_> {
self.renderer.gl.BlendFunc(ffi::ONE, ffi::ONE_MINUS_SRC_ALPHA);
}
}
self.renderer.profiler.exit(&self.renderer.gl, scope);
res
}
@ -2386,6 +2451,10 @@ impl GlesFrame<'_, '_> {
return Ok(SyncPoint::signaled());
}
let finish_gpu_span = self
.renderer
.profiler
.enter(gpu_span_location!("finish_internal"), &self.renderer.gl);
unsafe {
self.renderer.gl.Disable(ffi::SCISSOR_TEST);
self.renderer.gl.Disable(ffi::BLEND);
@ -2396,13 +2465,29 @@ impl GlesFrame<'_, '_> {
}
// delayed destruction until the next frame rendering.
self.renderer.cleanup();
{
let scope = self
.renderer
.profiler
.enter(gpu_span_location!("cleanup"), &self.renderer.gl);
self.renderer.cleanup();
self.renderer.profiler.exit(&self.renderer.gl, scope);
}
self.renderer.profiler.exit(&self.renderer.gl, finish_gpu_span);
if let Some(span) = self.gpu_span.take() {
self.renderer.profiler.exit(&self.renderer.gl, span);
}
// if we support egl fences we should use it
if let Some(sync_point) = self.renderer.export_sync_point() {
// Sync after glFlush in export_sync_point() and right before returning.
self.renderer.profiler.sync_gpu(&self.renderer.gl);
return Ok(sync_point);
}
self.renderer.profiler.sync_gpu(&self.renderer.gl);
// as a last option we force finish, this is unlikely to happen
unsafe {
self.renderer.gl.Finish();
@ -2487,6 +2572,10 @@ impl GlesFrame<'_, '_> {
}
let gl = &self.renderer.gl;
let _scope = self
.renderer
.profiler
.scope(gpu_span_location!("draw_solid"), &self.renderer.gl);
unsafe {
gl.UseProgram(self.renderer.solid_program.program);
gl.Uniform4f(
@ -2785,6 +2874,10 @@ impl GlesFrame<'_, '_> {
let sync_lock = tex.0.sync.read().unwrap();
unsafe {
sync_lock.wait_for_upload(gl);
let scope = self
.renderer
.profiler
.scope(gpu_span_location!("render_texture"), gl);
gl.ActiveTexture(ffi::TEXTURE0);
gl.BindTexture(target, tex.0.texture);
gl.TexParameteri(
@ -2852,11 +2945,20 @@ impl GlesFrame<'_, '_> {
);
if self.renderer.capabilities.contains(&Capability::Instancing) {
let _scope = self
.renderer
.profiler
.scope(gpu_span_location!("draw instanced"), gl);
gl.VertexAttribDivisor(program.attrib_vert as u32, 0);
gl.VertexAttribDivisor(program.attrib_vert_position as u32, 1);
gl.DrawArraysInstanced(ffi::TRIANGLE_STRIP, 0, 4, damage_len as i32);
} else {
let _scope = self
.renderer
.profiler
.scope(gpu_span_location!("draw batched"), gl);
// When we have more than 10 rectangles, draw them in batches of 10.
for i in 0..(damage_len - 1) / 10 {
gl.DrawArrays(ffi::TRIANGLES, 0, 60);
@ -2880,6 +2982,7 @@ impl GlesFrame<'_, '_> {
gl.BindTexture(target, 0);
gl.DisableVertexAttribArray(program.attrib_vert as u32);
gl.DisableVertexAttribArray(program.attrib_vert_position as u32);
drop(scope);
if self.renderer.capabilities.contains(&Capability::Fencing) {
sync_lock.update_read(gl);
@ -2970,6 +3073,10 @@ impl GlesFrame<'_, '_> {
// render
let gl = &self.renderer.gl;
unsafe {
let _scope = self
.renderer
.profiler
.scope(gpu_span_location!("render_pixel_shader_to"), gl);
gl.UseProgram(program.program);
gl.UniformMatrix3fv(program.uniform_matrix, 1, ffi::FALSE, matrix.as_ptr());

View file

@ -0,0 +1,320 @@
//! GPU profiling helpers
use super::ffi;
/// GPU profiling span location.
///
/// Create using the [`crate::gpu_span_location!`] macro.
///
/// When `tracy_gpu_profiling` feature is enabled, this wraps a Tracy span location.
/// When disabled, this is a zero-sized no-op type.
#[derive(Clone, Copy)]
#[allow(missing_debug_implementations)] // Tracy SpanLocation doens't impl Debug.
pub struct SpanLocation(#[cfg(feature = "tracy_gpu_profiling")] pub &'static tracy_client::SpanLocation);
/// Creates a GPU span location for profiling.
///
/// When `tracy_gpu_profiling` feature is enabled, this wraps `tracy_client::span_location!`.
/// When disabled, returns a no-op placeholder.
#[cfg(feature = "tracy_gpu_profiling")]
#[macro_export]
macro_rules! gpu_span_location {
() => {
$crate::backend::renderer::gles::profiler::SpanLocation(
$crate::reexports::tracy_client::span_location!(),
)
};
($name:expr) => {
$crate::backend::renderer::gles::profiler::SpanLocation(
$crate::reexports::tracy_client::span_location!($name),
)
};
}
/// Creates a GPU span location for profiling.
///
/// When `tracy_gpu_profiling` feature is enabled, this wraps `tracy_client::span_location!`.
/// When disabled, returns a no-op placeholder.
#[cfg(not(feature = "tracy_gpu_profiling"))]
#[macro_export]
macro_rules! gpu_span_location {
() => {
$crate::backend::renderer::gles::profiler::SpanLocation()
};
($name:expr) => {
$crate::backend::renderer::gles::profiler::SpanLocation()
};
}
/// An active GPU profiling span.
///
/// This type represents a GPU profiling span that has been entered but not yet exited.
/// It must be passed to `exit_gpu_span()` when the profiled code section is complete.
///
/// # Panics
///
/// Dropping this type without calling `exit_gpu_span()` will panic to ensure GPU profiling
/// spans are properly closed.
#[derive(Debug)]
pub struct GpuSpan {
active: bool,
}
impl Drop for GpuSpan {
fn drop(&mut self) {
assert!(!self.active, "GPU span must be properly exited");
}
}
pub(crate) struct ScopedGpuSpan<'a, 'b> {
span: Option<GpuSpan>,
profiler: &'a GpuProfiler,
gl: &'b ffi::Gles2,
}
impl<'a, 'b> Drop for ScopedGpuSpan<'a, 'b> {
fn drop(&mut self) {
let span = self.span.take().unwrap();
self.profiler.exit(self.gl, span);
}
}
#[cfg(feature = "tracy_gpu_profiling")]
mod imp {
use std::cell::Cell;
use super::{ffi, GpuSpan, SpanLocation};
// Number of timestamp queries in the pool. Limited by Tracy's use of u16 for query IDs.
const MAX_QUERIES: usize = u16::MAX as usize;
pub struct GpuProfiler {
// `None` means the required GL extension is not supported.
pool: Option<QueryPool>,
}
struct QueryPool {
context: tracy_client::GpuContext,
pool: Vec<GpuQuery>,
// Index of first free and first pending queries in the pool.
head_tail: Cell<(usize, usize)>,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[repr(transparent)]
struct GpuQuery(ffi::types::GLuint);
impl QueryPool {
fn new(context: tracy_client::GpuContext, gl: &ffi::Gles2) -> Self {
let mut pool = vec![GpuQuery(0); MAX_QUERIES];
unsafe {
gl.GenQueriesEXT(pool.len() as ffi::types::GLsizei, pool.as_mut_ptr().cast());
}
Self {
context,
pool,
head_tail: Cell::new((0, 0)),
}
}
fn next(&self, gl: &ffi::Gles2) -> GpuQuery {
let (head, tail) = self.head_tail.get();
let query = self.pool[head];
let new_head = (head + 1) % MAX_QUERIES;
assert_ne!(new_head, tail, "ran out of queries");
self.head_tail.set((new_head, tail));
unsafe { gl.QueryCounterEXT(query.0, ffi::TIMESTAMP_EXT) };
query
}
fn collect(&mut self, gl: &ffi::Gles2) {
let (head, tail) = self.head_tail.get_mut();
if tail == head {
return;
}
profiling::scope!("QueryPool::collect");
while tail != head {
let query = self.pool[*tail];
let mut available = 0;
unsafe {
while gl.GetError() != ffi::NO_ERROR {}
gl.GetQueryObjectuivEXT(query.0, ffi::QUERY_RESULT_AVAILABLE, &mut available);
if gl.GetError() != ffi::NO_ERROR {
// Don't really have a good way out of this.
available = 1;
}
}
if available == 0 {
return;
}
let mut timestamp = 0;
unsafe { gl.GetQueryObjecti64vEXT(query.0, ffi::QUERY_RESULT, &mut timestamp) };
self.context.upload_gpu_timestamp(query.0 as u16, timestamp);
*tail = (*tail + 1) % MAX_QUERIES;
}
}
/// Clean up the timestamp query pool by deleting them.
///
/// If `gl` is `None` then all queries are leaked. Pass `None` only when failing to obtain
/// the GL context during drop.
fn cleanup(&mut self, gl: Option<&ffi::Gles2>) {
if self.pool.is_empty() {
return;
}
if let Some(gl) = gl {
unsafe {
gl.DeleteQueriesEXT(self.pool.len() as ffi::types::GLsizei, self.pool.as_ptr().cast());
}
}
self.pool.clear();
}
}
impl Drop for QueryPool {
fn drop(&mut self) {
assert!(self.pool.is_empty(), "QueryPool must be cleaned up before drop");
}
}
impl GpuProfiler {
pub fn new(gl: &ffi::Gles2, extensions: &[String]) -> Self {
let has_timer_query = extensions.iter().any(|ext| ext == "GL_EXT_disjoint_timer_query");
if !has_timer_query {
return Self { pool: None };
}
let client = tracy_client::Client::start();
let mut gpu_timestamp = 0;
unsafe { gl.GetInteger64v(ffi::TIMESTAMP_EXT, &mut gpu_timestamp) };
let context = client
.new_gpu_context(
Some("GlesRenderer"),
tracy_client::GpuContextType::OpenGL,
gpu_timestamp,
1.0,
)
.unwrap();
let pool = QueryPool::new(context, gl);
Self { pool: Some(pool) }
}
pub fn enter(&self, span_location: SpanLocation, gl: &ffi::Gles2) -> GpuSpan {
let Some(pool) = &self.pool else {
return GpuSpan { active: false };
};
if !tracy_client::Client::is_connected() {
return GpuSpan { active: false };
}
let query = pool.next(gl);
pool.context.begin_span(span_location.0, query.0 as u16);
GpuSpan { active: true }
}
pub fn exit(&self, gl: &ffi::Gles2, mut entered: GpuSpan) {
if !entered.active {
return;
}
entered.active = false;
let Some(pool) = &self.pool else {
return;
};
let query = pool.next(gl);
pool.context.end_span(query.0 as u16);
}
/// Collect completed timestamp queries and send them to the profiler.
///
/// Must be called regularly to avoid filling up the query pool. A good place is right
/// before a batch of rendering operations.
pub fn collect(&mut self, gl: &ffi::Gles2) {
if let Some(pool) = &mut self.pool {
pool.collect(gl);
}
}
/// Sync the GPU and CPU times by uploading the current GPU timestamp to the profiler.
///
/// Necessary to avoid GPU timestamp drift. A good place to call this is right after
/// flushing a batch of rendering operations.
pub fn sync_gpu(&self, gl: &ffi::Gles2) {
let Some(pool) = &self.pool else {
return;
};
let mut gpu_timestamp = 0;
unsafe { gl.GetInteger64v(ffi::TIMESTAMP_EXT, &mut gpu_timestamp) };
pool.context.sync_gpu_time(gpu_timestamp);
}
/// Clean up the timestamp query pool by deleting them.
///
/// If `gl` is `None` then all queries are leaked. Pass `None` only when failing to obtain
/// the GL context during drop.
pub fn cleanup(&mut self, gl: Option<&ffi::Gles2>) {
if let Some(pool) = &mut self.pool {
pool.cleanup(gl);
}
}
}
}
#[cfg(not(feature = "tracy_gpu_profiling"))]
mod imp {
use super::{ffi, GpuSpan, SpanLocation};
pub struct GpuProfiler(());
impl GpuProfiler {
pub fn new(_gl: &ffi::Gles2, _extensions: &[String]) -> Self {
Self(())
}
pub fn enter(&self, _span_location: SpanLocation, _gl: &ffi::Gles2) -> GpuSpan {
GpuSpan { active: true }
}
pub fn exit(&self, _gl: &ffi::Gles2, mut entered: GpuSpan) {
entered.active = false;
}
pub fn collect(&mut self, _gl: &ffi::Gles2) {}
pub fn sync_gpu(&self, _gl: &ffi::Gles2) {}
pub fn cleanup(&mut self, _gl: Option<&ffi::Gles2>) {}
}
}
pub(crate) use imp::*;
impl GpuProfiler {
pub fn scope<'a, 'b>(&'a self, span_location: SpanLocation, gl: &'b ffi::Gles2) -> ScopedGpuSpan<'a, 'b> {
let span = self.enter(span_location, gl);
ScopedGpuSpan {
span: Some(span),
gl,
profiler: self,
}
}
}

View file

@ -14,6 +14,8 @@ pub use input;
#[cfg(feature = "renderer_pixman")]
pub use pixman;
pub use rustix;
#[cfg(feature = "tracy_gpu_profiling")]
pub use tracy_client;
#[cfg(feature = "backend_udev")]
pub use udev;
#[cfg(feature = "wayland_frontend")]