From 18c2b2dec6ce044ac2e3d52d6a9b4840f1a62ccf Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Fri, 2 Oct 2026 09:53:44 +0200 Subject: [PATCH 1/4] perf(preview): hand composed frames to the canvas as shared GPU textures The preview was bound by the trip of its pixels to the canvas, not by the compositor. Each frame was read back from the GPU, copied into a Vec, cloned across IPC and uploaded to the canvas again: at 1080p the round trip took ~24 ms, so the pull loop drew ~20 of the ~57 frames composed per second, and the transport alone kept ~57 % of the renderer's main thread busy. On Windows' hardware backend the render thread now copies each composed frame into one of four shared D3D11 textures (NT handles) and publishes its slot. The main process imports it with Electron 41's sharedTexture API and sends it to the frame that asked; the renderer draws the VideoFrame on the canvas. Measured end to end with a real 1080p60 recording: 53 frames drawn per second, renderer main thread ~2 % busy, main process under 1 %, and the pixels byte-identical to read-back. Read-back stays the path everywhere else (macOS, Linux, software backend, no GPU compositing, OPENSCREEN_PREVIEW_READBACK=1), and a view falls back to it when an import or a send fails, or when its first shared frame lands as nothing (a texture Chromium could not open). --- crates/Cargo.toml | 1 + crates/compositor-view-napi/src/lib.rs | 73 +++- crates/compositor/src/compositor_windows.rs | 6 + crates/compositor/src/lib.rs | 2 + crates/compositor/src/live.rs | 201 +++++++++-- crates/compositor/src/shared_frames.rs | 329 ++++++++++++++++++ .../compositor/tests/shared_frame_handoff.rs | 135 +++++++ electron/electron-env.d.ts | 9 + electron/ipc/nativeBridge.ts | 19 +- .../services/compositorViewService.test.ts | 161 +++++++++ .../services/compositorViewService.ts | 131 ++++++- electron/native/compositor-view/addon.d.ts | 25 ++ electron/preload.ts | 28 +- src/native/compositorViewClient.ts | 29 +- src/native/contracts.ts | 26 ++ .../hooks/useNativeCompositorView.test.ts | 133 ++++++- src/native/hooks/useNativeCompositorView.ts | 112 ++++-- .../architecture/preview.md | 42 ++- .../engineering/rendering-performance.md | 46 +++ 19 files changed, 1417 insertions(+), 91 deletions(-) create mode 100644 crates/compositor/src/shared_frames.rs create mode 100644 crates/compositor/tests/shared_frame_handoff.rs diff --git a/crates/Cargo.toml b/crates/Cargo.toml index 3caa7edfd..a65f9aaef 100644 --- a/crates/Cargo.toml +++ b/crates/Cargo.toml @@ -68,6 +68,7 @@ features = [ "Win32_Graphics_Direct2D_Common", # D2D1_COLOR_F / PIXEL_FORMAT "Win32_Graphics_DirectWrite", # mise en page + rastérisation du texte "Win32_Graphics_Gdi", # brosses/police GUI de la preview (poc-d3d/src/app.rs) + "Win32_Security", # IDXGIResource1::CreateSharedHandle (shared_frames.rs) "Win32_System_Performance", # QueryPerformanceCounter / Frequency (§10) "Win32_System_LibraryLoader", # GetModuleHandleW (hInstance) "Win32_UI_WindowsAndMessaging", # fenêtres (harnais live + GUI du POC) diff --git a/crates/compositor-view-napi/src/lib.rs b/crates/compositor-view-napi/src/lib.rs index 185f48d43..338beaa80 100644 --- a/crates/compositor-view-napi/src/lib.rs +++ b/crates/compositor-view-napi/src/lib.rs @@ -11,6 +11,7 @@ use napi::{Env, JsFunction, Task}; use napi_derive::napi; use openscreen_compositor::compositor::{live_params_from_scene, Compositor}; use openscreen_compositor::d3d::{Backend, Gpu}; +use openscreen_compositor::frame_geometry::FootageQuad; use openscreen_compositor::gif_export::{GifExportParams, GifStats}; use openscreen_compositor::gif_export_control::{GifExportCancelled, GifExportControl}; use openscreen_compositor::live::{LiveView, PausedPreviews}; @@ -249,12 +250,82 @@ pub fn read_frame(id: i32, since_gen: f64) -> Result> { width: w, height: h, data: Buffer::from(pixels), - footage: footage.map(|q| q.corners.iter().flatten().map(|&v| v as f64).collect()), + footage: footage_corners(footage), footage_projective: footage.is_some_and(|q| q.projective), } })) } +/// Les coins du métrage (TL, TR, BR, BL) aplatis en huit nombres, tels que JS les lit. +fn footage_corners(footage: Option) -> Option> { + footage.map(|q| q.corners.iter().flatten().map(|&v| v as f64).collect()) +} + +/// Une frame de preview posée dans une texture partagée : de quoi l'importer côté GPU +/// (`sharedTexture.importSharedTexture`) au lieu d'en recevoir les pixels. +#[napi(object)] +pub struct SharedFramePacket { + pub gen: f64, + /// Case de l'anneau qui porte la frame, à rendre avec `gen` par `release_shared_frame` + /// quand Chromium l'a relâchée. + pub slot: u32, + /// Handle NT de la texture, 8 octets little-endian : la forme qu'attend Electron + /// (`SharedTextureHandle.ntHandle`). Valable dans ce processus seulement. + pub handle: Buffer, + pub width: u32, + pub height: u32, + pub footage: Option>, + pub footage_projective: bool, +} + +/// Livrer les frames de la vue par textures partagées plutôt que par `read_frame`. Rend +/// `false` quand la machine ne le peut pas (hors Windows, backend logiciel) : la vue reste +/// alors au readback. +#[napi] +pub fn set_shared_frames(id: i32, enabled: bool) -> bool { + registry() + .lock() + .unwrap() + .get(&id) + .is_some_and(|v| v.set_shared_frames(enabled)) +} + +/// Le pendant de `read_frame` pour une vue en textures partagées : la dernière frame posée +/// dans l'anneau, si elle est plus récente que `since_gen`. Sa case reste tenue jusqu'à +/// `release_shared_frame`. `Ok(None)` aussi quand la vue relit en RAM. +#[napi] +pub fn read_shared_frame(id: i32, since_gen: f64) -> Result> { + let frame = match registry().lock().unwrap().get(&id) { + None => return Ok(None), + Some(v) => { + // Même relais que `read_frame` : sans lui, un thread de rendu mort ne se verrait + // que par un canvas noir. + if let Some(fatal) = v.fatal_error() { + return Err(Error::from_reason(fatal)); + } + v.take_shared_frame(since_gen.max(0.0) as u64) + } + }; + Ok(frame.map(|f| SharedFramePacket { + gen: f.gen as f64, + slot: f.slot, + handle: f.handle.to_le_bytes().to_vec().into(), + width: f.width, + height: f.height, + footage: footage_corners(f.footage), + footage_projective: f.footage.is_some_and(|q| q.projective), + })) +} + +/// Chromium a relâché la frame `gen` de la case `slot` (`allReferencesReleased` côté +/// Electron) : le thread de rendu peut y réécrire. Sans effet sur une vue détruite. +#[napi] +pub fn release_shared_frame(id: i32, slot: u32, gen: f64) { + if let Some(v) = registry().lock().unwrap().get(&id) { + v.release_shared_frame(slot, gen.max(0.0) as u64); + } +} + /// Param live (inspector). Le type de valeur route vers le bon setter : /// bool = switch (webcamMirror…), number = slider (shadow/roundness/motionBlur/backgroundBlur), /// string = sélection (backgroundColor "#rrggbb"). diff --git a/crates/compositor/src/compositor_windows.rs b/crates/compositor/src/compositor_windows.rs index cfe335143..1875d4a34 100644 --- a/crates/compositor/src/compositor_windows.rs +++ b/crates/compositor/src/compositor_windows.rs @@ -3008,6 +3008,12 @@ impl Compositor { Ok((rw, rh, out)) } + /// Le RT de la dernière composition, à `render_size()`, pour qui le copie ailleurs sans + /// passer par la RAM : l'anneau de textures partagées de la preview (`shared_frames`). + pub fn render_target(&self) -> &ID3D11Texture2D { + &self.rt + } + /// Comme `rgb_to_nv12`, mais vers `target_w`×`target_h` : si la cible diffère de la taille /// de rendu, le RT composé est d'abord redimensionné (bilinéaire, `ps_tex`/`sampler` déjà /// utilisés partout ailleurs dans le fichier, cf. `blit_resized`). Le RT suivant la sortie, diff --git a/crates/compositor/src/lib.rs b/crates/compositor/src/lib.rs index 8e5f63e17..767abac8a 100644 --- a/crates/compositor/src/lib.rs +++ b/crates/compositor/src/lib.rs @@ -51,6 +51,8 @@ pub mod sculpt; // `segmentation` ses deux entrées échouent proprement, ce qui garde le reste du crate // indépendant du choix de packaging d'ONNX Runtime. pub mod segmentation; +// Livraison de la preview par textures partagées (Windows) ; la tenue des cases est portable. +pub mod shared_frames; pub mod text_anim; pub mod text_fonts; pub mod text_plate; diff --git a/crates/compositor/src/live.rs b/crates/compositor/src/live.rs index af586f262..cbed49464 100644 --- a/crates/compositor/src/live.rs +++ b/crates/compositor/src/live.rs @@ -27,9 +27,12 @@ use crate::regions::{speed_at, ProgrammeClock}; use crate::scene::Scene; use crate::config::{self, Cfg}; use crate::cursor::CursorTrack; -use crate::d3d::Gpu; +use crate::d3d::{Backend, Gpu}; use crate::frame_geometry::webcam_is_real; use crate::pipeline::Decoder; +#[cfg(windows)] +use crate::shared_frames::SharedRing; +use crate::shared_frames::{SharedFrame, SlotBook}; use crate::timeline_walk::{frame_step, FrameStep, NextFrameTime}; use anyhow::Result; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; @@ -800,6 +803,15 @@ struct Shared { /// vide la ferait repartir à 1 — donc rejouer des générations déjà peintes. Monotone, /// jamais remise à zéro. frame_gen: AtomicU64, + /// La vue livre ses frames dans l'anneau de textures partagées (`take_shared_frame`) au + /// lieu de les relire en RAM (`latest_frame_since`). Posé par JS ; le thread de rendu le + /// rabat à `false` si l'anneau ne peut pas servir, et les frames repassent par la RAM. + shared_frames: AtomicBool, + /// Tenue des cases de l'anneau, entre le thread de rendu et le thread Node. + slot_book: Mutex, + /// Republier la frame courante au prochain tour, même en pause : posé quand le transport + /// change, pour que le consommateur ait une image sans attendre que quelque chose bouge. + republish: AtomicBool, /// Erreur fatale du thread de rendu (device D3D11 introuvable, décodeur qui refuse /// le fichier…). Le thread meurt sur la première erreur ; sans ce champ, elle /// finissait dans un `eprintln!` que personne ne lit et l'utilisateur n'avait @@ -853,6 +865,9 @@ impl LiveView { stop: AtomicBool::new(false), latest_frame: Mutex::new(None), frame_gen: AtomicU64::new(0), + shared_frames: AtomicBool::new(false), + slot_book: Mutex::new(SlotBook::default()), + republish: AtomicBool::new(false), fatal: Mutex::new(None), }); let sh = shared.clone(); @@ -929,6 +944,35 @@ impl LiveView { } } + /// Livrer les frames par textures partagées plutôt que par readback. Seul le backend + /// matériel de Windows sait le faire : ailleurs la demande est refusée et la vue continue + /// de relire en RAM. Rend l'état obtenu. + pub fn set_shared_frames(&self, enabled: bool) -> bool { + let on = enabled && cfg!(windows) && Gpu::probe() == Some(Backend::Hardware); + if self.shared.shared_frames.swap(on, Ordering::Relaxed) != on { + if !on { + if let Ok(mut book) = self.shared.slot_book.lock() { + book.clear_ready(); + } + } + self.shared.republish.store(true, Ordering::Relaxed); + } + on + } + + /// La dernière frame posée dans l'anneau, si elle est plus récente que `since_gen`. Sa + /// case reste tenue jusqu'à `release_shared_frame`. `None` aussi quand la vue relit en RAM. + pub fn take_shared_frame(&self, since_gen: u64) -> Option { + self.shared.slot_book.lock().ok()?.take(since_gen, Instant::now()) + } + + /// Chromium a relâché la frame `gen` de la case `slot` : le thread de rendu peut y réécrire. + pub fn release_shared_frame(&self, slot: u32, gen: u64) { + if let Ok(mut book) = self.shared.slot_book.lock() { + book.release(slot, gen); + } + } + /// Switch inspector (booléen). pub fn set_param_bool(&self, key: &str, value: bool) { if let Ok(mut p) = self.shared.inspector.lock() { @@ -1426,8 +1470,15 @@ unsafe fn render_thread( // appliquée, on refuse de jouer le layout fixture (POC) : un fallback fixture ne ferait que // MASQUER un scene-push cassé. On attend la scène avant de produire le 1er frame. let mut scene_applied = false; + // Textures partagées de la preview, créées au premier besoin (voir `publish_shared`). + let mut ring = Ring::default(); while !shared.stop.load(Ordering::SeqCst) { + // Le transport vient de changer : le consommateur doit recevoir la frame courante + // sans attendre qu'une autre soit composée (en pause, il n'y en aurait pas). + if shared.republish.swap(false, Ordering::Relaxed) { + first = true; + } // params inspector : booléens/taps → cfg ; valeurs continues → live_params let ip = *shared.inspector.lock().unwrap(); let mut clip_changed = false; @@ -1785,37 +1836,43 @@ unsafe fn render_thread( if stepped || first { if pw > 0 && ph > 0 { - // Step complet : `compose_frame` (déjà appelé par `step`/`present_frame`/ - // `recompose`) a rastérisé le RT à la géométrie de sortie ramenée au panneau. - // On lit ce RT DIRECTEMENT à sa résolution de rendu (`readback_direct` : copy - // rt → staging → Map/Unmap), sans le resize `blit_resized` qui, depuis la - // refonte ratio, n'était plus qu'une copie identité + une alloc NV12 inutile. - match comp.readback_direct() { - Ok((rw, rh, rgba)) => { - // Publie dans `latest_frame` : on remplace le buffer précédent - // (le canvas ne montre que la dernière frame, peu importe combien - // le renderer en a raté entre deux lectures napi). On incrémente - // la génération sous le MÊME lock que l'écriture du buffer, pour - // qu'un lecteur ne puisse jamais voir un `gen` neuf appairé à un - // buffer périmé (ou l'inverse). `+ 1` depuis la précédente, `1` au - // premier publish. Les dims publiées sont celles du RENDU (`rw`×`rh`) : - // le canvas JS s'y dimensionne (packet auto-descriptif) puis CSS met à - // l'échelle vers la boîte du panneau — plus de resize GPU intermédiaire. - // La génération vient d'un compteur atomique et non du slot : la - // livraison sans copie VIDE le slot en le lisant, et un - // `unwrap_or(1)` repartirait alors de 1 — le consommateur recevrait - // des générations déjà peintes et boucherait. Séquence identique à - // l'ancienne dérivation tant que le slot n'est pas vidé. - let next_gen = shared.frame_gen.fetch_add(1, Ordering::Relaxed) + 1; - if let Ok(mut slot) = shared.latest_frame.lock() { - *slot = Some((next_gen, rw, rh, rgba, comp.footage_quad())); + match publish_shared(&shared, &gpu, &comp, &mut ring) { + SharedPublish::Published => first = false, + // Chromium tient toutes les cases : la frame est sautée, et la boucle ne + // tourne pas à vide le temps qu'il en relâche une. + SharedPublish::NoFreeSlot => std::thread::sleep(Duration::from_millis(2)), + // Step complet : `compose_frame` (déjà appelé par `step`/`present_frame`/ + // `recompose`) a rastérisé le RT à la géométrie de sortie ramenée au panneau. + // On lit ce RT DIRECTEMENT à sa résolution de rendu (`readback_direct` : copy + // rt → staging → Map/Unmap), sans le resize `blit_resized` qui, depuis la + // refonte ratio, n'était plus qu'une copie identité + une alloc NV12 inutile. + SharedPublish::Off => match comp.readback_direct() { + Ok((rw, rh, rgba)) => { + // Publie dans `latest_frame` : on remplace le buffer précédent + // (le canvas ne montre que la dernière frame, peu importe combien + // le renderer en a raté entre deux lectures napi). On incrémente + // la génération sous le MÊME lock que l'écriture du buffer, pour + // qu'un lecteur ne puisse jamais voir un `gen` neuf appairé à un + // buffer périmé (ou l'inverse). `+ 1` depuis la précédente, `1` au + // premier publish. Les dims publiées sont celles du RENDU (`rw`×`rh`) : + // le canvas JS s'y dimensionne (packet auto-descriptif) puis CSS met à + // l'échelle vers la boîte du panneau — plus de resize GPU intermédiaire. + // La génération vient d'un compteur atomique et non du slot : la + // livraison sans copie VIDE le slot en le lisant, et un + // `unwrap_or(1)` repartirait alors de 1 — le consommateur recevrait + // des générations déjà peintes et boucherait. Séquence identique à + // l'ancienne dérivation tant que le slot n'est pas vidé. + let next_gen = shared.frame_gen.fetch_add(1, Ordering::Relaxed) + 1; + if let Ok(mut slot) = shared.latest_frame.lock() { + *slot = Some((next_gen, rw, rh, rgba, comp.footage_quad())); + } + first = false; } - first = false; - } - Err(e) => { - eprintln!("[live] readback_direct: {e:#}"); - std::thread::sleep(Duration::from_millis(8)); - } + Err(e) => { + eprintln!("[live] readback_direct: {e:#}"); + std::thread::sleep(Duration::from_millis(8)); + } + }, } } } else { @@ -1825,6 +1882,88 @@ unsafe fn render_thread( Ok(()) } +/// Ce que la publication dans l'anneau partagé a fait de la frame composée. +enum SharedPublish { + /// Posée dans une case de l'anneau. + Published, + /// Chromium tient toutes les cases : frame sautée. + NoFreeSlot, + /// Transport partagé coupé ou indisponible : la frame repasse par le readback. + Off, +} + +#[cfg(windows)] +type Ring = Option; +#[cfg(not(windows))] +type Ring = (); + +/// Pose la frame composée dans l'anneau de textures partagées, si la vue livre ainsi. +/// +/// Toute panne coupe le transport (`shared_frames` à `false`) et rend `Off` : la frame part +/// par le readback, et le service, qui lit les deux, n'a rien à décider. +#[cfg(windows)] +unsafe fn publish_shared( + shared: &Shared, + gpu: &Gpu, + comp: &Compositor, + ring: &mut Ring, +) -> SharedPublish { + if !shared.shared_frames.load(Ordering::Relaxed) { + return SharedPublish::Off; + } + let turn_off = |why: String| { + eprintln!("[live] textures partagées coupées ({why}) — retour au readback"); + shared.shared_frames.store(false, Ordering::Relaxed); + SharedPublish::Off + }; + // WARP ne partage rien avec le device de Chromium. + if gpu.backend != Backend::Hardware { + return turn_off("backend logiciel".into()); + } + if ring.is_none() { + match SharedRing::new(gpu) { + Ok(created) => *ring = Some(created), + Err(e) => return turn_off(format!("{e:#}")), + } + } + let Some(ring) = ring.as_mut() else { + return SharedPublish::Off; + }; + let claimed = shared + .slot_book + .lock() + .ok() + .and_then(|mut book| book.claim(crate::shared_frames::RING_SLOTS, Instant::now())); + let Some(slot) = claimed else { + return SharedPublish::NoFreeSlot; + }; + let (width, height) = comp.render_size(); + match ring.write(slot, comp.render_target(), width, height) { + Ok(handle) => { + if let Ok(mut book) = shared.slot_book.lock() { + // Le compteur du readback : une seule suite de générations quel que soit le + // transport, incrémentée sous le lock de la publication comme là-bas. + let gen = shared.frame_gen.fetch_add(1, Ordering::Relaxed) + 1; + book.publish(SharedFrame { + gen, + slot, + handle, + width, + height, + footage: comp.footage_quad(), + }); + } + SharedPublish::Published + } + Err(e) => turn_off(format!("{e:#}")), + } +} + +#[cfg(not(windows))] +unsafe fn publish_shared(_: &Shared, _: &Gpu, _: &Compositor, _: &mut Ring) -> SharedPublish { + SharedPublish::Off +} + // ---------- harnais standalone (poc-d3d.exe --live) ---------- // // Le harnais crée une fenêtre Win32 simple qui héberge la preview live ; c'est un diff --git a/crates/compositor/src/shared_frames.rs b/crates/compositor/src/shared_frames.rs new file mode 100644 index 000000000..bf0e1763a --- /dev/null +++ b/crates/compositor/src/shared_frames.rs @@ -0,0 +1,329 @@ +//! Livraison de la preview par textures partagées : le thread de rendu copie chaque frame +//! composée dans l'une des textures d'un petit anneau, et Electron l'importe côté GPU +//! (`sharedTexture.importSharedTexture`) au lieu d'en recevoir les pixels. Plus de `Map` +//! qui attend le GPU puis recopie l'image en RAM, plus de `Vec` de plusieurs Mo par frame +//! à travers l'IPC : mesuré, ce transport coûtait à lui seul 37 à 55 % du thread principal +//! du renderer à 30 images/s. +//! +//! Deux moitiés : +//! - `SlotBook`, portable et sans GPU : quelle case porte la frame prête, lesquelles +//! Chromium tient encore. Partagé entre le thread de rendu (`claim`, `publish`) et le +//! thread Node (`take`, `release`). +//! - `SharedRing` (Windows, backend matériel) : les textures D3D11 et leurs handles NT. +//! macOS (IOSurface) et Linux (dmabuf) restent sur le readback. + +use crate::frame_geometry::FootageQuad; +use std::time::{Duration, Instant}; + +/// Cases de l'anneau : une prête, une en transit vers le renderer (tenue jusqu'à ce que +/// Chromium la relâche), une en écriture, et une de marge pour la latence de la libération. +pub const RING_SLOTS: u32 = 4; + +/// Une case tenue plus longtemps que ça a perdu sa libération (renderer rechargé, import +/// échoué en route) : elle est reprise plutôt que de laisser l'anneau se vider et la preview +/// se figer. Même ordre que le délai d'Electron sur `sendSharedTexture` (1 s). +pub const LOST_SLOT_AFTER: Duration = Duration::from_secs(1); + +/// Une frame composée posée dans une case de l'anneau, telle que JS la reçoit. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct SharedFrame { + /// Même compteur que les frames lues en RAM : le consommateur ne voit qu'une suite de + /// générations, quel que soit le transport. + pub gen: u64, + pub slot: u32, + /// Valeur du handle NT de la texture de la case, valable dans CE processus. Electron le + /// duplique à l'import : la case garde le sien. + pub handle: u64, + pub width: u32, + pub height: u32, + pub footage: Option, +} + +/// Une case prise par JS : la génération qu'elle porte et quand elle est partie. +#[derive(Debug, Clone, Copy)] +struct Held { + slot: u32, + gen: u64, + since: Instant, +} + +/// Qui tient quelle case de l'anneau. +#[derive(Debug, Default)] +pub struct SlotBook { + /// La dernière frame publiée, que JS n'a pas encore prise. + ready: Option, + /// Les cases prises par JS, que Chromium n'a pas encore relâchées, plus anciennes d'abord. + held: Vec, +} + +impl SlotBook { + /// La case où écrire la prochaine frame. Une case libre d'abord ; à défaut, celle de la + /// frame prête que personne n'a prise, retirée puisqu'elle est périmée ; à défaut, une + /// case tenue depuis `LOST_SLOT_AFTER`. `None` quand Chromium tient légitimement toutes + /// les cases : la frame est sautée, la suivante passera. + pub fn claim(&mut self, slots: u32, now: Instant) -> Option { + let ready = self.ready.map(|frame| frame.slot); + let is_held = |slot: u32| self.held.iter().any(|held| held.slot == slot); + if let Some(free) = (0..slots).find(|slot| !is_held(*slot) && Some(*slot) != ready) { + return Some(free); + } + if let Some(stale) = ready { + self.ready = None; + return Some(stale); + } + let lost = self.held.first().filter(|held| now.duration_since(held.since) >= LOST_SLOT_AFTER)?; + let slot = lost.slot; + self.held.remove(0); + Some(slot) + } + + pub fn publish(&mut self, frame: SharedFrame) { + self.ready = Some(frame); + } + + /// La frame prête, si elle est plus récente que `since_gen`. Sa case reste tenue jusqu'à + /// `release` : le thread de rendu n'y écrira plus d'ici là. + pub fn take(&mut self, since_gen: u64, now: Instant) -> Option { + let frame = self.ready.filter(|frame| frame.gen > since_gen)?; + self.ready = None; + self.held.push(Held { slot: frame.slot, gen: frame.gen, since: now }); + Some(frame) + } + + /// Chromium a relâché la frame `gen` de la case `slot`. La génération évite qu'une + /// libération tardive, arrivée après que la case a été reprise, ne libère la frame + /// suivante. + pub fn release(&mut self, slot: u32, gen: u64) { + self.held.retain(|held| held.slot != slot || held.gen != gen); + } + + /// Oublie la frame prête : la vue repasse au readback, personne ne viendra la prendre. + pub fn clear_ready(&mut self) { + self.ready = None; + } +} + +#[cfg(windows)] +pub use ring::SharedRing; + +#[cfg(windows)] +mod ring { + use super::RING_SLOTS; + use crate::d3d::Gpu; + use anyhow::{bail, Result}; + use std::time::{Duration, Instant}; + use windows::core::{Interface, PCWSTR}; + use windows::Win32::Foundation::{CloseHandle, BOOL, HANDLE}; + use windows::Win32::Graphics::Direct3D11::{ + ID3D11Device, ID3D11DeviceContext, ID3D11Query, ID3D11Texture2D, + D3D11_ASYNC_GETDATA_DONOTFLUSH, D3D11_BIND_RENDER_TARGET, D3D11_BIND_SHADER_RESOURCE, + D3D11_QUERY_DESC, D3D11_QUERY_EVENT, D3D11_RESOURCE_MISC_SHARED, + D3D11_RESOURCE_MISC_SHARED_NTHANDLE, D3D11_TEXTURE2D_DESC, D3D11_USAGE_DEFAULT, + }; + use windows::Win32::Graphics::Dxgi::Common::{DXGI_FORMAT_R8G8B8A8_UNORM, DXGI_SAMPLE_DESC}; + use windows::Win32::Graphics::Dxgi::{ + IDXGIResource1, DXGI_SHARED_RESOURCE_READ, DXGI_SHARED_RESOURCE_WRITE, + }; + + /// Au-delà, le GPU est bloqué ou perdu : on rend une erreur plutôt que de geler la vue. + const COPY_TIMEOUT: Duration = Duration::from_millis(500); + + struct SlotTexture { + tex: ID3D11Texture2D, + handle: HANDLE, + width: u32, + height: u32, + } + + impl Drop for SlotTexture { + fn drop(&mut self) { + // Electron importe un duplicata du handle : fermer le nôtre ne retire rien à une + // frame que Chromium affiche encore. + unsafe { + let _ = CloseHandle(self.handle); + } + } + } + + /// Les textures partagées d'une vue live, créées à la demande à la taille du rendu. + pub struct SharedRing { + device: ID3D11Device, + context: ID3D11DeviceContext, + copied: ID3D11Query, + slots: Vec>, + } + + impl SharedRing { + pub fn new(gpu: &Gpu) -> Result { + let desc = D3D11_QUERY_DESC { Query: D3D11_QUERY_EVENT, MiscFlags: 0 }; + let mut copied = None; + unsafe { gpu.device.CreateQuery(&desc, Some(&mut copied))? }; + Ok(SharedRing { + device: gpu.device.clone(), + context: gpu.context.clone(), + copied: copied.expect("CreateQuery a réussi sans requête"), + slots: (0..RING_SLOTS).map(|_| None).collect(), + }) + } + + /// Copie `src` (`width`×`height`, RGBA8) dans la case `slot` et rend le handle de sa + /// texture une fois le GPU arrivé au bout de la copie. Chromium la lit depuis son propre + /// device, sans keyed mutex : quand le handle part, plus rien ne doit y écrire. L'attente + /// est celle que faisait déjà le `Map` du readback, la recopie en RAM en moins. + /// + /// La case ne doit être ni prête ni tenue (`SlotBook::claim`) : sa texture peut être + /// recréée ici quand la taille du rendu a changé. + pub unsafe fn write( + &mut self, + slot: u32, + src: &ID3D11Texture2D, + width: u32, + height: u32, + ) -> Result { + let entry = &mut self.slots[slot as usize]; + if !matches!(entry, Some(texture) if texture.width == width && texture.height == height) { + *entry = Some(create_slot(&self.device, width, height)?); + } + let target = entry.as_ref().expect("case créée juste au-dessus"); + self.context.CopyResource(&target.tex, src); + self.context.End(&self.copied); + self.context.Flush(); + let deadline = Instant::now() + COPY_TIMEOUT; + loop { + let mut done = BOOL(0); + self.context.GetData( + &self.copied, + Some(&mut done as *mut BOOL as *mut core::ffi::c_void), + std::mem::size_of::() as u32, + D3D11_ASYNC_GETDATA_DONOTFLUSH.0 as u32, + )?; + if done.as_bool() { + break; + } + if Instant::now() > deadline { + bail!("le GPU n'a pas fini de copier la frame en {COPY_TIMEOUT:?}"); + } + std::thread::yield_now(); + } + Ok(target.handle.0 as u64) + } + } + + unsafe fn create_slot(device: &ID3D11Device, width: u32, height: u32) -> Result { + let desc = D3D11_TEXTURE2D_DESC { + Width: width, + Height: height, + MipLevels: 1, + ArraySize: 1, + // Le format du RT composé : la copie est un `CopyResource`, sans conversion. + Format: DXGI_FORMAT_R8G8B8A8_UNORM, + SampleDesc: DXGI_SAMPLE_DESC { Count: 1, Quality: 0 }, + Usage: D3D11_USAGE_DEFAULT, + BindFlags: (D3D11_BIND_RENDER_TARGET.0 | D3D11_BIND_SHADER_RESOURCE.0) as u32, + CPUAccessFlags: 0, + // Handle NT, le seul qu'Electron importe, sans keyed mutex : `write` attend la fin + // de la copie avant de le livrer. + MiscFlags: (D3D11_RESOURCE_MISC_SHARED.0 | D3D11_RESOURCE_MISC_SHARED_NTHANDLE.0) as u32, + }; + let mut tex = None; + device.CreateTexture2D(&desc, None, Some(&mut tex))?; + let tex = tex.expect("CreateTexture2D a réussi sans texture"); + let resource: IDXGIResource1 = tex.cast()?; + let handle = resource.CreateSharedHandle( + None, + DXGI_SHARED_RESOURCE_READ.0 | DXGI_SHARED_RESOURCE_WRITE.0, + PCWSTR::null(), + )?; + Ok(SlotTexture { tex, handle, width, height }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn frame(gen: u64, slot: u32) -> SharedFrame { + SharedFrame { gen, slot, handle: 0, width: 2, height: 2, footage: None } + } + + /// Toutes les cases prises par JS, la case `n` portant la génération `n + 1`. + fn all_held(now: Instant) -> SlotBook { + let mut book = SlotBook::default(); + for slot in 0..RING_SLOTS { + book.publish(frame(u64::from(slot) + 1, slot)); + assert!(book.take(u64::from(slot), now).is_some()); + } + book + } + + #[test] + fn claims_a_free_slot_before_a_ready_one() { + let now = Instant::now(); + let mut book = SlotBook::default(); + book.publish(frame(1, 0)); + assert_eq!(book.claim(RING_SLOTS, now), Some(1)); + // La frame prête n'a pas bougé : JS peut toujours la prendre. + assert_eq!(book.take(0, now), Some(frame(1, 0))); + } + + #[test] + fn a_taken_slot_is_never_written_until_released() { + let now = Instant::now(); + let mut book = all_held(now); + assert_eq!(book.claim(RING_SLOTS, now), None, "Chromium tient toutes les cases"); + book.release(2, 3); + assert_eq!(book.claim(RING_SLOTS, now), Some(2)); + } + + #[test] + fn a_frame_nobody_took_gives_its_slot_back_when_nothing_else_is_free() { + let now = Instant::now(); + let mut book = SlotBook::default(); + for slot in 0..RING_SLOTS - 1 { + book.publish(frame(u64::from(slot) + 1, slot)); + book.take(u64::from(slot), now); + } + book.publish(frame(9, 3)); + assert_eq!(book.claim(RING_SLOTS, now), Some(3)); + // Retirée : la livrer maintenant, pendant qu'on la réécrit, donnerait une image déchirée. + assert_eq!(book.take(0, now), None); + } + + #[test] + fn take_hands_out_only_newer_generations() { + let now = Instant::now(); + let mut book = SlotBook::default(); + book.publish(frame(5, 1)); + assert_eq!(book.take(5, now), None); + assert_eq!(book.take(4, now), Some(frame(5, 1))); + assert_eq!(book.take(0, now), None, "déjà prise"); + } + + #[test] + fn a_release_only_frees_the_generation_it_names() { + let now = Instant::now(); + let mut book = SlotBook::default(); + book.publish(frame(1, 0)); + book.take(0, now); + book.release(3, 1); + book.release(0, 7); + assert_eq!(book.claim(1, now), None, "ni la case 3 ni la génération 7 ne sont tenues"); + book.release(0, 1); + assert_eq!(book.claim(1, now), Some(0)); + } + + #[test] + fn a_slot_whose_release_never_came_is_taken_back_after_a_second() { + let start = Instant::now(); + let mut book = all_held(start); + assert_eq!(book.claim(RING_SLOTS, start + LOST_SLOT_AFTER / 2), None); + // La plus ancienne d'abord. + assert_eq!(book.claim(RING_SLOTS, start + LOST_SLOT_AFTER), Some(0)); + // Sa libération tardive ne doit plus rien libérer : la case est déjà reprise. + book.release(0, 1); + book.publish(frame(10, 0)); + assert!(book.take(9, start + LOST_SLOT_AFTER).is_some()); + book.release(0, 1); + assert_eq!(book.claim(RING_SLOTS, start + LOST_SLOT_AFTER), Some(1)); + } +} diff --git a/crates/compositor/tests/shared_frame_handoff.rs b/crates/compositor/tests/shared_frame_handoff.rs new file mode 100644 index 000000000..2465f2b98 --- /dev/null +++ b/crates/compositor/tests/shared_frame_handoff.rs @@ -0,0 +1,135 @@ +//! La texture qu'une vue live livre par handle NT, rouverte par un AUTRE device D3D11 — la +//! place qu'occupe le processus GPU de Chromium quand Electron l'importe. Ce que le +//! consommateur lit doit être exactement ce que le compositeur a composé : pas une image +//! à moitié copiée, pas une case périmée, et une case redimensionnée doit livrer un handle +//! neuf à la nouvelle taille. +//! +//! Demande un device matériel : saute sans lui (runner sans adaptateur). + +#![cfg(windows)] + +use openscreen_compositor::d3d::Gpu; +use openscreen_compositor::shared_frames::SharedRing; +use windows::core::Interface; +use windows::Win32::Foundation::{HANDLE, HMODULE}; +use windows::Win32::Graphics::Direct3D::{D3D_DRIVER_TYPE_HARDWARE, D3D_FEATURE_LEVEL_11_1}; +use windows::Win32::Graphics::Direct3D11::{ + D3D11CreateDevice, ID3D11Device, ID3D11Device1, ID3D11DeviceContext, ID3D11Texture2D, + D3D11_BIND_SHADER_RESOURCE, D3D11_CPU_ACCESS_READ, D3D11_CREATE_DEVICE_FLAG, + D3D11_MAPPED_SUBRESOURCE, D3D11_MAP_READ, D3D11_SDK_VERSION, D3D11_SUBRESOURCE_DATA, + D3D11_TEXTURE2D_DESC, D3D11_USAGE_DEFAULT, D3D11_USAGE_STAGING, +}; +use windows::Win32::Graphics::Dxgi::Common::{DXGI_FORMAT_R8G8B8A8_UNORM, DXGI_SAMPLE_DESC}; + +fn pattern(width: u32, height: u32, seed: u8) -> Vec { + (0..width * height) + .flat_map(|i| { + let (x, y) = (i % width, i / width); + [x as u8 ^ seed, y as u8, seed, 255] + }) + .collect() +} + +fn texture_with(gpu: &Gpu, width: u32, height: u32, pixels: &[u8]) -> ID3D11Texture2D { + let desc = D3D11_TEXTURE2D_DESC { + Width: width, + Height: height, + MipLevels: 1, + ArraySize: 1, + Format: DXGI_FORMAT_R8G8B8A8_UNORM, + SampleDesc: DXGI_SAMPLE_DESC { Count: 1, Quality: 0 }, + Usage: D3D11_USAGE_DEFAULT, + BindFlags: D3D11_BIND_SHADER_RESOURCE.0 as u32, + CPUAccessFlags: 0, + MiscFlags: 0, + }; + let init = D3D11_SUBRESOURCE_DATA { + pSysMem: pixels.as_ptr().cast(), + SysMemPitch: width * 4, + SysMemSlicePitch: 0, + }; + let mut tex = None; + unsafe { gpu.device.CreateTexture2D(&desc, Some(&init), Some(&mut tex)) }.expect("texture"); + tex.expect("texture") +} + +/// Le second device, sans rien de commun avec celui du compositeur que l'adaptateur. +fn consumer() -> (ID3D11Device, ID3D11DeviceContext) { + let mut device = None; + let mut context = None; + unsafe { + D3D11CreateDevice( + None, + D3D_DRIVER_TYPE_HARDWARE, + HMODULE::default(), + D3D11_CREATE_DEVICE_FLAG(0), + Some(&[D3D_FEATURE_LEVEL_11_1]), + D3D11_SDK_VERSION, + Some(&mut device), + None, + Some(&mut context), + ) + } + .expect("second device"); + (device.unwrap(), context.unwrap()) +} + +/// Ce que le consommateur lit à travers le handle, en RGBA serré. +fn read_through(handle: u64, width: u32, height: u32) -> Vec { + let (device, context) = consumer(); + let device1: ID3D11Device1 = device.cast().expect("ID3D11Device1"); + let shared: ID3D11Texture2D = + unsafe { device1.OpenSharedResource1(HANDLE(handle as *mut core::ffi::c_void)) } + .expect("le handle doit s'ouvrir depuis un autre device"); + let mut desc = D3D11_TEXTURE2D_DESC::default(); + unsafe { shared.GetDesc(&mut desc) }; + assert_eq!((desc.Width, desc.Height), (width, height)); + desc.Usage = D3D11_USAGE_STAGING; + desc.BindFlags = 0; + desc.CPUAccessFlags = D3D11_CPU_ACCESS_READ.0 as u32; + desc.MiscFlags = 0; + let mut staging = None; + unsafe { device.CreateTexture2D(&desc, None, Some(&mut staging)) }.expect("staging"); + let staging = staging.unwrap(); + let mut mapped = D3D11_MAPPED_SUBRESOURCE::default(); + let mut out = vec![0u8; (width * height * 4) as usize]; + unsafe { + context.CopyResource(&staging, &shared); + context.Map(&staging, 0, D3D11_MAP_READ, 0, Some(&mut mapped)).expect("map"); + for y in 0..height as usize { + std::ptr::copy_nonoverlapping( + (mapped.pData as *const u8).add(y * mapped.RowPitch as usize), + out.as_mut_ptr().add(y * width as usize * 4), + width as usize * 4, + ); + } + context.Unmap(&staging, 0); + } + out +} + +#[test] +fn another_device_reads_exactly_the_composed_frame() { + let Ok(gpu) = Gpu::create(false) else { + eprintln!("pas de device D3D11 matériel — saute"); + return; + }; + let mut ring = SharedRing::new(&gpu).expect("anneau"); + + let (w, h) = (64, 36); + let first = pattern(w, h, 0x11); + let handle = unsafe { ring.write(0, &texture_with(&gpu, w, h, &first), w, h) }.expect("écriture"); + assert_eq!(read_through(handle, w, h), first); + + // La même case réécrite : même texture, même handle, nouveau contenu. + let second = pattern(w, h, 0x22); + let again = unsafe { ring.write(0, &texture_with(&gpu, w, h, &second), w, h) }.expect("réécriture"); + assert_eq!(again, handle); + assert_eq!(read_through(again, w, h), second); + + // Le rendu change de taille : la case est recréée, avec un handle à la nouvelle taille. + let (w2, h2) = (32, 18); + let resized = pattern(w2, h2, 0x33); + let fresh = unsafe { ring.write(0, &texture_with(&gpu, w2, h2, &resized), w2, h2) }.expect("taille"); + assert_eq!(read_through(fresh, w2, h2), resized); +} diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index e0fddd35c..26f2f5308 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -37,6 +37,15 @@ interface Window { * `compositor.export`/`compositor.exportMulti` runs. Distinct from `exportOnFrameAck`, * the OLD web/CPU pipeline's per-frame ack, not a progress signal. */ onNativeExportProgress?: (callback: (frames: number, exportId?: string) => void) => () => void; + /** Preview frames the main process hands over as shared GPU textures (Windows). The + * listener draws `frame` and closes it; one listener at a time. Returns the + * unsubscribe. Optional: shim/web contexts have no bridge. */ + onCompositorFrame?: ( + listener: ( + frame: VideoFrame, + meta: import("../src/native/contracts").CompositorSharedFrameMeta, + ) => void, + ) => () => void; getSources: (opts: Electron.SourcesOptions) => Promise; switchToEditor: () => Promise; switchToHud: () => Promise; diff --git a/electron/ipc/nativeBridge.ts b/electron/ipc/nativeBridge.ts index 452f42701..61ed8ef86 100644 --- a/electron/ipc/nativeBridge.ts +++ b/electron/ipc/nativeBridge.ts @@ -384,18 +384,23 @@ export function registerNativeBridgeHandlers(context: NativeBridgeContext) { compositorViewService.setRect(request.payload.id, request.payload.rect); return createSuccessResponse(requestId, { ok: true }); case "readFrame": { - // The renderer polls this every rAF tick (~30fps). It passes the - // generation it last painted as `sinceGen`; native returns `null` when - // nothing newer exists (idle path — no buffer copy). On a new frame it - // returns `{ gen, width, height, data }`. The response wrapper does NOT - // JSON-stringify — `ipcMain.handle` round-trips via structured clone, - // which preserves the nested `Buffer` in `.data` as binary. - const frame = compositorViewService.readFrame( + // The renderer polls this every rAF tick. It passes the generation it + // last painted as `sinceGen`; native returns `null` when nothing newer + // exists (idle path — no buffer copy). On a new frame it returns + // `{ gen, width, height, data }`, or — for a view on shared textures — + // sends the texture to the asking frame and returns its receipt. The + // response wrapper does NOT JSON-stringify — `ipcMain.handle` round-trips + // via structured clone, which preserves the nested `Buffer` as binary. + const frame = await compositorViewService.readFrame( request.payload.id, request.payload.sinceGen, + event.senderFrame, ); return createSuccessResponse(requestId, frame); } + case "stopSharedFrames": + compositorViewService.stopSharedFrames(request.payload.id); + return createSuccessResponse(requestId, { ok: true }); case "setParam": compositorViewService.setParam( request.payload.id, diff --git a/electron/native-bridge/services/compositorViewService.test.ts b/electron/native-bridge/services/compositorViewService.test.ts index eee8fe305..4a099a7f9 100644 --- a/electron/native-bridge/services/compositorViewService.test.ts +++ b/electron/native-bridge/services/compositorViewService.test.ts @@ -1,6 +1,7 @@ import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +import type { WebFrameMain } from "electron"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; import { CURSOR_THEMES, DEFAULT_CURSOR_SPRITES } from "../../../src/lib/cursor/cursorThemes"; import type { CompositorViewAddon, GifExportStats } from "../../native/compositor-view/addon"; @@ -58,6 +59,166 @@ describe("native GIF cancellation capability", () => { }); }); +describe("CompositorViewService frames handed over as shared GPU textures", () => { + const target = { frameTreeNodeId: 1 } as unknown as WebFrameMain; + const sharedFrame = { + gen: 3, + slot: 1, + handle: Buffer.from([1, 2, 3, 4, 5, 6, 7, 8]), + width: 4, + height: 2, + footage: null, + footageProjective: false, + }; + + function setup(options: { addon?: Partial; gpuCompositing?: boolean } = {}) { + const addon = { + createView: vi.fn(() => 7), + setSharedFrames: vi.fn(() => true), + readSharedFrame: vi.fn(() => sharedFrame), + readFrame: vi.fn(() => null), + releaseSharedFrame: vi.fn(), + destroyView: vi.fn(), + ...options.addon, + }; + const imported = { release: vi.fn() }; + let allReferencesReleased: (() => void) | undefined; + const sharedTexture = { + importSharedTexture: vi.fn((importOptions: Electron.ImportSharedTextureOptions) => { + allReferencesReleased = importOptions.allReferencesReleased as () => void; + return imported as unknown as Electron.SharedTextureImported; + }), + sendSharedTexture: vi.fn(async () => undefined), + }; + const service = new CompositorViewService({ + addon: addon as unknown as CompositorViewAddon, + sharedTexture, + gpuCompositing: () => options.gpuCompositing ?? true, + }); + const id = service.createView({ x: 0, y: 0, width: 4, height: 2 }); + return { + addon, + id, + imported, + service, + sharedTexture, + releaseEverywhere: () => allReferencesReleased?.(), + }; + } + + afterEach(() => { + vi.unstubAllEnvs(); + }); + + it("sends the texture to the frame that asked, and answers with its receipt", async () => { + const { addon, id, imported, service, sharedTexture } = setup(); + expect(addon.setSharedFrames).toHaveBeenCalledWith(id, true); + + const receipt = await service.readFrame(id, 2, target); + + expect(addon.readSharedFrame).toHaveBeenCalledWith(id, 2); + expect(sharedTexture.importSharedTexture).toHaveBeenCalledWith( + expect.objectContaining({ + textureInfo: { + pixelFormat: "rgba", + codedSize: { width: 4, height: 2 }, + handle: { ntHandle: sharedFrame.handle }, + }, + }), + ); + const meta = { + viewId: id, + gen: 3, + width: 4, + height: 2, + footage: null, + footageProjective: false, + }; + expect(sharedTexture.sendSharedTexture).toHaveBeenCalledWith( + { frame: target, importedSharedTexture: imported }, + meta, + ); + expect(receipt).toEqual({ ...meta, shared: true }); + // This process's reference goes at once: the renderer holds its own. + expect(imported.release).toHaveBeenCalled(); + expect(addon.releaseSharedFrame).not.toHaveBeenCalled(); + }); + + it("hands the slot back once Chromium has let go of the texture everywhere", async () => { + const { addon, id, releaseEverywhere, service } = setup(); + await service.readFrame(id, 0, target); + + releaseEverywhere(); + + expect(addon.releaseSharedFrame).toHaveBeenCalledWith(id, 1, 3); + }); + + it("goes back to read-back when a delivery fails", async () => { + const { addon, id, service, sharedTexture } = setup(); + sharedTexture.sendSharedTexture.mockRejectedValueOnce(new Error("timed out after 1000ms")); + const warn = vi.spyOn(console, "warn").mockImplementation(() => undefined); + try { + expect(await service.readFrame(id, 0, target)).toBeNull(); + } finally { + warn.mockRestore(); + } + expect(addon.setSharedFrames).toHaveBeenLastCalledWith(id, false); + + vi.mocked(addon.readSharedFrame).mockClear(); + await service.readFrame(id, 3, target); + expect(addon.readSharedFrame).not.toHaveBeenCalled(); + expect(addon.readFrame).toHaveBeenCalledWith(id, 3); + }); + + it("frees the slot itself when the texture never got imported", async () => { + const { addon, id, service, sharedTexture } = setup(); + sharedTexture.importSharedTexture.mockImplementationOnce(() => { + throw new TypeError("Invalid ntHandle value"); + }); + const warn = vi.spyOn(console, "warn").mockImplementation(() => undefined); + try { + await service.readFrame(id, 0, target); + } finally { + warn.mockRestore(); + } + expect(addon.releaseSharedFrame).toHaveBeenCalledWith(id, 1, 3); + }); + + it("reads pixels for a view whose render thread went back to read-back on its own", async () => { + const packet = { gen: 4, width: 4, height: 2, data: Buffer.alloc(32) }; + const { id, service, sharedTexture } = setup({ + addon: { readSharedFrame: vi.fn(() => null), readFrame: vi.fn(() => packet) }, + }); + + expect(await service.readFrame(id, 3, target)).toBe(packet); + expect(sharedTexture.sendSharedTexture).not.toHaveBeenCalled(); + }); + + it("keeps a view on read-back when Chromium does not composite on the GPU", async () => { + const { addon, id, service } = setup({ gpuCompositing: false }); + expect(addon.setSharedFrames).not.toHaveBeenCalled(); + + await service.readFrame(id, 0, target); + expect(addon.readSharedFrame).not.toHaveBeenCalled(); + expect(addon.readFrame).toHaveBeenCalledWith(id, 0); + }); + + it("keeps a view on read-back when OPENSCREEN_PREVIEW_READBACK=1", () => { + vi.stubEnv("OPENSCREEN_PREVIEW_READBACK", "1"); + const { addon } = setup(); + expect(addon.setSharedFrames).not.toHaveBeenCalled(); + }); + + it("keeps a view on read-back when the addon cannot share (not Windows, software backend)", async () => { + const { id, service, sharedTexture } = setup({ + addon: { setSharedFrames: vi.fn(() => false) }, + }); + + await service.readFrame(id, 0, target); + expect(sharedTexture.importSharedTexture).not.toHaveBeenCalled(); + }); +}); + /** A source checkout's `crates/.cargo/config.toml`, with `FFMPEG_DIR` written as `body`. */ function writeCargoConfig(root: string, body: string): void { const cargoDir = path.join(root, "crates", ".cargo"); diff --git a/electron/native-bridge/services/compositorViewService.ts b/electron/native-bridge/services/compositorViewService.ts index 4a765db2b..51e562979 100644 --- a/electron/native-bridge/services/compositorViewService.ts +++ b/electron/native-bridge/services/compositorViewService.ts @@ -2,12 +2,16 @@ import fs from "node:fs"; import { createRequire } from "node:module"; import path from "node:path"; import { fileURLToPath } from "node:url"; -import { app } from "electron"; +import { app, sharedTexture, type WebFrameMain } from "electron"; import { type CursorKind, readCursorAsArrow, resolveCursorSprites, } from "../../../src/lib/cursor/cursorThemes"; +import type { + CompositorSharedFrameMeta, + CompositorSharedFrameReceipt, +} from "../../../src/native/contracts"; import type { GifExportJob } from "../../ipc/gifExportJobs"; import type { ClipInput, @@ -20,6 +24,7 @@ import type { GifExportStats, GifParamsInput, NativeFramePacket, + NativeSharedFramePacket, RemuxStats, SegmentationSupport, } from "../../native/compositor-view/addon"; @@ -215,6 +220,26 @@ export interface CompositorViewServiceOptions { */ appRoot?: string; isPackaged?: boolean; + /** Electron's `sharedTexture` module, injectable for tests. `null` keeps every view on + * read-back. */ + sharedTexture?: SharedTextureApi | null; + /** Whether Chromium composites on the GPU, where a shared texture is imported. Injectable + * for tests; defaults to `app.getGPUFeatureStatus()`. */ + gpuCompositing?: () => boolean; +} + +/** The two calls a shared preview frame needs from Electron's `sharedTexture` module. */ +export type SharedTextureApi = Pick< + Electron.SharedTexture, + "importSharedTexture" | "sendSharedTexture" +>; + +function defaultGpuCompositing(): boolean { + try { + return String(app.getGPUFeatureStatus().gpu_compositing).startsWith("enabled"); + } catch { + return false; + } } function defaultAppRoot(): string { @@ -467,6 +492,8 @@ function tryLoadAddon(candidates: string[]): CompositorViewAddon | null { export class CompositorViewService { private readonly options: CompositorViewServiceOptions; private readonly rects = new Map(); + /** Views whose frames go out as shared GPU textures rather than RAM pixels. */ + private readonly sharedViews = new Set(); private addon: CompositorViewAddon | null = null; private loadAttempted = false; private syntheticIdCounter = 0; @@ -593,9 +620,41 @@ export class CompositorViewService { } const id = addon.createView(rect, paths?.screenPath, paths?.webcamPath, paths?.cursorPath); this.rects.set(id, rect); + this.shareFrames(addon, id); return id; } + /** Hands the view's frames over as shared GPU textures when this host can, instead of + * copying them through IPC: measured, that transport alone kept 37 to 55 % of the + * renderer's main thread busy at 30 fps (rendering-performance.md). Anything missing — + * the Electron API, GPU compositing, Windows' hardware backend native-side — leaves the + * view on read-back. `OPENSCREEN_PREVIEW_READBACK=1` forces read-back, to compare. */ + private shareFrames(addon: CompositorViewAddon, id: number): void { + if (process.env.OPENSCREEN_PREVIEW_READBACK === "1" || !this.sharedTextureApi()) { + return; + } + if (!(this.options.gpuCompositing ?? defaultGpuCompositing)()) { + return; + } + if (addon.setSharedFrames?.(id, true)) { + this.sharedViews.add(id); + } + } + + private sharedTextureApi(): SharedTextureApi | null { + const api = + this.options.sharedTexture === undefined ? sharedTexture : this.options.sharedTexture; + return typeof api?.importSharedTexture === "function" ? api : null; + } + + /** Takes the view back to read-back frames: a delivery failed, or the renderer saw a shared + * frame land as nothing (Chromium could not open the texture). The render thread + * republishes the current frame, so the canvas is not left empty. */ + stopSharedFrames(id: number): void { + this.sharedViews.delete(id); + this.ensureAddon()?.setSharedFrames?.(id, false); + } + setRect(id: number, rect: CompositorViewRect): void { const addon = this.ensureAddon(); this.rects.set(id, rect); @@ -605,19 +664,76 @@ export class CompositorViewService { addon.setRect(id, rect); } - /** Reads the most recently rendered frame for `id` as a self-describing packet - * (`{ gen, width, height, data }`), but only if its generation is newer than - * `sinceGen`. Returns `null` when the addon is absent, no frame is ready yet, - * OR the caller already holds the current generation — the idle path, where - * `null` comes back without any buffer copy. Byte order is RGBA. */ - readFrame(id: number, sinceGen: number): NativeFramePacket | null { + /** Reads the most recently rendered frame for `id`, but only if its generation is newer + * than `sinceGen`. Returns `null` when the addon is absent, no frame is ready yet, OR the + * caller already holds the current generation — the idle path, where nothing is copied. + * + * A view on shared textures sends the frame to `target` as a GPU texture, and answers with + * its receipt (`shared: true`, no pixels): the texture reaches the renderer before this + * reply does. Any other view answers with the RGBA pixels themselves. */ + async readFrame( + id: number, + sinceGen: number, + target?: WebFrameMain | null, + ): Promise { const addon = this.ensureAddon(); if (!addon) { return null; } + if (target && this.sharedViews.has(id)) { + const frame = addon.readSharedFrame?.(id, sinceGen); + if (frame) { + return this.sendSharedFrame(addon, id, frame, target); + } + } + // Also the answer for a view whose render thread went back to read-back on its own. return addon.readFrame(id, sinceGen); } + private async sendSharedFrame( + addon: CompositorViewAddon, + id: number, + frame: NativeSharedFramePacket, + target: WebFrameMain, + ): Promise { + const meta: CompositorSharedFrameMeta = { + viewId: id, + gen: frame.gen, + width: frame.width, + height: frame.height, + footage: frame.footage ?? null, + footageProjective: frame.footageProjective ?? false, + }; + const api = this.sharedTextureApi(); + let imported: Electron.SharedTextureImported | undefined; + try { + if (!api) { + throw new Error("the sharedTexture API is gone"); + } + imported = api.importSharedTexture({ + textureInfo: { + pixelFormat: "rgba", + codedSize: { width: frame.width, height: frame.height }, + handle: { ntHandle: frame.handle }, + }, + // Chromium is done with the texture in every process: the slot may be written again. + allReferencesReleased: () => addon.releaseSharedFrame?.(id, frame.slot, frame.gen), + }); + await api.sendSharedTexture({ frame: target, importedSharedTexture: imported }, meta); + return { ...meta, shared: true }; + } catch (error) { + console.warn("[compositor-view] shared frame not delivered; reading frames back:", error); + this.stopSharedFrames(id); + if (!imported) { + addon.releaseSharedFrame?.(id, frame.slot, frame.gen); + } + return null; + } finally { + // This process's reference only: the renderer holds its own until it has drawn. + imported?.release(); + } + } + setParam(id: number, key: string, value: CompositorParamValue): void { const addon = this.ensureAddon(); if (!addon) { @@ -668,6 +784,7 @@ export class CompositorViewService { destroyView(id: number): void { const addon = this.ensureAddon(); this.rects.delete(id); + this.sharedViews.delete(id); if (!addon) { return; } diff --git a/electron/native/compositor-view/addon.d.ts b/electron/native/compositor-view/addon.d.ts index ee29cd80f..631d70fdf 100644 --- a/electron/native/compositor-view/addon.d.ts +++ b/electron/native/compositor-view/addon.d.ts @@ -41,6 +41,20 @@ export interface NativeFramePacket { footageProjective?: boolean; } +/** A preview frame left in a shared GPU texture instead of copied into RAM (Windows, hardware + * backend — see `setSharedFrames`). `handle` is the texture's NT handle the way Electron's + * `sharedTexture.importSharedTexture` takes it: 8 bytes, little-endian, valid in this process + * only. Its `slot` stays reserved until `releaseSharedFrame(id, slot, gen)`. */ +export interface NativeSharedFramePacket { + gen: number; + slot: number; + handle: Buffer; + width: number; + height: number; + footage?: number[] | null; + footageProjective?: boolean; +} + export interface ExportStats { frames: number; wallS: number; @@ -158,6 +172,17 @@ export interface CompositorViewAddon { * still frame): `null` comes back WITHOUT cloning the buffer or crossing IPC. * Pass `sinceGen = 0` to force delivery of the current frame. */ readFrame(id: number, sinceGen: number): NativeFramePacket | null; + /** Deliver this view's frames as shared GPU textures (`readSharedFrame`) rather than RAM + * pixels (`readFrame`). Returns `false` where that cannot work (not Windows, software + * backend): the view keeps reading back. Optional: an older `.node` predates it. */ + setSharedFrames?(id: number, enabled: boolean): boolean; + /** The latest frame left in the view's shared texture ring, if newer than `sinceGen`. + * `null` too while the view reads back to RAM. Throws the render thread's fatal error, + * like `readFrame`. */ + readSharedFrame?(id: number, sinceGen: number): NativeSharedFramePacket | null; + /** Chromium let go of frame `gen` in `slot`, in every process: the render thread may write + * that slot again. A no-op for a destroyed view. */ + releaseSharedFrame?(id: number, slot: number, gen: number): void; setParam(id: number, key: string, value: CompositorParamValue): void; setPlaying(id: number, playing: boolean): void; /** Seeks the view to source-media `seconds` for the active clip. */ diff --git a/electron/preload.ts b/electron/preload.ts index a754c99b6..feb3942df 100644 --- a/electron/preload.ts +++ b/electron/preload.ts @@ -1,4 +1,4 @@ -import { contextBridge, ipcRenderer, webUtils } from "electron"; +import { contextBridge, ipcRenderer, sharedTexture, webUtils } from "electron"; import type { NativeLinuxRecordingRequest } from "../src/lib/nativeLinuxRecording"; import type { NativeMacRecordingRequest } from "../src/lib/nativeMacRecording"; import type { NativeWindowsRecordingRequest } from "../src/lib/nativeWindowsRecording"; @@ -8,6 +8,7 @@ import type { AiEditionChatEvent, AiEditionMcpHostRequest, AiEditionMcpHostResponse, + CompositorSharedFrameMeta, } from "../src/native/contracts"; import { AI_EDITION_MCP_HOST_CHANNEL, @@ -34,6 +35,23 @@ const assetBaseUrl = assetBaseUrlArg ? assetBaseUrlArg.slice(ASSET_BASE_URL_ARG_ // so a synchronous read here saves the renderer's every-call IPC round-trip. const PLATFORM = process.platform; +// Preview frames the main process hands over as shared GPU textures (see +// `compositorViewService`). Electron gives up on a send after 1 s without a receiver, so it is +// registered here, at load, and forwards to whichever listener the page has set. +type CompositorFrameListener = (frame: VideoFrame, meta: CompositorSharedFrameMeta) => void; +let compositorFrameListener: CompositorFrameListener | null = null; +sharedTexture?.setSharedTextureReceiver(async ({ importedSharedTexture }, meta) => { + const frame = importedSharedTexture.getVideoFrame(); + try { + compositorFrameListener?.(frame, meta as CompositorSharedFrameMeta); + } finally { + // The page drew from its own copy (contextBridge clones the frame) and closed it. Closing + // ours and releasing the import is what lets the native ring write the slot again. + frame.close(); + importedSharedTexture.release(); + } +}); + contextBridge.exposeInMainWorld("electronAPI", { assetBaseUrl, @@ -85,6 +103,14 @@ contextBridge.exposeInMainWorld("electronAPI", { ipcRenderer.on("export:native-progress", handler); return () => ipcRenderer.off("export:native-progress", handler); }, + onCompositorFrame: (listener: CompositorFrameListener) => { + compositorFrameListener = listener; + return () => { + if (compositorFrameListener === listener) { + compositorFrameListener = null; + } + }; + }, invokeNativeBridge: (request: NativeBridgeRequest) => { return ipcRenderer.invoke(NATIVE_BRIDGE_CHANNEL, request) as Promise; }, diff --git a/src/native/compositorViewClient.ts b/src/native/compositorViewClient.ts index 793521086..66f63ce56 100644 --- a/src/native/compositorViewClient.ts +++ b/src/native/compositorViewClient.ts @@ -18,6 +18,8 @@ import type { CompositorExportResult, CompositorFramePacket, CompositorParamValue, + CompositorSharedFrameMeta, + CompositorSharedFrameReceipt, CompositorViewRect, CompositorViewResult, SegmentationSupport, @@ -97,18 +99,39 @@ export function setCompositorRect(id: number, rect: CompositorViewRect): Promise * {@link CompositorFramePacket} on a new frame, or `null` when the addon is absent, * no frame is ready yet, OR the caller already holds the current generation — the * idle path, where `null` returns without any buffer crossing IPC. Pass `sinceGen = 0` - * to force delivery of the current frame. */ + * to force delivery of the current frame. + * + * A view on shared textures answers with a {@link CompositorSharedFrameReceipt} instead: + * its frame already went to {@link subscribeCompositorSharedFrames}, ahead of this reply. */ export function readCompositorFrame( id: number, sinceGen: number, -): Promise { - return requireNativeBridgeData({ +): Promise { + return requireNativeBridgeData({ domain: "compositor", action: "readFrame", payload: { id, sinceGen }, }); } +/** Preview frames handed over as shared GPU textures. The listener draws `frame` and closes + * it. Returns the unsubscribe; without the Electron bridge (pure web, jsdom) nothing ever + * arrives. */ +export function subscribeCompositorSharedFrames( + listener: (frame: VideoFrame, meta: CompositorSharedFrameMeta) => void, +): () => void { + return window.electronAPI?.onCompositorFrame?.(listener) ?? (() => undefined); +} + +/** Back to read-back frames for `id`: a shared frame reached the canvas as nothing. */ +export function stopSharedCompositorFrames(id: number): Promise<{ ok: true }> { + return requireNativeBridgeData<{ ok: true }>({ + domain: "compositor", + action: "stopSharedFrames", + payload: { id }, + }); +} + export function setCompositorParam( id: number, key: string, diff --git a/src/native/contracts.ts b/src/native/contracts.ts index 432a6ca69..d13a95f5c 100644 --- a/src/native/contracts.ts +++ b/src/native/contracts.ts @@ -167,6 +167,24 @@ export interface CompositorFramePacket { footageProjective?: boolean; } +/** What travels with a preview frame handed over as a shared GPU texture (Windows): all a + * {@link CompositorFramePacket} says but the pixels, plus the view the frame belongs to. The + * main process sends it alongside the texture, to `electronAPI.onCompositorFrame`. */ +export interface CompositorSharedFrameMeta { + viewId: number; + gen: number; + width: number; + height: number; + footage: number[] | null; + footageProjective: boolean; +} + +/** `readFrame`'s answer for a frame sent as a shared texture. The texture reached + * `onCompositorFrame` before this reply did, so only the generation is news here. */ +export interface CompositorSharedFrameReceipt extends CompositorSharedFrameMeta { + shared: true; +} + /** Un clip de la timeline pour l'export multiclip natif (fichiers screen+webcam + trim). */ export interface CompositorClipInput { screenPath: string; @@ -820,6 +838,14 @@ export type NativeBridgeRequest = payload: { id: number; sinceGen: number }; requestId?: string; } + | { + domain: "compositor"; + /** Back to read-back frames: a shared frame reached the canvas as nothing, i.e. + * Chromium could not open the texture. */ + action: "stopSharedFrames"; + payload: { id: number }; + requestId?: string; + } | { domain: "compositor"; action: "setParam"; diff --git a/src/native/hooks/useNativeCompositorView.test.ts b/src/native/hooks/useNativeCompositorView.test.ts index 55b89174b..ac182aceb 100644 --- a/src/native/hooks/useNativeCompositorView.test.ts +++ b/src/native/hooks/useNativeCompositorView.test.ts @@ -15,12 +15,18 @@ import { renderHook, waitFor } from "@testing-library/react"; import type { RefObject } from "react"; import { beforeEach, describe, expect, it, vi } from "vitest"; +import type { CompositorSharedFrameMeta } from "../contracts"; + +type SharedFrameListener = (frame: VideoFrame, meta: CompositorSharedFrameMeta) => void; const mocks = vi.hoisted(() => ({ createCompositorView: vi.fn(), readCompositorFrame: vi.fn(), destroyCompositorView: vi.fn(), setCompositorRect: vi.fn(async () => undefined), + stopSharedCompositorFrames: vi.fn(async () => ({ ok: true })), + /** What the preload would call with each frame sent as a shared texture. */ + sharedListener: null as SharedFrameListener | null, })); vi.mock("../compositorViewClient", () => ({ @@ -30,6 +36,15 @@ vi.mock("../compositorViewClient", () => ({ setCompositorParam: vi.fn(), setCompositorPlaying: vi.fn(), setCompositorRect: mocks.setCompositorRect, + stopSharedCompositorFrames: mocks.stopSharedCompositorFrames, + subscribeCompositorSharedFrames: (listener: SharedFrameListener) => { + mocks.sharedListener = listener; + return () => { + if (mocks.sharedListener === listener) { + mocks.sharedListener = null; + } + }; + }, })); import { useNativeCompositorView } from "./useNativeCompositorView"; @@ -48,22 +63,50 @@ globalThis.ResizeObserver = class { } } as unknown as typeof ResizeObserver; -/** A canvas with a stubbed 2D context — jsdom has none, and the pull loop bails without it. */ -function stubCanvasRef(): RefObject { +/** A canvas with a stubbed 2D context — jsdom has none, and the pull loop bails without it. + * `centreAlpha` is what reading back the middle pixel answers: opaque, as a composed frame + * always is, unless a test says otherwise. */ +function stubCanvasRef(centreAlpha = 255): RefObject { const canvas = document.createElement("canvas"); - canvas.getContext = vi.fn(() => ({ + const ctx = { drawImage: vi.fn(), putImageData: vi.fn(), - })) as unknown as HTMLCanvasElement["getContext"]; + getImageData: vi.fn(() => ({ data: new Uint8ClampedArray([0, 0, 0, centreAlpha]) })), + }; + canvas.getContext = vi.fn(() => ctx) as unknown as HTMLCanvasElement["getContext"]; return { current: canvas }; } +function context(ref: RefObject) { + return ref.current?.getContext("2d") as unknown as { + drawImage: ReturnType; + getImageData: ReturnType; + }; +} + +function sharedMeta(overrides: Partial = {}): CompositorSharedFrameMeta { + return { + viewId: 7, + gen: 3, + width: 4, + height: 2, + footage: null, + footageProjective: false, + ...overrides, + }; +} + +function fakeVideoFrame() { + return { close: vi.fn() } as unknown as VideoFrame & { close: ReturnType }; +} + const DEVICE_FAILURE = "this display adapter has no D3D11 video decoder (0x887A0004). OpenScreen decodes every preview and export frame with D3D11VA"; describe("useNativeCompositorView", () => { beforeEach(() => { vi.clearAllMocks(); + mocks.sharedListener = null; }); it("surfaces the native message when the render thread dies", async () => { @@ -317,4 +360,86 @@ describe("useNativeCompositorView", () => { resolveSecond({ id: 8 }); await waitFor(() => expect(result.current.viewId).toBe(8)); }); + + describe("frames handed over as shared GPU textures", () => { + it("draws the frame on the canvas, sized to it, and closes it", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame.mockResolvedValue(null); + const ref = stubCanvasRef(); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + + const frame = fakeVideoFrame(); + mocks.sharedListener?.(frame, sharedMeta()); + + expect(context(ref).drawImage).toHaveBeenCalledWith(frame, 0, 0); + expect(ref.current?.width).toBe(4); + expect(ref.current?.height).toBe(2); + expect(ref.current?.dataset.painted).toBe("true"); + expect(frame.close).toHaveBeenCalled(); + expect(mocks.stopSharedCompositorFrames).not.toHaveBeenCalled(); + }); + + it("closes a frame meant for another view without drawing it", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame.mockResolvedValue(null); + const ref = stubCanvasRef(); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + + const frame = fakeVideoFrame(); + mocks.sharedListener?.(frame, sharedMeta({ viewId: 8 })); + + expect(context(ref).drawImage).not.toHaveBeenCalled(); + expect(frame.close).toHaveBeenCalled(); + }); + + // Chromium opens the texture in its own GPU process. One it cannot open — a hybrid + // laptop's other adapter — draws as nothing, and the preview would stay empty. + it("goes back to read-back frames when the first one lands as nothing", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame.mockResolvedValue(null); + const ref = stubCanvasRef(0); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta()); + + expect(mocks.stopSharedCompositorFrames).toHaveBeenCalledWith(7); + }); + + it("checks only the view's first frame, not every frame", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame.mockResolvedValue(null); + const ref = stubCanvasRef(); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta({ gen: 3 })); + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta({ gen: 4 })); + + // A readback stalls the GPU pipeline: one per view, not one per frame. + expect(context(ref).getImageData).toHaveBeenCalledTimes(1); + }); + + it("takes a receipt as the generation to ask after, with nothing to draw", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame + .mockResolvedValueOnce({ ...sharedMeta({ gen: 5 }), shared: true }) + .mockResolvedValue(null); + const ref = stubCanvasRef(); + renderHook(() => useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } })); + + await waitFor(() => expect(mocks.readCompositorFrame).toHaveBeenCalledWith(7, 5)); + expect(context(ref).drawImage).not.toHaveBeenCalled(); + }); + }); }); diff --git a/src/native/hooks/useNativeCompositorView.ts b/src/native/hooks/useNativeCompositorView.ts index 416b57c77..a138c8b92 100644 --- a/src/native/hooks/useNativeCompositorView.ts +++ b/src/native/hooks/useNativeCompositorView.ts @@ -5,13 +5,17 @@ * rect (measured via ResizeObserver + window resize/scroll, rAF-coalesced * — the exact same sync machinery as before, repurposed: it now drives * the offscreen render-target resolution instead of a window position). - * 2. Polls `readCompositorFrame` on every other rAF tick (~30fps), passing the - * generation it last painted. Native returns a self-describing packet - * (`{ gen, width, height, data }`) ONLY when a newer frame exists — otherwise - * `null`, and the canvas is left untouched. So while the preview sits still - * (paused editing) nothing is cloned, sent over IPC, or repainted. The canvas - * drawing buffer is sized from the packet's own dims, so pixels and canvas - * can never drift apart. + * 2. Polls `readCompositorFrame` on rAF ticks, passing the generation it last + * painted. Native answers ONLY when a newer frame exists — otherwise `null`, and + * the canvas is left untouched, so while the preview sits still (paused editing) + * nothing is cloned, sent over IPC, or repainted. A new frame comes one of two ways: + * - as a shared GPU texture (Windows), sent to `subscribeCompositorSharedFrames` + * ahead of the reply, which then only names its generation. Nothing is copied + * through RAM, so it is pulled every tick; + * - as RGBA pixels (`{ gen, width, height, data }`) everywhere else, pulled every + * other tick (~30fps) to bound the copies. + * Either way the canvas drawing buffer is sized from the frame's own dims, so pixels + * and canvas can never drift apart. * * Every native-bridge call is wrapped in a try/catch that swallows + warns, * because the renderer may run without the bridge (pure web `npm run dev`, @@ -28,6 +32,8 @@ import { setCompositorParam, setCompositorPlaying, setCompositorRect, + stopSharedCompositorFrames, + subscribeCompositorSharedFrames, } from "../compositorViewClient"; import type { CompositorParamValue, CompositorViewRect } from "../contracts"; import { publishFootageQuad } from "../footageQuadStore"; @@ -78,9 +84,9 @@ function safelyCall(label: string, call: () => Promise) { } } -/** Throttle the rAF pull loop to roughly 30fps: process every other animation - * frame. Keeps IPC + GPU readback + putImageData cheap on high-refresh - * displays (120/144 Hz) without changing perceived preview smoothness. */ +/** Read-back frames are pulled on every other animation frame (~30fps on 60 Hz): each one + * is a GPU readback, a structured clone across IPC and a canvas upload. Shared-texture + * frames copy nothing through RAM and are pulled every tick. */ const PULL_LOOP_TICK_DIVISOR = 2; export function useNativeCompositorView( @@ -189,17 +195,68 @@ export function useNativeCompositorView( // the next frame's. The generation actually on the canvas lets an older one be dropped, // instead of bringing back its pixels and its buffer size until native sends another. let paintedGen = 0; + // Frames arrive as shared GPU textures (see `subscribeCompositorSharedFrames`): nothing + // to bound, so the pull loop stops skipping ticks. + let sharedTransport = false; + // Whether this view's first shared frame was checked to have actually landed. + let sharedChecked = false; - /** rAF pull loop: throttle to ~30fps and repaint ONLY when native reports a - * newer generation. The returned packet is self-describing (`gen` + dims + - * pixels), so the canvas is sized from the packet — pixels and canvas can - * never drift out of sync. Runs off the main thread so UI stays at 60/120fps. */ + /** Puts a frame on the canvas. The buffer takes the frame's size HERE, in the same task + * as the draw that refills it, never when the canvas box changes: anything awaited + * between the two (the rect's trip to native, `createImageBitmap`) is a frame the + * browser presents empty. Meanwhile CSS stretches the previous frame over the new box. + * An Auto format reshapes that box on every padding tick, so an early resize blinked + * the footage out and back while the slider moved. */ + const paint = (gen: number, width: number, height: number, draw: () => void): boolean => { + if (disposed || gen < paintedGen) { + return false; + } + setBufferSize(width, height); + draw(); + paintedGen = gen; + markPainted(); + return true; + }; + + // Frames the main process hands over as shared GPU textures land here, ahead of the + // `readCompositorFrame` reply that names their generation. + const unsubscribeShared = subscribeCompositorSharedFrames((frame, meta) => { + try { + const id = viewIdRef.current; + const ctx = canvas.getContext("2d"); + if (disposed || id == null || meta.viewId !== id || !ctx) { + return; + } + sharedTransport = true; + lastGen = Math.max(lastGen, meta.gen); + publishFootageQuad(meta.footage, meta.footageProjective); + const fresh = canvas.dataset.painted === undefined; + const drawn = paint(meta.gen, meta.width, meta.height, () => ctx.drawImage(frame, 0, 0)); + if (drawn && fresh && !sharedChecked) { + sharedChecked = true; + // Chromium opens the texture in its own GPU process, and one it cannot open (a + // hybrid laptop's other adapter) draws as nothing at all. The composed frame is + // opaque everywhere, so a transparent pixel on this fresh canvas is that failure. + if (ctx.getImageData(meta.width >> 1, meta.height >> 1, 1, 1).data[3] === 0) { + sharedTransport = false; + safelyCall("stopSharedFrames", () => stopSharedCompositorFrames(id)); + } + } + noteUiProbePreviewFrame(); + } finally { + frame.close(); + } + }); + + /** rAF pull loop: repaint ONLY when native reports a newer generation. A pixel packet + * is self-describing (`gen` + dims + pixels), so the canvas is sized from the packet — + * pixels and canvas can never drift out of sync. */ const pullLoop = () => { pullRafHandle = requestAnimationFrame(pullLoop); if (disposed || inFlight) { return; } - pullTick = (pullTick + 1) % PULL_LOOP_TICK_DIVISOR; + pullTick = (pullTick + 1) % (sharedTransport ? 1 : PULL_LOOP_TICK_DIVISOR); if (pullTick !== 0) { return; } @@ -220,6 +277,11 @@ export function useNativeCompositorView( if (disposed || !frame) { return; } + if ("shared" in frame) { + // Already drawn by the shared-frame listener: only the generation is news. + lastGen = Math.max(lastGen, frame.gen); + return; + } const { gen, width, height, data } = frame; // Defensive: the packet's byte count must match its own declared // dimensions. A mismatch would corrupt the image silently — bail. @@ -239,30 +301,15 @@ export function useNativeCompositorView( data.byteLength, ); const image = new ImageData(pixels, width, height); - // The buffer takes the packet's size HERE, in the same task as the draw that - // refills it, never when the canvas box changes: anything awaited between the - // two (the rect's trip to native, `createImageBitmap`) is a frame the browser - // presents empty. Meanwhile CSS stretches the previous frame over the new box. - // An Auto format reshapes that box on every padding tick, so an early resize - // blinked the footage out and back while the slider moved. - const paint = (draw: () => void) => { - if (disposed || gen < paintedGen) { - return; - } - setBufferSize(width, height); - draw(); - paintedGen = gen; - markPainted(); - }; // `createImageBitmap` decodes off the main thread (keeps UI at 60/120fps) // and snapshots `image`, so the view can be released after; `putImageData` // is the synchronous fallback if bitmap creation is unavailable. createImageBitmap(image) .then((bitmap) => { - paint(() => ctx.drawImage(bitmap, 0, 0)); + paint(gen, width, height, () => ctx.drawImage(bitmap, 0, 0)); bitmap.close(); }) - .catch(() => paint(() => ctx.putImageData(image, 0, 0))); + .catch(() => paint(gen, width, height, () => ctx.putImageData(image, 0, 0))); // Advance only after a successful, validated frame — so a dropped/ // malformed packet is retried rather than silently skipped. lastGen = gen; @@ -326,6 +373,7 @@ export function useNativeCompositorView( return () => { disposed = true; + unsubscribeShared(); publishFootageQuad(null); if (rectRafHandle !== 0) { cancelAnimationFrame(rectRafHandle); diff --git a/technical-documentation/architecture/preview.md b/technical-documentation/architecture/preview.md index 0a83ac5fd..9b7c34ed9 100644 --- a/technical-documentation/architecture/preview.md +++ b/technical-documentation/architecture/preview.md @@ -171,16 +171,45 @@ The contract, with the invariant on the consumer side: a mismatch would corrupt the image silently. The consumer never assumes a size — every draw is preceded by a resize of the canvas's drawing buffer to the packet's declared `width`/`height`. -5. **Pull loop cadence.** The renderer pulls on every other rAF tick (`PULL_LOOP_TICK_DIVISOR = 2`, - [`useNativeCompositorView.ts:70`](../../src/native/hooks/useNativeCompositorView.ts:70)), - so IPC + GPU readback + `putImageData` run at roughly 30 fps on 60/120 Hz - displays without changing perceived smoothness. +5. **Pull loop cadence.** Read-back frames are pulled on every other rAF tick + (`PULL_LOOP_TICK_DIVISOR = 2`), shared-texture frames (below) on every tick. Each + read-back frame is a GPU readback, a structured clone across IPC and a canvas upload, + and the tick counter does not advance while a read is in flight: an 8 MB frame whose + round trip passes 16.7 ms is pulled one tick in three, about 20 fps. The renderer-side wrapper mirrors this verbatim in the Electron main process -([`compositorViewService.ts:339`](../../electron/native-bridge/services/compositorViewService.ts:339)), +([`compositorViewService.ts`](../../electron/native-bridge/services/compositorViewService.ts)), so when the addon is absent the IPC layer returns `null` too — the renderer never has to special-case "addon missing". +### Shared textures (Windows) + +On Windows' hardware backend the pixels never leave the GPU. Copying them through RAM +was the preview's bottleneck, not the compositor: at the same ~57 composed frames per +second, read-back reached the canvas at 20-26 fps and kept 54-57 % of the renderer's main +thread busy (measurements in +[engineering/rendering-performance.md](../engineering/rendering-performance.md#preview-transport--2026-10-02)). + +- **Native side.** The render thread copies each composed frame into one of four shared + D3D11 textures (`shared_frames.rs`, NT handles, no keyed mutex), waits for the GPU to + finish the copy, and publishes `{ gen, slot, handle }`. `SlotBook` tracks which slot + holds the ready frame and which ones Chromium still holds; a frame nobody took gives + its slot back, and a slot whose release never came is reclaimed after 1 s. +- **Main process.** `readFrame` takes the frame (`readSharedFrame`), imports it with + `sharedTexture.importSharedTexture`, sends it with `sharedTexture.sendSharedTexture` to + the frame that asked, drops its own reference and answers with a receipt + (`{ …meta, shared: true }`, no pixels). `allReferencesReleased` returns the slot + (`releaseSharedFrame(id, slot, gen)`) once Chromium is done with it in every process. +- **Renderer.** The preload's `setSharedTextureReceiver` hands the frame to + `electronAPI.onCompositorFrame`, ahead of the receipt; the hook draws the `VideoFrame` + with `drawImage` and closes it. The pixels are byte-identical to read-back. +- **Fallbacks, all to read-back.** Not Windows, the software backend, Chromium without GPU + compositing, or `OPENSCREEN_PREVIEW_READBACK=1` never enable it. A failed import or send + turns it off for the view, and so does a first frame that lands transparent (a texture + Chromium could not open, e.g. on another adapter): the composed frame is cleared opaque, + so a transparent pixel can only be that. The render thread republishes the current + frame on every switch, so the canvas never waits for something to move. + ## Playback sync The native view runs its own clock while playing, so the renderer only pushes a @@ -307,6 +336,9 @@ was thrown away. measured at the bench in [engineering/rendering-performance.md](../engineering/rendering-performance.md) stay below the threshold in practice, but no systematic measurement exists. +- **Shared textures are Windows-only.** macOS (an `IOSurface`-backed Metal texture) and + Linux (a dmabuf exported from Vulkan) still read back: `sharedTexture` imports both, the + native halves are not written. - **Add-on absent = blank frame.** When `compositor_view.node` is missing (development with the addon not yet built, or a packaged build for an unsupported architecture) the overlay renders no pixels: only the DOM/CSS diff --git a/technical-documentation/engineering/rendering-performance.md b/technical-documentation/engineering/rendering-performance.md index 6fa913214..33de33cb2 100644 --- a/technical-documentation/engineering/rendering-performance.md +++ b/technical-documentation/engineering/rendering-performance.md @@ -212,6 +212,52 @@ cliffs above: fine for a real GPU, heavy for a per-pixel loop on the CPU rasteri > of the aurora export moves by up to 22/255 between frames 0 and 300, against 4/255 for the > still one. +## Preview transport — 2026-10-02 + +**The preview was bound by the trip of its pixels to the canvas, not by the compositor.** +The compositor composed ~57 frames per second of a 1080p60 recording in both arms below; what +reached the canvas, and what it cost the UI, depended only on how the frames travelled. + +Two throwaway Electron benches (not kept in the tree), run in a hidden window on a desktop — +Ryzen 7 5800X, GeForce RTX 4070 Ti, Windows 11 — not the reference laptop: the copies are CPU +and memory work, so an iGPU laptop pays more for read-back, not less. A hidden window +throttles nothing here (`backgroundThrottling: false`, ticks on timers rather than rAF), but +it renders no React and decodes nothing alongside, so the absolutes are a floor, not the app. +Main-thread load is measured as the gaps in a back-to-back `MessageChannel` ping loop +(`setImmediate` in the main process). + +**IPC alone** (synthetic buffers through `ipcRenderer.invoke` from a sandboxed preload and +`contextBridge`, as the app does), two runs: + +| frame | MB | round trip p50 / p90 | renderer main thread busy at 30/s | main process busy at 30/s | +|---|---:|---:|---:|---:| +| 1920×1080 | 8.3 | 24 / 31 ms | 54-55 % | 31-37 % | +| 1650×930 | 6.1 | 18-20 / 22-25 ms | 37-39 % | 19-21 % | +| 1100×620 | 2.7 | 8-10 / 10-13 ms | 17-21 % | 9-12 % | + +**End to end** (the built addon, a real 32 s 1080p60 recording with its 1080p30 webcam, the +hook's pull loop on 60 Hz ticks), two runs per arm: + +| preview | arm | frames on the canvas /s | composed /s | renderer busy | main busy | main import + send p50 / p90 | +|---|---|---:|---:|---:|---:|---:| +| 1920×1080 | read-back | 20-21 | 57-59 | 57 % | 37-38 % | — | +| 1920×1080 | shared texture | 53-54 | 56-57 | 1.8-2.0 % | 0.1-0.5 % | 0.67 / 0.97 ms | +| 1650×928 | read-back | 26 | 57 | 54-56 % | 35 % | — | +| 1650×928 | shared texture | 52.5-53 | 57-58 | 2.0-2.6 % | 0.6-1.7 % | 0.67-0.73 / 1.02-1.18 ms | + +The same paused frame, read back and drawn from the shared texture, differs in **0 of +8 294 400 bytes** at 1080p (0 of 6 124 800 at 1650×928). + +- **Read-back stalls on its own round trip.** At 18-24 ms per 8 MB frame, the hook's pull loop + (one read in flight, one tick in two) gets one frame every three ticks: ~20 fps on 60 Hz. +- **The transport was most of the renderer's load.** The ~44 % main-thread block measured + during playback on 2026-09-29 (`VirtualPreview.tsx`, another machine) is the order of this + cost alone. Every copy, structured clone and allocation landed on the thread React paints + from. +- **Shared textures leave ~7 % of the composed frames undrawn.** Two ~60 Hz clocks — the + render thread and the pull loop — beat against each other; a tick that finds nothing new + is followed by one that finds two, and only the newer is drawn. + ## The macOS export path — 2026-09-03/04 Everything above is the Windows reference machine. This section is a **different machine and a different pipeline**: a Mac mini M1 (8 cores, 8 GiB, macOS 26.5), Metal compositor, VideoToolbox on both ends. Nothing here transfers to the Windows numbers, and the reverse held too — of the three levers that mattered on Windows and Linux, **none applied here**. From dfc70487b984cc2f5232288eff428e603c73c7c8 Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Fri, 2 Oct 2026 10:11:20 +0200 Subject: [PATCH 2/4] perf(preview): pull frames by the clock, not by display ticks Measured in the editor on a 280 Hz display, pulling shared frames on every animation frame meant 280 IPC round trips a second, idle or not: the main process sat at ~11 % of a core with nothing playing. Counting ticks also under-pulled read-back frames on 60 Hz, where a tick spent waiting on a slow round trip pushed the next pull a whole tick further. Pulls are now spaced by time: every 8 ms while shared frames keep coming (within 250 ms of the last one, so a pull that lands between two frames does not end the fast cadence), ~30 a second otherwise. In the editor: idle main process ~11 % -> ~2 %, all ~500 frames of the 8 s take still drawn during playback. --- .../hooks/useNativeCompositorView.test.ts | 120 ++++++++++++++++++ src/native/hooks/useNativeCompositorView.ts | 46 +++++-- 2 files changed, 153 insertions(+), 13 deletions(-) diff --git a/src/native/hooks/useNativeCompositorView.test.ts b/src/native/hooks/useNativeCompositorView.test.ts index ac182aceb..7d2b0e756 100644 --- a/src/native/hooks/useNativeCompositorView.test.ts +++ b/src/native/hooks/useNativeCompositorView.test.ts @@ -442,4 +442,124 @@ describe("useNativeCompositorView", () => { expect(context(ref).drawImage).not.toHaveBeenCalled(); }); }); + + // The display sets the rAF rate and the recording the frame rate. Counting ticks pulled 280 + // times a second on a 280 Hz display, idle or not. + describe("pull cadence, by the clock rather than by ticks", () => { + /** rAF driven by hand, at `periodMs`: callbacks run with the timestamps a display of that + * rate would give them. */ + function manualFrames() { + let queue: FrameRequestCallback[] = []; + vi.stubGlobal("requestAnimationFrame", (callback: FrameRequestCallback) => { + queue.push(callback); + return queue.length; + }); + vi.stubGlobal("cancelAnimationFrame", () => undefined); + return async (durationMs: number, periodMs: number, start = 0) => { + for (let t = start; t < start + durationMs; t += periodMs) { + const due = queue; + queue = []; + for (const callback of due) { + callback(t); + } + // Let each pull's reply land before the next tick, as it would between frames. + for (let flush = 0; flush < 4; flush++) { + await Promise.resolve(); + } + } + }; + } + + async function mountedView() { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + const ref = stubCanvasRef(); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + mocks.readCompositorFrame.mockClear(); + } + + it("pulls ~30 times a second while nothing new comes, whatever the display rate", async () => { + const run = manualFrames(); + try { + mocks.readCompositorFrame.mockResolvedValue(null); + await mountedView(); + + await run(1000, 1000 / 280); + + // One pull every 9 ticks of 3.6 ms: 32 in the second, against 280 counting ticks. + const pulls = mocks.readCompositorFrame.mock.calls.length; + expect(pulls).toBeGreaterThanOrEqual(28); + expect(pulls).toBeLessThanOrEqual(33); + } finally { + vi.unstubAllGlobals(); + } + }); + + it("pulls every 8 ms or so while shared frames keep coming, and slows down once they stop", async () => { + const run = manualFrames(); + try { + let gen = 0; + mocks.readCompositorFrame.mockImplementation(async () => ({ + ...sharedMeta({ gen: ++gen }), + shared: true, + })); + await mountedView(); + // The first frame marks the transport as shared, as the preload's listener does. + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta({ gen: 1 })); + + await run(1000, 1000 / 280); + const flowing = mocks.readCompositorFrame.mock.calls.length; + expect(flowing).toBeGreaterThanOrEqual(100); + expect(flowing).toBeLessThanOrEqual(150); + + // Playback stops: nothing new comes any more. Past the flowing window the loop is + // back to ~30 pulls a second. + mocks.readCompositorFrame.mockReset(); + mocks.readCompositorFrame.mockResolvedValue(null); + await run(400, 1000 / 280, 1000); + mocks.readCompositorFrame.mockClear(); + await run(500, 1000 / 280, 1400); + expect(mocks.readCompositorFrame.mock.calls.length).toBeLessThanOrEqual(17); + } finally { + vi.unstubAllGlobals(); + } + }); + + // Read-back frames bound the copies: the fast cadence is for shared textures only. + it("keeps read-back frames at ~30 pulls a second even while they keep coming", async () => { + const run = manualFrames(); + vi.stubGlobal( + "ImageData", + class { + constructor( + public data: Uint8ClampedArray, + public width: number, + public height: number, + ) {} + }, + ); + vi.stubGlobal( + "createImageBitmap", + vi.fn(async () => ({ close: vi.fn() })), + ); + try { + let gen = 0; + mocks.readCompositorFrame.mockImplementation(async () => ({ + gen: ++gen, + width: 2, + height: 1, + data: new Uint8Array(8), + })); + await mountedView(); + + await run(1000, 1000 / 280); + + expect(mocks.readCompositorFrame.mock.calls.length).toBeLessThanOrEqual(33); + } finally { + vi.unstubAllGlobals(); + } + }); + }); }); diff --git a/src/native/hooks/useNativeCompositorView.ts b/src/native/hooks/useNativeCompositorView.ts index a138c8b92..d42fcfa89 100644 --- a/src/native/hooks/useNativeCompositorView.ts +++ b/src/native/hooks/useNativeCompositorView.ts @@ -11,9 +11,9 @@ * nothing is cloned, sent over IPC, or repainted. A new frame comes one of two ways: * - as a shared GPU texture (Windows), sent to `subscribeCompositorSharedFrames` * ahead of the reply, which then only names its generation. Nothing is copied - * through RAM, so it is pulled every tick; - * - as RGBA pixels (`{ gen, width, height, data }`) everywhere else, pulled every - * other tick (~30fps) to bound the copies. + * through RAM, so while frames keep coming it is pulled every 8 ms; + * - as RGBA pixels (`{ gen, width, height, data }`) everywhere else, pulled ~30 + * times a second to bound the copies. * Either way the canvas drawing buffer is sized from the frame's own dims, so pixels * and canvas can never drift apart. * @@ -84,10 +84,24 @@ function safelyCall(label: string, call: () => Promise) { } } -/** Read-back frames are pulled on every other animation frame (~30fps on 60 Hz): each one - * is a GPU readback, a structured clone across IPC and a canvas upload. Shared-texture - * frames copy nothing through RAM and are pulled every tick. */ -const PULL_LOOP_TICK_DIVISOR = 2; +/** Least time between two pulls, by the clock rather than in animation frames: the display + * sets the rAF rate, the recording sets the frame rate, and the two have nothing to do with + * each other. Counting ticks pulled 280 times a second on a 280 Hz display, idle or not, and + * under-pulled read-back frames on 60 Hz, where a tick lost to a slow round trip pushed the + * next pull a whole tick further. + * + * While shared-texture frames keep coming, a pull every 8 ms catches each one within half a + * 60 fps frame. Anything else — read-back frames, each a GPU readback, a structured clone + * across IPC and a canvas upload, or a view with nothing new — is pulled ~30 times a second. */ +const PULL_INTERVAL_FLOWING_MS = 8; +const PULL_INTERVAL_MS = 33; +/** Frames still count as coming this long after the last one. A pull that lands between two + * frames finds nothing new, and taking that for the end of playback dropped every next + * pull to the slow interval: ~40 frames a second drawn out of ~60. */ +const PULL_FLOWING_WINDOW_MS = 250; +/** rAF timestamps jitter around the display's period: without it, a 33 ms interval on a 60 Hz + * display would land on the third tick as often as the second. */ +const PULL_INTERVAL_SLACK_MS = 1.5; export function useNativeCompositorView( canvasRef: RefObject, @@ -122,7 +136,7 @@ export function useNativeCompositorView( let rectRafHandle = 0; let pullRafHandle = 0; - let pullTick = 0; + let lastPullAt = Number.NEGATIVE_INFINITY; let lastRect: CompositorViewRect | null = null; let disposed = false; // Fresh view (source or enablement changed) → the previous view's fatal error @@ -195,9 +209,10 @@ export function useNativeCompositorView( // the next frame's. The generation actually on the canvas lets an older one be dropped, // instead of bringing back its pixels and its buffer size until native sends another. let paintedGen = 0; - // Frames arrive as shared GPU textures (see `subscribeCompositorSharedFrames`): nothing - // to bound, so the pull loop stops skipping ticks. + // Frames arrive as shared GPU textures (see `subscribeCompositorSharedFrames`), and one + // came lately: while both hold, the loop pulls at its fast interval. let sharedTransport = false; + let lastFrameAt = Number.NEGATIVE_INFINITY; // Whether this view's first shared frame was checked to have actually landed. let sharedChecked = false; @@ -251,13 +266,14 @@ export function useNativeCompositorView( /** rAF pull loop: repaint ONLY when native reports a newer generation. A pixel packet * is self-describing (`gen` + dims + pixels), so the canvas is sized from the packet — * pixels and canvas can never drift out of sync. */ - const pullLoop = () => { + const pullLoop = (now: number) => { pullRafHandle = requestAnimationFrame(pullLoop); if (disposed || inFlight) { return; } - pullTick = (pullTick + 1) % (sharedTransport ? 1 : PULL_LOOP_TICK_DIVISOR); - if (pullTick !== 0) { + const flowing = sharedTransport && now - lastFrameAt < PULL_FLOWING_WINDOW_MS; + const interval = flowing ? PULL_INTERVAL_FLOWING_MS : PULL_INTERVAL_MS; + if (now - lastPullAt < interval - PULL_INTERVAL_SLACK_MS) { return; } const id = viewIdRef.current; @@ -269,9 +285,13 @@ export function useNativeCompositorView( return; } inFlight = true; + lastPullAt = now; readCompositorFrame(id, lastGen) .then((frame) => { inFlight = false; + if (frame) { + lastFrameAt = now; + } // `null` = nothing newer than `lastGen` (idle path — no pixels // crossed IPC) OR no frame yet. Either way, leave the canvas as-is. if (disposed || !frame) { From 773e41d31f398868f2510c1e089ec83e9497be5d Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Fri, 2 Oct 2026 10:37:30 +0200 Subject: [PATCH 3/4] fix(preview): republish a frame that found no free slot, and drop the fast cadence on read-back Two cases CodeRabbit found in the shared-texture transport: - A frame composed while every ring slot was still held was skipped. In playback the next one replaces it, but a frame composed once in pause (a seek, a parameter change) never reached the canvas until the next change. The render target keeps it, so it is now published as soon as a slot is free, without composing it again. - A view that fell back to read-back after a failed import or send kept the 8 ms pull cadence meant for shared textures, on frames that each cost a readback and a structured clone. A read-back frame now ends it. --- crates/compositor/src/live.rs | 21 ++++++--- .../hooks/useNativeCompositorView.test.ts | 44 +++++++++++++++++++ src/native/hooks/useNativeCompositorView.ts | 6 ++- 3 files changed, 65 insertions(+), 6 deletions(-) diff --git a/crates/compositor/src/live.rs b/crates/compositor/src/live.rs index cbed49464..8fd6a29e0 100644 --- a/crates/compositor/src/live.rs +++ b/crates/compositor/src/live.rs @@ -1472,6 +1472,10 @@ unsafe fn render_thread( let mut scene_applied = false; // Textures partagées de la preview, créées au premier besoin (voir `publish_shared`). let mut ring = Ring::default(); + // Une frame composée qui n'a trouvé aucune case libre dans l'anneau : le RT la garde, elle + // repart dès qu'une case se libère. Sans ça, une frame composée en pause (seek, réglage) + // restait invisible jusqu'au changement suivant — en lecture, la suivante la remplace. + let mut publish_pending = false; while !shared.stop.load(Ordering::SeqCst) { // Le transport vient de changer : le consommateur doit recevoir la frame courante @@ -1834,13 +1838,19 @@ unsafe fn render_thread( stepped = true; } - if stepped || first { + if stepped || first || publish_pending { if pw > 0 && ph > 0 { match publish_shared(&shared, &gpu, &comp, &mut ring) { - SharedPublish::Published => first = false, - // Chromium tient toutes les cases : la frame est sautée, et la boucle ne - // tourne pas à vide le temps qu'il en relâche une. - SharedPublish::NoFreeSlot => std::thread::sleep(Duration::from_millis(2)), + SharedPublish::Published => { + first = false; + publish_pending = false; + } + // Chromium tient toutes les cases : la frame attend dans le RT, et la boucle + // ne tourne pas à vide le temps qu'il en relâche une. + SharedPublish::NoFreeSlot => { + publish_pending = true; + std::thread::sleep(Duration::from_millis(2)); + } // Step complet : `compose_frame` (déjà appelé par `step`/`present_frame`/ // `recompose`) a rastérisé le RT à la géométrie de sortie ramenée au panneau. // On lit ce RT DIRECTEMENT à sa résolution de rendu (`readback_direct` : copy @@ -1867,6 +1877,7 @@ unsafe fn render_thread( *slot = Some((next_gen, rw, rh, rgba, comp.footage_quad())); } first = false; + publish_pending = false; } Err(e) => { eprintln!("[live] readback_direct: {e:#}"); diff --git a/src/native/hooks/useNativeCompositorView.test.ts b/src/native/hooks/useNativeCompositorView.test.ts index 7d2b0e756..4949bb5b7 100644 --- a/src/native/hooks/useNativeCompositorView.test.ts +++ b/src/native/hooks/useNativeCompositorView.test.ts @@ -527,6 +527,50 @@ describe("useNativeCompositorView", () => { } }); + // The service turns shared textures off for a view whose import or send failed: its next + // frames are read-back pixels, and the fast cadence must not outlive the transport. + it("drops back to ~30 pulls a second once the view falls back to read-back", async () => { + const run = manualFrames(); + vi.stubGlobal( + "ImageData", + class { + constructor( + public data: Uint8ClampedArray, + public width: number, + public height: number, + ) {} + }, + ); + vi.stubGlobal( + "createImageBitmap", + vi.fn(async () => ({ close: vi.fn() })), + ); + try { + let gen = 0; + mocks.readCompositorFrame.mockImplementation(async () => ({ + ...sharedMeta({ gen: ++gen }), + shared: true, + })); + await mountedView(); + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta({ gen: 1 })); + await run(300, 1000 / 280); + + mocks.readCompositorFrame.mockImplementation(async () => ({ + gen: ++gen, + width: 2, + height: 1, + data: new Uint8Array(8), + })); + await run(100, 1000 / 280, 300); + mocks.readCompositorFrame.mockClear(); + await run(500, 1000 / 280, 400); + + expect(mocks.readCompositorFrame.mock.calls.length).toBeLessThanOrEqual(17); + } finally { + vi.unstubAllGlobals(); + } + }); + // Read-back frames bound the copies: the fast cadence is for shared textures only. it("keeps read-back frames at ~30 pulls a second even while they keep coming", async () => { const run = manualFrames(); diff --git a/src/native/hooks/useNativeCompositorView.ts b/src/native/hooks/useNativeCompositorView.ts index d42fcfa89..94688c2bf 100644 --- a/src/native/hooks/useNativeCompositorView.ts +++ b/src/native/hooks/useNativeCompositorView.ts @@ -289,8 +289,12 @@ export function useNativeCompositorView( readCompositorFrame(id, lastGen) .then((frame) => { inFlight = false; - if (frame) { + // A receipt keeps the fast cadence going. Read-back pixels mean the view went back to + // read-back (a failed import or send), whose frames are pulled ~30 times a second. + if (frame && "shared" in frame) { lastFrameAt = now; + } else if (frame) { + sharedTransport = false; } // `null` = nothing newer than `lastGen` (idle path — no pixels // crossed IPC) OR no frame yet. Either way, leave the canvas as-is. From 72574f0a19f0a0287e31d8932e9a8df2ed6e28ad Mon Sep 17 00:00:00 2001 From: EtienneLescot Date: Fri, 2 Oct 2026 11:04:51 +0200 Subject: [PATCH 4/4] fix(preview): never rewrite a shared slot Chromium still holds A slot whose release had not come back after a second was reclaimed and rewritten, while Chromium might still read that texture. A held slot is now never written: when every slot is held and one has waited a second for its release, the view falls back to read-back instead. A briefly full ring still waits for the next release, as before. --- crates/compositor/src/live.rs | 17 +++--- crates/compositor/src/shared_frames.rs | 53 +++++++++---------- .../architecture/preview.md | 4 +- 3 files changed, 40 insertions(+), 34 deletions(-) diff --git a/crates/compositor/src/live.rs b/crates/compositor/src/live.rs index 8fd6a29e0..912741b2c 100644 --- a/crates/compositor/src/live.rs +++ b/crates/compositor/src/live.rs @@ -1897,7 +1897,7 @@ unsafe fn render_thread( enum SharedPublish { /// Posée dans une case de l'anneau. Published, - /// Chromium tient toutes les cases : frame sautée. + /// Chromium tient toutes les cases : la frame attend dans le RT qu'il en relâche une. NoFreeSlot, /// Transport partagé coupé ou indisponible : la frame repasse par le readback. Off, @@ -1940,12 +1940,17 @@ unsafe fn publish_shared( let Some(ring) = ring.as_mut() else { return SharedPublish::Off; }; - let claimed = shared - .slot_book - .lock() - .ok() - .and_then(|mut book| book.claim(crate::shared_frames::RING_SLOTS, Instant::now())); + let (claimed, lost) = match shared.slot_book.lock() { + Ok(mut book) => (book.claim(crate::shared_frames::RING_SLOTS), book.lost(Instant::now())), + Err(_) => (None, false), + }; let Some(slot) = claimed else { + // Chromium relâche d'ordinaire une case dans la milliseconde. Une case perdue ne + // reviendra plus : la réécrire déchirerait peut-être une image qu'il lit encore, et + // l'attendre figerait la preview. + if lost { + return turn_off("une case jamais relâchée par Chromium".into()); + } return SharedPublish::NoFreeSlot; }; let (width, height) = comp.render_size(); diff --git a/crates/compositor/src/shared_frames.rs b/crates/compositor/src/shared_frames.rs index bf0e1763a..421897131 100644 --- a/crates/compositor/src/shared_frames.rs +++ b/crates/compositor/src/shared_frames.rs @@ -20,8 +20,8 @@ use std::time::{Duration, Instant}; pub const RING_SLOTS: u32 = 4; /// Une case tenue plus longtemps que ça a perdu sa libération (renderer rechargé, import -/// échoué en route) : elle est reprise plutôt que de laisser l'anneau se vider et la preview -/// se figer. Même ordre que le délai d'Electron sur `sendSharedTexture` (1 s). +/// échoué en route) : Chromium relâche les autres dans la milliseconde. Même ordre que le +/// délai d'Electron sur `sendSharedTexture` (1 s). pub const LOST_SLOT_AFTER: Duration = Duration::from_secs(1); /// Une frame composée posée dans une case de l'anneau, telle que JS la reçoit. @@ -58,10 +58,10 @@ pub struct SlotBook { impl SlotBook { /// La case où écrire la prochaine frame. Une case libre d'abord ; à défaut, celle de la - /// frame prête que personne n'a prise, retirée puisqu'elle est périmée ; à défaut, une - /// case tenue depuis `LOST_SLOT_AFTER`. `None` quand Chromium tient légitimement toutes - /// les cases : la frame est sautée, la suivante passera. - pub fn claim(&mut self, slots: u32, now: Instant) -> Option { + /// frame prête que personne n'a prise, retirée puisqu'elle est périmée. `None` quand + /// Chromium tient toutes les cases : une case tenue n'est jamais réécrite, même perdue + /// (`lost`), puisqu'il peut encore la lire. + pub fn claim(&mut self, slots: u32) -> Option { let ready = self.ready.map(|frame| frame.slot); let is_held = |slot: u32| self.held.iter().any(|held| held.slot == slot); if let Some(free) = (0..slots).find(|slot| !is_held(*slot) && Some(*slot) != ready) { @@ -71,10 +71,12 @@ impl SlotBook { self.ready = None; return Some(stale); } - let lost = self.held.first().filter(|held| now.duration_since(held.since) >= LOST_SLOT_AFTER)?; - let slot = lost.slot; - self.held.remove(0); - Some(slot) + None + } + + /// Une case est tenue depuis `LOST_SLOT_AFTER` : sa libération ne viendra plus. + pub fn lost(&self, now: Instant) -> bool { + self.held.first().is_some_and(|held| now.duration_since(held.since) >= LOST_SLOT_AFTER) } pub fn publish(&mut self, frame: SharedFrame) { @@ -91,8 +93,7 @@ impl SlotBook { } /// Chromium a relâché la frame `gen` de la case `slot`. La génération évite qu'une - /// libération tardive, arrivée après que la case a été reprise, ne libère la frame - /// suivante. + /// libération en double ne libère la frame suivante posée dans la même case. pub fn release(&mut self, slot: u32, gen: u64) { self.held.retain(|held| held.slot != slot || held.gen != gen); } @@ -261,7 +262,7 @@ mod tests { let now = Instant::now(); let mut book = SlotBook::default(); book.publish(frame(1, 0)); - assert_eq!(book.claim(RING_SLOTS, now), Some(1)); + assert_eq!(book.claim(RING_SLOTS), Some(1)); // La frame prête n'a pas bougé : JS peut toujours la prendre. assert_eq!(book.take(0, now), Some(frame(1, 0))); } @@ -270,9 +271,9 @@ mod tests { fn a_taken_slot_is_never_written_until_released() { let now = Instant::now(); let mut book = all_held(now); - assert_eq!(book.claim(RING_SLOTS, now), None, "Chromium tient toutes les cases"); + assert_eq!(book.claim(RING_SLOTS), None, "Chromium tient toutes les cases"); book.release(2, 3); - assert_eq!(book.claim(RING_SLOTS, now), Some(2)); + assert_eq!(book.claim(RING_SLOTS), Some(2)); } #[test] @@ -284,7 +285,7 @@ mod tests { book.take(u64::from(slot), now); } book.publish(frame(9, 3)); - assert_eq!(book.claim(RING_SLOTS, now), Some(3)); + assert_eq!(book.claim(RING_SLOTS), Some(3)); // Retirée : la livrer maintenant, pendant qu'on la réécrit, donnerait une image déchirée. assert_eq!(book.take(0, now), None); } @@ -307,23 +308,21 @@ mod tests { book.take(0, now); book.release(3, 1); book.release(0, 7); - assert_eq!(book.claim(1, now), None, "ni la case 3 ni la génération 7 ne sont tenues"); + assert_eq!(book.claim(1), None, "ni la case 3 ni la génération 7 ne sont tenues"); book.release(0, 1); - assert_eq!(book.claim(1, now), Some(0)); + assert_eq!(book.claim(1), Some(0)); } #[test] - fn a_slot_whose_release_never_came_is_taken_back_after_a_second() { + fn a_slot_whose_release_never_came_is_reported_lost_but_never_rewritten() { let start = Instant::now(); let mut book = all_held(start); - assert_eq!(book.claim(RING_SLOTS, start + LOST_SLOT_AFTER / 2), None); - // La plus ancienne d'abord. - assert_eq!(book.claim(RING_SLOTS, start + LOST_SLOT_AFTER), Some(0)); - // Sa libération tardive ne doit plus rien libérer : la case est déjà reprise. - book.release(0, 1); - book.publish(frame(10, 0)); - assert!(book.take(9, start + LOST_SLOT_AFTER).is_some()); + assert!(!book.lost(start + LOST_SLOT_AFTER / 2)); + assert!(book.lost(start + LOST_SLOT_AFTER)); + // Chromium peut encore lire une case perdue : la réécrire déchirerait son image. + assert_eq!(book.claim(RING_SLOTS), None); + // Si sa libération finit par venir, la case sert de nouveau. book.release(0, 1); - assert_eq!(book.claim(RING_SLOTS, start + LOST_SLOT_AFTER), Some(1)); + assert_eq!(book.claim(RING_SLOTS), Some(0)); } } diff --git a/technical-documentation/architecture/preview.md b/technical-documentation/architecture/preview.md index 9b7c34ed9..f28b0bde5 100644 --- a/technical-documentation/architecture/preview.md +++ b/technical-documentation/architecture/preview.md @@ -194,7 +194,9 @@ thread busy (measurements in D3D11 textures (`shared_frames.rs`, NT handles, no keyed mutex), waits for the GPU to finish the copy, and publishes `{ gen, slot, handle }`. `SlotBook` tracks which slot holds the ready frame and which ones Chromium still holds; a frame nobody took gives - its slot back, and a slot whose release never came is reclaimed after 1 s. + its slot back. A held slot is never rewritten, since Chromium may still read it: when + every slot is held and one has waited 1 s for its release, the view falls back to + read-back. - **Main process.** `readFrame` takes the frame (`readSharedFrame`), imports it with `sharedTexture.importSharedTexture`, sends it with `sharedTexture.sendSharedTexture` to the frame that asked, drops its own reference and answers with a receipt