diff --git a/crates/Cargo.toml b/crates/Cargo.toml index 3caa7edfd..a65f9aaef 100644 --- a/crates/Cargo.toml +++ b/crates/Cargo.toml @@ -68,6 +68,7 @@ features = [ "Win32_Graphics_Direct2D_Common", # D2D1_COLOR_F / PIXEL_FORMAT "Win32_Graphics_DirectWrite", # mise en page + rastérisation du texte "Win32_Graphics_Gdi", # brosses/police GUI de la preview (poc-d3d/src/app.rs) + "Win32_Security", # IDXGIResource1::CreateSharedHandle (shared_frames.rs) "Win32_System_Performance", # QueryPerformanceCounter / Frequency (§10) "Win32_System_LibraryLoader", # GetModuleHandleW (hInstance) "Win32_UI_WindowsAndMessaging", # fenêtres (harnais live + GUI du POC) diff --git a/crates/compositor-view-napi/src/lib.rs b/crates/compositor-view-napi/src/lib.rs index 185f48d43..338beaa80 100644 --- a/crates/compositor-view-napi/src/lib.rs +++ b/crates/compositor-view-napi/src/lib.rs @@ -11,6 +11,7 @@ use napi::{Env, JsFunction, Task}; use napi_derive::napi; use openscreen_compositor::compositor::{live_params_from_scene, Compositor}; use openscreen_compositor::d3d::{Backend, Gpu}; +use openscreen_compositor::frame_geometry::FootageQuad; use openscreen_compositor::gif_export::{GifExportParams, GifStats}; use openscreen_compositor::gif_export_control::{GifExportCancelled, GifExportControl}; use openscreen_compositor::live::{LiveView, PausedPreviews}; @@ -249,12 +250,82 @@ pub fn read_frame(id: i32, since_gen: f64) -> Result> { width: w, height: h, data: Buffer::from(pixels), - footage: footage.map(|q| q.corners.iter().flatten().map(|&v| v as f64).collect()), + footage: footage_corners(footage), footage_projective: footage.is_some_and(|q| q.projective), } })) } +/// Les coins du métrage (TL, TR, BR, BL) aplatis en huit nombres, tels que JS les lit. +fn footage_corners(footage: Option) -> Option> { + footage.map(|q| q.corners.iter().flatten().map(|&v| v as f64).collect()) +} + +/// Une frame de preview posée dans une texture partagée : de quoi l'importer côté GPU +/// (`sharedTexture.importSharedTexture`) au lieu d'en recevoir les pixels. +#[napi(object)] +pub struct SharedFramePacket { + pub gen: f64, + /// Case de l'anneau qui porte la frame, à rendre avec `gen` par `release_shared_frame` + /// quand Chromium l'a relâchée. + pub slot: u32, + /// Handle NT de la texture, 8 octets little-endian : la forme qu'attend Electron + /// (`SharedTextureHandle.ntHandle`). Valable dans ce processus seulement. + pub handle: Buffer, + pub width: u32, + pub height: u32, + pub footage: Option>, + pub footage_projective: bool, +} + +/// Livrer les frames de la vue par textures partagées plutôt que par `read_frame`. Rend +/// `false` quand la machine ne le peut pas (hors Windows, backend logiciel) : la vue reste +/// alors au readback. +#[napi] +pub fn set_shared_frames(id: i32, enabled: bool) -> bool { + registry() + .lock() + .unwrap() + .get(&id) + .is_some_and(|v| v.set_shared_frames(enabled)) +} + +/// Le pendant de `read_frame` pour une vue en textures partagées : la dernière frame posée +/// dans l'anneau, si elle est plus récente que `since_gen`. Sa case reste tenue jusqu'à +/// `release_shared_frame`. `Ok(None)` aussi quand la vue relit en RAM. +#[napi] +pub fn read_shared_frame(id: i32, since_gen: f64) -> Result> { + let frame = match registry().lock().unwrap().get(&id) { + None => return Ok(None), + Some(v) => { + // Même relais que `read_frame` : sans lui, un thread de rendu mort ne se verrait + // que par un canvas noir. + if let Some(fatal) = v.fatal_error() { + return Err(Error::from_reason(fatal)); + } + v.take_shared_frame(since_gen.max(0.0) as u64) + } + }; + Ok(frame.map(|f| SharedFramePacket { + gen: f.gen as f64, + slot: f.slot, + handle: f.handle.to_le_bytes().to_vec().into(), + width: f.width, + height: f.height, + footage: footage_corners(f.footage), + footage_projective: f.footage.is_some_and(|q| q.projective), + })) +} + +/// Chromium a relâché la frame `gen` de la case `slot` (`allReferencesReleased` côté +/// Electron) : le thread de rendu peut y réécrire. Sans effet sur une vue détruite. +#[napi] +pub fn release_shared_frame(id: i32, slot: u32, gen: f64) { + if let Some(v) = registry().lock().unwrap().get(&id) { + v.release_shared_frame(slot, gen.max(0.0) as u64); + } +} + /// Param live (inspector). Le type de valeur route vers le bon setter : /// bool = switch (webcamMirror…), number = slider (shadow/roundness/motionBlur/backgroundBlur), /// string = sélection (backgroundColor "#rrggbb"). diff --git a/crates/compositor/src/compositor_windows.rs b/crates/compositor/src/compositor_windows.rs index cfe335143..1875d4a34 100644 --- a/crates/compositor/src/compositor_windows.rs +++ b/crates/compositor/src/compositor_windows.rs @@ -3008,6 +3008,12 @@ impl Compositor { Ok((rw, rh, out)) } + /// Le RT de la dernière composition, à `render_size()`, pour qui le copie ailleurs sans + /// passer par la RAM : l'anneau de textures partagées de la preview (`shared_frames`). + pub fn render_target(&self) -> &ID3D11Texture2D { + &self.rt + } + /// Comme `rgb_to_nv12`, mais vers `target_w`×`target_h` : si la cible diffère de la taille /// de rendu, le RT composé est d'abord redimensionné (bilinéaire, `ps_tex`/`sampler` déjà /// utilisés partout ailleurs dans le fichier, cf. `blit_resized`). Le RT suivant la sortie, diff --git a/crates/compositor/src/lib.rs b/crates/compositor/src/lib.rs index 8e5f63e17..767abac8a 100644 --- a/crates/compositor/src/lib.rs +++ b/crates/compositor/src/lib.rs @@ -51,6 +51,8 @@ pub mod sculpt; // `segmentation` ses deux entrées échouent proprement, ce qui garde le reste du crate // indépendant du choix de packaging d'ONNX Runtime. pub mod segmentation; +// Livraison de la preview par textures partagées (Windows) ; la tenue des cases est portable. +pub mod shared_frames; pub mod text_anim; pub mod text_fonts; pub mod text_plate; diff --git a/crates/compositor/src/live.rs b/crates/compositor/src/live.rs index af586f262..912741b2c 100644 --- a/crates/compositor/src/live.rs +++ b/crates/compositor/src/live.rs @@ -27,9 +27,12 @@ use crate::regions::{speed_at, ProgrammeClock}; use crate::scene::Scene; use crate::config::{self, Cfg}; use crate::cursor::CursorTrack; -use crate::d3d::Gpu; +use crate::d3d::{Backend, Gpu}; use crate::frame_geometry::webcam_is_real; use crate::pipeline::Decoder; +#[cfg(windows)] +use crate::shared_frames::SharedRing; +use crate::shared_frames::{SharedFrame, SlotBook}; use crate::timeline_walk::{frame_step, FrameStep, NextFrameTime}; use anyhow::Result; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; @@ -800,6 +803,15 @@ struct Shared { /// vide la ferait repartir à 1 — donc rejouer des générations déjà peintes. Monotone, /// jamais remise à zéro. frame_gen: AtomicU64, + /// La vue livre ses frames dans l'anneau de textures partagées (`take_shared_frame`) au + /// lieu de les relire en RAM (`latest_frame_since`). Posé par JS ; le thread de rendu le + /// rabat à `false` si l'anneau ne peut pas servir, et les frames repassent par la RAM. + shared_frames: AtomicBool, + /// Tenue des cases de l'anneau, entre le thread de rendu et le thread Node. + slot_book: Mutex, + /// Republier la frame courante au prochain tour, même en pause : posé quand le transport + /// change, pour que le consommateur ait une image sans attendre que quelque chose bouge. + republish: AtomicBool, /// Erreur fatale du thread de rendu (device D3D11 introuvable, décodeur qui refuse /// le fichier…). Le thread meurt sur la première erreur ; sans ce champ, elle /// finissait dans un `eprintln!` que personne ne lit et l'utilisateur n'avait @@ -853,6 +865,9 @@ impl LiveView { stop: AtomicBool::new(false), latest_frame: Mutex::new(None), frame_gen: AtomicU64::new(0), + shared_frames: AtomicBool::new(false), + slot_book: Mutex::new(SlotBook::default()), + republish: AtomicBool::new(false), fatal: Mutex::new(None), }); let sh = shared.clone(); @@ -929,6 +944,35 @@ impl LiveView { } } + /// Livrer les frames par textures partagées plutôt que par readback. Seul le backend + /// matériel de Windows sait le faire : ailleurs la demande est refusée et la vue continue + /// de relire en RAM. Rend l'état obtenu. + pub fn set_shared_frames(&self, enabled: bool) -> bool { + let on = enabled && cfg!(windows) && Gpu::probe() == Some(Backend::Hardware); + if self.shared.shared_frames.swap(on, Ordering::Relaxed) != on { + if !on { + if let Ok(mut book) = self.shared.slot_book.lock() { + book.clear_ready(); + } + } + self.shared.republish.store(true, Ordering::Relaxed); + } + on + } + + /// La dernière frame posée dans l'anneau, si elle est plus récente que `since_gen`. Sa + /// case reste tenue jusqu'à `release_shared_frame`. `None` aussi quand la vue relit en RAM. + pub fn take_shared_frame(&self, since_gen: u64) -> Option { + self.shared.slot_book.lock().ok()?.take(since_gen, Instant::now()) + } + + /// Chromium a relâché la frame `gen` de la case `slot` : le thread de rendu peut y réécrire. + pub fn release_shared_frame(&self, slot: u32, gen: u64) { + if let Ok(mut book) = self.shared.slot_book.lock() { + book.release(slot, gen); + } + } + /// Switch inspector (booléen). pub fn set_param_bool(&self, key: &str, value: bool) { if let Ok(mut p) = self.shared.inspector.lock() { @@ -1426,8 +1470,19 @@ unsafe fn render_thread( // appliquée, on refuse de jouer le layout fixture (POC) : un fallback fixture ne ferait que // MASQUER un scene-push cassé. On attend la scène avant de produire le 1er frame. let mut scene_applied = false; + // Textures partagées de la preview, créées au premier besoin (voir `publish_shared`). + let mut ring = Ring::default(); + // Une frame composée qui n'a trouvé aucune case libre dans l'anneau : le RT la garde, elle + // repart dès qu'une case se libère. Sans ça, une frame composée en pause (seek, réglage) + // restait invisible jusqu'au changement suivant — en lecture, la suivante la remplace. + let mut publish_pending = false; while !shared.stop.load(Ordering::SeqCst) { + // Le transport vient de changer : le consommateur doit recevoir la frame courante + // sans attendre qu'une autre soit composée (en pause, il n'y en aurait pas). + if shared.republish.swap(false, Ordering::Relaxed) { + first = true; + } // params inspector : booléens/taps → cfg ; valeurs continues → live_params let ip = *shared.inspector.lock().unwrap(); let mut clip_changed = false; @@ -1783,39 +1838,52 @@ unsafe fn render_thread( stepped = true; } - if stepped || first { + if stepped || first || publish_pending { if pw > 0 && ph > 0 { - // Step complet : `compose_frame` (déjà appelé par `step`/`present_frame`/ - // `recompose`) a rastérisé le RT à la géométrie de sortie ramenée au panneau. - // On lit ce RT DIRECTEMENT à sa résolution de rendu (`readback_direct` : copy - // rt → staging → Map/Unmap), sans le resize `blit_resized` qui, depuis la - // refonte ratio, n'était plus qu'une copie identité + une alloc NV12 inutile. - match comp.readback_direct() { - Ok((rw, rh, rgba)) => { - // Publie dans `latest_frame` : on remplace le buffer précédent - // (le canvas ne montre que la dernière frame, peu importe combien - // le renderer en a raté entre deux lectures napi). On incrémente - // la génération sous le MÊME lock que l'écriture du buffer, pour - // qu'un lecteur ne puisse jamais voir un `gen` neuf appairé à un - // buffer périmé (ou l'inverse). `+ 1` depuis la précédente, `1` au - // premier publish. Les dims publiées sont celles du RENDU (`rw`×`rh`) : - // le canvas JS s'y dimensionne (packet auto-descriptif) puis CSS met à - // l'échelle vers la boîte du panneau — plus de resize GPU intermédiaire. - // La génération vient d'un compteur atomique et non du slot : la - // livraison sans copie VIDE le slot en le lisant, et un - // `unwrap_or(1)` repartirait alors de 1 — le consommateur recevrait - // des générations déjà peintes et boucherait. Séquence identique à - // l'ancienne dérivation tant que le slot n'est pas vidé. - let next_gen = shared.frame_gen.fetch_add(1, Ordering::Relaxed) + 1; - if let Ok(mut slot) = shared.latest_frame.lock() { - *slot = Some((next_gen, rw, rh, rgba, comp.footage_quad())); - } + match publish_shared(&shared, &gpu, &comp, &mut ring) { + SharedPublish::Published => { first = false; + publish_pending = false; } - Err(e) => { - eprintln!("[live] readback_direct: {e:#}"); - std::thread::sleep(Duration::from_millis(8)); + // Chromium tient toutes les cases : la frame attend dans le RT, et la boucle + // ne tourne pas à vide le temps qu'il en relâche une. + SharedPublish::NoFreeSlot => { + publish_pending = true; + std::thread::sleep(Duration::from_millis(2)); } + // Step complet : `compose_frame` (déjà appelé par `step`/`present_frame`/ + // `recompose`) a rastérisé le RT à la géométrie de sortie ramenée au panneau. + // On lit ce RT DIRECTEMENT à sa résolution de rendu (`readback_direct` : copy + // rt → staging → Map/Unmap), sans le resize `blit_resized` qui, depuis la + // refonte ratio, n'était plus qu'une copie identité + une alloc NV12 inutile. + SharedPublish::Off => match comp.readback_direct() { + Ok((rw, rh, rgba)) => { + // Publie dans `latest_frame` : on remplace le buffer précédent + // (le canvas ne montre que la dernière frame, peu importe combien + // le renderer en a raté entre deux lectures napi). On incrémente + // la génération sous le MÊME lock que l'écriture du buffer, pour + // qu'un lecteur ne puisse jamais voir un `gen` neuf appairé à un + // buffer périmé (ou l'inverse). `+ 1` depuis la précédente, `1` au + // premier publish. Les dims publiées sont celles du RENDU (`rw`×`rh`) : + // le canvas JS s'y dimensionne (packet auto-descriptif) puis CSS met à + // l'échelle vers la boîte du panneau — plus de resize GPU intermédiaire. + // La génération vient d'un compteur atomique et non du slot : la + // livraison sans copie VIDE le slot en le lisant, et un + // `unwrap_or(1)` repartirait alors de 1 — le consommateur recevrait + // des générations déjà peintes et boucherait. Séquence identique à + // l'ancienne dérivation tant que le slot n'est pas vidé. + let next_gen = shared.frame_gen.fetch_add(1, Ordering::Relaxed) + 1; + if let Ok(mut slot) = shared.latest_frame.lock() { + *slot = Some((next_gen, rw, rh, rgba, comp.footage_quad())); + } + first = false; + publish_pending = false; + } + Err(e) => { + eprintln!("[live] readback_direct: {e:#}"); + std::thread::sleep(Duration::from_millis(8)); + } + }, } } } else { @@ -1825,6 +1893,93 @@ unsafe fn render_thread( Ok(()) } +/// Ce que la publication dans l'anneau partagé a fait de la frame composée. +enum SharedPublish { + /// Posée dans une case de l'anneau. + Published, + /// Chromium tient toutes les cases : la frame attend dans le RT qu'il en relâche une. + NoFreeSlot, + /// Transport partagé coupé ou indisponible : la frame repasse par le readback. + Off, +} + +#[cfg(windows)] +type Ring = Option; +#[cfg(not(windows))] +type Ring = (); + +/// Pose la frame composée dans l'anneau de textures partagées, si la vue livre ainsi. +/// +/// Toute panne coupe le transport (`shared_frames` à `false`) et rend `Off` : la frame part +/// par le readback, et le service, qui lit les deux, n'a rien à décider. +#[cfg(windows)] +unsafe fn publish_shared( + shared: &Shared, + gpu: &Gpu, + comp: &Compositor, + ring: &mut Ring, +) -> SharedPublish { + if !shared.shared_frames.load(Ordering::Relaxed) { + return SharedPublish::Off; + } + let turn_off = |why: String| { + eprintln!("[live] textures partagées coupées ({why}) — retour au readback"); + shared.shared_frames.store(false, Ordering::Relaxed); + SharedPublish::Off + }; + // WARP ne partage rien avec le device de Chromium. + if gpu.backend != Backend::Hardware { + return turn_off("backend logiciel".into()); + } + if ring.is_none() { + match SharedRing::new(gpu) { + Ok(created) => *ring = Some(created), + Err(e) => return turn_off(format!("{e:#}")), + } + } + let Some(ring) = ring.as_mut() else { + return SharedPublish::Off; + }; + let (claimed, lost) = match shared.slot_book.lock() { + Ok(mut book) => (book.claim(crate::shared_frames::RING_SLOTS), book.lost(Instant::now())), + Err(_) => (None, false), + }; + let Some(slot) = claimed else { + // Chromium relâche d'ordinaire une case dans la milliseconde. Une case perdue ne + // reviendra plus : la réécrire déchirerait peut-être une image qu'il lit encore, et + // l'attendre figerait la preview. + if lost { + return turn_off("une case jamais relâchée par Chromium".into()); + } + return SharedPublish::NoFreeSlot; + }; + let (width, height) = comp.render_size(); + match ring.write(slot, comp.render_target(), width, height) { + Ok(handle) => { + if let Ok(mut book) = shared.slot_book.lock() { + // Le compteur du readback : une seule suite de générations quel que soit le + // transport, incrémentée sous le lock de la publication comme là-bas. + let gen = shared.frame_gen.fetch_add(1, Ordering::Relaxed) + 1; + book.publish(SharedFrame { + gen, + slot, + handle, + width, + height, + footage: comp.footage_quad(), + }); + } + SharedPublish::Published + } + Err(e) => turn_off(format!("{e:#}")), + } +} + +#[cfg(not(windows))] +unsafe fn publish_shared(_: &Shared, _: &Gpu, _: &Compositor, _: &mut Ring) -> SharedPublish { + SharedPublish::Off +} + // ---------- harnais standalone (poc-d3d.exe --live) ---------- // // Le harnais crée une fenêtre Win32 simple qui héberge la preview live ; c'est un diff --git a/crates/compositor/src/shared_frames.rs b/crates/compositor/src/shared_frames.rs new file mode 100644 index 000000000..421897131 --- /dev/null +++ b/crates/compositor/src/shared_frames.rs @@ -0,0 +1,328 @@ +//! Livraison de la preview par textures partagées : le thread de rendu copie chaque frame +//! composée dans l'une des textures d'un petit anneau, et Electron l'importe côté GPU +//! (`sharedTexture.importSharedTexture`) au lieu d'en recevoir les pixels. Plus de `Map` +//! qui attend le GPU puis recopie l'image en RAM, plus de `Vec` de plusieurs Mo par frame +//! à travers l'IPC : mesuré, ce transport coûtait à lui seul 37 à 55 % du thread principal +//! du renderer à 30 images/s. +//! +//! Deux moitiés : +//! - `SlotBook`, portable et sans GPU : quelle case porte la frame prête, lesquelles +//! Chromium tient encore. Partagé entre le thread de rendu (`claim`, `publish`) et le +//! thread Node (`take`, `release`). +//! - `SharedRing` (Windows, backend matériel) : les textures D3D11 et leurs handles NT. +//! macOS (IOSurface) et Linux (dmabuf) restent sur le readback. + +use crate::frame_geometry::FootageQuad; +use std::time::{Duration, Instant}; + +/// Cases de l'anneau : une prête, une en transit vers le renderer (tenue jusqu'à ce que +/// Chromium la relâche), une en écriture, et une de marge pour la latence de la libération. +pub const RING_SLOTS: u32 = 4; + +/// Une case tenue plus longtemps que ça a perdu sa libération (renderer rechargé, import +/// échoué en route) : Chromium relâche les autres dans la milliseconde. Même ordre que le +/// délai d'Electron sur `sendSharedTexture` (1 s). +pub const LOST_SLOT_AFTER: Duration = Duration::from_secs(1); + +/// Une frame composée posée dans une case de l'anneau, telle que JS la reçoit. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct SharedFrame { + /// Même compteur que les frames lues en RAM : le consommateur ne voit qu'une suite de + /// générations, quel que soit le transport. + pub gen: u64, + pub slot: u32, + /// Valeur du handle NT de la texture de la case, valable dans CE processus. Electron le + /// duplique à l'import : la case garde le sien. + pub handle: u64, + pub width: u32, + pub height: u32, + pub footage: Option, +} + +/// Une case prise par JS : la génération qu'elle porte et quand elle est partie. +#[derive(Debug, Clone, Copy)] +struct Held { + slot: u32, + gen: u64, + since: Instant, +} + +/// Qui tient quelle case de l'anneau. +#[derive(Debug, Default)] +pub struct SlotBook { + /// La dernière frame publiée, que JS n'a pas encore prise. + ready: Option, + /// Les cases prises par JS, que Chromium n'a pas encore relâchées, plus anciennes d'abord. + held: Vec, +} + +impl SlotBook { + /// La case où écrire la prochaine frame. Une case libre d'abord ; à défaut, celle de la + /// frame prête que personne n'a prise, retirée puisqu'elle est périmée. `None` quand + /// Chromium tient toutes les cases : une case tenue n'est jamais réécrite, même perdue + /// (`lost`), puisqu'il peut encore la lire. + pub fn claim(&mut self, slots: u32) -> Option { + let ready = self.ready.map(|frame| frame.slot); + let is_held = |slot: u32| self.held.iter().any(|held| held.slot == slot); + if let Some(free) = (0..slots).find(|slot| !is_held(*slot) && Some(*slot) != ready) { + return Some(free); + } + if let Some(stale) = ready { + self.ready = None; + return Some(stale); + } + None + } + + /// Une case est tenue depuis `LOST_SLOT_AFTER` : sa libération ne viendra plus. + pub fn lost(&self, now: Instant) -> bool { + self.held.first().is_some_and(|held| now.duration_since(held.since) >= LOST_SLOT_AFTER) + } + + pub fn publish(&mut self, frame: SharedFrame) { + self.ready = Some(frame); + } + + /// La frame prête, si elle est plus récente que `since_gen`. Sa case reste tenue jusqu'à + /// `release` : le thread de rendu n'y écrira plus d'ici là. + pub fn take(&mut self, since_gen: u64, now: Instant) -> Option { + let frame = self.ready.filter(|frame| frame.gen > since_gen)?; + self.ready = None; + self.held.push(Held { slot: frame.slot, gen: frame.gen, since: now }); + Some(frame) + } + + /// Chromium a relâché la frame `gen` de la case `slot`. La génération évite qu'une + /// libération en double ne libère la frame suivante posée dans la même case. + pub fn release(&mut self, slot: u32, gen: u64) { + self.held.retain(|held| held.slot != slot || held.gen != gen); + } + + /// Oublie la frame prête : la vue repasse au readback, personne ne viendra la prendre. + pub fn clear_ready(&mut self) { + self.ready = None; + } +} + +#[cfg(windows)] +pub use ring::SharedRing; + +#[cfg(windows)] +mod ring { + use super::RING_SLOTS; + use crate::d3d::Gpu; + use anyhow::{bail, Result}; + use std::time::{Duration, Instant}; + use windows::core::{Interface, PCWSTR}; + use windows::Win32::Foundation::{CloseHandle, BOOL, HANDLE}; + use windows::Win32::Graphics::Direct3D11::{ + ID3D11Device, ID3D11DeviceContext, ID3D11Query, ID3D11Texture2D, + D3D11_ASYNC_GETDATA_DONOTFLUSH, D3D11_BIND_RENDER_TARGET, D3D11_BIND_SHADER_RESOURCE, + D3D11_QUERY_DESC, D3D11_QUERY_EVENT, D3D11_RESOURCE_MISC_SHARED, + D3D11_RESOURCE_MISC_SHARED_NTHANDLE, D3D11_TEXTURE2D_DESC, D3D11_USAGE_DEFAULT, + }; + use windows::Win32::Graphics::Dxgi::Common::{DXGI_FORMAT_R8G8B8A8_UNORM, DXGI_SAMPLE_DESC}; + use windows::Win32::Graphics::Dxgi::{ + IDXGIResource1, DXGI_SHARED_RESOURCE_READ, DXGI_SHARED_RESOURCE_WRITE, + }; + + /// Au-delà, le GPU est bloqué ou perdu : on rend une erreur plutôt que de geler la vue. + const COPY_TIMEOUT: Duration = Duration::from_millis(500); + + struct SlotTexture { + tex: ID3D11Texture2D, + handle: HANDLE, + width: u32, + height: u32, + } + + impl Drop for SlotTexture { + fn drop(&mut self) { + // Electron importe un duplicata du handle : fermer le nôtre ne retire rien à une + // frame que Chromium affiche encore. + unsafe { + let _ = CloseHandle(self.handle); + } + } + } + + /// Les textures partagées d'une vue live, créées à la demande à la taille du rendu. + pub struct SharedRing { + device: ID3D11Device, + context: ID3D11DeviceContext, + copied: ID3D11Query, + slots: Vec>, + } + + impl SharedRing { + pub fn new(gpu: &Gpu) -> Result { + let desc = D3D11_QUERY_DESC { Query: D3D11_QUERY_EVENT, MiscFlags: 0 }; + let mut copied = None; + unsafe { gpu.device.CreateQuery(&desc, Some(&mut copied))? }; + Ok(SharedRing { + device: gpu.device.clone(), + context: gpu.context.clone(), + copied: copied.expect("CreateQuery a réussi sans requête"), + slots: (0..RING_SLOTS).map(|_| None).collect(), + }) + } + + /// Copie `src` (`width`×`height`, RGBA8) dans la case `slot` et rend le handle de sa + /// texture une fois le GPU arrivé au bout de la copie. Chromium la lit depuis son propre + /// device, sans keyed mutex : quand le handle part, plus rien ne doit y écrire. L'attente + /// est celle que faisait déjà le `Map` du readback, la recopie en RAM en moins. + /// + /// La case ne doit être ni prête ni tenue (`SlotBook::claim`) : sa texture peut être + /// recréée ici quand la taille du rendu a changé. + pub unsafe fn write( + &mut self, + slot: u32, + src: &ID3D11Texture2D, + width: u32, + height: u32, + ) -> Result { + let entry = &mut self.slots[slot as usize]; + if !matches!(entry, Some(texture) if texture.width == width && texture.height == height) { + *entry = Some(create_slot(&self.device, width, height)?); + } + let target = entry.as_ref().expect("case créée juste au-dessus"); + self.context.CopyResource(&target.tex, src); + self.context.End(&self.copied); + self.context.Flush(); + let deadline = Instant::now() + COPY_TIMEOUT; + loop { + let mut done = BOOL(0); + self.context.GetData( + &self.copied, + Some(&mut done as *mut BOOL as *mut core::ffi::c_void), + std::mem::size_of::() as u32, + D3D11_ASYNC_GETDATA_DONOTFLUSH.0 as u32, + )?; + if done.as_bool() { + break; + } + if Instant::now() > deadline { + bail!("le GPU n'a pas fini de copier la frame en {COPY_TIMEOUT:?}"); + } + std::thread::yield_now(); + } + Ok(target.handle.0 as u64) + } + } + + unsafe fn create_slot(device: &ID3D11Device, width: u32, height: u32) -> Result { + let desc = D3D11_TEXTURE2D_DESC { + Width: width, + Height: height, + MipLevels: 1, + ArraySize: 1, + // Le format du RT composé : la copie est un `CopyResource`, sans conversion. + Format: DXGI_FORMAT_R8G8B8A8_UNORM, + SampleDesc: DXGI_SAMPLE_DESC { Count: 1, Quality: 0 }, + Usage: D3D11_USAGE_DEFAULT, + BindFlags: (D3D11_BIND_RENDER_TARGET.0 | D3D11_BIND_SHADER_RESOURCE.0) as u32, + CPUAccessFlags: 0, + // Handle NT, le seul qu'Electron importe, sans keyed mutex : `write` attend la fin + // de la copie avant de le livrer. + MiscFlags: (D3D11_RESOURCE_MISC_SHARED.0 | D3D11_RESOURCE_MISC_SHARED_NTHANDLE.0) as u32, + }; + let mut tex = None; + device.CreateTexture2D(&desc, None, Some(&mut tex))?; + let tex = tex.expect("CreateTexture2D a réussi sans texture"); + let resource: IDXGIResource1 = tex.cast()?; + let handle = resource.CreateSharedHandle( + None, + DXGI_SHARED_RESOURCE_READ.0 | DXGI_SHARED_RESOURCE_WRITE.0, + PCWSTR::null(), + )?; + Ok(SlotTexture { tex, handle, width, height }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn frame(gen: u64, slot: u32) -> SharedFrame { + SharedFrame { gen, slot, handle: 0, width: 2, height: 2, footage: None } + } + + /// Toutes les cases prises par JS, la case `n` portant la génération `n + 1`. + fn all_held(now: Instant) -> SlotBook { + let mut book = SlotBook::default(); + for slot in 0..RING_SLOTS { + book.publish(frame(u64::from(slot) + 1, slot)); + assert!(book.take(u64::from(slot), now).is_some()); + } + book + } + + #[test] + fn claims_a_free_slot_before_a_ready_one() { + let now = Instant::now(); + let mut book = SlotBook::default(); + book.publish(frame(1, 0)); + assert_eq!(book.claim(RING_SLOTS), Some(1)); + // La frame prête n'a pas bougé : JS peut toujours la prendre. + assert_eq!(book.take(0, now), Some(frame(1, 0))); + } + + #[test] + fn a_taken_slot_is_never_written_until_released() { + let now = Instant::now(); + let mut book = all_held(now); + assert_eq!(book.claim(RING_SLOTS), None, "Chromium tient toutes les cases"); + book.release(2, 3); + assert_eq!(book.claim(RING_SLOTS), Some(2)); + } + + #[test] + fn a_frame_nobody_took_gives_its_slot_back_when_nothing_else_is_free() { + let now = Instant::now(); + let mut book = SlotBook::default(); + for slot in 0..RING_SLOTS - 1 { + book.publish(frame(u64::from(slot) + 1, slot)); + book.take(u64::from(slot), now); + } + book.publish(frame(9, 3)); + assert_eq!(book.claim(RING_SLOTS), Some(3)); + // Retirée : la livrer maintenant, pendant qu'on la réécrit, donnerait une image déchirée. + assert_eq!(book.take(0, now), None); + } + + #[test] + fn take_hands_out_only_newer_generations() { + let now = Instant::now(); + let mut book = SlotBook::default(); + book.publish(frame(5, 1)); + assert_eq!(book.take(5, now), None); + assert_eq!(book.take(4, now), Some(frame(5, 1))); + assert_eq!(book.take(0, now), None, "déjà prise"); + } + + #[test] + fn a_release_only_frees_the_generation_it_names() { + let now = Instant::now(); + let mut book = SlotBook::default(); + book.publish(frame(1, 0)); + book.take(0, now); + book.release(3, 1); + book.release(0, 7); + assert_eq!(book.claim(1), None, "ni la case 3 ni la génération 7 ne sont tenues"); + book.release(0, 1); + assert_eq!(book.claim(1), Some(0)); + } + + #[test] + fn a_slot_whose_release_never_came_is_reported_lost_but_never_rewritten() { + let start = Instant::now(); + let mut book = all_held(start); + assert!(!book.lost(start + LOST_SLOT_AFTER / 2)); + assert!(book.lost(start + LOST_SLOT_AFTER)); + // Chromium peut encore lire une case perdue : la réécrire déchirerait son image. + assert_eq!(book.claim(RING_SLOTS), None); + // Si sa libération finit par venir, la case sert de nouveau. + book.release(0, 1); + assert_eq!(book.claim(RING_SLOTS), Some(0)); + } +} diff --git a/crates/compositor/tests/shared_frame_handoff.rs b/crates/compositor/tests/shared_frame_handoff.rs new file mode 100644 index 000000000..2465f2b98 --- /dev/null +++ b/crates/compositor/tests/shared_frame_handoff.rs @@ -0,0 +1,135 @@ +//! La texture qu'une vue live livre par handle NT, rouverte par un AUTRE device D3D11 — la +//! place qu'occupe le processus GPU de Chromium quand Electron l'importe. Ce que le +//! consommateur lit doit être exactement ce que le compositeur a composé : pas une image +//! à moitié copiée, pas une case périmée, et une case redimensionnée doit livrer un handle +//! neuf à la nouvelle taille. +//! +//! Demande un device matériel : saute sans lui (runner sans adaptateur). + +#![cfg(windows)] + +use openscreen_compositor::d3d::Gpu; +use openscreen_compositor::shared_frames::SharedRing; +use windows::core::Interface; +use windows::Win32::Foundation::{HANDLE, HMODULE}; +use windows::Win32::Graphics::Direct3D::{D3D_DRIVER_TYPE_HARDWARE, D3D_FEATURE_LEVEL_11_1}; +use windows::Win32::Graphics::Direct3D11::{ + D3D11CreateDevice, ID3D11Device, ID3D11Device1, ID3D11DeviceContext, ID3D11Texture2D, + D3D11_BIND_SHADER_RESOURCE, D3D11_CPU_ACCESS_READ, D3D11_CREATE_DEVICE_FLAG, + D3D11_MAPPED_SUBRESOURCE, D3D11_MAP_READ, D3D11_SDK_VERSION, D3D11_SUBRESOURCE_DATA, + D3D11_TEXTURE2D_DESC, D3D11_USAGE_DEFAULT, D3D11_USAGE_STAGING, +}; +use windows::Win32::Graphics::Dxgi::Common::{DXGI_FORMAT_R8G8B8A8_UNORM, DXGI_SAMPLE_DESC}; + +fn pattern(width: u32, height: u32, seed: u8) -> Vec { + (0..width * height) + .flat_map(|i| { + let (x, y) = (i % width, i / width); + [x as u8 ^ seed, y as u8, seed, 255] + }) + .collect() +} + +fn texture_with(gpu: &Gpu, width: u32, height: u32, pixels: &[u8]) -> ID3D11Texture2D { + let desc = D3D11_TEXTURE2D_DESC { + Width: width, + Height: height, + MipLevels: 1, + ArraySize: 1, + Format: DXGI_FORMAT_R8G8B8A8_UNORM, + SampleDesc: DXGI_SAMPLE_DESC { Count: 1, Quality: 0 }, + Usage: D3D11_USAGE_DEFAULT, + BindFlags: D3D11_BIND_SHADER_RESOURCE.0 as u32, + CPUAccessFlags: 0, + MiscFlags: 0, + }; + let init = D3D11_SUBRESOURCE_DATA { + pSysMem: pixels.as_ptr().cast(), + SysMemPitch: width * 4, + SysMemSlicePitch: 0, + }; + let mut tex = None; + unsafe { gpu.device.CreateTexture2D(&desc, Some(&init), Some(&mut tex)) }.expect("texture"); + tex.expect("texture") +} + +/// Le second device, sans rien de commun avec celui du compositeur que l'adaptateur. +fn consumer() -> (ID3D11Device, ID3D11DeviceContext) { + let mut device = None; + let mut context = None; + unsafe { + D3D11CreateDevice( + None, + D3D_DRIVER_TYPE_HARDWARE, + HMODULE::default(), + D3D11_CREATE_DEVICE_FLAG(0), + Some(&[D3D_FEATURE_LEVEL_11_1]), + D3D11_SDK_VERSION, + Some(&mut device), + None, + Some(&mut context), + ) + } + .expect("second device"); + (device.unwrap(), context.unwrap()) +} + +/// Ce que le consommateur lit à travers le handle, en RGBA serré. +fn read_through(handle: u64, width: u32, height: u32) -> Vec { + let (device, context) = consumer(); + let device1: ID3D11Device1 = device.cast().expect("ID3D11Device1"); + let shared: ID3D11Texture2D = + unsafe { device1.OpenSharedResource1(HANDLE(handle as *mut core::ffi::c_void)) } + .expect("le handle doit s'ouvrir depuis un autre device"); + let mut desc = D3D11_TEXTURE2D_DESC::default(); + unsafe { shared.GetDesc(&mut desc) }; + assert_eq!((desc.Width, desc.Height), (width, height)); + desc.Usage = D3D11_USAGE_STAGING; + desc.BindFlags = 0; + desc.CPUAccessFlags = D3D11_CPU_ACCESS_READ.0 as u32; + desc.MiscFlags = 0; + let mut staging = None; + unsafe { device.CreateTexture2D(&desc, None, Some(&mut staging)) }.expect("staging"); + let staging = staging.unwrap(); + let mut mapped = D3D11_MAPPED_SUBRESOURCE::default(); + let mut out = vec![0u8; (width * height * 4) as usize]; + unsafe { + context.CopyResource(&staging, &shared); + context.Map(&staging, 0, D3D11_MAP_READ, 0, Some(&mut mapped)).expect("map"); + for y in 0..height as usize { + std::ptr::copy_nonoverlapping( + (mapped.pData as *const u8).add(y * mapped.RowPitch as usize), + out.as_mut_ptr().add(y * width as usize * 4), + width as usize * 4, + ); + } + context.Unmap(&staging, 0); + } + out +} + +#[test] +fn another_device_reads_exactly_the_composed_frame() { + let Ok(gpu) = Gpu::create(false) else { + eprintln!("pas de device D3D11 matériel — saute"); + return; + }; + let mut ring = SharedRing::new(&gpu).expect("anneau"); + + let (w, h) = (64, 36); + let first = pattern(w, h, 0x11); + let handle = unsafe { ring.write(0, &texture_with(&gpu, w, h, &first), w, h) }.expect("écriture"); + assert_eq!(read_through(handle, w, h), first); + + // La même case réécrite : même texture, même handle, nouveau contenu. + let second = pattern(w, h, 0x22); + let again = unsafe { ring.write(0, &texture_with(&gpu, w, h, &second), w, h) }.expect("réécriture"); + assert_eq!(again, handle); + assert_eq!(read_through(again, w, h), second); + + // Le rendu change de taille : la case est recréée, avec un handle à la nouvelle taille. + let (w2, h2) = (32, 18); + let resized = pattern(w2, h2, 0x33); + let fresh = unsafe { ring.write(0, &texture_with(&gpu, w2, h2, &resized), w2, h2) }.expect("taille"); + assert_eq!(read_through(fresh, w2, h2), resized); +} diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index e0fddd35c..26f2f5308 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -37,6 +37,15 @@ interface Window { * `compositor.export`/`compositor.exportMulti` runs. Distinct from `exportOnFrameAck`, * the OLD web/CPU pipeline's per-frame ack, not a progress signal. */ onNativeExportProgress?: (callback: (frames: number, exportId?: string) => void) => () => void; + /** Preview frames the main process hands over as shared GPU textures (Windows). The + * listener draws `frame` and closes it; one listener at a time. Returns the + * unsubscribe. Optional: shim/web contexts have no bridge. */ + onCompositorFrame?: ( + listener: ( + frame: VideoFrame, + meta: import("../src/native/contracts").CompositorSharedFrameMeta, + ) => void, + ) => () => void; getSources: (opts: Electron.SourcesOptions) => Promise; switchToEditor: () => Promise; switchToHud: () => Promise; diff --git a/electron/ipc/nativeBridge.ts b/electron/ipc/nativeBridge.ts index 452f42701..61ed8ef86 100644 --- a/electron/ipc/nativeBridge.ts +++ b/electron/ipc/nativeBridge.ts @@ -384,18 +384,23 @@ export function registerNativeBridgeHandlers(context: NativeBridgeContext) { compositorViewService.setRect(request.payload.id, request.payload.rect); return createSuccessResponse(requestId, { ok: true }); case "readFrame": { - // The renderer polls this every rAF tick (~30fps). It passes the - // generation it last painted as `sinceGen`; native returns `null` when - // nothing newer exists (idle path — no buffer copy). On a new frame it - // returns `{ gen, width, height, data }`. The response wrapper does NOT - // JSON-stringify — `ipcMain.handle` round-trips via structured clone, - // which preserves the nested `Buffer` in `.data` as binary. - const frame = compositorViewService.readFrame( + // The renderer polls this every rAF tick. It passes the generation it + // last painted as `sinceGen`; native returns `null` when nothing newer + // exists (idle path — no buffer copy). On a new frame it returns + // `{ gen, width, height, data }`, or — for a view on shared textures — + // sends the texture to the asking frame and returns its receipt. The + // response wrapper does NOT JSON-stringify — `ipcMain.handle` round-trips + // via structured clone, which preserves the nested `Buffer` as binary. + const frame = await compositorViewService.readFrame( request.payload.id, request.payload.sinceGen, + event.senderFrame, ); return createSuccessResponse(requestId, frame); } + case "stopSharedFrames": + compositorViewService.stopSharedFrames(request.payload.id); + return createSuccessResponse(requestId, { ok: true }); case "setParam": compositorViewService.setParam( request.payload.id, diff --git a/electron/native-bridge/services/compositorViewService.test.ts b/electron/native-bridge/services/compositorViewService.test.ts index eee8fe305..4a099a7f9 100644 --- a/electron/native-bridge/services/compositorViewService.test.ts +++ b/electron/native-bridge/services/compositorViewService.test.ts @@ -1,6 +1,7 @@ import fs from "node:fs"; import os from "node:os"; import path from "node:path"; +import type { WebFrameMain } from "electron"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; import { CURSOR_THEMES, DEFAULT_CURSOR_SPRITES } from "../../../src/lib/cursor/cursorThemes"; import type { CompositorViewAddon, GifExportStats } from "../../native/compositor-view/addon"; @@ -58,6 +59,166 @@ describe("native GIF cancellation capability", () => { }); }); +describe("CompositorViewService frames handed over as shared GPU textures", () => { + const target = { frameTreeNodeId: 1 } as unknown as WebFrameMain; + const sharedFrame = { + gen: 3, + slot: 1, + handle: Buffer.from([1, 2, 3, 4, 5, 6, 7, 8]), + width: 4, + height: 2, + footage: null, + footageProjective: false, + }; + + function setup(options: { addon?: Partial; gpuCompositing?: boolean } = {}) { + const addon = { + createView: vi.fn(() => 7), + setSharedFrames: vi.fn(() => true), + readSharedFrame: vi.fn(() => sharedFrame), + readFrame: vi.fn(() => null), + releaseSharedFrame: vi.fn(), + destroyView: vi.fn(), + ...options.addon, + }; + const imported = { release: vi.fn() }; + let allReferencesReleased: (() => void) | undefined; + const sharedTexture = { + importSharedTexture: vi.fn((importOptions: Electron.ImportSharedTextureOptions) => { + allReferencesReleased = importOptions.allReferencesReleased as () => void; + return imported as unknown as Electron.SharedTextureImported; + }), + sendSharedTexture: vi.fn(async () => undefined), + }; + const service = new CompositorViewService({ + addon: addon as unknown as CompositorViewAddon, + sharedTexture, + gpuCompositing: () => options.gpuCompositing ?? true, + }); + const id = service.createView({ x: 0, y: 0, width: 4, height: 2 }); + return { + addon, + id, + imported, + service, + sharedTexture, + releaseEverywhere: () => allReferencesReleased?.(), + }; + } + + afterEach(() => { + vi.unstubAllEnvs(); + }); + + it("sends the texture to the frame that asked, and answers with its receipt", async () => { + const { addon, id, imported, service, sharedTexture } = setup(); + expect(addon.setSharedFrames).toHaveBeenCalledWith(id, true); + + const receipt = await service.readFrame(id, 2, target); + + expect(addon.readSharedFrame).toHaveBeenCalledWith(id, 2); + expect(sharedTexture.importSharedTexture).toHaveBeenCalledWith( + expect.objectContaining({ + textureInfo: { + pixelFormat: "rgba", + codedSize: { width: 4, height: 2 }, + handle: { ntHandle: sharedFrame.handle }, + }, + }), + ); + const meta = { + viewId: id, + gen: 3, + width: 4, + height: 2, + footage: null, + footageProjective: false, + }; + expect(sharedTexture.sendSharedTexture).toHaveBeenCalledWith( + { frame: target, importedSharedTexture: imported }, + meta, + ); + expect(receipt).toEqual({ ...meta, shared: true }); + // This process's reference goes at once: the renderer holds its own. + expect(imported.release).toHaveBeenCalled(); + expect(addon.releaseSharedFrame).not.toHaveBeenCalled(); + }); + + it("hands the slot back once Chromium has let go of the texture everywhere", async () => { + const { addon, id, releaseEverywhere, service } = setup(); + await service.readFrame(id, 0, target); + + releaseEverywhere(); + + expect(addon.releaseSharedFrame).toHaveBeenCalledWith(id, 1, 3); + }); + + it("goes back to read-back when a delivery fails", async () => { + const { addon, id, service, sharedTexture } = setup(); + sharedTexture.sendSharedTexture.mockRejectedValueOnce(new Error("timed out after 1000ms")); + const warn = vi.spyOn(console, "warn").mockImplementation(() => undefined); + try { + expect(await service.readFrame(id, 0, target)).toBeNull(); + } finally { + warn.mockRestore(); + } + expect(addon.setSharedFrames).toHaveBeenLastCalledWith(id, false); + + vi.mocked(addon.readSharedFrame).mockClear(); + await service.readFrame(id, 3, target); + expect(addon.readSharedFrame).not.toHaveBeenCalled(); + expect(addon.readFrame).toHaveBeenCalledWith(id, 3); + }); + + it("frees the slot itself when the texture never got imported", async () => { + const { addon, id, service, sharedTexture } = setup(); + sharedTexture.importSharedTexture.mockImplementationOnce(() => { + throw new TypeError("Invalid ntHandle value"); + }); + const warn = vi.spyOn(console, "warn").mockImplementation(() => undefined); + try { + await service.readFrame(id, 0, target); + } finally { + warn.mockRestore(); + } + expect(addon.releaseSharedFrame).toHaveBeenCalledWith(id, 1, 3); + }); + + it("reads pixels for a view whose render thread went back to read-back on its own", async () => { + const packet = { gen: 4, width: 4, height: 2, data: Buffer.alloc(32) }; + const { id, service, sharedTexture } = setup({ + addon: { readSharedFrame: vi.fn(() => null), readFrame: vi.fn(() => packet) }, + }); + + expect(await service.readFrame(id, 3, target)).toBe(packet); + expect(sharedTexture.sendSharedTexture).not.toHaveBeenCalled(); + }); + + it("keeps a view on read-back when Chromium does not composite on the GPU", async () => { + const { addon, id, service } = setup({ gpuCompositing: false }); + expect(addon.setSharedFrames).not.toHaveBeenCalled(); + + await service.readFrame(id, 0, target); + expect(addon.readSharedFrame).not.toHaveBeenCalled(); + expect(addon.readFrame).toHaveBeenCalledWith(id, 0); + }); + + it("keeps a view on read-back when OPENSCREEN_PREVIEW_READBACK=1", () => { + vi.stubEnv("OPENSCREEN_PREVIEW_READBACK", "1"); + const { addon } = setup(); + expect(addon.setSharedFrames).not.toHaveBeenCalled(); + }); + + it("keeps a view on read-back when the addon cannot share (not Windows, software backend)", async () => { + const { id, service, sharedTexture } = setup({ + addon: { setSharedFrames: vi.fn(() => false) }, + }); + + await service.readFrame(id, 0, target); + expect(sharedTexture.importSharedTexture).not.toHaveBeenCalled(); + }); +}); + /** A source checkout's `crates/.cargo/config.toml`, with `FFMPEG_DIR` written as `body`. */ function writeCargoConfig(root: string, body: string): void { const cargoDir = path.join(root, "crates", ".cargo"); diff --git a/electron/native-bridge/services/compositorViewService.ts b/electron/native-bridge/services/compositorViewService.ts index 4a765db2b..51e562979 100644 --- a/electron/native-bridge/services/compositorViewService.ts +++ b/electron/native-bridge/services/compositorViewService.ts @@ -2,12 +2,16 @@ import fs from "node:fs"; import { createRequire } from "node:module"; import path from "node:path"; import { fileURLToPath } from "node:url"; -import { app } from "electron"; +import { app, sharedTexture, type WebFrameMain } from "electron"; import { type CursorKind, readCursorAsArrow, resolveCursorSprites, } from "../../../src/lib/cursor/cursorThemes"; +import type { + CompositorSharedFrameMeta, + CompositorSharedFrameReceipt, +} from "../../../src/native/contracts"; import type { GifExportJob } from "../../ipc/gifExportJobs"; import type { ClipInput, @@ -20,6 +24,7 @@ import type { GifExportStats, GifParamsInput, NativeFramePacket, + NativeSharedFramePacket, RemuxStats, SegmentationSupport, } from "../../native/compositor-view/addon"; @@ -215,6 +220,26 @@ export interface CompositorViewServiceOptions { */ appRoot?: string; isPackaged?: boolean; + /** Electron's `sharedTexture` module, injectable for tests. `null` keeps every view on + * read-back. */ + sharedTexture?: SharedTextureApi | null; + /** Whether Chromium composites on the GPU, where a shared texture is imported. Injectable + * for tests; defaults to `app.getGPUFeatureStatus()`. */ + gpuCompositing?: () => boolean; +} + +/** The two calls a shared preview frame needs from Electron's `sharedTexture` module. */ +export type SharedTextureApi = Pick< + Electron.SharedTexture, + "importSharedTexture" | "sendSharedTexture" +>; + +function defaultGpuCompositing(): boolean { + try { + return String(app.getGPUFeatureStatus().gpu_compositing).startsWith("enabled"); + } catch { + return false; + } } function defaultAppRoot(): string { @@ -467,6 +492,8 @@ function tryLoadAddon(candidates: string[]): CompositorViewAddon | null { export class CompositorViewService { private readonly options: CompositorViewServiceOptions; private readonly rects = new Map(); + /** Views whose frames go out as shared GPU textures rather than RAM pixels. */ + private readonly sharedViews = new Set(); private addon: CompositorViewAddon | null = null; private loadAttempted = false; private syntheticIdCounter = 0; @@ -593,9 +620,41 @@ export class CompositorViewService { } const id = addon.createView(rect, paths?.screenPath, paths?.webcamPath, paths?.cursorPath); this.rects.set(id, rect); + this.shareFrames(addon, id); return id; } + /** Hands the view's frames over as shared GPU textures when this host can, instead of + * copying them through IPC: measured, that transport alone kept 37 to 55 % of the + * renderer's main thread busy at 30 fps (rendering-performance.md). Anything missing — + * the Electron API, GPU compositing, Windows' hardware backend native-side — leaves the + * view on read-back. `OPENSCREEN_PREVIEW_READBACK=1` forces read-back, to compare. */ + private shareFrames(addon: CompositorViewAddon, id: number): void { + if (process.env.OPENSCREEN_PREVIEW_READBACK === "1" || !this.sharedTextureApi()) { + return; + } + if (!(this.options.gpuCompositing ?? defaultGpuCompositing)()) { + return; + } + if (addon.setSharedFrames?.(id, true)) { + this.sharedViews.add(id); + } + } + + private sharedTextureApi(): SharedTextureApi | null { + const api = + this.options.sharedTexture === undefined ? sharedTexture : this.options.sharedTexture; + return typeof api?.importSharedTexture === "function" ? api : null; + } + + /** Takes the view back to read-back frames: a delivery failed, or the renderer saw a shared + * frame land as nothing (Chromium could not open the texture). The render thread + * republishes the current frame, so the canvas is not left empty. */ + stopSharedFrames(id: number): void { + this.sharedViews.delete(id); + this.ensureAddon()?.setSharedFrames?.(id, false); + } + setRect(id: number, rect: CompositorViewRect): void { const addon = this.ensureAddon(); this.rects.set(id, rect); @@ -605,19 +664,76 @@ export class CompositorViewService { addon.setRect(id, rect); } - /** Reads the most recently rendered frame for `id` as a self-describing packet - * (`{ gen, width, height, data }`), but only if its generation is newer than - * `sinceGen`. Returns `null` when the addon is absent, no frame is ready yet, - * OR the caller already holds the current generation — the idle path, where - * `null` comes back without any buffer copy. Byte order is RGBA. */ - readFrame(id: number, sinceGen: number): NativeFramePacket | null { + /** Reads the most recently rendered frame for `id`, but only if its generation is newer + * than `sinceGen`. Returns `null` when the addon is absent, no frame is ready yet, OR the + * caller already holds the current generation — the idle path, where nothing is copied. + * + * A view on shared textures sends the frame to `target` as a GPU texture, and answers with + * its receipt (`shared: true`, no pixels): the texture reaches the renderer before this + * reply does. Any other view answers with the RGBA pixels themselves. */ + async readFrame( + id: number, + sinceGen: number, + target?: WebFrameMain | null, + ): Promise { const addon = this.ensureAddon(); if (!addon) { return null; } + if (target && this.sharedViews.has(id)) { + const frame = addon.readSharedFrame?.(id, sinceGen); + if (frame) { + return this.sendSharedFrame(addon, id, frame, target); + } + } + // Also the answer for a view whose render thread went back to read-back on its own. return addon.readFrame(id, sinceGen); } + private async sendSharedFrame( + addon: CompositorViewAddon, + id: number, + frame: NativeSharedFramePacket, + target: WebFrameMain, + ): Promise { + const meta: CompositorSharedFrameMeta = { + viewId: id, + gen: frame.gen, + width: frame.width, + height: frame.height, + footage: frame.footage ?? null, + footageProjective: frame.footageProjective ?? false, + }; + const api = this.sharedTextureApi(); + let imported: Electron.SharedTextureImported | undefined; + try { + if (!api) { + throw new Error("the sharedTexture API is gone"); + } + imported = api.importSharedTexture({ + textureInfo: { + pixelFormat: "rgba", + codedSize: { width: frame.width, height: frame.height }, + handle: { ntHandle: frame.handle }, + }, + // Chromium is done with the texture in every process: the slot may be written again. + allReferencesReleased: () => addon.releaseSharedFrame?.(id, frame.slot, frame.gen), + }); + await api.sendSharedTexture({ frame: target, importedSharedTexture: imported }, meta); + return { ...meta, shared: true }; + } catch (error) { + console.warn("[compositor-view] shared frame not delivered; reading frames back:", error); + this.stopSharedFrames(id); + if (!imported) { + addon.releaseSharedFrame?.(id, frame.slot, frame.gen); + } + return null; + } finally { + // This process's reference only: the renderer holds its own until it has drawn. + imported?.release(); + } + } + setParam(id: number, key: string, value: CompositorParamValue): void { const addon = this.ensureAddon(); if (!addon) { @@ -668,6 +784,7 @@ export class CompositorViewService { destroyView(id: number): void { const addon = this.ensureAddon(); this.rects.delete(id); + this.sharedViews.delete(id); if (!addon) { return; } diff --git a/electron/native/compositor-view/addon.d.ts b/electron/native/compositor-view/addon.d.ts index ee29cd80f..631d70fdf 100644 --- a/electron/native/compositor-view/addon.d.ts +++ b/electron/native/compositor-view/addon.d.ts @@ -41,6 +41,20 @@ export interface NativeFramePacket { footageProjective?: boolean; } +/** A preview frame left in a shared GPU texture instead of copied into RAM (Windows, hardware + * backend — see `setSharedFrames`). `handle` is the texture's NT handle the way Electron's + * `sharedTexture.importSharedTexture` takes it: 8 bytes, little-endian, valid in this process + * only. Its `slot` stays reserved until `releaseSharedFrame(id, slot, gen)`. */ +export interface NativeSharedFramePacket { + gen: number; + slot: number; + handle: Buffer; + width: number; + height: number; + footage?: number[] | null; + footageProjective?: boolean; +} + export interface ExportStats { frames: number; wallS: number; @@ -158,6 +172,17 @@ export interface CompositorViewAddon { * still frame): `null` comes back WITHOUT cloning the buffer or crossing IPC. * Pass `sinceGen = 0` to force delivery of the current frame. */ readFrame(id: number, sinceGen: number): NativeFramePacket | null; + /** Deliver this view's frames as shared GPU textures (`readSharedFrame`) rather than RAM + * pixels (`readFrame`). Returns `false` where that cannot work (not Windows, software + * backend): the view keeps reading back. Optional: an older `.node` predates it. */ + setSharedFrames?(id: number, enabled: boolean): boolean; + /** The latest frame left in the view's shared texture ring, if newer than `sinceGen`. + * `null` too while the view reads back to RAM. Throws the render thread's fatal error, + * like `readFrame`. */ + readSharedFrame?(id: number, sinceGen: number): NativeSharedFramePacket | null; + /** Chromium let go of frame `gen` in `slot`, in every process: the render thread may write + * that slot again. A no-op for a destroyed view. */ + releaseSharedFrame?(id: number, slot: number, gen: number): void; setParam(id: number, key: string, value: CompositorParamValue): void; setPlaying(id: number, playing: boolean): void; /** Seeks the view to source-media `seconds` for the active clip. */ diff --git a/electron/preload.ts b/electron/preload.ts index a754c99b6..feb3942df 100644 --- a/electron/preload.ts +++ b/electron/preload.ts @@ -1,4 +1,4 @@ -import { contextBridge, ipcRenderer, webUtils } from "electron"; +import { contextBridge, ipcRenderer, sharedTexture, webUtils } from "electron"; import type { NativeLinuxRecordingRequest } from "../src/lib/nativeLinuxRecording"; import type { NativeMacRecordingRequest } from "../src/lib/nativeMacRecording"; import type { NativeWindowsRecordingRequest } from "../src/lib/nativeWindowsRecording"; @@ -8,6 +8,7 @@ import type { AiEditionChatEvent, AiEditionMcpHostRequest, AiEditionMcpHostResponse, + CompositorSharedFrameMeta, } from "../src/native/contracts"; import { AI_EDITION_MCP_HOST_CHANNEL, @@ -34,6 +35,23 @@ const assetBaseUrl = assetBaseUrlArg ? assetBaseUrlArg.slice(ASSET_BASE_URL_ARG_ // so a synchronous read here saves the renderer's every-call IPC round-trip. const PLATFORM = process.platform; +// Preview frames the main process hands over as shared GPU textures (see +// `compositorViewService`). Electron gives up on a send after 1 s without a receiver, so it is +// registered here, at load, and forwards to whichever listener the page has set. +type CompositorFrameListener = (frame: VideoFrame, meta: CompositorSharedFrameMeta) => void; +let compositorFrameListener: CompositorFrameListener | null = null; +sharedTexture?.setSharedTextureReceiver(async ({ importedSharedTexture }, meta) => { + const frame = importedSharedTexture.getVideoFrame(); + try { + compositorFrameListener?.(frame, meta as CompositorSharedFrameMeta); + } finally { + // The page drew from its own copy (contextBridge clones the frame) and closed it. Closing + // ours and releasing the import is what lets the native ring write the slot again. + frame.close(); + importedSharedTexture.release(); + } +}); + contextBridge.exposeInMainWorld("electronAPI", { assetBaseUrl, @@ -85,6 +103,14 @@ contextBridge.exposeInMainWorld("electronAPI", { ipcRenderer.on("export:native-progress", handler); return () => ipcRenderer.off("export:native-progress", handler); }, + onCompositorFrame: (listener: CompositorFrameListener) => { + compositorFrameListener = listener; + return () => { + if (compositorFrameListener === listener) { + compositorFrameListener = null; + } + }; + }, invokeNativeBridge: (request: NativeBridgeRequest) => { return ipcRenderer.invoke(NATIVE_BRIDGE_CHANNEL, request) as Promise; }, diff --git a/src/native/compositorViewClient.ts b/src/native/compositorViewClient.ts index 793521086..66f63ce56 100644 --- a/src/native/compositorViewClient.ts +++ b/src/native/compositorViewClient.ts @@ -18,6 +18,8 @@ import type { CompositorExportResult, CompositorFramePacket, CompositorParamValue, + CompositorSharedFrameMeta, + CompositorSharedFrameReceipt, CompositorViewRect, CompositorViewResult, SegmentationSupport, @@ -97,18 +99,39 @@ export function setCompositorRect(id: number, rect: CompositorViewRect): Promise * {@link CompositorFramePacket} on a new frame, or `null` when the addon is absent, * no frame is ready yet, OR the caller already holds the current generation — the * idle path, where `null` returns without any buffer crossing IPC. Pass `sinceGen = 0` - * to force delivery of the current frame. */ + * to force delivery of the current frame. + * + * A view on shared textures answers with a {@link CompositorSharedFrameReceipt} instead: + * its frame already went to {@link subscribeCompositorSharedFrames}, ahead of this reply. */ export function readCompositorFrame( id: number, sinceGen: number, -): Promise { - return requireNativeBridgeData({ +): Promise { + return requireNativeBridgeData({ domain: "compositor", action: "readFrame", payload: { id, sinceGen }, }); } +/** Preview frames handed over as shared GPU textures. The listener draws `frame` and closes + * it. Returns the unsubscribe; without the Electron bridge (pure web, jsdom) nothing ever + * arrives. */ +export function subscribeCompositorSharedFrames( + listener: (frame: VideoFrame, meta: CompositorSharedFrameMeta) => void, +): () => void { + return window.electronAPI?.onCompositorFrame?.(listener) ?? (() => undefined); +} + +/** Back to read-back frames for `id`: a shared frame reached the canvas as nothing. */ +export function stopSharedCompositorFrames(id: number): Promise<{ ok: true }> { + return requireNativeBridgeData<{ ok: true }>({ + domain: "compositor", + action: "stopSharedFrames", + payload: { id }, + }); +} + export function setCompositorParam( id: number, key: string, diff --git a/src/native/contracts.ts b/src/native/contracts.ts index 432a6ca69..d13a95f5c 100644 --- a/src/native/contracts.ts +++ b/src/native/contracts.ts @@ -167,6 +167,24 @@ export interface CompositorFramePacket { footageProjective?: boolean; } +/** What travels with a preview frame handed over as a shared GPU texture (Windows): all a + * {@link CompositorFramePacket} says but the pixels, plus the view the frame belongs to. The + * main process sends it alongside the texture, to `electronAPI.onCompositorFrame`. */ +export interface CompositorSharedFrameMeta { + viewId: number; + gen: number; + width: number; + height: number; + footage: number[] | null; + footageProjective: boolean; +} + +/** `readFrame`'s answer for a frame sent as a shared texture. The texture reached + * `onCompositorFrame` before this reply did, so only the generation is news here. */ +export interface CompositorSharedFrameReceipt extends CompositorSharedFrameMeta { + shared: true; +} + /** Un clip de la timeline pour l'export multiclip natif (fichiers screen+webcam + trim). */ export interface CompositorClipInput { screenPath: string; @@ -820,6 +838,14 @@ export type NativeBridgeRequest = payload: { id: number; sinceGen: number }; requestId?: string; } + | { + domain: "compositor"; + /** Back to read-back frames: a shared frame reached the canvas as nothing, i.e. + * Chromium could not open the texture. */ + action: "stopSharedFrames"; + payload: { id: number }; + requestId?: string; + } | { domain: "compositor"; action: "setParam"; diff --git a/src/native/hooks/useNativeCompositorView.test.ts b/src/native/hooks/useNativeCompositorView.test.ts index 55b89174b..4949bb5b7 100644 --- a/src/native/hooks/useNativeCompositorView.test.ts +++ b/src/native/hooks/useNativeCompositorView.test.ts @@ -15,12 +15,18 @@ import { renderHook, waitFor } from "@testing-library/react"; import type { RefObject } from "react"; import { beforeEach, describe, expect, it, vi } from "vitest"; +import type { CompositorSharedFrameMeta } from "../contracts"; + +type SharedFrameListener = (frame: VideoFrame, meta: CompositorSharedFrameMeta) => void; const mocks = vi.hoisted(() => ({ createCompositorView: vi.fn(), readCompositorFrame: vi.fn(), destroyCompositorView: vi.fn(), setCompositorRect: vi.fn(async () => undefined), + stopSharedCompositorFrames: vi.fn(async () => ({ ok: true })), + /** What the preload would call with each frame sent as a shared texture. */ + sharedListener: null as SharedFrameListener | null, })); vi.mock("../compositorViewClient", () => ({ @@ -30,6 +36,15 @@ vi.mock("../compositorViewClient", () => ({ setCompositorParam: vi.fn(), setCompositorPlaying: vi.fn(), setCompositorRect: mocks.setCompositorRect, + stopSharedCompositorFrames: mocks.stopSharedCompositorFrames, + subscribeCompositorSharedFrames: (listener: SharedFrameListener) => { + mocks.sharedListener = listener; + return () => { + if (mocks.sharedListener === listener) { + mocks.sharedListener = null; + } + }; + }, })); import { useNativeCompositorView } from "./useNativeCompositorView"; @@ -48,22 +63,50 @@ globalThis.ResizeObserver = class { } } as unknown as typeof ResizeObserver; -/** A canvas with a stubbed 2D context — jsdom has none, and the pull loop bails without it. */ -function stubCanvasRef(): RefObject { +/** A canvas with a stubbed 2D context — jsdom has none, and the pull loop bails without it. + * `centreAlpha` is what reading back the middle pixel answers: opaque, as a composed frame + * always is, unless a test says otherwise. */ +function stubCanvasRef(centreAlpha = 255): RefObject { const canvas = document.createElement("canvas"); - canvas.getContext = vi.fn(() => ({ + const ctx = { drawImage: vi.fn(), putImageData: vi.fn(), - })) as unknown as HTMLCanvasElement["getContext"]; + getImageData: vi.fn(() => ({ data: new Uint8ClampedArray([0, 0, 0, centreAlpha]) })), + }; + canvas.getContext = vi.fn(() => ctx) as unknown as HTMLCanvasElement["getContext"]; return { current: canvas }; } +function context(ref: RefObject) { + return ref.current?.getContext("2d") as unknown as { + drawImage: ReturnType; + getImageData: ReturnType; + }; +} + +function sharedMeta(overrides: Partial = {}): CompositorSharedFrameMeta { + return { + viewId: 7, + gen: 3, + width: 4, + height: 2, + footage: null, + footageProjective: false, + ...overrides, + }; +} + +function fakeVideoFrame() { + return { close: vi.fn() } as unknown as VideoFrame & { close: ReturnType }; +} + const DEVICE_FAILURE = "this display adapter has no D3D11 video decoder (0x887A0004). OpenScreen decodes every preview and export frame with D3D11VA"; describe("useNativeCompositorView", () => { beforeEach(() => { vi.clearAllMocks(); + mocks.sharedListener = null; }); it("surfaces the native message when the render thread dies", async () => { @@ -317,4 +360,250 @@ describe("useNativeCompositorView", () => { resolveSecond({ id: 8 }); await waitFor(() => expect(result.current.viewId).toBe(8)); }); + + describe("frames handed over as shared GPU textures", () => { + it("draws the frame on the canvas, sized to it, and closes it", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame.mockResolvedValue(null); + const ref = stubCanvasRef(); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + + const frame = fakeVideoFrame(); + mocks.sharedListener?.(frame, sharedMeta()); + + expect(context(ref).drawImage).toHaveBeenCalledWith(frame, 0, 0); + expect(ref.current?.width).toBe(4); + expect(ref.current?.height).toBe(2); + expect(ref.current?.dataset.painted).toBe("true"); + expect(frame.close).toHaveBeenCalled(); + expect(mocks.stopSharedCompositorFrames).not.toHaveBeenCalled(); + }); + + it("closes a frame meant for another view without drawing it", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame.mockResolvedValue(null); + const ref = stubCanvasRef(); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + + const frame = fakeVideoFrame(); + mocks.sharedListener?.(frame, sharedMeta({ viewId: 8 })); + + expect(context(ref).drawImage).not.toHaveBeenCalled(); + expect(frame.close).toHaveBeenCalled(); + }); + + // Chromium opens the texture in its own GPU process. One it cannot open — a hybrid + // laptop's other adapter — draws as nothing, and the preview would stay empty. + it("goes back to read-back frames when the first one lands as nothing", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame.mockResolvedValue(null); + const ref = stubCanvasRef(0); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta()); + + expect(mocks.stopSharedCompositorFrames).toHaveBeenCalledWith(7); + }); + + it("checks only the view's first frame, not every frame", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame.mockResolvedValue(null); + const ref = stubCanvasRef(); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta({ gen: 3 })); + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta({ gen: 4 })); + + // A readback stalls the GPU pipeline: one per view, not one per frame. + expect(context(ref).getImageData).toHaveBeenCalledTimes(1); + }); + + it("takes a receipt as the generation to ask after, with nothing to draw", async () => { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + mocks.readCompositorFrame + .mockResolvedValueOnce({ ...sharedMeta({ gen: 5 }), shared: true }) + .mockResolvedValue(null); + const ref = stubCanvasRef(); + renderHook(() => useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } })); + + await waitFor(() => expect(mocks.readCompositorFrame).toHaveBeenCalledWith(7, 5)); + expect(context(ref).drawImage).not.toHaveBeenCalled(); + }); + }); + + // The display sets the rAF rate and the recording the frame rate. Counting ticks pulled 280 + // times a second on a 280 Hz display, idle or not. + describe("pull cadence, by the clock rather than by ticks", () => { + /** rAF driven by hand, at `periodMs`: callbacks run with the timestamps a display of that + * rate would give them. */ + function manualFrames() { + let queue: FrameRequestCallback[] = []; + vi.stubGlobal("requestAnimationFrame", (callback: FrameRequestCallback) => { + queue.push(callback); + return queue.length; + }); + vi.stubGlobal("cancelAnimationFrame", () => undefined); + return async (durationMs: number, periodMs: number, start = 0) => { + for (let t = start; t < start + durationMs; t += periodMs) { + const due = queue; + queue = []; + for (const callback of due) { + callback(t); + } + // Let each pull's reply land before the next tick, as it would between frames. + for (let flush = 0; flush < 4; flush++) { + await Promise.resolve(); + } + } + }; + } + + async function mountedView() { + mocks.createCompositorView.mockResolvedValue({ id: 7 }); + const ref = stubCanvasRef(); + const { result } = renderHook(() => + useNativeCompositorView(ref, { sources: { screenPath: "rec.mp4" } }), + ); + await waitFor(() => expect(result.current.viewId).toBe(7)); + mocks.readCompositorFrame.mockClear(); + } + + it("pulls ~30 times a second while nothing new comes, whatever the display rate", async () => { + const run = manualFrames(); + try { + mocks.readCompositorFrame.mockResolvedValue(null); + await mountedView(); + + await run(1000, 1000 / 280); + + // One pull every 9 ticks of 3.6 ms: 32 in the second, against 280 counting ticks. + const pulls = mocks.readCompositorFrame.mock.calls.length; + expect(pulls).toBeGreaterThanOrEqual(28); + expect(pulls).toBeLessThanOrEqual(33); + } finally { + vi.unstubAllGlobals(); + } + }); + + it("pulls every 8 ms or so while shared frames keep coming, and slows down once they stop", async () => { + const run = manualFrames(); + try { + let gen = 0; + mocks.readCompositorFrame.mockImplementation(async () => ({ + ...sharedMeta({ gen: ++gen }), + shared: true, + })); + await mountedView(); + // The first frame marks the transport as shared, as the preload's listener does. + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta({ gen: 1 })); + + await run(1000, 1000 / 280); + const flowing = mocks.readCompositorFrame.mock.calls.length; + expect(flowing).toBeGreaterThanOrEqual(100); + expect(flowing).toBeLessThanOrEqual(150); + + // Playback stops: nothing new comes any more. Past the flowing window the loop is + // back to ~30 pulls a second. + mocks.readCompositorFrame.mockReset(); + mocks.readCompositorFrame.mockResolvedValue(null); + await run(400, 1000 / 280, 1000); + mocks.readCompositorFrame.mockClear(); + await run(500, 1000 / 280, 1400); + expect(mocks.readCompositorFrame.mock.calls.length).toBeLessThanOrEqual(17); + } finally { + vi.unstubAllGlobals(); + } + }); + + // The service turns shared textures off for a view whose import or send failed: its next + // frames are read-back pixels, and the fast cadence must not outlive the transport. + it("drops back to ~30 pulls a second once the view falls back to read-back", async () => { + const run = manualFrames(); + vi.stubGlobal( + "ImageData", + class { + constructor( + public data: Uint8ClampedArray, + public width: number, + public height: number, + ) {} + }, + ); + vi.stubGlobal( + "createImageBitmap", + vi.fn(async () => ({ close: vi.fn() })), + ); + try { + let gen = 0; + mocks.readCompositorFrame.mockImplementation(async () => ({ + ...sharedMeta({ gen: ++gen }), + shared: true, + })); + await mountedView(); + mocks.sharedListener?.(fakeVideoFrame(), sharedMeta({ gen: 1 })); + await run(300, 1000 / 280); + + mocks.readCompositorFrame.mockImplementation(async () => ({ + gen: ++gen, + width: 2, + height: 1, + data: new Uint8Array(8), + })); + await run(100, 1000 / 280, 300); + mocks.readCompositorFrame.mockClear(); + await run(500, 1000 / 280, 400); + + expect(mocks.readCompositorFrame.mock.calls.length).toBeLessThanOrEqual(17); + } finally { + vi.unstubAllGlobals(); + } + }); + + // Read-back frames bound the copies: the fast cadence is for shared textures only. + it("keeps read-back frames at ~30 pulls a second even while they keep coming", async () => { + const run = manualFrames(); + vi.stubGlobal( + "ImageData", + class { + constructor( + public data: Uint8ClampedArray, + public width: number, + public height: number, + ) {} + }, + ); + vi.stubGlobal( + "createImageBitmap", + vi.fn(async () => ({ close: vi.fn() })), + ); + try { + let gen = 0; + mocks.readCompositorFrame.mockImplementation(async () => ({ + gen: ++gen, + width: 2, + height: 1, + data: new Uint8Array(8), + })); + await mountedView(); + + await run(1000, 1000 / 280); + + expect(mocks.readCompositorFrame.mock.calls.length).toBeLessThanOrEqual(33); + } finally { + vi.unstubAllGlobals(); + } + }); + }); }); diff --git a/src/native/hooks/useNativeCompositorView.ts b/src/native/hooks/useNativeCompositorView.ts index 416b57c77..94688c2bf 100644 --- a/src/native/hooks/useNativeCompositorView.ts +++ b/src/native/hooks/useNativeCompositorView.ts @@ -5,13 +5,17 @@ * rect (measured via ResizeObserver + window resize/scroll, rAF-coalesced * — the exact same sync machinery as before, repurposed: it now drives * the offscreen render-target resolution instead of a window position). - * 2. Polls `readCompositorFrame` on every other rAF tick (~30fps), passing the - * generation it last painted. Native returns a self-describing packet - * (`{ gen, width, height, data }`) ONLY when a newer frame exists — otherwise - * `null`, and the canvas is left untouched. So while the preview sits still - * (paused editing) nothing is cloned, sent over IPC, or repainted. The canvas - * drawing buffer is sized from the packet's own dims, so pixels and canvas - * can never drift apart. + * 2. Polls `readCompositorFrame` on rAF ticks, passing the generation it last + * painted. Native answers ONLY when a newer frame exists — otherwise `null`, and + * the canvas is left untouched, so while the preview sits still (paused editing) + * nothing is cloned, sent over IPC, or repainted. A new frame comes one of two ways: + * - as a shared GPU texture (Windows), sent to `subscribeCompositorSharedFrames` + * ahead of the reply, which then only names its generation. Nothing is copied + * through RAM, so while frames keep coming it is pulled every 8 ms; + * - as RGBA pixels (`{ gen, width, height, data }`) everywhere else, pulled ~30 + * times a second to bound the copies. + * Either way the canvas drawing buffer is sized from the frame's own dims, so pixels + * and canvas can never drift apart. * * Every native-bridge call is wrapped in a try/catch that swallows + warns, * because the renderer may run without the bridge (pure web `npm run dev`, @@ -28,6 +32,8 @@ import { setCompositorParam, setCompositorPlaying, setCompositorRect, + stopSharedCompositorFrames, + subscribeCompositorSharedFrames, } from "../compositorViewClient"; import type { CompositorParamValue, CompositorViewRect } from "../contracts"; import { publishFootageQuad } from "../footageQuadStore"; @@ -78,10 +84,24 @@ function safelyCall(label: string, call: () => Promise) { } } -/** Throttle the rAF pull loop to roughly 30fps: process every other animation - * frame. Keeps IPC + GPU readback + putImageData cheap on high-refresh - * displays (120/144 Hz) without changing perceived preview smoothness. */ -const PULL_LOOP_TICK_DIVISOR = 2; +/** Least time between two pulls, by the clock rather than in animation frames: the display + * sets the rAF rate, the recording sets the frame rate, and the two have nothing to do with + * each other. Counting ticks pulled 280 times a second on a 280 Hz display, idle or not, and + * under-pulled read-back frames on 60 Hz, where a tick lost to a slow round trip pushed the + * next pull a whole tick further. + * + * While shared-texture frames keep coming, a pull every 8 ms catches each one within half a + * 60 fps frame. Anything else — read-back frames, each a GPU readback, a structured clone + * across IPC and a canvas upload, or a view with nothing new — is pulled ~30 times a second. */ +const PULL_INTERVAL_FLOWING_MS = 8; +const PULL_INTERVAL_MS = 33; +/** Frames still count as coming this long after the last one. A pull that lands between two + * frames finds nothing new, and taking that for the end of playback dropped every next + * pull to the slow interval: ~40 frames a second drawn out of ~60. */ +const PULL_FLOWING_WINDOW_MS = 250; +/** rAF timestamps jitter around the display's period: without it, a 33 ms interval on a 60 Hz + * display would land on the third tick as often as the second. */ +const PULL_INTERVAL_SLACK_MS = 1.5; export function useNativeCompositorView( canvasRef: RefObject, @@ -116,7 +136,7 @@ export function useNativeCompositorView( let rectRafHandle = 0; let pullRafHandle = 0; - let pullTick = 0; + let lastPullAt = Number.NEGATIVE_INFINITY; let lastRect: CompositorViewRect | null = null; let disposed = false; // Fresh view (source or enablement changed) → the previous view's fatal error @@ -189,18 +209,71 @@ export function useNativeCompositorView( // the next frame's. The generation actually on the canvas lets an older one be dropped, // instead of bringing back its pixels and its buffer size until native sends another. let paintedGen = 0; + // Frames arrive as shared GPU textures (see `subscribeCompositorSharedFrames`), and one + // came lately: while both hold, the loop pulls at its fast interval. + let sharedTransport = false; + let lastFrameAt = Number.NEGATIVE_INFINITY; + // Whether this view's first shared frame was checked to have actually landed. + let sharedChecked = false; + + /** Puts a frame on the canvas. The buffer takes the frame's size HERE, in the same task + * as the draw that refills it, never when the canvas box changes: anything awaited + * between the two (the rect's trip to native, `createImageBitmap`) is a frame the + * browser presents empty. Meanwhile CSS stretches the previous frame over the new box. + * An Auto format reshapes that box on every padding tick, so an early resize blinked + * the footage out and back while the slider moved. */ + const paint = (gen: number, width: number, height: number, draw: () => void): boolean => { + if (disposed || gen < paintedGen) { + return false; + } + setBufferSize(width, height); + draw(); + paintedGen = gen; + markPainted(); + return true; + }; - /** rAF pull loop: throttle to ~30fps and repaint ONLY when native reports a - * newer generation. The returned packet is self-describing (`gen` + dims + - * pixels), so the canvas is sized from the packet — pixels and canvas can - * never drift out of sync. Runs off the main thread so UI stays at 60/120fps. */ - const pullLoop = () => { + // Frames the main process hands over as shared GPU textures land here, ahead of the + // `readCompositorFrame` reply that names their generation. + const unsubscribeShared = subscribeCompositorSharedFrames((frame, meta) => { + try { + const id = viewIdRef.current; + const ctx = canvas.getContext("2d"); + if (disposed || id == null || meta.viewId !== id || !ctx) { + return; + } + sharedTransport = true; + lastGen = Math.max(lastGen, meta.gen); + publishFootageQuad(meta.footage, meta.footageProjective); + const fresh = canvas.dataset.painted === undefined; + const drawn = paint(meta.gen, meta.width, meta.height, () => ctx.drawImage(frame, 0, 0)); + if (drawn && fresh && !sharedChecked) { + sharedChecked = true; + // Chromium opens the texture in its own GPU process, and one it cannot open (a + // hybrid laptop's other adapter) draws as nothing at all. The composed frame is + // opaque everywhere, so a transparent pixel on this fresh canvas is that failure. + if (ctx.getImageData(meta.width >> 1, meta.height >> 1, 1, 1).data[3] === 0) { + sharedTransport = false; + safelyCall("stopSharedFrames", () => stopSharedCompositorFrames(id)); + } + } + noteUiProbePreviewFrame(); + } finally { + frame.close(); + } + }); + + /** rAF pull loop: repaint ONLY when native reports a newer generation. A pixel packet + * is self-describing (`gen` + dims + pixels), so the canvas is sized from the packet — + * pixels and canvas can never drift out of sync. */ + const pullLoop = (now: number) => { pullRafHandle = requestAnimationFrame(pullLoop); if (disposed || inFlight) { return; } - pullTick = (pullTick + 1) % PULL_LOOP_TICK_DIVISOR; - if (pullTick !== 0) { + const flowing = sharedTransport && now - lastFrameAt < PULL_FLOWING_WINDOW_MS; + const interval = flowing ? PULL_INTERVAL_FLOWING_MS : PULL_INTERVAL_MS; + if (now - lastPullAt < interval - PULL_INTERVAL_SLACK_MS) { return; } const id = viewIdRef.current; @@ -212,14 +285,27 @@ export function useNativeCompositorView( return; } inFlight = true; + lastPullAt = now; readCompositorFrame(id, lastGen) .then((frame) => { inFlight = false; + // A receipt keeps the fast cadence going. Read-back pixels mean the view went back to + // read-back (a failed import or send), whose frames are pulled ~30 times a second. + if (frame && "shared" in frame) { + lastFrameAt = now; + } else if (frame) { + sharedTransport = false; + } // `null` = nothing newer than `lastGen` (idle path — no pixels // crossed IPC) OR no frame yet. Either way, leave the canvas as-is. if (disposed || !frame) { return; } + if ("shared" in frame) { + // Already drawn by the shared-frame listener: only the generation is news. + lastGen = Math.max(lastGen, frame.gen); + return; + } const { gen, width, height, data } = frame; // Defensive: the packet's byte count must match its own declared // dimensions. A mismatch would corrupt the image silently — bail. @@ -239,30 +325,15 @@ export function useNativeCompositorView( data.byteLength, ); const image = new ImageData(pixels, width, height); - // The buffer takes the packet's size HERE, in the same task as the draw that - // refills it, never when the canvas box changes: anything awaited between the - // two (the rect's trip to native, `createImageBitmap`) is a frame the browser - // presents empty. Meanwhile CSS stretches the previous frame over the new box. - // An Auto format reshapes that box on every padding tick, so an early resize - // blinked the footage out and back while the slider moved. - const paint = (draw: () => void) => { - if (disposed || gen < paintedGen) { - return; - } - setBufferSize(width, height); - draw(); - paintedGen = gen; - markPainted(); - }; // `createImageBitmap` decodes off the main thread (keeps UI at 60/120fps) // and snapshots `image`, so the view can be released after; `putImageData` // is the synchronous fallback if bitmap creation is unavailable. createImageBitmap(image) .then((bitmap) => { - paint(() => ctx.drawImage(bitmap, 0, 0)); + paint(gen, width, height, () => ctx.drawImage(bitmap, 0, 0)); bitmap.close(); }) - .catch(() => paint(() => ctx.putImageData(image, 0, 0))); + .catch(() => paint(gen, width, height, () => ctx.putImageData(image, 0, 0))); // Advance only after a successful, validated frame — so a dropped/ // malformed packet is retried rather than silently skipped. lastGen = gen; @@ -326,6 +397,7 @@ export function useNativeCompositorView( return () => { disposed = true; + unsubscribeShared(); publishFootageQuad(null); if (rectRafHandle !== 0) { cancelAnimationFrame(rectRafHandle); diff --git a/technical-documentation/architecture/preview.md b/technical-documentation/architecture/preview.md index 0a83ac5fd..f28b0bde5 100644 --- a/technical-documentation/architecture/preview.md +++ b/technical-documentation/architecture/preview.md @@ -171,16 +171,47 @@ The contract, with the invariant on the consumer side: a mismatch would corrupt the image silently. The consumer never assumes a size — every draw is preceded by a resize of the canvas's drawing buffer to the packet's declared `width`/`height`. -5. **Pull loop cadence.** The renderer pulls on every other rAF tick (`PULL_LOOP_TICK_DIVISOR = 2`, - [`useNativeCompositorView.ts:70`](../../src/native/hooks/useNativeCompositorView.ts:70)), - so IPC + GPU readback + `putImageData` run at roughly 30 fps on 60/120 Hz - displays without changing perceived smoothness. +5. **Pull loop cadence.** Read-back frames are pulled on every other rAF tick + (`PULL_LOOP_TICK_DIVISOR = 2`), shared-texture frames (below) on every tick. Each + read-back frame is a GPU readback, a structured clone across IPC and a canvas upload, + and the tick counter does not advance while a read is in flight: an 8 MB frame whose + round trip passes 16.7 ms is pulled one tick in three, about 20 fps. The renderer-side wrapper mirrors this verbatim in the Electron main process -([`compositorViewService.ts:339`](../../electron/native-bridge/services/compositorViewService.ts:339)), +([`compositorViewService.ts`](../../electron/native-bridge/services/compositorViewService.ts)), so when the addon is absent the IPC layer returns `null` too — the renderer never has to special-case "addon missing". +### Shared textures (Windows) + +On Windows' hardware backend the pixels never leave the GPU. Copying them through RAM +was the preview's bottleneck, not the compositor: at the same ~57 composed frames per +second, read-back reached the canvas at 20-26 fps and kept 54-57 % of the renderer's main +thread busy (measurements in +[engineering/rendering-performance.md](../engineering/rendering-performance.md#preview-transport--2026-10-02)). + +- **Native side.** The render thread copies each composed frame into one of four shared + D3D11 textures (`shared_frames.rs`, NT handles, no keyed mutex), waits for the GPU to + finish the copy, and publishes `{ gen, slot, handle }`. `SlotBook` tracks which slot + holds the ready frame and which ones Chromium still holds; a frame nobody took gives + its slot back. A held slot is never rewritten, since Chromium may still read it: when + every slot is held and one has waited 1 s for its release, the view falls back to + read-back. +- **Main process.** `readFrame` takes the frame (`readSharedFrame`), imports it with + `sharedTexture.importSharedTexture`, sends it with `sharedTexture.sendSharedTexture` to + the frame that asked, drops its own reference and answers with a receipt + (`{ …meta, shared: true }`, no pixels). `allReferencesReleased` returns the slot + (`releaseSharedFrame(id, slot, gen)`) once Chromium is done with it in every process. +- **Renderer.** The preload's `setSharedTextureReceiver` hands the frame to + `electronAPI.onCompositorFrame`, ahead of the receipt; the hook draws the `VideoFrame` + with `drawImage` and closes it. The pixels are byte-identical to read-back. +- **Fallbacks, all to read-back.** Not Windows, the software backend, Chromium without GPU + compositing, or `OPENSCREEN_PREVIEW_READBACK=1` never enable it. A failed import or send + turns it off for the view, and so does a first frame that lands transparent (a texture + Chromium could not open, e.g. on another adapter): the composed frame is cleared opaque, + so a transparent pixel can only be that. The render thread republishes the current + frame on every switch, so the canvas never waits for something to move. + ## Playback sync The native view runs its own clock while playing, so the renderer only pushes a @@ -307,6 +338,9 @@ was thrown away. measured at the bench in [engineering/rendering-performance.md](../engineering/rendering-performance.md) stay below the threshold in practice, but no systematic measurement exists. +- **Shared textures are Windows-only.** macOS (an `IOSurface`-backed Metal texture) and + Linux (a dmabuf exported from Vulkan) still read back: `sharedTexture` imports both, the + native halves are not written. - **Add-on absent = blank frame.** When `compositor_view.node` is missing (development with the addon not yet built, or a packaged build for an unsupported architecture) the overlay renders no pixels: only the DOM/CSS diff --git a/technical-documentation/engineering/rendering-performance.md b/technical-documentation/engineering/rendering-performance.md index 6fa913214..33de33cb2 100644 --- a/technical-documentation/engineering/rendering-performance.md +++ b/technical-documentation/engineering/rendering-performance.md @@ -212,6 +212,52 @@ cliffs above: fine for a real GPU, heavy for a per-pixel loop on the CPU rasteri > of the aurora export moves by up to 22/255 between frames 0 and 300, against 4/255 for the > still one. +## Preview transport — 2026-10-02 + +**The preview was bound by the trip of its pixels to the canvas, not by the compositor.** +The compositor composed ~57 frames per second of a 1080p60 recording in both arms below; what +reached the canvas, and what it cost the UI, depended only on how the frames travelled. + +Two throwaway Electron benches (not kept in the tree), run in a hidden window on a desktop — +Ryzen 7 5800X, GeForce RTX 4070 Ti, Windows 11 — not the reference laptop: the copies are CPU +and memory work, so an iGPU laptop pays more for read-back, not less. A hidden window +throttles nothing here (`backgroundThrottling: false`, ticks on timers rather than rAF), but +it renders no React and decodes nothing alongside, so the absolutes are a floor, not the app. +Main-thread load is measured as the gaps in a back-to-back `MessageChannel` ping loop +(`setImmediate` in the main process). + +**IPC alone** (synthetic buffers through `ipcRenderer.invoke` from a sandboxed preload and +`contextBridge`, as the app does), two runs: + +| frame | MB | round trip p50 / p90 | renderer main thread busy at 30/s | main process busy at 30/s | +|---|---:|---:|---:|---:| +| 1920×1080 | 8.3 | 24 / 31 ms | 54-55 % | 31-37 % | +| 1650×930 | 6.1 | 18-20 / 22-25 ms | 37-39 % | 19-21 % | +| 1100×620 | 2.7 | 8-10 / 10-13 ms | 17-21 % | 9-12 % | + +**End to end** (the built addon, a real 32 s 1080p60 recording with its 1080p30 webcam, the +hook's pull loop on 60 Hz ticks), two runs per arm: + +| preview | arm | frames on the canvas /s | composed /s | renderer busy | main busy | main import + send p50 / p90 | +|---|---|---:|---:|---:|---:|---:| +| 1920×1080 | read-back | 20-21 | 57-59 | 57 % | 37-38 % | — | +| 1920×1080 | shared texture | 53-54 | 56-57 | 1.8-2.0 % | 0.1-0.5 % | 0.67 / 0.97 ms | +| 1650×928 | read-back | 26 | 57 | 54-56 % | 35 % | — | +| 1650×928 | shared texture | 52.5-53 | 57-58 | 2.0-2.6 % | 0.6-1.7 % | 0.67-0.73 / 1.02-1.18 ms | + +The same paused frame, read back and drawn from the shared texture, differs in **0 of +8 294 400 bytes** at 1080p (0 of 6 124 800 at 1650×928). + +- **Read-back stalls on its own round trip.** At 18-24 ms per 8 MB frame, the hook's pull loop + (one read in flight, one tick in two) gets one frame every three ticks: ~20 fps on 60 Hz. +- **The transport was most of the renderer's load.** The ~44 % main-thread block measured + during playback on 2026-09-29 (`VirtualPreview.tsx`, another machine) is the order of this + cost alone. Every copy, structured clone and allocation landed on the thread React paints + from. +- **Shared textures leave ~7 % of the composed frames undrawn.** Two ~60 Hz clocks — the + render thread and the pull loop — beat against each other; a tick that finds nothing new + is followed by one that finds two, and only the newer is drawn. + ## The macOS export path — 2026-09-03/04 Everything above is the Windows reference machine. This section is a **different machine and a different pipeline**: a Mac mini M1 (8 cores, 8 GiB, macOS 26.5), Metal compositor, VideoToolbox on both ends. Nothing here transfers to the Windows numbers, and the reverse held too — of the three levers that mattered on Windows and Linux, **none applied here**.