diff --git a/CLAUDE.md b/CLAUDE.md index 03483cdd9..47b62d16a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -79,6 +79,7 @@ Vendor display drivers ship as **plug-in DLLs** from their own repos (ADR-019). - `XR_DXR_android_surface_binding` — app passes its own Android Surface/`ANativeWindow`; `xrSetAndroidSurfaceDXR` republishes it across background/resume and `xrSetAndroidWindowGeometryDXR` feeds the per-frame window rect (ADR-036 D6). **Required for multi-window on Android** — the runtime-spawned SurfaceView is `_hosted`-fullscreen-only - `XR_DXR_mcp_tools` — app registers its own MCP tools (agent control surface); event-queue dispatch via `XrEventDataMCPToolCallDXR` - `XR_DXR_depth_budget` — advisory **rear depth budget**: how far behind the display plane a transparent app may render (`farOffsetVH`, 0 = clip at the ZDP, 1000 = unrestricted), chained on `XrViewState` at `xrLocateViews`. The runtime owns the policy (it measures the background's horizontal-disparity cue), the DP owns pixels, the app owns geometry — ADR-040 +- `XR_DXR_lift` — 2D→3D **conversion service**: a vendor plug-in's module (depth / SBS / N-view / photo→splats) exposed generically as async latest-wins streams (IPC-only, D3D11 service lift thread); SBS/N-view results are woven on the ordinary weave path (a weave rect can be flagged "lift me"), and a READY module supersedes the browser/SDK open default — ADR-042 The list above is highlights, not the catalog — **`docs/specs/extensions/index.json` is the catalog**: one hand-written note (group, title, one-line summary) per published extension, joined to the `XR_DXR_*.h` headers by `scripts/gen_extensions_index.py`. It generates the `displayxr-extensions` mirror's `README.md` + machine-readable `extensions.json`, and `displayxr-website` merges its longer editorial prose onto that by name — so adding a header is all it takes for an extension to appear on every public surface. `lint.yml` runs `--check` on every PR, so a header with no note (or a note with no header) fails the build. That guard exists because the mirror's README was a frozen heredoc that documented 5 of 16 extensions for months (displayxr-extensions#2). diff --git a/docs/README.md b/docs/README.md index 78a44d4a2..92c3ecfdf 100644 --- a/docs/README.md +++ b/docs/README.md @@ -147,6 +147,7 @@ Integrate your 3D display hardware into DisplayXR. - [ADR-039](adr/ADR-039-one-fill-engine-for-every-tier.md) — One fill engine for every tier (same-adapter split) - [ADR-040](adr/ADR-040-rear-depth-budget.md) — Rear depth budget — the runtime owns the policy, the plug-in owns pixels, the app owns geometry - [ADR-041](adr/ADR-041-fixed-view-count-with-per-frame-activity.md) — Fixed view count with per-frame activity — inactive views alias, they do not disappear +- [ADR-042](adr/ADR-042-vendor-2d3d-conversion-supersedes-default.md) — A vendor 2D→3D conversion module supersedes the open default — the runtime exposes it, weaving stays the DP's - [ADR-043](adr/ADR-043-stereo-camera-source.md) — A display's stereo camera is a plug-in-provided source, owned by the service and privacy-gated by the runtime - [ADR-044](adr/ADR-044-colour-contract-per-backend.md) — The colour contract, per backend and swapchain format diff --git a/docs/adr/ADR-042-vendor-2d3d-conversion-supersedes-default.md b/docs/adr/ADR-042-vendor-2d3d-conversion-supersedes-default.md new file mode 100644 index 000000000..ae00fa1fc --- /dev/null +++ b/docs/adr/ADR-042-vendor-2d3d-conversion-supersedes-default.md @@ -0,0 +1,137 @@ +# ADR-042: A vendor 2D→3D conversion module supersedes the open default — the runtime exposes it, weaving stays the DP's + +**Status:** Accepted (2026-09-25) · introduces +[`XR_DXR_lift`](../specs/extensions/XR_DXR_lift.md) · appends five optional D3D11 +display-processor slots and one optional plug-in factory under the +[ADR-020](ADR-020-plugin-abi-compatibility-policy.md) append-at-end rule · related: +[ADR-007](ADR-007-compositor-never-weaves.md), +[ADR-019](ADR-019-vendor-plugin-aux-boundary.md), +[ADR-040](ADR-040-rear-depth-budget.md) + +## Context + +Turning ordinary 2D content — a video in a page, a photo, a call participant — into something +the panel can show in 3D is done today by **open defaults that live in the consumer**: the +DisplayXR Browser and the web SDK run an open monocular depth estimator plus a view generator +(and, for photos, an open depth + splat generator) in the page, then hand the result to the +weave like any other stereo content. + +Display vendors have their own conversion modules — trained and tuned for their optics, often +running on dedicated inference paths the page cannot reach (Leia's NeurD over DirectML/CUDA is +the first; a Leia-owned photo → Gaussian-splat model is next). The vendor plug-in is already the +one component that knows the panel, and it already ships per machine. What was missing is a +generic way for the runtime to expose such a module, and a rule for who wins when both exist. + +Three questions have to be settled together: + +1. **Who wins?** An app — or the SDK inside it — that has its own converter, on a machine whose + plug-in also has one. +2. **What crosses the boundary?** Woven pixels would make the conversion a second weaver and + break ADR-007; depth alone would push view synthesis back into every consumer. +3. **How does it meet the frame loop?** Models take tens of milliseconds (seconds for photo → + splats). The weave is a ~1 ms synchronous service on the present thread of a browser. + +## Decision + +### 1. A READY vendor module supersedes the open default + +When `xrGetLiftPropertiesDXR` reports `READY` with the needed mode bit, a consumer that also +ships an open converter uses the runtime's. It falls back to its open default only when the +runtime reports `UNAVAILABLE` (no module, a failed one, a non-Windows service, an in-process +session) — and treats `ACTIVATING` as "not yet", polling, never as a permanent fallback. + +This is the same shape as weaving: the vendor's calibrated implementation, behind the plug-in, +beats a generic one in the app. The runtime exposes the capability **generically** — modes, +limits, state, an informational backend name — never which model runs. + +**Vendor Gaussian modules supersede the SDK's open MoGe + generator lift the same way NeurD +supersedes the open depth (VDA-class) default.** A plug-in advertising `GAUSSIANS` is preferred +for photo → splats; the SDK passes the photo's focal length (`focalPx`, e.g. the MoGe-estimated +`fx`) so the vendor model gets the intrinsics it takes as input. + +### 2. The module returns pre-weave views; weaving stays the DP's + +SBS and N-view results are ordinary pre-weave views (two / N views side by side, not woven), +DEPTH is a depth map, GAUSSIANS a blob. The runtime weaves SBS/N-view results on the existing +weave path, through the same display processor as everything else (ADR-007). So: + +- one weaver per panel, never two; the conversion module is never on the present path; +- a consumer may also take the views or depth itself (effects, look-around re-render) — the + result is useful beyond weaving; +- a vendor module needs no knowledge of windows, phase, or the interlace. + +### 3. Asynchronous, one frame behind; geometry from the weave rect + +Conversion runs on a runtime-owned **lift thread** with its **own device**, off the weave, render +and IPC threads. A submit is a snapshot into a **latest-wins mailbox** (a slow module lags a +frame; it never builds a backlog); an acquire returns the newest finished result. Nothing on any +latency-sensitive thread waits for the model. + +When lifted content is woven, the caller flags the weave rect (`XrWeaveSubmitLiftRectsDXR`); the +service snapshots the rect's 2D content into the stream **and weaves the stream's latest result +at the rect's current position** in the same submit. Drag, resize and scroll therefore stay exact +and real-time — they come from this frame's rect — while only the depth is one conversion +behind. Until the first result the rect is woven flat. + +### 4. Scheduling is runtime policy + +Multiple streams share one module. The runtime schedules them by a per-stream priority (HIGH +every round, NORMAL round-robin, LOW every 4th round, PAUSED never) and reports each stream's +effective rate. The plug-in converts one frame per call and knows nothing about streams' +relative importance. + +### 5. The plug-in contract is minimal and synchronous + +Five appended D3D11 DP slots (`lift_get_caps`, `lift_stream_create`, `lift_stream_destroy`, +`lift_convert`, `lift_convert_blob`) and one appended plug-in factory +(`create_dp_d3d11_lift` — a DP that serves only lift: no weaver, no tracker session), all +optional (ADR-020, no ABI bump). The runtime always passes explicit viewpoints (the panel's +predicted tracked eyes, or the app's), so the lift DP needs no tracker. A module's warm-up +(licence, model load) is reported as ACTIVATING and polled. + +## Consumer contract (browser + web SDK) + +| Situation | Browser / SDK does | +|---|---| +| `READY` + mode bit | use the runtime: `XrWeaveSubmitLiftRectsDXR` for inline 2D video/images lifted in place; `xrSubmitLiftFrameDXR` + `xrAcquireLiftResultDXR` where the page wants views/depth itself; `xrAcquireLiftBlobDXR` for photo → splats | +| `ACTIVATING` | keep the current rendering (flat, or the open default if already running); poll ≤ 2 Hz; switch to the runtime on READY | +| `UNAVAILABLE` / `XR_ERROR_FEATURE_UNSUPPORTED` | open default | +| a lift call returns `XR_ERROR_RUNTIME_FAILURE` | transient: retry next frame, do not fall back | +| `XR_ERROR_INSTANCE_LOST` | the weave §4b recovery (new instance), then re-query properties | + +- Draw lifted content into the weave input as 2D (the whole rect on the batch layout; every tile + on the N-view layout) — never pre-convert it when the runtime will. +- Key results on `sourceTime` when the pairing matters; otherwise take the freshest. +- Set stream priority from what the user is looking at (active speaker, focused video); + PAUSED for off-screen content keeps its last result without spending the module. +- Gate the chained structs on the extension being enabled, not on a spec version (v1). + +## Consequences + +- **+** The best available converter on each machine is used without the app knowing which it is. +- **+** No second weaver; ADR-007 holds. The conversion never touches the present path. +- **+** Geometry of lifted content is exact at weave time; depth latency is bounded by the + module, visible per stream, and shaped by priority. +- **+** Hardware-free end to end: sim_display's env-gated fake module exercises every path in CI; + `displayxr-cli lift probe` measures a real module on a panel box. +- **−** One more thread and one more D3D11 device in the service (lazily created — zero cost on a + machine that never asks), and a second DP instance of the vendor plug-in (lift-only). +- **−** Frames cross devices as keyed-mutex shared textures: one extra copy each way. On a hybrid + box the lift device is on the service's render adapter; a module that runs on another adapter + (its own device) pays its own bridge. +- **−** Windows/D3D11 only in v1. Other platforms advertise the extension and report + `supportedModes = 0`, which the consumer contract already handles. + +## Alternatives rejected + +- **The module weaves its own output.** Two weavers per panel, calibration duplicated, and a + conversion on the present path — ADR-007 exists to prevent exactly this. +- **Return depth only; consumers synthesize views.** Pushes the vendor-specific part (view + synthesis tuned to the optics) back into every consumer. +- **Synchronous conversion inside `xrWeaveSubmitDXR`.** Turns a ~1 ms service into a tens-of-ms + one on the browser's present thread. +- **Reuse the weaving DP instance for lift.** It is driven on the service's single shared + immediate context under the render lock and is recreated on presenter changes; a conversion + would stall every weave for its whole duration. +- **Priority as a per-frame submit field.** Priority changes when attention changes, not per + frame; a per-stream setter says so and keeps the submit minimal. diff --git a/docs/adr/README.md b/docs/adr/README.md index fd2db195c..78e8744db 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -45,5 +45,6 @@ - [ADR-039](ADR-039-one-fill-engine-for-every-tier.md) — One fill engine for every tier (same-adapter split) - [ADR-040](ADR-040-rear-depth-budget.md) — Rear depth budget — the runtime owns the policy, the plug-in owns pixels, the app owns geometry - [ADR-041](ADR-041-fixed-view-count-with-per-frame-activity.md) — Fixed view count with per-frame activity — inactive views alias, they do not disappear +- [ADR-042](ADR-042-vendor-2d3d-conversion-supersedes-default.md) — A vendor 2D→3D conversion module supersedes the open default — the runtime exposes it, weaving stays the DP's - [ADR-043](ADR-043-stereo-camera-source.md) — A display's stereo camera is a plug-in-provided source, owned by the service and privacy-gated by the runtime - [ADR-044](ADR-044-colour-contract-per-backend.md) — The colour contract, per backend and swapchain format diff --git a/docs/architecture/service-architecture.md b/docs/architecture/service-architecture.md index ebfcd6d6e..34ff0eeac 100644 --- a/docs/architecture/service-architecture.md +++ b/docs/architecture/service-architecture.md @@ -302,6 +302,7 @@ closed with **no message to the client**. An evicted-but-alive client (#925 S4) | window-op worker | `comp_d3d11_service.cpp:1398` | `SetWindowPlacement`-family restores off the render path (#925 rank 7) | try/catch, no restart | | WinRT capture pool | `d3d11_capture.cpp:285-300` | `on_frame_arrived` → `CopyResource` on the **shared immediate context, outside `render_mutex`** (relies on `SetMultithreadProtected(TRUE)`, `comp_d3d11_service.cpp:14697-14705`) | partial | | provider threads | provider-owned (`ultraleap_provider.cpp:568` poll thread; net_input hub) | LeapC polling, #941 idle watchdog (a branch of the poll thread) | none | +| lift (ADR-042) | `d3d11_lift.cpp` `d3d11_lift_create`, lazily on the first `XR_DXR_lift` call | the ONLY thread that touches the vendor's lift display processor: brings up a dedicated **lift device** on the service adapter (ID3D11Multithread-protected) + the plug-in's lift-only DP, then loops: priority-scheduled round (`u_lift_sched`) → take a stream's newest pending input → `lift_convert` / `lift_convert_blob` (ms to s) → copy into the output ring. Never takes `render_mutex` or `immediate_ctx_mutex`; calls back for tracked eyes with no lift lock held. Joined first in `system_destroy` | `DXR_LIFT=0` kill switch | ### 3.2 Lock order (as of #964–#966) @@ -348,6 +349,15 @@ c->mutex → render_mutex → { ws_snapshot_mutex, active_compositor_mutex, resume path to maintain: `multi_compositor_register_client` restarts the thread and the presenter re-bind builds a DP on the first frame. The grace window is what keeps an app restart or a shell relaunch from recreating the vendor weaver. +- **Lift (ADR-042) adds one leaf and no edge into the panel lock.** `d3d11_lift::mtx` + guards the stream table and every lift mailbox and is never held across a GPU + wait, a keyed-mutex acquire or a vendor call. Where both are taken the order is + `immediate_ctx_mutex → lift mtx` (a lift-flagged weave rect's snapshot, inside + `weave_submit`); the lift thread never takes `immediate_ctx_mutex` (it owns a + device of its own), and the IPC-side lift calls take `immediate_ctx_mutex` alone, + for one blit or one copy. Frames cross between the service device and the lift + device only as keyed-mutex shared textures (two input slots, two output-ring + slots per stream), so neither thread ever waits on the other's GPU queue. - Never join the render thread under `render_mutex`; never hold `global_state.lock` across a compositor call. @@ -364,6 +374,7 @@ c->mutex → render_mutex → { ws_snapshot_mutex, active_compositor_mutex, | `hub->mutex` | Ultraleap provider | joint sets | poll thread + every consumer's `get_hand_tracking` | | `ipc_c->mutex` | per client process | the whole pipe round trip | every RPC | | `usys->sessions.mutex` | per system | session list; event push (unbounded malloc'd per-session list, `u_session.c:34-42`) | broadcasts | +| `d3d11_lift::mtx` | per service (`d3d11_lift.cpp`), leaf | lift stream table, every stream's `u_lift_mailbox` (input slots, output ring, pins, stats), caps | lift thread between conversions; IPC threads for submit / acquire / stats (ns-µs, never across GPU or vendor work); `weave_submit` for lift-flagged rects (under `immediate_ctx_mutex`) | **Nesting observed:** `global_state.lock → render_mutex` at ≥ 11 handler sites (`ipc_server_handler.c:3735, 4079, 4110, 4204, 4370, 4429, 4475, 4696, 4792, 4968, 5106` diff --git a/docs/reference/xrt_plugin_iface.md b/docs/reference/xrt_plugin_iface.md index e509c638c..8a87f3082 100644 --- a/docs/reference/xrt_plugin_iface.md +++ b/docs/reference/xrt_plugin_iface.md @@ -392,6 +392,81 @@ bool (*get_background_preview)(struct xrt_display_processor_d3d11 *xdp, four floats zeroed is read as the documented normal case, `0,0,1,1`. - **Purely additive**, so no `XRT_PLUGIN_API_VERSION_CURRENT` bump (ADR-020). +## Turning 2D into 3D: the lift slots (ADR-042, `XR_DXR_lift`) + +A vendor may ship a **2D→3D conversion module** — monocular depth, stereo or N-view synthesis, +photo → Gaussian splats. The runtime exposes it to apps generically as +[`XR_DXR_lift`](../specs/extensions/XR_DXR_lift.md), and, per +[ADR-042](../adr/ADR-042-vendor-2d3d-conversion-supersedes-default.md), a READY vendor module +supersedes the open default a browser or the web SDK would otherwise run. The plug-in side is +five optional slots appended to `xrt_display_processor_d3d11` and one optional factory appended +to `xrt_plugin_iface`, all announced by `XRT_DP_D3D11_HAS_LIFT` / +`XRT_PLUGIN_IFACE_HAS_D3D11_LIFT_FACTORY`: + +```c +/* xrt_dp_lift.h */ +struct xrt_dp_lift_caps { uint32_t struct_size; uint32_t modes; /* 1 DEPTH, 2 SBS, 4 NVIEW, 8 GAUSSIANS */ + uint32_t max_streams; uint32_t max_views; + uint32_t depth_semantics; /* 0 relative, 1 metric */ + uint32_t state; /* 0 unavailable, 1 activating, 2 ready */ + uint64_t typical_latency_ns; char backend[32]; }; +struct xrt_dp_lift_stream_info { uint32_t struct_size; uint32_t mode; uint32_t content_hint; /* 0 video, 1 photo */ + float input_scale; }; +struct xrt_dp_lift_params { uint32_t struct_size; float convergence; /* [0,1] relative depth at the glass; <0 AUTO */ + float strength; uint32_t inpaint; uint32_t view_count; + float focal_px; /* appended: input focal length in px; <=0 unknown (GAUSSIANS) */ }; + +/* xrt_display_processor_d3d11, slots 25..29 */ +bool (*lift_get_caps)(xdp, struct xrt_dp_lift_caps *out); +bool (*lift_stream_create)(xdp, const struct xrt_dp_lift_stream_info *info, uint64_t *out_id); +void (*lift_stream_destroy)(xdp, uint64_t id); +bool (*lift_convert)(xdp, uint64_t id, void *d3d11_context, void *input_resource /* RGBA8, w x h */, + uint32_t w, uint32_t h, const struct xrt_dp_lift_params *p, + const float *viewpoints_xyz, uint32_t viewpoint_floats, + void **out_resource /* valid until the next call on this stream */, + uint32_t *out_w, uint32_t *out_h, uint32_t *out_format /* DXGI_FORMAT */); +bool (*lift_convert_blob)(xdp, uint64_t id, void *d3d11_context, void *input_resource, uint32_t w, uint32_t h, + const struct xrt_dp_lift_params *p, uint32_t *out_format /* 1 PLY_3DGS, 2 SOG */, + const void **out_bytes /* valid until the next call */, size_t *out_size); + +/* xrt_plugin_iface, appended */ +xrt_dp_factory_d3d11_fn_t create_dp_d3d11_lift; /* a DP that serves ONLY the lift slots */ +``` + +- **Who calls, from where.** Exactly one runtime thread — the D3D11 service's *lift thread* — + on exactly one DP instance per service process, created through **`create_dp_d3d11_lift`** + on a **dedicated device** on the service's adapter (its immediate context has + `ID3D11Multithread` protection on, so a module may flush / signal it from its own worker). + Window handle NULL, never `process_atlas`, never a mode request. **That DP must build no + weaver and open no tracker session.** Without the lift-only factory the runtime falls back to + `create_dp_d3d11` with a NULL window, and destroys the instance at once if it has no lift + slots — which is why a plug-in with a module should provide the explicit factory. +- **Why not the weaving DP.** A conversion blocks for tens of milliseconds (seconds for + GAUSSIANS); the weaving DP is driven on the service's single shared immediate context under + the render lock and is recreated on presenter/focus changes. Sharing it would stall every + weave for the length of a conversion. +- **Synchronous, stateless in time.** One call converts one frame. The plug-in never queues, + drops, timestamps or schedules: the runtime owns the latest-wins mailbox, the output ring + (it copies `out_resource` / `out_bytes` before the next call), fences, per-stream priority + (`XrLiftPriorityDXR`) and timestamps. A `false` return means "no output this frame"; the + runtime keeps the previous result. +- **Viewpoints are always explicit.** For SBS/N-view the runtime passes `viewpoints_xyz` — + the app's EXPLICIT viewpoints, else the panel DP's predicted tracked eye pair (display space, + metres). Give them precedence over any tracker of your own; the lift DP has none. NULL/0 only + while no eyes are known. +- **Output layout.** SBS = two views side by side; NVIEW = `view_count` views in one row, view 0 + leftmost; DEPTH = one channel (any single-channel or RGBA format — say which in + `out_format`). The runtime weaves SBS/NVIEW results itself on the ordinary weave path + (ADR-007) — the plug-in never weaves a lift result. +- **Caps are polled.** `lift_get_caps` at DP creation, then ≤ 1 Hz while not READY, so an + ACTIVATING module (licence check, model load) shows up as READY without an app restart. + Modes 0 / a NULL slot / an older `struct_size` all read as "no module". +- **Knobs live in the SERVICE's environment.** A vendor module that reads env vars (backend + choice, gains) reads them in `displayxr-service.exe`'s process, not the app's. +- **sim_display** fills the slots only under `SIM_DISPLAY_FAKE_LIFT=1` (shifted SBS/N-view, + gradient depth, a two-layer 3DGS PLY) so the whole path runs hardware-free. +- **Purely additive**, so no `XRT_PLUGIN_API_VERSION_CURRENT` bump (ADR-020). + ## Where the window may LAND: `snap_window_rect` A windowed weave takes its interlace phase from the window's absolute position on the panel. diff --git a/docs/roadmap/control-panel-performance-settings.md b/docs/roadmap/control-panel-performance-settings.md index ce9c2f009..03288919e 100644 --- a/docs/roadmap/control-panel-performance-settings.md +++ b/docs/roadmap/control-panel-performance-settings.md @@ -84,7 +84,7 @@ means deleting the static and reading a service-held snapshot — mechanically c `DXR_COMMIT_PACE` · `DXR_FENCE_WAIT_MS` · `DXR_EVICT_IDLE_MS` · `DXR_EVICT_ENDED_MS` · `DXR_DP_GRAVEYARD_MS` · `DXR_DEVICE_REMOVED_EXIT_MS` · `DXR_HEALTH_MS` · -`DXR_IDLE_QUIESCE_MS`. +`DXR_IDLE_QUIESCE_MS` · `DXR_LIFT_MAX_INPUT_EDGE` · `DXR_LIFT_LETTERBOX`. Semantic caveat: `DXR_COMMIT_PACE` changes backpressure discipline; flipping it under a live client is a behaviour change mid-flight, not just a number change. @@ -324,7 +324,7 @@ never claim a mode the runtime is not in, and "Custom" falls out for free. ## Appendix A — census Every `DXR_*` name read at runtime under `src/xrt`, with its read site, mechanism, default -and tier. **98 distinct names**; the two `DXR_BG2D_*` knobs reach the environment through +and tier. **100 distinct names**; the two `DXR_BG2D_*` knobs reach the environment through `bg2d_int_knob()` rather than a literal `getenv` at the listed line. Process column: **App** = the runtime DLL, loaded into the OpenXR app's process · @@ -386,6 +386,8 @@ library linked into both · **CLI** = `displayxr-cli.exe` (reporting only, contr | `DXR_PRESENT_OPAQUE` | `d3d11/comp_d3d11_compositor.cpp:104`; `d3d11/comp_d3d11_target.cpp:32`; `d3d12/comp_d3d12_compositor.cpp:94`; `d3d12/comp_d3d12_target.cpp:33`; `vk_native/comp_vk_native_compositor.c:138`; `vk_native/comp_vk_native_target.cpp:75` | `DEBUG_GET_ONCE_BOOL` ×6 **independent TU caches** | off | App | 1 | Opaque flip chain instead of the composed chain. Sets swapchain `AlphaMode` | | `DXR_APP_HWND_LATENCY` | `compositor/d3d11_service/comp_d3d11_service.cpp:263` | `getenv`, **uncached** | 2 | Svc | 1 | Frame-latency depth on the app-HWND present path. Consumed at `SetMaximumFrameLatency` | | `DXR_COMMIT_PACE` | `compositor/d3d11_service/comp_d3d11_service.cpp:19741` | `getenv`, `static` cached | on (1; 0/1/2) | Svc | 2 | Commit backpressure. `=0` lets a client run free; `=2` is the legacy tick-align shape | +| `DXR_LIFT_MAX_INPUT_EDGE` | `compositor/d3d11_service/d3d11_lift.cpp` (`d3d11_lift_create`, parsed by `u_lift_max_input_edge_parse`; applied per snapshot in `submit_locked` via `u_lift_cap_dims`) | `getenv` once at lift-module create | **1920** (`0` = off; 1-255 clamp to 256) | Svc | 2 | XR_DXR_lift (ADR-042): caps the long edge of a lift-flagged weave rect's snapshot before the vendor module (aspect kept, dims even, box-filtered). The module synthesizes views at INPUT resolution and the result is stretched back into the rect, so a device-pixel snapshot only costs time: measured on an 8K Leia panel (NeurD DirectML) a fullscreen 7680x4319 player converted at 132 ms (~6.5 Hz) vs 48 ms at 1080p-class input. Never applied to an app's explicit `xrSubmitLiftFrameDXR` frame. Latched only by the create-time read — could go live. **User-facing: yes** — the natural "Convert-to-3D quality" tier (e.g. Fast 1280 / Balanced 1920 / Sharp 3840 / Native 0) | +| `DXR_LIFT_LETTERBOX` | `compositor/d3d11_service/d3d11_lift.cpp` (`d3d11_lift_create`, parsed by `u_lift_letterbox_parse`; measured per snapshot in `letterbox_step`, decided by `u_lift_letterbox_update`, applied in `submit_locked`; recomposed in `comp_d3d11_service.cpp` `lift_weave_rect_batch` / `lift_weave_rects_nview`) | `getenv` once at lift-module create | **on** (`0` = off) | Svc | 2 | XR_DXR_lift (ADR-042): crops a lift-flagged weave rect's snapshot to its ACTIVE area when it holds black letterbox/pillarbox bars (GPU row/column non-black profile, async readback, ~45-frame settle, instant un-crop, symmetry rule so subtitles in a bar stay in it), before the `DXR_LIFT_MAX_INPUT_EDGE` cap. The bars (subtitles included) are woven FLAT, identical in both views. Measured on the 8K Leia panel: 3386x1904 rect with a 2.39:1 still + subtitle → active 3386x1422, bars |L-R| = 0.00, module input 1920x806 instead of 1920x1080. Never applied to an app's explicit `xrSubmitLiftFrameDXR` frame. **User-facing: maybe** — an on/off "Crop black bars" toggle next to the Convert-to-3D quality tier | | `DXR_FENCE_WAIT_MS` | `compositor/d3d11_service/comp_d3d11_service.cpp:15911` | `getenv`, `static` cached | 4 (clamped 0..16) | Svc | 2 | Budget for the workspace-sync fence wait on a client's IPC thread | | `DXR_CMD_QUEUE` | `compositor/d3d11_service/comp_d3d11_service.cpp:2341` | `getenv`, `static` cached | on | Svc | 1 | Command-queue submission path vs. the fair-lock path (A/B) | | `DXR_COMPOSE_FROM_COPY` | `compositor/d3d11_service/comp_d3d11_service.cpp:2328` | `getenv`, `static` cached | **off** ("until soaked") | Svc | 1 | Compose from a service-owned `CopyResource` instead of sampling the shared handle | diff --git a/docs/specs/extensions/XR_DXR_lift.md b/docs/specs/extensions/XR_DXR_lift.md new file mode 100644 index 000000000..107be52a9 --- /dev/null +++ b/docs/specs/extensions/XR_DXR_lift.md @@ -0,0 +1,382 @@ +# XR_DXR_lift — 2D→3D Conversion Service + +| Field | Value | +|---|---| +| **Extension Name** | `XR_DXR_lift` | +| **Spec Version** | 1 | +| **Extension Type** | Instance extension (service path — Windows/D3D11; advertised on every desktop platform, where a service without a module reports `supportedModes = 0`) | +| **Header** | `src/external/openxr_includes/openxr/XR_DXR_lift.h` (canonical; auto-syncs to `displayxr-extensions`) | +| **Status** | Provisional (`1004999270–280` block, pending Khronos registry) | +| **Decision record** | [ADR-042](../../adr/ADR-042-vendor-2d3d-conversion-supersedes-default.md) | +| **Plug-in contract** | `src/xrt/include/xrt/xrt_dp_lift.h` + the lift slots of `xrt_display_processor_d3d11` (`XRT_DP_D3D11_HAS_LIFT`), [`xrt_plugin_iface.md` § lift](../../reference/xrt_plugin_iface.md#turning-2d-into-3d-the-lift-slots-adr-042-xr_dxr_lift) | + +## 1. What it is + +A display vendor may ship a **2D→3D conversion module** with its plug-in: monocular depth, +stereo synthesis, N-view synthesis, or photo → Gaussian splats. `XR_DXR_lift` exposes that +module to apps **generically**, exactly the way `XR_DXR_weave` exposes the vendor's weaver: the +caller learns what the module can do and what it produced, never which model runs. + +Four ideas carry the whole extension: + +1. **A READY vendor module supersedes the open default** (ADR-042). A consumer that ships its + own open converter — the DisplayXR Browser's and the web SDK's depth estimator + view + generator — uses the runtime's module whenever `xrGetLiftPropertiesDXR` reports it READY for + the mode it needs. The vendor module is calibrated for the panel it is plugged into. +2. **Lift never weaves.** Results are pre-weave SBS / N-view pixels, a depth map, or a splat + blob. Weaving stays the display processor's job on the ordinary weave path (ADR-007). +3. **Asynchronous, one frame behind.** Conversion runs on a runtime-owned thread. A submit is a + snapshot into a latest-wins mailbox; an acquire returns the newest finished result. Nothing + on the caller's frame path — and nothing on the service's weave, render or IPC threads — + ever waits for the model. +4. **Geometry comes from the weave rect.** When lifted content is woven + (`XrWeaveSubmitLiftRectsDXR`), the service weaves the latest result at the rect's *current* + position every frame. Drag, resize and scroll are exact and real-time; only the depth lags. + +```c +XrLiftPropertiesDXR props = {XR_TYPE_LIFT_PROPERTIES_DXR}; +xrGetLiftPropertiesDXR(session, &props); // cheap; kicks activation +if (props.state == XR_LIFT_STATE_READY_DXR && (props.supportedModes & XR_LIFT_MODE_SBS_BIT_DXR)) { + XrLiftStreamCreateInfoDXR ci = {XR_TYPE_LIFT_STREAM_CREATE_INFO_DXR, NULL, + XR_LIFT_MODE_SBS_DXR, XR_LIFT_CONTENT_HINT_VIDEO_DXR, 1.0f}; + xrCreateLiftStreamDXR(session, &ci, &stream); + // per frame: + xrSubmitLiftFrameDXR(stream, &submit, &frameId); // never waits for the model + XrLiftResultDXR res = {XR_TYPE_LIFT_RESULT_DXR}; + if (xrAcquireLiftResultDXR(stream, &res) == XR_SUCCESS) { /* newer result */ } +} +``` + +## 2. Availability, states and discovery + +| Situation | `xrGetLiftPropertiesDXR` | +|---|---| +| In-process session (any platform) | `XR_ERROR_FEATURE_UNSUPPORTED` — lift is IPC-only, like the weave service | +| Service without a module (macOS, Linux, Android, or a Windows plug-in without lift slots, sim_display) | `XR_SUCCESS`, `supportedModes = 0`, `UNAVAILABLE` | +| First query on a Windows service with a module | `ACTIVATING` (the service is bringing its lift device + the plug-in's lift DP up in the background) | +| Module loading (model weights, licence check, engine build) | `ACTIVATING` — poll ≤ 2 Hz, do not fall back permanently | +| Module up | `READY`, `supportedModes` = the module's bits, `backend` = its name | +| Module failed / `DXR_LIFT=0` on the service | `UNAVAILABLE` | + +The first `xrGetLiftPropertiesDXR` (or `xrCreateLiftStreamDXR`) of the service's lifetime starts +activation; the call itself never waits. `typicalLatency` is the module's own estimate +(submit → result); the measured value per stream is in `xrGetLiftStreamStatsDXR` (§7). + +`maxStreams` is **service-wide**, not per session. `maxViews` bounds `XrLiftOptionsDXR::viewCount` +for NVIEW. `depthSemantics` says whether a DEPTH result is relative (larger = farther, no unit) +or metric (metres). + +**Streams may be created while ACTIVATING** (their frames wait, latest-wins, until READY). A +mode not in `supportedModes` of a READY module is `XR_ERROR_FEATURE_UNSUPPORTED`; creating past +`maxStreams` is `XR_ERROR_LIMIT_REACHED`. + +## 3. Streams, modes and result layouts + +| Mode | Result | Acquire with | Layout | +|---|---|---|---| +| `DEPTH` | depth map | `xrAcquireLiftResultDXR` | one channel (`format` tells: `R32_FLOAT`, `R8_UNORM`, …) at the module's inference resolution | +| `SBS` | stereo pair, **not woven** | `xrAcquireLiftResultDXR` | two views side by side, left view left; `viewCount = 2` | +| `NVIEW` | N views, **not woven** | `xrAcquireLiftResultDXR` | `viewCount` views side by side in ONE row, view 0 leftmost | +| `GAUSSIANS` | 3D Gaussian splats | `xrAcquireLiftBlobDXR` | a binary little-endian reference-3DGS PLY, or a PlayCanvas SOG container (`format`) | + +The tile size is the module's inference resolution, not necessarily the input size — always read +`extent` / `viewCount`. `contentHint` VIDEO lets a module keep temporal state across frames; +PHOTO asks for quality over latency. GAUSSIANS streams take PHOTO only. `inputScale` in (0, 1] +asks the module to convert at a reduced resolution — an advisory latency lever. + +## 4. Submitting frames + +```c +XrLiftOptionsDXR opt = {XR_TYPE_LIFT_OPTIONS_DXR, NULL, + XR_LIFT_CONVERGENCE_AUTO_DXR, /*strength*/ 1.0f, /*inpaint*/ XR_TRUE, + XR_LIFT_VIEWPOINT_SOURCE_TRACKED_DXR, /*viewCount*/ 2, /*viewpoints*/ NULL, /*focalPx*/ 0.0f}; +XrLiftFrameSubmitInfoDXR submit = {XR_TYPE_LIFT_FRAME_SUBMIT_INFO_DXR, &opt, + sharedHandle, /*inputIsDxgi*/ XR_FALSE, {w, h}, sourceTime}; +xrSubmitLiftFrameDXR(stream, &submit, &frameId); +``` + +**Non-blocking.** The service acquires the input's keyed mutex (key 0 = "caller done writing", +the weave contract; 4 ms budget), **snapshots** `extent` (the top-left sub-rect of +`inputTexture`) into the stream's mailbox with one blit on its own GPU, releases the mutex, and +returns. The caller may overwrite the texture immediately. The call costs one IPC round trip + +one blit — the same order as a weave submit, never the model's latency. + +**Latest wins.** The mailbox has two input slots. While the module converts one frame, the other +slot always accepts a new one; a frame still waiting when the next arrives is **dropped** +(counted in `framesDropped`), never queued. A slow module therefore lags by one frame instead +of building a backlog. + +**Handle kinds** are XR_DXR_weave v3's: a D3D11 NT shared handle, or a legacy global DXGI +handle with `inputIsDxgi = XR_TRUE` (Low-integrity callers, #743). RGBA8 or BGRA8. The service +caches the import per stream (NT handles by kernel-object identity, DXGI handles by value). + +**`frameId`** is per stream, monotonic from 1. `0` means the frame was **not taken** because the +module is not up yet (still ACTIVATING); submit again next frame. + +**Options** (`XrLiftOptionsDXR`, optional; omitted = the stream's last options, initially the +module defaults): + +| Field | Meaning | +|---|---| +| `convergence` | RELATIVE depth placed at the display plane, in [0, 1] over the frame's depth range (0 = nearest content on the glass, 1 = farthest, 0.5 = mid). Negative = AUTO. >1 is clamped. The plug-in maps it to its model's units and calibrates it for its panel. | +| `strength` | disparity scale; 1 = the module's calibrated budget, 0 = flat | +| `inpaint` | fill disocclusions (XR_TRUE) or leave them | +| `viewpointSource` | TRACKED (the runtime passes the panel's predicted tracked eyes) or EXPLICIT (`viewpoints`, `viewCount` display-space positions in metres) | +| `viewCount` | views for NVIEW (≤ `maxViews`); 2 for SBS | +| `focalPx` | the input image's focal length in pixels of `extent`; ≤ 0 = unknown (the module assumes its default FOV). Photo → Gaussians modules take it as input; DEPTH/SBS/NVIEW modules ignore it | + +**Viewpoints always reach the module explicitly.** With TRACKED, the lift thread samples the +panel display processor's predicted eyes just before each conversion and hands the pair to the +module; a module spreads N views around it. The lift display processor has no tracker session +of its own. + +## 5. Acquiring results + +### 5.1 Textures (DEPTH / SBS / NVIEW) + +`xrAcquireLiftResultDXR` returns the newest finished result **newer than the last one it +returned**, or the success code `XR_LIFT_NOT_READY_DXR` (keep presenting the previous result). + +Handles follow the XR_DXR_weave output contract. The service copies the result into a +per-stream **export texture** and signals the export **fence**: + +- `outputTexture` / `fence` are shared HANDLEs, handed out on the **first** acquire and again + whenever the export texture is **reallocated** (result size or format changed — watch + `extent` / `format`); `NULL` on steady-state acquires. The caller imports them once per + allocation and closes its handles when done. +- The caller GPU-waits `fence` to `fenceValue` before sampling, and **finishes sampling before + its next acquire on the stream** — the next result is copied into the same texture. +- The runtime latches "exported" only when both handles were produced (the #1427 rule), so a + transient export miss retries on the next acquire rather than leaving the caller handle-less. + +`sourceTime` is the submitting call's value, verbatim; `frameId` identifies the frame; +`latency` is submit → conversion finished on the runtime clock. + +### 5.2 Blobs (GAUSSIANS) + +`xrAcquireLiftBlobDXR` uses the standard two-call idiom on `byteCapacityInput`: + +1. Capacity 0 → fills `byteCountOutput`, `frameId`, `sourceTime`, `format` and **latches** that + blob in the service. +2. Capacity ≥ `byteCountOutput` → the latched blob's bytes (the SAME frame, even if a newer one + finished in between); the latch is consumed. + +A non-zero capacity smaller than the blob is `XR_ERROR_SIZE_INSUFFICIENT` (latch kept). +`XR_LIFT_NOT_READY_DXR` when nothing newer than the last blob handed out exists. A texture +acquire on a GAUSSIANS stream (and vice versa) is `XR_ERROR_VALIDATION_FAILURE`. Blobs larger +than 256 MiB are refused by the transport (`XR_ERROR_RUNTIME_FAILURE`). + +Why a separate entry point rather than a struct chained on the texture acquire: the two result +kinds never coexist on a stream, and a blob needs the two-call size negotiation a texture does +not. One call per result kind keeps each contract minimal. + +## 6. Lifting weave rects + +```c +XrLiftOptionsDXR opt = {XR_TYPE_LIFT_OPTIONS_DXR, NULL, XR_LIFT_CONVERGENCE_AUTO_DXR, 1.0f, XR_TRUE, + XR_LIFT_VIEWPOINT_SOURCE_TRACKED_DXR, 2, NULL, 0.0f}; +XrWeaveRectLiftDXR lift = {XR_TYPE_WEAVE_RECT_LIFT_DXR, &opt, /*rectIndex*/ 3, sbsStream}; +XrWeaveSubmitLiftRectsDXR lifts = {XR_TYPE_WEAVE_SUBMIT_LIFT_RECTS_DXR, NULL, 1, &lift}; +XrWeaveSubmitRectsDXR rects = {XR_TYPE_WEAVE_SUBMIT_RECTS_DXR, &lifts, n, rectArray}; +XrWeaveSubmitInfoDXR in = {XR_TYPE_WEAVE_SUBMIT_INFO_DXR, &rects, input, ...}; +xrWeaveSubmitDXR(session, &in, &out); +``` + +"The content of weave rect `rectIndex` is 2D — lift it before weaving." An `XrRect2Di` has no +`next`, so the association is by index into `XrWeaveSubmitRectsDXR::rects`; up to +`XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR` (8) rects per submit. The stream must be an SBS or NVIEW +stream of the same session; options on the weave path must use TRACKED viewpoints. + +What the caller draws: + +| Weave layout | Where the 2D frame goes | +|---|---| +| Batch (v3/v4/v5) — window-sized input | the WHOLE rect holds the 2D frame (not squeezed SBS) | +| N-view atlas (v6) | the 2D frame at the rect's scaled position in EVERY tile (tile 0's copy is lifted) | + +What the service does, inside the same `xrWeaveSubmitDXR`: + +1. **Snapshot.** The rect's region (tile 0's on v6) is blitted into the stream's mailbox — + exactly a `xrSubmitLiftFrameDXR`, stamped with the runtime clock — except that the service + **downsamples** it so its long edge is at most a cap (default **1920**, aspect kept, both + dims even; service env `DXR_LIFT_MAX_INPUT_EDGE`, `0` = off, smaller values clamp to 256). + The rect is in device pixels (a fullscreen player on an 8K panel is 7680x4319), the module + synthesizes its views at *input* resolution, and step 2 stretches the result back into the + rect anyway — so an uncapped snapshot buys little sharpness at a large conversion-rate cost. + The module sees the capped size as the frame's input size (a `focalPx` hint is scaled with + it). An app's own `xrSubmitLiftFrameDXR` frame is never capped: its size is the app's choice. + + **Letterbox crop** (before the cap; service env `DXR_LIFT_LETTERBOX`, default on, `0` = off). + The service measures each rect's rows and columns (the fraction of non-black pixels per + bucket, a small GPU reduction read back asynchronously) and, once black bars have held for + ~45 frames, snapshots only the **active** area. A bar GROWS only into truly black buckets + (< 3 % non-black — a dark scene's edge is rarely that empty), and SHRINKS only when picture + (≥ 25 %) intrudes for 6 consecutive frames and by more than the profile's jitter, so a + caption burst does not un-crop. A shorter bar is extended to match the opposite one when the + extra band is separated from the picture by a black gap (a subtitle line, however dense) or + is only sparsely lit (< 60 %). Frames without picture (a cut to black) change nothing; a + rect resize resets the crop. One WARN per stream per crop change. Like the cap, never + applied to an app's own `xrSubmitLiftFrameDXR` frame. +2. **Weave the latest result at the CURRENT rect.** The stream's newest result (from an earlier + frame) is stretched into the rect's current position — into its **active** part when the + letterbox crop was in effect for that result; the bars (subtitles included) are then the 2D + frame, written identically into both views AFTER the result and a few pixels into the active + area (covering the profile's one-bucket edge slack and the module's edge seam) (v3), i.e. + flat on the screen plane, or left as + the caller drew them (v6): the middle stereo pair into the SBS scratch's left/right tiles + (v3), or view *v* of the result into tile *v* (v6; equal view counts map 1:1, otherwise + proportionally). Then the ordinary single weave of the whole window runs. +3. **Flat until the first result.** With no result yet, the rect is woven flat: the service + writes the 2D frame into both views (v3), or leaves the caller's identical tiles as drawn (v6). + +So position and size are always this frame's, and only the depth is one conversion behind. +A v6 frame carrying lift rects never takes the zero-copy path (the service writes into its crop +copy, never the caller's texture). + +**Wire.** The lifted-rect set travels as its own IPC call (`lift_weave_rects`) immediately +before the `weave_submit` it belongs to, on the same connection; the service consumes it with +that one submit. The weave wire itself is byte-identical. A rect naming a stream the connection +does not own, or a DEPTH/GAUSSIANS stream, is refused (non-fatal) and woven as drawn. + +**Compatibility.** A runtime without `XR_DXR_lift` ignores the unknown chained struct and weaves +the rect as whatever the caller drew — flat 2D, never a misread. No version gate is needed; the +extension being enabled is the gate. + +## 7. Scheduling, priority and stats + +The module converts one frame at a time (one GPU, often one serialised inference queue), so +concurrent streams share its throughput. The lift thread runs **rounds**; each round converts, +in order: + +| Priority | Converted | +|---|---| +| `HIGH` | every HIGH stream with a new frame, every round | +| `NORMAL` (default) | ONE NORMAL stream with a new frame per round, round-robin | +| `LOW` | every LOW stream with a new frame, on every 4th round | +| `PAUSED` | never — it keeps serving its last result (the weave keeps weaving it; acquires report NOT READY) | + +A round is counted only when some non-paused stream had a frame, so an idle service does not +spend LOW rounds. Example — a call with four tiles and the active speaker HIGH: the speaker +converts every round, the other three rotate one per round. + +`xrSetLiftStreamPriorityDXR` is a per-stream setter (one IPC call, effective at the next round), +because priority changes rarely — when the active speaker changes — and a per-frame field would +cost nothing extra but invite churn. Scheduling is runtime policy; the vendor module never sees +it. + +`xrGetLiftStreamStatsDXR` returns the stream's counters (`framesSubmitted`, `framesConverted`, +`framesDropped`, `framesFailed`), submit → result latency (last, moving average, min, max) and +the **effective conversion rate** (results per second, moving average over publish intervals) — +what the priority actually buys under the current load. One IPC round trip, no GPU work: poll at +UI rates. + +## 8. Timestamps and latency + +- `sourceTime` is the caller's (a video PTS, an `XrTime`); the runtime never interprets it and + echoes it on the result produced from that frame, so a late result can be paired with its + source. +- `latency` / stats latencies are the runtime clock from the moment the snapshot was committed to + the moment the result was published into the ring (i.e. excluding the caller's acquire poll). +- A consumer that wants "the depth for the frame I am showing" keys on `sourceTime`; one that + wants "the freshest depth" ignores it. The weave path always uses the freshest. + +## 9. Service implementation (for reviewers) + +- **One lift thread** per service (`d3d11_lift.cpp`), created on the first lift call. It owns a + **dedicated D3D11 device** on the service adapter (`ID3D11Multithread` protection on) and the + plug-in's **lift-only display processor** (`xrt_plugin_iface::create_dp_d3d11_lift`; fallback: + the ordinary factory with a NULL window). It is the only thread that touches that DP. +- **Per stream:** a latest-wins mailbox (two input slots) and a two-slot output ring, both + keyed-mutex shared textures crossing between the service device and the lift device; the + module's output is copied into the ring before its next call. Consumers **pin** a ring slot for + the length of one GPU-copy issue; the worker never overwrites the latest or a pinned slot. The + state machine is platform-neutral and unit-tested (`u_lift_mailbox.h`, + `tests/tests_lift_mailbox.cpp`), as is the round scheduler. +- **Locks:** the lift mutex is a leaf, never held across GPU or vendor work. IPC calls take the + service's `immediate_ctx_mutex` alone, for one blit / copy; the weave path takes the lift mutex + under it. The lift thread never takes the service's context or render locks — see + [service-architecture §3](../../architecture/service-architecture.md#3-threads-and-locks). +- **Logging:** one WARN per module state change and per stream create/destroy; per-stream + statistics at INFO, throttled to one line per 5 s. +- **Kill switch:** `DXR_LIFT=0` in the service's environment → UNAVAILABLE, the vendor DP is + never created. +- **Vendor knobs live in the SERVICE's environment** (e.g. a module's backend selection), not + the app's — the conversion runs in `displayxr-service.exe`. +- **Streams belong to the IPC connection**, not the session, and die with it; the calls need no + compositor session, which is what lets `displayxr-cli lift` drive them headless. + +## 10. Error codes and connection loss + +Mirrors XR_DXR_weave §4b. + +| Result | When | Session still usable? | +|---|---|---| +| `XR_LIFT_NOT_READY_DXR` (success) | no result newer than the last acquired | yes | +| `XR_ERROR_VALIDATION_FAILURE` | a struct out of contract (mode not one bit, bad viewCount, EXPLICIT without viewpoints, EXPLICIT on a weave rect, wrong acquire for the mode, bad `rectIndex`) | yes — caller bug, nothing sent | +| `XR_ERROR_FEATURE_UNSUPPORTED` | in-process session; mode not supported by a READY module; no module | yes — permanent for that request | +| `XR_ERROR_LIMIT_REACHED` | `maxStreams` reached | yes | +| `XR_ERROR_SIZE_INSUFFICIENT` | blob capacity too small (latch kept) | yes | +| `XR_ERROR_RUNTIME_FAILURE` | the service refused this call over a healthy pipe — most commonly the 4 ms input keyed-mutex miss, or a stream the service no longer knows | **yes — retry next frame** | +| `XR_ERROR_INSTANCE_LOST` | the IPC connection is gone | no — recover with a new instance (weave §4b) | +| `XR_ERROR_SESSION_LOST` | any call after that | no | + +## 11. Vendor contract, in one table + +| Runtime owns | Plug-in owns | +|---|---| +| the lift thread, device, DP lifetime | one synchronous conversion per call | +| latest-wins mailbox, drops, frame ids | the model, its backend, licensing, warm-up (reported as ACTIVATING) | +| output ring copies, export textures, fences | an output valid until its next call | +| priority scheduling across streams | nothing about scheduling | +| tracked eyes → explicit viewpoints | honouring explicit viewpoints over any tracker of its own | +| weaving SBS / N-view results (ADR-007) | never weaving a lift result | +| timestamps, latency, stats | reporting `typical_latency_ns` | + +Full slot reference: [`xrt_plugin_iface.md` § lift](../../reference/xrt_plugin_iface.md#turning-2d-into-3d-the-lift-slots-adr-042-xr_dxr_lift). +sim_display carries an env-gated fake (`SIM_DISPLAY_FAKE_LIFT=1`; `SIM_DISPLAY_FAKE_LIFT_LATENCY_MS`, +default 8): shifted SBS/N-view, a gradient depth, a two-layer 3DGS PLY — the whole path runs in CI +without hardware. + +## 12. Probe and diagnostics + +``` +displayxr-cli lift caps [--json] [--wait S] +displayxr-cli lift probe [--mode depth|sbs|nview|gaussians] [--n N] + [--views N] [--strength F] [--convergence F] [--focal PX] + [--priority paused|low|normal|high] [--pipelined] [--fps F] [--out DIR] +``` + +Connects over IPC as a DIAG client (non-elevated prompt on Windows), waits through ACTIVATING, +creates a stream, submits the image(s) through a shared keyed-mutex texture exactly as a +browser does, reads each result back through the export texture + fence and writes +`lift_out_.png` (depth normalised to 8-bit grey) or `lift_out_.ply`. It prints per frame +the submit → acquire latency seen by the caller and the service's own latency, and at the end +the stream stats and a latency summary. `--pipelined` submits at `--fps` without waiting, to +measure throughput and drops. `displayxr-cli selftest` includes a `lift_caps` check (headless +WARP + the lift-only factory): modes 0 passes; only malformed caps fail +(`CLI_SELFTEST_BAD_LIFT_CAPS`, exit 11). + +## 13. Consumers + +| Consumer | Path | Uses | +|---|---|---| +| DisplayXR Browser (Chromium fork) | GPU process → service | `xrGetLiftPropertiesDXR` to decide vendor vs open default (ADR-042); `XrWeaveSubmitLiftRectsDXR` for inline 2D video/images lifted in place; `xrSubmitLiftFrameDXR` + acquire where the page wants the depth/views itself | +| DisplayXR web SDK (`@displayxr/inline3d`) | via the browser | the same policy: prefer the runtime module when READY; GAUSSIANS via the blob path supersedes the SDK's open photo → splats lift | +| 3D calling (up to 4 mono tiles) | app → service | one SBS stream per tile; `xrSetLiftStreamPriorityDXR` HIGH for the active speaker | +| `displayxr-cli lift` | DIAG IPC | caps + the N0 probe (§12) | + +When changing the header, byte-sync every consumer's vendored copy and rebuild it — coupled-PR +order: runtime → extensions auto-sync → consumers. + +## 14. Version history + +| Version | Change | +|---|---| +| 1 | Initial: properties + states, streams (DEPTH / SBS / NVIEW / GAUSSIANS), non-blocking latest-wins submit, texture acquire (weave-style handles + fence), blob acquire (two-call latch), weave-rect lift chain, per-stream priority scheduling + stats, `focalPx`. | + +## Probing on a Windows box — gotchas (first N0 run, 2026-09-25) + +- **Selecting the display processor.** `XRT_PREFERRED_PLUGIN_ID` does *not* switch the active DP when an installed vendor plug-in also probes; `DXR_PLUGIN_EXCLUSIVE=` in the **service's** environment does (relaunch the service non-elevated from a `.bat` that sets it). `SIM_DISPLAY_FAKE_LIFT=1` alongside `DXR_PLUGIN_EXCLUSIVE=sim-display` exercises the whole path hardware-free (caps modes = DEPTH|SBS|NVIEW|GAUSSIANS). +- **`displayxr-cli lift …` must run non-elevated** — an elevated prompt reports "not connected to the service" (same integrity-level rule as every other IPC client). +- **Vendor module version gate.** The Leia plug-in logs the NeurD version it loaded and reports `UNAVAILABLE` when it is outside the supported range (needs 0.4.3+): e.g. a box with Immersity Live's NeurD 0.3.7 shows `state -> ACTIVATING` then `UNAVAILABLE` with the reason in the service log. Upgrade the NeurD runtime; the plug-in never crashes on a mismatch. +- **Probe timing.** `lift probe` stamps submit→acquire before any readback; use `--no-write` (or `--write-every N`) for throughput runs — encoding a 4K SBS PNG takes seconds and would otherwise cap the pipelined rate. diff --git a/docs/specs/extensions/index.json b/docs/specs/extensions/index.json index 7c53d27b5..c10cb4d56 100644 --- a/docs/specs/extensions/index.json +++ b/docs/specs/extensions/index.json @@ -77,6 +77,11 @@ "title": "Window Weave Service", "summary": "Window-bound synchronous weave for present-owners: hand the runtime a stereo texture and a window rect, get back a weaved shared texture and a fence" }, + "XR_DXR_lift": { + "group": "rendering", + "title": "2D-to-3D Conversion", + "summary": "Asynchronous access to a vendor's 2D-to-3D conversion module (depth, stereo, N-view, or photo-to-Gaussian-splats): submit 2D frames, acquire the latest converted result, or flag a weave rect as 2D so the service lifts it before weaving" + }, "XR_DXR_win32_window_binding": { "group": "windowing", "title": "Win32 Window Binding", diff --git a/src/external/openxr_includes/openxr/README.md b/src/external/openxr_includes/openxr/README.md index 345abb979..69b1720a6 100644 --- a/src/external/openxr_includes/openxr/README.md +++ b/src/external/openxr_includes/openxr/README.md @@ -46,7 +46,9 @@ Rules: | 1004999240–249 | `XR_DXR_weave` (v9+ additions) | 240 = `XR_TYPE_WEAVE_IPC_CONNECTION_DXR` (`xrWeaveExportIpcConnectionDXR`, spec v9, browser#103); 241 = `XR_TYPE_WEAVE_DMABUF_DESC_DXR`, 242 = `XR_TYPE_WEAVE_OVERLAY_DMABUF_DESC_DXR`, 243 = `XR_TYPE_WEAVE_OUTPUT_DMABUF_DXR`, 244 = `XR_TYPE_WEAVE_SUBMIT_SYNC_DXR`, 245 = `XR_TYPE_WEAVE_OUTPUT_SYNC_DXR` (desktop-Linux dma-buf transport + sync_file fences, spec v10, #1699); 246 = `XR_TYPE_WEAVE_SNAP_GRID_INFO_DXR` (`xrWeaveSnapWindowGridDXR`, bulk grid snap, spec v11, #1723); 247 = `XR_TYPE_WEAVE_OUTPUT_ORIGIN_DXR`, 248 = `XR_TYPE_WEAVE_WINDOW_LOGICAL_ORIGIN_DXR` (woven origin per output for present-owner move sync, spec v12, browser-pvt#180). A fresh decade because the original 190–199 block is exhausted (199 is reserved for the wish mask) | | 1004999250–259 | `XR_DXR_wayland_surface_binding` | 250 = `XR_TYPE_WAYLAND_SURFACE_BINDING_CREATE_INFO_DXR` (#757), 251 = `XR_TYPE_WAYLAND_SURFACE_GEOMETRY_DXR` (spec v2) — the app-declared surface size + output refresh, because a `wl_surface` has no intrinsic size and the compositor geometry service cannot bootstrap one. **250 was renumbered from 1004999210**, which collided with `XR_TYPE_DISPLAY_DESKTOP_POSITION_DXR` — the 210–219 decade was already `XR_DXR_display_info`'s. Safe to move: SPEC_VERSION 1, no shipped app chains it | | 1004999260–269 | `XR_DXR_depth_budget` | 260 = `XR_TYPE_REAR_DEPTH_BUDGET_DXR`, 261 = `XR_TYPE_CONTENT_BOUNDS_DXR`, 262 = `XR_TYPE_EVENT_DATA_REAR_DEPTH_BUDGET_STATE_CHANGED_DXR`, 263 = `XR_TYPE_CONTENT_MASK_DXR` (ADR-040). Was taken without a registry row — recorded retroactively in #1486 PR 2 | -| 1004999270+ | **next free** | | +| 1004999270–279 | `XR_DXR_lift` | 270 = `XR_TYPE_LIFT_PROPERTIES_DXR`, 271 = `XR_TYPE_LIFT_STREAM_CREATE_INFO_DXR`, 272 = `XR_TYPE_LIFT_FRAME_SUBMIT_INFO_DXR`, 273 = `XR_TYPE_LIFT_OPTIONS_DXR`, 274 = `XR_TYPE_LIFT_RESULT_DXR`, 275 = `XR_TYPE_WEAVE_SUBMIT_LIFT_RECTS_DXR`, 276 = `XR_TYPE_WEAVE_RECT_LIFT_DXR`, 277 = `XR_TYPE_LIFT_BLOB_DXR` (spec v1, ADR-042). 278 = `XR_OBJECT_TYPE_LIFT_STREAM_DXR` — an `XrObjectType`, not an `XrStructureType`. 279 = `XR_LIFT_NOT_READY_DXR` — a **success**-class `XrResult` (positive) | +| 1004999280–289 | `XR_DXR_lift` (continued) | 280 = `XR_TYPE_LIFT_STREAM_STATS_DXR` (spec v1). A second decade because 270–279 is full | +| 1004999290+ | **next free** | | `XR_DXR_android_surface_binding` was implemented in #1037 and now has its own header and decade — see the row above for why its create-info type value sits diff --git a/src/external/openxr_includes/openxr/XR_DXR_lift.h b/src/external/openxr_includes/openxr/XR_DXR_lift.h new file mode 100644 index 000000000..c661443ab --- /dev/null +++ b/src/external/openxr_includes/openxr/XR_DXR_lift.h @@ -0,0 +1,441 @@ +// Copyright 2026, DisplayXR +// SPDX-License-Identifier: Apache-2.0 +// +// PROVISIONAL — DXR is DisplayXR's Khronos-registered OpenXR author ID, but +// the XR_DXR_* extensions in this header are NOT yet registered in the +// Khronos OpenXR registry: extension numbers and XrStructureType values sit +// in a provisional experimental block (1004999xxx) pending official +// assignment. Extension names are expected to be stable; numeric values are +// not. +// See GOVERNANCE.md. +// +/*! + * @file + * @brief Header for XR_DXR_lift extension + * @author David Fattal + * @ingroup external_openxr + * + * 2D→3D CONVERSION ("lift") as a runtime service. A vendor display plug-in may + * ship a conversion module — monocular depth, stereo (SBS) synthesis, N-view + * synthesis, or photo → Gaussian splats — and this extension exposes it + * generically, the same way XR_DXR_weave exposes the vendor's weaver: the + * caller never learns which model runs, only what it can do (@ref + * XrLiftPropertiesDXR) and what it produced. + * + * Policy (ADR-042): when the runtime reports a READY module, a consumer that + * also ships an OPEN default converter (the browser's / web SDK's depth + * estimator + view generator) must prefer the runtime's. The vendor module is + * calibrated for the panel it is plugged into; the open default is not. + * + * Output is NEVER woven here. A lift stream returns pre-weave SBS / N-view + * pixels (or a depth map, or a splat blob); weaving stays the display + * processor's job on the ordinary weave path (ADR-007). The one place the two + * meet is XrWeaveSubmitLiftRectsDXR: a weave rect flagged "this content is + * 2D — lift it first", which the service routes through a lift stream and then + * weaves at the rect's CURRENT position with the latest converted result. + * + * Asynchronous by construction. Conversion runs on a runtime-owned thread, one + * frame (or, for photos → splats, seconds) behind the submit: + * + * xrCreateLiftStreamDXR(session, &createInfo, &stream); + * // per frame, never blocks on the model: + * xrSubmitLiftFrameDXR(stream, &submit, &frameId); // latest-wins mailbox + * XrResult r = xrAcquireLiftResultDXR(stream, &result); // newest finished frame + * if (r == XR_LIFT_NOT_READY_DXR) { keep showing the last result } + * + * @c sourceTime is the caller's own timestamp for the submitted frame (a video + * PTS, an XrTime — the runtime never interprets it) and comes back verbatim on + * the result it produced, so a caller can pair a late result with its frame. + * + * Availability: out-of-process (service / IPC) sessions only, on the Windows + * D3D11 service, exactly like XR_DXR_weave. An in-process session reports + * XR_ERROR_FEATURE_UNSUPPORTED from every entry point. + * + * Version history: 1 = initial (properties, streams, texture results, the + * weave-rect lift chain, the Gaussian-splat blob path, per-stream priority + * scheduling and stream stats). + */ +#ifndef XR_DXR_LIFT_H +#define XR_DXR_LIFT_H 1 + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +#define XR_DXR_lift 1 +#define XR_DXR_lift_SPEC_VERSION 1 +#define XR_DXR_LIFT_EXTENSION_NAME "XR_DXR_lift" + +// Reserved 1004999270..279. Allocation registry: README.md in this directory. +#define XR_TYPE_LIFT_PROPERTIES_DXR ((XrStructureType)1004999270) +#define XR_TYPE_LIFT_STREAM_CREATE_INFO_DXR ((XrStructureType)1004999271) +#define XR_TYPE_LIFT_FRAME_SUBMIT_INFO_DXR ((XrStructureType)1004999272) +#define XR_TYPE_LIFT_OPTIONS_DXR ((XrStructureType)1004999273) +#define XR_TYPE_LIFT_RESULT_DXR ((XrStructureType)1004999274) +#define XR_TYPE_WEAVE_SUBMIT_LIFT_RECTS_DXR ((XrStructureType)1004999275) +#define XR_TYPE_WEAVE_RECT_LIFT_DXR ((XrStructureType)1004999276) +#define XR_TYPE_LIFT_BLOB_DXR ((XrStructureType)1004999277) +//! XrObjectType of an XrLiftStreamDXR (debug-utils naming). +#define XR_OBJECT_TYPE_LIFT_STREAM_DXR ((XrObjectType)1004999278) +//! SUCCESS-class result: the acquire found no result newer than the last one +//! it handed out. Not an error — keep presenting the previous result. +#define XR_LIFT_NOT_READY_DXR ((XrResult)1004999279) +// Second decade 1004999280..289 (the first is full). +#define XR_TYPE_LIFT_STREAM_STATS_DXR ((XrStructureType)1004999280) + +//! Size of XrLiftPropertiesDXR::backend, NUL included. +#define XR_LIFT_BACKEND_NAME_MAX_SIZE_DXR 32 +//! Upper bound on XrLiftOptionsDXR::viewCount (and explicit viewpoints). +#define XR_LIFT_MAX_VIEWS_DXR 8 +//! XrLiftOptionsDXR::convergence value asking the module to pick the +//! zero-disparity depth itself. +#define XR_LIFT_CONVERGENCE_AUTO_DXR (-1.0f) +//! Upper bound on lifted rects carried by one XrWeaveSubmitLiftRectsDXR. +#define XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR 8 + +XR_DEFINE_HANDLE(XrLiftStreamDXR) + +typedef XrFlags64 XrLiftModeFlagsDXR; +//! Monocular depth map (one channel; see XrLiftDepthSemanticsDXR). +static const XrLiftModeFlagsDXR XR_LIFT_MODE_DEPTH_BIT_DXR = 0x00000001; +//! Stereo pair, side by side (left view in the left half), NOT woven. +static const XrLiftModeFlagsDXR XR_LIFT_MODE_SBS_BIT_DXR = 0x00000002; +//! N views side by side in one row (view 0 leftmost), NOT woven. +static const XrLiftModeFlagsDXR XR_LIFT_MODE_NVIEW_BIT_DXR = 0x00000004; +//! Photo → 3D Gaussian splats, returned as a blob (xrAcquireLiftBlobDXR). +//! PHOTO content only; seconds, not frames. +static const XrLiftModeFlagsDXR XR_LIFT_MODE_GAUSSIANS_BIT_DXR = 0x00000008; + +//! The ONE mode a stream runs in (a single bit of XrLiftModeFlagsDXR). +typedef enum XrLiftModeDXR { + XR_LIFT_MODE_DEPTH_DXR = 1, + XR_LIFT_MODE_SBS_DXR = 2, + XR_LIFT_MODE_NVIEW_DXR = 4, + XR_LIFT_MODE_GAUSSIANS_DXR = 8, + XR_LIFT_MODE_MAX_ENUM_DXR = 0x7FFFFFFF +} XrLiftModeDXR; + +typedef enum XrLiftDepthSemanticsDXR { + //! Relative (affine-invariant) depth: larger = farther, no unit. + XR_LIFT_DEPTH_SEMANTICS_RELATIVE_DXR = 0, + //! Metric depth in metres. + XR_LIFT_DEPTH_SEMANTICS_METRIC_DXR = 1, + XR_LIFT_DEPTH_SEMANTICS_MAX_ENUM_DXR = 0x7FFFFFFF +} XrLiftDepthSemanticsDXR; + +typedef enum XrLiftStateDXR { + //! No module, or the module failed. supportedModes is 0. + XR_LIFT_STATE_UNAVAILABLE_DXR = 0, + //! A module exists and is loading (model weights, engine build). Poll. + XR_LIFT_STATE_ACTIVATING_DXR = 1, + //! Streams may be created and fed. + XR_LIFT_STATE_READY_DXR = 2, + XR_LIFT_STATE_MAX_ENUM_DXR = 0x7FFFFFFF +} XrLiftStateDXR; + +typedef enum XrLiftContentHintDXR { + //! Temporally coherent frames: the module may use temporal state. + XR_LIFT_CONTENT_HINT_VIDEO_DXR = 0, + //! Independent stills: favour quality over latency. + XR_LIFT_CONTENT_HINT_PHOTO_DXR = 1, + XR_LIFT_CONTENT_HINT_MAX_ENUM_DXR = 0x7FFFFFFF +} XrLiftContentHintDXR; + +typedef enum XrLiftViewpointSourceDXR { + //! Synthesize for the runtime's tracked eyes (the normal display case). + XR_LIFT_VIEWPOINT_SOURCE_TRACKED_DXR = 0, + //! Synthesize for XrLiftOptionsDXR::viewpoints (display space, metres). + XR_LIFT_VIEWPOINT_SOURCE_EXPLICIT_DXR = 1, + XR_LIFT_VIEWPOINT_SOURCE_MAX_ENUM_DXR = 0x7FFFFFFF +} XrLiftViewpointSourceDXR; + +/*! + * How the service's lift thread schedules a stream against the others (the + * module converts one frame at a time — one GPU, often one serialised + * inference queue — so N concurrent streams share its throughput). Each + * scheduling ROUND converts, in order: + * + * - every HIGH stream that has a new frame; + * - ONE NORMAL stream with a new frame, round-robin across NORMAL streams; + * - every LOW stream with a new frame, but only on every 4th round; + * - PAUSED streams never — they keep serving their last result (the weave + * keeps weaving it; an acquire keeps reporting NOT READY). + * + * Default NORMAL. A per-STREAM setting, changed rarely (e.g. when the active + * speaker of a call changes), so it is its own call rather than a per-frame + * field. Scheduling is runtime policy: the vendor module never sees it. + */ +typedef enum XrLiftPriorityDXR { + XR_LIFT_PRIORITY_PAUSED_DXR = 0, + XR_LIFT_PRIORITY_LOW_DXR = 1, + XR_LIFT_PRIORITY_NORMAL_DXR = 2, + XR_LIFT_PRIORITY_HIGH_DXR = 3, + XR_LIFT_PRIORITY_MAX_ENUM_DXR = 0x7FFFFFFF +} XrLiftPriorityDXR; + +typedef enum XrLiftBlobFormatDXR { + //! Binary little-endian PLY in the reference 3DGS layout (x y z nx ny nz + //! f_dc_0..2 [f_rest_*] opacity scale_0..2 rot_0..3). + XR_LIFT_BLOB_FORMAT_PLY_3DGS_DXR = 1, + //! PlayCanvas SOG (self-organizing Gaussians) container. + XR_LIFT_BLOB_FORMAT_SOG_DXR = 2, + XR_LIFT_BLOB_FORMAT_MAX_ENUM_DXR = 0x7FFFFFFF +} XrLiftBlobFormatDXR; + +/*! + * What the runtime's conversion module can do, and whether it can do it NOW. + * + * A consumer with its own open default converter treats + * @c state == XR_LIFT_STATE_READY_DXR plus the needed mode bit as "use the + * runtime's" (ADR-042). ACTIVATING is transient — poll (≤ 2 Hz) rather than + * falling back permanently. UNAVAILABLE with supportedModes 0 is the answer on + * every display whose plug-in ships no module, and on sim_display. + */ +typedef struct XrLiftPropertiesDXR { + XrStructureType type; //!< XR_TYPE_LIFT_PROPERTIES_DXR + void* XR_MAY_ALIAS next; + XrLiftModeFlagsDXR supportedModes; //!< XR_LIFT_MODE_*_BIT_DXR; 0 = none + uint32_t maxStreams; //!< concurrent streams across the whole runtime + uint32_t maxViews; //!< upper bound on XrLiftOptionsDXR::viewCount (NVIEW) + XrLiftDepthSemanticsDXR depthSemantics; //!< meaning of a DEPTH result + XrLiftStateDXR state; + char backend[XR_LIFT_BACKEND_NAME_MAX_SIZE_DXR]; //!< vendor module name, informational + XrDuration typicalLatency; //!< submit→result, ns, as the module reports it; 0 = unknown +} XrLiftPropertiesDXR; + +/*! + * One conversion stream. @c mode is ONE bit the runtime reported in + * XrLiftPropertiesDXR::supportedModes. @c inputScale (0, 1] asks the module to + * convert at a reduced resolution (1.0 = native; 0 is read as 1.0) — a latency + * lever, advisory. + */ +typedef struct XrLiftStreamCreateInfoDXR { + XrStructureType type; //!< XR_TYPE_LIFT_STREAM_CREATE_INFO_DXR + const void* XR_MAY_ALIAS next; + XrLiftModeDXR mode; + XrLiftContentHintDXR contentHint; + float inputScale; +} XrLiftStreamCreateInfoDXR; + +/*! + * Per-frame conversion parameters. Chain on XrLiftFrameSubmitInfoDXR::next or + * XrWeaveRectLiftDXR::next; omitted = the module's defaults (auto convergence, + * strength 1, inpainting on, tracked eyes, 2 views). + * + * @c convergence is the RELATIVE depth placed at the display plane, in [0, 1] + * over the frame's depth range (0 = the nearest content sits on the glass, + * 1 = the farthest does, 0.5 = mid-range); XR_LIFT_CONVERGENCE_AUTO_DXR (any + * negative) lets the module choose. Values above 1 are clamped. @c strength scales the disparity (0 = flat, 1 = the module's + * calibrated budget). @c viewCount is the number of views an NVIEW stream + * produces (2 for SBS; ignored for DEPTH / GAUSSIANS), ≤ maxViews. With + * EXPLICIT viewpoints, @c viewpoints holds @c viewCount display-space positions + * (metres); on the weave path viewpoints must be TRACKED. @c focalPx is the + * submitted image's focal length in pixels of @c extent (for a photo, e.g. the + * fx of an estimated intrinsics); <= 0 = unknown, the module assumes its + * default field of view. Photo → Gaussians modules take it as input; DEPTH / + * SBS / NVIEW modules ignore it. + */ +typedef struct XrLiftOptionsDXR { + XrStructureType type; //!< XR_TYPE_LIFT_OPTIONS_DXR + const void* XR_MAY_ALIAS next; + float convergence; //!< XR_LIFT_CONVERGENCE_AUTO_DXR = auto + float strength; //!< disparity scale, >= 0 + XrBool32 inpaint; //!< fill disocclusions (XR_TRUE) or leave them + XrLiftViewpointSourceDXR viewpointSource; + uint32_t viewCount; //!< 1..XR_LIFT_MAX_VIEWS_DXR (NVIEW); 2 for SBS + const XrVector3f* viewpoints; //!< viewCount entries when EXPLICIT, else ignored + float focalPx; //!< input focal length in input pixels; <= 0 = unknown (GAUSSIANS) +} XrLiftOptionsDXR; + +/*! + * One frame into a stream. NON-BLOCKING: the runtime snapshots the input's + * @c extent (top-left sub-rect of @c inputTexture) into its own mailbox before + * returning — the caller may overwrite the texture immediately after — and + * the frame then waits for the conversion thread. The mailbox is LATEST-WINS: + * a frame still waiting when the next one arrives is dropped (never queued), + * so a slow module lags one frame instead of building a backlog. + * + * @c inputTexture has the same handle kinds as XR_DXR_weave v3: a D3D11 NT + * shared handle, or a legacy global DXGI handle with @c inputIsDxgi = XR_TRUE, + * carrying an IDXGIKeyedMutex (key 0 = "caller done writing"). RGBA8 or BGRA8. + */ +typedef struct XrLiftFrameSubmitInfoDXR { + XrStructureType type; //!< XR_TYPE_LIFT_FRAME_SUBMIT_INFO_DXR + const void* XR_MAY_ALIAS next; //!< chain XrLiftOptionsDXR here + void* inputTexture; //!< shared texture HANDLE (keyed mutex, key 0) + XrBool32 inputIsDxgi; //!< XR_TRUE for a legacy global DXGI handle + XrExtent2Di extent; //!< region of inputTexture to convert, from (0,0) + XrTime sourceTime; //!< caller's timestamp; echoed on the result +} XrLiftFrameSubmitInfoDXR; + +/*! + * The newest finished conversion (texture modes: DEPTH / SBS / NVIEW). + * + * Handles follow the XR_DXR_weave output contract: @c outputTexture and + * @c fence are shared HANDLEs handed out on the FIRST successful acquire and + * again whenever the output is reallocated (size or format change — @c extent + * / @c format tell you); NULL on steady-state acquires. The caller waits + * @c fence to @c fenceValue before sampling, and finishes sampling before its + * next acquire on this stream (the runtime copies the next result into the + * same texture). The texture is runtime-owned; the caller closes its handles. + * + * Layout: DEPTH = one channel at the input's aspect; SBS = two views side by + * side (@c viewCount 2); NVIEW = @c viewCount views side by side, view 0 + * leftmost. @c format is a DXGI_FORMAT value. + */ +typedef struct XrLiftResultDXR { + XrStructureType type; //!< XR_TYPE_LIFT_RESULT_DXR + void* XR_MAY_ALIAS next; + uint64_t frameId; //!< the xrSubmitLiftFrameDXR frame this converts + XrTime sourceTime; //!< that frame's sourceTime, verbatim + void* outputTexture; //!< shared HANDLE on first acquire / realloc, else NULL + void* fence; //!< shared fence HANDLE on first acquire, else NULL + uint64_t fenceValue; //!< wait the fence to this before sampling + XrExtent2Di extent; //!< output texture size (all views) + int64_t format; //!< DXGI_FORMAT + uint32_t viewCount; //!< 1 (DEPTH), 2 (SBS), N (NVIEW) + XrDuration latency; //!< submit → conversion finished, ns (runtime clock) +} XrLiftResultDXR; + +/*! + * The newest finished GAUSSIANS conversion, as bytes (spec v1). + * + * Standard two-call idiom on @c byteCapacityInput: 0 writes @c byteCountOutput + * (and @c frameId / @c sourceTime / @c format) and LATCHES that blob, so the + * second call — with a buffer at least that large — returns the SAME frame even + * if a newer one finished in between. XR_ERROR_SIZE_INSUFFICIENT if the buffer + * is too small (the latch is kept). XR_LIFT_NOT_READY_DXR when nothing newer + * than the last blob handed out exists. + */ +typedef struct XrLiftBlobDXR { + XrStructureType type; //!< XR_TYPE_LIFT_BLOB_DXR + void* XR_MAY_ALIAS next; + uint64_t frameId; + XrTime sourceTime; + XrLiftBlobFormatDXR format; + uint32_t byteCapacityInput; + uint32_t byteCountOutput; + uint8_t* bytes; +} XrLiftBlobDXR; + +/*! + * "The content of weave rect @c rectIndex is 2D — lift it before weaving." + * + * An element of XrWeaveSubmitLiftRectsDXR (an XrRect2Di has no @c next, so the + * association is by index into the submit's XrWeaveSubmitRectsDXR::rects). + * Chain XrLiftOptionsDXR on @c next for per-rect parameters. + * + * The caller draws the element's 2D pixels into the rect exactly as it would + * draw any other element — on the batch (v3) layout the WHOLE rect holds the 2D + * frame (not squeezed SBS); on the v6 N-view layout the 2D frame goes into + * EVERY tile at the rect's scaled position (tile 0's copy is what gets lifted). + * The service snapshots that region into @c stream's mailbox, and weaves the + * stream's LATEST converted output at the rect's CURRENT position — so geometry + * (drag, resize, scroll) is exact and real-time while the conversion itself + * runs one frame behind. Until the stream's first result exists the rect is + * woven FLAT: the service writes the 2D frame into both views (v3), or leaves + * the caller's identical tiles as drawn (v6). + * + * @c stream must be an SBS or NVIEW stream of the same session. + */ +typedef struct XrWeaveRectLiftDXR { + XrStructureType type; //!< XR_TYPE_WEAVE_RECT_LIFT_DXR + const void* XR_MAY_ALIAS next; //!< chain XrLiftOptionsDXR here + uint32_t rectIndex; //!< index into XrWeaveSubmitRectsDXR::rects + XrLiftStreamDXR stream; +} XrWeaveRectLiftDXR; + +/*! + * Chain on XrWeaveSubmitInfoDXR::next (together with XrWeaveSubmitRectsDXR) + * to flag up to XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR rects as 2D-to-lift. + * Requires XR_DXR_lift enabled; ignored by a runtime without it (the rects are + * then woven as whatever the caller drew — flat 2D). + */ +typedef struct XrWeaveSubmitLiftRectsDXR { + XrStructureType type; //!< XR_TYPE_WEAVE_SUBMIT_LIFT_RECTS_DXR + const void* XR_MAY_ALIAS next; + uint32_t liftCount; //!< 0..XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR + const XrWeaveRectLiftDXR* lifts; +} XrWeaveSubmitLiftRectsDXR; + +/*! + * One stream's counters and EFFECTIVE conversion rate — what its priority + * actually buys it under the current load (xrGetLiftStreamStatsDXR). + */ +typedef struct XrLiftStreamStatsDXR { + XrStructureType type; //!< XR_TYPE_LIFT_STREAM_STATS_DXR + void* XR_MAY_ALIAS next; + XrLiftPriorityDXR priority; + uint64_t framesSubmitted; //!< frames the mailbox accepted + uint64_t framesConverted; //!< results published + uint64_t framesDropped; //!< superseded before conversion (latest wins) + uint64_t framesFailed; //!< conversions the module declined + XrDuration latencyLast; //!< submit → result, ns + XrDuration latencyAverage; //!< moving average, ns + XrDuration latencyMin; + XrDuration latencyMax; + float conversionRate; //!< results per second, moving average (0 = none yet) +} XrLiftStreamStatsDXR; + +typedef XrResult (XRAPI_PTR *PFN_xrGetLiftPropertiesDXR)( + XrSession session, XrLiftPropertiesDXR* properties); +typedef XrResult (XRAPI_PTR *PFN_xrCreateLiftStreamDXR)( + XrSession session, const XrLiftStreamCreateInfoDXR* createInfo, XrLiftStreamDXR* stream); +typedef XrResult (XRAPI_PTR *PFN_xrDestroyLiftStreamDXR)(XrLiftStreamDXR stream); +typedef XrResult (XRAPI_PTR *PFN_xrSubmitLiftFrameDXR)( + XrLiftStreamDXR stream, const XrLiftFrameSubmitInfoDXR* submitInfo, uint64_t* frameId); +typedef XrResult (XRAPI_PTR *PFN_xrAcquireLiftResultDXR)(XrLiftStreamDXR stream, XrLiftResultDXR* result); +typedef XrResult (XRAPI_PTR *PFN_xrAcquireLiftBlobDXR)(XrLiftStreamDXR stream, XrLiftBlobDXR* blob); +typedef XrResult (XRAPI_PTR *PFN_xrSetLiftStreamPriorityDXR)(XrLiftStreamDXR stream, XrLiftPriorityDXR priority); +typedef XrResult (XRAPI_PTR *PFN_xrGetLiftStreamStatsDXR)(XrLiftStreamDXR stream, XrLiftStreamStatsDXR* stats); + +#ifndef XR_NO_PROTOTYPES + +//! Query the conversion module. Cheap and non-blocking: the first call may +//! report ACTIVATING while the runtime brings the module up in the background. +XRAPI_ATTR XrResult XRAPI_CALL xrGetLiftPropertiesDXR( + XrSession session, XrLiftPropertiesDXR* properties); + +//! Create a stream in one mode. XR_ERROR_FEATURE_UNSUPPORTED if the mode is not +//! in supportedModes (or the session is in-process); XR_ERROR_LIMIT_REACHED +//! past maxStreams. May be called while ACTIVATING (frames wait for READY). +XRAPI_ATTR XrResult XRAPI_CALL xrCreateLiftStreamDXR( + XrSession session, const XrLiftStreamCreateInfoDXR* createInfo, XrLiftStreamDXR* stream); + +//! Destroy a stream. Its textures stay valid in the caller until it closes +//! its own handles. Destroying the session destroys its streams. +XRAPI_ATTR XrResult XRAPI_CALL xrDestroyLiftStreamDXR(XrLiftStreamDXR stream); + +//! Hand one frame to the stream's latest-wins mailbox; returns its frameId +//! (monotonic per stream, from 1). Never waits for the conversion. +//! XR_ERROR_RUNTIME_FAILURE (non-fatal, retry next frame) when the input could +//! not be acquired within the service's 4 ms keyed-mutex budget. +XRAPI_ATTR XrResult XRAPI_CALL xrSubmitLiftFrameDXR( + XrLiftStreamDXR stream, const XrLiftFrameSubmitInfoDXR* submitInfo, uint64_t* frameId); + +//! The newest finished texture result newer than the last one acquired, or +//! XR_LIFT_NOT_READY_DXR. XR_ERROR_VALIDATION_FAILURE on a GAUSSIANS stream. +XRAPI_ATTR XrResult XRAPI_CALL xrAcquireLiftResultDXR(XrLiftStreamDXR stream, XrLiftResultDXR* result); + +//! The newest finished blob (GAUSSIANS streams), two-call idiom — see +//! XrLiftBlobDXR. XR_ERROR_VALIDATION_FAILURE on a texture-mode stream. +XRAPI_ATTR XrResult XRAPI_CALL xrAcquireLiftBlobDXR(XrLiftStreamDXR stream, XrLiftBlobDXR* blob); + +//! Set the stream's scheduling priority (see XrLiftPriorityDXR). Cheap; takes +//! effect at the lift thread's next round. Default NORMAL. +XRAPI_ATTR XrResult XRAPI_CALL xrSetLiftStreamPriorityDXR(XrLiftStreamDXR stream, XrLiftPriorityDXR priority); + +//! The stream's counters and effective conversion rate. Cheap (one IPC round +//! trip, no GPU work) — poll it at UI rates, not per frame. +XRAPI_ATTR XrResult XRAPI_CALL xrGetLiftStreamStatsDXR(XrLiftStreamDXR stream, XrLiftStreamStatsDXR* stats); + +#endif /* !XR_NO_PROTOTYPES */ + +#ifdef __cplusplus +} +#endif + +#endif // XR_DXR_LIFT_H diff --git a/src/xrt/auxiliary/util/CMakeLists.txt b/src/xrt/auxiliary/util/CMakeLists.txt index aa1983fda..6fa0f2db1 100644 --- a/src/xrt/auxiliary/util/CMakeLists.txt +++ b/src/xrt/auxiliary/util/CMakeLists.txt @@ -24,6 +24,8 @@ add_library( u_bg_neutrality.c u_bg_neutrality.h u_bitwise.c + u_lift_mailbox.c + u_lift_mailbox.h u_snap_grid.c u_snap_grid.h u_bitwise.h diff --git a/src/xrt/auxiliary/util/u_lift_mailbox.c b/src/xrt/auxiliary/util/u_lift_mailbox.c new file mode 100644 index 000000000..4d62f75ce --- /dev/null +++ b/src/xrt/auxiliary/util/u_lift_mailbox.c @@ -0,0 +1,611 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief XR_DXR_lift per-stream mailbox + output ring state machine. + * @ingroup aux_util + * + * See u_lift_mailbox.h. No locks here: the caller serializes. + */ + +#include "util/u_lift_mailbox.h" + +#include + +static bool +slot_ok_in(int32_t slot) +{ + return slot >= 0 && slot < U_LIFT_INPUT_SLOTS; +} + +static bool +slot_ok_out(int32_t slot) +{ + return slot >= 0 && slot < U_LIFT_RING_SIZE; +} + +void +u_lift_mailbox_init(struct u_lift_mailbox *mb) +{ + memset(mb, 0, sizeof(*mb)); + mb->latest = -1; +} + +bool +u_lift_mailbox_begin_submit(struct u_lift_mailbox *mb, int32_t *out_slot) +{ + int32_t pick = -1; + for (int32_t i = 0; i < U_LIFT_INPUT_SLOTS; i++) { + if (mb->in_state[i] == U_LIFT_IN_FREE) { + pick = i; + break; + } + } + if (pick < 0) { + // No free slot: overwrite the pending one — that frame never reaches + // the module (latest wins). + for (int32_t i = 0; i < U_LIFT_INPUT_SLOTS; i++) { + if (mb->in_state[i] == U_LIFT_IN_PENDING) { + pick = i; + mb->dropped++; + break; + } + } + } + if (pick < 0) { + return false; + } + mb->in_state[pick] = U_LIFT_IN_WRITING; + memset(&mb->in_meta[pick], 0, sizeof(mb->in_meta[pick])); + *out_slot = pick; + return true; +} + +uint64_t +u_lift_mailbox_commit_submit( + struct u_lift_mailbox *mb, int32_t slot, int64_t source_time, uint64_t now_ns, uint32_t width, uint32_t height) +{ + if (!slot_ok_in(slot) || mb->in_state[slot] != U_LIFT_IN_WRITING) { + return 0; + } + // An OLDER pending frame is superseded by this one. + for (int32_t i = 0; i < U_LIFT_INPUT_SLOTS; i++) { + if (i != slot && mb->in_state[i] == U_LIFT_IN_PENDING) { + mb->in_state[i] = U_LIFT_IN_FREE; + mb->dropped++; + } + } + struct u_lift_frame_meta *m = &mb->in_meta[slot]; + m->frame_id = ++mb->last_frame_id; + m->source_time = source_time; + m->submit_ns = now_ns; + m->convert_start_ns = 0; + m->done_ns = 0; + m->width = width; + m->height = height; + mb->in_state[slot] = U_LIFT_IN_PENDING; + mb->submitted++; + return m->frame_id; +} + +void +u_lift_mailbox_abort_submit(struct u_lift_mailbox *mb, int32_t slot) +{ + if (slot_ok_in(slot) && mb->in_state[slot] == U_LIFT_IN_WRITING) { + mb->in_state[slot] = U_LIFT_IN_FREE; + } +} + +bool +u_lift_mailbox_has_pending(const struct u_lift_mailbox *mb) +{ + for (int32_t i = 0; i < U_LIFT_INPUT_SLOTS; i++) { + if (mb->in_state[i] == U_LIFT_IN_PENDING) { + return true; + } + } + return false; +} + +bool +u_lift_mailbox_take_pending(struct u_lift_mailbox *mb, + uint64_t now_ns, + int32_t *out_slot, + struct u_lift_frame_meta *out_meta) +{ + int32_t pick = -1; + for (int32_t i = 0; i < U_LIFT_INPUT_SLOTS; i++) { + if (mb->in_state[i] != U_LIFT_IN_PENDING) { + continue; + } + if (pick < 0 || mb->in_meta[i].frame_id > mb->in_meta[pick].frame_id) { + pick = i; + } + } + if (pick < 0) { + return false; + } + mb->in_state[pick] = U_LIFT_IN_CONVERTING; + mb->in_meta[pick].convert_start_ns = now_ns; + *out_slot = pick; + if (out_meta != NULL) { + *out_meta = mb->in_meta[pick]; + } + return true; +} + +void +u_lift_mailbox_finish_input(struct u_lift_mailbox *mb, int32_t slot) +{ + if (slot_ok_in(slot) && mb->in_state[slot] == U_LIFT_IN_CONVERTING) { + mb->in_state[slot] = U_LIFT_IN_FREE; + } +} + +bool +u_lift_mailbox_begin_output(struct u_lift_mailbox *mb, int32_t *out_slot) +{ + for (int32_t i = 0; i < U_LIFT_RING_SIZE; i++) { + if (i == mb->latest || mb->out_pins[i] > 0 || mb->out_state[i] == U_LIFT_OUT_WRITING) { + continue; + } + mb->out_state[i] = U_LIFT_OUT_WRITING; + *out_slot = i; + return true; + } + return false; +} + +void +u_lift_mailbox_publish_output(struct u_lift_mailbox *mb, + int32_t slot, + const struct u_lift_frame_meta *meta, + uint64_t now_ns) +{ + if (!slot_ok_out(slot) || mb->out_state[slot] != U_LIFT_OUT_WRITING || meta == NULL) { + return; + } + struct u_lift_frame_meta *m = &mb->out_meta[slot]; + *m = *meta; + m->done_ns = now_ns; + mb->out_state[slot] = U_LIFT_OUT_READY; + mb->latest = slot; + mb->converted++; + + if (mb->last_publish_ns != 0 && now_ns > mb->last_publish_ns) { + uint64_t iv = now_ns - mb->last_publish_ns; + if (mb->interval_ema_ns == 0) { + mb->interval_ema_ns = iv; + } else if (iv >= mb->interval_ema_ns) { + mb->interval_ema_ns += (iv - mb->interval_ema_ns) / 8; + } else { + mb->interval_ema_ns -= (mb->interval_ema_ns - iv) / 8; + } + } + mb->last_publish_ns = now_ns; + + uint64_t lat = now_ns >= m->submit_ns ? now_ns - m->submit_ns : 0; + mb->lat_last_ns = lat; + if (mb->converted == 1) { + mb->lat_min_ns = lat; + mb->lat_max_ns = lat; + mb->lat_ema_ns = lat; + } else { + if (lat < mb->lat_min_ns) { + mb->lat_min_ns = lat; + } + if (lat > mb->lat_max_ns) { + mb->lat_max_ns = lat; + } + // alpha = 1/8, integer: ema += (lat - ema) / 8, signed-safe. + if (lat >= mb->lat_ema_ns) { + mb->lat_ema_ns += (lat - mb->lat_ema_ns) / 8; + } else { + mb->lat_ema_ns -= (mb->lat_ema_ns - lat) / 8; + } + } +} + +void +u_lift_mailbox_abort_output(struct u_lift_mailbox *mb, int32_t slot) +{ + if (slot_ok_out(slot) && mb->out_state[slot] == U_LIFT_OUT_WRITING) { + mb->out_state[slot] = U_LIFT_OUT_EMPTY; + memset(&mb->out_meta[slot], 0, sizeof(mb->out_meta[slot])); + mb->failed++; + } +} + +bool +u_lift_mailbox_pin_latest(struct u_lift_mailbox *mb, + bool only_newer, + int32_t *out_slot, + struct u_lift_frame_meta *out_meta) +{ + int32_t s = mb->latest; + if (!slot_ok_out(s) || mb->out_state[s] != U_LIFT_OUT_READY) { + return false; + } + if (only_newer) { + if (mb->out_meta[s].frame_id <= mb->last_acquired_frame_id) { + return false; + } + mb->last_acquired_frame_id = mb->out_meta[s].frame_id; + } + mb->out_pins[s]++; + *out_slot = s; + if (out_meta != NULL) { + *out_meta = mb->out_meta[s]; + } + return true; +} + +void +u_lift_mailbox_unpin(struct u_lift_mailbox *mb, int32_t slot) +{ + if (slot_ok_out(slot) && mb->out_pins[slot] > 0) { + mb->out_pins[slot]--; + } +} + +bool +u_lift_mailbox_has_result(const struct u_lift_mailbox *mb) +{ + return slot_ok_out(mb->latest) && mb->out_state[mb->latest] == U_LIFT_OUT_READY; +} + +float +u_lift_mailbox_rate_hz(const struct u_lift_mailbox *mb) +{ + return mb->interval_ema_ns > 0 ? (float)(1e9 / (double)mb->interval_ema_ns) : 0.0f; +} + +void +u_lift_sched_init(struct u_lift_sched *s) +{ + memset(s, 0, sizeof(*s)); +} + +uint32_t +u_lift_sched_plan( + struct u_lift_sched *s, const struct u_lift_sched_entry *entries, uint32_t count, uint64_t *out_ids, uint32_t max) +{ + uint32_t n = 0; + bool any = false; + for (uint32_t i = 0; i < count; i++) { + if (entries[i].pending && entries[i].priority != U_LIFT_PRIORITY_PAUSED) { + any = true; + } + } + if (!any) { + return 0; + } + const bool low_round = (s->round % U_LIFT_LOW_EVERY_N) == 0; + s->round++; + + // Every HIGH stream with a new frame. + for (uint32_t i = 0; i < count && n < max; i++) { + if (entries[i].pending && entries[i].priority >= U_LIFT_PRIORITY_HIGH) { + out_ids[n++] = entries[i].id; + } + } + + // ONE NORMAL stream, round-robin: the first after the last served, else wrap. + int32_t pick = -1; + for (uint32_t i = 0; i < count; i++) { + if (entries[i].pending && entries[i].priority == U_LIFT_PRIORITY_NORMAL && + entries[i].id > s->normal_rr_last) { + pick = (int32_t)i; + break; + } + } + if (pick < 0) { + for (uint32_t i = 0; i < count; i++) { + if (entries[i].pending && entries[i].priority == U_LIFT_PRIORITY_NORMAL) { + pick = (int32_t)i; + break; + } + } + } + if (pick >= 0 && n < max) { + out_ids[n++] = entries[pick].id; + s->normal_rr_last = entries[pick].id; + } + + // LOW: every stream with a new frame, every Nth round. + if (low_round) { + for (uint32_t i = 0; i < count && n < max; i++) { + if (entries[i].pending && entries[i].priority == U_LIFT_PRIORITY_LOW) { + out_ids[n++] = entries[i].id; + } + } + } + return n; +} + + +/* + * + * Snapshot size cap. + * + */ + +uint32_t +u_lift_max_input_edge_parse(const char *value) +{ + if (value == NULL || value[0] == '\0') { + return U_LIFT_MAX_INPUT_EDGE_DEFAULT; + } + uint64_t v = 0; + for (const char *p = value; *p != '\0'; p++) { + if (*p < '0' || *p > '9') { + return U_LIFT_MAX_INPUT_EDGE_DEFAULT; + } + v = v * 10u + (uint64_t)(*p - '0'); + if (v > 0xffffu) { + v = 0xffffu; // far beyond any D3D11 texture edge; saturate + } + } + if (v == 0) { + return 0; + } + return v < U_LIFT_MAX_INPUT_EDGE_MIN ? U_LIFT_MAX_INPUT_EDGE_MIN : (uint32_t)v; +} + +bool +u_lift_cap_dims(uint32_t w, uint32_t h, uint32_t cap, uint32_t *out_w, uint32_t *out_h) +{ + *out_w = w; + *out_h = h; + const uint32_t long_edge = w > h ? w : h; + if (cap == 0 || long_edge <= cap || w == 0 || h == 0) { + return false; + } + const uint64_t ce = (uint64_t)(cap & ~1u) < 2u ? 2u : (uint64_t)(cap & ~1u); + const uint64_t short_edge = w > h ? h : w; + // Nearest even: 2 * round(short * ce / (2 * long)). + uint64_t se = 2u * ((short_edge * ce + long_edge) / (2u * (uint64_t)long_edge)); + se = se < 2u ? 2u : se; + se = se > ce ? ce : se; + if (w >= h) { + *out_w = (uint32_t)ce; + *out_h = (uint32_t)se; + } else { + *out_w = (uint32_t)se; + *out_h = (uint32_t)ce; + } + return true; +} + + +/* + * + * Letterbox crop. + * + */ + +bool +u_lift_letterbox_parse(const char *value) +{ + return !(value != NULL && value[0] == '0' && value[1] == '\0'); +} + +bool +u_lift_crop_active(const struct u_lift_crop *c) +{ + return c->top != 0 || c->bottom != 0 || c->left != 0 || c->right != 0; +} + + +/*! + * Bar sizes (pixels) at both ends of one axis of length @p len profiled in + * @p n buckets. A bar is the run from an edge of buckets below @p thr. + * + * Symmetry: a film is centred, so when one bar measures shorter, the shorter + * one exists, and the extra band on its side is either separated from the + * picture by a black gap (a subtitle line, however dense) or only sparsely lit + * throughout, the shorter bar is really as long as the other. + */ +static void +letterbox_axis(const float *prof, uint32_t n, uint32_t len, float thr, uint32_t *out_lo, uint32_t *out_hi) +{ + *out_lo = 0; + *out_hi = 0; + if (prof == NULL || n == 0 || len == 0) { + return; + } + uint32_t lo = 0; + while (lo < n && prof[lo] < thr) { + lo++; + } + if (lo == n) { + return; // no picture anywhere on this axis: not a measurement of bars + } + uint32_t hi = 0; + while (hi < n && prof[n - 1 - hi] < thr) { + hi++; + } + // idx(i) = the i-th bucket counted from the SHORT bar's edge. + for (int side = 0; side < 2; side++) { + uint32_t *shrt = side == 0 ? &hi : &lo; + const uint32_t lng = side == 0 ? lo : hi; + if (!(lng > *shrt && *shrt > 0 && n - lng > *shrt)) { + continue; + } +#define LB_IDX(i) (side == 0 ? n - 1 - (i) : (i)) + // The band is [*shrt, lng) from the short edge; lng - 1 touches the picture. + bool gap = prof[LB_IDX(lng - 1)] < U_LIFT_LETTERBOX_PICTURE_FRAC; + bool sparse = true; + for (uint32_t i = *shrt; i < lng && sparse; i++) { + sparse = prof[LB_IDX(i)] < U_LIFT_LETTERBOX_SPARSE_FRAC; + } +#undef LB_IDX + if (gap || sparse) { + *shrt = lng; + } + } + // Bucket k starts at pixel k*len/n: the bars end where the first picture + // bucket starts, so no picture row is ever inside a bar. + uint32_t lo_px = (uint32_t)(((uint64_t)lo * len) / n); + uint32_t hi_px = len - (uint32_t)(((uint64_t)(n - hi) * len) / n); + if (lo_px < U_LIFT_LETTERBOX_MIN_BAR_PX) { + lo_px = 0; + } + if (hi_px < U_LIFT_LETTERBOX_MIN_BAR_PX) { + hi_px = 0; + } + if ((float)(len - lo_px - hi_px) < U_LIFT_LETTERBOX_MIN_ACTIVE_FRAC * (float)len) { + return; // implausible (a mostly-dark frame): no crop from this frame + } + *out_lo = lo_px; + *out_hi = hi_px; +} + +//! Each bar of @p a is no larger than the matching bar of @p b. +static bool +crop_within(const struct u_lift_crop *a, const struct u_lift_crop *b) +{ + return a->top <= b->top && a->bottom <= b->bottom && a->left <= b->left && a->right <= b->right; +} + +static uint32_t +min_u32(uint32_t a, uint32_t b) +{ + return a < b ? a : b; +} + +static bool +near_u32(uint32_t a, uint32_t b, uint32_t tol) +{ + return a > b ? a - b <= tol : b - a <= tol; +} + +static bool +crop_near(const struct u_lift_crop *a, const struct u_lift_crop *b, uint32_t tol_v, uint32_t tol_h) +{ + return near_u32(a->top, b->top, tol_v) && near_u32(a->bottom, b->bottom, tol_v) && + near_u32(a->left, b->left, tol_h) && near_u32(a->right, b->right, tol_h); +} + +static void +crop_min_into(struct u_lift_crop *acc, const struct u_lift_crop *m) +{ + acc->top = min_u32(acc->top, m->top); + acc->bottom = min_u32(acc->bottom, m->bottom); + acc->left = min_u32(acc->left, m->left); + acc->right = min_u32(acc->right, m->right); +} + +bool +u_lift_letterbox_update(struct u_lift_letterbox *lb, + uint32_t w, + uint32_t h, + const float *rows, + uint32_t nr, + const float *cols, + uint32_t nc) +{ + const struct u_lift_crop none = {0, 0, 0, 0}; + bool changed = false; + if (lb->w != w || lb->h != h) { + changed = u_lift_crop_active(&lb->committed); + lb->w = w; + lb->h = h; + lb->committed = none; + lb->pending = none; + lb->pending_frames = 0; + lb->shrink = none; + lb->shrink_frames = 0; + } + + // A frame with no picture on either axis (black, a fade) says nothing. + bool content = false; + for (uint32_t i = 0; i < nr && !content; i++) { + content = rows[i] >= U_LIFT_LETTERBOX_PICTURE_FRAC; + } + if (!content) { + return changed; + } + + // Two readings of the same profile: bars as far as the rows are truly + // BLACK (what the crop may grow into), and as far as no PICTURE intrudes + // (what the crop must shrink back to). + struct u_lift_crop grow = none, keep = none; + letterbox_axis(rows, nr, h, U_LIFT_LETTERBOX_BLACK_FRAC, &grow.top, &grow.bottom); + letterbox_axis(cols, nc, w, U_LIFT_LETTERBOX_BLACK_FRAC, &grow.left, &grow.right); + letterbox_axis(rows, nr, h, U_LIFT_LETTERBOX_PICTURE_FRAC, &keep.top, &keep.bottom); + letterbox_axis(cols, nc, w, U_LIFT_LETTERBOX_PICTURE_FRAC, &keep.left, &keep.right); + + // Profiles jitter by a bucket frame to frame: "the same" within a tolerance, + // settling on the SMALLEST bars seen (never crop picture). + const uint32_t tol_v = h / 64u > 8u ? h / 64u : 8u; + const uint32_t tol_h = w / 64u > 8u ? w / 64u : 8u; + + // An intrusion within the jitter tolerance is bucket rounding (the symmetry + // rule mirrors a bar's BUCKET count, which lands a few pixels off the other + // bar's pixel size), not picture: it does not shrink the crop. +#define LB_RELAX(e, t) \ + if (lb->committed.e > keep.e && lb->committed.e - keep.e <= (t)) { \ + keep.e = lb->committed.e; \ + } + LB_RELAX(top, tol_v) + LB_RELAX(bottom, tol_v) + LB_RELAX(left, tol_h) + LB_RELAX(right, tol_h) +#undef LB_RELAX + + // Shrink: picture inside a committed bar, held for SHRINK_FRAMES frames. + if (!crop_within(&lb->committed, &keep)) { + struct u_lift_crop target = lb->committed; + crop_min_into(&target, &keep); + if (lb->shrink_frames > 0 && crop_near(&target, &lb->shrink, tol_v, tol_h)) { + crop_min_into(&lb->shrink, &target); + lb->shrink_frames++; + } else { + lb->shrink = target; + lb->shrink_frames = 1; + } + if (lb->shrink_frames >= U_LIFT_LETTERBOX_SHRINK_FRAMES) { + lb->committed = lb->shrink; + lb->shrink_frames = 0; + lb->pending = lb->committed; + lb->pending_frames = 0; + return true; + } + return changed; // no growth while picture is intruding + } + lb->shrink_frames = 0; + + // Grow: larger truly-black bars have to settle first. + if (crop_within(&grow, &lb->committed)) { + lb->pending = lb->committed; + lb->pending_frames = 0; + return changed; + } + if (lb->pending_frames > 0 && crop_near(&grow, &lb->pending, tol_v, tol_h)) { + crop_min_into(&lb->pending, &grow); + lb->pending_frames++; + } else { + lb->pending = grow; + lb->pending_frames = 1; + } + if (lb->pending_frames >= U_LIFT_LETTERBOX_SETTLE_FRAMES) { + // Never grow a bar beyond what the no-picture reading allows either. + struct u_lift_crop next = lb->pending; + crop_min_into(&next, &keep); + // Deadband: once a crop is in effect, growing it by a few pixels buys + // nothing (the recompose's flat bars already reach into the active + // area) and would re-size the module's input every settle period. + const uint32_t db_v = h / 256u > 8u ? h / 256u : 8u; + const uint32_t db_h = w / 256u > 8u ? w / 256u : 8u; + const struct u_lift_crop *c = &lb->committed; + const bool worth = !u_lift_crop_active(c) || next.top > c->top + db_v || next.bottom > c->bottom + db_v || + next.left > c->left + db_h || next.right > c->right + db_h; + if (worth && memcmp(&next, &lb->committed, sizeof(next)) != 0) { + lb->committed = next; + changed = true; + } + lb->pending_frames = 0; + } + return changed; +} diff --git a/src/xrt/auxiliary/util/u_lift_mailbox.h b/src/xrt/auxiliary/util/u_lift_mailbox.h new file mode 100644 index 000000000..805fe3314 --- /dev/null +++ b/src/xrt/auxiliary/util/u_lift_mailbox.h @@ -0,0 +1,409 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief XR_DXR_lift (ADR-042): the per-stream bookkeeping between the + * submitting thread, the conversion thread and the consumers. + * + * A lift stream moves frames through three hands: + * + * - PRODUCER — an IPC thread (xrSubmitLiftFrameDXR, or a lift-flagged weave + * rect inside xrWeaveSubmitDXR). It snapshots the caller's pixels into an + * INPUT SLOT and must never wait for the model. + * - WORKER — the one lift thread. It takes the newest pending input, runs the + * vendor module (synchronous, tens of ms or seconds), and copies the module's + * output — valid only until the module's next call — into an OUTPUT SLOT. + * - CONSUMERS — the weave (every frame, "latest result, whatever it is") and + * xrAcquireLiftResultDXR ("latest result newer than the one I already have"). + * A consumer PINS the slot it reads so the worker can never overwrite it + * mid-copy. + * + * Two input slots make the mailbox LATEST-WINS without ever blocking the + * producer: at most one slot is CONVERTING, so the other one is always + * writable; a pending frame the producer overwrites (or that a newer commit + * supersedes) is DROPPED and counted — never queued. Two output slots (the + * ring) let the worker write the next result while a consumer reads the latest; + * the worker waits only while a stale slot is still pinned, which lasts one + * GPU-copy issue. + * + * This file is the pure state machine — no locks, no GPU, no clock. The caller + * serializes every call under its own stream mutex and passes timestamps in, so + * the whole contract (slot choice, drop accounting, frame ids, newer-than + * acquire, pin exclusion, latency stats) is pinned host-side by + * tests/tests_lift_mailbox.cpp. The D3D11 service (d3d11_lift.cpp) is the one + * user today. + * + * @ingroup aux_util + */ + +#pragma once + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +//! Input slots per stream. Two is the minimum for a never-blocking producer. +#define U_LIFT_INPUT_SLOTS 2 +//! Output ring slots per stream (the module's output is only valid until its +//! next call, so every result is copied into one of these). +#define U_LIFT_RING_SIZE 2 + +enum u_lift_in_state +{ + U_LIFT_IN_FREE = 0, + U_LIFT_IN_WRITING, //!< producer is snapshotting into it + U_LIFT_IN_PENDING, //!< complete, waiting for the worker + U_LIFT_IN_CONVERTING, //!< the worker owns it +}; + +enum u_lift_out_state +{ + U_LIFT_OUT_EMPTY = 0, + U_LIFT_OUT_WRITING, //!< the worker is copying a result into it + U_LIFT_OUT_READY, //!< holds a complete result +}; + +//! Everything the runtime knows about one frame's trip through a stream. +struct u_lift_frame_meta +{ + uint64_t frame_id; //!< per stream, monotonic from 1 (0 = none) + int64_t source_time; //!< caller's timestamp, echoed verbatim + uint64_t submit_ns; //!< producer commit time (runtime clock) + uint64_t convert_start_ns; //!< worker took it + uint64_t done_ns; //!< result published + uint32_t width, height; //!< input extent +}; + +struct u_lift_mailbox +{ + enum u_lift_in_state in_state[U_LIFT_INPUT_SLOTS]; + struct u_lift_frame_meta in_meta[U_LIFT_INPUT_SLOTS]; + + enum u_lift_out_state out_state[U_LIFT_RING_SIZE]; + struct u_lift_frame_meta out_meta[U_LIFT_RING_SIZE]; + uint32_t out_pins[U_LIFT_RING_SIZE]; + //! Ring slot holding the newest result, -1 = none yet. + int32_t latest; + + //! Last frame id handed out; the next commit gets this + 1. + uint64_t last_frame_id; + //! Newest frame id a "newer-than" acquire has returned. + uint64_t last_acquired_frame_id; + + //! Counters (monotonic). + uint64_t submitted; //!< committed frames + uint64_t dropped; //!< pending frames superseded before the worker took them + uint64_t converted; //!< results published + uint64_t failed; //!< conversions the worker abandoned + + //! submit → published latency, ns. + uint64_t lat_last_ns; + uint64_t lat_min_ns; + uint64_t lat_max_ns; + uint64_t lat_ema_ns; //!< exponential moving average, alpha = 1/8 + + //! publish → publish interval, for the effective conversion rate. + uint64_t last_publish_ns; + uint64_t interval_ema_ns; //!< alpha = 1/8; 0 until two results exist +}; + +//! Reset @p mb to "no frames, no results". +void +u_lift_mailbox_init(struct u_lift_mailbox *mb); + +/* + * Producer. + */ + +/*! + * Pick an input slot to snapshot into and mark it WRITING. Prefers a FREE slot; + * otherwise overwrites the PENDING one (that frame is dropped and counted). + * Fails only if no slot is FREE or PENDING — impossible with one producer per + * stream, reported rather than asserted because the wire is untrusted. + */ +bool +u_lift_mailbox_begin_submit(struct u_lift_mailbox *mb, int32_t *out_slot); + +/*! + * The snapshot into @p slot is complete: assign the next frame id, mark it + * PENDING, and drop any OLDER pending frame (latest wins). Returns the frame id. + */ +uint64_t +u_lift_mailbox_commit_submit( + struct u_lift_mailbox *mb, int32_t slot, int64_t source_time, uint64_t now_ns, uint32_t width, uint32_t height); + +//! The snapshot into @p slot failed: return it to FREE (no frame id consumed). +void +u_lift_mailbox_abort_submit(struct u_lift_mailbox *mb, int32_t slot); + +/* + * Worker. + */ + +//! True when an input is waiting for the worker. +bool +u_lift_mailbox_has_pending(const struct u_lift_mailbox *mb); + +/*! + * Take the newest PENDING input (-> CONVERTING), stamping @p now_ns as its + * conversion start. False when nothing is pending. + */ +bool +u_lift_mailbox_take_pending(struct u_lift_mailbox *mb, + uint64_t now_ns, + int32_t *out_slot, + struct u_lift_frame_meta *out_meta); + +//! The worker is done reading input @p slot (-> FREE). +void +u_lift_mailbox_finish_input(struct u_lift_mailbox *mb, int32_t slot); + +/*! + * Pick an output slot to write: never the latest result, never a pinned slot, + * never one already being written. False = wait (a consumer still holds the + * only candidate). + */ +bool +u_lift_mailbox_begin_output(struct u_lift_mailbox *mb, int32_t *out_slot); + +//! Result copied into @p slot: it becomes the latest; latency stats update. +void +u_lift_mailbox_publish_output(struct u_lift_mailbox *mb, + int32_t slot, + const struct u_lift_frame_meta *meta, + uint64_t now_ns); + +//! The worker gave up on @p slot's result (-> EMPTY; failed++). +void +u_lift_mailbox_abort_output(struct u_lift_mailbox *mb, int32_t slot); + +/* + * Consumers. + */ + +/*! + * Pin the latest result for reading. With @p only_newer, succeed only when it + * is newer than the last result a newer-than pin returned (and record it as + * returned); without, always return the latest (the weave's case). False when + * no result qualifies. Every successful pin must be matched by one + * @ref u_lift_mailbox_unpin. + */ +bool +u_lift_mailbox_pin_latest(struct u_lift_mailbox *mb, + bool only_newer, + int32_t *out_slot, + struct u_lift_frame_meta *out_meta); + +void +u_lift_mailbox_unpin(struct u_lift_mailbox *mb, int32_t slot); + +//! True once any result has been published (and not since aborted). +bool +u_lift_mailbox_has_result(const struct u_lift_mailbox *mb); + +//! Effective conversion rate in results per second (0 until two results). +float +u_lift_mailbox_rate_hz(const struct u_lift_mailbox *mb); + + +/* + * + * Cross-stream scheduling (XrLiftPriorityDXR). + * + * The module converts one frame at a time, so concurrent streams share it. The + * lift thread asks for a PLAN each round and converts the planned streams in + * order: + * + * - every HIGH stream with a pending frame; + * - ONE NORMAL stream with a pending frame, round-robin; + * - every LOW stream with a pending frame, only on every + * U_LIFT_LOW_EVERY_N-th round; + * - PAUSED streams never. + * + * A round is counted only when something (non-paused) was pending, so an idle + * service does not "use up" LOW rounds. + * + */ + +enum u_lift_priority +{ + U_LIFT_PRIORITY_PAUSED = 0, + U_LIFT_PRIORITY_LOW = 1, + U_LIFT_PRIORITY_NORMAL = 2, + U_LIFT_PRIORITY_HIGH = 3, +}; + +//! LOW streams convert on every Nth round. +#define U_LIFT_LOW_EVERY_N 4 + +struct u_lift_sched +{ + uint64_t round; + uint64_t normal_rr_last; //!< id of the NORMAL stream served last +}; + +//! One stream as the scheduler sees it. Entries must be in ascending id order. +struct u_lift_sched_entry +{ + uint64_t id; + uint32_t priority; //!< enum u_lift_priority + bool pending; +}; + +void +u_lift_sched_init(struct u_lift_sched *s); + +/*! + * Plan one round: write up to @p max stream ids to convert, in order, to + * @p out_ids and return how many. May return 0 with a pending LOW stream (not + * its round) — call again. + */ +uint32_t +u_lift_sched_plan( + struct u_lift_sched *s, const struct u_lift_sched_entry *entries, uint32_t count, uint64_t *out_ids, uint32_t max); + + +/* + * + * Snapshot size cap (service policy, ADR-042). + * + * A lift-flagged weave rect is snapshotted at DEVICE pixels, and the module + * synthesizes its views at INPUT resolution — a fullscreen player on an 8K + * panel is a 7680x4319 input and a 15360-wide SBS per frame. The result is + * stretched back into the rect's current position anyway, so the service + * downsamples the snapshot before the DP. An app's explicit + * xrSubmitLiftFrameDXR frame is NOT capped: its size is the app's choice. + * + */ + +//! Default cap on the snapshot's long edge (DXR_LIFT_MAX_INPUT_EDGE unset). +#define U_LIFT_MAX_INPUT_EDGE_DEFAULT 1920u + +//! Smallest cap honoured; lower non-zero values clamp to it. +#define U_LIFT_MAX_INPUT_EDGE_MIN 256u + +/*! + * Parse a DXR_LIFT_MAX_INPUT_EDGE value. NULL, empty or non-numeric = the + * default; "0" = no cap (returns 0); 1..255 clamp to + * U_LIFT_MAX_INPUT_EDGE_MIN. + */ +uint32_t +u_lift_max_input_edge_parse(const char *value); + +/*! + * Scale @p w x @p h so the long edge is at most @p cap, keeping the aspect. + * The long edge becomes @p cap rounded DOWN to even; the short edge is scaled + * by the same factor and rounded to the NEAREST even value, minimum 2. When + * @p cap is 0 or the long edge already fits, the dims pass through unchanged + * and the function returns false; true = the dims were reduced. + */ +bool +u_lift_cap_dims(uint32_t w, uint32_t h, uint32_t cap, uint32_t *out_w, uint32_t *out_h); + + +/* + * + * Letterbox crop (service policy, ADR-042). + * + * A lift-flagged weave rect often holds a film with black bars (2.39:1 in a + * 16:9 player), sometimes with subtitles drawn in the bars. Converting the bars + * wastes module time, and the hard black edge confuses depth. The service + * measures, per row / column of the rect, the fraction of non-black pixels + * (GPU reduction, read back asynchronously), and this helper turns those + * profiles into a stable crop: the lifted input is the ACTIVE area only, and + * the bars (subtitles included) are woven flat, identical in both eyes. + * + * A bar is the run of rows (columns) from an edge whose non-black fraction + * stays below U_LIFT_LETTERBOX_PICTURE_FRAC — low enough that subtitle text in + * a bar does not end it, high enough that picture rows do. Bars GROW only after + * the same measurement has held for U_LIFT_LETTERBOX_SETTLE_FRAMES frames that + * carry content (a fade or cut to black is not a letterbox), and SHRINK at once + * when picture appears in them. A wrong crop in a dark scene costs little: the + * cropped rows are dark and are simply woven flat. + * + */ + +//! Row / column profile length cap (buckets per axis). One bucket per row / +//! column up to 8K, so a bar edge is exact to the pixel and the lifted area is +//! exactly the picture (no flat overlap needed to hide a bucket's slack). +#define U_LIFT_LETTERBOX_BINS_MAX 8192u + +//! A bucket whose non-black fraction reaches this is picture intruding into a +//! bar (the SHRINK test). Subtitle text stays below it or is caught by the +//! symmetry rule. +#define U_LIFT_LETTERBOX_PICTURE_FRAC 0.25f + +//! A bar only GROWS into buckets at most this lit (truly black): a dark scene +//! edge is rarely this empty, so dark stretches do not crop picture. +#define U_LIFT_LETTERBOX_BLACK_FRAC 0.03f + +//! Consecutive frames picture must intrude into a bar before it shrinks — one +//! caption frame must not un-crop; a 16:9 ad still un-crops within ~0.1 s. +#define U_LIFT_LETTERBOX_SHRINK_FRAMES 6u + +/*! + * Rows below this non-black fraction count as sparse (subtitle text, not + * picture) when a bar is extended to match the opposite one — see + * u_lift_letterbox_update's symmetry rule. Dense subtitles can pass + * U_LIFT_LETTERBOX_PICTURE_FRAC; a picture edge row stays above this. + */ +#define U_LIFT_LETTERBOX_SPARSE_FRAC 0.6f + +//! Frames a larger bar must hold before the crop grows into it. +#define U_LIFT_LETTERBOX_SETTLE_FRAMES 45u + +//! Bars below this many pixels are ignored (no crop on that edge). +#define U_LIFT_LETTERBOX_MIN_BAR_PX 4u + +//! The active area must keep at least this fraction of each dimension. +#define U_LIFT_LETTERBOX_MIN_ACTIVE_FRAC 0.3f + +//! Crop of a @c w x @c h rect: bar sizes in rect pixels (0 = no bar). +struct u_lift_crop +{ + uint32_t top, bottom, left, right; +}; + +struct u_lift_letterbox +{ + uint32_t w, h; //!< rect dims the state belongs to (0 = none yet) + struct u_lift_crop committed; //!< the crop in effect + struct u_lift_crop pending; //!< a larger crop waiting to settle + uint32_t pending_frames; //!< consecutive content frames @c pending held + struct u_lift_crop shrink; //!< a smaller crop (picture in a bar) waiting to hold + uint32_t shrink_frames; //!< consecutive content frames @c shrink held +}; + +/*! + * Parse DXR_LIFT_LETTERBOX: NULL / empty / anything but "0" = enabled. + */ +bool +u_lift_letterbox_parse(const char *value); + +/*! + * Feed one measurement of a @p w x @p h rect. @p rows holds @p nr per-bucket + * non-black fractions top to bottom (bucket i covers rows [i*h/nr, (i+1)*h/nr)), + * @p cols @p nc buckets left to right; either may be NULL/0 (that axis then + * never crops). A change of @p w / @p h resets the state (no crop until the new + * bars settle). Returns true when the committed crop changed. + */ +bool +u_lift_letterbox_update(struct u_lift_letterbox *lb, + uint32_t w, + uint32_t h, + const float *rows, + uint32_t nr, + const float *cols, + uint32_t nc); + +//! True when @p c crops anything. +bool +u_lift_crop_active(const struct u_lift_crop *c); + + +#ifdef __cplusplus +} +#endif diff --git a/src/xrt/compositor/d3d11_service/CMakeLists.txt b/src/xrt/compositor/d3d11_service/CMakeLists.txt index 9f0073b00..76b216bfe 100644 --- a/src/xrt/compositor/d3d11_service/CMakeLists.txt +++ b/src/xrt/compositor/d3d11_service/CMakeLists.txt @@ -33,6 +33,8 @@ add_library( d3d11_service_shaders.h d3d11_capture.cpp d3d11_capture.h + d3d11_lift.cpp + d3d11_lift.h d3d11_icon_loader.cpp d3d11_icon_loader.h displayxr_logo_data.h diff --git a/src/xrt/compositor/d3d11_service/comp_d3d11_service.cpp b/src/xrt/compositor/d3d11_service/comp_d3d11_service.cpp index cf7a14e33..d45bbc4f5 100644 --- a/src/xrt/compositor/d3d11_service/comp_d3d11_service.cpp +++ b/src/xrt/compositor/d3d11_service/comp_d3d11_service.cpp @@ -58,6 +58,7 @@ // compositor (PR 1/6 made it a standalone static lib for exactly this). #include "comp_xbridge.h" #include "comp_split_gate.h" +#include "d3d11_lift.h" // XR_DXR_lift (ADR-042) #include "util/u_hud.h" #include "util/u_tiling.h" @@ -777,6 +778,16 @@ struct d3d11_client_render_resources //! One-shot: this client has been reported as using the legacy path (#1058). bool weave_legacy_warned; + //! XR_DXR_lift (ADR-042): the lift-flagged rects of this client's NEXT + //! weave submit (comp_d3d11_service_lift_set_weave_rects), consumed and + //! cleared by that submit. Same IPC thread as the submit — no lock. + uint32_t lift_rect_count; + uint64_t lift_owner; + struct xrt_lift_weave_rect lift_rects[XRT_LIFT_WEAVE_RECTS_MAX]; + //! v6 path: RTV on weave_crop_tex, so lifted views can be written into + //! the crop before the weave (a lift frame never takes the zero-copy path). + wil::com_ptr weave_crop_rtv; + //! Cached import of the caller's v4 overlay atlas (browser#18), keyed by the //! shared-handle value exactly like weave_input_* above. A window-sized //! premultiplied-alpha RGBA atlas the runtime composites OVER the woven @@ -1704,6 +1715,12 @@ struct d3d11_service_system //! render_mutex / c->mutex / atlas_submit_mutex → this. std::mutex immediate_ctx_mutex; + //! XR_DXR_lift (ADR-042): the lift module — its own thread, device and + //! display processor (d3d11_lift.h). Created lazily on the first lift call + //! under @ref lift_create_mutex; destroyed first in system_destroy. + struct d3d11_lift *lift = nullptr; + std::mutex lift_create_mutex; + //! Count of threads currently blocked acquiring render_mutex through //! render_mutex_fair_lock (every acquirer EXCEPT the capture render //! thread). std::recursive_mutex has no waiter fairness, and with SR v2 @@ -6789,6 +6806,8 @@ fini_client_render_resources(struct d3d11_client_render_resources *res) res->weave_saw_frame_first = false; res->weave_gen_count = 0; res->weave_legacy_warned = false; + res->lift_rect_count = 0; + res->weave_crop_rtv.reset(); res->weave_overlay_handle_cached = nullptr; res->weave_overlay_km.reset(); res->weave_overlay_srv.reset(); @@ -24334,6 +24353,219 @@ comp_d3d11_service_compositor_export_transparent_output_fence(struct xrt_composi * output sub-rect. The DP does all weaving (ADR-007 / ADR-019). */ +/* + * + * XR_DXR_lift (ADR-042) — lift-flagged weave rects. + * + * The weave never waits for the module: each rect's 2D content is SNAPSHOTTED + * into its stream's latest-wins mailbox (one blit on this context), and the + * stream's LATEST result — one frame (or more) behind — is woven at the rect's + * CURRENT position. Geometry is therefore exact and real-time; only the depth + * lags. Caller holds render_mutex + immediate_ctx_mutex and the input's keyed + * mutex (the same state the surrounding blits run under). + * + */ + +static const struct xrt_lift_weave_rect * +lift_binding_for_rect(const struct xrt_lift_weave_rect *rects, uint32_t count, uint32_t rect_index) +{ + for (uint32_t k = 0; k < count; k++) { + if (rects[k].rect_index == rect_index) { + return &rects[k]; + } + } + return nullptr; +} + +//! The stereo pair out of an N-view result: the middle two views. +static void +lift_pick_pair(uint32_t views, uint32_t *out_l, uint32_t *out_r) +{ + if (views <= 2) { + *out_l = 0; + *out_r = views == 2 ? 1 : 0; + return; + } + *out_l = views / 2 - 1; + *out_r = views / 2; +} + +/*! + * Batch (v3) layout: rect @p rect is window-sized-input pixels holding a 2D + * frame. Snapshot it, then blit the stream's latest pair into the left / right + * tiles of the SBS scratch at the rect's position (or the 2D frame into both, + * FLAT, until the first result). Returns false when @p rect_index is not lifted + * (the caller then blits it as ordinary squeezed SBS). + */ +static bool +lift_weave_rect_batch(struct d3d11_service_system *sys, + struct d3d11_service_compositor *c, + struct d3d11_lift *lift, + uint64_t owner, + const struct xrt_lift_weave_rect *bindings, + uint32_t binding_count, + uint32_t rect_index, + ID3D11ShaderResourceView *in_srv, + uint32_t in_w, + uint32_t in_h, + const struct xrt_rect &rect, + uint32_t win_w, + uint32_t win_h) +{ + const struct xrt_lift_weave_rect *b = lift_binding_for_rect(bindings, binding_count, rect_index); + if (b == nullptr) { + return false; + } + // Clip to the input (xrt_offset fields are named w/h). + int32_t x0 = rect.offset.w < 0 ? 0 : rect.offset.w; + int32_t y0 = rect.offset.h < 0 ? 0 : rect.offset.h; + int32_t x1 = rect.offset.w + rect.extent.w; + int32_t y1 = rect.offset.h + rect.extent.h; + x1 = x1 > (int32_t)in_w ? (int32_t)in_w : x1; + y1 = y1 > (int32_t)in_h ? (int32_t)in_h : y1; + if (x1 <= x0 || y1 <= y0) { + return true; // nothing visible to lift; nothing to weave either + } + uint64_t frame_id = 0; + (void)d3d11_lift_submit_srv_locked(lift, owner, b->stream_id, in_srv, in_w, in_h, (uint32_t)x0, (uint32_t)y0, + (uint32_t)(x1 - x0), (uint32_t)(y1 - y0), (int64_t)os_monotonic_get_ns(), + b->has_params ? &b->params : nullptr, &frame_id); + + const float rx = (float)rect.offset.w, ry = (float)rect.offset.h; + const float rw = (float)rect.extent.w, rh = (float)rect.extent.h; + ID3D11RenderTargetView *rtv = c->render.weave_sbs_rtv.get(); + const float atw = (float)(win_w * 2), ath = (float)win_h; + + struct d3d11_lift_pin pin = {}; + if (d3d11_lift_pin_latest(lift, owner, b->stream_id, &pin) && pin.view_count >= 1) { + uint32_t vl = 0, vr = 0; + lift_pick_pair(pin.view_count, &vl, &vr); + const float vw = (float)pin.width / (float)pin.view_count; + // Letterbox: the result covers only the ACTIVE part of the rect; the + // bars (subtitles in them included) are the 2D input, FLAT — the same + // pixels at the same place in both views, i.e. on the screen plane. + const float ax = rx + pin.active[0] * rw, ay = ry + pin.active[1] * rh; + const float aw = (pin.active[2] - pin.active[0]) * rw, ah = (pin.active[3] - pin.active[1]) * rh; + blit_to_atlas_texture(sys, &c->render, pin.srv, vw * (float)vl, 0.0f, vw, (float)pin.height, + (float)pin.width, (float)pin.height, ax, ay, aw, ah, /*is_srgb*/ false, + /*blend*/ nullptr, rtv, atw, ath); + blit_to_atlas_texture(sys, &c->render, pin.srv, vw * (float)vr, 0.0f, vw, (float)pin.height, + (float)pin.width, (float)pin.height, (float)win_w + ax, ay, aw, ah, + /*is_srgb*/ false, /*blend*/ nullptr, rtv, atw, ath); + // The bars: the 2D input, FLAT, tiling the rect exactly around the active + // area — no overlap, so the woven picture meets them pixel-exact (the + // profile is per row, so the active area holds no bar rows to hide). + const float bar[4][4] = { + {rx, ry, rw, ay - ry}, // top + {rx, ay + ah, rw, (ry + rh) - (ay + ah)}, // bottom + {rx, ay, ax - rx, ah}, // left + {ax + aw, ay, (rx + rw) - (ax + aw), ah}, // right + }; + for (int k = 0; k < 4; k++) { + const float *s = bar[k]; + if (s[2] < 0.5f || s[3] < 0.5f) { + continue; // no bar on this edge + } + for (int eye = 0; eye < 2; eye++) { + blit_to_atlas_texture(sys, &c->render, in_srv, s[0], s[1], s[2], s[3], (float)in_w, (float)in_h, + (float)(eye * win_w) + s[0], s[1], s[2], s[3], /*is_srgb*/ false, + /*blend*/ nullptr, rtv, atw, ath); + } + } + d3d11_lift_unpin(&pin); + } else { + // No result yet: weave it FLAT — the same 2D frame in both views. + blit_to_atlas_texture(sys, &c->render, in_srv, rx, ry, rw, rh, (float)in_w, (float)in_h, rx, ry, rw, rh, + /*is_srgb*/ false, /*blend*/ nullptr, rtv, atw, ath); + blit_to_atlas_texture(sys, &c->render, in_srv, rx, ry, rw, rh, (float)in_w, (float)in_h, + (float)win_w + rx, ry, rw, rh, /*is_srgb*/ false, /*blend*/ nullptr, rtv, atw, ath); + } + return true; +} + +/*! + * v6 N-view layout: the crop (@c weave_crop_tex) holds the caller's packed + * atlas. For each lifted rect: snapshot its region of TILE 0 into the stream, + * then overwrite that region of every tile with the matching view of the + * stream's latest result. Until the first result the tiles stay as the caller + * drew them (a caller draws the 2D frame into every tile, i.e. flat). + */ +static void +lift_weave_rects_nview(struct d3d11_service_system *sys, + struct d3d11_service_compositor *c, + struct d3d11_lift *lift, + uint64_t owner, + const struct xrt_lift_weave_rect *bindings, + uint32_t binding_count, + const struct xrt_rect *rects, + uint32_t rect_count, + const struct xrt_weave_atlas_layout *layout, + uint32_t win_w, + uint32_t win_h) +{ + if (rects == nullptr || win_w == 0 || win_h == 0) { + return; + } + const uint32_t cvw = layout->content_view_w, cvh = layout->content_view_h; + const float sx = (float)cvw / (float)win_w, sy = (float)cvh / (float)win_h; + const uint32_t packed_w = layout->tile_columns * cvw, packed_h = layout->tile_rows * cvh; + ID3D11ShaderResourceView *crop_srv = c->render.weave_crop_srv.get(); + ID3D11RenderTargetView *crop_rtv = c->render.weave_crop_rtv.get(); + + for (uint32_t k = 0; k < binding_count; k++) { + const struct xrt_lift_weave_rect *b = &bindings[k]; + if (b->rect_index >= rect_count) { + continue; + } + const struct xrt_rect &r = rects[b->rect_index]; + // Tile-0 region, clipped to tile 0. + float fx0 = (float)r.offset.w * sx, fy0 = (float)r.offset.h * sy; + float fx1 = fx0 + (float)r.extent.w * sx, fy1 = fy0 + (float)r.extent.h * sy; + fx0 = fx0 < 0.0f ? 0.0f : fx0; + fy0 = fy0 < 0.0f ? 0.0f : fy0; + fx1 = fx1 > (float)cvw ? (float)cvw : fx1; + fy1 = fy1 > (float)cvh ? (float)cvh : fy1; + if (fx1 - fx0 < 1.0f || fy1 - fy0 < 1.0f) { + continue; + } + const uint32_t x0 = (uint32_t)fx0, y0 = (uint32_t)fy0; + const uint32_t w = (uint32_t)(fx1 - fx0), h = (uint32_t)(fy1 - fy0); + uint64_t frame_id = 0; + (void)d3d11_lift_submit_srv_locked(lift, owner, b->stream_id, crop_srv, packed_w, packed_h, x0, y0, w, h, + (int64_t)os_monotonic_get_ns(), b->has_params ? &b->params : nullptr, + &frame_id); + + struct d3d11_lift_pin pin = {}; + if (!d3d11_lift_pin_latest(lift, owner, b->stream_id, &pin) || pin.view_count == 0) { + continue; + } + const float vw = (float)pin.width / (float)pin.view_count; + // Letterbox: the result covers only the ACTIVE part of the rect; the + // tiles already hold the caller's flat 2D frame in the bars. + const float ax = (float)x0 + pin.active[0] * (float)w, ay = (float)y0 + pin.active[1] * (float)h; + const float aw = (pin.active[2] - pin.active[0]) * (float)w; + const float ah = (pin.active[3] - pin.active[1]) * (float)h; + for (uint32_t v = 0; v < layout->view_count; v++) { + // Map the atlas view onto the result's views (equal counts: 1:1). + uint32_t src_v = v; + if (pin.view_count != layout->view_count) { + src_v = layout->view_count > 1 + ? (uint32_t)((float)v * (float)(pin.view_count - 1) / (float)(layout->view_count - 1) + + 0.5f) + : pin.view_count / 2; + } + const float dx = (float)((v % layout->tile_columns) * cvw) + ax; + const float dy = (float)((v / layout->tile_columns) * cvh) + ay; + blit_to_atlas_texture(sys, &c->render, pin.srv, vw * (float)src_v, 0.0f, vw, (float)pin.height, + (float)pin.width, (float)pin.height, dx, dy, aw, ah, + /*is_srgb*/ false, /*blend*/ nullptr, crop_rtv, (float)packed_w, + (float)packed_h); + } + d3d11_lift_unpin(&pin); + } +} + + //! (Re)allocate the server-owned weaved-output texture + RTV (+ the persistent //! fence on first use) sized to @p w × @p h. Caller holds sys->render_mutex. static bool @@ -25112,6 +25344,23 @@ comp_d3d11_service_weave_submit(struct xrt_compositor *xc, return false; } + // XR_DXR_lift (ADR-042): take this submit's lift-flagged rects NOW, so every + // exit path below consumes them (a set is valid for exactly one submit). + struct xrt_lift_weave_rect lift_rects[XRT_LIFT_WEAVE_RECTS_MAX]; + uint32_t lift_count = c->render.lift_rect_count; + const uint64_t lift_owner = c->render.lift_owner; + if (lift_count > XRT_LIFT_WEAVE_RECTS_MAX) { + lift_count = XRT_LIFT_WEAVE_RECTS_MAX; + } + if (lift_count > 0) { + memcpy(lift_rects, c->render.lift_rects, lift_count * sizeof(lift_rects[0])); + } + c->render.lift_rect_count = 0; + struct d3d11_lift *lift = lift_count > 0 ? sys->lift : nullptr; + if (lift == nullptr) { + lift_count = 0; + } + // #625: choose the display processor that performs the weave. A standalone // present-owner (the WebXR / inline-3D browser bridge, the CEF host, the weave // probe) gets its OWN per-client DP in init_client_render_resources — but ONLY @@ -25539,10 +25788,15 @@ comp_d3d11_service_weave_submit(struct xrt_compositor *xc, // atlas to fill the ACTIVE mode's atlas exactly. With a swapchain sized // at the max over all modes, only the worst-case-achieving mode // qualifies, and only at fullscreen (ADR-030). - const bool zero_copy = (packed_w == idesc.Width && packed_h == idesc.Height); + // + // XR_DXR_lift (ADR-042): a frame with lift-flagged rects writes lifted + // views INTO the atlas, which must never be the caller's own texture — + // so it always takes the crop. + const bool zero_copy = (packed_w == idesc.Width && packed_h == idesc.Height) && lift_count == 0; if (!zero_copy) { if (!c->render.weave_crop_tex || c->render.weave_crop_w != packed_w || c->render.weave_crop_h != packed_h || c->render.weave_crop_format != idesc.Format) { + c->render.weave_crop_rtv.reset(); c->render.weave_crop_srv.reset(); c->render.weave_crop_tex.reset(); @@ -25554,13 +25808,22 @@ comp_d3d11_service_weave_submit(struct xrt_compositor *xc, cd.Format = idesc.Format; cd.SampleDesc.Count = 1; cd.Usage = D3D11_USAGE_DEFAULT; - cd.BindFlags = D3D11_BIND_SHADER_RESOURCE; + // RENDER_TARGET too: XR_DXR_lift writes lifted views into it. + cd.BindFlags = D3D11_BIND_SHADER_RESOURCE | D3D11_BIND_RENDER_TARGET; hr = sys->device->CreateTexture2D(&cd, nullptr, c->render.weave_crop_tex.put()); if (SUCCEEDED(hr)) { hr = sys->device->CreateShaderResourceView(c->render.weave_crop_tex.get(), nullptr, c->render.weave_crop_srv.put()); } + if (SUCCEEDED(hr)) { + // Non-fatal: without it only lifted rects are skipped. + if (FAILED(sys->device->CreateRenderTargetView(c->render.weave_crop_tex.get(), + nullptr, + c->render.weave_crop_rtv.put()))) { + c->render.weave_crop_rtv.reset(); + } + } if (FAILED(hr)) { U_LOG_E("#625 weave v6: crop texture %ux%u create failed: 0x%08lx", packed_w, packed_h, hr); @@ -25587,6 +25850,14 @@ comp_d3d11_service_weave_submit(struct xrt_compositor *xc, sys->context->CopySubresourceRegion(c->render.weave_crop_tex.get(), 0, 0, 0, 0, in_tex, 0, &box); dp_srv = c->render.weave_crop_srv.get(); + + // XR_DXR_lift (ADR-042): lift-flagged rects — snapshot tile 0's + // region into each stream, then overwrite that region of EVERY + // tile with the stream's latest view set. + if (lift_count > 0 && c->render.weave_crop_rtv) { + lift_weave_rects_nview(sys, c, lift, lift_owner, lift_rects, lift_count, rects, rect_count, + layout, win_w, win_h); + } } // Keyed-mutex release must straddle the DP read, NOT precede it. On the @@ -25781,6 +26052,14 @@ comp_d3d11_service_weave_submit(struct xrt_compositor *xc, if (rw <= 0.0f || rh <= 0.0f) { continue; } + // XR_DXR_lift (ADR-042): a lift-flagged rect holds 2D, not SBS. + // Snapshot it into its stream and weave the stream's latest + // stereo pair here, at the rect's CURRENT position. + if (lift_count > 0 && lift_weave_rect_batch(sys, c, lift, lift_owner, lift_rects, lift_count, i, + in_srv, idesc.Width, idesc.Height, rects[i], win_w, + win_h)) { + continue; + } // Left view: the rect's left half -> the left tile, at the // rect's own window position, stretched to full rect width. blit_to_atlas_texture(sys, &c->render, in_srv, rx, ry, rw / 2.0f, rh, src_tw, src_th, @@ -26550,6 +26829,9 @@ system_set_workspace_view_rig(struct xrt_system_compositor *xsysc, const struct return comp_d3d11_service_set_workspace_view_rig(xsysc, rig); } +static void +svc_lift_destroy(struct d3d11_service_system *sys); + static void system_destroy(struct xrt_system_compositor *xsysc) { @@ -26557,6 +26839,10 @@ system_destroy(struct xrt_system_compositor *xsysc) U_LOG_I("Destroying D3D11 service system compositor"); + // XR_DXR_lift (ADR-042): first — the lift thread calls back into this + // system for eyes, and owns a DP + device of its own. + svc_lift_destroy(sys); + /* * #918: stop the bridge BEFORE the multi-compositor goes, and before either * device does. `quiesce` stops submissions, joins the watchdog and drains @@ -30858,3 +31144,261 @@ comp_d3d11_service_poll_mcp_capture(struct xrt_system_compositor *xsysc) } mcp_capture_complete(&sys->mcp_capture, ok); } + + +/* + * + * XR_DXR_lift (ADR-042) — system-level entry points. Thin: resolve the system, + * create the lift module on first use, forward to d3d11_lift.cpp. None of these + * takes render_mutex; the ones that touch the shared immediate context take + * immediate_ctx_mutex inside d3d11_lift for one copy / blit only. + * + */ + +//! The lift thread's eye source: the panel DP's predicted eyes (display space). +static bool +svc_lift_eyes(void *ud, struct xrt_eye_positions *out) +{ + return comp_d3d11_service_get_predicted_eye_positions_full((struct xrt_system_compositor *)ud, out); +} + +static struct d3d11_lift * +svc_lift(struct xrt_system_compositor *xsysc) +{ + if (!comp_d3d11_service_is_d3d11_service(xsysc)) { + return nullptr; + } + struct d3d11_service_system *sys = d3d11_service_system_from_xrt(xsysc); + if (sys == nullptr || sys->device == nullptr || sys->context == nullptr) { + return nullptr; + } + std::lock_guard g(sys->lift_create_mutex); + if (sys->lift == nullptr) { + void *lift_factory = sys->base.info.dp_factory_d3d11_lift; + void *fallback = comp_dp_factory_for_window(&sys->base.info, COMP_DP_PRIMARY_MONITOR, COMP_DP_API_D3D11); + sys->lift = d3d11_lift_create(sys->device.get(), sys->context.get(), &sys->immediate_ctx_mutex, + lift_factory, fallback, svc_lift_eyes, (void *)xsysc); + U_LOG_W("[lift] lift module created (lift-only factory %s, fallback factory %s)", + lift_factory != nullptr ? "present" : "absent", fallback != nullptr ? "present" : "absent"); + } + return sys->lift; +} + +//! Called first thing in system_destroy: joins the lift thread (which calls +//! back into the system for eyes) before anything it uses goes away. +static void +svc_lift_destroy(struct d3d11_service_system *sys) +{ + std::lock_guard g(sys->lift_create_mutex); + d3d11_lift_destroy(&sys->lift); +} + +extern "C" void +comp_d3d11_service_lift_get_caps(struct xrt_system_compositor *xsysc, struct xrt_dp_lift_caps *out) +{ + xrt_dp_lift_caps_init(out); + struct d3d11_lift *l = svc_lift(xsysc); + if (l != nullptr) { + d3d11_lift_get_caps(l, out); + } +} + +extern "C" xrt_result_t +comp_d3d11_service_lift_stream_create(struct xrt_system_compositor *xsysc, + uint64_t owner, + const struct xrt_dp_lift_stream_info *info, + uint64_t *out_id) +{ + struct d3d11_lift *l = svc_lift(xsysc); + if (l == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + return d3d11_lift_stream_create(l, owner, info, out_id); +} + +extern "C" void +comp_d3d11_service_lift_stream_destroy(struct xrt_system_compositor *xsysc, uint64_t owner, uint64_t id) +{ + struct d3d11_lift *l = svc_lift(xsysc); + if (l != nullptr) { + d3d11_lift_stream_destroy(l, owner, id); + } +} + +extern "C" void +comp_d3d11_service_lift_release_owner(struct xrt_system_compositor *xsysc, uint64_t owner) +{ + if (!comp_d3d11_service_is_d3d11_service(xsysc)) { + return; + } + struct d3d11_service_system *sys = d3d11_service_system_from_xrt(xsysc); + std::lock_guard g(sys->lift_create_mutex); + if (sys->lift != nullptr) { // never CREATE the module just to release nothing + d3d11_lift_release_owner(sys->lift, owner); + } +} + +extern "C" xrt_result_t +comp_d3d11_service_lift_submit(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + xrt_graphics_buffer_handle_t handle, + bool is_dxgi, + uint32_t w, + uint32_t h, + int64_t source_time, + const struct xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_floats, + uint64_t *out_frame_id) +{ + struct d3d11_lift *l = svc_lift(xsysc); + if (l == nullptr) { + if (handle != nullptr && !is_dxgi) { + CloseHandle((HANDLE)handle); + } + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + return d3d11_lift_submit_handle(l, owner, id, (HANDLE)handle, is_dxgi, w, h, source_time, params, viewpoints, + viewpoint_floats, out_frame_id); +} + +extern "C" xrt_result_t +comp_d3d11_service_lift_acquire(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + bool *out_ready, + struct xrt_lift_result *out) +{ + *out_ready = false; + memset(out, 0, sizeof(*out)); + struct d3d11_lift *l = svc_lift(xsysc); + if (l == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + struct d3d11_lift_result_info r = {}; + xrt_result_t xret = d3d11_lift_acquire_result(l, owner, id, out_ready, &r); + out->frame_id = r.frame_id; + out->source_time = r.source_time; + out->fence_value = r.fence_value; + out->latency_ns = r.latency_ns; + out->width = r.width; + out->height = r.height; + out->format = r.format; + out->view_count = r.view_count; + out->output_realloc = r.output_realloc; + return xret; +} + +extern "C" bool +comp_d3d11_service_lift_export_output(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + xrt_graphics_buffer_handle_t *out_handle, + uint32_t *out_width, + uint32_t *out_height, + uint32_t *out_format) +{ + struct d3d11_lift *l = svc_lift(xsysc); + HANDLE h = nullptr; + if (l == nullptr || !d3d11_lift_export_output(l, owner, id, &h, out_width, out_height, out_format)) { + return false; + } + *out_handle = (xrt_graphics_buffer_handle_t)h; + return true; +} + +extern "C" bool +comp_d3d11_service_lift_export_fence(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + xrt_graphics_sync_handle_t *out_handle) +{ + struct d3d11_lift *l = svc_lift(xsysc); + HANDLE h = nullptr; + if (l == nullptr || !d3d11_lift_export_fence(l, owner, id, &h)) { + return false; + } + *out_handle = (xrt_graphics_sync_handle_t)h; + return true; +} + +extern "C" xrt_result_t +comp_d3d11_service_lift_acquire_blob(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + uint64_t capacity, + bool *out_ready, + struct xrt_lift_blob_info *out_info, + uint8_t **out_bytes) +{ + *out_ready = false; + memset(out_info, 0, sizeof(*out_info)); + *out_bytes = nullptr; + struct d3d11_lift *l = svc_lift(xsysc); + if (l == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + struct d3d11_lift_blob_info bi = {}; + xrt_result_t xret = d3d11_lift_acquire_blob(l, owner, id, capacity, out_ready, &bi, out_bytes); + out_info->frame_id = bi.frame_id; + out_info->source_time = bi.source_time; + out_info->format = bi.format; + out_info->byte_count = bi.byte_count; + return xret; +} + +extern "C" bool +comp_d3d11_service_lift_set_weave_rects(struct xrt_compositor *xc, + uint64_t owner, + uint32_t count, + const struct xrt_lift_weave_rect *rects) +{ + if (xc == nullptr || xc->destroy != compositor_destroy || count > XRT_LIFT_WEAVE_RECTS_MAX || + (count > 0 && rects == nullptr)) { + return false; + } + struct d3d11_service_compositor *c = d3d11_service_compositor_from_xrt(xc); + if (c->sys == nullptr) { + return false; + } + if (count > 0) { + // Create the module HERE (outside every weave lock), never inside the + // submit that consumes these rects. + if (svc_lift(&c->sys->base) == nullptr) { + return false; + } + // Only this owner's live SBS / NVIEW streams can be woven. + for (uint32_t i = 0; i < count; i++) { + uint32_t mode = d3d11_lift_stream_mode(c->sys->lift, owner, rects[i].stream_id); + if (mode != XRT_DP_LIFT_MODE_SBS && mode != XRT_DP_LIFT_MODE_NVIEW) { + return false; + } + } + memcpy(c->render.lift_rects, rects, count * sizeof(rects[0])); + } + c->render.lift_owner = owner; + c->render.lift_rect_count = count; + return true; +} + +extern "C" xrt_result_t +comp_d3d11_service_lift_set_priority(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + uint32_t priority) +{ + struct d3d11_lift *l = svc_lift(xsysc); + return l != nullptr ? d3d11_lift_set_priority(l, owner, id, priority) : XRT_ERROR_FEATURE_NOT_SUPPORTED; +} + +extern "C" xrt_result_t +comp_d3d11_service_lift_get_stats(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + struct xrt_lift_stream_stats *out) +{ + memset(out, 0, sizeof(*out)); + struct d3d11_lift *l = svc_lift(xsysc); + return l != nullptr ? d3d11_lift_get_stats(l, owner, id, out) : XRT_ERROR_FEATURE_NOT_SUPPORTED; +} diff --git a/src/xrt/compositor/d3d11_service/comp_d3d11_service.h b/src/xrt/compositor/d3d11_service/comp_d3d11_service.h index 895079bac..de1214c9a 100644 --- a/src/xrt/compositor/d3d11_service/comp_d3d11_service.h +++ b/src/xrt/compositor/d3d11_service/comp_d3d11_service.h @@ -21,6 +21,8 @@ #include "xrt/xrt_display_metrics.h" #include "xrt/xrt_handles.h" #include "xrt/xrt_system.h" +#include "xrt/xrt_dp_lift.h" +#include "xrt/xrt_lift.h" #ifdef __cplusplus extern "C" { @@ -1193,6 +1195,116 @@ comp_d3d11_service_weave_snap_window_rect(struct xrt_compositor *xc, int32_t *out_x, int32_t *out_y); +/* + * XR_DXR_lift (ADR-042) — 2D→3D conversion streams. SYSTEM-level (a stream is + * owned by an IPC connection, identified by @p owner, not by a session), so a + * headless probe can drive them without a compositor. The work runs on the + * service's lift thread (d3d11_lift.h); none of these waits for a conversion. + * Every function is a no-op / FEATURE_NOT_SUPPORTED when @p xsysc is not the + * D3D11 service. + */ + +//! Value types: xrt_lift_result / xrt_lift_blob_info / xrt_lift_weave_rect (xrt_lift.h). + +//! The conversion module's caps (cached; kicks module activation on first call). +void +comp_d3d11_service_lift_get_caps(struct xrt_system_compositor *xsysc, struct xrt_dp_lift_caps *out); + +xrt_result_t +comp_d3d11_service_lift_stream_create(struct xrt_system_compositor *xsysc, + uint64_t owner, + const struct xrt_dp_lift_stream_info *info, + uint64_t *out_id); + +void +comp_d3d11_service_lift_stream_destroy(struct xrt_system_compositor *xsysc, uint64_t owner, uint64_t id); + +//! Destroy every stream @p owner created (IPC client teardown). +void +comp_d3d11_service_lift_release_owner(struct xrt_system_compositor *xsysc, uint64_t owner); + +/*! + * xrSubmitLiftFrameDXR: snapshot @p w x @p h of the caller's shared texture into + * the stream's latest-wins mailbox and return its frame id (0 = the module is + * not up yet; the frame was not taken). XRT_ERROR_WEAVE_REFUSED = transient + * (keyed-mutex miss), retry next frame. Takes ownership of an NT @p handle. + */ +xrt_result_t +comp_d3d11_service_lift_submit(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + xrt_graphics_buffer_handle_t handle, + bool is_dxgi, + uint32_t w, + uint32_t h, + int64_t source_time, + const struct xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_floats, + uint64_t *out_frame_id); + +//! xrAcquireLiftResultDXR. @p out_ready false = nothing newer (NOT READY). +xrt_result_t +comp_d3d11_service_lift_acquire(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + bool *out_ready, + struct xrt_lift_result *out); + +bool +comp_d3d11_service_lift_export_output(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + xrt_graphics_buffer_handle_t *out_handle, + uint32_t *out_width, + uint32_t *out_height, + uint32_t *out_format); + +bool +comp_d3d11_service_lift_export_fence(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + xrt_graphics_sync_handle_t *out_handle); + +/*! + * xrAcquireLiftBlobDXR (two-call latch). On delivery @p out_bytes is a malloc'd + * copy the caller frees. + */ +xrt_result_t +comp_d3d11_service_lift_acquire_blob(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + uint64_t capacity, + bool *out_ready, + struct xrt_lift_blob_info *out_info, + uint8_t **out_bytes); + +//! XrLiftPriorityDXR (0 paused .. 3 high) for one stream. +xrt_result_t +comp_d3d11_service_lift_set_priority(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + uint32_t priority); + +//! One stream's counters + effective conversion rate. +xrt_result_t +comp_d3d11_service_lift_get_stats(struct xrt_system_compositor *xsysc, + uint64_t owner, + uint64_t id, + struct xrt_lift_stream_stats *out); + +/*! + * Latch the lift-flagged rects of this client's NEXT weave submit + * (XrWeaveSubmitLiftRectsDXR). Consumed — and cleared — by that submit. Each + * rect's content is snapshotted into its stream and the stream's LATEST result + * is woven at the rect's CURRENT position (flat until the first result). + */ +bool +comp_d3d11_service_lift_set_weave_rects(struct xrt_compositor *xc, + uint64_t owner, + uint32_t count, + const struct xrt_lift_weave_rect *rects); + /*! @} */ diff --git a/src/xrt/compositor/d3d11_service/d3d11_lift.cpp b/src/xrt/compositor/d3d11_service/d3d11_lift.cpp new file mode 100644 index 000000000..efbd19b09 --- /dev/null +++ b/src/xrt/compositor/d3d11_service/d3d11_lift.cpp @@ -0,0 +1,2147 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief XR_DXR_lift (ADR-042) on the D3D11 service — see d3d11_lift.h. + * + * Deliberately raw COM (no WIL) so the file is portable to the MinGW + * compile-check (scripts/build-mingw-check.sh cannot build WIL). + * + * @ingroup comp_d3d11_service + */ + +#include "d3d11_lift.h" + +#include "xrt/xrt_display_processor_d3d11.h" + +#include "util/u_lift_mailbox.h" +#include "util/u_logging.h" +#include "os/os_time.h" + +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + + +/* + * + * Small helpers. + * + */ + +template +static void +rel(T *&p) +{ + if (p != nullptr) { + p->Release(); + p = nullptr; + } +} + +static void +close_handle(HANDLE &h) +{ + if (h != nullptr && h != INVALID_HANDLE_VALUE) { + CloseHandle(h); + } + h = nullptr; +} + +//! Round up to a 64 px multiple — input slots grow, they don't chase every size. +static uint32_t +round_cap(uint32_t v) +{ + return (v + 63u) & ~63u; +} + +static const char * +state_str(uint32_t s) +{ + switch (s) { + case XRT_DP_LIFT_STATE_READY: return "READY"; + case XRT_DP_LIFT_STATE_ACTIVATING: return "ACTIVATING"; + default: return "UNAVAILABLE"; + } +} + +//! CompareObjectHandles (Windows 10+), resolved at runtime so older SDKs build. +static bool +same_kernel_object(HANDLE a, HANDLE b) +{ + typedef BOOL(WINAPI * pfn_t)(HANDLE, HANDLE); + static pfn_t fn = []() -> pfn_t { + HMODULE m = GetModuleHandleA("kernelbase.dll"); + return m != nullptr ? (pfn_t)(void *)GetProcAddress(m, "CompareObjectHandles") : nullptr; + }(); + if (fn == nullptr || a == nullptr || b == nullptr) { + return false; + } + return fn(a, b) != FALSE; +} + +static DXGI_FORMAT +typed_srv_format(DXGI_FORMAT f) +{ + switch (f) { + case DXGI_FORMAT_R8G8B8A8_TYPELESS: return DXGI_FORMAT_R8G8B8A8_UNORM; + case DXGI_FORMAT_B8G8R8A8_TYPELESS: return DXGI_FORMAT_B8G8R8A8_UNORM; + case DXGI_FORMAT_R10G10B10A2_TYPELESS: return DXGI_FORMAT_R10G10B10A2_UNORM; + case DXGI_FORMAT_R16G16B16A16_TYPELESS: return DXGI_FORMAT_R16G16B16A16_FLOAT; + default: return f; + } +} + + +/* + * + * Snapshot blit (service device): sample a source sub-rect into a slot — 1:1 + * (point) when the snapshot is uncapped, box-filtered down when the service's + * DXR_LIFT_MAX_INPUT_EDGE cap engages (u_lift_cap_dims). + * + */ + +static const char *k_snap_vs = R"( +struct VSO { float4 pos : SV_Position; float2 uv : TEXCOORD0; }; +VSO main(uint id : SV_VertexID) { + VSO o; + o.uv = float2(id & 1, id >> 1); + o.pos = float4(o.uv * float2(2, -2) + float2(-1, 1), 0, 1); + return o; +} +)"; + +// src_rect = (x, y, w, h) in source texels, src_size = (tw, th). +static const char *k_snap_ps = R"( +cbuffer C : register(b0) { float4 src_rect; float4 src_size; }; +Texture2D src : register(t0); +SamplerState samp : register(s0); +float4 main(float4 pos : SV_Position, float2 uv : TEXCOORD0) : SV_Target { + float2 p = src_rect.xy + uv * src_rect.zw; + float4 c = src.SampleLevel(samp, p / src_size.xy, 0); + return float4(c.rgb, 1.0); +} +)"; + +/* + * Downscaling variant (capped snapshot). One destination texel covers + * step = src_rect.zw / dst_size.xy source texels; four bilinear taps at the + * quarter points of that footprint are an exact box filter up to a 4x reduction + * (7680 -> 1920) and a much better one than a single tap beyond it. Taps are + * clamped to texel centres inside the sub-rect, so nothing outside the lifted + * rect bleeds in at its edges. + */ +static const char *k_snap_ps_scaled = R"( +cbuffer C : register(b0) { float4 src_rect; float4 src_size; float4 dst_size; }; +Texture2D src : register(t0); +SamplerState samp : register(s0); +float3 tap(float2 p, float2 lo, float2 hi) { + return src.SampleLevel(samp, clamp(p, lo, hi) / src_size.xy, 0).rgb; +} +float4 main(float4 pos : SV_Position, float2 uv : TEXCOORD0) : SV_Target { + float2 p = src_rect.xy + uv * src_rect.zw; + float2 q = 0.25 * src_rect.zw / dst_size.xy; + float2 lo = src_rect.xy + 0.5; + float2 hi = src_rect.xy + src_rect.zw - 0.5; + float3 c = tap(p + float2(-q.x, -q.y), lo, hi) + tap(p + float2(q.x, -q.y), lo, hi) + + tap(p + float2(-q.x, q.y), lo, hi) + tap(p + float2(q.x, q.y), lo, hi); + return float4(c * 0.25, 1.0); +} +)"; + +struct snap_cb +{ + float src_rect[4]; + float src_size[4]; + float dst_size[4]; //!< read by k_snap_ps_scaled only +}; + +/* + * Letterbox profile (DXR_LIFT_LETTERBOX). Target: R32_FLOAT, + * U_LIFT_LETTERBOX_BINS_MAX x 2 — row 0 = the rect's row buckets top to bottom, + * row 1 = its column buckets left to right. Each texel is the non-black + * fraction of its bucket: the max over the bucket's first, middle and last + * line, each line sampled at 96 points across the rect. Taking the max over + * the first and LAST line means a bucket holding the picture's first row is + * already picture, so a bar never swallows a picture row. + * bins = (row buckets, column buckets, -, -). + */ +static const char *k_lb_ps = R"( +cbuffer C : register(b0) { float4 src_rect; float4 bins; }; +Texture2D src : register(t0); +float line_frac(bool rows, float a, float span) { + float cnt = 0.0; + [loop] for (int k = 0; k < 96; k++) { + float b = floor((k + 0.5) * span / 96.0); + int2 p = rows ? int2(src_rect.x + b, src_rect.y + a) : int2(src_rect.x + a, src_rect.y + b); + float3 c = src.Load(int3(p, 0)).rgb; + cnt += dot(c, float3(0.2126, 0.7152, 0.0722)) > 0.07 ? 1.0 : 0.0; + } + return cnt / 96.0; +} +float4 main(float4 pos : SV_Position) : SV_Target { + uint i = (uint)pos.x; + bool rows = pos.y < 1.0; + float n = rows ? bins.x : bins.y; + if ((float)i >= n) { + return 0.0; + } + float len = rows ? src_rect.w : src_rect.z; // along the profile + float span = rows ? src_rect.z : src_rect.w; // across it + float a0 = floor(i * len / n); + float a1 = max(a0, floor((i + 1) * len / n) - 1.0); + float am = floor((a0 + a1) * 0.5); + return max(line_frac(rows, a0, span), max(line_frac(rows, am, span), line_frac(rows, a1, span))); +} +)"; + +struct lb_cb +{ + float src_rect[4]; + float bins[4]; +}; + +//! Staging readbacks in flight per stream (async: never stall the weave). +#define LB_RING 3 + + +/* + * + * Stream. + * + */ + +//! One input slot: created on the SERVICE device, opened on the LIFT device. +struct lift_in_slot +{ + ID3D11Texture2D *svc_tex = nullptr; + ID3D11RenderTargetView *svc_rtv = nullptr; + IDXGIKeyedMutex *svc_km = nullptr; + HANDLE share = nullptr; + ID3D11Texture2D *lift_tex = nullptr; + IDXGIKeyedMutex *lift_km = nullptr; + uint32_t cap_w = 0, cap_h = 0; + + // Parameters of the frame currently in this slot. + xrt_dp_lift_params params = {}; + float viewpoints[3 * XRT_DP_LIFT_MAX_EXPLICIT_VIEWPOINTS] = {}; + uint32_t viewpoint_floats = 0; + //! The part of the submitted rect this frame holds (letterbox crop): + //! x0, y0, x1, y1 normalised to the rect. {0,0,1,1} = the whole rect. + float active[4] = {0.0f, 0.0f, 1.0f, 1.0f}; +}; + +//! One output ring slot: created on the LIFT device, opened on the SERVICE device. +struct lift_out_slot +{ + ID3D11Texture2D *lift_tex = nullptr; + IDXGIKeyedMutex *lift_km = nullptr; + HANDLE share = nullptr; + ID3D11Texture2D *svc_tex = nullptr; + ID3D11ShaderResourceView *svc_srv = nullptr; + IDXGIKeyedMutex *svc_km = nullptr; + uint32_t w = 0, h = 0, format = 0, view_count = 0; + std::shared_ptr> blob; //!< GAUSSIANS + uint32_t blob_format = 0; + float active[4] = {0.0f, 0.0f, 1.0f, 1.0f}; //!< the input's lift_in_slot::active +}; + +//! One letterbox profile readback (service device, producer thread only). +struct lift_lb_readback +{ + ID3D11Texture2D *staging = nullptr; + bool pending = false; + uint32_t w = 0, h = 0; //!< rect dims it measured + uint32_t nr = 0, nc = 0; //!< buckets used +}; + +struct lift_stream +{ + uint64_t id = 0; + uint64_t owner = 0; + xrt_dp_lift_stream_info info = {}; + u_lift_mailbox mb = {}; + lift_in_slot in[U_LIFT_INPUT_SLOTS]; + lift_out_slot out[U_LIFT_RING_SIZE]; + xrt_dp_lift_params last_params = {}; + float last_viewpoints[3 * XRT_DP_LIFT_MAX_EXPLICIT_VIEWPOINTS] = {}; + uint32_t last_viewpoint_floats = 0; + + uint32_t priority = U_LIFT_PRIORITY_NORMAL; //!< XrLiftPriorityDXR + bool dead = false; //!< destroy requested; the lift thread reaps it + bool converting = false; //!< the lift thread is inside a conversion for it + + // Lift thread only. + uint64_t dp_id = 0; + bool dp_created = false; + bool dp_failed = false; + ID3D11Texture2D *exact_tex = nullptr; + uint32_t exact_w = 0, exact_h = 0; + + // Caller-input import cache (service device, producer thread only). + HANDLE imp_handle = nullptr; + bool imp_dxgi = false; + ID3D11Texture2D *imp_tex = nullptr; + ID3D11ShaderResourceView *imp_srv = nullptr; + IDXGIKeyedMutex *imp_km = nullptr; + uint32_t imp_w = 0, imp_h = 0; + + // Client export (service device) — the weave output pattern. + ID3D11Texture2D *exp_tex = nullptr; + HANDLE exp_handle = nullptr; + ID3D11Fence *exp_fence = nullptr; + HANDLE exp_fence_handle = nullptr; + uint64_t exp_fence_value = 0; + uint32_t exp_w = 0, exp_h = 0, exp_format = 0; + + // Blob latch (xrAcquireLiftBlobDXR two-call idiom). + std::shared_ptr> blob_latched; + u_lift_frame_meta blob_latched_meta = {}; + uint32_t blob_latched_format = 0; + + // Throttled INFO. + uint64_t last_stats_log_ns = 0; + + // Snapshot cap: the last capped size WARNed about (producer thread, under mtx). + uint32_t cap_logged_w = 0, cap_logged_h = 0; + + // Letterbox crop (producer thread only; the crop is read under mtx at submit). + u_lift_letterbox lb = {}; + lift_lb_readback lb_rb[LB_RING]; + uint32_t lb_next = 0; //!< next ring entry to issue into +}; + +struct d3d11_lift +{ + // Service side (borrowed). + ID3D11Device *svc_device = nullptr; + ID3D11Device1 *svc_device1 = nullptr; + ID3D11DeviceContext *svc_context = nullptr; + ID3D11DeviceContext4 *svc_context4 = nullptr; + std::mutex *svc_ctx_mutex = nullptr; + xrt_dp_factory_d3d11_fn_t lift_factory = nullptr; + xrt_dp_factory_d3d11_fn_t fallback_factory = nullptr; + bool (*eyes_fn)(void *ud, struct xrt_eye_positions *out) = nullptr; + void *eyes_ud = nullptr; + + // Snapshot blit (service device). + ID3D11VertexShader *snap_vs = nullptr; + ID3D11PixelShader *snap_ps = nullptr; + ID3D11PixelShader *snap_ps_scaled = nullptr; + ID3D11SamplerState *snap_sampler = nullptr; + ID3D11SamplerState *snap_sampler_linear = nullptr; + ID3D11Buffer *snap_cb = nullptr; + bool snap_ok = false; + //! DXR_LIFT_MAX_INPUT_EDGE, read once at create: long-edge cap on weave-rect + //! snapshots (0 = off). Never applied to xrSubmitLiftFrameDXR frames. + uint32_t max_input_edge = U_LIFT_MAX_INPUT_EDGE_DEFAULT; + //! DXR_LIFT_LETTERBOX, read once at create (default on). Weave-rect + //! snapshots only, like the cap. + bool letterbox = true; + ID3D11PixelShader *lb_ps = nullptr; + ID3D11Texture2D *lb_tex = nullptr; //!< R32F BINS_MAX x 2 profile target + ID3D11RenderTargetView *lb_rtv = nullptr; + ID3D11Buffer *lb_cbuf = nullptr; + bool lb_ok = false; + + // Lift side (lift thread only, after activation). + ID3D11Device *lift_device = nullptr; + ID3D11Device1 *lift_device1 = nullptr; + ID3D11DeviceContext *lift_context = nullptr; + struct xrt_display_processor_d3d11 *dp = nullptr; + + // Shared state, under mtx. + std::mutex mtx; + std::condition_variable cv; + bool stop = false; + bool activate_requested = false; + bool activated = false; + xrt_dp_lift_caps caps = {}; + uint32_t logged_state = 0xffffffffu; + uint64_t next_stream_id = 0; + uint32_t live_streams = 0; + std::map> streams; + u_lift_sched sched = {}; //!< cross-stream priority scheduling (u_lift_mailbox.h) + + std::thread thread; +}; + +static void +caps_set_state(d3d11_lift *l, uint32_t state, const char *why) +{ + // Caller holds mtx. WARN once per state CHANGE, never per poll. + l->caps.state = state; + if (l->logged_state != state) { + l->logged_state = state; + U_LOG_W("[lift] module state -> %s (modes=0x%x backend='%s' max_streams=%u max_views=%u)%s%s", + state_str(state), l->caps.modes, l->caps.backend, l->caps.max_streams, l->caps.max_views, + why != nullptr ? " — " : "", why != nullptr ? why : ""); + } +} + + +/* + * + * Slot allocation. + * + */ + +static void +in_slot_release(lift_in_slot &s) +{ + rel(s.lift_km); + rel(s.lift_tex); + close_handle(s.share); + rel(s.svc_km); + rel(s.svc_rtv); + rel(s.svc_tex); + s.cap_w = s.cap_h = 0; +} + +static void +out_slot_release(lift_out_slot &s) +{ + rel(s.svc_km); + rel(s.svc_srv); + rel(s.svc_tex); + close_handle(s.share); + rel(s.lift_km); + rel(s.lift_tex); + s.w = s.h = s.format = s.view_count = 0; + s.blob.reset(); +} + +/*! + * Make input slot @p s hold at least @p w x @p h. Created on the service device + * (RGBA8, RT for the snapshot blit, keyed mutex, NT handle), opened on the lift + * device. Needs the lift device, so a stream can only be fed after activation. + */ +static bool +in_slot_ensure(d3d11_lift *l, lift_in_slot &s, uint32_t w, uint32_t h) +{ + if (s.svc_tex != nullptr && s.cap_w >= w && s.cap_h >= h) { + return true; + } + if (l->lift_device1 == nullptr) { + return false; + } + uint32_t cw = round_cap(w > s.cap_w ? w : s.cap_w); + uint32_t ch = round_cap(h > s.cap_h ? h : s.cap_h); + in_slot_release(s); + + D3D11_TEXTURE2D_DESC td = {}; + td.Width = cw; + td.Height = ch; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = DXGI_FORMAT_R8G8B8A8_UNORM; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_DEFAULT; + td.BindFlags = D3D11_BIND_RENDER_TARGET | D3D11_BIND_SHADER_RESOURCE; + td.MiscFlags = D3D11_RESOURCE_MISC_SHARED_NTHANDLE | D3D11_RESOURCE_MISC_SHARED_KEYEDMUTEX; + HRESULT hr = l->svc_device->CreateTexture2D(&td, nullptr, &s.svc_tex); + if (SUCCEEDED(hr)) { + hr = l->svc_device->CreateRenderTargetView(s.svc_tex, nullptr, &s.svc_rtv); + } + if (SUCCEEDED(hr)) { + hr = s.svc_tex->QueryInterface(__uuidof(IDXGIKeyedMutex), (void **)&s.svc_km); + } + IDXGIResource1 *r1 = nullptr; + if (SUCCEEDED(hr)) { + hr = s.svc_tex->QueryInterface(__uuidof(IDXGIResource1), (void **)&r1); + } + if (SUCCEEDED(hr)) { + hr = r1->CreateSharedHandle(nullptr, DXGI_SHARED_RESOURCE_READ | DXGI_SHARED_RESOURCE_WRITE, nullptr, + &s.share); + } + rel(r1); + if (SUCCEEDED(hr)) { + hr = l->lift_device1->OpenSharedResource1(s.share, __uuidof(ID3D11Texture2D), (void **)&s.lift_tex); + } + if (SUCCEEDED(hr)) { + hr = s.lift_tex->QueryInterface(__uuidof(IDXGIKeyedMutex), (void **)&s.lift_km); + } + if (FAILED(hr)) { + U_LOG_E("[lift] input slot %ux%u create/share failed: 0x%08lx", cw, ch, (unsigned long)hr); + in_slot_release(s); + return false; + } + s.cap_w = cw; + s.cap_h = ch; + U_LOG_I("[lift] input slot %ux%u ready", cw, ch); + return true; +} + +/*! + * Make output slot @p s exactly @p w x @p h in @p format. Created on the lift + * device (keyed mutex, NT handle), opened on the service device. Lift thread. + */ +static bool +out_slot_ensure(d3d11_lift *l, lift_out_slot &s, uint32_t w, uint32_t h, uint32_t format) +{ + if (s.lift_tex != nullptr && s.w == w && s.h == h && s.format == format) { + return true; + } + std::shared_ptr> keep_blob = s.blob; + out_slot_release(s); + s.blob = keep_blob; + + D3D11_TEXTURE2D_DESC td = {}; + td.Width = w; + td.Height = h; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = (DXGI_FORMAT)format; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_DEFAULT; + td.BindFlags = D3D11_BIND_SHADER_RESOURCE; + td.MiscFlags = D3D11_RESOURCE_MISC_SHARED_NTHANDLE | D3D11_RESOURCE_MISC_SHARED_KEYEDMUTEX; + HRESULT hr = l->lift_device->CreateTexture2D(&td, nullptr, &s.lift_tex); + if (SUCCEEDED(hr)) { + hr = s.lift_tex->QueryInterface(__uuidof(IDXGIKeyedMutex), (void **)&s.lift_km); + } + IDXGIResource1 *r1 = nullptr; + if (SUCCEEDED(hr)) { + hr = s.lift_tex->QueryInterface(__uuidof(IDXGIResource1), (void **)&r1); + } + if (SUCCEEDED(hr)) { + hr = r1->CreateSharedHandle(nullptr, DXGI_SHARED_RESOURCE_READ | DXGI_SHARED_RESOURCE_WRITE, nullptr, + &s.share); + } + rel(r1); + if (SUCCEEDED(hr)) { + hr = l->svc_device1->OpenSharedResource1(s.share, __uuidof(ID3D11Texture2D), (void **)&s.svc_tex); + } + if (SUCCEEDED(hr)) { + hr = s.svc_tex->QueryInterface(__uuidof(IDXGIKeyedMutex), (void **)&s.svc_km); + } + if (SUCCEEDED(hr)) { + D3D11_SHADER_RESOURCE_VIEW_DESC sd = {}; + sd.Format = typed_srv_format((DXGI_FORMAT)format); + sd.ViewDimension = D3D11_SRV_DIMENSION_TEXTURE2D; + sd.Texture2D.MipLevels = 1; + hr = l->svc_device->CreateShaderResourceView(s.svc_tex, &sd, &s.svc_srv); + } + if (FAILED(hr)) { + U_LOG_E("[lift] output slot %ux%u fmt=%u create/share failed: 0x%08lx", w, h, format, + (unsigned long)hr); + out_slot_release(s); + return false; + } + s.w = w; + s.h = h; + s.format = format; + U_LOG_I("[lift] output slot %ux%u fmt=%u ready", w, h, format); + return true; +} + +static void +stream_release_gpu(lift_stream &st) +{ + for (auto &s : st.in) { + in_slot_release(s); + } + for (auto &s : st.out) { + out_slot_release(s); + } + rel(st.exact_tex); + rel(st.imp_km); + rel(st.imp_srv); + rel(st.imp_tex); + if (!st.imp_dxgi) { + close_handle(st.imp_handle); + } + st.imp_handle = nullptr; + rel(st.exp_fence); + close_handle(st.exp_fence_handle); + rel(st.exp_tex); + close_handle(st.exp_handle); + st.blob_latched.reset(); + for (auto &r : st.lb_rb) { + rel(r.staging); + r.pending = false; + } +} + + +/* + * + * Lift thread. + * + */ + +//! Bring up the lift device + the vendor module's DP. Lift thread, no lock held. +static void +lift_activate(d3d11_lift *l) +{ + const char *kill = getenv("DXR_LIFT"); + if (kill != nullptr && strcmp(kill, "0") == 0) { + std::lock_guard g(l->mtx); + l->activated = true; + caps_set_state(l, XRT_DP_LIFT_STATE_UNAVAILABLE, "DXR_LIFT=0 (kill switch)"); + return; + } + if (l->lift_factory == nullptr && l->fallback_factory == nullptr) { + std::lock_guard g(l->mtx); + l->activated = true; + caps_set_state(l, XRT_DP_LIFT_STATE_UNAVAILABLE, "no D3D11 display-processor factory"); + return; + } + + // Same adapter as the service device: shared textures cross between them. + IDXGIDevice *dxgi_dev = nullptr; + IDXGIAdapter *adapter = nullptr; + HRESULT hr = l->svc_device->QueryInterface(__uuidof(IDXGIDevice), (void **)&dxgi_dev); + if (SUCCEEDED(hr)) { + hr = dxgi_dev->GetAdapter(&adapter); + } + rel(dxgi_dev); + D3D_FEATURE_LEVEL levels[] = {D3D_FEATURE_LEVEL_11_1, D3D_FEATURE_LEVEL_11_0}; + D3D_FEATURE_LEVEL got = D3D_FEATURE_LEVEL_11_0; + if (SUCCEEDED(hr)) { + hr = D3D11CreateDevice(adapter, D3D_DRIVER_TYPE_UNKNOWN, nullptr, D3D11_CREATE_DEVICE_BGRA_SUPPORT, + levels, 2, D3D11_SDK_VERSION, &l->lift_device, &got, &l->lift_context); + } + rel(adapter); + if (SUCCEEDED(hr)) { + hr = l->lift_device->QueryInterface(__uuidof(ID3D11Device1), (void **)&l->lift_device1); + } + if (SUCCEEDED(hr)) { + // A conversion module may flush / signal this context from its own + // worker thread — the same protection the vendor weaver relies on. + ID3D11Multithread *mt = nullptr; + if (SUCCEEDED(l->lift_context->QueryInterface(__uuidof(ID3D11Multithread), (void **)&mt))) { + mt->SetMultithreadProtected(TRUE); + rel(mt); + } + } + if (FAILED(hr)) { + rel(l->lift_device1); + rel(l->lift_context); + rel(l->lift_device); + std::lock_guard g(l->mtx); + l->activated = true; + caps_set_state(l, XRT_DP_LIFT_STATE_UNAVAILABLE, "lift device creation failed"); + return; + } + + // The lift DP — never weaves, carries only the module. The plug-in's + // explicit lift-only factory when it has one (no weaver, no tracker + // session); else its ordinary factory with a NULL window. + const bool lift_only = l->lift_factory != nullptr; + xrt_result_t xret = + (lift_only ? l->lift_factory : l->fallback_factory)(l->lift_device, l->lift_context, nullptr, &l->dp); + U_LOG_W("[lift] lift DP %s via the plug-in's %s factory (dedicated device, NULL window)", + xret == XRT_SUCCESS && l->dp != nullptr ? "created" : "REFUSED", + lift_only ? "lift-only" : "ordinary (no lift-only factory)"); + xrt_dp_lift_caps caps; + bool have = false; + if (xret == XRT_SUCCESS && l->dp != nullptr && xrt_display_processor_d3d11_has_lift(l->dp)) { + have = xrt_display_processor_d3d11_lift_get_caps(l->dp, &caps); + } else { + xrt_dp_lift_caps_init(&caps); + } + if (!have && l->dp != nullptr) { + // No module on this plug-in: don't keep a vendor DP instance alive for nothing. + xrt_display_processor_d3d11_destroy(&l->dp); + } + + std::lock_guard g(l->mtx); + l->activated = true; + if (!have) { + l->caps = caps; + caps_set_state(l, XRT_DP_LIFT_STATE_UNAVAILABLE, + xret != XRT_SUCCESS ? "lift DP factory refused (NULL window)" + : "display processor ships no conversion module"); + return; + } + l->caps = caps; + if (l->caps.state > XRT_DP_LIFT_STATE_READY) { + l->caps.state = XRT_DP_LIFT_STATE_UNAVAILABLE; + } + if (l->caps.modes == 0) { + l->caps.state = XRT_DP_LIFT_STATE_UNAVAILABLE; + } + l->logged_state = 0xffffffffu; + caps_set_state(l, l->caps.state, nullptr); +} + +//! Poll caps while not READY (a module warming up). Lift thread, no lock held. +static void +lift_poll_caps(d3d11_lift *l) +{ + if (l->dp == nullptr) { + return; + } + xrt_dp_lift_caps caps; + bool have = xrt_display_processor_d3d11_lift_get_caps(l->dp, &caps); + std::lock_guard g(l->mtx); + if (!have) { + caps_set_state(l, XRT_DP_LIFT_STATE_UNAVAILABLE, "module stopped answering"); + return; + } + uint32_t st = caps.state > XRT_DP_LIFT_STATE_READY ? XRT_DP_LIFT_STATE_UNAVAILABLE : caps.state; + l->caps = caps; + l->caps.state = l->logged_state; // keep the logged value until caps_set_state compares + caps_set_state(l, st, nullptr); +} + +//! Destroy dead streams nobody still reads. Caller holds mtx (via @p lk). +static void +lift_reap(d3d11_lift *l, std::unique_lock &lk) +{ + for (auto it = l->streams.begin(); it != l->streams.end();) { + lift_stream &st = *it->second; + bool pinned = false; + for (uint32_t i = 0; i < U_LIFT_RING_SIZE; i++) { + pinned = pinned || st.mb.out_pins[i] > 0; + } + if (!st.dead || st.converting || pinned) { + ++it; + continue; + } + std::unique_ptr victim = std::move(it->second); + it = l->streams.erase(it); + lk.unlock(); + if (victim->dp_created && l->dp != nullptr) { + l->dp->lift_stream_destroy(l->dp, victim->dp_id); + } + stream_release_gpu(*victim); + U_LOG_I("[lift] stream %llu reaped (submitted=%llu converted=%llu dropped=%llu failed=%llu)", + (unsigned long long)victim->id, (unsigned long long)victim->mb.submitted, + (unsigned long long)victim->mb.converted, (unsigned long long)victim->mb.dropped, + (unsigned long long)victim->mb.failed); + victim.reset(); + lk.lock(); + it = l->streams.begin(); // the map may have changed while unlocked + } +} + +/*! + * Convert the pending frame of @p st. Called with @p lk HELD; drops it around + * every GPU / module call and returns with it held. + */ +static void +lift_convert_one(d3d11_lift *l, lift_stream &st, std::unique_lock &lk) +{ + int32_t in_slot = -1; + u_lift_frame_meta meta = {}; + if (!u_lift_mailbox_take_pending(&st.mb, os_monotonic_get_ns(), &in_slot, &meta)) { + return; + } + st.converting = true; + lift_in_slot &in = st.in[in_slot]; + const xrt_dp_lift_params params = in.params; + float vps[3 * XRT_DP_LIFT_MAX_EXPLICIT_VIEWPOINTS]; + memcpy(vps, in.viewpoints, sizeof(vps)); + uint32_t vp_floats = in.viewpoint_floats; + float active[4]; + memcpy(active, in.active, sizeof(active)); + const uint32_t w = meta.width, h = meta.height; + lk.unlock(); + + // TRACKED viewpoints are resolved HERE, per conversion, from the panel's + // predicted eyes, and always handed to the module explicitly: the lift DP + // has no tracker session of its own. The pair (a module spreads N views + // around it); nothing when no eyes are known yet. + if (vp_floats == 0 && l->eyes_fn != nullptr && st.info.mode != XRT_DP_LIFT_MODE_GAUSSIANS && + st.info.mode != XRT_DP_LIFT_MODE_DEPTH) { + struct xrt_eye_positions eyes = {}; + if (l->eyes_fn(l->eyes_ud, &eyes) && eyes.valid && eyes.count >= 2) { + uint32_t n = eyes.count == params.view_count ? eyes.count : 2; + n = n > XRT_DP_LIFT_MAX_EXPLICIT_VIEWPOINTS ? XRT_DP_LIFT_MAX_EXPLICIT_VIEWPOINTS : n; + for (uint32_t i = 0; i < n; i++) { + vps[3 * i + 0] = eyes.eyes[i].x; + vps[3 * i + 1] = eyes.eyes[i].y; + vps[3 * i + 2] = eyes.eyes[i].z; + } + vp_floats = 3 * n; + } + } + + bool ok = true; + // 1. The module's own stream, created lazily (module contract: lift thread only). + if (!st.dp_created && !st.dp_failed) { + if (l->dp->lift_stream_create(l->dp, &st.info, &st.dp_id)) { + st.dp_created = true; + } else { + st.dp_failed = true; + U_LOG_W("[lift] module refused stream %llu (mode=%u hint=%u) — its frames will never convert", + (unsigned long long)st.id, st.info.mode, st.info.content_hint); + } + } + ok = st.dp_created; + + // 2. Exact-size RGBA8 copy of the input on the lift device (the module + // contract: input is exactly w x h). + if (ok && (st.exact_tex == nullptr || st.exact_w != w || st.exact_h != h)) { + rel(st.exact_tex); + D3D11_TEXTURE2D_DESC td = {}; + td.Width = w; + td.Height = h; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = DXGI_FORMAT_R8G8B8A8_UNORM; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_DEFAULT; + td.BindFlags = D3D11_BIND_SHADER_RESOURCE; + ok = SUCCEEDED(l->lift_device->CreateTexture2D(&td, nullptr, &st.exact_tex)); + st.exact_w = ok ? w : 0; + st.exact_h = ok ? h : 0; + } + if (ok) { + HRESULT hr = in.lift_km->AcquireSync(0, 100); + ok = SUCCEEDED(hr) && hr != (HRESULT)WAIT_TIMEOUT; + if (ok) { + D3D11_BOX box = {0, 0, 0, w, h, 1}; + l->lift_context->CopySubresourceRegion(st.exact_tex, 0, 0, 0, 0, in.lift_tex, 0, &box); + in.lift_km->ReleaseSync(0); + } + } + + // 3. The conversion itself — synchronous, possibly long. + void *out_res = nullptr; + uint32_t ow = 0, oh = 0, of = 0; + const void *blob_bytes = nullptr; + size_t blob_size = 0; + uint32_t blob_format = 0; + const bool is_blob = st.info.mode == XRT_DP_LIFT_MODE_GAUSSIANS; + if (ok) { + if (is_blob) { + ok = xrt_display_processor_d3d11_has_lift_blob(l->dp) && + l->dp->lift_convert_blob(l->dp, st.dp_id, l->lift_context, st.exact_tex, w, h, ¶ms, + &blob_format, &blob_bytes, &blob_size) && + blob_bytes != nullptr && blob_size > 0; + } else { + ok = l->dp->lift_convert(l->dp, st.dp_id, l->lift_context, st.exact_tex, w, h, ¶ms, + vp_floats > 0 ? vps : nullptr, vp_floats, &out_res, &ow, &oh, &of) && + out_res != nullptr && ow > 0 && oh > 0; + } + } + + // 4. Copy the module's output (valid only until its next call) into the ring. + std::shared_ptr> blob_copy; + if (ok && is_blob) { + blob_copy = std::make_shared>((const uint8_t *)blob_bytes, + (const uint8_t *)blob_bytes + blob_size); + } + + lk.lock(); + u_lift_mailbox_finish_input(&st.mb, in_slot); + int32_t out_slot = -1; + if (ok) { + // Wait while a consumer still pins the only writable slot (one GPU-copy + // issue long); give up if the stream dies or we are stopping. + while (!u_lift_mailbox_begin_output(&st.mb, &out_slot)) { + if (st.dead || l->stop) { + ok = false; + break; + } + l->cv.wait_for(lk, std::chrono::milliseconds(5)); + } + } + lk.unlock(); + + if (ok && !is_blob) { + lift_out_slot &o = st.out[out_slot]; + uint32_t views = 1; + if (st.info.mode == XRT_DP_LIFT_MODE_SBS) { + views = 2; + } else if (st.info.mode == XRT_DP_LIFT_MODE_NVIEW) { + views = params.view_count >= 1 ? params.view_count : 2; + } + ok = out_slot_ensure(l, o, ow, oh, of); + if (ok) { + HRESULT hr = o.lift_km->AcquireSync(0, 100); + ok = SUCCEEDED(hr) && hr != (HRESULT)WAIT_TIMEOUT; + if (ok) { + D3D11_BOX box = {0, 0, 0, ow, oh, 1}; + l->lift_context->CopySubresourceRegion(o.lift_tex, 0, 0, 0, 0, + (ID3D11Resource *)out_res, 0, &box); + o.lift_km->ReleaseSync(0); + l->lift_context->Flush(); + o.view_count = views; + memcpy(o.active, active, sizeof(o.active)); + } + } + } else if (ok && is_blob) { + lift_out_slot &o = st.out[out_slot]; + o.blob = blob_copy; + o.blob_format = blob_format; + } + + lk.lock(); + st.converting = false; + if (out_slot >= 0) { + if (ok) { + u_lift_mailbox_publish_output(&st.mb, out_slot, &meta, os_monotonic_get_ns()); + } else { + u_lift_mailbox_abort_output(&st.mb, out_slot); + } + } else if (!ok) { + st.mb.failed++; + } + // Throttled INFO (never per frame at WARN). + uint64_t now = os_monotonic_get_ns(); + if (now - st.last_stats_log_ns > 5ull * 1000 * 1000 * 1000) { + st.last_stats_log_ns = now; + U_LOG_I( + "[lift] stream %llu: submitted=%llu converted=%llu dropped=%llu failed=%llu " + "latency last=%.1fms ema=%.1fms min=%.1fms max=%.1fms", + (unsigned long long)st.id, (unsigned long long)st.mb.submitted, (unsigned long long)st.mb.converted, + (unsigned long long)st.mb.dropped, (unsigned long long)st.mb.failed, st.mb.lat_last_ns / 1e6, + st.mb.lat_ema_ns / 1e6, st.mb.lat_min_ns / 1e6, st.mb.lat_max_ns / 1e6); + } + l->cv.notify_all(); +} + +static bool +any_dead(d3d11_lift *l) +{ + for (auto &kv : l->streams) { + if (kv.second->dead) { + return true; + } + } + return false; +} + +static bool +any_work(d3d11_lift *l) +{ + for (auto &kv : l->streams) { + const lift_stream &st = *kv.second; + if (st.dead || (st.priority != U_LIFT_PRIORITY_PAUSED && u_lift_mailbox_has_pending(&st.mb))) { + return true; + } + } + return false; +} + +static void +lift_thread_main(d3d11_lift *l) +{ + uint64_t last_poll_ns = 0; + std::unique_lock lk(l->mtx); + for (;;) { + l->cv.wait_for(lk, std::chrono::milliseconds(250), [&] { + // Frames only count as work once the module is READY (they wait, + // latest-wins, until then); dead streams always do (reaping). + return l->stop || (l->activate_requested && !l->activated) || + (l->activated && + (any_dead(l) || (l->caps.state == XRT_DP_LIFT_STATE_READY && any_work(l)))); + }); + if (l->stop) { + break; + } + if (l->activate_requested && !l->activated) { + lk.unlock(); + lift_activate(l); + lk.lock(); + continue; + } + if (!l->activated) { + continue; + } + + // A module warming up: poll ≤ 1 Hz until READY. + uint64_t now = os_monotonic_get_ns(); + if (l->dp != nullptr && l->caps.state != XRT_DP_LIFT_STATE_READY && + now - last_poll_ns > 1000000000ull) { + last_poll_ns = now; + lk.unlock(); + lift_poll_caps(l); + lk.lock(); + } + + lift_reap(l, lk); + + if (l->caps.state != XRT_DP_LIFT_STATE_READY || l->dp == nullptr) { + continue; // frames wait (latest wins) until the module is READY + } + + // A failed stream's frames can never convert: drain them so the wait + // predicate does not spin on them. + for (auto &kv : l->streams) { + lift_stream &st = *kv.second; + int32_t slot = -1; + while (st.dp_failed && u_lift_mailbox_take_pending(&st.mb, now, &slot, nullptr)) { + u_lift_mailbox_finish_input(&st.mb, slot); + st.mb.failed++; + } + } + + // One scheduling round (XrLiftPriorityDXR): HIGH streams with a new + // frame, one NORMAL round-robin, LOW every Nth round, PAUSED never. + std::vector entries; + entries.reserve(l->streams.size()); + for (auto &kv : l->streams) { // std::map: ascending id, as the planner wants + const lift_stream &st = *kv.second; + if (st.dead || st.dp_failed) { + continue; + } + entries.push_back({st.id, st.priority, u_lift_mailbox_has_pending(&st.mb)}); + } + uint64_t plan_ids[64]; + uint32_t n = u_lift_sched_plan(&l->sched, entries.data(), (uint32_t)entries.size(), plan_ids, 64); + if (n == 0) { + if (any_work(l)) { + // Only not-this-round LOW / dead-but-pinned left: back off briefly. + l->cv.wait_for(lk, std::chrono::milliseconds(5)); + } + continue; + } + for (uint32_t i = 0; i < n && !l->stop; i++) { + auto it = l->streams.find(plan_ids[i]); // re-look-up: unlocked between conversions + if (it == l->streams.end() || it->second->dead) { + continue; + } + lift_convert_one(l, *it->second, lk); + } + } + + // Shutdown: every stream, then the DP, on this thread. + for (auto &kv : l->streams) { + lift_stream &st = *kv.second; + if (st.dp_created && l->dp != nullptr) { + l->dp->lift_stream_destroy(l->dp, st.dp_id); + } + stream_release_gpu(st); + } + l->streams.clear(); + lk.unlock(); + if (l->dp != nullptr) { + xrt_display_processor_d3d11_destroy(&l->dp); + } +} + + +/* + * + * Public API. + * + */ + +static lift_stream * +find_live(d3d11_lift *l, uint64_t owner, uint64_t id) +{ + auto it = l->streams.find(id); + if (it == l->streams.end() || it->second->dead || it->second->owner != owner) { + return nullptr; + } + return it->second.get(); +} + +struct d3d11_lift * +d3d11_lift_create(ID3D11Device *svc_device, + ID3D11DeviceContext *svc_context, + std::mutex *svc_ctx_mutex, + void *lift_factory, + void *fallback_factory, + bool (*eyes_fn)(void *ud, struct xrt_eye_positions *out), + void *eyes_ud) +{ + if (svc_device == nullptr || svc_context == nullptr || svc_ctx_mutex == nullptr) { + return nullptr; + } + auto *l = new d3d11_lift(); + l->svc_device = svc_device; + l->svc_device->AddRef(); + l->svc_context = svc_context; + l->svc_context->AddRef(); + l->svc_ctx_mutex = svc_ctx_mutex; + l->lift_factory = (xrt_dp_factory_d3d11_fn_t)lift_factory; + l->fallback_factory = (xrt_dp_factory_d3d11_fn_t)fallback_factory; + l->eyes_fn = eyes_fn; + l->eyes_ud = eyes_ud; + (void)svc_device->QueryInterface(__uuidof(ID3D11Device1), (void **)&l->svc_device1); + (void)svc_context->QueryInterface(__uuidof(ID3D11DeviceContext4), (void **)&l->svc_context4); + + // Snapshot blit resources (service device). + ID3DBlob *b = nullptr; + ID3DBlob *err = nullptr; + bool ok = + SUCCEEDED(D3DCompile(k_snap_vs, strlen(k_snap_vs), nullptr, nullptr, nullptr, "main", "vs_5_0", 0, 0, &b, + &err)) && + SUCCEEDED(svc_device->CreateVertexShader(b->GetBufferPointer(), b->GetBufferSize(), nullptr, &l->snap_vs)); + rel(b); + rel(err); + ok = ok && + SUCCEEDED(D3DCompile(k_snap_ps, strlen(k_snap_ps), nullptr, nullptr, nullptr, "main", "ps_5_0", 0, 0, &b, + &err)) && + SUCCEEDED(svc_device->CreatePixelShader(b->GetBufferPointer(), b->GetBufferSize(), nullptr, &l->snap_ps)); + rel(b); + rel(err); + ok = ok && + SUCCEEDED(D3DCompile(k_snap_ps_scaled, strlen(k_snap_ps_scaled), nullptr, nullptr, nullptr, "main", + "ps_5_0", 0, 0, &b, &err)) && + SUCCEEDED( + svc_device->CreatePixelShader(b->GetBufferPointer(), b->GetBufferSize(), nullptr, &l->snap_ps_scaled)); + rel(b); + rel(err); + if (ok) { + D3D11_SAMPLER_DESC sd = {}; + sd.Filter = D3D11_FILTER_MIN_MAG_MIP_POINT; // 1:1 snapshot: exact texels + sd.AddressU = sd.AddressV = sd.AddressW = D3D11_TEXTURE_ADDRESS_CLAMP; + sd.MaxLOD = D3D11_FLOAT32_MAX; + ok = SUCCEEDED(svc_device->CreateSamplerState(&sd, &l->snap_sampler)); + if (ok) { + sd.Filter = D3D11_FILTER_MIN_MAG_LINEAR_MIP_POINT; // capped snapshot: box-filter taps + ok = SUCCEEDED(svc_device->CreateSamplerState(&sd, &l->snap_sampler_linear)); + } + } + + // Weave-rect snapshot cap, read once in the service process (ADR-042: + // the runtime owns the policy; the vendor's own input autoscaling only + // affects inference, not the view synthesis that dominates at 8K). + l->max_input_edge = u_lift_max_input_edge_parse(getenv("DXR_LIFT_MAX_INPUT_EDGE")); + if (l->max_input_edge != U_LIFT_MAX_INPUT_EDGE_DEFAULT) { + U_LOG_W("[lift] DXR_LIFT_MAX_INPUT_EDGE=%u%s", l->max_input_edge, + l->max_input_edge == 0 ? " (weave-rect snapshots uncapped)" : ""); + } + if (ok) { + D3D11_BUFFER_DESC bd = {}; + bd.ByteWidth = sizeof(snap_cb); + bd.Usage = D3D11_USAGE_DEFAULT; + bd.BindFlags = D3D11_BIND_CONSTANT_BUFFER; + ok = SUCCEEDED(svc_device->CreateBuffer(&bd, nullptr, &l->snap_cb)); + } + l->snap_ok = ok && l->svc_device1 != nullptr && l->svc_context4 != nullptr; + if (!l->snap_ok) { + U_LOG_E("[lift] snapshot pipeline init failed — lift will report UNAVAILABLE"); + } + + // Letterbox crop (optional: a failure here only disables the crop). + l->letterbox = u_lift_letterbox_parse(getenv("DXR_LIFT_LETTERBOX")); + if (l->snap_ok && l->letterbox) { + bool lok = SUCCEEDED(D3DCompile(k_lb_ps, strlen(k_lb_ps), nullptr, nullptr, nullptr, "main", "ps_5_0", 0, + 0, &b, &err)) && + SUCCEEDED(svc_device->CreatePixelShader(b->GetBufferPointer(), b->GetBufferSize(), nullptr, + &l->lb_ps)); + if (err != nullptr) { + U_LOG_E("[lift] letterbox shader: %s", (const char *)err->GetBufferPointer()); + } + rel(b); + rel(err); + if (lok) { + D3D11_TEXTURE2D_DESC td = {}; + td.Width = U_LIFT_LETTERBOX_BINS_MAX; + td.Height = 2; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = DXGI_FORMAT_R32_FLOAT; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_DEFAULT; + td.BindFlags = D3D11_BIND_RENDER_TARGET; + lok = SUCCEEDED(svc_device->CreateTexture2D(&td, nullptr, &l->lb_tex)) && + SUCCEEDED(svc_device->CreateRenderTargetView(l->lb_tex, nullptr, &l->lb_rtv)); + } + if (lok) { + D3D11_BUFFER_DESC bd = {}; + bd.ByteWidth = sizeof(lb_cb); + bd.Usage = D3D11_USAGE_DEFAULT; + bd.BindFlags = D3D11_BIND_CONSTANT_BUFFER; + lok = SUCCEEDED(svc_device->CreateBuffer(&bd, nullptr, &l->lb_cbuf)); + } + l->lb_ok = lok; + if (!lok) { + U_LOG_E("[lift] letterbox profile init failed — weave-rect snapshots are not cropped"); + } + } else if (!l->letterbox) { + U_LOG_W("[lift] DXR_LIFT_LETTERBOX=0 (weave-rect snapshots never cropped to the active area)"); + } + + xrt_dp_lift_caps_init(&l->caps); + l->caps.state = l->snap_ok ? XRT_DP_LIFT_STATE_ACTIVATING : XRT_DP_LIFT_STATE_UNAVAILABLE; + if (!l->snap_ok) { + l->activated = true; + } + l->thread = std::thread(lift_thread_main, l); + return l; +} + +void +d3d11_lift_destroy(struct d3d11_lift **lift_ptr) +{ + if (lift_ptr == nullptr || *lift_ptr == nullptr) { + return; + } + d3d11_lift *l = *lift_ptr; + { + std::lock_guard g(l->mtx); + l->stop = true; + } + l->cv.notify_all(); + if (l->thread.joinable()) { + l->thread.join(); + } + rel(l->lb_cbuf); + rel(l->lb_rtv); + rel(l->lb_tex); + rel(l->lb_ps); + rel(l->snap_cb); + rel(l->snap_sampler_linear); + rel(l->snap_sampler); + rel(l->snap_ps_scaled); + rel(l->snap_ps); + rel(l->snap_vs); + rel(l->lift_device1); + rel(l->lift_context); + rel(l->lift_device); + rel(l->svc_context4); + rel(l->svc_device1); + rel(l->svc_context); + rel(l->svc_device); + delete l; + *lift_ptr = nullptr; +} + +void +d3d11_lift_get_caps(struct d3d11_lift *l, struct xrt_dp_lift_caps *out) +{ + xrt_dp_lift_caps_init(out); + if (l == nullptr) { + return; + } + bool kick = false; + { + std::lock_guard g(l->mtx); + *out = l->caps; + out->struct_size = (uint32_t)sizeof(*out); + if (!l->activate_requested) { + l->activate_requested = true; + kick = true; + } + } + if (kick) { + l->cv.notify_all(); + } +} + +xrt_result_t +d3d11_lift_stream_create(struct d3d11_lift *l, + uint64_t owner, + const struct xrt_dp_lift_stream_info *info, + uint64_t *out_id) +{ + if (l == nullptr || info == nullptr || out_id == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + const uint32_t m = info->mode; + if (m != XRT_DP_LIFT_MODE_DEPTH && m != XRT_DP_LIFT_MODE_SBS && m != XRT_DP_LIFT_MODE_NVIEW && + m != XRT_DP_LIFT_MODE_GAUSSIANS) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + std::lock_guard g(l->mtx); + l->activate_requested = true; + if (l->caps.state == XRT_DP_LIFT_STATE_UNAVAILABLE && l->activated) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + if (l->caps.state == XRT_DP_LIFT_STATE_READY && (l->caps.modes & m) == 0) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + uint32_t max_streams = l->caps.max_streams > 0 ? l->caps.max_streams : 4; + if (l->live_streams >= max_streams) { + return XRT_ERROR_CLIENT_LIMIT_REACHED; + } + auto st = std::make_unique(); + st->id = ++l->next_stream_id; + st->owner = owner; + st->info = *info; + st->info.struct_size = (uint32_t)sizeof(st->info); + if (!(st->info.input_scale > 0.0f) || st->info.input_scale > 1.0f) { + st->info.input_scale = 1.0f; + } + u_lift_mailbox_init(&st->mb); + st->last_params.struct_size = (uint32_t)sizeof(st->last_params); + st->last_params.convergence = -1.0f; + st->last_params.strength = 1.0f; + st->last_params.inpaint = 1; + st->last_params.view_count = m == XRT_DP_LIFT_MODE_NVIEW ? 4 : 2; + *out_id = st->id; + U_LOG_W("[lift] stream %llu created (mode=%u hint=%u scale=%.2f, owner=%llu)", (unsigned long long)st->id, m, + info->content_hint, st->info.input_scale, (unsigned long long)owner); + l->streams[st->id] = std::move(st); + l->live_streams++; + l->cv.notify_all(); + return XRT_SUCCESS; +} + +void +d3d11_lift_stream_destroy(struct d3d11_lift *l, uint64_t owner, uint64_t id) +{ + if (l == nullptr) { + return; + } + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + if (st != nullptr) { + st->dead = true; + l->live_streams--; + U_LOG_W("[lift] stream %llu destroyed", (unsigned long long)id); + l->cv.notify_all(); + } +} + +void +d3d11_lift_release_owner(struct d3d11_lift *l, uint64_t owner) +{ + if (l == nullptr || owner == 0) { + return; + } + std::lock_guard g(l->mtx); + uint32_t n = 0; + for (auto &kv : l->streams) { + if (kv.second->owner == owner && !kv.second->dead) { + kv.second->dead = true; + l->live_streams--; + n++; + } + } + if (n > 0) { + U_LOG_W("[lift] client gone: %u stream(s) released", n); + l->cv.notify_all(); + } +} + +uint32_t +d3d11_lift_stream_mode(struct d3d11_lift *l, uint64_t owner, uint64_t id) +{ + if (l == nullptr) { + return 0; + } + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + return st != nullptr ? st->info.mode : 0; +} + +/*! + * Snapshot @p src region (@p x, @p y, @p w, @p h) into the top-left + * @p dst_w x @p dst_h of input slot @p s's service texture — 1:1 when the dims + * match, box-filtered down when the snapshot cap shrank them. Caller holds the + * service context mutex; takes the slot's keyed mutex itself. + */ +static bool +snapshot_blit(d3d11_lift *l, + lift_in_slot &s, + ID3D11ShaderResourceView *src, + uint32_t src_tw, + uint32_t src_th, + uint32_t x, + uint32_t y, + uint32_t w, + uint32_t h, + uint32_t dst_w, + uint32_t dst_h) +{ + const bool scaled = dst_w != w || dst_h != h; + HRESULT hr = s.svc_km->AcquireSync(0, 4); + if (FAILED(hr) || hr == (HRESULT)WAIT_TIMEOUT) { + return false; + } + ID3D11DeviceContext *ctx = l->svc_context; + snap_cb cb = {}; + cb.src_rect[0] = (float)x; + cb.src_rect[1] = (float)y; + cb.src_rect[2] = (float)w; + cb.src_rect[3] = (float)h; + cb.src_size[0] = (float)src_tw; + cb.src_size[1] = (float)src_th; + cb.dst_size[0] = (float)dst_w; + cb.dst_size[1] = (float)dst_h; + ctx->UpdateSubresource(l->snap_cb, 0, nullptr, &cb, 0, 0); + + // State this sequence completely: the context is shared (immediate_ctx_mutex + // serializes SEQUENCES, and every sequence states its own state). The fixed- + // function states are restored afterwards, because this runs in the MIDDLE + // of a weave sequence whose later draws may rely on what they set earlier. + ID3D11RasterizerState *prev_rs = nullptr; + ID3D11BlendState *prev_bs = nullptr; + float prev_bf[4] = {0, 0, 0, 0}; + UINT prev_mask = 0xffffffff; + ID3D11DepthStencilState *prev_ds = nullptr; + UINT prev_ref = 0; + ctx->RSGetState(&prev_rs); + ctx->OMGetBlendState(&prev_bs, prev_bf, &prev_mask); + ctx->OMGetDepthStencilState(&prev_ds, &prev_ref); + + D3D11_VIEWPORT vp = {}; + vp.Width = (float)dst_w; + vp.Height = (float)dst_h; + vp.MaxDepth = 1.0f; + ctx->RSSetViewports(1, &vp); + D3D11_RECT sc = {0, 0, (LONG)dst_w, (LONG)dst_h}; + ctx->RSSetScissorRects(1, &sc); + ctx->RSSetState(nullptr); + ctx->OMSetBlendState(nullptr, nullptr, 0xffffffff); + ctx->OMSetDepthStencilState(nullptr, 0); + ctx->OMSetRenderTargets(1, &s.svc_rtv, nullptr); + ctx->IASetPrimitiveTopology(D3D11_PRIMITIVE_TOPOLOGY_TRIANGLESTRIP); + ctx->IASetInputLayout(nullptr); + ctx->VSSetShader(l->snap_vs, nullptr, 0); + ctx->GSSetShader(nullptr, nullptr, 0); + ctx->PSSetShader(scaled ? l->snap_ps_scaled : l->snap_ps, nullptr, 0); + ctx->PSSetConstantBuffers(0, 1, &l->snap_cb); + ctx->PSSetSamplers(0, 1, scaled ? &l->snap_sampler_linear : &l->snap_sampler); + ctx->PSSetShaderResources(0, 1, &src); + ctx->Draw(4, 0); + ID3D11ShaderResourceView *null_srv = nullptr; + ctx->PSSetShaderResources(0, 1, &null_srv); + ID3D11RenderTargetView *null_rtv = nullptr; + ctx->OMSetRenderTargets(1, &null_rtv, nullptr); + ctx->RSSetState(prev_rs); + ctx->OMSetBlendState(prev_bs, prev_bf, prev_mask); + ctx->OMSetDepthStencilState(prev_ds, prev_ref); + rel(prev_rs); + rel(prev_bs); + rel(prev_ds); + + s.svc_km->ReleaseSync(0); + return true; +} + +/*! + * Letterbox: collect finished profile readbacks (never waits) into the stream's + * crop state, then measure this frame's rect into the next free readback. The + * crop therefore lags the picture by a few frames, which the settle hysteresis + * dwarfs anyway. Producer thread (service context mutex held); @p st is not + * reaped meanwhile (its input slot is WRITING). + */ +static void +letterbox_step(d3d11_lift *l, + lift_stream &st, + ID3D11ShaderResourceView *src, + uint32_t x, + uint32_t y, + uint32_t w, + uint32_t h) +{ + ID3D11DeviceContext *ctx = l->svc_context; + + // 1. Oldest-first collection, stop at the first still in flight. + for (uint32_t k = 0; k < LB_RING; k++) { + lift_lb_readback &r = st.lb_rb[(st.lb_next + k) % LB_RING]; + if (!r.pending) { + continue; + } + D3D11_MAPPED_SUBRESOURCE ms = {}; + HRESULT hr = ctx->Map(r.staging, 0, D3D11_MAP_READ, D3D11_MAP_FLAG_DO_NOT_WAIT, &ms); + if (hr == DXGI_ERROR_WAS_STILL_DRAWING) { + break; + } + r.pending = false; + if (FAILED(hr)) { + continue; + } + const float *rows = (const float *)ms.pData; + const float *cols = (const float *)((const uint8_t *)ms.pData + ms.RowPitch); + const bool changed = u_lift_letterbox_update(&st.lb, r.w, r.h, rows, r.nr, cols, r.nc); + // The row profile that decided it, top to bottom in 64 bands, each the + // band's max non-black fraction as a digit 0-9 — so a surprising crop + // change in the field says what was in the bars (captions, player UI). + char prof[65] = {0}; + if (changed) { + for (uint32_t k = 0; k < 64; k++) { + const uint32_t b0 = k * r.nr / 64, b1 = (k + 1) * r.nr / 64 > b0 ? (k + 1) * r.nr / 64 : b0 + 1; + float mx = 0.0f; + for (uint32_t i = b0; i < b1 && i < r.nr; i++) { + mx = rows[i] > mx ? rows[i] : mx; + } + const int d = (int)(mx * 10.0f); + prof[k] = (char)('0' + (d > 9 ? 9 : (d < 0 ? 0 : d))); + } + } + ctx->Unmap(r.staging, 0); + if (changed) { + const u_lift_crop &c = st.lb.committed; + if (u_lift_crop_active(&c)) { + U_LOG_W("[lift] stream %llu: letterbox crop %ux%u -> active %ux%u (bars top %u bottom %u left %u " + "right %u; bars woven flat) rows[%s]", + (unsigned long long)st.id, r.w, r.h, r.w - c.left - c.right, r.h - c.top - c.bottom, + c.top, c.bottom, c.left, c.right, prof); + } else { + U_LOG_W("[lift] stream %llu: letterbox crop off (%ux%u lifted whole) rows[%s]", + (unsigned long long)st.id, r.w, r.h, prof); + } + } + } + + // 2. Measure this frame into the next ring entry, if it is free. + lift_lb_readback &r = st.lb_rb[st.lb_next]; + if (r.pending) { + return; // GPU behind by LB_RING readbacks: skip a measurement, never stall + } + if (r.staging == nullptr) { + D3D11_TEXTURE2D_DESC td = {}; + td.Width = U_LIFT_LETTERBOX_BINS_MAX; + td.Height = 2; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = DXGI_FORMAT_R32_FLOAT; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_STAGING; + td.CPUAccessFlags = D3D11_CPU_ACCESS_READ; + if (FAILED(l->svc_device->CreateTexture2D(&td, nullptr, &r.staging))) { + return; + } + } + r.w = w; + r.h = h; + r.nr = h < U_LIFT_LETTERBOX_BINS_MAX ? h : U_LIFT_LETTERBOX_BINS_MAX; + r.nc = w < U_LIFT_LETTERBOX_BINS_MAX ? w : U_LIFT_LETTERBOX_BINS_MAX; + + lb_cb cb = {}; + cb.src_rect[0] = (float)x; + cb.src_rect[1] = (float)y; + cb.src_rect[2] = (float)w; + cb.src_rect[3] = (float)h; + cb.bins[0] = (float)r.nr; + cb.bins[1] = (float)r.nc; + ctx->UpdateSubresource(l->lb_cbuf, 0, nullptr, &cb, 0, 0); + + // Same state discipline as snapshot_blit: this runs mid weave sequence. + ID3D11RasterizerState *prev_rs = nullptr; + ID3D11BlendState *prev_bs = nullptr; + float prev_bf[4] = {0, 0, 0, 0}; + UINT prev_mask = 0xffffffff; + ID3D11DepthStencilState *prev_ds = nullptr; + UINT prev_ref = 0; + ctx->RSGetState(&prev_rs); + ctx->OMGetBlendState(&prev_bs, prev_bf, &prev_mask); + ctx->OMGetDepthStencilState(&prev_ds, &prev_ref); + + D3D11_VIEWPORT vp = {}; + vp.Width = (float)U_LIFT_LETTERBOX_BINS_MAX; + vp.Height = 2.0f; + vp.MaxDepth = 1.0f; + ctx->RSSetViewports(1, &vp); + D3D11_RECT sc = {0, 0, (LONG)U_LIFT_LETTERBOX_BINS_MAX, 2}; + ctx->RSSetScissorRects(1, &sc); + ctx->RSSetState(nullptr); + ctx->OMSetBlendState(nullptr, nullptr, 0xffffffff); + ctx->OMSetDepthStencilState(nullptr, 0); + ctx->OMSetRenderTargets(1, &l->lb_rtv, nullptr); + ctx->IASetPrimitiveTopology(D3D11_PRIMITIVE_TOPOLOGY_TRIANGLESTRIP); + ctx->IASetInputLayout(nullptr); + ctx->VSSetShader(l->snap_vs, nullptr, 0); + ctx->GSSetShader(nullptr, nullptr, 0); + ctx->PSSetShader(l->lb_ps, nullptr, 0); + ctx->PSSetConstantBuffers(0, 1, &l->lb_cbuf); + ctx->PSSetShaderResources(0, 1, &src); + ctx->Draw(4, 0); + ID3D11ShaderResourceView *null_srv = nullptr; + ctx->PSSetShaderResources(0, 1, &null_srv); + ID3D11RenderTargetView *null_rtv = nullptr; + ctx->OMSetRenderTargets(1, &null_rtv, nullptr); + ctx->RSSetState(prev_rs); + ctx->OMSetBlendState(prev_bs, prev_bf, prev_mask); + ctx->OMSetDepthStencilState(prev_ds, prev_ref); + rel(prev_rs); + rel(prev_bs); + rel(prev_ds); + + ctx->CopyResource(r.staging, l->lb_tex); + r.pending = true; + st.lb_next = (st.lb_next + 1) % LB_RING; +} + +//! Producer common path. Caller holds the service context mutex. +static xrt_result_t +submit_locked(d3d11_lift *l, + uint64_t owner, + uint64_t id, + ID3D11ShaderResourceView *src, + uint32_t src_tw, + uint32_t src_th, + uint32_t x, + uint32_t y, + uint32_t w, + uint32_t h, + int64_t source_time, + const xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_floats, + uint32_t max_input_edge, + bool letterbox, + uint64_t *out_frame_id) +{ + if (w == 0 || h == 0 || x + w > src_tw || y + h > src_th) { + return XRT_ERROR_OUTPUT_REQUEST_FAILURE; // unknown stream / bad extent: non-fatal + } + letterbox = letterbox && l->lb_ok; + // Service policy (ADR-042), in this order: letterbox crop to the active + // area, then the long-edge cap. The module synthesizes at INPUT resolution + // and the result is stretched back into the (active part of the) rect, so + // neither costs much on the panel and both save a lot in the module. + uint32_t cx = x, cy = y, cw = w, ch = h; + uint32_t dw = w, dh = h; + bool capped = false; + int32_t slot = -1; + lift_stream *st = nullptr; + { + std::lock_guard g(l->mtx); + st = find_live(l, owner, id); + if (st == nullptr) { + return XRT_ERROR_OUTPUT_REQUEST_FAILURE; // unknown stream / bad extent: non-fatal + } + if (letterbox && st->lb.w == w && st->lb.h == h && u_lift_crop_active(&st->lb.committed)) { + const u_lift_crop &c = st->lb.committed; + cx = x + c.left; + cy = y + c.top; + cw = w - c.left - c.right; + ch = h - c.top - c.bottom; + } + dw = cw; + dh = ch; + capped = u_lift_cap_dims(cw, ch, max_input_edge, &dw, &dh); + if (!l->activated || l->lift_device1 == nullptr) { + // Not up yet: there is no lift device to share the slot with. The + // frame is simply not taken (the caller submits again next frame). + l->activate_requested = true; + l->cv.notify_all(); + *out_frame_id = 0; + return XRT_SUCCESS; + } + if (!u_lift_mailbox_begin_submit(&st->mb, &slot)) { + return XRT_ERROR_WEAVE_REFUSED; + } + // Stream is not reaped while a slot is WRITING (not converting, but the + // reaper only runs for dead streams, and a dead stream is not found above). + if (params != nullptr) { + st->last_params = *params; + st->last_params.struct_size = (uint32_t)sizeof(st->last_params); + if (st->last_params.convergence > 1.0f) { + st->last_params.convergence = 1.0f; // [0,1]; negative = AUTO passes through + } + if (!(st->last_params.strength >= 0.0f)) { + st->last_params.strength = 1.0f; + } + if (st->last_params.view_count == 0) { + st->last_params.view_count = st->info.mode == XRT_DP_LIFT_MODE_NVIEW ? 4 : 2; + } + uint32_t n = viewpoint_floats; + if (n > 3 * XRT_DP_LIFT_MAX_EXPLICIT_VIEWPOINTS) { + n = 3 * XRT_DP_LIFT_MAX_EXPLICIT_VIEWPOINTS; + } + st->last_viewpoint_floats = viewpoints != nullptr ? n - n % 3 : 0; + if (st->last_viewpoint_floats > 0) { + memcpy(st->last_viewpoints, viewpoints, st->last_viewpoint_floats * sizeof(float)); + } + } + lift_in_slot &in = st->in[slot]; + in.params = st->last_params; + if (capped && in.params.focal_px > 0.0f) { + in.params.focal_px *= (float)dw / (float)cw; // focal is in INPUT pixels (a crop keeps the scale) + } + in.viewpoint_floats = st->last_viewpoint_floats; + memcpy(in.viewpoints, st->last_viewpoints, sizeof(in.viewpoints)); + in.active[0] = (float)(cx - x) / (float)w; + in.active[1] = (float)(cy - y) / (float)h; + in.active[2] = (float)(cx - x + cw) / (float)w; + in.active[3] = (float)(cy - y + ch) / (float)h; + } + + // Allocation + blit outside mtx: this slot is WRITING, nobody else touches it. + lift_in_slot &in = st->in[slot]; + bool ok = in_slot_ensure(l, in, dw, dh) && snapshot_blit(l, in, src, src_tw, src_th, cx, cy, cw, ch, dw, dh); + if (ok && letterbox) { + letterbox_step(l, *st, src, x, y, w, h); // measures the WHOLE rect, bars included + } + + std::lock_guard g(l->mtx); + if (!ok) { + u_lift_mailbox_abort_submit(&st->mb, slot); + return XRT_ERROR_WEAVE_REFUSED; + } + // WARN once per stream when the cap engages or its capped size changes + // (a rect resize), never per frame. + if (capped && (st->cap_logged_w != dw || st->cap_logged_h != dh)) { + st->cap_logged_w = dw; + st->cap_logged_h = dh; + U_LOG_W("[lift] stream %llu: snapshot %ux%u capped to %ux%u (DXR_LIFT_MAX_INPUT_EDGE=%u)", + (unsigned long long)id, cw, ch, dw, dh, max_input_edge); + } + // The mailbox carries the REAL input dims: the lift thread copies exactly + // dw x dh out of the slot and hands the module that size. + *out_frame_id = u_lift_mailbox_commit_submit(&st->mb, slot, source_time, os_monotonic_get_ns(), dw, dh); + l->cv.notify_all(); + return XRT_SUCCESS; +} + +xrt_result_t +d3d11_lift_submit_srv_locked(struct d3d11_lift *l, + uint64_t owner, + uint64_t id, + ID3D11ShaderResourceView *src, + uint32_t src_tw, + uint32_t src_th, + uint32_t x, + uint32_t y, + uint32_t w, + uint32_t h, + int64_t source_time, + const struct xrt_dp_lift_params *params, + uint64_t *out_frame_id) +{ + if (l == nullptr || !l->snap_ok || src == nullptr || out_frame_id == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + // Weave-rect snapshot: the service's size cap applies. + return submit_locked(l, owner, id, src, src_tw, src_th, x, y, w, h, source_time, params, nullptr, 0, + l->max_input_edge, l->letterbox, out_frame_id); +} + +xrt_result_t +d3d11_lift_submit_handle(struct d3d11_lift *l, + uint64_t owner, + uint64_t id, + HANDLE handle, + bool is_dxgi, + uint32_t w, + uint32_t h, + int64_t source_time, + const struct xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_floats, + uint64_t *out_frame_id) +{ + if (out_frame_id != nullptr) { + *out_frame_id = 0; + } + if (l == nullptr || !l->snap_ok || handle == nullptr || out_frame_id == nullptr) { + if (handle != nullptr && !is_dxgi) { + CloseHandle(handle); + } + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + + lift_stream *st = nullptr; + { + std::lock_guard g(l->mtx); + st = find_live(l, owner, id); + } + if (st == nullptr) { + if (!is_dxgi) { + CloseHandle(handle); + } + return XRT_ERROR_OUTPUT_REQUEST_FAILURE; // unknown stream / bad extent: non-fatal + } + + // Import cache, per stream (producer thread only — one IPC thread per + // connection). A duplicated NT handle is a new value every submit, so the + // cache compares kernel objects; a legacy DXGI handle is a stable global. + bool reuse = false; + if (st->imp_tex != nullptr && st->imp_dxgi == is_dxgi) { + reuse = is_dxgi ? (st->imp_handle == handle) : same_kernel_object(st->imp_handle, handle); + } + if (reuse) { + if (!is_dxgi) { + CloseHandle(handle); + } + } else { + rel(st->imp_km); + rel(st->imp_srv); + rel(st->imp_tex); + if (!st->imp_dxgi) { + close_handle(st->imp_handle); + } + st->imp_handle = nullptr; + HRESULT hr; + if (is_dxgi) { + hr = + l->svc_device->OpenSharedResource(handle, __uuidof(ID3D11Texture2D), (void **)&st->imp_tex); + } else { + hr = l->svc_device1->OpenSharedResource1(handle, __uuidof(ID3D11Texture2D), + (void **)&st->imp_tex); + } + D3D11_TEXTURE2D_DESC d = {}; + if (SUCCEEDED(hr)) { + st->imp_tex->GetDesc(&d); + D3D11_SHADER_RESOURCE_VIEW_DESC sd = {}; + sd.Format = typed_srv_format(d.Format); + sd.ViewDimension = D3D11_SRV_DIMENSION_TEXTURE2D; + sd.Texture2D.MipLevels = 1; + hr = l->svc_device->CreateShaderResourceView(st->imp_tex, &sd, &st->imp_srv); + } + if (FAILED(hr)) { + U_LOG_E("[lift] input OpenSharedResource(%s) failed: 0x%08lx", is_dxgi ? "DXGI" : "NT", + (unsigned long)hr); + rel(st->imp_srv); + rel(st->imp_tex); + if (!is_dxgi) { + CloseHandle(handle); + } + return XRT_ERROR_WEAVE_REFUSED; + } + (void)st->imp_tex->QueryInterface(__uuidof(IDXGIKeyedMutex), (void **)&st->imp_km); + st->imp_handle = handle; // kept open (NT) for the identity compare + st->imp_dxgi = is_dxgi; + st->imp_w = d.Width; + st->imp_h = d.Height; + U_LOG_I("[lift] stream %llu input import cached (%s, %ux%u)", (unsigned long long)id, + is_dxgi ? "DXGI" : "NT", d.Width, d.Height); + } + + // The caller's keyed mutex: 4 ms, the budget every weave-side acquire uses. + bool acquired = false; + if (st->imp_km != nullptr) { + HRESULT hr = st->imp_km->AcquireSync(0, 4); + if (FAILED(hr) || hr == (HRESULT)WAIT_TIMEOUT) { + return XRT_ERROR_WEAVE_REFUSED; + } + acquired = true; + } + xrt_result_t xret; + { + std::lock_guard ctx_lock(*l->svc_ctx_mutex); + // An app's explicit frame: its size and content are the app's choice — + // never capped, never letterbox-cropped. + xret = submit_locked(l, owner, id, st->imp_srv, st->imp_w, st->imp_h, 0, 0, w, h, source_time, params, + viewpoints, viewpoint_floats, /*max_input_edge*/ 0, /*letterbox*/ false, out_frame_id); + } + if (acquired) { + st->imp_km->ReleaseSync(0); + } + return xret; +} + +bool +d3d11_lift_pin_latest(struct d3d11_lift *l, uint64_t owner, uint64_t id, struct d3d11_lift_pin *out) +{ + *out = d3d11_lift_pin{}; + if (l == nullptr) { + return false; + } + int32_t slot = -1; + lift_out_slot *o = nullptr; + { + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + if (st == nullptr || st->info.mode == XRT_DP_LIFT_MODE_GAUSSIANS || + !u_lift_mailbox_pin_latest(&st->mb, false, &slot, nullptr)) { + return false; + } + o = &st->out[slot]; + if (o->svc_srv == nullptr || o->svc_km == nullptr) { + u_lift_mailbox_unpin(&st->mb, slot); + return false; + } + } + // Pinned: the lift thread will not write this slot, so its mutex is free. + HRESULT hr = o->svc_km->AcquireSync(0, 4); + if (FAILED(hr) || hr == (HRESULT)WAIT_TIMEOUT) { + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + if (st != nullptr) { + u_lift_mailbox_unpin(&st->mb, slot); + } + return false; + } + out->lift = l; + out->stream_id = id; + out->slot = slot; + out->km = o->svc_km; + out->srv = o->svc_srv; + out->width = o->w; + out->height = o->h; + out->view_count = o->view_count; + memcpy(out->active, o->active, sizeof(out->active)); + out->valid = true; + return true; +} + +void +d3d11_lift_unpin(struct d3d11_lift_pin *pin) +{ + if (pin == nullptr || !pin->valid || pin->lift == nullptr) { + return; + } + pin->km->ReleaseSync(0); + d3d11_lift *l = pin->lift; + { + std::lock_guard g(l->mtx); + auto it = l->streams.find(pin->stream_id); // dead streams stay mapped until unpinned + if (it != l->streams.end()) { + u_lift_mailbox_unpin(&it->second->mb, pin->slot); + } + } + l->cv.notify_all(); + *pin = d3d11_lift_pin{}; +} + +xrt_result_t +d3d11_lift_acquire_result( + struct d3d11_lift *l, uint64_t owner, uint64_t id, bool *out_ready, struct d3d11_lift_result_info *out) +{ + *out_ready = false; + *out = d3d11_lift_result_info{}; + if (l == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + int32_t slot = -1; + u_lift_frame_meta meta = {}; + lift_stream *st = nullptr; + lift_out_slot *o = nullptr; + { + std::lock_guard g(l->mtx); + st = find_live(l, owner, id); + if (st == nullptr) { + return XRT_ERROR_OUTPUT_REQUEST_FAILURE; // unknown stream / bad extent: non-fatal + } + if (st->info.mode == XRT_DP_LIFT_MODE_GAUSSIANS) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + if (!u_lift_mailbox_pin_latest(&st->mb, true, &slot, &meta)) { + return XRT_SUCCESS; // nothing newer: NOT READY + } + o = &st->out[slot]; + } + + auto unpin = [&]() { + std::lock_guard g(l->mtx); + auto it = l->streams.find(id); + if (it != l->streams.end()) { + u_lift_mailbox_unpin(&it->second->mb, slot); + } + l->cv.notify_all(); + }; + + // Export texture: sized to the result, re-created on size / format change. + bool realloc = false; + if (st->exp_tex == nullptr || st->exp_w != o->w || st->exp_h != o->h || st->exp_format != o->format) { + rel(st->exp_tex); + close_handle(st->exp_handle); + D3D11_TEXTURE2D_DESC td = {}; + td.Width = o->w; + td.Height = o->h; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = (DXGI_FORMAT)o->format; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_DEFAULT; + td.BindFlags = D3D11_BIND_SHADER_RESOURCE; + td.MiscFlags = D3D11_RESOURCE_MISC_SHARED_NTHANDLE | D3D11_RESOURCE_MISC_SHARED; + HRESULT hr = l->svc_device->CreateTexture2D(&td, nullptr, &st->exp_tex); + IDXGIResource1 *r1 = nullptr; + if (SUCCEEDED(hr)) { + hr = st->exp_tex->QueryInterface(__uuidof(IDXGIResource1), (void **)&r1); + } + if (SUCCEEDED(hr)) { + hr = r1->CreateSharedHandle(nullptr, DXGI_SHARED_RESOURCE_READ | DXGI_SHARED_RESOURCE_WRITE, + nullptr, &st->exp_handle); + } + rel(r1); + if (SUCCEEDED(hr) && st->exp_fence == nullptr) { + ID3D11Device5 *d5 = nullptr; + hr = l->svc_device->QueryInterface(__uuidof(ID3D11Device5), (void **)&d5); + if (SUCCEEDED(hr)) { + hr = d5->CreateFence(0, D3D11_FENCE_FLAG_SHARED, __uuidof(ID3D11Fence), + (void **)&st->exp_fence); + } + rel(d5); + if (SUCCEEDED(hr)) { + hr = st->exp_fence->CreateSharedHandle(nullptr, GENERIC_ALL, nullptr, + &st->exp_fence_handle); + } + } + if (FAILED(hr)) { + U_LOG_E("[lift] export texture %ux%u fmt=%u create failed: 0x%08lx", o->w, o->h, o->format, + (unsigned long)hr); + rel(st->exp_tex); + close_handle(st->exp_handle); + unpin(); + return XRT_ERROR_WEAVE_REFUSED; + } + st->exp_w = o->w; + st->exp_h = o->h; + st->exp_format = o->format; + realloc = true; + U_LOG_W("[lift] stream %llu export texture %ux%u fmt=%u + fence ready", (unsigned long long)id, o->w, + o->h, o->format); + } + + HRESULT hr = o->svc_km->AcquireSync(0, 4); + if (FAILED(hr) || hr == (HRESULT)WAIT_TIMEOUT) { + unpin(); + return XRT_ERROR_WEAVE_REFUSED; + } + { + std::lock_guard ctx_lock(*l->svc_ctx_mutex); + l->svc_context->CopyResource(st->exp_tex, o->svc_tex); + st->exp_fence_value++; + l->svc_context4->Signal(st->exp_fence, st->exp_fence_value); + l->svc_context->Flush(); + } + o->svc_km->ReleaseSync(0); + + out->frame_id = meta.frame_id; + out->source_time = meta.source_time; + out->fence_value = st->exp_fence_value; + out->latency_ns = meta.done_ns >= meta.submit_ns ? meta.done_ns - meta.submit_ns : 0; + out->width = o->w; + out->height = o->h; + out->format = o->format; + out->view_count = o->view_count; + out->output_realloc = realloc; + *out_ready = true; + unpin(); + return XRT_SUCCESS; +} + +bool +d3d11_lift_export_output(struct d3d11_lift *l, + uint64_t owner, + uint64_t id, + HANDLE *out_handle, + uint32_t *out_w, + uint32_t *out_h, + uint32_t *out_format) +{ + if (l == nullptr) { + return false; + } + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + if (st == nullptr || st->exp_handle == nullptr) { + return false; + } + *out_handle = st->exp_handle; + *out_w = st->exp_w; + *out_h = st->exp_h; + *out_format = st->exp_format; + return true; +} + +bool +d3d11_lift_export_fence(struct d3d11_lift *l, uint64_t owner, uint64_t id, HANDLE *out_handle) +{ + if (l == nullptr) { + return false; + } + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + if (st == nullptr || st->exp_fence_handle == nullptr) { + return false; + } + *out_handle = st->exp_fence_handle; + return true; +} + +xrt_result_t +d3d11_lift_acquire_blob(struct d3d11_lift *l, + uint64_t owner, + uint64_t id, + uint64_t capacity, + bool *out_ready, + struct d3d11_lift_blob_info *info, + uint8_t **out_bytes) +{ + *out_ready = false; + *info = d3d11_lift_blob_info{}; + *out_bytes = nullptr; + if (l == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + std::shared_ptr> blob; + u_lift_frame_meta meta = {}; + uint32_t format = 0; + { + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + if (st == nullptr) { + return XRT_ERROR_OUTPUT_REQUEST_FAILURE; // unknown stream / bad extent: non-fatal + } + if (st->info.mode != XRT_DP_LIFT_MODE_GAUSSIANS) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + if (!st->blob_latched) { + int32_t slot = -1; + if (!u_lift_mailbox_pin_latest(&st->mb, true, &slot, &st->blob_latched_meta)) { + return XRT_SUCCESS; // nothing newer + } + st->blob_latched = st->out[slot].blob; + st->blob_latched_format = st->out[slot].blob_format; + u_lift_mailbox_unpin(&st->mb, slot); // the shared_ptr keeps the bytes alive + if (!st->blob_latched) { + return XRT_SUCCESS; + } + } + blob = st->blob_latched; + meta = st->blob_latched_meta; + format = st->blob_latched_format; + if (capacity >= blob->size()) { + st->blob_latched.reset(); // delivered: the latch is consumed + } + } + info->frame_id = meta.frame_id; + info->source_time = meta.source_time; + info->format = format; + info->byte_count = blob->size(); + *out_ready = true; + if (capacity >= blob->size() && !blob->empty()) { + *out_bytes = (uint8_t *)malloc(blob->size()); + if (*out_bytes == nullptr) { + return XRT_ERROR_ALLOCATION; + } + memcpy(*out_bytes, blob->data(), blob->size()); + } + return XRT_SUCCESS; +} + +xrt_result_t +d3d11_lift_set_priority(struct d3d11_lift *l, uint64_t owner, uint64_t id, uint32_t priority) +{ + if (l == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + if (priority > U_LIFT_PRIORITY_HIGH) { + return XRT_ERROR_OUTPUT_REQUEST_FAILURE; + } + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + if (st == nullptr) { + return XRT_ERROR_OUTPUT_REQUEST_FAILURE; // unknown stream: non-fatal + } + if (st->priority != priority) { + U_LOG_I("[lift] stream %llu priority %u -> %u", (unsigned long long)id, st->priority, priority); + st->priority = priority; + l->cv.notify_all(); // a stream un-paused may have a frame waiting + } + return XRT_SUCCESS; +} + +xrt_result_t +d3d11_lift_get_stats(struct d3d11_lift *l, uint64_t owner, uint64_t id, struct xrt_lift_stream_stats *out) +{ + memset(out, 0, sizeof(*out)); + if (l == nullptr) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + std::lock_guard g(l->mtx); + lift_stream *st = find_live(l, owner, id); + if (st == nullptr) { + return XRT_ERROR_OUTPUT_REQUEST_FAILURE; + } + out->priority = st->priority; + out->submitted = st->mb.submitted; + out->converted = st->mb.converted; + out->dropped = st->mb.dropped; + out->failed = st->mb.failed; + out->latency_last_ns = st->mb.lat_last_ns; + out->latency_avg_ns = st->mb.lat_ema_ns; + out->latency_min_ns = st->mb.lat_min_ns; + out->latency_max_ns = st->mb.lat_max_ns; + out->rate_hz = u_lift_mailbox_rate_hz(&st->mb); + return XRT_SUCCESS; +} diff --git a/src/xrt/compositor/d3d11_service/d3d11_lift.h b/src/xrt/compositor/d3d11_service/d3d11_lift.h new file mode 100644 index 000000000..f0669050d --- /dev/null +++ b/src/xrt/compositor/d3d11_service/d3d11_lift.h @@ -0,0 +1,248 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief XR_DXR_lift (ADR-042) on the D3D11 service: the lift thread, its + * device, the vendor module's display processor, and every stream. + * + * ## Shape + * + * One @ref d3d11_lift per service system, created lazily on the first lift + * call. It owns: + * + * - the LIFT THREAD, the only thread that ever touches the lift display + * processor (the vendor module's contract, xrt_display_processor_d3d11.h); + * - a dedicated LIFT DEVICE on the service's adapter, with ID3D11Multithread + * protection on its immediate context (a module may flush / signal it from + * its own worker). A conversion takes milliseconds to seconds, so it must + * never run on the service's shared immediate context — that context is + * the one every weave, commit and render sequence serializes on; + * - per stream: a latest-wins MAILBOX (two input slots) and an output RING + * (two slots), both shared textures with keyed mutexes that cross between + * the service device and the lift device, driven by the platform-neutral + * state machine in util/u_lift_mailbox.h; + * - per stream, for xrAcquireLiftResultDXR: a client-facing export texture + + * fence on the service device (the weave output pattern). + * + * ## Threads and locks + * + * - `mtx` (inside d3d11_lift) guards the stream table and every mailbox. It is + * never held across a GPU wait, a keyed-mutex acquire, or a module call. + * - The service's `immediate_ctx_mutex` is passed in and taken ONLY around + * draws/copies on the service context (snapshot blit, export copy). Order: + * immediate_ctx_mutex → mtx, never the reverse. The lift thread never takes + * it. + * - Producers (IPC threads) never wait for the module: a submit is one + * snapshot blit on the service context. Consumers (the weave, an acquire) + * pin a result slot for the duration of one GPU copy issue. + * + * @ingroup comp_d3d11_service + */ + +#pragma once + +#include "xrt/xrt_results.h" +#include "xrt/xrt_dp_lift.h" +#include "xrt/xrt_display_metrics.h" +#include "xrt/xrt_lift.h" + +#define WIN32_LEAN_AND_MEAN +#include +#include + +#include +#include + +struct d3d11_lift; + +//! One acquired texture result (xrAcquireLiftResultDXR). +struct d3d11_lift_result_info +{ + uint64_t frame_id; + int64_t source_time; + uint64_t fence_value; + uint64_t latency_ns; + uint32_t width; + uint32_t height; + uint32_t format; //!< DXGI_FORMAT + uint32_t view_count; + bool output_realloc; //!< the export texture changed: the caller must re-import it +}; + +//! One acquired blob result (xrAcquireLiftBlobDXR). +struct d3d11_lift_blob_info +{ + uint64_t frame_id; + int64_t source_time; + uint32_t format; //!< XRT_DP_LIFT_BLOB_* + uint64_t byte_count; +}; + +/*! + * A result pinned for reading on the SERVICE device (the weave's use). While + * pinned, the lift thread will not overwrite it and its keyed mutex is held. + */ +struct d3d11_lift_pin +{ + struct d3d11_lift *lift; + uint64_t stream_id; + int32_t slot; + IDXGIKeyedMutex *km; + ID3D11ShaderResourceView *srv; //!< service-device SRV of the result + uint32_t width; //!< whole result (all views) + uint32_t height; + uint32_t view_count; + //! The part of the lifted rect this result covers (letterbox crop): + //! x0, y0, x1, y1 normalised to the rect as it was snapshotted. Outside it + //! (the bars) the caller weaves the 2D input FLAT, identical in every view. + float active[4]; + bool valid; +}; + +/*! + * Create the lift module for a service. Cheap: the lift device and the vendor + * display processor are brought up on the lift thread, on first demand. + * + * @param svc_device the service's render device (ID3D11Device1-capable). + * @param svc_context its immediate context. + * @param svc_ctx_mutex the service's immediate_ctx_mutex. + * @param lift_factory the plug-in's LIFT-ONLY factory + * (xrt_plugin_iface::create_dp_d3d11_lift), or NULL. + * @param fallback_factory its ordinary D3D11 factory, used with a NULL window + * when @p lift_factory is NULL. Both NULL ⟹ UNAVAILABLE. + * @param eyes_fn returns the panel's predicted tracked eyes (display + * space, metres) — the viewpoints the lift thread passes + * to every TRACKED conversion. Called from the lift + * thread with no lift lock held. May be NULL. + * @param eyes_ud user data for @p eyes_fn. + */ +struct d3d11_lift * +d3d11_lift_create(ID3D11Device *svc_device, + ID3D11DeviceContext *svc_context, + std::mutex *svc_ctx_mutex, + void *lift_factory, + void *fallback_factory, + bool (*eyes_fn)(void *ud, struct xrt_eye_positions *out), + void *eyes_ud); + +//! Stop the lift thread, destroy every stream and the lift DP/device. +void +d3d11_lift_destroy(struct d3d11_lift **lift_ptr); + +//! Current caps (cached; never blocks). Kicks activation on first call. +void +d3d11_lift_get_caps(struct d3d11_lift *lift, struct xrt_dp_lift_caps *out); + +//! Create a stream owned by @p owner (an IPC connection token). +xrt_result_t +d3d11_lift_stream_create(struct d3d11_lift *lift, + uint64_t owner, + const struct xrt_dp_lift_stream_info *info, + uint64_t *out_id); + +void +d3d11_lift_stream_destroy(struct d3d11_lift *lift, uint64_t owner, uint64_t id); + +//! Destroy every stream @p owner created (client teardown). +void +d3d11_lift_release_owner(struct d3d11_lift *lift, uint64_t owner); + +//! The stream's mode bit, 0 when @p id is not @p owner's live stream. +uint32_t +d3d11_lift_stream_mode(struct d3d11_lift *lift, uint64_t owner, uint64_t id); + +/*! + * xrSubmitLiftFrameDXR: open the caller's shared texture (cached per stream), + * acquire its keyed mutex (4 ms), snapshot @p w x @p h into the mailbox, post. + * Takes the service context mutex internally — call WITHOUT it held. + * The handle is the service's to close (a duplicated NT handle) unless + * @p is_dxgi. @p params NULL = the stream's last parameters. + */ +xrt_result_t +d3d11_lift_submit_handle(struct d3d11_lift *lift, + uint64_t owner, + uint64_t id, + HANDLE handle, + bool is_dxgi, + uint32_t w, + uint32_t h, + int64_t source_time, + const struct xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_floats, + uint64_t *out_frame_id); + +/*! + * The weave path's submit: snapshot region (@p x, @p y, @p w, @p h) of @p src + * (an SRV on the service device, @p src_tw x @p src_th) into the mailbox. + * The CALLER holds the service context mutex and whatever keyed mutex guards + * @p src. The snapshot's long edge is capped at DXR_LIFT_MAX_INPUT_EDGE + * (default 1920, 0 = off; u_lift_cap_dims) — the result is stretched back into + * the rect, so consumers must sample it, never assume result size == rect size. + */ +xrt_result_t +d3d11_lift_submit_srv_locked(struct d3d11_lift *lift, + uint64_t owner, + uint64_t id, + ID3D11ShaderResourceView *src, + uint32_t src_tw, + uint32_t src_th, + uint32_t x, + uint32_t y, + uint32_t w, + uint32_t h, + int64_t source_time, + const struct xrt_dp_lift_params *params, + uint64_t *out_frame_id); + +//! Pin the stream's latest result for a service-device read (the weave). +bool +d3d11_lift_pin_latest(struct d3d11_lift *lift, uint64_t owner, uint64_t id, struct d3d11_lift_pin *out); + +void +d3d11_lift_unpin(struct d3d11_lift_pin *pin); + +/*! + * xrAcquireLiftResultDXR: copy the newest result not yet acquired into the + * stream's export texture, signal its fence. @p out_ready false = nothing new. + * Takes the service context mutex internally — call WITHOUT it held. + */ +xrt_result_t +d3d11_lift_acquire_result( + struct d3d11_lift *lift, uint64_t owner, uint64_t id, bool *out_ready, struct d3d11_lift_result_info *out); + +//! The export texture's NT handle (service-owned; the IPC layer duplicates it). +bool +d3d11_lift_export_output(struct d3d11_lift *lift, + uint64_t owner, + uint64_t id, + HANDLE *out_handle, + uint32_t *out_w, + uint32_t *out_h, + uint32_t *out_format); + +bool +d3d11_lift_export_fence(struct d3d11_lift *lift, uint64_t owner, uint64_t id, HANDLE *out_handle); + +/*! + * xrAcquireLiftBlobDXR with the two-call latch (see XrLiftBlobDXR). On + * delivery (@p capacity >= byte_count) @p out_bytes receives a malloc'd copy + * the caller frees; otherwise it stays NULL and the blob stays latched (the + * caller reports XR_ERROR_SIZE_INSUFFICIENT for 0 < capacity < byte_count). + */ +xrt_result_t +d3d11_lift_acquire_blob(struct d3d11_lift *lift, + uint64_t owner, + uint64_t id, + uint64_t capacity, + bool *out_ready, + struct d3d11_lift_blob_info *info, + uint8_t **out_bytes); + +//! Set a stream's scheduling priority (XrLiftPriorityDXR: 0 paused .. 3 high). +xrt_result_t +d3d11_lift_set_priority(struct d3d11_lift *lift, uint64_t owner, uint64_t id, uint32_t priority); + +//! A stream's counters + effective conversion rate. +xrt_result_t +d3d11_lift_get_stats(struct d3d11_lift *lift, uint64_t owner, uint64_t id, struct xrt_lift_stream_stats *out); diff --git a/src/xrt/drivers/CMakeLists.txt b/src/xrt/drivers/CMakeLists.txt index 4475fc62e..1f412e877 100644 --- a/src/xrt/drivers/CMakeLists.txt +++ b/src/xrt/drivers/CMakeLists.txt @@ -101,7 +101,10 @@ endif() # D3D11 display processor (HLSL shaders compiled at runtime) if(WIN32) list(APPEND SIM_DISPLAY_SOURCES - sim_display/sim_display_processor_d3d11.cpp) + sim_display/sim_display_processor_d3d11.cpp + # ADR-042: env-gated FAKE lift module (SIM_DISPLAY_FAKE_LIFT=1) + sim_display/sim_display_lift_d3d11.cpp + sim_display/sim_display_fake_ply.c) endif() # D3D12 display processor (HLSL shaders compiled at runtime into PSOs) diff --git a/src/xrt/drivers/sim_display/sim_display_fake_ply.c b/src/xrt/drivers/sim_display/sim_display_fake_ply.c new file mode 100644 index 000000000..f9cd5c66f --- /dev/null +++ b/src/xrt/drivers/sim_display/sim_display_fake_ply.c @@ -0,0 +1,109 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief sim_display's fake Gaussian-splat PLY (see sim_display_fake_ply.h). + * @ingroup drv_sim_display + */ + +#include "sim_display_fake_ply.h" + +#include +#include +#include + +static const char *const k_props[SIM_FAKE_PLY_FLOATS_PER_SPLAT] = { + "x", "y", "z", "nx", "ny", "nz", "f_dc_0", "f_dc_1", "f_dc_2", + "opacity", "scale_0", "scale_1", "scale_2", "rot_0", "rot_1", "rot_2", "rot_3", +}; + +static size_t +header_write(char *buf, size_t cap) +{ + int n = snprintf(buf, cap, "ply\nformat binary_little_endian 1.0\nelement vertex %d\n", SIM_FAKE_PLY_SPLATS); + size_t len = n > 0 ? (size_t)n : 0; + for (int i = 0; i < SIM_FAKE_PLY_FLOATS_PER_SPLAT; i++) { + n = snprintf(buf != NULL && len < cap ? buf + len : NULL, buf != NULL && len < cap ? cap - len : 0, + "property float %s\n", k_props[i]); + len += n > 0 ? (size_t)n : 0; + } + n = snprintf(buf != NULL && len < cap ? buf + len : NULL, buf != NULL && len < cap ? cap - len : 0, + "end_header\n"); + len += n > 0 ? (size_t)n : 0; + return len; +} + +static void +put_f32(uint8_t *dst, float v) +{ + // PLY binary_little_endian: every supported host is little-endian, but be + // explicit so the format never depends on it. + uint32_t u; + memcpy(&u, &v, sizeof(u)); + dst[0] = (uint8_t)(u & 0xff); + dst[1] = (uint8_t)((u >> 8) & 0xff); + dst[2] = (uint8_t)((u >> 16) & 0xff); + dst[3] = (uint8_t)((u >> 24) & 0xff); +} + +size_t +sim_fake_ply_write(uint8_t *dst, size_t cap, const uint8_t *rgba, uint32_t w, uint32_t h, uint32_t row_pitch) +{ + char hdr[1024]; + size_t hlen = header_write(hdr, sizeof(hdr)); + size_t need = hlen + (size_t)SIM_FAKE_PLY_SPLATS * SIM_FAKE_PLY_FLOATS_PER_SPLAT * 4u; + if (dst == NULL || cap < need) { + return need; + } + memcpy(dst, hdr, hlen); + uint8_t *p = dst + hlen; + + const float aspect = (w > 0 && h > 0) ? (float)w / (float)h : 1.0f; + const float sh_c0 = 0.28209479177387814f; // 3DGS: colour = 0.5 + SH_C0 * f_dc + for (int layer = 0; layer < 2; layer++) { + const float z = layer == 0 ? 0.0f : 0.5f; // back layer sits behind + const float shade = layer == 0 ? 1.0f : 0.45f; // and darker + const float scale = logf(layer == 0 ? 0.03f : 0.05f); + for (int gy = 0; gy < SIM_FAKE_PLY_GRID_Y; gy++) { + for (int gx = 0; gx < SIM_FAKE_PLY_GRID_X; gx++) { + float u = ((float)gx + 0.5f) / (float)SIM_FAKE_PLY_GRID_X; + float v = ((float)gy + 0.5f) / (float)SIM_FAKE_PLY_GRID_Y; + float rgb[3] = {0.5f, 0.5f, 0.5f}; + if (rgba != NULL && w > 0 && h > 0) { + uint32_t px = (uint32_t)(u * (float)w); + uint32_t py = (uint32_t)(v * (float)h); + px = px >= w ? w - 1 : px; + py = py >= h ? h - 1 : py; + const uint8_t *s = rgba + (size_t)py * row_pitch + (size_t)px * 4u; + rgb[0] = (float)s[0] / 255.0f; + rgb[1] = (float)s[1] / 255.0f; + rgb[2] = (float)s[2] / 255.0f; + } + float f[SIM_FAKE_PLY_FLOATS_PER_SPLAT] = { + (u - 0.5f) * aspect, // x + (0.5f - v), // y (up) + z, // z + 0.0f, + 0.0f, + 0.0f, // normal + (rgb[0] * shade - 0.5f) / sh_c0, // f_dc + (rgb[1] * shade - 0.5f) / sh_c0, + (rgb[2] * shade - 0.5f) / sh_c0, + 2.0f, // opacity logit (~0.88) + scale, + scale, + scale, // log scale + 1.0f, + 0.0f, + 0.0f, + 0.0f, // rotation (w x y z), identity + }; + for (int k = 0; k < SIM_FAKE_PLY_FLOATS_PER_SPLAT; k++) { + put_f32(p, f[k]); + p += 4; + } + } + } + } + return need; +} diff --git a/src/xrt/drivers/sim_display/sim_display_fake_ply.h b/src/xrt/drivers/sim_display/sim_display_fake_ply.h new file mode 100644 index 000000000..58f0d7b80 --- /dev/null +++ b/src/xrt/drivers/sim_display/sim_display_fake_ply.h @@ -0,0 +1,44 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief sim_display's FAKE photo → Gaussian-splat output (SIM_DISPLAY_FAKE_LIFT, + * XR_DXR_lift GAUSSIANS mode, ADR-042). + * + * Not a model: a tiny, VALID reference-3DGS binary PLY — two layers of splats + * (a front layer coloured from the photo on a grid, a darker back layer behind + * it) — so the blob path (lift thread → IPC varlen → xrAcquireLiftBlobDXR → + * a splat viewer) can be exercised end to end without vendor hardware. + * Platform-neutral so its format is pinned host-side (tests_lift_mailbox.cpp). + * + * @ingroup drv_sim_display + */ + +#pragma once + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +//! Splats per layer along x / y. +#define SIM_FAKE_PLY_GRID_X 16 +#define SIM_FAKE_PLY_GRID_Y 12 +//! Two layers. +#define SIM_FAKE_PLY_SPLATS (2 * SIM_FAKE_PLY_GRID_X * SIM_FAKE_PLY_GRID_Y) +//! Floats per splat: x y z nx ny nz f_dc_0..2 opacity scale_0..2 rot_0..3. +#define SIM_FAKE_PLY_FLOATS_PER_SPLAT 17 + +/*! + * Write the fake PLY for an RGBA8 image into @p dst (capacity @p cap). + * Returns the number of bytes the PLY needs; writes nothing when @p dst is + * NULL or @p cap is smaller than that. @p rgba may be NULL (flat grey). + */ +size_t +sim_fake_ply_write(uint8_t *dst, size_t cap, const uint8_t *rgba, uint32_t w, uint32_t h, uint32_t row_pitch); + +#ifdef __cplusplus +} +#endif diff --git a/src/xrt/drivers/sim_display/sim_display_lift_d3d11.cpp b/src/xrt/drivers/sim_display/sim_display_lift_d3d11.cpp new file mode 100644 index 000000000..044f83931 --- /dev/null +++ b/src/xrt/drivers/sim_display/sim_display_lift_d3d11.cpp @@ -0,0 +1,449 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief sim_display's FAKE lift module (see sim_display_lift_d3d11.h). + * @ingroup drv_sim_display + */ + +#include "sim_display_lift_d3d11.h" +#include "sim_display_fake_ply.h" + +#include "util/u_debug.h" +#include "util/u_logging.h" +#include "os/os_time.h" + +#define WIN32_LEAN_AND_MEAN +#include +#include +#include + +#include +#include +#include + +DEBUG_GET_ONCE_BOOL_OPTION(sim_display_fake_lift, "SIM_DISPLAY_FAKE_LIFT", false) +DEBUG_GET_ONCE_NUM_OPTION(sim_display_fake_lift_latency_ms, "SIM_DISPLAY_FAKE_LIFT_LATENCY_MS", 8) + +#define SIM_FAKE_LIFT_MAX_STREAMS 8 +#define SIM_FAKE_LIFT_MAX_VIEWS 8 + +static const char *k_vs = R"( +struct VS_OUTPUT { float4 pos : SV_Position; float2 uv : TEXCOORD0; }; +VS_OUTPUT main(uint id : SV_VertexID) { + VS_OUTPUT o; + o.uv = float2(id & 1, id >> 1); + o.pos = float4(o.uv * float2(2, -2) + float2(-1, 1), 0, 1); + return o; +} +)"; + +// Views side by side; view v samples the input shifted by (v - (n-1)/2) * shift. +static const char *k_ps_views = R"( +cbuffer P : register(b0) { float view_count; float shift; float2 pad; }; +Texture2D src : register(t0); +SamplerState samp : register(s0); +float4 main(float4 pos : SV_Position, float2 uv : TEXCOORD0) : SV_Target { + float v = min(floor(uv.x * view_count), view_count - 1.0); + float lu = uv.x * view_count - v; + float off = (v - (view_count - 1.0) * 0.5) * shift; + float4 c = src.Sample(samp, float2(saturate(lu + off), uv.y)); + return float4(c.rgb, c.a); // keep the source alpha (runtime#1742: transparent gaps must stay transparent) +} +)"; + +// Relative depth: a plain vertical gradient, 0 (near) at the top. +static const char *k_ps_depth = R"( +float main(float4 pos : SV_Position, float2 uv : TEXCOORD0) : SV_Target { return uv.y; } +)"; + +struct fake_cb +{ + float view_count; + float shift; + float pad[2]; +}; + +struct fake_stream +{ + bool used; + uint64_t id; + uint32_t mode; + ID3D11Texture2D *out; + ID3D11RenderTargetView *rtv; + uint32_t out_w, out_h; + DXGI_FORMAT out_fmt; + std::vector blob; +}; + +struct sim_fake_lift +{ + ID3D11Device *device; + ID3D11VertexShader *vs; + ID3D11PixelShader *ps_views; + ID3D11PixelShader *ps_depth; + ID3D11SamplerState *sampler; + ID3D11Buffer *cb; + uint64_t next_id; + fake_stream streams[SIM_FAKE_LIFT_MAX_STREAMS]; +}; + +static HRESULT +compile(const char *src, const char *target, ID3DBlob **out) +{ + ID3DBlob *err = nullptr; + HRESULT hr = D3DCompile(src, strlen(src), nullptr, nullptr, nullptr, "main", target, 0, 0, out, &err); + if (FAILED(hr) && err != nullptr) { + U_LOG_E("sim_display fake lift: shader compile error: %s", (const char *)err->GetBufferPointer()); + } + if (err != nullptr) { + err->Release(); + } + return hr; +} + +template +static void +safe_release(T *&p) +{ + if (p != nullptr) { + p->Release(); + p = nullptr; + } +} + +static void +stream_release(fake_stream &s) +{ + safe_release(s.rtv); + safe_release(s.out); + s.blob.clear(); + s.blob.shrink_to_fit(); + s.used = false; +} + +static fake_stream * +find_stream(struct sim_fake_lift *fl, uint64_t id) +{ + for (auto &s : fl->streams) { + if (s.used && s.id == id) { + return &s; + } + } + return nullptr; +} + +static void +fake_latency(void) +{ + int64_t ms = debug_get_num_option_sim_display_fake_lift_latency_ms(); + if (ms > 0) { + os_nanosleep(ms * 1000 * 1000); + } +} + +extern "C" bool +sim_fake_lift_enabled(void) +{ + return debug_get_bool_option_sim_display_fake_lift(); +} + +extern "C" struct sim_fake_lift * +sim_fake_lift_create(void *d3d11_device) +{ + if (!sim_fake_lift_enabled() || d3d11_device == nullptr) { + return nullptr; + } + auto *fl = new sim_fake_lift{}; + fl->device = static_cast(d3d11_device); + fl->device->AddRef(); + + ID3DBlob *b = nullptr; + bool ok = + SUCCEEDED(compile(k_vs, "vs_5_0", &b)) && + SUCCEEDED(fl->device->CreateVertexShader(b->GetBufferPointer(), b->GetBufferSize(), nullptr, &fl->vs)); + safe_release(b); + ok = + ok && SUCCEEDED(compile(k_ps_views, "ps_5_0", &b)) && + SUCCEEDED(fl->device->CreatePixelShader(b->GetBufferPointer(), b->GetBufferSize(), nullptr, &fl->ps_views)); + safe_release(b); + ok = + ok && SUCCEEDED(compile(k_ps_depth, "ps_5_0", &b)) && + SUCCEEDED(fl->device->CreatePixelShader(b->GetBufferPointer(), b->GetBufferSize(), nullptr, &fl->ps_depth)); + safe_release(b); + + if (ok) { + D3D11_SAMPLER_DESC sd = {}; + sd.Filter = D3D11_FILTER_MIN_MAG_MIP_LINEAR; + sd.AddressU = sd.AddressV = sd.AddressW = D3D11_TEXTURE_ADDRESS_CLAMP; + sd.MaxLOD = D3D11_FLOAT32_MAX; + ok = SUCCEEDED(fl->device->CreateSamplerState(&sd, &fl->sampler)); + } + if (ok) { + D3D11_BUFFER_DESC bd = {}; + bd.ByteWidth = sizeof(fake_cb); + bd.Usage = D3D11_USAGE_DEFAULT; + bd.BindFlags = D3D11_BIND_CONSTANT_BUFFER; + ok = SUCCEEDED(fl->device->CreateBuffer(&bd, nullptr, &fl->cb)); + } + if (!ok) { + U_LOG_E("sim_display fake lift: init failed — lift stays unavailable"); + sim_fake_lift_destroy(fl); + return nullptr; + } + U_LOG_W( + "sim_display: FAKE lift module active (SIM_DISPLAY_FAKE_LIFT) — shifted SBS/N-view, gradient " + "depth, two-layer PLY; not a model"); + return fl; +} + +extern "C" void +sim_fake_lift_destroy(struct sim_fake_lift *fl) +{ + if (fl == nullptr) { + return; + } + for (auto &s : fl->streams) { + stream_release(s); + } + safe_release(fl->cb); + safe_release(fl->sampler); + safe_release(fl->ps_depth); + safe_release(fl->ps_views); + safe_release(fl->vs); + safe_release(fl->device); + delete fl; +} + +extern "C" bool +sim_fake_lift_get_caps(struct sim_fake_lift *fl, struct xrt_dp_lift_caps *out) +{ + if (fl == nullptr || out == nullptr || out->struct_size < sizeof(struct xrt_dp_lift_caps)) { + return false; + } + out->modes = + XRT_DP_LIFT_MODE_DEPTH | XRT_DP_LIFT_MODE_SBS | XRT_DP_LIFT_MODE_NVIEW | XRT_DP_LIFT_MODE_GAUSSIANS; + out->max_streams = SIM_FAKE_LIFT_MAX_STREAMS; + out->max_views = SIM_FAKE_LIFT_MAX_VIEWS; + out->depth_semantics = XRT_DP_LIFT_DEPTH_RELATIVE; + out->state = XRT_DP_LIFT_STATE_READY; + out->typical_latency_ns = (uint64_t)debug_get_num_option_sim_display_fake_lift_latency_ms() * 1000000ull; + snprintf(out->backend, sizeof(out->backend), "sim_display-fake"); + return true; +} + +extern "C" bool +sim_fake_lift_stream_create(struct sim_fake_lift *fl, const struct xrt_dp_lift_stream_info *info, uint64_t *out_id) +{ + if (fl == nullptr || info == nullptr || out_id == nullptr) { + return false; + } + const uint32_t m = info->mode; + if (m != XRT_DP_LIFT_MODE_DEPTH && m != XRT_DP_LIFT_MODE_SBS && m != XRT_DP_LIFT_MODE_NVIEW && + m != XRT_DP_LIFT_MODE_GAUSSIANS) { + return false; + } + for (auto &s : fl->streams) { + if (!s.used) { + s = fake_stream{}; + s.used = true; + s.id = ++fl->next_id; + s.mode = m; + *out_id = s.id; + return true; + } + } + return false; +} + +extern "C" void +sim_fake_lift_stream_destroy(struct sim_fake_lift *fl, uint64_t id) +{ + if (fl == nullptr) { + return; + } + fake_stream *s = find_stream(fl, id); + if (s != nullptr) { + stream_release(*s); + } +} + +static bool +ensure_out(struct sim_fake_lift *fl, fake_stream &s, uint32_t w, uint32_t h, DXGI_FORMAT fmt) +{ + if (s.out != nullptr && s.out_w == w && s.out_h == h && s.out_fmt == fmt) { + return true; + } + safe_release(s.rtv); + safe_release(s.out); + D3D11_TEXTURE2D_DESC td = {}; + td.Width = w; + td.Height = h; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = fmt; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_DEFAULT; + td.BindFlags = D3D11_BIND_RENDER_TARGET | D3D11_BIND_SHADER_RESOURCE; + if (FAILED(fl->device->CreateTexture2D(&td, nullptr, &s.out)) || + FAILED(fl->device->CreateRenderTargetView(s.out, nullptr, &s.rtv))) { + safe_release(s.rtv); + safe_release(s.out); + return false; + } + s.out_w = w; + s.out_h = h; + s.out_fmt = fmt; + return true; +} + +extern "C" bool +sim_fake_lift_convert(struct sim_fake_lift *fl, + uint64_t id, + void *d3d11_context, + void *input_resource, + uint32_t w, + uint32_t h, + const struct xrt_dp_lift_params *p, + void **out_resource, + uint32_t *out_w, + uint32_t *out_h, + uint32_t *out_format) +{ + if (fl == nullptr || d3d11_context == nullptr || input_resource == nullptr || w == 0 || h == 0 || + out_resource == nullptr || out_w == nullptr || out_h == nullptr || out_format == nullptr) { + return false; + } + fake_stream *s = find_stream(fl, id); + if (s == nullptr || s->mode == XRT_DP_LIFT_MODE_GAUSSIANS) { + return false; + } + auto *ctx = static_cast(d3d11_context); + + uint32_t views = 1; + if (s->mode == XRT_DP_LIFT_MODE_SBS) { + views = 2; + } else if (s->mode == XRT_DP_LIFT_MODE_NVIEW) { + views = (p != nullptr && p->view_count >= 1) ? p->view_count : 4; + views = views > SIM_FAKE_LIFT_MAX_VIEWS ? SIM_FAKE_LIFT_MAX_VIEWS : views; + } + const bool depth = s->mode == XRT_DP_LIFT_MODE_DEPTH; + const DXGI_FORMAT fmt = depth ? DXGI_FORMAT_R32_FLOAT : DXGI_FORMAT_R8G8B8A8_UNORM; + const uint32_t ow = w * views; + if (!ensure_out(fl, *s, ow, h, fmt)) { + return false; + } + + ID3D11ShaderResourceView *srv = nullptr; + if (!depth) { + D3D11_SHADER_RESOURCE_VIEW_DESC sd = {}; + sd.Format = DXGI_FORMAT_R8G8B8A8_UNORM; + sd.ViewDimension = D3D11_SRV_DIMENSION_TEXTURE2D; + sd.Texture2D.MipLevels = 1; + if (FAILED(fl->device->CreateShaderResourceView(static_cast(input_resource), &sd, + &srv))) { + return false; + } + } + + // ~2% of the view width per view step at strength 1: visible, obviously fake. + fake_cb cb = {}; + cb.view_count = (float)views; + cb.shift = 0.02f * (p != nullptr && p->strength > 0.0f ? p->strength : 1.0f); + ctx->UpdateSubresource(fl->cb, 0, nullptr, &cb, 0, 0); + + D3D11_VIEWPORT vp = {}; + vp.Width = (float)ow; + vp.Height = (float)h; + vp.MaxDepth = 1.0f; + ctx->RSSetViewports(1, &vp); + D3D11_RECT sc = {0, 0, (LONG)ow, (LONG)h}; + ctx->RSSetScissorRects(1, &sc); + ctx->OMSetRenderTargets(1, &s->rtv, nullptr); + ctx->OMSetBlendState(nullptr, nullptr, 0xffffffff); + ctx->IASetPrimitiveTopology(D3D11_PRIMITIVE_TOPOLOGY_TRIANGLESTRIP); + ctx->IASetInputLayout(nullptr); + ctx->VSSetShader(fl->vs, nullptr, 0); + ctx->PSSetShader(depth ? fl->ps_depth : fl->ps_views, nullptr, 0); + ctx->PSSetConstantBuffers(0, 1, &fl->cb); + ctx->PSSetSamplers(0, 1, &fl->sampler); + ctx->PSSetShaderResources(0, 1, &srv); + ctx->Draw(4, 0); + ID3D11ShaderResourceView *null_srv = nullptr; + ctx->PSSetShaderResources(0, 1, &null_srv); + ID3D11RenderTargetView *null_rtv = nullptr; + ctx->OMSetRenderTargets(1, &null_rtv, nullptr); + safe_release(srv); + + fake_latency(); + + *out_resource = s->out; + *out_w = ow; + *out_h = h; + *out_format = (uint32_t)fmt; + return true; +} + +extern "C" bool +sim_fake_lift_convert_blob(struct sim_fake_lift *fl, + uint64_t id, + void *d3d11_context, + void *input_resource, + uint32_t w, + uint32_t h, + uint32_t *out_format, + const void **out_bytes, + size_t *out_size) +{ + if (fl == nullptr || d3d11_context == nullptr || input_resource == nullptr || out_format == nullptr || + out_bytes == nullptr || out_size == nullptr) { + return false; + } + fake_stream *s = find_stream(fl, id); + if (s == nullptr || s->mode != XRT_DP_LIFT_MODE_GAUSSIANS) { + return false; + } + auto *ctx = static_cast(d3d11_context); + + // Read the photo back (the fake colours its front layer from it). + D3D11_TEXTURE2D_DESC td = {}; + td.Width = w; + td.Height = h; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = DXGI_FORMAT_R8G8B8A8_UNORM; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_STAGING; + td.CPUAccessFlags = D3D11_CPU_ACCESS_READ; + ID3D11Texture2D *staging = nullptr; + const uint8_t *rgba = nullptr; + uint32_t pitch = 0; + D3D11_MAPPED_SUBRESOURCE map = {}; + bool mapped = false; + if (SUCCEEDED(fl->device->CreateTexture2D(&td, nullptr, &staging))) { + D3D11_BOX box = {0, 0, 0, w, h, 1}; + ctx->CopySubresourceRegion(staging, 0, 0, 0, 0, static_cast(input_resource), 0, &box); + if (SUCCEEDED(ctx->Map(staging, 0, D3D11_MAP_READ, 0, &map))) { + mapped = true; + rgba = static_cast(map.pData); + pitch = map.RowPitch; + } + } + + size_t need = sim_fake_ply_write(nullptr, 0, nullptr, w, h, 0); + s->blob.resize(need); + sim_fake_ply_write(s->blob.data(), s->blob.size(), rgba, w, h, pitch); + + if (mapped) { + ctx->Unmap(staging, 0); + } + safe_release(staging); + + // Photo → splats is seconds on a real module; the fake is quick but not free. + fake_latency(); + + *out_format = XRT_DP_LIFT_BLOB_PLY_3DGS; + *out_bytes = s->blob.data(); + *out_size = s->blob.size(); + return true; +} diff --git a/src/xrt/drivers/sim_display/sim_display_lift_d3d11.h b/src/xrt/drivers/sim_display/sim_display_lift_d3d11.h new file mode 100644 index 000000000..a38108463 --- /dev/null +++ b/src/xrt/drivers/sim_display/sim_display_lift_d3d11.h @@ -0,0 +1,84 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief sim_display's env-gated FAKE 2D→3D conversion module + * (SIM_DISPLAY_FAKE_LIFT=1; XR_DXR_lift, ADR-042). + * + * sim_display ships no conversion module, so by default its lift slots stay + * NULL and the runtime reports supportedModes 0. With SIM_DISPLAY_FAKE_LIFT=1 + * the D3D11 DP installs this fake so the whole path — lift thread, mailbox, + * ring, IPC, weave-rect lifting, blob transport — can run hardware-free: + * + * - SBS / NVIEW: the input, horizontally SHIFTED per view (a constant + * parallax, not depth-aware), views side by side. + * - DEPTH: a vertical gradient (R32_FLOAT, 0 at the top, 1 at the bottom). + * - GAUSSIANS: a tiny valid two-layer 3DGS PLY (sim_display_fake_ply.h). + * + * SIM_DISPLAY_FAKE_LIFT_LATENCY_MS (default 8) sleeps inside each convert so + * the runtime's asynchrony and latest-wins dropping are observable. + * + * @ingroup drv_sim_display + */ + +#pragma once + +#include "xrt/xrt_dp_lift.h" + +#include +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +struct sim_fake_lift; + +//! True when SIM_DISPLAY_FAKE_LIFT is set. +bool +sim_fake_lift_enabled(void); + +//! Create the fake on @p d3d11_device (ID3D11Device*). NULL on failure. +struct sim_fake_lift * +sim_fake_lift_create(void *d3d11_device); + +void +sim_fake_lift_destroy(struct sim_fake_lift *fl); + +bool +sim_fake_lift_get_caps(struct sim_fake_lift *fl, struct xrt_dp_lift_caps *out); + +bool +sim_fake_lift_stream_create(struct sim_fake_lift *fl, const struct xrt_dp_lift_stream_info *info, uint64_t *out_id); + +void +sim_fake_lift_stream_destroy(struct sim_fake_lift *fl, uint64_t id); + +bool +sim_fake_lift_convert(struct sim_fake_lift *fl, + uint64_t id, + void *d3d11_context, + void *input_resource, + uint32_t w, + uint32_t h, + const struct xrt_dp_lift_params *p, + void **out_resource, + uint32_t *out_w, + uint32_t *out_h, + uint32_t *out_format); + +bool +sim_fake_lift_convert_blob(struct sim_fake_lift *fl, + uint64_t id, + void *d3d11_context, + void *input_resource, + uint32_t w, + uint32_t h, + uint32_t *out_format, + const void **out_bytes, + size_t *out_size); + +#ifdef __cplusplus +} +#endif diff --git a/src/xrt/drivers/sim_display/sim_display_plugin.c b/src/xrt/drivers/sim_display/sim_display_plugin.c index 9eeae5bc8..dca190627 100644 --- a/src/xrt/drivers/sim_display/sim_display_plugin.c +++ b/src/xrt/drivers/sim_display/sim_display_plugin.c @@ -254,6 +254,17 @@ static struct xrt_plugin_iface g_sim_display_iface = { .set_pose_source = sim_display_plugin_set_pose_source, .probe_displays = sim_display_plugin_probe_displays, + + /* + * ADR-042 lift-only D3D11 DP. sim_display builds no weaver or tracker for + * any DP, so its ordinary factory is already "lift-only"-cheap; the lift + * slots are filled only under SIM_DISPLAY_FAKE_LIFT=1. + */ +#if defined(_WIN32) + .create_dp_d3d11_lift = sim_display_dp_factory_d3d11, +#else + .create_dp_d3d11_lift = NULL, +#endif }; diff --git a/src/xrt/drivers/sim_display/sim_display_processor_d3d11.cpp b/src/xrt/drivers/sim_display/sim_display_processor_d3d11.cpp index 820fe991c..838067006 100644 --- a/src/xrt/drivers/sim_display/sim_display_processor_d3d11.cpp +++ b/src/xrt/drivers/sim_display/sim_display_processor_d3d11.cpp @@ -15,6 +15,7 @@ #include "sim_display_interface.h" #include "sim_display_zone_common.h" #include "sim_display_scanout_common.h" +#include "sim_display_lift_d3d11.h" #include "xrt/xrt_display_processor_d3d11.h" #include "xrt/xrt_display_metrics.h" @@ -130,7 +131,10 @@ float4 main(float4 pos : SV_Position, float2 uv : TEXCOORD0) : SV_Target { float2 right_uv = float2((uv.x + col1) * tile_cols_inv, (uv.y + row1) * tile_rows_inv); float4 left = atlas_tex.Sample(samp, left_uv); float4 right = atlas_tex.Sample(samp, right_uv); - return out_finish(float4(left.r, right.g, right.b, 1.0), uv); + // Carry the atlas alpha: a present-owner weave (browser weave_frame_first) draws the tiles back + // whole-window and relies on transparent gaps to show the page through. Forcing alpha=1 painted + // everything outside the 3D tiles black (runtime#1742). + return out_finish(float4(left.r, right.g, right.b, max(left.a, right.a)), uv); } )"; @@ -241,6 +245,10 @@ struct sim_display_processor_d3d11_impl ID3D11ShaderResourceView *wish_copy_srv; uint32_t wish_copy_w, wish_copy_h; bool wish_active; //!< wish_copy holds a live (not cleared) publish. + + //! ADR-042: the FAKE lift module, only when SIM_DISPLAY_FAKE_LIFT=1 (else + //! NULL and the five lift slots stay NULL — sim_display ships no module). + struct sim_fake_lift *fake_lift; }; static inline struct sim_display_processor_d3d11_impl * @@ -447,6 +455,8 @@ sim_dp_d3d11_destroy(struct xrt_display_processor_d3d11 *xdp) if (sdp->wish_copy != nullptr) { sdp->wish_copy->Release(); } + sim_fake_lift_destroy(sdp->fake_lift); + sdp->fake_lift = nullptr; free(sdp); } @@ -763,6 +773,71 @@ sim_dp_d3d11_get_scanout_caps(struct xrt_display_processor_d3d11 *xdp, struct xr } +/* + * + * ADR-042 lift slots — installed only with SIM_DISPLAY_FAKE_LIFT=1. + * + */ + +static bool +sim_dp_d3d11_lift_get_caps(struct xrt_display_processor_d3d11 *xdp, struct xrt_dp_lift_caps *out) +{ + return sim_fake_lift_get_caps(sim_dp_d3d11(xdp)->fake_lift, out); +} + +static bool +sim_dp_d3d11_lift_stream_create(struct xrt_display_processor_d3d11 *xdp, + const struct xrt_dp_lift_stream_info *info, + uint64_t *out_id) +{ + return sim_fake_lift_stream_create(sim_dp_d3d11(xdp)->fake_lift, info, out_id); +} + +static void +sim_dp_d3d11_lift_stream_destroy(struct xrt_display_processor_d3d11 *xdp, uint64_t id) +{ + sim_fake_lift_stream_destroy(sim_dp_d3d11(xdp)->fake_lift, id); +} + +static bool +sim_dp_d3d11_lift_convert(struct xrt_display_processor_d3d11 *xdp, + uint64_t id, + void *d3d11_context, + void *input_resource, + uint32_t w, + uint32_t h, + const struct xrt_dp_lift_params *p, + const float *viewpoints_xyz, + uint32_t viewpoint_floats, + void **out_resource, + uint32_t *out_w, + uint32_t *out_h, + uint32_t *out_format) +{ + // The fake synthesizes a constant parallax; it has no use for viewpoints. + (void)viewpoints_xyz; + (void)viewpoint_floats; + return sim_fake_lift_convert(sim_dp_d3d11(xdp)->fake_lift, id, d3d11_context, input_resource, w, h, p, + out_resource, out_w, out_h, out_format); +} + +static bool +sim_dp_d3d11_lift_convert_blob(struct xrt_display_processor_d3d11 *xdp, + uint64_t id, + void *d3d11_context, + void *input_resource, + uint32_t w, + uint32_t h, + const struct xrt_dp_lift_params *p, + uint32_t *out_format, + const void **out_bytes, + size_t *out_size) +{ + (void)p; + return sim_fake_lift_convert_blob(sim_dp_d3d11(xdp)->fake_lift, id, d3d11_context, input_resource, w, h, + out_format, out_bytes, out_size); +} + extern "C" xrt_result_t sim_display_processor_d3d11_create(enum sim_display_output_mode mode, void *d3d11_device, @@ -887,6 +962,19 @@ sim_display_processor_d3d11_create(enum sim_display_output_mode mode, return XRT_ERROR_VULKAN; } + // ADR-042: the FAKE lift module, env-gated. Without it every lift slot stays + // NULL (calloc) and the runtime reports XR_DXR_lift supportedModes = 0. + if (sim_fake_lift_enabled()) { + sdp->fake_lift = sim_fake_lift_create(d3d11_device); + if (sdp->fake_lift != nullptr) { + sdp->base.lift_get_caps = sim_dp_d3d11_lift_get_caps; + sdp->base.lift_stream_create = sim_dp_d3d11_lift_stream_create; + sdp->base.lift_stream_destroy = sim_dp_d3d11_lift_stream_destroy; + sdp->base.lift_convert = sim_dp_d3d11_lift_convert; + sdp->base.lift_convert_blob = sim_dp_d3d11_lift_convert_blob; + } + } + // Set the initial output mode (atomic global read by process_atlas each frame) sim_display_set_output_mode(mode); diff --git a/src/xrt/include/xrt/xrt_compositor.h b/src/xrt/include/xrt/xrt_compositor.h index a0a901b8b..092c81e6e 100644 --- a/src/xrt/include/xrt/xrt_compositor.h +++ b/src/xrt/include/xrt/xrt_compositor.h @@ -3036,6 +3036,13 @@ struct xrt_system_compositor_info //! Signature: xrt_dp_factory_gl_fn_t (see xrt_display_processor_gl.h). void *dp_factory_gl; + //! ADR-042: the plug-in's LIFT-ONLY D3D11 display processor factory + //! (xrt_plugin_iface::create_dp_d3d11_lift) — a DP that serves only the + //! lift slots and builds no weaver. NULL = the plug-in has none; the D3D11 + //! service then falls back to dp_factory_d3d11 with a NULL window. + //! Signature: xrt_dp_factory_d3d11_fn_t. + void *dp_factory_d3d11_lift; + /*! * Optional callback: re-derive the dp_factory_* pointers above from the * runtime's plug-in loader, and re-pull the plug-in's display info diff --git a/src/xrt/include/xrt/xrt_display_processor_d3d11.h b/src/xrt/include/xrt/xrt_display_processor_d3d11.h index b52497032..a3826b3c6 100644 --- a/src/xrt/include/xrt/xrt_display_processor_d3d11.h +++ b/src/xrt/include/xrt/xrt_display_processor_d3d11.h @@ -23,6 +23,7 @@ #include "xrt/xrt_display_color.h" #include "xrt/xrt_display_zones.h" #include "xrt/xrt_display_scanout.h" +#include "xrt/xrt_dp_lift.h" #include #include @@ -583,6 +584,105 @@ struct xrt_display_processor_d3d11 */ bool (*get_background_preview)(struct xrt_display_processor_d3d11 *xdp, struct xrt_dp_background_preview *out_preview); + + /* + * ── 2D→3D conversion ("lift", ADR-042, XR_DXR_lift) ───────────────────── + * + * Five optional slots, appended together per ADR-020 and announced by ONE + * define, XRT_DP_D3D11_HAS_LIFT. A plug-in with no conversion module leaves + * all five NULL (or an older plug-in's struct_size stops short of them); + * the runtime then reports XR_DXR_lift supportedModes = 0, state + * UNAVAILABLE. + * + * THREADING. The runtime calls every lift slot from ONE thread — its lift + * thread — on a display processor instance it created FOR lift, through + * the plug-in's lift-only factory (xrt_plugin_iface::create_dp_d3d11_lift; + * fallback: create_dp_d3d11 with a NULL window): a dedicated D3D11 device + * on the service's adapter, @p d3d11_context its immediate context. That + * instance is never asked to weave, so it must build no weaver and open no + * tracker session. The context has ID3D11Multithread protection enabled, + * so a module that flushes or signals it from its own worker is safe. + * + * SYNCHRONOUS. lift_convert / lift_convert_blob run one conversion and + * return its output. Asynchrony, latest-wins dropping, timestamps and the + * copy of the output into a runtime ring are RUNTIME code — the plug-in + * never queues or timestamps. + */ + + /*! + * Report the module's capabilities and state (@ref xrt_dp_lift_caps). The + * caller pre-sets struct_size; write only fields within it. Called at DP + * creation and then ≤ 1 Hz while the state is not READY (to see + * ACTIVATING → READY). Return false = no module (same as a NULL slot). + */ + bool (*lift_get_caps)(struct xrt_display_processor_d3d11 *xdp, struct xrt_dp_lift_caps *out); + + /*! + * Create a conversion stream; return the plug-in's own id in @p out_id. + * False = refused (mode unsupported, too many streams): the runtime fails + * the app's stream. + */ + bool (*lift_stream_create)(struct xrt_display_processor_d3d11 *xdp, + const struct xrt_dp_lift_stream_info *info, + uint64_t *out_id); + + //! Destroy stream @p id and every resource it returned. + void (*lift_stream_destroy)(struct xrt_display_processor_d3d11 *xdp, uint64_t id); + + /*! + * Convert one frame of stream @p id (texture modes: DEPTH, SBS, NVIEW). + * + * @param input_resource ID3D11Resource* on the lift device, RGBA8, exactly + * @p w x @p h. Valid for the duration of the call. + * @param p per-frame parameters (never NULL). + * @param viewpoints_xyz @p viewpoint_floats / 3 display-space eye positions + * (metres) to synthesize for. The runtime ALWAYS + * passes them when it has them: the app's explicit + * viewpoints, else the panel DP's predicted tracked + * eyes (the pair; a module spreads N views around + * it). A plug-in must give these precedence over any + * tracker of its own — the lift DP has no tracker + * session. NULL / 0 only when no eyes are known yet. + * @param out_resource ID3D11Resource* the DP owns, valid until the NEXT + * call on this stream (the runtime copies it out + * before then). SBS = 2 views side by side; NVIEW = + * view_count views side by side, view 0 leftmost; + * DEPTH = one channel. + * @param out_format DXGI_FORMAT of @p out_resource. + * @return false = no output this frame (the runtime keeps the previous one). + */ + bool (*lift_convert)(struct xrt_display_processor_d3d11 *xdp, + uint64_t id, + void *d3d11_context, + void *input_resource, + uint32_t w, + uint32_t h, + const struct xrt_dp_lift_params *p, + const float *viewpoints_xyz, + uint32_t viewpoint_floats, + void **out_resource, + uint32_t *out_w, + uint32_t *out_h, + uint32_t *out_format); + + /*! + * Convert one photo of a GAUSSIANS stream @p id into a splat blob. + * + * Same threading and input contract as @ref lift_convert. Returns the blob + * as bytes the DP owns, valid until the NEXT call on this stream (the + * runtime copies them), in @p out_format (XRT_DP_LIFT_BLOB_*). Expected to + * take seconds; the runtime calls it off every latency-sensitive thread. + */ + bool (*lift_convert_blob)(struct xrt_display_processor_d3d11 *xdp, + uint64_t id, + void *d3d11_context, + void *input_resource, + uint32_t w, + uint32_t h, + const struct xrt_dp_lift_params *p, + uint32_t *out_format, + const void **out_bytes, + size_t *out_size); }; @@ -658,7 +758,22 @@ XRT_DP_ABI_ASSERT(offsetof(struct xrt_display_processor_d3d11, set_predicted_sca */ #define XRT_DP_D3D11_HAS_PREDICTED_SCANOUT 1 XRT_DP_ABI_ASSERT(offsetof(struct xrt_display_processor_d3d11, get_background_preview) == XRT_DP_D3D11_BASE_OFF + 24 * sizeof(void *), XRT_DP_ABI_MSG); -XRT_DP_ABI_ASSERT(sizeof(struct xrt_display_processor_d3d11) == XRT_DP_D3D11_BASE_OFF + 25 * sizeof(void *), XRT_DP_ABI_MSG); +XRT_DP_ABI_ASSERT(offsetof(struct xrt_display_processor_d3d11, lift_get_caps) == XRT_DP_D3D11_BASE_OFF + 25 * sizeof(void *), XRT_DP_ABI_MSG); +XRT_DP_ABI_ASSERT(offsetof(struct xrt_display_processor_d3d11, lift_stream_create) == XRT_DP_D3D11_BASE_OFF + 26 * sizeof(void *), XRT_DP_ABI_MSG); +XRT_DP_ABI_ASSERT(offsetof(struct xrt_display_processor_d3d11, lift_stream_destroy) == XRT_DP_D3D11_BASE_OFF + 27 * sizeof(void *), XRT_DP_ABI_MSG); +XRT_DP_ABI_ASSERT(offsetof(struct xrt_display_processor_d3d11, lift_convert) == XRT_DP_D3D11_BASE_OFF + 28 * sizeof(void *), XRT_DP_ABI_MSG); +XRT_DP_ABI_ASSERT(offsetof(struct xrt_display_processor_d3d11, lift_convert_blob) == XRT_DP_D3D11_BASE_OFF + 29 * sizeof(void *), XRT_DP_ABI_MSG); +XRT_DP_ABI_ASSERT(sizeof(struct xrt_display_processor_d3d11) == XRT_DP_D3D11_BASE_OFF + 30 * sizeof(void *), XRT_DP_ABI_MSG); + +/*! + * Defined when this header carries the five lift slots (lift_get_caps, + * lift_stream_create, lift_stream_destroy, lift_convert, lift_convert_blob — + * ADR-042, XR_DXR_lift), so a plug-in built against an older runtime can + * #ifdef-guard its conversion module — the coupled-ABI-addition pattern used + * by every other appended slot. Purely additive: no + * XRT_PLUGIN_API_VERSION_CURRENT bump (ADR-020). + */ +#define XRT_DP_D3D11_HAS_LIFT 1 /*! * Defined when this header carries the get_background_preview slot, so a @@ -1160,6 +1275,50 @@ xrt_display_processor_d3d11_get_background_preview(struct xrt_display_processor_ return xdp->get_background_preview(xdp, out_preview); } +/*! + * @copydoc xrt_display_processor_d3d11::lift_get_caps + * + * Returns false when the slot is absent (older plug-in `struct_size`), NULL, or + * the DP has no module — every one of those reads as "no lift": modes 0, + * state UNAVAILABLE. @p out is initialised here either way. + * + * @public @memberof xrt_display_processor_d3d11 + */ +static inline bool +xrt_display_processor_d3d11_lift_get_caps(struct xrt_display_processor_d3d11 *xdp, struct xrt_dp_lift_caps *out) +{ + xrt_dp_lift_caps_init(out); + if (!XRT_DP_HAS_SLOT(xdp, lift_get_caps) || xdp->lift_get_caps == NULL) { + return false; + } + if (!xdp->lift_get_caps(xdp, out)) { + xrt_dp_lift_caps_init(out); + return false; + } + out->backend[sizeof(out->backend) - 1] = '\0'; + return true; +} + +/*! + * True when @p xdp carries the whole texture-mode lift contract (caps + stream + * create/destroy + convert). A DP must provide all four or none. + * + * @public @memberof xrt_display_processor_d3d11 + */ +static inline bool +xrt_display_processor_d3d11_has_lift(struct xrt_display_processor_d3d11 *xdp) +{ + return XRT_DP_HAS_SLOT(xdp, lift_convert) && xdp->lift_get_caps != NULL && xdp->lift_stream_create != NULL && + xdp->lift_stream_destroy != NULL && xdp->lift_convert != NULL; +} + +//! True when @p xdp also carries lift_convert_blob (GAUSSIANS). +static inline bool +xrt_display_processor_d3d11_has_lift_blob(struct xrt_display_processor_d3d11 *xdp) +{ + return XRT_DP_HAS_SLOT(xdp, lift_convert_blob) && xdp->lift_convert_blob != NULL; +} + #ifdef __cplusplus } #endif diff --git a/src/xrt/include/xrt/xrt_dp_lift.h b/src/xrt/include/xrt/xrt_dp_lift.h new file mode 100644 index 000000000..f2f25c3ff --- /dev/null +++ b/src/xrt/include/xrt/xrt_dp_lift.h @@ -0,0 +1,151 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief Display-processor 2D→3D conversion ("lift") contract types (ADR-042). + * + * A vendor plug-in may ship a conversion module (monocular depth, stereo or + * N-view synthesis, photo → Gaussian splats). The runtime reaches it through + * optional appended slots on the per-API DP vtable — today only + * @ref xrt_display_processor_d3d11 (lift_get_caps … lift_convert_blob, guarded + * by XRT_DP_D3D11_HAS_LIFT) — and exposes it to apps as XR_DXR_lift. + * + * The split of labour (ADR-042, ADR-007): + * - the PLUG-IN converts, synchronously, one frame per call. It never weaves + * the result, never queues, never timestamps. + * - the RUNTIME owns everything around the call: the lift thread, the + * latest-wins mailbox, the output ring (a slot's output is valid only until + * that slot's next call), fences, timestamps, IPC, and weaving the SBS / + * N-view result on the ordinary weave path. + * + * Every struct starts with @c struct_size, set by the CALLER (the runtime for + * caps / params / stream info), so either side may grow its struct by + * appending fields without an ABI bump (ADR-020). + * + * @ingroup xrt_iface + */ + +#pragma once + +#include + +#ifdef __cplusplus +extern "C" { +#endif + +/*! + * @name Lift mode bits (xrt_dp_lift_caps::modes, xrt_dp_lift_stream_info::mode) + * Values match XR_LIFT_MODE_*_BIT_DXR. + * @{ + */ +#define XRT_DP_LIFT_MODE_DEPTH 1u //!< monocular depth map +#define XRT_DP_LIFT_MODE_SBS 2u //!< stereo, side by side, NOT woven +#define XRT_DP_LIFT_MODE_NVIEW 4u //!< N views side by side in one row, NOT woven +#define XRT_DP_LIFT_MODE_GAUSSIANS 8u //!< photo → Gaussian splats, via lift_convert_blob +/*! @} */ + +/*! + * @name Lift module state (xrt_dp_lift_caps::state) + * Values are the DP-side encoding; XR_LIFT_STATE_*_DXR maps them 1:1 in name + * (not in value — the OpenXR enum orders UNAVAILABLE=0, ACTIVATING=1, READY=2 + * too, so they do coincide; keep it that way). + * @{ + */ +#define XRT_DP_LIFT_STATE_UNAVAILABLE 0u +#define XRT_DP_LIFT_STATE_ACTIVATING 1u +#define XRT_DP_LIFT_STATE_READY 2u +/*! @} */ + +//! xrt_dp_lift_caps::depth_semantics +#define XRT_DP_LIFT_DEPTH_RELATIVE 0u +#define XRT_DP_LIFT_DEPTH_METRIC 1u + +//! xrt_dp_lift_stream_info::content_hint +#define XRT_DP_LIFT_CONTENT_VIDEO 0u +#define XRT_DP_LIFT_CONTENT_PHOTO 1u + +/*! + * @name Blob formats (lift_convert_blob's out_format) + * Values match XR_LIFT_BLOB_FORMAT_*_DXR. + * @{ + */ +#define XRT_DP_LIFT_BLOB_PLY_3DGS 1u //!< binary little-endian PLY, reference 3DGS layout +#define XRT_DP_LIFT_BLOB_SOG 2u //!< PlayCanvas SOG container +/*! @} */ + +/*! + * What the plug-in's conversion module can do. The runtime pre-sets + * @c struct_size (zeroing the rest); the DP writes only fields within it. + * Cheap: the runtime polls it from its lift thread (≤ 1 Hz while not READY). + */ +struct xrt_dp_lift_caps +{ + uint32_t struct_size; + uint32_t modes; //!< bits: 1 DEPTH, 2 SBS, 4 NVIEW, 8 GAUSSIANS + uint32_t max_streams; + uint32_t max_views; + uint32_t depth_semantics; //!< 0 relative, 1 metric + uint32_t state; //!< 0 unavailable, 1 activating, 2 ready + uint64_t typical_latency_ns; //!< submit→result as the module expects it; 0 = unknown + char backend[32]; //!< NUL-terminated module name (informational) +}; + +//! Stream creation parameters (runtime-filled). +struct xrt_dp_lift_stream_info +{ + uint32_t struct_size; + uint32_t mode; //!< ONE XRT_DP_LIFT_MODE_* bit + uint32_t content_hint; //!< 0 video, 1 photo + float input_scale; //!< (0,1]: convert at reduced resolution; 1 = native +}; + +//! Upper bound on explicit viewpoints the runtime passes to lift_convert. +#define XRT_DP_LIFT_MAX_EXPLICIT_VIEWPOINTS 8 + +/*! + * Per-frame conversion parameters (runtime-filled). + * + * @c focal_px (appended) is the input image's focal length in pixels; <= 0 = + * unknown. + * + * @c convergence is the RELATIVE depth placed at the display plane (zero + * disparity), normalised to [0, 1] over the frame's depth range: 0 = the + * nearest content sits on the glass (everything else behind it), 1 = the + * farthest does (everything pops out), 0.5 = the middle of the range. Any + * negative value = AUTO: the module picks it. The plug-in maps this to its + * model's own units and calibrates it on its panel; the runtime never + * interprets it beyond clamping to [0, 1] (or passing a negative through). + */ +struct xrt_dp_lift_params +{ + uint32_t struct_size; + float convergence; //!< relative depth at the display plane, [0,1]; < 0 = AUTO + float strength; //!< disparity scale; 1 = the module's calibrated budget + uint32_t inpaint; //!< non-zero = fill disocclusions + uint32_t view_count; //!< views to produce (2 for SBS, N for NVIEW; ignored otherwise) + /*! + * Focal length of the submitted image, in PIXELS of the input (w x h) — + * what a photo → Gaussians module (SHARP-class) takes as its intrinsics + * input. <= 0 = unknown: the module assumes its default field of view. + * Appended (read only when struct_size covers it); DEPTH / SBS / NVIEW + * modules ignore it. + */ + float focal_px; +}; + +/*! + * Pre-set @p caps for a lift_get_caps call: zero it and stamp struct_size. + */ +static inline void +xrt_dp_lift_caps_init(struct xrt_dp_lift_caps *caps) +{ + uint8_t *p = (uint8_t *)caps; + for (uint32_t i = 0; i < (uint32_t)sizeof(*caps); i++) { + p[i] = 0; + } + caps->struct_size = (uint32_t)sizeof(*caps); +} + +#ifdef __cplusplus +} +#endif diff --git a/src/xrt/include/xrt/xrt_lift.h b/src/xrt/include/xrt/xrt_lift.h new file mode 100644 index 000000000..ea0ad5af8 --- /dev/null +++ b/src/xrt/include/xrt/xrt_lift.h @@ -0,0 +1,81 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief Runtime-internal XR_DXR_lift (ADR-042) value types shared by the + * state tracker, the IPC bridges and the service compositor. + * + * The DP-facing contract is xrt_dp_lift.h; these are the runtime's own + * plumbing types between xrAcquireLiftResultDXR & co. and the service. + * + * @ingroup xrt_iface + */ + +#pragma once + +#include "xrt/xrt_dp_lift.h" + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +//! Max lift-flagged rects per weave submit (mirrors XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR). +#define XRT_LIFT_WEAVE_RECTS_MAX 8 +//! Max explicit viewpoints per submit (mirrors XR_LIFT_MAX_VIEWS_DXR). +#define XRT_LIFT_MAX_VIEWS 8 + +//! One acquired texture result. +struct xrt_lift_result +{ + uint64_t frame_id; + int64_t source_time; + uint64_t fence_value; + uint64_t latency_ns; + uint32_t width; + uint32_t height; + uint32_t format; //!< DXGI_FORMAT + uint32_t view_count; + bool output_realloc; //!< export texture changed: re-export it to the caller +}; + +//! One acquired blob result. +struct xrt_lift_blob_info +{ + uint64_t frame_id; + int64_t source_time; + uint32_t format; //!< XRT_DP_LIFT_BLOB_* + uint64_t byte_count; +}; + +//! A lift-flagged rect of the NEXT weave submit (XrWeaveRectLiftDXR). +struct xrt_lift_weave_rect +{ + uint64_t stream_id; + uint32_t rect_index; + bool has_params; + struct xrt_dp_lift_params params; +}; + +//! One stream's counters + effective rate (XrLiftStreamStatsDXR). +struct xrt_lift_stream_stats +{ + uint32_t priority; //!< XrLiftPriorityDXR value (0 paused .. 3 high) + uint32_t reserved; + uint64_t submitted; + uint64_t converted; + uint64_t dropped; + uint64_t failed; + uint64_t latency_last_ns; + uint64_t latency_avg_ns; + uint64_t latency_min_ns; + uint64_t latency_max_ns; + float rate_hz; + uint32_t reserved2; +}; + +#ifdef __cplusplus +} +#endif diff --git a/src/xrt/include/xrt/xrt_openxr_includes.h b/src/xrt/include/xrt/xrt_openxr_includes.h index e64c6787b..4016e59f6 100644 --- a/src/xrt/include/xrt/xrt_openxr_includes.h +++ b/src/xrt/include/xrt/xrt_openxr_includes.h @@ -86,3 +86,4 @@ typedef __eglMustCastToProperFunctionPointerType (*PFNEGLGETPROCADDRESSPROC)(con #include "openxr/XR_DXR_depth_budget.h" #include "openxr/XR_DXR_display_zones.h" #include "openxr/XR_DXR_weave.h" +#include "openxr/XR_DXR_lift.h" diff --git a/src/xrt/include/xrt/xrt_plugin.h b/src/xrt/include/xrt/xrt_plugin.h index 3a805b088..16a8f6109 100644 --- a/src/xrt/include/xrt/xrt_plugin.h +++ b/src/xrt/include/xrt/xrt_plugin.h @@ -784,8 +784,42 @@ struct xrt_plugin_iface * Set to `(uint32_t)offsetof(struct vk_bundle, vkGetInstanceProcAddr)`. */ uint32_t vk_bundle_fn_table_offset; + + /*! + * Create a D3D11 display processor that serves ONLY the 2D→3D lift slots + * (ADR-042, XR_DXR_lift) — the explicit "lift-only" factory. + * + * The runtime's D3D11 service creates exactly one lift DP per process, on + * a dedicated device on the service adapter (@p d3d11_context has + * ID3D11Multithread protection on), from its lift thread, and never asks + * it to weave: no process_atlas, no window (@p window_handle is always + * NULL), no mode requests. So the DP returned here must build NO weaver + * and open NO tracker session — only what its conversion module needs — + * and must fill the lift_* slots of @ref xrt_display_processor_d3d11 + * (XRT_DP_D3D11_HAS_LIFT). Same signature as @ref create_dp_d3d11. + * + * Why a separate factory rather than reusing the weaving DP: a conversion + * blocks for tens of ms (seconds for GAUSSIANS), and the weaving DP is + * driven on the service's one shared immediate context under the render + * lock and is recreated on presenter / focus changes. Running lift on it + * would stall every weave for the length of a conversion. + * + * Optional. NULL (or a plug-in whose `struct_size` predates this field) ⟹ + * the runtime falls back to @ref create_dp_d3d11 with a NULL window and, if + * that DP carries no lift slots, destroys it at once. Appended per ADR-020 + * (append-only within a major; gated by @ref struct_size; no + * XRT_PLUGIN_API_VERSION_CURRENT bump). + */ + xrt_dp_factory_d3d11_fn_t create_dp_d3d11_lift; }; +/*! + * Defined when @ref xrt_plugin_iface carries @ref + * xrt_plugin_iface::create_dp_d3d11_lift (ADR-042), so a plug-in built against + * an older runtime header can #ifdef-guard filling it. + */ +#define XRT_PLUGIN_IFACE_HAS_D3D11_LIFT_FACTORY 1 + /* * diff --git a/src/xrt/ipc/CMakeLists.txt b/src/xrt/ipc/CMakeLists.txt index 435d7e8ea..3ae3b601f 100644 --- a/src/xrt/ipc/CMakeLists.txt +++ b/src/xrt/ipc/CMakeLists.txt @@ -68,6 +68,8 @@ add_library( client/ipc_client.h client/ipc_client_compositor.c client/ipc_client_connection.c + client/ipc_client_lift.c + client/ipc_client_lift.h client/ipc_client_device.c client/ipc_client_hmd.c client/ipc_client_instance.c diff --git a/src/xrt/ipc/client/ipc_client_compositor.c b/src/xrt/ipc/client/ipc_client_compositor.c index c718d174c..52af8d9a7 100644 --- a/src/xrt/ipc/client/ipc_client_compositor.c +++ b/src/xrt/ipc/client/ipc_client_compositor.c @@ -29,6 +29,7 @@ #include "shared/ipc_protocol.h" #include "client/ipc_client.h" #include "client/ipc_client_connection.h" +#include "client/ipc_client_lift.h" #include "ipc_client_generated.h" // Phase 2.D: the workspace input-event bridge translates wire events into @@ -3027,3 +3028,121 @@ ipc_client_create_system_compositor(struct ipc_connection *ipc_c, return XRT_SUCCESS; } + + +/* + * XR_DXR_lift bridges (ADR-042) — thin accessors the OpenXR state tracker + * (oxr_lift.c) forward-declares, like the weave bridges above: extract the + * session's connection and forward to ipc_client_lift.c. Streams belong to the + * connection, so a session's streams die with its connection. + */ +static struct ipc_connection * +lift_conn(struct xrt_compositor *xc) +{ + if (xc == NULL) { + return NULL; + } + struct ipc_client_compositor *icc = ipc_client_compositor(xc); + return icc != NULL ? icc->ipc_c : NULL; +} + +xrt_result_t +comp_ipc_client_compositor_lift_get_properties(struct xrt_compositor *xc, struct xrt_dp_lift_caps *out_caps) +{ + return ipc_client_lift_get_properties(lift_conn(xc), out_caps); +} + +xrt_result_t +comp_ipc_client_compositor_lift_stream_create(struct xrt_compositor *xc, + uint32_t mode, + uint32_t content_hint, + float input_scale, + uint64_t *out_stream_id) +{ + return ipc_client_lift_stream_create(lift_conn(xc), mode, content_hint, input_scale, out_stream_id); +} + +xrt_result_t +comp_ipc_client_compositor_lift_stream_destroy(struct xrt_compositor *xc, uint64_t stream_id) +{ + return ipc_client_lift_stream_destroy(lift_conn(xc), stream_id); +} + +xrt_result_t +comp_ipc_client_compositor_lift_submit(struct xrt_compositor *xc, + uint64_t stream_id, + xrt_graphics_buffer_handle_t handle, + bool is_dxgi, + uint32_t width, + uint32_t height, + int64_t source_time, + const struct xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_count, + uint64_t *out_frame_id) +{ + return ipc_client_lift_submit(lift_conn(xc), stream_id, handle, is_dxgi, width, height, source_time, params, + viewpoints, viewpoint_count, out_frame_id); +} + +xrt_result_t +comp_ipc_client_compositor_lift_acquire(struct xrt_compositor *xc, + uint64_t stream_id, + bool *out_ready, + struct xrt_lift_result *out_result) +{ + return ipc_client_lift_acquire(lift_conn(xc), stream_id, out_ready, out_result); +} + +xrt_result_t +comp_ipc_client_compositor_lift_get_output(struct xrt_compositor *xc, + uint64_t stream_id, + bool *out_have, + xrt_graphics_buffer_handle_t *out_handle) +{ + return ipc_client_lift_get_output(lift_conn(xc), stream_id, out_have, NULL, NULL, NULL, out_handle); +} + +xrt_result_t +comp_ipc_client_compositor_lift_get_fence(struct xrt_compositor *xc, + uint64_t stream_id, + bool *out_have, + xrt_graphics_sync_handle_t *out_handle) +{ + return ipc_client_lift_get_fence(lift_conn(xc), stream_id, out_have, out_handle); +} + +xrt_result_t +comp_ipc_client_compositor_lift_acquire_blob(struct xrt_compositor *xc, + uint64_t stream_id, + uint64_t capacity, + uint8_t *out_bytes, + bool *out_ready, + bool *out_delivered, + struct xrt_lift_blob_info *out_info) +{ + return ipc_client_lift_acquire_blob(lift_conn(xc), stream_id, capacity, out_bytes, out_ready, out_delivered, + out_info); +} + +xrt_result_t +comp_ipc_client_compositor_lift_weave_rects(struct xrt_compositor *xc, + uint32_t count, + const struct xrt_lift_weave_rect *rects) +{ + return ipc_client_lift_weave_rects(lift_conn(xc), count, rects); +} + +xrt_result_t +comp_ipc_client_compositor_lift_set_priority(struct xrt_compositor *xc, uint64_t stream_id, uint32_t priority) +{ + return ipc_client_lift_set_priority(lift_conn(xc), stream_id, priority); +} + +xrt_result_t +comp_ipc_client_compositor_lift_stats(struct xrt_compositor *xc, + uint64_t stream_id, + struct xrt_lift_stream_stats *out_stats) +{ + return ipc_client_lift_stats(lift_conn(xc), stream_id, out_stats); +} diff --git a/src/xrt/ipc/client/ipc_client_lift.c b/src/xrt/ipc/client/ipc_client_lift.c new file mode 100644 index 000000000..801904d4f --- /dev/null +++ b/src/xrt/ipc/client/ipc_client_lift.c @@ -0,0 +1,288 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief XR_DXR_lift (ADR-042) client-side IPC calls (see ipc_client_lift.h). + * @ingroup ipc_client + */ + +#include "xrt/xrt_config_os.h" + +#include "util/u_misc.h" + +#include "shared/ipc_protocol.h" +#include "client/ipc_client.h" +#include "client/ipc_client_connection.h" +#include "client/ipc_client_lift.h" +#include "ipc_client_generated.h" + +#include +#include +#include + +static void +params_to_ipc(const struct xrt_dp_lift_params *p, + const float *viewpoints, + uint32_t viewpoint_count, + struct ipc_lift_params *out) +{ + U_ZERO(out); + out->convergence = p->convergence; + out->strength = p->strength; + out->inpaint = p->inpaint; + out->view_count = p->view_count; + out->focal_px = + p->struct_size >= offsetof(struct xrt_dp_lift_params, focal_px) + sizeof(float) ? p->focal_px : 0.0f; + if (viewpoints != NULL && viewpoint_count > 0) { + uint32_t n = viewpoint_count > IPC_LIFT_MAX_VIEWS ? IPC_LIFT_MAX_VIEWS : viewpoint_count; + out->viewpoint_count = n; + memcpy(out->viewpoints, viewpoints, (size_t)n * 3 * sizeof(float)); + } +} + +xrt_result_t +ipc_client_lift_get_properties(struct ipc_connection *ipc_c, struct xrt_dp_lift_caps *out_caps) +{ + xrt_dp_lift_caps_init(out_caps); + if (ipc_c == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + struct ipc_lift_properties props; + U_ZERO(&props); + xrt_result_t xret = ipc_call_lift_get_properties(ipc_c, &props); + if (xret != XRT_SUCCESS) { + return xret; + } + out_caps->modes = props.modes; + out_caps->max_streams = props.max_streams; + out_caps->max_views = props.max_views; + out_caps->depth_semantics = props.depth_semantics; + out_caps->state = props.state; + out_caps->typical_latency_ns = props.typical_latency_ns; + memcpy(out_caps->backend, props.backend, sizeof(out_caps->backend)); + out_caps->backend[sizeof(out_caps->backend) - 1] = '\0'; + return XRT_SUCCESS; +} + +xrt_result_t +ipc_client_lift_stream_create( + struct ipc_connection *ipc_c, uint32_t mode, uint32_t content_hint, float input_scale, uint64_t *out_stream_id) +{ + if (ipc_c == NULL || out_stream_id == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + return ipc_call_lift_stream_create(ipc_c, mode, content_hint, input_scale, out_stream_id); +} + +xrt_result_t +ipc_client_lift_stream_destroy(struct ipc_connection *ipc_c, uint64_t stream_id) +{ + if (ipc_c == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + return ipc_call_lift_stream_destroy(ipc_c, stream_id); +} + +xrt_result_t +ipc_client_lift_submit(struct ipc_connection *ipc_c, + uint64_t stream_id, + xrt_graphics_buffer_handle_t handle, + bool is_dxgi, + uint32_t width, + uint32_t height, + int64_t source_time, + const struct xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_count, + uint64_t *out_frame_id) +{ + if (ipc_c == NULL || out_frame_id == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + *out_frame_id = 0; + struct ipc_arg_lift_submit args; + U_ZERO(&args); + args.stream_id = stream_id; + args.source_time = source_time; + args.width = width; + args.height = height; + if (params != NULL) { + args.has_params = 1; + params_to_ipc(params, viewpoints, viewpoint_count, &args.params); + } + xrt_graphics_buffer_handle_t handles[1] = {handle}; +#if defined(XRT_GRAPHICS_BUFFER_HANDLE_IS_WIN32_HANDLE) + // Legacy DXGI handles cross raw, low-bit tagged (the weave_submit convention). + if (is_dxgi && handles[0] != XRT_GRAPHICS_BUFFER_HANDLE_INVALID) { + handles[0] = (void *)((size_t)handles[0] | 1); + } +#else + (void)is_dxgi; +#endif + return ipc_call_lift_submit_frame(ipc_c, &args, handles, 1, out_frame_id); +} + +xrt_result_t +ipc_client_lift_acquire(struct ipc_connection *ipc_c, + uint64_t stream_id, + bool *out_ready, + struct xrt_lift_result *out_result) +{ + if (ipc_c == NULL || out_ready == NULL || out_result == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + *out_ready = false; + U_ZERO(out_result); + struct ipc_lift_result r; + U_ZERO(&r); + bool ready = false; + xrt_result_t xret = ipc_call_lift_acquire_result(ipc_c, stream_id, &ready, &r); + if (xret != XRT_SUCCESS) { + return xret; + } + *out_ready = ready; + out_result->frame_id = r.frame_id; + out_result->source_time = r.source_time; + out_result->fence_value = r.fence_value; + out_result->latency_ns = r.latency_ns; + out_result->width = r.width; + out_result->height = r.height; + out_result->format = r.format; + out_result->view_count = r.view_count; + out_result->output_realloc = r.output_realloc != 0; + return XRT_SUCCESS; +} + +xrt_result_t +ipc_client_lift_get_output(struct ipc_connection *ipc_c, + uint64_t stream_id, + bool *out_have, + uint32_t *out_width, + uint32_t *out_height, + uint32_t *out_format, + xrt_graphics_buffer_handle_t *out_handle) +{ + if (ipc_c == NULL || out_have == NULL || out_handle == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + *out_have = false; + *out_handle = XRT_GRAPHICS_BUFFER_HANDLE_INVALID; + uint32_t w = 0, h = 0, f = 0; + xrt_result_t xret = ipc_call_lift_get_output(ipc_c, stream_id, out_have, &w, &h, &f, out_handle, 1); + if (out_width != NULL) { + *out_width = w; + } + if (out_height != NULL) { + *out_height = h; + } + if (out_format != NULL) { + *out_format = f; + } + return xret; +} + +xrt_result_t +ipc_client_lift_get_fence(struct ipc_connection *ipc_c, + uint64_t stream_id, + bool *out_have, + xrt_graphics_sync_handle_t *out_handle) +{ + if (ipc_c == NULL || out_have == NULL || out_handle == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + *out_have = false; + *out_handle = XRT_GRAPHICS_SYNC_HANDLE_INVALID; + return ipc_call_lift_get_fence(ipc_c, stream_id, out_have, out_handle, 1); +} + +xrt_result_t +ipc_client_lift_acquire_blob(struct ipc_connection *ipc_c, + uint64_t stream_id, + uint64_t capacity, + uint8_t *out_bytes, + bool *out_ready, + bool *out_delivered, + struct xrt_lift_blob_info *out_info) +{ + if (ipc_c == NULL || out_ready == NULL || out_delivered == NULL || out_info == NULL || + (capacity > 0 && out_bytes == NULL)) { + return XRT_ERROR_IPC_FAILURE; + } + *out_ready = false; + *out_delivered = false; + U_ZERO(out_info); + + ipc_client_connection_lock(ipc_c); + xrt_result_t xret = ipc_send_lift_acquire_blob_locked(ipc_c, stream_id, capacity); + if (xret != XRT_SUCCESS) { + ipc_client_connection_unlock(ipc_c); + return xret; + } + bool ready = false; + uint64_t frame_id = 0, byte_count = 0; + int64_t source_time = 0; + uint32_t format = 0; + xrt_result_t result = + ipc_receive_lift_acquire_blob_locked(ipc_c, &ready, &frame_id, &source_time, &format, &byte_count); + if (result != XRT_SUCCESS) { + // Receive failure (pipe) or a service-side rejection: no payload follows. + ipc_client_connection_unlock(ipc_c); + return result; + } + // The service sends the bytes exactly when it had them and capacity sufficed. + if (ready && byte_count > 0 && capacity >= byte_count) { + xret = ipc_receive(&ipc_c->imc, out_bytes, (size_t)byte_count); + *out_delivered = xret == XRT_SUCCESS; + } + ipc_client_connection_unlock(ipc_c); + if (xret != XRT_SUCCESS) { + return xret; + } + *out_ready = ready; + out_info->frame_id = frame_id; + out_info->source_time = source_time; + out_info->format = format; + out_info->byte_count = byte_count; + return XRT_SUCCESS; +} + +xrt_result_t +ipc_client_lift_weave_rects(struct ipc_connection *ipc_c, uint32_t count, const struct xrt_lift_weave_rect *rects) +{ + if (ipc_c == NULL || count > IPC_LIFT_WEAVE_RECTS_MAX || (count > 0 && rects == NULL)) { + return XRT_ERROR_IPC_FAILURE; + } + struct ipc_arg_lift_weave_rects args; + U_ZERO(&args); + args.count = count; + for (uint32_t i = 0; i < count; i++) { + args.rects[i].stream_id = rects[i].stream_id; + args.rects[i].rect_index = rects[i].rect_index; + args.rects[i].has_params = rects[i].has_params ? 1u : 0u; + args.rects[i].convergence = rects[i].params.convergence; + args.rects[i].strength = rects[i].params.strength; + args.rects[i].inpaint = rects[i].params.inpaint; + args.rects[i].view_count = rects[i].params.view_count; + args.rects[i].focal_px = rects[i].params.focal_px; + } + return ipc_call_lift_weave_rects(ipc_c, &args); +} + +xrt_result_t +ipc_client_lift_set_priority(struct ipc_connection *ipc_c, uint64_t stream_id, uint32_t priority) +{ + if (ipc_c == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + return ipc_call_lift_stream_set_priority(ipc_c, stream_id, priority); +} + +xrt_result_t +ipc_client_lift_stats(struct ipc_connection *ipc_c, uint64_t stream_id, struct xrt_lift_stream_stats *out_stats) +{ + if (ipc_c == NULL || out_stats == NULL) { + return XRT_ERROR_IPC_FAILURE; + } + U_ZERO(out_stats); + return ipc_call_lift_stream_stats(ipc_c, stream_id, out_stats); +} diff --git a/src/xrt/ipc/client/ipc_client_lift.h b/src/xrt/ipc/client/ipc_client_lift.h new file mode 100644 index 000000000..8fe026d78 --- /dev/null +++ b/src/xrt/ipc/client/ipc_client_lift.h @@ -0,0 +1,109 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief XR_DXR_lift (ADR-042) client side: connection-level calls shared by + * the OpenXR state tracker's bridges and `displayxr-cli lift`. + * + * Lift streams belong to the IPC connection, not a session, so every call here + * takes a bare @ref ipc_connection. Results use the xrt_lift.h value types. + * + * @ingroup ipc_client + */ + +#pragma once + +#include "xrt/xrt_handles.h" +#include "xrt/xrt_results.h" +#include "xrt/xrt_lift.h" + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +struct ipc_connection; + +xrt_result_t +ipc_client_lift_get_properties(struct ipc_connection *ipc_c, struct xrt_dp_lift_caps *out_caps); + +xrt_result_t +ipc_client_lift_stream_create( + struct ipc_connection *ipc_c, uint32_t mode, uint32_t content_hint, float input_scale, uint64_t *out_stream_id); + +xrt_result_t +ipc_client_lift_stream_destroy(struct ipc_connection *ipc_c, uint64_t stream_id); + +/*! + * Submit one frame. @p params NULL = the stream's last parameters. + * @p viewpoints holds @p viewpoint_count xyz triplets (EXPLICIT source), or + * NULL/0 (TRACKED). On Windows a legacy DXGI handle is low-bit tagged on the + * wire (@p is_dxgi), as for weave_submit. + */ +xrt_result_t +ipc_client_lift_submit(struct ipc_connection *ipc_c, + uint64_t stream_id, + xrt_graphics_buffer_handle_t handle, + bool is_dxgi, + uint32_t width, + uint32_t height, + int64_t source_time, + const struct xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_count, + uint64_t *out_frame_id); + +//! @p out_ready false = nothing newer than the last acquire (NOT READY). +xrt_result_t +ipc_client_lift_acquire(struct ipc_connection *ipc_c, + uint64_t stream_id, + bool *out_ready, + struct xrt_lift_result *out_result); + +//! The stream's export texture (a NEW handle, caller-owned) + its layout. +xrt_result_t +ipc_client_lift_get_output(struct ipc_connection *ipc_c, + uint64_t stream_id, + bool *out_have, + uint32_t *out_width, + uint32_t *out_height, + uint32_t *out_format, + xrt_graphics_buffer_handle_t *out_handle); + +//! The stream's export fence (a NEW handle, caller-owned). +xrt_result_t +ipc_client_lift_get_fence(struct ipc_connection *ipc_c, + uint64_t stream_id, + bool *out_have, + xrt_graphics_sync_handle_t *out_handle); + +/*! + * Two-call blob acquire. @p capacity 0 = size query (the service latches the + * blob); otherwise, when @p capacity >= byte_count, the bytes are written to + * @p out_bytes. @p out_delivered says whether they were. + */ +xrt_result_t +ipc_client_lift_acquire_blob(struct ipc_connection *ipc_c, + uint64_t stream_id, + uint64_t capacity, + uint8_t *out_bytes, + bool *out_ready, + bool *out_delivered, + struct xrt_lift_blob_info *out_info); + +//! XrLiftPriorityDXR (0 paused .. 3 high). +xrt_result_t +ipc_client_lift_set_priority(struct ipc_connection *ipc_c, uint64_t stream_id, uint32_t priority); + +xrt_result_t +ipc_client_lift_stats(struct ipc_connection *ipc_c, uint64_t stream_id, struct xrt_lift_stream_stats *out_stats); + +//! Latch the lift-flagged rects of the NEXT weave_submit on this connection. +xrt_result_t +ipc_client_lift_weave_rects(struct ipc_connection *ipc_c, uint32_t count, const struct xrt_lift_weave_rect *rects); + +#ifdef __cplusplus +} +#endif diff --git a/src/xrt/ipc/server/ipc_server.h b/src/xrt/ipc/server/ipc_server.h index bdce2fcb7..321adb306 100644 --- a/src/xrt/ipc/server/ipc_server.h +++ b/src/xrt/ipc/server/ipc_server.h @@ -205,6 +205,15 @@ struct ipc_client_state int weave_deferred_close_fds[4]; uint32_t weave_deferred_close_count; #endif + + /*! + * XR_DXR_lift (ADR-042): this connection's lift-stream owner token, taken + * from a process-wide counter on first lift use (0 = never used lift). A + * token, not the ics pointer: the thread slot is reused by later + * connections, and a stale stream must never resolve for them. Released — + * every stream it owns destroyed — at client teardown. + */ + uint64_t lift_owner; }; enum ipc_thread_state @@ -590,6 +599,13 @@ void ipc_server_client_weave_flush_deferred_fds(volatile struct ipc_client_state *ics); #endif +/*! + * XR_DXR_lift (ADR-042): destroy every lift stream this client created. Called + * once at client teardown; a no-op for a client that never used lift. + */ +void +ipc_server_client_lift_release(volatile struct ipc_client_state *ics); + /*! * @defgroup ipc_server_internals Server Internals * @brief These are only called by the platform-specific mainloop polling code. diff --git a/src/xrt/ipc/server/ipc_server_handler.c b/src/xrt/ipc/server/ipc_server_handler.c index 02cd518a6..cbc8de267 100644 --- a/src/xrt/ipc/server/ipc_server_handler.c +++ b/src/xrt/ipc/server/ipc_server_handler.c @@ -8140,3 +8140,459 @@ ipc_handle_device_set_brightness(volatile struct ipc_client_state *ics, uint32_t return xrt_device_set_brightness(xdev, brightness, relative); } + + +/* + * + * XR_DXR_lift (ADR-042) — 2D→3D conversion streams. + * + * Streams belong to the CONNECTION (ics->lift_owner), not to a session: the + * calls need no compositor, so `displayxr-cli lift` can drive them headless. + * Implemented on the Windows D3D11 service; every other service answers + * XRT_ERROR_FEATURE_NOT_SUPPORTED over a healthy pipe (the client reports + * XR_ERROR_FEATURE_UNSUPPORTED). A transient refusal (keyed-mutex miss) is + * XRT_ERROR_WEAVE_REFUSED, never XRT_ERROR_IPC_FAILURE — the latter would mark + * the caller's session lost over a perfectly healthy pipe (browser#103). + * + */ + +//! Who may use lift: the browser (PRESENT_OWNER), ordinary apps, and the +//! diagnostic probe (DIAG, `displayxr-cli lift`). +static xrt_result_t +require_lift_client(volatile struct ipc_client_state *ics, const char *what) +{ + uint32_t cls = ics->client_state.client_class; + if (cls != XRT_CLIENT_CLASS_PRESENT_OWNER && cls != XRT_CLIENT_CLASS_APP && cls != XRT_CLIENT_CLASS_DIAG) { + IPC_WARN(ics->server, "%s: denied — caller pid %ld is class %s (lift: PRESENT_OWNER / APP / DIAG).", what, + ics->peer_pid, ipc_server_client_class_str(cls)); + return XRT_ERROR_NOT_AUTHORIZED; + } + return XRT_SUCCESS; +} + +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) +//! This connection's owner token, taken on first use from a server-wide counter. +static uint64_t +lift_owner(volatile struct ipc_client_state *ics) +{ + if (ics->lift_owner == 0) { + static uint64_t s_next_owner = 0; + os_mutex_lock(&ics->server->global_state.lock); + ics->lift_owner = ++s_next_owner; + os_mutex_unlock(&ics->server->global_state.lock); + } + return ics->lift_owner; +} + +static struct xrt_system_compositor * +lift_xsysc(volatile struct ipc_client_state *ics) +{ + struct xrt_system_compositor *xsysc = ics->server != NULL ? ics->server->xsysc : NULL; + return comp_d3d11_service_is_d3d11_service(xsysc) ? xsysc : NULL; +} + +static void +lift_params_from_ipc(const struct ipc_lift_params *in, struct xrt_dp_lift_params *out) +{ + U_ZERO(out); + out->struct_size = (uint32_t)sizeof(*out); + out->convergence = in->convergence; + out->strength = in->strength; + out->inpaint = in->inpaint; + out->view_count = in->view_count > IPC_LIFT_MAX_VIEWS ? IPC_LIFT_MAX_VIEWS : in->view_count; + out->focal_px = in->focal_px; +} +#endif + +void +ipc_server_client_lift_release(volatile struct ipc_client_state *ics) +{ + if (ics == NULL || ics->lift_owner == 0) { + return; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc != NULL) { + comp_d3d11_service_lift_release_owner(xsysc, ics->lift_owner); + } +#endif + ics->lift_owner = 0; +} + +xrt_result_t +ipc_handle_lift_get_properties(volatile struct ipc_client_state *ics, struct ipc_lift_properties *out_props) +{ + IPC_TRACE_MARKER(); + U_ZERO(out_props); + xrt_result_t auth = require_lift_client(ics, "lift_get_properties"); + if (auth != XRT_SUCCESS) { + return auth; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc != NULL) { + struct xrt_dp_lift_caps caps; + comp_d3d11_service_lift_get_caps(xsysc, &caps); + out_props->modes = caps.modes; + out_props->max_streams = caps.max_streams; + out_props->max_views = caps.max_views; + out_props->depth_semantics = caps.depth_semantics; + out_props->state = caps.state; + out_props->typical_latency_ns = caps.typical_latency_ns; + memcpy(out_props->backend, caps.backend, sizeof(out_props->backend)); + out_props->backend[sizeof(out_props->backend) - 1] = '\0'; + } +#endif + // Anything else: modes 0, UNAVAILABLE — a successful answer, not an error. + return XRT_SUCCESS; +} + +xrt_result_t +ipc_handle_lift_stream_create(volatile struct ipc_client_state *ics, + uint32_t mode, + uint32_t content_hint, + float input_scale, + uint64_t *out_stream_id) +{ + IPC_TRACE_MARKER(); + *out_stream_id = 0; + xrt_result_t auth = require_lift_client(ics, "lift_stream_create"); + if (auth != XRT_SUCCESS) { + return auth; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc != NULL) { + struct xrt_dp_lift_stream_info info = {0}; + info.struct_size = (uint32_t)sizeof(info); + info.mode = mode; + info.content_hint = content_hint; + info.input_scale = input_scale; + return comp_d3d11_service_lift_stream_create(xsysc, lift_owner(ics), &info, out_stream_id); + } +#else + (void)mode; + (void)content_hint; + (void)input_scale; +#endif + return XRT_ERROR_FEATURE_NOT_SUPPORTED; +} + +xrt_result_t +ipc_handle_lift_stream_destroy(volatile struct ipc_client_state *ics, uint64_t stream_id) +{ + IPC_TRACE_MARKER(); + xrt_result_t auth = require_lift_client(ics, "lift_stream_destroy"); + if (auth != XRT_SUCCESS) { + return auth; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc != NULL && ics->lift_owner != 0) { + comp_d3d11_service_lift_stream_destroy(xsysc, ics->lift_owner, stream_id); + } +#else + (void)stream_id; +#endif + return XRT_SUCCESS; +} + +xrt_result_t +ipc_handle_lift_submit_frame(volatile struct ipc_client_state *ics, + const struct ipc_arg_lift_submit *args, + uint64_t *out_frame_id, + const xrt_graphics_buffer_handle_t *handles, + uint32_t handle_count) +{ + IPC_TRACE_MARKER(); + *out_frame_id = 0; + xrt_result_t auth = require_lift_client(ics, "lift_submit_frame"); + if (auth != XRT_SUCCESS) { + weave_submit_release_handles(handles, handle_count); + return auth; + } + if (handle_count < 1) { + return XRT_ERROR_IPC_FAILURE; // malformed request: the wire is not trusted + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + xrt_graphics_buffer_handle_t in_handle = handles[0]; + bool in_is_dxgi = false; +#if defined(XRT_GRAPHICS_BUFFER_HANDLE_IS_WIN32_HANDLE) + if ((size_t)in_handle & 1) { + in_handle = (HANDLE)((size_t)in_handle - 1); + in_is_dxgi = true; + } +#endif + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc == NULL) { + if (!in_is_dxgi && in_handle != NULL) { + CloseHandle(in_handle); + } + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + struct xrt_dp_lift_params params; + lift_params_from_ipc(&args->params, ¶ms); + uint32_t vp_count = args->params.viewpoint_count > IPC_LIFT_MAX_VIEWS ? IPC_LIFT_MAX_VIEWS + : args->params.viewpoint_count; + // The handle is the service's from here (closed or cached inside). + return comp_d3d11_service_lift_submit(xsysc, lift_owner(ics), args->stream_id, in_handle, in_is_dxgi, + args->width, args->height, args->source_time, + args->has_params ? ¶ms : NULL, args->params.viewpoints, + args->has_params ? 3 * vp_count : 0, out_frame_id); +#else + (void)args; + weave_submit_release_handles(handles, handle_count); + return XRT_ERROR_FEATURE_NOT_SUPPORTED; +#endif +} + +xrt_result_t +ipc_handle_lift_acquire_result(volatile struct ipc_client_state *ics, + uint64_t stream_id, + bool *out_ready, + struct ipc_lift_result *out_lift_result) +{ + IPC_TRACE_MARKER(); + *out_ready = false; + U_ZERO(out_lift_result); + xrt_result_t auth = require_lift_client(ics, "lift_acquire_result"); + if (auth != XRT_SUCCESS) { + return auth; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc == NULL) { + return XRT_ERROR_FEATURE_NOT_SUPPORTED; + } + struct xrt_lift_result r; + xrt_result_t xret = comp_d3d11_service_lift_acquire(xsysc, lift_owner(ics), stream_id, out_ready, &r); + out_lift_result->frame_id = r.frame_id; + out_lift_result->source_time = r.source_time; + out_lift_result->fence_value = r.fence_value; + out_lift_result->latency_ns = r.latency_ns; + out_lift_result->width = r.width; + out_lift_result->height = r.height; + out_lift_result->format = r.format; + out_lift_result->view_count = r.view_count; + out_lift_result->output_realloc = r.output_realloc ? 1u : 0u; + return xret; +#else + (void)stream_id; + return XRT_ERROR_FEATURE_NOT_SUPPORTED; +#endif +} + +xrt_result_t +ipc_handle_lift_get_output(volatile struct ipc_client_state *ics, + uint64_t stream_id, + bool *out_have_output, + uint32_t *out_width, + uint32_t *out_height, + uint32_t *out_format, + uint32_t max_handle_count, + xrt_graphics_buffer_handle_t *out_handles, + uint32_t *out_handle_count) +{ + IPC_TRACE_MARKER(); + *out_have_output = false; + *out_width = 0; + *out_height = 0; + *out_format = 0; + *out_handle_count = 0; + xrt_result_t auth = require_lift_client(ics, "lift_get_output"); + if (auth != XRT_SUCCESS) { + return auth; + } + if (max_handle_count < 1) { + return XRT_SUCCESS; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + xrt_graphics_buffer_handle_t h = XRT_GRAPHICS_BUFFER_HANDLE_INVALID; + if (xsysc != NULL && comp_d3d11_service_lift_export_output(xsysc, lift_owner(ics), stream_id, &h, out_width, + out_height, out_format)) { + // Service-owned: the transport DuplicateHandle's it into the caller. + out_handles[0] = h; + *out_handle_count = 1; + *out_have_output = true; + } +#else + (void)stream_id; + (void)out_handles; +#endif + return XRT_SUCCESS; +} + +xrt_result_t +ipc_handle_lift_get_fence(volatile struct ipc_client_state *ics, + uint64_t stream_id, + bool *out_have_fence, + uint32_t max_handle_count, + xrt_graphics_sync_handle_t *out_handles, + uint32_t *out_handle_count) +{ + IPC_TRACE_MARKER(); + *out_have_fence = false; + *out_handle_count = 0; + xrt_result_t auth = require_lift_client(ics, "lift_get_fence"); + if (auth != XRT_SUCCESS) { + return auth; + } + if (max_handle_count < 1) { + return XRT_SUCCESS; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + xrt_graphics_sync_handle_t h = XRT_GRAPHICS_SYNC_HANDLE_INVALID; + if (xsysc != NULL && comp_d3d11_service_lift_export_fence(xsysc, lift_owner(ics), stream_id, &h)) { + out_handles[0] = h; + *out_handle_count = 1; + *out_have_fence = true; + } +#else + (void)stream_id; + (void)out_handles; +#endif + return XRT_SUCCESS; +} + +/*! + * Varlen reply: struct ipc_lift_acquire_blob_reply {result, ready, frame_id, + * source_time, format, byte_count}, then — only when result is XRT_SUCCESS, + * ready, and capacity >= byte_count — byte_count bytes in ONE send. With a + * smaller capacity only the header goes (the blob stays latched service-side + * for the second call). Rejections ride in reply.result over a healthy pipe. + */ +xrt_result_t +ipc_handle_lift_acquire_blob(volatile struct ipc_client_state *ics, uint64_t stream_id, uint64_t capacity) +{ + IPC_TRACE_MARKER(); + struct ipc_message_channel *imc = (struct ipc_message_channel *)&ics->imc; + struct ipc_lift_acquire_blob_reply reply = XRT_STRUCT_INIT; + reply.result = require_lift_client(ics, "lift_acquire_blob"); + uint8_t *bytes = NULL; + +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + if (reply.result == XRT_SUCCESS) { + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc == NULL) { + reply.result = XRT_ERROR_FEATURE_NOT_SUPPORTED; + } else { + uint64_t cap = capacity > IPC_LIFT_BLOB_MAX_BYTES ? IPC_LIFT_BLOB_MAX_BYTES : capacity; + struct xrt_lift_blob_info info; + bool ready = false; + reply.result = comp_d3d11_service_lift_acquire_blob(xsysc, lift_owner(ics), stream_id, cap, &ready, + &info, &bytes); + reply.ready = ready; + reply.frame_id = info.frame_id; + reply.source_time = info.source_time; + reply.format = info.format; + reply.byte_count = info.byte_count; + if (reply.result == XRT_SUCCESS && info.byte_count > IPC_LIFT_BLOB_MAX_BYTES) { + // Too big for one reply: report it, deliver nothing. + reply.result = XRT_ERROR_ALLOCATION; + free(bytes); + bytes = NULL; + } + } + } +#else + (void)stream_id; + (void)capacity; + if (reply.result == XRT_SUCCESS) { + reply.result = XRT_ERROR_FEATURE_NOT_SUPPORTED; + } +#endif + + xrt_result_t xret = ipc_send(imc, &reply, sizeof(reply)); + if (xret == XRT_SUCCESS && reply.result == XRT_SUCCESS && bytes != NULL && reply.byte_count > 0) { + xret = ipc_send(imc, bytes, (size_t)reply.byte_count); + } + if (xret != XRT_SUCCESS) { + IPC_ERROR(ics->server, "lift_acquire_blob: failed to send the reply"); + } + free(bytes); + return xret; +} + +xrt_result_t +ipc_handle_lift_weave_rects(volatile struct ipc_client_state *ics, const struct ipc_arg_lift_weave_rects *args) +{ + IPC_TRACE_MARKER(); + xrt_result_t auth = require_present_owner(ics, "lift_weave_rects"); + if (auth != XRT_SUCCESS) { + return auth; + } + if (ics->xc == NULL) { + return XRT_ERROR_IPC_SESSION_NOT_CREATED; + } + if (args->count > IPC_LIFT_WEAVE_RECTS_MAX) { + return XRT_ERROR_IPC_FAILURE; // untrusted wire + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_lift_weave_rect rects[IPC_LIFT_WEAVE_RECTS_MAX]; + for (uint32_t i = 0; i < args->count; i++) { + const struct ipc_lift_weave_rect *w = &args->rects[i]; + U_ZERO(&rects[i]); + rects[i].stream_id = w->stream_id; + rects[i].rect_index = w->rect_index; + rects[i].has_params = w->has_params != 0; + rects[i].params.struct_size = (uint32_t)sizeof(rects[i].params); + rects[i].params.convergence = w->convergence; + rects[i].params.strength = w->strength; + rects[i].params.inpaint = w->inpaint; + rects[i].params.view_count = w->view_count > IPC_LIFT_MAX_VIEWS ? IPC_LIFT_MAX_VIEWS : w->view_count; + rects[i].params.focal_px = w->focal_px; + } + if (!comp_d3d11_service_lift_set_weave_rects(ics->xc, lift_owner(ics), args->count, rects)) { + // A rect naming a stream this connection does not own (or a non-SBS/NVIEW + // one): refused, non-fatal. The submit then weaves those rects as drawn. + return XRT_ERROR_WEAVE_REFUSED; + } + return XRT_SUCCESS; +#else + return args->count == 0 ? XRT_SUCCESS : XRT_ERROR_FEATURE_NOT_SUPPORTED; +#endif +} + +xrt_result_t +ipc_handle_lift_stream_set_priority(volatile struct ipc_client_state *ics, uint64_t stream_id, uint32_t priority) +{ + IPC_TRACE_MARKER(); + xrt_result_t auth = require_lift_client(ics, "lift_stream_set_priority"); + if (auth != XRT_SUCCESS) { + return auth; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc != NULL) { + return comp_d3d11_service_lift_set_priority(xsysc, lift_owner(ics), stream_id, priority); + } +#else + (void)stream_id; + (void)priority; +#endif + return XRT_ERROR_FEATURE_NOT_SUPPORTED; +} + +xrt_result_t +ipc_handle_lift_stream_stats(volatile struct ipc_client_state *ics, + uint64_t stream_id, + struct xrt_lift_stream_stats *out_stats) +{ + IPC_TRACE_MARKER(); + U_ZERO(out_stats); + xrt_result_t auth = require_lift_client(ics, "lift_stream_stats"); + if (auth != XRT_SUCCESS) { + return auth; + } +#if defined(XRT_HAVE_D3D11_SERVICE_COMPOSITOR) + struct xrt_system_compositor *xsysc = lift_xsysc(ics); + if (xsysc != NULL) { + return comp_d3d11_service_lift_get_stats(xsysc, lift_owner(ics), stream_id, out_stats); + } +#else + (void)stream_id; +#endif + return XRT_ERROR_FEATURE_NOT_SUPPORTED; +} diff --git a/src/xrt/ipc/server/ipc_server_per_client_thread.c b/src/xrt/ipc/server/ipc_server_per_client_thread.c index c531dfe49..8701188df 100644 --- a/src/xrt/ipc/server/ipc_server_per_client_thread.c +++ b/src/xrt/ipc/server/ipc_server_per_client_thread.c @@ -124,6 +124,9 @@ common_shutdown(volatile struct ipc_client_state *ics) ipc_server_client_weave_flush_deferred_fds(ics); #endif + // XR_DXR_lift (ADR-042): streams are owned by the connection, not a session. + ipc_server_client_lift_release(ics); + // Make sure undestroyed spaces are unreferenced for (uint32_t i = 0; i < IPC_MAX_CLIENT_SPACES; i++) { // Cast away volatile. diff --git a/src/xrt/ipc/shared/ipc_protocol.h b/src/xrt/ipc/shared/ipc_protocol.h index d11b97881..f8d5a985c 100644 --- a/src/xrt/ipc/shared/ipc_protocol.h +++ b/src/xrt/ipc/shared/ipc_protocol.h @@ -11,6 +11,7 @@ #pragma once +#include "xrt/xrt_lift.h" #include "xrt/xrt_limits.h" #include "xrt/xrt_compiler.h" #include "xrt/xrt_compositor.h" @@ -988,6 +989,130 @@ struct ipc_weave_dmabuf_output uint32_t strides[IPC_WEAVE_DMABUF_MAX_PLANES]; }; +/* + * + * XR_DXR_lift (ADR-042) — 2D→3D conversion streams over IPC. + * + * Streams are owned by the IPC CLIENT (not a session): a stream created on a + * connection dies with it. The lift calls need no compositor session, so a + * headless probe (`displayxr-cli lift`) can drive them on a raw connection. + * + */ + +//! Max explicit viewpoints on the wire. Mirrors XR_LIFT_MAX_VIEWS_DXR. +#define IPC_LIFT_MAX_VIEWS 8 +//! Max lifted weave rects per submit. Mirrors XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR. +#define IPC_LIFT_WEAVE_RECTS_MAX 8 +//! Largest blob the service will send in one lift_acquire_blob reply. +#define IPC_LIFT_BLOB_MAX_BYTES (256u * 1024u * 1024u) + +/*! + * The conversion module's capabilities (xrGetLiftPropertiesDXR). Field values + * are the DP-side encoding (xrt_dp_lift.h), which XR_DXR_lift mirrors. + * + * @ingroup ipc + */ +struct ipc_lift_properties +{ + uint32_t modes; + uint32_t max_streams; + uint32_t max_views; + uint32_t depth_semantics; + uint32_t state; + uint32_t reserved; + uint64_t typical_latency_ns; + char backend[32]; +}; + +/*! + * Per-frame conversion parameters on the wire (XrLiftOptionsDXR). + * + * @ingroup ipc + */ +struct ipc_lift_params +{ + float convergence; //!< < 0 = auto + float strength; + uint32_t inpaint; + uint32_t view_count; //!< 0 = module default + uint32_t viewpoint_count; //!< 0 = tracked eyes; else explicit, <= IPC_LIFT_MAX_VIEWS + float focal_px; //!< input focal length, input pixels; <= 0 = unknown + float viewpoints[3 * IPC_LIFT_MAX_VIEWS]; //!< xyz, display space, metres +}; + +/*! + * lift_submit_frame arguments. The input texture rides as in_handle[0] (a + * D3D11 shared handle; legacy DXGI handles low-bit tagged, as weave_submit). + * + * @ingroup ipc + */ +struct ipc_arg_lift_submit +{ + uint64_t stream_id; + int64_t source_time; //!< caller's timestamp, echoed on the result + uint32_t width; //!< region of the input to convert, from (0,0) + uint32_t height; + uint32_t has_params; //!< 0 = the stream's last / default parameters + uint32_t reserved; + struct ipc_lift_params params; +}; + +/*! + * One finished texture result (xrAcquireLiftResultDXR). The shared texture and + * fence travel separately (lift_get_output / lift_get_fence), only when + * @c output_realloc says the caller needs them (first acquire, size/format + * change) — the weave output pattern. + * + * @ingroup ipc + */ +struct ipc_lift_result +{ + uint64_t frame_id; + int64_t source_time; + uint64_t fence_value; + uint64_t latency_ns; + uint32_t width; + uint32_t height; + uint32_t format; //!< DXGI_FORMAT + uint32_t view_count; + uint32_t output_realloc; //!< 1 = the export texture changed since the last acquire + uint32_t reserved; +}; + +/*! + * One lift-flagged weave rect. Weave-path lifts always synthesize for the + * tracked eyes, so no viewpoints ride here (keeps 8 rects well inside + * IPC_BUF_SIZE). + * + * @ingroup ipc + */ +struct ipc_lift_weave_rect +{ + uint64_t stream_id; + uint32_t rect_index; //!< index into the NEXT weave_submit's rects[] + uint32_t has_params; + float convergence; + float strength; + uint32_t inpaint; + uint32_t view_count; + float focal_px; + uint32_t reserved; +}; + +/*! + * lift_weave_rects arguments: the lift-flagged rects of the weave_submit that + * immediately follows on the same connection (count 0 = none; the service + * consumes the set with that submit). + * + * @ingroup ipc + */ +struct ipc_arg_lift_weave_rects +{ + uint32_t count; + uint32_t reserved; + struct ipc_lift_weave_rect rects[IPC_LIFT_WEAVE_RECTS_MAX]; +}; + /*! * XR_DXR_weave v12 (browser-pvt#180): the origin this submit's woven output was * woven for, replied by weave_submit_dmabuf as its LAST out field (after the diff --git a/src/xrt/ipc/shared/proto.json b/src/xrt/ipc/shared/proto.json index 4ea457cc5..67c40f178 100644 --- a/src/xrt/ipc/shared/proto.json +++ b/src/xrt/ipc/shared/proto.json @@ -827,6 +827,109 @@ ] }, + "lift_get_properties": { + "out": [ + {"name": "props", "type": "struct ipc_lift_properties"} + ] + }, + + "lift_stream_create": { + "in": [ + {"name": "mode", "type": "uint32_t"}, + {"name": "content_hint", "type": "uint32_t"}, + {"name": "input_scale", "type": "float"} + ], + "out": [ + {"name": "stream_id", "type": "uint64_t"} + ] + }, + + "lift_stream_destroy": { + "in": [ + {"name": "stream_id", "type": "uint64_t"} + ] + }, + + "lift_submit_frame": { + "in": [ + {"name": "args", "type": "struct ipc_arg_lift_submit"} + ], + "out": [ + {"name": "frame_id", "type": "uint64_t"} + ], + "in_handles": {"type": "xrt_graphics_buffer_handle_t"} + }, + + "lift_acquire_result": { + "in": [ + {"name": "stream_id", "type": "uint64_t"} + ], + "out": [ + {"name": "ready", "type": "bool"}, + {"name": "lift_result", "type": "struct ipc_lift_result"} + ] + }, + + "lift_get_output": { + "in": [ + {"name": "stream_id", "type": "uint64_t"} + ], + "out": [ + {"name": "have_output", "type": "bool"}, + {"name": "width", "type": "uint32_t"}, + {"name": "height", "type": "uint32_t"}, + {"name": "format", "type": "uint32_t"} + ], + "out_handles": {"type": "xrt_graphics_buffer_handle_t"} + }, + + "lift_get_fence": { + "in": [ + {"name": "stream_id", "type": "uint64_t"} + ], + "out": [ + {"name": "have_fence", "type": "bool"} + ], + "out_handles": {"type": "xrt_graphics_sync_handle_t"} + }, + + "lift_acquire_blob": { + "varlen": true, + "in": [ + {"name": "stream_id", "type": "uint64_t"}, + {"name": "capacity", "type": "uint64_t"} + ], + "out": [ + {"name": "ready", "type": "bool"}, + {"name": "frame_id", "type": "uint64_t"}, + {"name": "source_time", "type": "int64_t"}, + {"name": "format", "type": "uint32_t"}, + {"name": "byte_count", "type": "uint64_t"} + ] + }, + + "lift_stream_set_priority": { + "in": [ + {"name": "stream_id", "type": "uint64_t"}, + {"name": "priority", "type": "uint32_t"} + ] + }, + + "lift_stream_stats": { + "in": [ + {"name": "stream_id", "type": "uint64_t"} + ], + "out": [ + {"name": "stats", "type": "struct xrt_lift_stream_stats"} + ] + }, + + "lift_weave_rects": { + "in": [ + {"name": "args", "type": "struct ipc_arg_lift_weave_rects"} + ] + }, + "compositor_semaphore_destroy": { "in": [ {"name": "id", "type": "uint32_t"} diff --git a/src/xrt/state_trackers/oxr/CMakeLists.txt b/src/xrt/state_trackers/oxr/CMakeLists.txt index 41dfa7248..a141f7efa 100644 --- a/src/xrt/state_trackers/oxr/CMakeLists.txt +++ b/src/xrt/state_trackers/oxr/CMakeLists.txt @@ -60,6 +60,7 @@ add_library( oxr_views_change.c oxr_views_change.h oxr_weave.c + oxr_lift.c oxr_workspace.c oxr_workspace_modal_win32.c oxr_workspace_file_dialog.c diff --git a/src/xrt/state_trackers/oxr/oxr_api_funcs.h b/src/xrt/state_trackers/oxr/oxr_api_funcs.h index 0db9dbe30..7774d11ef 100644 --- a/src/xrt/state_trackers/oxr/oxr_api_funcs.h +++ b/src/xrt/state_trackers/oxr/oxr_api_funcs.h @@ -1078,6 +1078,40 @@ XRAPI_ATTR XrResult XRAPI_CALL oxr_xrWeaveExportIpcConnectionDXR(XrInstance instance, XrWeaveIpcConnectionDXR *connection); #endif +#ifdef OXR_HAVE_DXR_lift +//! OpenXR API function @ep{xrGetLiftPropertiesDXR} +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrGetLiftPropertiesDXR(XrSession session, XrLiftPropertiesDXR *properties); + +//! OpenXR API function @ep{xrCreateLiftStreamDXR} +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrCreateLiftStreamDXR(XrSession session, const XrLiftStreamCreateInfoDXR *createInfo, XrLiftStreamDXR *stream); + +//! OpenXR API function @ep{xrDestroyLiftStreamDXR} +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrDestroyLiftStreamDXR(XrLiftStreamDXR stream); + +//! OpenXR API function @ep{xrSubmitLiftFrameDXR} +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrSubmitLiftFrameDXR(XrLiftStreamDXR stream, const XrLiftFrameSubmitInfoDXR *submitInfo, uint64_t *frameId); + +//! OpenXR API function @ep{xrAcquireLiftResultDXR} +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrAcquireLiftResultDXR(XrLiftStreamDXR stream, XrLiftResultDXR *result); + +//! OpenXR API function @ep{xrAcquireLiftBlobDXR} +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrAcquireLiftBlobDXR(XrLiftStreamDXR stream, XrLiftBlobDXR *blob); + +//! OpenXR API function @ep{xrSetLiftStreamPriorityDXR} +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrSetLiftStreamPriorityDXR(XrLiftStreamDXR stream, XrLiftPriorityDXR priority); + +//! OpenXR API function @ep{xrGetLiftStreamStatsDXR} +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrGetLiftStreamStatsDXR(XrLiftStreamDXR stream, XrLiftStreamStatsDXR *stats); +#endif + #ifdef OXR_HAVE_EXT_conformance_automation //! OpenXR API function @ep{xrSetInputDeviceActiveEXT} XRAPI_ATTR XrResult XRAPI_CALL diff --git a/src/xrt/state_trackers/oxr/oxr_api_negotiate.c b/src/xrt/state_trackers/oxr/oxr_api_negotiate.c index 10df85277..0a8fe9634 100644 --- a/src/xrt/state_trackers/oxr/oxr_api_negotiate.c +++ b/src/xrt/state_trackers/oxr/oxr_api_negotiate.c @@ -516,6 +516,16 @@ handle_non_null(struct oxr_instance *inst, struct oxr_logger *log, const char *n ENTRY_IF_EXT(xrWeaveSetScreenFlatRegionsDXR, DXR_weave); ENTRY_IF_EXT(xrWeaveExportIpcConnectionDXR, DXR_weave); #endif +#ifdef OXR_HAVE_DXR_lift + ENTRY_IF_EXT(xrGetLiftPropertiesDXR, DXR_lift); + ENTRY_IF_EXT(xrCreateLiftStreamDXR, DXR_lift); + ENTRY_IF_EXT(xrDestroyLiftStreamDXR, DXR_lift); + ENTRY_IF_EXT(xrSubmitLiftFrameDXR, DXR_lift); + ENTRY_IF_EXT(xrAcquireLiftResultDXR, DXR_lift); + ENTRY_IF_EXT(xrAcquireLiftBlobDXR, DXR_lift); + ENTRY_IF_EXT(xrSetLiftStreamPriorityDXR, DXR_lift); + ENTRY_IF_EXT(xrGetLiftStreamStatsDXR, DXR_lift); +#endif #ifdef OXR_HAVE_DXR_workspace_file_dialog ENTRY_IF_EXT(xrRequestFilePickerDXR, DXR_workspace_file_dialog); diff --git a/src/xrt/state_trackers/oxr/oxr_api_verify.h b/src/xrt/state_trackers/oxr/oxr_api_verify.h index f27d66d44..8ea5037cb 100644 --- a/src/xrt/state_trackers/oxr/oxr_api_verify.h +++ b/src/xrt/state_trackers/oxr/oxr_api_verify.h @@ -94,6 +94,8 @@ struct oxr_subaction_paths; OXR_VERIFY_AND_SET_AND_INIT(log, thing, new_thing, oxr_plane_detector_ext, PLANEDET, name, new_thing->sess->sys->inst) #define OXR_VERIFY_LOCAL_3D_ZONE_AND_INIT_LOG(log, thing, new_thing, name) \ OXR_VERIFY_AND_SET_AND_INIT(log, thing, new_thing, oxr_local_3d_zone_ext, LOCAL3DZONE, name, new_thing->sess->sys->inst) +#define OXR_VERIFY_LIFT_STREAM_AND_INIT_LOG(log, thing, new_thing, name) \ + OXR_VERIFY_AND_SET_AND_INIT(log, thing, new_thing, oxr_lift_stream_dxr, LIFTSTREAM, name, new_thing->sess->sys->inst) // clang-format on #define OXR_VERIFY_INSTANCE_NOT_NULL(log, arg, new_arg) OXR_VERIFY_SET(log, arg, new_arg, oxr_instance, INSTANCE); diff --git a/src/xrt/state_trackers/oxr/oxr_defines.h b/src/xrt/state_trackers/oxr/oxr_defines.h index 6422f112c..dbb77e62d 100644 --- a/src/xrt/state_trackers/oxr/oxr_defines.h +++ b/src/xrt/state_trackers/oxr/oxr_defines.h @@ -36,6 +36,8 @@ #define OXR_XR_DEBUG_PLANEDET (*(uint64_t *)"oxrplan\0") // local 3D zone mask #define OXR_XR_DEBUG_LOCAL3DZONE (*(uint64_t *)"oxrl3dz\0") +// XR_DXR_lift stream (ADR-042) +#define OXR_XR_DEBUG_LIFTSTREAM (*(uint64_t *)"oxrlift\0") // clang-format on /*! diff --git a/src/xrt/state_trackers/oxr/oxr_extension_support.h b/src/xrt/state_trackers/oxr/oxr_extension_support.h index 8c21f3b69..3574034b8 100644 --- a/src/xrt/state_trackers/oxr/oxr_extension_support.h +++ b/src/xrt/state_trackers/oxr/oxr_extension_support.h @@ -749,6 +749,26 @@ #endif +/* + * XR_DXR_lift + * + * Hand-added DisplayXR extension (generate_oxr_ext_support.py knows nothing of + * the DisplayXR blocks — keep them when regenerating). 2D→3D conversion streams + * (ADR-042). Advertised on every desktop platform the weave service is (it is + * an IPC-only service like XR_DXR_weave); the conversion module itself is + * implemented by the Windows D3D11 service, and everywhere else + * xrGetLiftPropertiesDXR honestly reports supportedModes 0 / UNAVAILABLE. + */ +#if defined(XR_DXR_lift) && (defined(XR_USE_PLATFORM_WIN32) || defined(XR_USE_PLATFORM_MACOS) || \ + (defined(XRT_OS_LINUX) && !defined(XRT_OS_ANDROID))) +#define OXR_HAVE_DXR_lift +#define OXR_EXTENSION_SUPPORT_DXR_lift(_) \ + _(DXR_lift, DXR_LIFT) +#else +#define OXR_EXTENSION_SUPPORT_DXR_lift(_) +#endif + + /* * XR_DXR_workspace_file_dialog */ @@ -1261,6 +1281,7 @@ OXR_EXTENSION_SUPPORT_DXR_depth_budget(_) \ OXR_EXTENSION_SUPPORT_DXR_display_zones(_) \ OXR_EXTENSION_SUPPORT_DXR_weave(_) \ + OXR_EXTENSION_SUPPORT_DXR_lift(_) \ OXR_EXTENSION_SUPPORT_DXR_workspace_file_dialog(_) \ OXR_EXTENSION_SUPPORT_DXR_mcp_tools(_) \ OXR_EXTENSION_SUPPORT_BD_controller_interaction(_) \ diff --git a/src/xrt/state_trackers/oxr/oxr_lift.c b/src/xrt/state_trackers/oxr/oxr_lift.c new file mode 100644 index 000000000..be32b00f4 --- /dev/null +++ b/src/xrt/state_trackers/oxr/oxr_lift.c @@ -0,0 +1,531 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief XR_DXR_lift API entry points — 2D→3D conversion streams (ADR-042). + * @author David Fattal + * @ingroup oxr_api + * + * The conversion runs in the service (d3d11_lift.cpp, its own thread and + * device); these entry points validate, forward to thin IPC-client bridges + * (ipc_client_compositor.c — st_oxr does not pull the ipc_client include path, + * so the symbols resolve at link time, the oxr_weave.c pattern) and translate + * results. IPC-only, like XR_DXR_weave: an in-process session reports + * XR_ERROR_FEATURE_UNSUPPORTED from every entry point. + * + * Error contract (mirrors XR_DXR_weave §4b): + * - a dead pipe (XRT_ERROR_IPC_FAILURE) marks the session lost → + * XR_ERROR_INSTANCE_LOST, then XR_ERROR_SESSION_LOST; + * - a transient service refusal (XRT_ERROR_WEAVE_REFUSED — keyed-mutex miss) + * is a non-fatal XR_ERROR_RUNTIME_FAILURE, retry next frame; + * - no module / mode unsupported (XRT_ERROR_FEATURE_NOT_SUPPORTED) is + * XR_ERROR_FEATURE_UNSUPPORTED, permanent for that mode. + */ + +#include "oxr_objects.h" +#include "oxr_logger.h" +#include "oxr_xret.h" +#include "oxr_handle.h" +#include "oxr_api_funcs.h" +#include "oxr_api_verify.h" +#include "oxr_chain.h" + +#include "util/u_misc.h" +#include "util/u_trace_marker.h" +#include "util/u_logging.h" + +#include "xrt/xrt_lift.h" + +#include +#include + +#ifdef OXR_HAVE_DXR_lift + +// IPC-bridge wrappers (defined in ipc_client_compositor.c). +xrt_result_t +comp_ipc_client_compositor_lift_get_properties(struct xrt_compositor *xc, struct xrt_dp_lift_caps *out_caps); +xrt_result_t +comp_ipc_client_compositor_lift_stream_create( + struct xrt_compositor *xc, uint32_t mode, uint32_t content_hint, float input_scale, uint64_t *out_stream_id); +xrt_result_t +comp_ipc_client_compositor_lift_stream_destroy(struct xrt_compositor *xc, uint64_t stream_id); +xrt_result_t +comp_ipc_client_compositor_lift_submit(struct xrt_compositor *xc, + uint64_t stream_id, + xrt_graphics_buffer_handle_t handle, + bool is_dxgi, + uint32_t width, + uint32_t height, + int64_t source_time, + const struct xrt_dp_lift_params *params, + const float *viewpoints, + uint32_t viewpoint_count, + uint64_t *out_frame_id); +xrt_result_t +comp_ipc_client_compositor_lift_acquire(struct xrt_compositor *xc, + uint64_t stream_id, + bool *out_ready, + struct xrt_lift_result *out_result); +xrt_result_t +comp_ipc_client_compositor_lift_get_output(struct xrt_compositor *xc, + uint64_t stream_id, + bool *out_have, + xrt_graphics_buffer_handle_t *out_handle); +xrt_result_t +comp_ipc_client_compositor_lift_get_fence(struct xrt_compositor *xc, + uint64_t stream_id, + bool *out_have, + xrt_graphics_sync_handle_t *out_handle); +xrt_result_t +comp_ipc_client_compositor_lift_acquire_blob(struct xrt_compositor *xc, + uint64_t stream_id, + uint64_t capacity, + uint8_t *out_bytes, + bool *out_ready, + bool *out_delivered, + struct xrt_lift_blob_info *out_info); + +xrt_result_t +comp_ipc_client_compositor_lift_set_priority(struct xrt_compositor *xc, uint64_t stream_id, uint32_t priority); +xrt_result_t +comp_ipc_client_compositor_lift_stats(struct xrt_compositor *xc, + uint64_t stream_id, + struct xrt_lift_stream_stats *out_stats); + +// The OpenXR and DP encodings must agree (the wire carries the DP one). +_Static_assert(XR_LIFT_MODE_DEPTH_DXR == XRT_DP_LIFT_MODE_DEPTH, "lift mode mismatch"); +_Static_assert(XR_LIFT_MODE_SBS_DXR == XRT_DP_LIFT_MODE_SBS, "lift mode mismatch"); +_Static_assert(XR_LIFT_MODE_NVIEW_DXR == XRT_DP_LIFT_MODE_NVIEW, "lift mode mismatch"); +_Static_assert(XR_LIFT_MODE_GAUSSIANS_DXR == XRT_DP_LIFT_MODE_GAUSSIANS, "lift mode mismatch"); +_Static_assert(XR_LIFT_STATE_READY_DXR == XRT_DP_LIFT_STATE_READY, "lift state mismatch"); +_Static_assert(XR_LIFT_STATE_ACTIVATING_DXR == XRT_DP_LIFT_STATE_ACTIVATING, "lift state mismatch"); +_Static_assert(XR_LIFT_BLOB_FORMAT_PLY_3DGS_DXR == XRT_DP_LIFT_BLOB_PLY_3DGS, "blob format mismatch"); +_Static_assert(XR_LIFT_BLOB_FORMAT_SOG_DXR == XRT_DP_LIFT_BLOB_SOG, "blob format mismatch"); +_Static_assert(XR_LIFT_PRIORITY_PAUSED_DXR == 0 && XR_LIFT_PRIORITY_LOW_DXR == 1 && XR_LIFT_PRIORITY_NORMAL_DXR == 2 && + XR_LIFT_PRIORITY_HIGH_DXR == 3, + "lift priority values are the wire values (u_lift_priority)"); +_Static_assert(XR_LIFT_MAX_VIEWS_DXR == XRT_LIFT_MAX_VIEWS, "lift view bound mismatch"); +_Static_assert(XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR == XRT_LIFT_WEAVE_RECTS_MAX, "lift rect bound mismatch"); +_Static_assert(XR_LIFT_BACKEND_NAME_MAX_SIZE_DXR == sizeof(((struct xrt_dp_lift_caps *)0)->backend), + "backend name size mismatch"); + +//! Same rule as oxr_weave.c: IPC sessions carry no in-process native flag. +static bool +lift_session_is_ipc(struct oxr_session *sess) +{ + if (sess == NULL || sess->xcn == NULL || sess->sys == NULL || sess->sys->xsysc == NULL) { + return false; + } + bool inprocess = sess->is_d3d11_native_compositor || sess->is_d3d12_native_compositor || + sess->is_metal_native_compositor || sess->is_gl_native_compositor || + sess->is_vk_native_compositor; + return !inprocess; +} + +//! Map a service result; returns XR_SUCCESS only for XRT_SUCCESS. +static XrResult +lift_xret(struct oxr_logger *log, struct oxr_session *sess, xrt_result_t xret, const char *what) +{ + switch (xret) { + case XRT_SUCCESS: return XR_SUCCESS; + case XRT_ERROR_IPC_FAILURE: + sess->has_lost = true; + return oxr_error(log, XR_ERROR_INSTANCE_LOST, "%s: the runtime service connection is gone", what); + case XRT_ERROR_FEATURE_NOT_SUPPORTED: + return oxr_error(log, XR_ERROR_FEATURE_UNSUPPORTED, "%s: no conversion module for this request", what); + case XRT_ERROR_CLIENT_LIMIT_REACHED: + return oxr_error(log, XR_ERROR_LIMIT_REACHED, "%s: maxStreams reached", what); + default: + // Transient (XRT_ERROR_WEAVE_REFUSED) or a stream the service no longer + // knows: non-fatal, the session stays usable. + return oxr_error(log, XR_ERROR_RUNTIME_FAILURE, "%s: refused by the service (xrt_result=%d)", what, + (int)xret); + } +} + +static void +lift_params_from_xr(const XrLiftOptionsDXR *o, uint32_t mode, struct xrt_dp_lift_params *out) +{ + memset(out, 0, sizeof(*out)); + out->struct_size = (uint32_t)sizeof(*out); + out->convergence = o->convergence < 0.0f ? -1.0f : (o->convergence > 1.0f ? 1.0f : o->convergence); + out->strength = o->strength >= 0.0f ? o->strength : 1.0f; + out->inpaint = o->inpaint == XR_TRUE ? 1u : 0u; + out->view_count = o->viewCount; + out->focal_px = o->focalPx > 0.0f ? o->focalPx : 0.0f; + if (out->view_count == 0) { + out->view_count = mode == XRT_DP_LIFT_MODE_NVIEW ? 4u : 2u; + } +} + +static XrResult +lift_validate_options(struct oxr_logger *log, const XrLiftOptionsDXR *o, bool weave_path) +{ + if (o->viewCount > XR_LIFT_MAX_VIEWS_DXR) { + return oxr_error(log, XR_ERROR_VALIDATION_FAILURE, "XrLiftOptionsDXR::viewCount (%u) > %u", + o->viewCount, (uint32_t)XR_LIFT_MAX_VIEWS_DXR); + } + if (o->viewpointSource != XR_LIFT_VIEWPOINT_SOURCE_TRACKED_DXR && + o->viewpointSource != XR_LIFT_VIEWPOINT_SOURCE_EXPLICIT_DXR) { + return oxr_error(log, XR_ERROR_VALIDATION_FAILURE, "XrLiftOptionsDXR::viewpointSource (%d) invalid", + (int)o->viewpointSource); + } + if (o->viewpointSource == XR_LIFT_VIEWPOINT_SOURCE_EXPLICIT_DXR) { + if (weave_path) { + return oxr_error(log, XR_ERROR_VALIDATION_FAILURE, + "XrLiftOptionsDXR: EXPLICIT viewpoints are not allowed on a weave rect " + "(the weave path always synthesizes for the tracked eyes)"); + } + if (o->viewCount == 0 || o->viewpoints == NULL) { + return oxr_error(log, XR_ERROR_VALIDATION_FAILURE, + "XrLiftOptionsDXR: EXPLICIT viewpoints need viewCount >= 1 and viewpoints"); + } + } + return XR_SUCCESS; +} + +static XrResult +oxr_lift_stream_destroy_cb(struct oxr_logger *log, struct oxr_handle_base *hb) +{ + struct oxr_lift_stream_dxr *st = (struct oxr_lift_stream_dxr *)hb; + // Best effort: a dead connection already took every stream with it. + if (st->sess != NULL && st->sess->xcn != NULL && !st->sess->has_lost) { + (void)comp_ipc_client_compositor_lift_stream_destroy(&st->sess->xcn->base, st->id); + } + free(st); + return XR_SUCCESS; +} + + +/* + * + * Entry points. + * + */ + +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrGetLiftPropertiesDXR(XrSession session, XrLiftPropertiesDXR *properties) +{ + OXR_TRACE_MARKER(); + + struct oxr_session *sess = NULL; + struct oxr_logger log; + OXR_VERIFY_SESSION_AND_INIT_LOG(&log, session, sess, "xrGetLiftPropertiesDXR"); + OXR_VERIFY_SESSION_NOT_LOST(&log, sess); + OXR_VERIFY_EXTENSION(&log, sess->sys->inst, DXR_lift); + OXR_VERIFY_ARG_TYPE_AND_NOT_NULL(&log, properties, XR_TYPE_LIFT_PROPERTIES_DXR); + + if (!lift_session_is_ipc(sess)) { + return oxr_error(&log, XR_ERROR_FEATURE_UNSUPPORTED, + "xrGetLiftPropertiesDXR: 2D→3D conversion is only available on the out-of-process " + "(service) path"); + } + + struct xrt_dp_lift_caps caps; + xrt_result_t xret = comp_ipc_client_compositor_lift_get_properties(&sess->xcn->base, &caps); + XrResult r = lift_xret(&log, sess, xret, "xrGetLiftPropertiesDXR"); + if (r != XR_SUCCESS) { + return r; + } + properties->supportedModes = (XrLiftModeFlagsDXR)caps.modes; + properties->maxStreams = caps.max_streams; + properties->maxViews = caps.max_views; + properties->depthSemantics = caps.depth_semantics == XRT_DP_LIFT_DEPTH_METRIC + ? XR_LIFT_DEPTH_SEMANTICS_METRIC_DXR + : XR_LIFT_DEPTH_SEMANTICS_RELATIVE_DXR; + properties->state = caps.state == XRT_DP_LIFT_STATE_READY ? XR_LIFT_STATE_READY_DXR + : caps.state == XRT_DP_LIFT_STATE_ACTIVATING ? XR_LIFT_STATE_ACTIVATING_DXR + : XR_LIFT_STATE_UNAVAILABLE_DXR; + memcpy(properties->backend, caps.backend, sizeof(properties->backend)); + properties->backend[sizeof(properties->backend) - 1] = '\0'; + properties->typicalLatency = (XrDuration)caps.typical_latency_ns; + return XR_SUCCESS; +} + +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrCreateLiftStreamDXR(XrSession session, const XrLiftStreamCreateInfoDXR *createInfo, XrLiftStreamDXR *stream) +{ + OXR_TRACE_MARKER(); + + struct oxr_session *sess = NULL; + struct oxr_logger log; + OXR_VERIFY_SESSION_AND_INIT_LOG(&log, session, sess, "xrCreateLiftStreamDXR"); + OXR_VERIFY_SESSION_NOT_LOST(&log, sess); + OXR_VERIFY_EXTENSION(&log, sess->sys->inst, DXR_lift); + OXR_VERIFY_ARG_TYPE_AND_NOT_NULL(&log, createInfo, XR_TYPE_LIFT_STREAM_CREATE_INFO_DXR); + OXR_VERIFY_ARG_NOT_NULL(&log, stream); + + const uint32_t mode = (uint32_t)createInfo->mode; + if (mode != XRT_DP_LIFT_MODE_DEPTH && mode != XRT_DP_LIFT_MODE_SBS && mode != XRT_DP_LIFT_MODE_NVIEW && + mode != XRT_DP_LIFT_MODE_GAUSSIANS) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "XrLiftStreamCreateInfoDXR::mode (%d) is not one mode", (int)createInfo->mode); + } + if (createInfo->contentHint != XR_LIFT_CONTENT_HINT_VIDEO_DXR && + createInfo->contentHint != XR_LIFT_CONTENT_HINT_PHOTO_DXR) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "XrLiftStreamCreateInfoDXR::contentHint (%d) invalid", (int)createInfo->contentHint); + } + if (mode == XRT_DP_LIFT_MODE_GAUSSIANS && createInfo->contentHint != XR_LIFT_CONTENT_HINT_PHOTO_DXR) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "XrLiftStreamCreateInfoDXR: GAUSSIANS streams take PHOTO content only"); + } + if (createInfo->inputScale < 0.0f || createInfo->inputScale > 1.0f) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "XrLiftStreamCreateInfoDXR::inputScale (%f) must be in [0, 1]", + (double)createInfo->inputScale); + } + if (!lift_session_is_ipc(sess)) { + return oxr_error(&log, XR_ERROR_FEATURE_UNSUPPORTED, + "xrCreateLiftStreamDXR: only available on the out-of-process (service) path"); + } + + uint64_t id = 0; + xrt_result_t xret = comp_ipc_client_compositor_lift_stream_create( + &sess->xcn->base, mode, (uint32_t)createInfo->contentHint, + createInfo->inputScale > 0.0f ? createInfo->inputScale : 1.0f, &id); + XrResult r = lift_xret(&log, sess, xret, "xrCreateLiftStreamDXR"); + if (r != XR_SUCCESS) { + return r; + } + + struct oxr_lift_stream_dxr *st = NULL; + OXR_ALLOCATE_HANDLE_OR_RETURN(&log, st, OXR_XR_DEBUG_LIFTSTREAM, oxr_lift_stream_destroy_cb, &sess->handle); + st->sess = sess; + st->id = id; + st->mode = mode; + *stream = XRT_CAST_PTR_TO_OXR_HANDLE(XrLiftStreamDXR, st); + return XR_SUCCESS; +} + +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrDestroyLiftStreamDXR(XrLiftStreamDXR stream) +{ + OXR_TRACE_MARKER(); + + struct oxr_lift_stream_dxr *st = NULL; + struct oxr_logger log; + OXR_VERIFY_LIFT_STREAM_AND_INIT_LOG(&log, stream, st, "xrDestroyLiftStreamDXR"); + return oxr_handle_destroy(&log, &st->handle); +} + +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrSubmitLiftFrameDXR(XrLiftStreamDXR stream, const XrLiftFrameSubmitInfoDXR *submitInfo, uint64_t *frameId) +{ + OXR_TRACE_MARKER(); + + struct oxr_lift_stream_dxr *st = NULL; + struct oxr_logger log; + OXR_VERIFY_LIFT_STREAM_AND_INIT_LOG(&log, stream, st, "xrSubmitLiftFrameDXR"); + struct oxr_session *sess = st->sess; + OXR_VERIFY_SESSION_NOT_LOST(&log, sess); + OXR_VERIFY_ARG_TYPE_AND_NOT_NULL(&log, submitInfo, XR_TYPE_LIFT_FRAME_SUBMIT_INFO_DXR); + OXR_VERIFY_ARG_NOT_NULL(&log, frameId); + *frameId = 0; + + if (submitInfo->inputTexture == NULL) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, "XrLiftFrameSubmitInfoDXR::inputTexture is NULL"); + } + if (submitInfo->extent.width <= 0 || submitInfo->extent.height <= 0) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "XrLiftFrameSubmitInfoDXR::extent (%dx%d) must be positive", submitInfo->extent.width, + submitInfo->extent.height); + } + + struct xrt_dp_lift_params params; + const struct xrt_dp_lift_params *pp = NULL; + float vps[3 * XR_LIFT_MAX_VIEWS_DXR]; + uint32_t vp_count = 0; + const XrLiftOptionsDXR *opt = OXR_GET_INPUT_FROM_CHAIN(submitInfo, XR_TYPE_LIFT_OPTIONS_DXR, XrLiftOptionsDXR); + if (opt != NULL) { + XrResult vr = lift_validate_options(&log, opt, /*weave_path*/ false); + if (vr != XR_SUCCESS) { + return vr; + } + lift_params_from_xr(opt, st->mode, ¶ms); + pp = ¶ms; + if (opt->viewpointSource == XR_LIFT_VIEWPOINT_SOURCE_EXPLICIT_DXR) { + vp_count = opt->viewCount; + for (uint32_t i = 0; i < vp_count; i++) { + vps[3 * i + 0] = opt->viewpoints[i].x; + vps[3 * i + 1] = opt->viewpoints[i].y; + vps[3 * i + 2] = opt->viewpoints[i].z; + } + } + } + + xrt_result_t xret = comp_ipc_client_compositor_lift_submit( + &sess->xcn->base, st->id, (xrt_graphics_buffer_handle_t)(intptr_t)submitInfo->inputTexture, + submitInfo->inputIsDxgi == XR_TRUE, (uint32_t)submitInfo->extent.width, (uint32_t)submitInfo->extent.height, + (int64_t)submitInfo->sourceTime, pp, vp_count > 0 ? vps : NULL, vp_count, frameId); + return lift_xret(&log, sess, xret, "xrSubmitLiftFrameDXR"); +} + +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrAcquireLiftResultDXR(XrLiftStreamDXR stream, XrLiftResultDXR *result) +{ + OXR_TRACE_MARKER(); + + struct oxr_lift_stream_dxr *st = NULL; + struct oxr_logger log; + OXR_VERIFY_LIFT_STREAM_AND_INIT_LOG(&log, stream, st, "xrAcquireLiftResultDXR"); + struct oxr_session *sess = st->sess; + OXR_VERIFY_SESSION_NOT_LOST(&log, sess); + OXR_VERIFY_ARG_TYPE_AND_NOT_NULL(&log, result, XR_TYPE_LIFT_RESULT_DXR); + + if (st->mode == XRT_DP_LIFT_MODE_GAUSSIANS) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "xrAcquireLiftResultDXR: a GAUSSIANS stream returns a blob (xrAcquireLiftBlobDXR)"); + } + + result->outputTexture = NULL; + result->fence = NULL; + + bool ready = false; + struct xrt_lift_result r; + xrt_result_t xret = comp_ipc_client_compositor_lift_acquire(&sess->xcn->base, st->id, &ready, &r); + XrResult xr = lift_xret(&log, sess, xret, "xrAcquireLiftResultDXR"); + if (xr != XR_SUCCESS) { + return xr; + } + if (!ready) { + return XR_LIFT_NOT_READY_DXR; + } + + result->frameId = r.frame_id; + result->sourceTime = (XrTime)r.source_time; + result->fenceValue = r.fence_value; + result->extent.width = (int32_t)r.width; + result->extent.height = (int32_t)r.height; + result->format = (int64_t)r.format; + result->viewCount = r.view_count; + result->latency = (XrDuration)r.latency_ns; + + // Handles: first acquire and every reallocation (the weave output pattern). + if (r.output_realloc || !st->exported) { + bool have_tex = false, have_fence = false; + xrt_graphics_buffer_handle_t th = XRT_GRAPHICS_BUFFER_HANDLE_INVALID; + xrt_graphics_sync_handle_t fh = XRT_GRAPHICS_SYNC_HANDLE_INVALID; + xret = comp_ipc_client_compositor_lift_get_output(&sess->xcn->base, st->id, &have_tex, &th); + xr = lift_xret(&log, sess, xret, "xrAcquireLiftResultDXR (output export)"); + if (xr != XR_SUCCESS) { + return xr; + } + xret = comp_ipc_client_compositor_lift_get_fence(&sess->xcn->base, st->id, &have_fence, &fh); + xr = lift_xret(&log, sess, xret, "xrAcquireLiftResultDXR (fence export)"); + if (xr != XR_SUCCESS) { + return xr; + } + const bool got_tex = have_tex && th != XRT_GRAPHICS_BUFFER_HANDLE_INVALID; + const bool got_fence = have_fence && fh != XRT_GRAPHICS_SYNC_HANDLE_INVALID; + if (got_tex) { + result->outputTexture = (void *)(intptr_t)th; + } + if (got_fence) { + result->fence = (void *)(intptr_t)fh; + } + // Latch only on a complete export (the #1427 rule): a miss retries. + st->exported = got_tex && got_fence; + } + return XR_SUCCESS; +} + +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrAcquireLiftBlobDXR(XrLiftStreamDXR stream, XrLiftBlobDXR *blob) +{ + OXR_TRACE_MARKER(); + + struct oxr_lift_stream_dxr *st = NULL; + struct oxr_logger log; + OXR_VERIFY_LIFT_STREAM_AND_INIT_LOG(&log, stream, st, "xrAcquireLiftBlobDXR"); + struct oxr_session *sess = st->sess; + OXR_VERIFY_SESSION_NOT_LOST(&log, sess); + OXR_VERIFY_ARG_TYPE_AND_NOT_NULL(&log, blob, XR_TYPE_LIFT_BLOB_DXR); + + if (st->mode != XRT_DP_LIFT_MODE_GAUSSIANS) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "xrAcquireLiftBlobDXR: only a GAUSSIANS stream returns a blob"); + } + if (blob->byteCapacityInput > 0 && blob->bytes == NULL) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, "XrLiftBlobDXR::bytes is NULL with a capacity"); + } + + bool ready = false, delivered = false; + struct xrt_lift_blob_info info; + xrt_result_t xret = comp_ipc_client_compositor_lift_acquire_blob( + &sess->xcn->base, st->id, blob->byteCapacityInput, blob->bytes, &ready, &delivered, &info); + XrResult xr = lift_xret(&log, sess, xret, "xrAcquireLiftBlobDXR"); + if (xr != XR_SUCCESS) { + return xr; + } + if (!ready) { + blob->byteCountOutput = 0; + return XR_LIFT_NOT_READY_DXR; + } + if (info.byte_count > UINT32_MAX) { + return oxr_error(&log, XR_ERROR_RUNTIME_FAILURE, "xrAcquireLiftBlobDXR: blob exceeds 4 GiB"); + } + blob->frameId = info.frame_id; + blob->sourceTime = (XrTime)info.source_time; + blob->format = (XrLiftBlobFormatDXR)info.format; + blob->byteCountOutput = (uint32_t)info.byte_count; + if (blob->byteCapacityInput == 0) { + return XR_SUCCESS; // size query; the service holds the blob latched + } + if (!delivered) { + return oxr_error(&log, XR_ERROR_SIZE_INSUFFICIENT, "xrAcquireLiftBlobDXR: capacity %u < %u", + blob->byteCapacityInput, blob->byteCountOutput); + } + return XR_SUCCESS; +} + +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrSetLiftStreamPriorityDXR(XrLiftStreamDXR stream, XrLiftPriorityDXR priority) +{ + OXR_TRACE_MARKER(); + + struct oxr_lift_stream_dxr *st = NULL; + struct oxr_logger log; + OXR_VERIFY_LIFT_STREAM_AND_INIT_LOG(&log, stream, st, "xrSetLiftStreamPriorityDXR"); + struct oxr_session *sess = st->sess; + OXR_VERIFY_SESSION_NOT_LOST(&log, sess); + if ((int)priority < (int)XR_LIFT_PRIORITY_PAUSED_DXR || (int)priority > (int)XR_LIFT_PRIORITY_HIGH_DXR) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, "xrSetLiftStreamPriorityDXR: priority (%d) invalid", + (int)priority); + } + xrt_result_t xret = comp_ipc_client_compositor_lift_set_priority(&sess->xcn->base, st->id, (uint32_t)priority); + return lift_xret(&log, sess, xret, "xrSetLiftStreamPriorityDXR"); +} + +XRAPI_ATTR XrResult XRAPI_CALL +oxr_xrGetLiftStreamStatsDXR(XrLiftStreamDXR stream, XrLiftStreamStatsDXR *stats) +{ + OXR_TRACE_MARKER(); + + struct oxr_lift_stream_dxr *st = NULL; + struct oxr_logger log; + OXR_VERIFY_LIFT_STREAM_AND_INIT_LOG(&log, stream, st, "xrGetLiftStreamStatsDXR"); + struct oxr_session *sess = st->sess; + OXR_VERIFY_SESSION_NOT_LOST(&log, sess); + OXR_VERIFY_ARG_TYPE_AND_NOT_NULL(&log, stats, XR_TYPE_LIFT_STREAM_STATS_DXR); + + struct xrt_lift_stream_stats s; + xrt_result_t xret = comp_ipc_client_compositor_lift_stats(&sess->xcn->base, st->id, &s); + XrResult xr = lift_xret(&log, sess, xret, "xrGetLiftStreamStatsDXR"); + if (xr != XR_SUCCESS) { + return xr; + } + stats->priority = (XrLiftPriorityDXR)s.priority; + stats->framesSubmitted = s.submitted; + stats->framesConverted = s.converted; + stats->framesDropped = s.dropped; + stats->framesFailed = s.failed; + stats->latencyLast = (XrDuration)s.latency_last_ns; + stats->latencyAverage = (XrDuration)s.latency_avg_ns; + stats->latencyMin = (XrDuration)s.latency_min_ns; + stats->latencyMax = (XrDuration)s.latency_max_ns; + stats->conversionRate = s.rate_hz; + return XR_SUCCESS; +} + +#endif // OXR_HAVE_DXR_lift diff --git a/src/xrt/state_trackers/oxr/oxr_objects.h b/src/xrt/state_trackers/oxr/oxr_objects.h index a9ba7e88e..745f1b71b 100644 --- a/src/xrt/state_trackers/oxr/oxr_objects.h +++ b/src/xrt/state_trackers/oxr/oxr_objects.h @@ -134,6 +134,7 @@ struct oxr_body_tracker_fb; struct oxr_xdev_list; struct oxr_plane_detector_ext; struct oxr_local_3d_zone_ext; +struct oxr_lift_stream_dxr; #define XRT_MAX_HANDLE_CHILDREN 256 #define OXR_MAX_BINDINGS_PER_ACTION 32 @@ -4332,6 +4333,28 @@ struct oxr_local_3d_zone_ext }; #endif // OXR_HAVE_DXR_local_3d_zone +#ifdef OXR_HAVE_DXR_lift +/*! + * One XR_DXR_lift conversion stream (ADR-042). The conversion itself lives in + * the service (d3d11_lift.cpp); this is the handle, the service-side id, and + * the export latch for the stream's result texture + fence (the weave output + * pattern: handed out on the first acquire and on every reallocation). + * + * Parent type/handle is @ref oxr_session + * + * @obj{XrLiftStreamDXR} + * @extends oxr_handle_base + */ +struct oxr_lift_stream_dxr +{ + struct oxr_handle_base handle; + struct oxr_session *sess; + uint64_t id; //!< service-side stream id + uint32_t mode; //!< XRT_DP_LIFT_MODE_* (one bit) + bool exported; //!< the caller holds the current export texture + fence +}; +#endif // OXR_HAVE_DXR_lift + #ifdef OXR_HAVE_EXT_user_presence XrResult diff --git a/src/xrt/state_trackers/oxr/oxr_weave.c b/src/xrt/state_trackers/oxr/oxr_weave.c index 2516c7e65..50ecb43a3 100644 --- a/src/xrt/state_trackers/oxr/oxr_weave.c +++ b/src/xrt/state_trackers/oxr/oxr_weave.c @@ -100,6 +100,16 @@ static_assert(XR_WEAVE_SNAP_GRID_MAX_POINTS_DXR == U_SNAP_GRID_MAX_POINTS, "grid static_assert(XR_WEAVE_SNAP_GRID_MAX_AXIS_DXR == U_SNAP_GRID_MAX_AXIS, "grid axis bound mismatch"); static_assert(XR_WEAVE_SNAP_GRID_NO_DELTA_DXR == U_SNAP_GRID_NO_DELTA, "grid sentinel mismatch"); +#ifdef OXR_HAVE_DXR_lift +#include "xrt/xrt_lift.h" +#include "oxr_handle.h" +//! XR_DXR_lift (ADR-042) bridge: latch the lift-flagged rects of the next submit. +xrt_result_t +comp_ipc_client_compositor_lift_weave_rects(struct xrt_compositor *xc, + uint32_t count, + const struct xrt_lift_weave_rect *rects); +#endif + // Forward decls of the IPC-bridge wrappers (defined in ipc_client_compositor.c). struct xrt_compositor; @@ -702,6 +712,72 @@ oxr_xrWeaveSubmitDXR(XrSession session, const XrWeaveSubmitInfoDXR *submitInfo, } #endif +#ifdef OXR_HAVE_DXR_lift + // XR_DXR_lift (ADR-042): rects whose content is 2D and must be lifted + // before weaving. Sent as its own call right before the submit it belongs + // to (same connection, so the order is the service's order); the service + // consumes the set with that submit. Ignored unless XR_DXR_lift is enabled. + const XrWeaveSubmitLiftRectsDXR *lifts = + OXR_GET_INPUT_FROM_CHAIN(submitInfo, XR_TYPE_WEAVE_SUBMIT_LIFT_RECTS_DXR, XrWeaveSubmitLiftRectsDXR); + if (lifts != NULL && lifts->liftCount > 0 && sess->sys->inst->extensions.DXR_lift) { + if (lifts->liftCount > XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR || lifts->lifts == NULL) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "xrWeaveSubmitDXR: XrWeaveSubmitLiftRectsDXR::liftCount (%u) must be 1..%u with " + "a non-NULL lifts array", + lifts->liftCount, (uint32_t)XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR); + } + if (rect_count == 0) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "xrWeaveSubmitDXR: XrWeaveSubmitLiftRectsDXR needs XrWeaveSubmitRectsDXR (the " + "lift indexes its rects)"); + } + struct xrt_lift_weave_rect lr[XR_WEAVE_SUBMIT_MAX_LIFT_RECTS_DXR]; + memset(lr, 0, sizeof(lr)); + for (uint32_t i = 0; i < lifts->liftCount; i++) { + const XrWeaveRectLiftDXR *e = &lifts->lifts[i]; + if (e->type != XR_TYPE_WEAVE_RECT_LIFT_DXR || e->rectIndex >= rect_count) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "xrWeaveSubmitDXR: lifts[%u] has a bad type or rectIndex (%u >= %u)", i, + e->rectIndex, rect_count); + } + struct oxr_lift_stream_dxr *ls = XRT_CAST_OXR_HANDLE_TO_PTR(struct oxr_lift_stream_dxr *, e->stream); + if (ls == NULL || ls->handle.debug != OXR_XR_DEBUG_LIFTSTREAM || + ls->handle.state != OXR_HANDLE_STATE_LIVE || ls->sess != sess) { + return oxr_error(&log, XR_ERROR_HANDLE_INVALID, "xrWeaveSubmitDXR: lifts[%u].stream", i); + } + if (ls->mode != XRT_DP_LIFT_MODE_SBS && ls->mode != XRT_DP_LIFT_MODE_NVIEW) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "xrWeaveSubmitDXR: lifts[%u].stream must be an SBS or NVIEW stream", i); + } + lr[i].stream_id = ls->id; + lr[i].rect_index = e->rectIndex; + const XrLiftOptionsDXR *o = OXR_GET_INPUT_FROM_CHAIN(e, XR_TYPE_LIFT_OPTIONS_DXR, XrLiftOptionsDXR); + if (o != NULL) { + if (o->viewpointSource == XR_LIFT_VIEWPOINT_SOURCE_EXPLICIT_DXR || + o->viewCount > XR_LIFT_MAX_VIEWS_DXR) { + return oxr_error(&log, XR_ERROR_VALIDATION_FAILURE, + "xrWeaveSubmitDXR: lifts[%u] options: TRACKED viewpoints only on the " + "weave path, viewCount <= %u", + i, (uint32_t)XR_LIFT_MAX_VIEWS_DXR); + } + lr[i].has_params = true; + lr[i].params.struct_size = (uint32_t)sizeof(lr[i].params); + lr[i].params.convergence = + o->convergence < 0.0f ? -1.0f : (o->convergence > 1.0f ? 1.0f : o->convergence); + lr[i].params.strength = o->strength >= 0.0f ? o->strength : 1.0f; + lr[i].params.inpaint = o->inpaint == XR_TRUE ? 1u : 0u; + lr[i].params.view_count = o->viewCount; + lr[i].params.focal_px = o->focalPx > 0.0f ? o->focalPx : 0.0f; + } + } + xrt_result_t lx = comp_ipc_client_compositor_lift_weave_rects(&sess->xcn->base, lifts->liftCount, lr); + if (lx == XRT_ERROR_IPC_FAILURE) { + OXR_CHECK_XRET_MSG(&log, sess, lx, "xrWeaveSubmitDXR: lift rects failed (xrt_result=%d)", (int)lx); + } + // Any other refusal is non-fatal: the rects are then woven as drawn. + } +#endif + bool have_out = false; uint32_t w = 0, h = 0; uint64_t fence_value = 0; diff --git a/src/xrt/targets/cli/CMakeLists.txt b/src/xrt/targets/cli/CMakeLists.txt index fe106f2f1..6778feaee 100644 --- a/src/xrt/targets/cli/CMakeLists.txt +++ b/src/xrt/targets/cli/CMakeLists.txt @@ -42,6 +42,7 @@ add_executable( cli_cmd_probe.c cli_cmd_test.c cli_cmd_clients.c + cli_cmd_lift.cpp cli_common.h cli_main.c ) @@ -88,4 +89,11 @@ if(XRT_FEATURE_SERVICE AND XRT_MODULE_IPC) target_compile_definitions(cli PRIVATE CLI_HAVE_IPC) endif() +# ADR-042: `displayxr-cli lift probe` loads / writes images (stb, header-only, +# compiled static into cli_cmd_lift.cpp) and drives D3D11 shared textures. +target_include_directories(cli PRIVATE ${PROJECT_SOURCE_DIR}/src/external/stb) +if(WIN32) + target_link_libraries(cli PRIVATE d3d11 dxgi) +endif() + install(TARGETS cli RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}) diff --git a/src/xrt/targets/cli/cli_cmd_lift.cpp b/src/xrt/targets/cli/cli_cmd_lift.cpp new file mode 100644 index 000000000..2d8028ff0 --- /dev/null +++ b/src/xrt/targets/cli/cli_cmd_lift.cpp @@ -0,0 +1,729 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief `displayxr-cli lift` — XR_DXR_lift (ADR-042) caps + the N0 probe. + * + * displayxr-cli lift caps [--json] [--wait S] + * displayxr-cli lift probe [--mode depth|sbs|nview|gaussians] + * [--n N] [--views N] [--strength F] [--convergence F] + * [--focal PX] [--priority paused|low|normal|high] + * [--pipelined] [--fps F] [--out DIR] [--wait S] + * --no-write skip all readback/encode (measure pure submit->acquire; PNG encode of a 4K SBS takes seconds) + * --write-every N write only every Nth result (default 1) + * + * Connects to the running service over IPC as a DIAG client (lift streams are + * owned by the connection, no session needed) — run it from a NON-elevated + * prompt, like `displayxr-cli clients`. `probe` is Windows-only: it hands the + * service D3D11 shared textures, exactly as the browser does, and reads the + * results back through the exported texture + fence. + * + * Per frame it prints the submit→acquire latency measured here (the number a + * consumer sees), the service's own submit→converted latency, and the output + * layout; results are written as lift_out_.png (depth normalised to 8-bit + * grey) or lift_out_.ply (gaussians). The vendor module reads its knobs + * (e.g. a DirectML/CUDA backend switch) from the SERVICE's environment, not + * this process's — see docs/specs/extensions/XR_DXR_lift.md §9. + */ + +#include "cli_common.h" + +#include +#include +#include + +#ifdef CLI_HAVE_IPC + +#include "xrt/xrt_instance.h" +#include "xrt/xrt_results.h" +#include "xrt/xrt_config_os.h" +#include "xrt/xrt_lift.h" +#include "util/u_logging.h" +#include "os/os_time.h" + +#include "client/ipc_client.h" +#include "client/ipc_client_connection.h" +#include "client/ipc_client_lift.h" + +#include +#include +#include + +#ifdef XRT_OS_WINDOWS +#define WIN32_LEAN_AND_MEAN +#include +#include +#include + +#define STB_IMAGE_STATIC +#define STB_IMAGE_IMPLEMENTATION +#include "stb_image.h" +#define STB_IMAGE_WRITE_STATIC +#define STB_IMAGE_WRITE_IMPLEMENTATION +#include "stb_image_write.h" +#endif + +namespace { + +const char * +state_str(uint32_t s) +{ + switch (s) { + case XRT_DP_LIFT_STATE_READY: return "READY"; + case XRT_DP_LIFT_STATE_ACTIVATING: return "ACTIVATING"; + default: return "UNAVAILABLE"; + } +} + +std::string +modes_str(uint32_t m) +{ + std::string s; + if (m & XRT_DP_LIFT_MODE_DEPTH) { + s += "DEPTH "; + } + if (m & XRT_DP_LIFT_MODE_SBS) { + s += "SBS "; + } + if (m & XRT_DP_LIFT_MODE_NVIEW) { + s += "NVIEW "; + } + if (m & XRT_DP_LIFT_MODE_GAUSSIANS) { + s += "GAUSSIANS "; + } + if (s.empty()) { + s = "(none)"; + } else { + s.pop_back(); + } + return s; +} + +const char * +opt_value(int argc, const char **argv, const char *name, const char *def) +{ + for (int i = 3; i + 1 < argc; i++) { + if (strcmp(argv[i], name) == 0) { + return argv[i + 1]; + } + } + return def; +} + +bool +connect(struct ipc_connection *ipc_c) +{ + struct xrt_instance_info ii = {}; + snprintf(ii.app_info.application_name, sizeof(ii.app_info.application_name), "%s", "displayxr-cli lift"); + ii.app_info.declared_client_class = XRT_CLIENT_CLASS_DIAG; + xrt_result_t xret = ipc_client_connection_init(ipc_c, U_LOGGING_ERROR, &ii); + if (xret != XRT_SUCCESS) { + printf("displayxr-cli lift: not connected to the service (xrt_result=%d).\n", (int)xret); + printf(" Is displayxr-service running? On Windows, run from a NON-elevated prompt.\n"); + return false; + } + return true; +} + +//! Poll caps until the module leaves ACTIVATING or @p wait_s passes. +bool +wait_caps(struct ipc_connection *ipc_c, double wait_s, struct xrt_dp_lift_caps *caps, bool verbose) +{ + uint64_t t0 = os_monotonic_get_ns(); + uint32_t last = 0xffffffffu; + for (;;) { + xrt_result_t xret = ipc_client_lift_get_properties(ipc_c, caps); + if (xret != XRT_SUCCESS) { + printf("lift_get_properties failed: %d\n", (int)xret); + return false; + } + if (verbose && caps->state != last) { + printf(" state: %s\n", state_str(caps->state)); + last = caps->state; + } + if (caps->state != XRT_DP_LIFT_STATE_ACTIVATING) { + return true; + } + if ((double)(os_monotonic_get_ns() - t0) / 1e9 > wait_s) { + return true; + } + os_nanosleep(250 * 1000 * 1000); + } +} + +void +print_caps(const struct xrt_dp_lift_caps &c, bool json) +{ + if (json) { + printf( + "{\"connected\": true, \"state\": \"%s\", \"modes\": %u, \"modes_str\": \"%s\", " + "\"max_streams\": %u, \"max_views\": %u, \"depth_semantics\": \"%s\", " + "\"typical_latency_ms\": %.2f, \"backend\": \"%s\"}\n", + state_str(c.state), c.modes, modes_str(c.modes).c_str(), c.max_streams, c.max_views, + c.depth_semantics == XRT_DP_LIFT_DEPTH_METRIC ? "metric" : "relative", + (double)c.typical_latency_ns / 1e6, c.backend); + return; + } + printf("XR_DXR_lift conversion module:\n"); + printf(" state: %s\n", state_str(c.state)); + printf(" backend: %s\n", c.backend[0] != '\0' ? c.backend : "(none)"); + printf(" modes: %s (0x%x)\n", modes_str(c.modes).c_str(), c.modes); + printf(" max streams: %u\n", c.max_streams); + printf(" max views: %u\n", c.max_views); + printf(" depth semantics: %s\n", c.depth_semantics == XRT_DP_LIFT_DEPTH_METRIC ? "metric" : "relative"); + printf(" typical latency: %.2f ms\n", (double)c.typical_latency_ns / 1e6); +} + +int +cmd_caps(int argc, const char **argv) +{ + bool json = cli_has_flag(argc, argv, "--json"); + double wait_s = atof(opt_value(argc, argv, "--wait", "10")); + struct ipc_connection ipc_c = {}; + if (!connect(&ipc_c)) { + return 2; + } + struct xrt_dp_lift_caps caps; + bool ok = wait_caps(&ipc_c, wait_s, &caps, !json); + if (ok) { + print_caps(caps, json); + } + ipc_client_connection_fini(&ipc_c); + return ok ? 0 : 2; +} + +#ifdef XRT_OS_WINDOWS + +template +void +rel(T *&p) +{ + if (p != nullptr) { + p->Release(); + p = nullptr; + } +} + +struct probe_gpu +{ + ID3D11Device *dev = nullptr; + ID3D11Device1 *dev1 = nullptr; + ID3D11Device5 *dev5 = nullptr; + ID3D11DeviceContext *ctx = nullptr; + ID3D11DeviceContext4 *ctx4 = nullptr; + + // Input: a shared keyed-mutex texture, re-created on size change. + ID3D11Texture2D *in_tex = nullptr; + IDXGIKeyedMutex *in_km = nullptr; + HANDLE in_handle = nullptr; + uint32_t in_w = 0, in_h = 0; + + // Output: the service's export texture + fence, re-opened on realloc. + ID3D11Texture2D *out_tex = nullptr; + ID3D11Fence *out_fence = nullptr; + ID3D11Texture2D *staging = nullptr; + uint32_t st_w = 0, st_h = 0, st_fmt = 0; + + ~probe_gpu() + { + rel(staging); + rel(out_fence); + rel(out_tex); + if (in_handle != nullptr) { + CloseHandle(in_handle); + } + rel(in_km); + rel(in_tex); + rel(ctx4); + rel(ctx); + rel(dev5); + rel(dev1); + rel(dev); + } +}; + +bool +gpu_init(probe_gpu &g) +{ + D3D_FEATURE_LEVEL fl = D3D_FEATURE_LEVEL_11_1; + HRESULT hr = D3D11CreateDevice(nullptr, D3D_DRIVER_TYPE_HARDWARE, nullptr, D3D11_CREATE_DEVICE_BGRA_SUPPORT, + &fl, 1, D3D11_SDK_VERSION, &g.dev, nullptr, &g.ctx); + if (FAILED(hr)) { + printf("D3D11CreateDevice failed: 0x%08lx\n", (unsigned long)hr); + return false; + } + g.dev->QueryInterface(__uuidof(ID3D11Device1), (void **)&g.dev1); + g.dev->QueryInterface(__uuidof(ID3D11Device5), (void **)&g.dev5); + g.ctx->QueryInterface(__uuidof(ID3D11DeviceContext4), (void **)&g.ctx4); + return g.dev1 != nullptr && g.dev5 != nullptr && g.ctx4 != nullptr; +} + +bool +upload(probe_gpu &g, const uint8_t *rgba, uint32_t w, uint32_t h) +{ + if (g.in_tex == nullptr || g.in_w != w || g.in_h != h) { + if (g.in_handle != nullptr) { + CloseHandle(g.in_handle); + g.in_handle = nullptr; + } + rel(g.in_km); + rel(g.in_tex); + D3D11_TEXTURE2D_DESC td = {}; + td.Width = w; + td.Height = h; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = DXGI_FORMAT_R8G8B8A8_UNORM; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_DEFAULT; + td.BindFlags = D3D11_BIND_SHADER_RESOURCE | D3D11_BIND_RENDER_TARGET; + td.MiscFlags = D3D11_RESOURCE_MISC_SHARED_NTHANDLE | D3D11_RESOURCE_MISC_SHARED_KEYEDMUTEX; + HRESULT hr = g.dev->CreateTexture2D(&td, nullptr, &g.in_tex); + IDXGIResource1 *r1 = nullptr; + if (SUCCEEDED(hr)) { + hr = g.in_tex->QueryInterface(__uuidof(IDXGIResource1), (void **)&r1); + } + if (SUCCEEDED(hr)) { + hr = r1->CreateSharedHandle(nullptr, DXGI_SHARED_RESOURCE_READ | DXGI_SHARED_RESOURCE_WRITE, + nullptr, &g.in_handle); + } + rel(r1); + if (SUCCEEDED(hr)) { + hr = g.in_tex->QueryInterface(__uuidof(IDXGIKeyedMutex), (void **)&g.in_km); + } + if (FAILED(hr)) { + printf("input texture %ux%u create/share failed: 0x%08lx\n", w, h, (unsigned long)hr); + return false; + } + g.in_w = w; + g.in_h = h; + } + // Key 0 = "caller done writing" — the service's AcquireSync(0) waits on it. + if (FAILED(g.in_km->AcquireSync(0, INFINITE))) { + return false; + } + g.ctx->UpdateSubresource(g.in_tex, 0, nullptr, rgba, w * 4, 0); + g.in_km->ReleaseSync(0); + g.ctx->Flush(); + return true; +} + +//! Read the export texture back (after waiting its fence) and write a PNG. +bool +read_back_png(probe_gpu &g, uint64_t fence_value, uint32_t w, uint32_t h, uint32_t fmt, const std::string &path) +{ + if (g.out_tex == nullptr || g.out_fence == nullptr) { + return false; + } + if (g.staging == nullptr || g.st_w != w || g.st_h != h || g.st_fmt != fmt) { + rel(g.staging); + D3D11_TEXTURE2D_DESC td = {}; + td.Width = w; + td.Height = h; + td.MipLevels = 1; + td.ArraySize = 1; + td.Format = (DXGI_FORMAT)fmt; + td.SampleDesc.Count = 1; + td.Usage = D3D11_USAGE_STAGING; + td.CPUAccessFlags = D3D11_CPU_ACCESS_READ; + if (FAILED(g.dev->CreateTexture2D(&td, nullptr, &g.staging))) { + printf(" staging %ux%u fmt=%u create failed\n", w, h, fmt); + return false; + } + g.st_w = w; + g.st_h = h; + g.st_fmt = fmt; + } + g.ctx4->Wait(g.out_fence, fence_value); + g.ctx->CopyResource(g.staging, g.out_tex); + D3D11_MAPPED_SUBRESOURCE m = {}; + if (FAILED(g.ctx->Map(g.staging, 0, D3D11_MAP_READ, 0, &m))) { + return false; + } + std::vector rgba((size_t)w * h * 4); + const uint8_t *src = (const uint8_t *)m.pData; + if (fmt == DXGI_FORMAT_R8G8B8A8_UNORM || fmt == DXGI_FORMAT_R8G8B8A8_UNORM_SRGB || + fmt == DXGI_FORMAT_B8G8R8A8_UNORM || fmt == DXGI_FORMAT_B8G8R8A8_UNORM_SRGB) { + const bool bgra = fmt == DXGI_FORMAT_B8G8R8A8_UNORM || fmt == DXGI_FORMAT_B8G8R8A8_UNORM_SRGB; + for (uint32_t y = 0; y < h; y++) { + const uint8_t *row = src + (size_t)y * m.RowPitch; + for (uint32_t x = 0; x < w; x++) { + uint8_t *o = &rgba[((size_t)y * w + x) * 4]; + o[0] = row[x * 4 + (bgra ? 2 : 0)]; + o[1] = row[x * 4 + 1]; + o[2] = row[x * 4 + (bgra ? 0 : 2)]; + o[3] = 255; + } + } + } else if (fmt == DXGI_FORMAT_R32_FLOAT || fmt == DXGI_FORMAT_R8_UNORM || fmt == DXGI_FORMAT_R16_UNORM) { + // Depth: normalise min..max to 8-bit grey. + std::vector v((size_t)w * h); + for (uint32_t y = 0; y < h; y++) { + const uint8_t *row = src + (size_t)y * m.RowPitch; + for (uint32_t x = 0; x < w; x++) { + float f = 0.0f; + if (fmt == DXGI_FORMAT_R32_FLOAT) { + memcpy(&f, row + x * 4, 4); + } else if (fmt == DXGI_FORMAT_R8_UNORM) { + f = row[x] / 255.0f; + } else { + uint16_t u; + memcpy(&u, row + x * 2, 2); + f = u / 65535.0f; + } + v[(size_t)y * w + x] = f; + } + } + auto mm = std::minmax_element(v.begin(), v.end()); + float lo = *mm.first, span = *mm.second - *mm.first; + for (size_t i = 0; i < v.size(); i++) { + uint8_t gy = span > 0 ? (uint8_t)std::min(255.0f, 255.0f * (v[i] - lo) / span) : 0; + rgba[i * 4 + 0] = rgba[i * 4 + 1] = rgba[i * 4 + 2] = gy; + rgba[i * 4 + 3] = 255; + } + } else { + g.ctx->Unmap(g.staging, 0); + printf(" (format %u not written as PNG)\n", fmt); + return false; + } + g.ctx->Unmap(g.staging, 0); + return stbi_write_png(path.c_str(), (int)w, (int)h, 4, rgba.data(), (int)w * 4) != 0; +} + +std::vector +list_frames(const std::string &path) +{ + std::vector out; + DWORD attr = GetFileAttributesA(path.c_str()); + if (attr == INVALID_FILE_ATTRIBUTES) { + return out; + } + if ((attr & FILE_ATTRIBUTE_DIRECTORY) == 0) { + out.push_back(path); + return out; + } + WIN32_FIND_DATAA fd; + HANDLE h = FindFirstFileA((path + "\\*").c_str(), &fd); + if (h == INVALID_HANDLE_VALUE) { + return out; + } + do { + std::string n = fd.cFileName; + std::string lower = n; + std::transform(lower.begin(), lower.end(), lower.begin(), ::tolower); + auto ends = [&](const char *e) { + size_t l = strlen(e); + return lower.size() >= l && lower.compare(lower.size() - l, l, e) == 0; + }; + if (ends(".png") || ends(".jpg") || ends(".jpeg") || ends(".bmp")) { + out.push_back(path + "\\" + n); + } + } while (FindNextFileA(h, &fd)); + FindClose(h); + std::sort(out.begin(), out.end()); + return out; +} + +int +cmd_probe(int argc, const char **argv) +{ + if (argc < 4 || argv[3][0] == '-') { + printf( + "usage: displayxr-cli lift probe [--mode depth|sbs|nview|gaussians] [--n N]\n" + " [--views N] [--strength F] [--convergence F] [--focal PX]\n" + " [--priority paused|low|normal|high] [--pipelined] [--fps F] [--out DIR] [--wait S]\n"); + return 1; + } + const std::string src = argv[3]; + const std::string mode_s = opt_value(argc, argv, "--mode", "sbs"); + const int n_frames = atoi(opt_value(argc, argv, "--n", "100")); + const std::string out_dir = opt_value(argc, argv, "--out", "."); + const bool pipelined = cli_has_flag(argc, argv, "--pipelined"); + const bool no_write = cli_has_flag(argc, argv, "--no-write"); + const int write_every = std::max(1, atoi(opt_value(argc, argv, "--write-every", "1"))); + const double fps = atof(opt_value(argc, argv, "--fps", "30")); + const double wait_s = atof(opt_value(argc, argv, "--wait", "30")); + const std::string prio_s = opt_value(argc, argv, "--priority", "normal"); + + uint32_t mode = XRT_DP_LIFT_MODE_SBS; + if (mode_s == "depth") { + mode = XRT_DP_LIFT_MODE_DEPTH; + } else if (mode_s == "nview") { + mode = XRT_DP_LIFT_MODE_NVIEW; + } else if (mode_s == "gaussians") { + mode = XRT_DP_LIFT_MODE_GAUSSIANS; + } else if (mode_s != "sbs") { + printf("unknown --mode '%s'\n", mode_s.c_str()); + return 1; + } + uint32_t prio = prio_s == "paused" ? 0 : prio_s == "low" ? 1 : prio_s == "high" ? 3 : 2; + + xrt_dp_lift_params params = {}; + params.struct_size = (uint32_t)sizeof(params); + params.convergence = (float)atof(opt_value(argc, argv, "--convergence", "-1")); + params.strength = (float)atof(opt_value(argc, argv, "--strength", "1")); + params.inpaint = 1; + params.view_count = + (uint32_t)atoi(opt_value(argc, argv, "--views", mode == XRT_DP_LIFT_MODE_NVIEW ? "4" : "2")); + params.focal_px = (float)atof(opt_value(argc, argv, "--focal", "0")); + + std::vector frames = list_frames(src); + if (frames.empty()) { + printf("no input image(s) at '%s'\n", src.c_str()); + return 1; + } + + struct ipc_connection ipc_c = {}; + if (!connect(&ipc_c)) { + return 2; + } + struct xrt_dp_lift_caps caps; + printf("waiting for the conversion module (up to %.0f s)...\n", wait_s); + if (!wait_caps(&ipc_c, wait_s, &caps, true)) { + ipc_client_connection_fini(&ipc_c); + return 2; + } + print_caps(caps, false); + if (caps.state != XRT_DP_LIFT_STATE_READY || (caps.modes & mode) == 0) { + printf("module not READY for mode %s — nothing to probe.\n", mode_s.c_str()); + ipc_client_connection_fini(&ipc_c); + return 3; + } + + probe_gpu g; + if (!gpu_init(g)) { + ipc_client_connection_fini(&ipc_c); + return 2; + } + + uint64_t sid = 0; + xrt_result_t xret = ipc_client_lift_stream_create( + &ipc_c, mode, mode == XRT_DP_LIFT_MODE_GAUSSIANS ? XRT_DP_LIFT_CONTENT_PHOTO : XRT_DP_LIFT_CONTENT_VIDEO, + 1.0f, &sid); + if (xret != XRT_SUCCESS) { + printf("lift_stream_create failed: %d\n", (int)xret); + ipc_client_connection_fini(&ipc_c); + return 2; + } + (void)ipc_client_lift_set_priority(&ipc_c, sid, prio); + printf("stream %llu (%s, priority %s), %d frame(s) from %zu image(s), %s\n", (unsigned long long)sid, + mode_s.c_str(), prio_s.c_str(), n_frames, frames.size(), + pipelined ? "pipelined (latest wins)" : "sequential (wait for each result)"); + printf("%6s %10s %12s %12s %12s %s\n", "frame", "frame_id", "submit->acq", "svc_latency", "out", "file"); + + std::vector lat_ms; + int written = 0; + int cached_idx = -1; + int iw = 0, ih = 0; + stbi_uc *pixels = nullptr; + uint64_t next_submit_ns = os_monotonic_get_ns(); + for (int i = 0; i < n_frames; i++) { + int idx = (int)(i % frames.size()); + if (idx != cached_idx) { + if (pixels != nullptr) { + stbi_image_free(pixels); + } + int comp = 0; + pixels = stbi_load(frames[idx].c_str(), &iw, &ih, &comp, 4); + cached_idx = idx; + if (pixels == nullptr) { + printf("cannot load '%s'\n", frames[idx].c_str()); + break; + } + } + if (!upload(g, pixels, (uint32_t)iw, (uint32_t)ih)) { + printf("upload failed\n"); + break; + } + if (pipelined) { + uint64_t now = os_monotonic_get_ns(); + if (next_submit_ns > now) { + os_nanosleep((int64_t)(next_submit_ns - now)); + } + next_submit_ns += (uint64_t)(1e9 / (fps > 0 ? fps : 30)); + } + uint64_t t_submit = os_monotonic_get_ns(); + uint64_t frame_id = 0; + xret = ipc_client_lift_submit(&ipc_c, sid, (xrt_graphics_buffer_handle_t)g.in_handle, false, + (uint32_t)iw, (uint32_t)ih, (int64_t)i, ¶ms, nullptr, 0, &frame_id); + if (xret != XRT_SUCCESS) { + printf("%6d submit refused (xrt_result=%d) — retrying next frame\n", i, (int)xret); + continue; + } + + // Sequential: wait (bounded) for THIS frame's result. Pipelined: take + // whatever is newest right now. + const uint64_t deadline = t_submit + (uint64_t)(mode == XRT_DP_LIFT_MODE_GAUSSIANS ? 30e9 : 5e9); + for (;;) { + bool got = false; + uint64_t t_acq = 0; // stamped when the result is acquired, BEFORE any readback/encode + std::string file; + std::string outdesc; + uint64_t got_id = 0; + uint64_t svc_lat = 0; + if (mode == XRT_DP_LIFT_MODE_GAUSSIANS) { + bool ready = false, delivered = false; + struct xrt_lift_blob_info bi = {}; + xret = ipc_client_lift_acquire_blob(&ipc_c, sid, 0, nullptr, &ready, &delivered, &bi); + if (xret == XRT_SUCCESS && ready) { + std::vector bytes((size_t)bi.byte_count); + xret = ipc_client_lift_acquire_blob(&ipc_c, sid, bi.byte_count, bytes.data(), + &ready, &delivered, &bi); + if (xret == XRT_SUCCESS && delivered) { + got = true; + t_acq = os_monotonic_get_ns(); + got_id = bi.frame_id; + char name[64]; + snprintf(name, sizeof(name), "lift_out_%d.%s", i, + bi.format == XRT_DP_LIFT_BLOB_SOG ? "sog" : "ply"); + file = out_dir + "\\" + name; + if (!no_write && (i % write_every == 0)) { + FILE *f = fopen(file.c_str(), "wb"); + if (f != nullptr) { + fwrite(bytes.data(), 1, bytes.size(), f); + fclose(f); + written++; + } + } else { + file = "(not written)"; + } + char d[64]; + snprintf(d, sizeof(d), "%llu B", (unsigned long long)bi.byte_count); + outdesc = d; + } + } + } else { + bool ready = false; + struct xrt_lift_result r = {}; + xret = ipc_client_lift_acquire(&ipc_c, sid, &ready, &r); + if (xret == XRT_SUCCESS && ready) { + got = true; + t_acq = os_monotonic_get_ns(); + got_id = r.frame_id; + svc_lat = r.latency_ns; + if (r.output_realloc || g.out_tex == nullptr) { + rel(g.out_tex); + rel(g.out_fence); + bool have = false; + xrt_graphics_buffer_handle_t th = XRT_GRAPHICS_BUFFER_HANDLE_INVALID; + uint32_t ow = 0, oh = 0, of = 0; + if (ipc_client_lift_get_output(&ipc_c, sid, &have, &ow, &oh, &of, + &th) == XRT_SUCCESS && + have) { + g.dev1->OpenSharedResource1( + (HANDLE)th, __uuidof(ID3D11Texture2D), (void **)&g.out_tex); + CloseHandle((HANDLE)th); + } + xrt_graphics_sync_handle_t fh = XRT_GRAPHICS_SYNC_HANDLE_INVALID; + if (ipc_client_lift_get_fence(&ipc_c, sid, &have, &fh) == XRT_SUCCESS && + have) { + g.dev5->OpenSharedFence((HANDLE)fh, __uuidof(ID3D11Fence), + (void **)&g.out_fence); + CloseHandle((HANDLE)fh); + } + } + char name[64]; + snprintf(name, sizeof(name), "lift_out_%d.png", i); + file = out_dir + "\\" + name; + const bool want_write = !no_write && (i % write_every == 0); + if (want_write && read_back_png(g, r.fence_value, r.width, r.height, r.format, file)) { + written++; + } else { + file = "(not written)"; + } + char d[64]; + snprintf(d, sizeof(d), "%ux%u/%uv", r.width, r.height, r.view_count); + outdesc = d; + } + } + if (xret != XRT_SUCCESS) { + printf("%6d acquire failed (xrt_result=%d)\n", i, (int)xret); + break; + } + if (got && (pipelined || got_id >= frame_id)) { + double ms = (double)((t_acq ? t_acq : (uint64_t)os_monotonic_get_ns()) - t_submit) / 1e6; + lat_ms.push_back(ms); + printf("%6d %10llu %10.2fms %10.2fms %12s %s\n", i, (unsigned long long)got_id, ms, + (double)svc_lat / 1e6, outdesc.c_str(), file.c_str()); + break; + } + if (pipelined || (uint64_t)os_monotonic_get_ns() > deadline) { + if (!pipelined) { + printf("%6d no result within the deadline\n", i); + } + break; + } + os_nanosleep(1000 * 1000); + } + } + if (pixels != nullptr) { + stbi_image_free(pixels); + } + + struct xrt_lift_stream_stats st = {}; + if (ipc_client_lift_stats(&ipc_c, sid, &st) == XRT_SUCCESS) { + printf( + "stream stats: submitted=%llu converted=%llu dropped=%llu failed=%llu rate=%.1f Hz " + "latency last=%.2f avg=%.2f min=%.2f max=%.2f ms\n", + (unsigned long long)st.submitted, (unsigned long long)st.converted, (unsigned long long)st.dropped, + (unsigned long long)st.failed, st.rate_hz, st.latency_last_ns / 1e6, st.latency_avg_ns / 1e6, + st.latency_min_ns / 1e6, st.latency_max_ns / 1e6); + } + if (!lat_ms.empty()) { + std::vector s = lat_ms; + std::sort(s.begin(), s.end()); + double sum = 0; + for (double v : s) { + sum += v; + } + printf("submit->acquire: n=%zu avg=%.2f p50=%.2f p95=%.2f min=%.2f max=%.2f ms; %d file(s) written\n", + s.size(), sum / s.size(), s[s.size() / 2], s[(size_t)(s.size() * 0.95)], s.front(), s.back(), + written); + } + (void)ipc_client_lift_stream_destroy(&ipc_c, sid); + ipc_client_connection_fini(&ipc_c); + return lat_ms.empty() ? 4 : 0; +} + +#else // !XRT_OS_WINDOWS + +int +cmd_probe(int argc, const char **argv) +{ + (void)argc; + (void)argv; + printf("displayxr-cli lift probe: Windows-only (D3D11 shared textures). `lift caps` works everywhere.\n"); + return 2; +} + +#endif // XRT_OS_WINDOWS + +} // namespace + +int +cli_cmd_lift(int argc, const char **argv) +{ + if (argc >= 3 && strcmp(argv[2], "caps") == 0) { + return cmd_caps(argc, argv); + } + if (argc >= 3 && strcmp(argv[2], "probe") == 0) { + return cmd_probe(argc, argv); + } + printf( + "usage: displayxr-cli lift caps [--json] [--wait S]\n" + " displayxr-cli lift probe [--mode depth|sbs|nview|gaussians] [--n N] ...\n"); + return 1; +} + +#else // !CLI_HAVE_IPC + +int +cli_cmd_lift(int argc, const char **argv) +{ + (void)argc; + (void)argv; + printf("displayxr-cli lift: this build has no IPC client (XRT_FEATURE_SERVICE off).\n"); + return 2; +} + +#endif diff --git a/src/xrt/targets/cli/cli_common.h b/src/xrt/targets/cli/cli_common.h index 879f61df5..9d32e5ab8 100644 --- a/src/xrt/targets/cli/cli_common.h +++ b/src/xrt/targets/cli/cli_common.h @@ -48,6 +48,10 @@ cli_cmd_clients(int argc, const char **argv); int cli_cmd_test(int argc, const char **argv); +//! `lift` — XR_DXR_lift (ADR-042): `lift caps`, `lift probe `. +int +cli_cmd_lift(int argc, const char **argv); + /*! * True if @p flag appears anywhere in argv[2..]. Commands take raw * argc/argv; this is the one shared option-scan (e.g. `--json`). diff --git a/src/xrt/targets/cli/cli_main.c b/src/xrt/targets/cli/cli_main.c index 65f46bd96..eb017a351 100644 --- a/src/xrt/targets/cli/cli_main.c +++ b/src/xrt/targets/cli/cli_main.c @@ -48,6 +48,9 @@ cli_print_help(int argc, const char **argv) P(" [--claims] - Also show which plug-in claims each display (loads plug-ins).\n"); P(" clients [--json] - List the running service's IPC clients with their verified class\n"); P(" (#960). Connects over IPC as a DIAG client; non-elevated on Windows.\n"); + P(" lift <...> - 2D->3D conversion module (XR_DXR_lift, ADR-042), over IPC (DIAG).\n"); + P(" 'lift caps [--json]', 'lift probe [--mode depth|sbs|\n"); + P(" nview|gaussians] [--n N]' (Windows) — writes lift_out_.png/.ply.\n"); P(" test - List found devices and role assignments, for prober testing.\n"); P(" probe - Just probe and then exit.\n"); @@ -100,5 +103,8 @@ main(int argc, const char **argv) if (strcmp(argv[1], "probe") == 0) { return cli_cmd_probe(argc, argv); } + if (strcmp(argv[1], "lift") == 0) { + return cli_cmd_lift(argc, argv); + } return cli_print_help(argc, argv); } diff --git a/src/xrt/targets/cli/cli_query.c b/src/xrt/targets/cli/cli_query.c index 53047162e..db9869cd6 100644 --- a/src/xrt/targets/cli/cli_query.c +++ b/src/xrt/targets/cli/cli_query.c @@ -158,6 +158,84 @@ probe_zone_caps_d3d11(struct cli_query_result *r, const struct xrt_plugin_iface ID3D11Device_Release(device); } +/*! + * ADR-042 — headless lift-caps probe (XR_DXR_lift). Same shape as the zone + * probe: a WARP device, the plug-in's LIFT-ONLY factory when it has one + * (xrt_plugin_iface::create_dp_d3d11_lift, struct_size-gated) — else its + * ordinary factory with a NULL window, which is what the service falls back + * to — then lift_get_caps. ABSENCE NEVER FAILS (sim_display, every plug-in + * without a module: modes 0 passes); only a malformed answer does. + */ +static void +probe_lift_caps_d3d11(struct cli_query_result *r, const struct xrt_plugin_iface *iface) +{ + xrt_dp_factory_d3d11_fn_t factory = iface->create_dp_d3d11; + const char *which = "ordinary"; + if (iface->struct_size >= offsetof(struct xrt_plugin_iface, create_dp_d3d11_lift) + + sizeof(iface->create_dp_d3d11_lift) && + iface->create_dp_d3d11_lift != NULL) { + factory = iface->create_dp_d3d11_lift; + which = "lift-only"; + } + if (factory == NULL) { + snprintf(r->lift_probe_note, sizeof(r->lift_probe_note), "not probed: no D3D11 DP factory (OK)"); + return; + } + + ID3D11Device *device = NULL; + ID3D11DeviceContext *context = NULL; + D3D_FEATURE_LEVEL fl; + HRESULT hr = + D3D11CreateDevice(NULL, D3D_DRIVER_TYPE_WARP, NULL, 0, NULL, 0, D3D11_SDK_VERSION, &device, &fl, &context); + if (FAILED(hr) || device == NULL || context == NULL) { + snprintf(r->lift_probe_note, sizeof(r->lift_probe_note), + "not probed: WARP D3D11 device creation failed (0x%08lx, OK)", (unsigned long)hr); + if (context != NULL) { + ID3D11DeviceContext_Release(context); + } + if (device != NULL) { + ID3D11Device_Release(device); + } + return; + } + + struct xrt_display_processor_d3d11 *xdp = NULL; + xrt_result_t xret = factory(device, context, NULL, &xdp); + if (xret != XRT_SUCCESS || xdp == NULL) { + snprintf(r->lift_probe_note, sizeof(r->lift_probe_note), + "not probed: %s D3D11 DP factory declined (xret=%d, OK)", which, (int)xret); + ID3D11DeviceContext_Release(context); + ID3D11Device_Release(device); + return; + } + + struct xrt_dp_lift_caps caps; + bool got = xrt_display_processor_d3d11_has_lift(xdp) && xrt_display_processor_d3d11_lift_get_caps(xdp, &caps); + if (!got) { + snprintf(r->lift_probe_note, sizeof(r->lift_probe_note), + "no conversion module (modes 0) via the %s factory (OK)", which); + } else { + r->lift_caps_probed = true; + r->lift_caps = caps; + const uint32_t known = XRT_DP_LIFT_MODE_DEPTH | XRT_DP_LIFT_MODE_SBS | XRT_DP_LIFT_MODE_NVIEW | + XRT_DP_LIFT_MODE_GAUSSIANS; + bool malformed = (caps.modes & ~known) != 0 || caps.state > XRT_DP_LIFT_STATE_READY || + (caps.modes != 0 && caps.state == XRT_DP_LIFT_STATE_READY && caps.max_streams == 0) || + ((caps.modes & XRT_DP_LIFT_MODE_NVIEW) != 0 && caps.max_views < 2) || + caps.depth_semantics > XRT_DP_LIFT_DEPTH_METRIC; + r->lift_caps_malformed = malformed; + snprintf(r->lift_probe_note, sizeof(r->lift_probe_note), + "%smodes=0x%x state=%u streams=%u views=%u depth=%s latency=%.1fms backend='%s' (%s factory)", + malformed ? "MALFORMED caps: " : "", caps.modes, caps.state, caps.max_streams, caps.max_views, + caps.depth_semantics == XRT_DP_LIFT_DEPTH_METRIC ? "metric" : "relative", + (double)caps.typical_latency_ns / 1e6, caps.backend, which); + } + + xrt_display_processor_d3d11_destroy(&xdp); + ID3D11DeviceContext_Release(context); + ID3D11Device_Release(device); +} + //! Pack a Windows LUID the way Vulkan reports deviceLUID (raw bytes, LowPart //! first) — the same packing @ref d3d_scanout_adapter_luid returns, so the two //! can be compared directly. @@ -1459,8 +1537,15 @@ cli_query_fill(struct cli_query_result *r, struct cli_query_handles *h, const st if (r->zone_caps_malformed) { r->result_code = CLI_SELFTEST_BAD_ZONE_CAPS; } + // ADR-042 — lift caps. Absence never fails; only malformed caps do. + probe_lift_caps_d3d11(r, iface); + if (r->lift_caps_malformed && r->result_code == CLI_SELFTEST_PASS) { + r->result_code = CLI_SELFTEST_BAD_LIFT_CAPS; + } #else snprintf(r->zone_probe_note, sizeof(r->zone_probe_note), "not probed: zone-caps probe is Windows-only (OK)"); + snprintf(r->lift_probe_note, sizeof(r->lift_probe_note), + "not probed: the lift module is a D3D11-service feature (Windows-only); modes 0 (OK)"); #endif // #1234 / #902 — can the Vulkan loader still reach the queue-lock layer? @@ -1885,6 +1970,9 @@ cli_query_print_info_text(const struct cli_query_result *r) } else { PT("%s\n", r->zone_probe_note[0] != '\0' ? r->zone_probe_note : "not evaluated"); } + + P(" :: 2D->3D conversion module (XR_DXR_lift / ADR-042, headless D3D11 WARP probe)\n"); + PT("%s\n", r->lift_probe_note[0] != '\0' ? r->lift_probe_note : "not evaluated"); } cJSON * @@ -2144,6 +2232,21 @@ cli_query_info_to_cjson(const struct cli_query_result *r) } } + // ADR-042 lift-caps probe. + { + cJSON *lc = cJSON_AddObjectToObject(root, "lift_caps"); + cJSON_AddBoolToObject(lc, "probed", r->lift_caps_probed); + cJSON_AddBoolToObject(lc, "malformed", r->lift_caps_malformed); + cJSON_AddStringToObject(lc, "note", r->lift_probe_note[0] != '\0' ? r->lift_probe_note : "not evaluated"); + if (r->lift_caps_probed) { + cJSON_AddNumberToObject(lc, "modes", (double)r->lift_caps.modes); + cJSON_AddNumberToObject(lc, "state", (double)r->lift_caps.state); + cJSON_AddNumberToObject(lc, "max_streams", (double)r->lift_caps.max_streams); + cJSON_AddNumberToObject(lc, "max_views", (double)r->lift_caps.max_views); + cJSON_AddStringToObject(lc, "backend", r->lift_caps.backend); + } + } + // #224 / ADR-027 P4 zone-caps probe. { cJSON *zc = cJSON_AddObjectToObject(root, "zone_caps"); @@ -2403,6 +2506,14 @@ build_checks(const struct cli_query_result *r, struct check *out) snprintf(c->detail, sizeof(c->detail), "%s", r->zone_probe_note[0] != '\0' ? r->zone_probe_note : "not evaluated"); + // ADR-042 — lift-caps probe (XR_DXR_lift). ABSENCE NEVER FAILS: modes 0 + // passes; only a present-but-malformed caps struct fails (BAD_LIFT_CAPS). + c = &out[n++]; + c->name = "lift_caps"; + c->ok = !r->lift_caps_malformed; + snprintf(c->detail, sizeof(c->detail), "%s", + r->lift_probe_note[0] != '\0' ? r->lift_probe_note : "not evaluated"); + // ADR-034 / #823 — input-provider check. ABSENCE NEVER FAILS: ok // stays true with no provider registered, ForceQwerty set, every // provider declining, or the active provider's hardware unplugged. diff --git a/src/xrt/targets/cli/cli_query.h b/src/xrt/targets/cli/cli_query.h index 6a2636952..3bbf724c2 100644 --- a/src/xrt/targets/cli/cli_query.h +++ b/src/xrt/targets/cli/cli_query.h @@ -23,6 +23,7 @@ #include "os/os_display_scale.h" #include "xrt/xrt_device.h" #include "xrt/xrt_display_zones.h" +#include "xrt/xrt_dp_lift.h" #include #include @@ -109,6 +110,14 @@ enum cli_selftest_result * navigating provider. */ CLI_SELFTEST_BAD_RIG = 10, + /*! + * The plug-in's 2D→3D conversion module (XR_DXR_lift, ADR-042) answered + * a MALFORMED caps struct (unknown mode bits, a state out of range, a + * READY module with no streams, an NVIEW module with < 2 views). ABSENCE + * NEVER FAILS: no module (modes 0) — the state of every plug-in without + * one, and of sim_display — passes. + */ + CLI_SELFTEST_BAD_LIFT_CAPS = 11, }; //! Hardware adapters reported by the GPU-topology probe (#918). @@ -268,6 +277,15 @@ struct cli_query_result char zone_probe_note[128]; struct xrt_dp_local_zone_caps zone_caps; + /* ADR-042 lift-caps probe (Windows-only: WARP D3D11 device + the plug-in's + * lift-only DP factory, else its ordinary one with a NULL window, then + * lift_get_caps). Same outcome contract as the zone probe: modes 0 / no + * slots / no factory pass; only a malformed answer fails. */ + bool lift_caps_probed; + bool lift_caps_malformed; + char lift_probe_note[160]; + struct xrt_dp_lift_caps lift_caps; + /* Input-provider checks (ADR-034 / #823). Absence never fails: with * no provider registered — or the ForceQwerty override set, or the * provider's hardware simply not plugged in — the fields stay "not diff --git a/src/xrt/targets/common/target_instance.c b/src/xrt/targets/common/target_instance.c index 08ae0e5a4..461def035 100644 --- a/src/xrt/targets/common/target_instance.c +++ b/src/xrt/targets/common/target_instance.c @@ -127,6 +127,14 @@ fill_dp_factories_from_plugin(struct xrt_system_compositor_info *info, const str if (plugin->create_dp_d3d11 != NULL) { info->dp_factory_d3d11 = (void *)plugin->create_dp_d3d11; } + // ADR-042: the optional lift-only factory, appended to the iface (read only + // when the plug-in's struct_size covers it). + info->dp_factory_d3d11_lift = NULL; + if (plugin->struct_size >= offsetof(struct xrt_plugin_iface, create_dp_d3d11_lift) + + sizeof(plugin->create_dp_d3d11_lift) && + plugin->create_dp_d3d11_lift != NULL) { + info->dp_factory_d3d11_lift = (void *)plugin->create_dp_d3d11_lift; + } if (plugin->create_dp_d3d12 != NULL) { info->dp_factory_d3d12 = (void *)plugin->create_dp_d3d12; } diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index b8eedf879..9ab14e607 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -28,6 +28,7 @@ set(tests tests_dp_weave_scope tests_dp_vk_snap_window_rect tests_weave_snap_grid + tests_lift_mailbox tests_space_overseer_rig tests_rig_composer tests_aux_x11_scale @@ -696,3 +697,8 @@ endif() add_test(NAME tests_ipc_proto COMMAND ${PYTHON_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/tests_ipc_proto.py ${PROJECT_SOURCE_DIR}/src/xrt/ipc/shared ) + +# ADR-042: the sim_display FAKE lift's splat PLY is pinned host-side next to the +# lift mailbox tests (the module that emits it is Windows/D3D11-only). +target_sources(tests_lift_mailbox PRIVATE ${PROJECT_SOURCE_DIR}/src/xrt/drivers/sim_display/sim_display_fake_ply.c) +target_include_directories(tests_lift_mailbox PRIVATE ${PROJECT_SOURCE_DIR}/src/xrt/drivers/sim_display) diff --git a/tests/tests_lift_mailbox.cpp b/tests/tests_lift_mailbox.cpp new file mode 100644 index 000000000..06ef9adec --- /dev/null +++ b/tests/tests_lift_mailbox.cpp @@ -0,0 +1,680 @@ +// Copyright 2026, The DisplayXR Project +// SPDX-License-Identifier: BSL-1.0 +/*! + * @file + * @brief XR_DXR_lift (ADR-042): the per-stream mailbox / output-ring state + * machine the D3D11 service's lift thread runs on. + * + * What must hold, because the weave and IPC threads are built on it: + * + * - the PRODUCER never waits: with one frame converting there is always a + * writable input slot; + * - LATEST WINS: a pending frame superseded before the worker takes it is + * dropped and counted, never queued behind the newer one; + * - frame ids are per-stream, monotonic from 1, and an aborted snapshot + * consumes none; + * - the worker never writes the LATEST result nor a PINNED slot (a consumer is + * mid-copy), and waits (begin_output false) rather than breaking that; + * - a "newer-than" acquire returns each result at most once; the weave's + * plain pin always sees the latest; + * - sourceTime is echoed verbatim and latency is submit → publish. + */ + +#include "catch_amalgamated.hpp" + +#include "util/u_lift_mailbox.h" + +#include "sim_display_fake_ply.h" + +#include +#include +#include +#include + +namespace { + +//! Submit one frame end to end on the producer side; returns its frame id. +uint64_t +submit(u_lift_mailbox &mb, int64_t source_time, uint64_t now_ns) +{ + int32_t slot = -1; + REQUIRE(u_lift_mailbox_begin_submit(&mb, &slot)); + return u_lift_mailbox_commit_submit(&mb, slot, source_time, now_ns, 640, 360); +} + +//! Worker: convert whatever is pending into an output slot at @p done_ns. +bool +convert(u_lift_mailbox &mb, uint64_t start_ns, uint64_t done_ns, u_lift_frame_meta *out = nullptr) +{ + int32_t in = -1; + u_lift_frame_meta meta = {}; + if (!u_lift_mailbox_take_pending(&mb, start_ns, &in, &meta)) { + return false; + } + int32_t o = -1; + REQUIRE(u_lift_mailbox_begin_output(&mb, &o)); + u_lift_mailbox_finish_input(&mb, in); + u_lift_mailbox_publish_output(&mb, o, &meta, done_ns); + if (out != nullptr) { + *out = meta; + } + return true; +} + +} // namespace + +TEST_CASE("lift mailbox: frame ids start at 1 and are monotonic", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + CHECK_FALSE(u_lift_mailbox_has_pending(&mb)); + CHECK_FALSE(u_lift_mailbox_has_result(&mb)); + CHECK(submit(mb, 100, 1000) == 1); + CHECK(submit(mb, 200, 2000) == 2); + CHECK(submit(mb, 300, 3000) == 3); + CHECK(mb.submitted == 3); +} + +TEST_CASE("lift mailbox: latest wins — an unconverted frame is dropped, not queued", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + submit(mb, 10, 1000); + submit(mb, 20, 2000); + submit(mb, 30, 3000); + CHECK(mb.dropped == 2); + + int32_t in = -1; + u_lift_frame_meta meta = {}; + REQUIRE(u_lift_mailbox_take_pending(&mb, 3500, &in, &meta)); + CHECK(meta.frame_id == 3); + CHECK(meta.source_time == 30); + CHECK(meta.convert_start_ns == 3500); + // Nothing older is left behind. + CHECK_FALSE(u_lift_mailbox_has_pending(&mb)); +} + +TEST_CASE("lift mailbox: the producer never blocks while a frame converts", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + submit(mb, 1, 100); + int32_t in = -1; + REQUIRE(u_lift_mailbox_take_pending(&mb, 150, &in, nullptr)); + + // Many submits while one frame is converting: each one gets a slot, the + // converting slot is never handed out, and only the newest stays pending. + for (int i = 0; i < 10; i++) { + int32_t s = -1; + REQUIRE(u_lift_mailbox_begin_submit(&mb, &s)); + CHECK(s != in); + u_lift_mailbox_commit_submit(&mb, s, 2 + i, 200 + i, 1, 1); + } + CHECK(mb.dropped == 9); + u_lift_mailbox_finish_input(&mb, in); + + u_lift_frame_meta meta = {}; + REQUIRE(u_lift_mailbox_take_pending(&mb, 500, &in, &meta)); + CHECK(meta.frame_id == 11); + CHECK(meta.source_time == 11); +} + +TEST_CASE("lift mailbox: an aborted snapshot consumes no frame id", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + int32_t s = -1; + REQUIRE(u_lift_mailbox_begin_submit(&mb, &s)); + u_lift_mailbox_abort_submit(&mb, s); + CHECK_FALSE(u_lift_mailbox_has_pending(&mb)); + CHECK(submit(mb, 5, 50) == 1); +} + +TEST_CASE("lift mailbox: overwriting the pending slot counts one drop", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + submit(mb, 1, 10); // pending in one slot + int32_t conv = -1; + REQUIRE(u_lift_mailbox_take_pending(&mb, 11, &conv, nullptr)); // converting + submit(mb, 2, 20); // pending in the other + CHECK(mb.dropped == 0); + + // Both slots are busy (converting + pending): the next begin must take the + // PENDING one, counting its frame as dropped. + int32_t s = -1; + REQUIRE(u_lift_mailbox_begin_submit(&mb, &s)); + CHECK(s != conv); + CHECK(mb.dropped == 1); + CHECK(u_lift_mailbox_commit_submit(&mb, s, 3, 30, 1, 1) == 3); +} + +TEST_CASE("lift mailbox: results, newer-than acquire, sourceTime echo, latency", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + int32_t slot = -1; + CHECK_FALSE(u_lift_mailbox_pin_latest(&mb, true, &slot, nullptr)); + CHECK_FALSE(u_lift_mailbox_pin_latest(&mb, false, &slot, nullptr)); + + submit(mb, 777, 1000); + REQUIRE(convert(mb, 1100, 1000 + 25)); + + u_lift_frame_meta m = {}; + REQUIRE(u_lift_mailbox_pin_latest(&mb, true, &slot, &m)); + CHECK(m.frame_id == 1); + CHECK(m.source_time == 777); + CHECK(m.done_ns - m.submit_ns == 25); + CHECK(mb.lat_last_ns == 25); + u_lift_mailbox_unpin(&mb, slot); + + // Same result again: a newer-than acquire says NOT READY... + CHECK_FALSE(u_lift_mailbox_pin_latest(&mb, true, &slot, nullptr)); + // ...the weave's plain pin still gets it. + REQUIRE(u_lift_mailbox_pin_latest(&mb, false, &slot, &m)); + CHECK(m.frame_id == 1); + u_lift_mailbox_unpin(&mb, slot); + + submit(mb, 888, 2000); + REQUIRE(convert(mb, 2010, 2000 + 45)); + REQUIRE(u_lift_mailbox_pin_latest(&mb, true, &slot, &m)); + CHECK(m.frame_id == 2); + CHECK(m.source_time == 888); + u_lift_mailbox_unpin(&mb, slot); + + CHECK(mb.converted == 2); + CHECK(mb.lat_min_ns == 25); + CHECK(mb.lat_max_ns == 45); + CHECK(mb.lat_ema_ns > 25); + CHECK(mb.lat_ema_ns < 45); +} + +TEST_CASE("lift mailbox: the worker never overwrites the latest or a pinned slot", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + + submit(mb, 1, 10); + REQUIRE(convert(mb, 11, 12)); // result A = latest + int32_t a = mb.latest; + + // A consumer pins A (the weave is copying it). + int32_t pinned = -1; + REQUIRE(u_lift_mailbox_pin_latest(&mb, false, &pinned, nullptr)); + CHECK(pinned == a); + + // Worker publishes B into the other slot — allowed. + submit(mb, 2, 20); + REQUIRE(convert(mb, 21, 22)); + int32_t b = mb.latest; + CHECK(b != a); + + // Next result: the only non-latest slot is A, still pinned → must wait. + int32_t o = -1; + CHECK_FALSE(u_lift_mailbox_begin_output(&mb, &o)); + + // Unpin → A becomes writable, B stays untouched. + u_lift_mailbox_unpin(&mb, pinned); + REQUIRE(u_lift_mailbox_begin_output(&mb, &o)); + CHECK(o == a); + // ...and while A is being written, a consumer still reads B. + int32_t r = -1; + u_lift_frame_meta m = {}; + REQUIRE(u_lift_mailbox_pin_latest(&mb, false, &r, &m)); + CHECK(r == b); + CHECK(m.frame_id == 2); + u_lift_mailbox_unpin(&mb, r); +} + +TEST_CASE("lift mailbox: an abandoned conversion leaves the previous result in place", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + submit(mb, 1, 10); + REQUIRE(convert(mb, 11, 12)); + int32_t first = mb.latest; + + submit(mb, 2, 20); + int32_t in = -1; + REQUIRE(u_lift_mailbox_take_pending(&mb, 21, &in, nullptr)); + int32_t o = -1; + REQUIRE(u_lift_mailbox_begin_output(&mb, &o)); + u_lift_mailbox_finish_input(&mb, in); + u_lift_mailbox_abort_output(&mb, o); + + CHECK(mb.failed == 1); + CHECK(mb.latest == first); + CHECK(u_lift_mailbox_has_result(&mb)); + int32_t r = -1; + u_lift_frame_meta m = {}; + REQUIRE(u_lift_mailbox_pin_latest(&mb, false, &r, &m)); + CHECK(m.frame_id == 1); + u_lift_mailbox_unpin(&mb, r); +} + +TEST_CASE("lift mailbox: out-of-range slots from the wire are inert", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + CHECK(u_lift_mailbox_commit_submit(&mb, 7, 0, 0, 1, 1) == 0); + CHECK(u_lift_mailbox_commit_submit(&mb, -1, 0, 0, 1, 1) == 0); + u_lift_mailbox_abort_submit(&mb, 99); + u_lift_mailbox_finish_input(&mb, -3); + u_lift_mailbox_unpin(&mb, 5); + u_lift_frame_meta meta = {}; + u_lift_mailbox_publish_output(&mb, 9, &meta, 0); + u_lift_mailbox_abort_output(&mb, 9); + CHECK(mb.submitted == 0); + CHECK(mb.latest == -1); + // Committing a slot that was never begun is refused too. + CHECK(u_lift_mailbox_commit_submit(&mb, 0, 0, 0, 1, 1) == 0); +} + +/* + * + * Cross-stream scheduling (XrLiftPriorityDXR). + * + */ + +namespace { + +std::vector +plan(u_lift_sched &s, const std::vector &e) +{ + uint64_t out[32] = {}; + uint32_t n = u_lift_sched_plan(&s, e.data(), (uint32_t)e.size(), out, 32); + return std::vector(out, out + n); +} + +} // namespace + +TEST_CASE("lift sched: HIGH every round, NORMAL round-robin one per round", "[lift][sched]") +{ + u_lift_sched s; + u_lift_sched_init(&s); + // A call: the active speaker HIGH, three other tiles NORMAL. + std::vector e = { + {1, U_LIFT_PRIORITY_NORMAL, true}, + {2, U_LIFT_PRIORITY_HIGH, true}, + {3, U_LIFT_PRIORITY_NORMAL, true}, + {4, U_LIFT_PRIORITY_NORMAL, true}, + }; + CHECK(plan(s, e) == std::vector{2, 1}); + CHECK(plan(s, e) == std::vector{2, 3}); + CHECK(plan(s, e) == std::vector{2, 4}); + CHECK(plan(s, e) == std::vector{2, 1}); // wraps +} + +TEST_CASE("lift sched: only streams with a new frame are planned", "[lift][sched]") +{ + u_lift_sched s; + u_lift_sched_init(&s); + std::vector e = { + {1, U_LIFT_PRIORITY_HIGH, false}, + {2, U_LIFT_PRIORITY_NORMAL, false}, + {3, U_LIFT_PRIORITY_NORMAL, true}, + }; + CHECK(plan(s, e) == std::vector{3}); + e[2].pending = false; + CHECK(plan(s, e).empty()); + CHECK(s.round == 1); // an idle service does not advance rounds +} + +TEST_CASE("lift sched: LOW every 4th round, PAUSED never", "[lift][sched]") +{ + u_lift_sched s; + u_lift_sched_init(&s); + std::vector e = { + {1, U_LIFT_PRIORITY_NORMAL, true}, + {2, U_LIFT_PRIORITY_LOW, true}, + {3, U_LIFT_PRIORITY_PAUSED, true}, + }; + int low_hits = 0; + for (int r = 0; r < 16; r++) { + auto p = plan(s, e); + for (uint64_t id : p) { + CHECK(id != 3); + low_hits += id == 2 ? 1 : 0; + } + CHECK(std::find(p.begin(), p.end(), 1) != p.end()); // NORMAL every round + } + CHECK(low_hits == 16 / U_LIFT_LOW_EVERY_N); + + // Only a PAUSED stream pending: nothing, and no round consumed. + std::vector paused = {{3, U_LIFT_PRIORITY_PAUSED, true}}; + uint64_t before = s.round; + CHECK(plan(s, paused).empty()); + CHECK(s.round == before); +} + +TEST_CASE("lift mailbox: effective rate from publish intervals", "[lift]") +{ + u_lift_mailbox mb; + u_lift_mailbox_init(&mb); + CHECK(u_lift_mailbox_rate_hz(&mb) == 0.0f); + uint64_t t = 1000; + for (int i = 0; i < 20; i++) { + submit(mb, i, t); + REQUIRE(convert(mb, t + 1, t + 2)); + t += 40 * 1000 * 1000; // 25 Hz + } + CHECK(u_lift_mailbox_rate_hz(&mb) == Catch::Approx(25.0f).epsilon(0.01)); +} + +TEST_CASE("sim fake lift: the splat PLY is a valid two-layer 3DGS binary PLY", "[lift][ply]") +{ + const uint32_t w = 64, h = 48; + std::vector rgba(w * h * 4, 0); + for (uint32_t i = 0; i < w * h; i++) { + rgba[i * 4 + 0] = 255; // red photo + rgba[i * 4 + 3] = 255; + } + size_t need = sim_fake_ply_write(nullptr, 0, rgba.data(), w, h, w * 4); + REQUIRE(need > 0); + std::vector ply(need); + CHECK(sim_fake_ply_write(ply.data(), ply.size() - 1, rgba.data(), w, h, w * 4) == need); // too small: nothing + CHECK(sim_fake_ply_write(ply.data(), ply.size(), rgba.data(), w, h, w * 4) == need); + + std::string text(reinterpret_cast(ply.data()), std::min(ply.size(), 1024)); + REQUIRE(text.rfind("ply\nformat binary_little_endian 1.0\n", 0) == 0); + CHECK(text.find("element vertex " + std::to_string(SIM_FAKE_PLY_SPLATS) + "\n") != std::string::npos); + for (const char *prop : {"property float f_dc_0\n", "property float opacity\n", "property float rot_3\n"}) { + CHECK(text.find(prop) != std::string::npos); + } + size_t hdr_end = text.find("end_header\n"); + REQUIRE(hdr_end != std::string::npos); + hdr_end += strlen("end_header\n"); + CHECK(ply.size() - hdr_end == (size_t)SIM_FAKE_PLY_SPLATS * SIM_FAKE_PLY_FLOATS_PER_SPLAT * 4); + CHECK(SIM_FAKE_PLY_SPLATS >= 200); // "a few hundred splats" + + // First splat: front layer (z = 0), red dominant in its SH DC term. + float f[SIM_FAKE_PLY_FLOATS_PER_SPLAT]; + memcpy(f, ply.data() + hdr_end, sizeof(f)); + CHECK(f[2] == 0.0f); + CHECK(f[6] > f[7]); + CHECK(f[6] > f[8]); + // Last splat: back layer, behind the front one. + memcpy(f, ply.data() + ply.size() - sizeof(f), sizeof(f)); + CHECK(f[2] > 0.0f); +} + +TEST_CASE("lift snapshot cap: long edge capped, aspect kept, dims even", "[lift][cap]") +{ + uint32_t w = 0, h = 0; + + // The measured case: a fullscreen player on an 8K panel. + CHECK(u_lift_cap_dims(7680, 4319, 1920, &w, &h)); + CHECK(w == 1920); + CHECK(h == 1080); + + // Portrait: the long edge is the height. + CHECK(u_lift_cap_dims(2160, 3840, 1920, &w, &h)); + CHECK(w == 1080); + CHECK(h == 1920); + + // Square, and an odd cap rounds DOWN to even (never exceeds the cap). + CHECK(u_lift_cap_dims(4000, 4000, 1921, &w, &h)); + CHECK(w == 1920); + CHECK(h == 1920); + + // Extreme aspect: the short edge never drops below 2. + CHECK(u_lift_cap_dims(8000, 1, 1920, &w, &h)); + CHECK(w == 1920); + CHECK(h == 2); + + // Already fits, exactly at the cap, and cap 0: unchanged, not "capped". + CHECK_FALSE(u_lift_cap_dims(1280, 721, 1920, &w, &h)); + CHECK(w == 1280); + CHECK(h == 721); + CHECK_FALSE(u_lift_cap_dims(1920, 1080, 1920, &w, &h)); + CHECK(w == 1920); + CHECK(h == 1080); + CHECK_FALSE(u_lift_cap_dims(7680, 4319, 0, &w, &h)); + CHECK(w == 7680); + CHECK(h == 4319); + + // Every result is even and inside the cap, aspect within one even step. + for (uint32_t sw = 1921; sw < 8000; sw += 377) { + for (uint32_t sh = 3; sh < 5000; sh += 211) { + REQUIRE(u_lift_cap_dims(sw, sh, 1920, &w, &h) == ((sw > sh ? sw : sh) > 1920)); + if ((sw > sh ? sw : sh) <= 1920) { + continue; + } + CHECK(w % 2 == 0); + CHECK(h % 2 == 0); + CHECK((w > h ? w : h) == 1920); + const double want = (double)(sw < sh ? sw : sh) * 1920.0 / (double)(sw > sh ? sw : sh); + const double got = (double)(w < h ? w : h); + CHECK(((got - want <= 1.0 && want - got <= 1.0) || got == 2.0)); + } + } +} + +TEST_CASE("lift snapshot cap: DXR_LIFT_MAX_INPUT_EDGE parsing", "[lift][cap]") +{ + CHECK(u_lift_max_input_edge_parse(nullptr) == U_LIFT_MAX_INPUT_EDGE_DEFAULT); + CHECK(u_lift_max_input_edge_parse("") == U_LIFT_MAX_INPUT_EDGE_DEFAULT); + CHECK(u_lift_max_input_edge_parse("abc") == U_LIFT_MAX_INPUT_EDGE_DEFAULT); + CHECK(u_lift_max_input_edge_parse("-5") == U_LIFT_MAX_INPUT_EDGE_DEFAULT); + CHECK(u_lift_max_input_edge_parse("0") == 0); + CHECK(u_lift_max_input_edge_parse("1") == U_LIFT_MAX_INPUT_EDGE_MIN); + CHECK(u_lift_max_input_edge_parse("255") == U_LIFT_MAX_INPUT_EDGE_MIN); + CHECK(u_lift_max_input_edge_parse("256") == 256); + CHECK(u_lift_max_input_edge_parse("3840") == 3840); + CHECK(u_lift_max_input_edge_parse("99999999999") == 0xffffu); +} + +// A row profile of @p nr buckets for an @p h-row rect whose picture spans +// rows [pic0, pic1): picture buckets read 0.9, bar buckets 0 (or @p sub in the +// subtitle band [sub0, sub1) of the bottom bar). +static std::vector +lb_rows(uint32_t h, uint32_t nr, uint32_t pic0, uint32_t pic1, float sub = 0.0f, uint32_t sub0 = 0, uint32_t sub1 = 0) +{ + std::vector p(nr, 0.0f); + for (uint32_t i = 0; i < nr; i++) { + const uint32_t a0 = (uint32_t)((uint64_t)i * h / nr); + const uint32_t a1 = (uint32_t)((uint64_t)(i + 1) * h / nr); // exclusive + if (a1 > pic0 && a0 < pic1) { + p[i] = 0.9f; + } else if (a1 > sub0 && a0 < sub1) { + p[i] = sub; + } + } + return p; +} + +TEST_CASE("lift letterbox: 2.39:1 in 16:9 settles, then crops", "[lift][letterbox]") +{ + // 1920x1080 rect, 1920x803 picture centred: bars of 138 / 139 rows. + const uint32_t w = 1920, h = 1080, nr = 512; + const std::vector rows = lb_rows(h, nr, 138, 941); + const std::vector cols(512, 0.9f); + u_lift_letterbox lb = {}; + + for (uint32_t f = 1; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + REQUIRE_FALSE(u_lift_letterbox_update(&lb, w, h, rows.data(), nr, cols.data(), 512)); + REQUIRE_FALSE(u_lift_crop_active(&lb.committed)); + } + CHECK(u_lift_letterbox_update(&lb, w, h, rows.data(), nr, cols.data(), 512)); + CHECK(u_lift_crop_active(&lb.committed)); + // Never into the picture, within a bucket (~2 px) + even rounding of the bar. + CHECK(lb.committed.top <= 138); + CHECK(lb.committed.top >= 132); + CHECK(lb.committed.bottom <= 139); + CHECK(lb.committed.bottom >= 132); + CHECK(lb.committed.left == 0); + CHECK(lb.committed.right == 0); +} + +TEST_CASE("lift letterbox: subtitles in the bar stay in the bar", "[lift][letterbox]") +{ + const uint32_t w = 1920, h = 1080, nr = 512; + // Subtitle text in the bottom bar: ~15% of each of those rows is non-black. + const std::vector rows = lb_rows(h, nr, 138, 941, 0.15f, 980, 1040); + const std::vector cols(512, 0.9f); + u_lift_letterbox lb = {}; + for (uint32_t f = 0; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + (void)u_lift_letterbox_update(&lb, w, h, rows.data(), nr, cols.data(), 512); + } + CHECK(lb.committed.bottom >= 132); // the bar runs to the picture, subtitles inside it + + // DENSE subtitles (big text, ~40% of each row lit) end the bottom bar at the + // picture threshold; the symmetry rule extends it to match the top bar. + const std::vector dense = lb_rows(h, nr, 138, 941, 0.4f, 980, 1040); + u_lift_letterbox lb2 = {}; + for (uint32_t f = 0; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + (void)u_lift_letterbox_update(&lb2, w, h, dense.data(), nr, cols.data(), 512); + } + CHECK(lb2.committed.bottom >= 132); + CHECK(lb2.committed.top >= 132); + + // But a genuinely asymmetric frame (picture reaching the bottom edge's + // neighbourhood, dense) is not extended: 0.9 rows are picture, not text. + const std::vector asym = lb_rows(h, nr, 138, 1040); + u_lift_letterbox lb3 = {}; + for (uint32_t f = 0; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + (void)u_lift_letterbox_update(&lb3, w, h, asym.data(), nr, cols.data(), 512); + } + CHECK(lb3.committed.bottom <= 40); +} + +TEST_CASE("lift letterbox: picture in a bar un-crops at once; black frames change nothing", "[lift][letterbox]") +{ + const uint32_t w = 1920, h = 1080, nr = 512; + const std::vector film = lb_rows(h, nr, 138, 941); + const std::vector full(nr, 0.9f); + const std::vector black(nr, 0.0f); + const std::vector cols(512, 0.9f); + u_lift_letterbox lb = {}; + for (uint32_t f = 0; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + (void)u_lift_letterbox_update(&lb, w, h, film.data(), nr, cols.data(), 512); + } + REQUIRE(u_lift_crop_active(&lb.committed)); + const u_lift_crop before = lb.committed; + + // A cut to black: no content, no change (neither shrink nor grow). + for (int f = 0; f < 200; f++) { + CHECK_FALSE(u_lift_letterbox_update(&lb, w, h, black.data(), nr, cols.data(), 512)); + } + CHECK(lb.committed.top == before.top); + CHECK(lb.committed.bottom == before.bottom); + + // A few frames of picture in the bars (a caption burst, a flash): the crop holds. + for (uint32_t f = 1; f < U_LIFT_LETTERBOX_SHRINK_FRAMES; f++) { + CHECK_FALSE(u_lift_letterbox_update(&lb, w, h, full.data(), nr, cols.data(), 512)); + } + (void)u_lift_letterbox_update(&lb, w, h, film.data(), nr, cols.data(), 512); + CHECK(lb.committed.top == before.top); + CHECK(lb.committed.bottom == before.bottom); + + // Full-frame picture that persists (a 16:9 ad): released after SHRINK_FRAMES. + for (uint32_t f = 1; f < U_LIFT_LETTERBOX_SHRINK_FRAMES; f++) { + CHECK_FALSE(u_lift_letterbox_update(&lb, w, h, full.data(), nr, cols.data(), 512)); + } + CHECK(u_lift_letterbox_update(&lb, w, h, full.data(), nr, cols.data(), 512)); + CHECK_FALSE(u_lift_crop_active(&lb.committed)); + + // A mostly-dark frame with a thin bright strip is not a letterbox. + const std::vector strip = lb_rows(h, nr, 500, 560); + for (uint32_t f = 0; f < 2 * U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + CHECK_FALSE(u_lift_letterbox_update(&lb, w, h, strip.data(), nr, cols.data(), 512)); + } + CHECK_FALSE(u_lift_crop_active(&lb.committed)); +} + +TEST_CASE("lift letterbox: a rect resize resets; pillarbox crops columns", "[lift][letterbox]") +{ + const uint32_t nr = 512; + u_lift_letterbox lb = {}; + // 4:3 content pillarboxed in 1920x1080: columns [240, 1680) are picture. + const std::vector rows(nr, 0.9f); + const std::vector cols = lb_rows(1920, 512, 240, 1680); + for (uint32_t f = 0; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + (void)u_lift_letterbox_update(&lb, 1920, 1080, rows.data(), nr, cols.data(), 512); + } + CHECK(lb.committed.left <= 240); + CHECK(lb.committed.left >= 230); + CHECK(lb.committed.right <= 240); + CHECK(lb.committed.right >= 230); + CHECK(lb.committed.top == 0); + + // Fullscreen: new dims, crop dropped until the new bars settle. + CHECK(u_lift_letterbox_update(&lb, 7680, 4320, rows.data(), nr, cols.data(), 512)); + CHECK_FALSE(u_lift_crop_active(&lb.committed)); +} + +TEST_CASE("lift letterbox: DXR_LIFT_LETTERBOX parse", "[lift][letterbox]") +{ + CHECK(u_lift_letterbox_parse(nullptr)); + CHECK(u_lift_letterbox_parse("")); + CHECK(u_lift_letterbox_parse("1")); + CHECK_FALSE(u_lift_letterbox_parse("0")); +} + +TEST_CASE("lift letterbox: dark picture edges never grow a bar", "[lift][letterbox]") +{ + // A dark scene: the outer columns are dim picture (8% lit), not black bars. + const uint32_t nr = 512; + const std::vector rows(nr, 0.9f); + std::vector cols(512, 0.9f); + for (uint32_t i = 0; i < 140; i++) { + cols[i] = 0.08f; + cols[511 - i] = 0.08f; + } + u_lift_letterbox lb = {}; + for (uint32_t f = 0; f < 3 * U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + CHECK_FALSE(u_lift_letterbox_update(&lb, 2562, 1440, rows.data(), nr, cols.data(), 512)); + } + CHECK_FALSE(u_lift_crop_active(&lb.committed)); +} + +TEST_CASE("lift letterbox: flickering captions keep the bottom bar", "[lift][letterbox]") +{ + // YouTube-style: top bar 184 of 1440, captions come and go in the bottom bar + // (dense, but always with a black gap above them). + const uint32_t w = 2562, h = 1440, nr = 512; + const std::vector plain = lb_rows(h, nr, 184, 1254); + const std::vector caption = lb_rows(h, nr, 184, 1254, 0.7f, 1300, 1400); + const std::vector cols(512, 0.9f); + u_lift_letterbox lb = {}; + for (uint32_t f = 0; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + (void)u_lift_letterbox_update(&lb, w, h, plain.data(), nr, cols.data(), 512); + } + REQUIRE(lb.committed.bottom >= 176); + const u_lift_crop before = lb.committed; + for (int f = 0; f < 300; f++) { + const std::vector &p = (f / 20) % 2 ? caption : plain; + (void)u_lift_letterbox_update(&lb, w, h, p.data(), nr, cols.data(), 512); + REQUIRE(lb.committed.bottom == before.bottom); + REQUIRE(lb.committed.top == before.top); + } +} + +TEST_CASE("lift letterbox: no creep by a few pixels once cropped", "[lift][letterbox]") +{ + // Measured on the panel: bars settling at 250, then 254, then 256 — each a + // re-crop. Once cropped, only a growth beyond the deadband re-crops. + const uint32_t w = 2561, h = 1440, nr = 512; + const std::vector cols(512, 0.9f); + u_lift_letterbox lb = {}; + const std::vector a = lb_rows(h, nr, 252, 1188); + for (uint32_t f = 0; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + (void)u_lift_letterbox_update(&lb, w, h, a.data(), nr, cols.data(), 512); + } + REQUIRE(u_lift_crop_active(&lb.committed)); + const u_lift_crop first = lb.committed; + // Bars 4 px thicker: within the deadband, no re-crop. + const std::vector b = lb_rows(h, nr, 256, 1184); + for (uint32_t f = 0; f < 3 * U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + CHECK_FALSE(u_lift_letterbox_update(&lb, w, h, b.data(), nr, cols.data(), 512)); + } + CHECK(lb.committed.top == first.top); + // Bars 40 px thicker (a genuinely wider aspect): re-crops after settling. + const std::vector c = lb_rows(h, nr, 292, 1148); + bool grew = false; + for (uint32_t f = 0; f < U_LIFT_LETTERBOX_SETTLE_FRAMES; f++) { + grew = u_lift_letterbox_update(&lb, w, h, c.data(), nr, cols.data(), 512) || grew; + } + CHECK(grew); + CHECK(lb.committed.top > first.top + 16); +}