From 0cc0572dd136a8ff07faca05c92ce12cad5b4ff6 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Fri, 18 Sep 2026 16:20:44 +0200 Subject: [PATCH] 10-bit HDR mixer: SDR/HLG/PQ conversion, native 4:2:2, HDR10 outputs, shutdown fixes Every source is normalised to the canvas contract (sdr, hlg, pq) on the GPU by the new tonemap_cuda filter; P010/P210 canvases with 8-bit sources promoted in the compositor; v210 sources stay 4:2:2; HEVC Main10 with HLG/PQ signalling and HDR10 static metadata on outputs; filter EOF-tail, Python shutdown and seek latch fixes; per-node GIL release. Compositor and orchestrator split into modules under src/mixer (primitives/, orchestrator/), libav/avcpp reused where the code re-implemented them; the mixer library ships as pyplumber.mixer. Show files gain working_format, color, latency_ms, v210 sources, tone-mapped renditions and HDR10 metadata; docs and examples follow. Squash of PR #33 (78 commits) onto develop after #34. Co-Authored-By: Claude Fable 5.1 --- .gitattributes | 2 +- AGENTS.md | 4 + Makefile | 7 +- avpmixer/__init__.py | 16 - avpmixer/janus.py | 106 - demos/dmabuf-browser/compose.mixer.yaml | 2 +- demos/dmabuf-browser/consumer/Dockerfile.cuda | 2 +- demos/mixer/Dockerfile | 3 +- demos/mixer/README.md | 37 +- demos/mixer/config.example.hdr.json | 46 + demos/mixer/config.example.json | 6 + demos/mixer/docs/config.md | 155 +- demos/mixer/docs/guide.md | 2 +- demos/mixer/make_config.py | 63 +- demos/mixer/mixer.py | 597 +++--- demos/mixer/smoke_test.py | 4 +- demos/mixer/tests/check_interruptions.py | 4 +- .../mixer/tests/check_transition_recovery.py | 2 +- demos/mixer/tests/conftest.py | 2 +- demos/mixer/tests/sdr_patterns.py | 53 + demos/mixer/tests/smoke_live_takes.py | 84 + demos/mixer/tests/test_control.py | 2 +- demos/mixer/tests/test_graph.py | 305 ++- demos/mixer/tests/test_prewarm.py | 8 +- demos/mixer/tui.py | 4 +- demos/mixer/webui.py | 2 +- demos/playlist/Dockerfile | 2 +- demos/playlist/control.py | 2 +- demos/playlist/docs/guide.md | 6 +- demos/playlist/engine.py | 8 +- demos/playlist/server.py | 2 +- demos/playlist/tests/conftest.py | 2 +- demos/playlist/tests/test_container.py | 2 +- .../0008-avfilter-transition-cuda-10bit.patch | 235 ++ .../ffmpeg/8/0009-avfilter-tonemap-cuda.patch | 1000 +++++++++ deps/ffmpeg/8/bases.env | 6 +- deps/ffmpeg/README.md | 8 + doc/NODES.md | 20 + doc/{mixer_orchestrator.md => mixer.md} | 19 +- .../2026-09-07-playlist-on-mixer-plan.md | 6 +- .../2026-09-08-mixer-config-schema.md | 4 +- doc/research/2026-09-08-mixer-dmabuf-plan.md | 6 +- ...6-09-07-mixer-graph-optimization-design.md | 2 +- pyplumber/__init__.py | 17 +- pyplumber/mixer.py | 5 - pyplumber/mixer/__init__.py | 17 + {avpmixer => pyplumber/mixer}/clipcache.py | 0 pyplumber/mixer/color.py | 133 ++ {avpmixer => pyplumber/mixer}/config.py | 238 ++- {avpmixer => pyplumber/mixer}/control.py | 0 .../mixer}/dmabuf_inputs.py | 2 +- {avpmixer => pyplumber/mixer}/graph.py | 110 +- {avpmixer => pyplumber/mixer}/inputs.py | 56 +- pyplumber/mixer/janus.py | 141 ++ {avpmixer => pyplumber/mixer}/models.py | 3 + {avpmixer => pyplumber/mixer}/prewarm.py | 0 pyplumber/node.py | 4 + src/avplumber.cpp | 13 +- src/avplumber_pybind.cpp | 22 +- src/graph_core.hpp | 11 +- src/hdr_metadata.hpp | 50 + src/instance_shared.hpp | 6 +- src/mixer/Playout.hpp | 14 +- src/mixer/TransitionScheduler.cpp | 111 + src/mixer/TransitionScheduler.hpp | 26 + src/mixer/graph_ops.cpp | 235 ++ src/mixer/graph_ops.hpp | 70 + src/mixer/mixer_orchestrator.cpp | 1898 ----------------- .../MixerOrchestrator.hpp} | 40 +- src/mixer/orchestrator/core.cpp | 325 +++ src/mixer/orchestrator/cut.cpp | 143 ++ src/mixer/orchestrator/fade.cpp | 177 ++ src/mixer/orchestrator/internal.hpp | 21 + src/mixer/orchestrator/overlay.cpp | 113 + src/mixer/orchestrator/scene.cpp | 325 +++ src/mixer/orchestrator/wipe.cpp | 373 ++++ src/mixer/{ => primitives}/Cadence.hpp | 6 +- src/mixer/{ => primitives}/CutLatency.hpp | 0 .../{ => primitives}/CutLatencyProbe.hpp | 16 +- src/mixer/{ => primitives}/MixerState.hpp | 14 +- src/mixer/{ => primitives}/MonotonicClock.hpp | 0 src/mixer/{ => primitives}/OutputSnapshot.hpp | 4 +- src/mixer/{ => primitives}/Snapshot.hpp | 0 .../TickGrid.hpp} | 18 +- src/mixer/primitives/TransitionGuard.hpp | 34 + src/mixer/primitives/compositor_color.hpp | 17 + .../primitives/compositor_geometry.hpp} | 12 +- src/mixer/primitives/compositor_layers.hpp | 217 ++ src/mixer/primitives/pixel_layout.hpp | 210 ++ src/mixer/routing.hpp | 110 + src/nodes/clip_cache/clip_cache.cpp | 2 +- src/nodes/encoders.cpp | 12 +- src/nodes/filters.cpp | 20 +- src/nodes/force_fps.cpp | 4 +- src/nodes/hwaccel/cuda_rect_draw.cpp | 303 +++ src/nodes/hwaccel/cuda_rect_draw.hpp | 81 + src/nodes/hwaccel/cuda_rect_overlay.cpp | 706 +----- src/nodes/hwaccel/cuda_rect_scale.cu | 201 +- src/nodes/hwaccel/egl_image_cuda_overlay.cpp | 5 +- src/nodes/hwaccel/v210_to_cuda.cpp | 229 ++ src/nodes/hwaccel/v210_unpack.cu | 38 + src/nodes/mixer_snapshot.cpp | 22 +- src/nodes/repeat_last_frame.cpp | 4 +- src/nodes/source_switcher.cpp | 11 +- tests/conftest.py | 31 + tests/cpp/test_compositor_color.cpp | 29 + tests/cpp/test_compositor_geometry.cpp | 7 +- tests/cpp/test_cut_latency.cpp | 2 +- tests/cpp/test_hdr_metadata.cpp | 52 + tests/cpp/test_mixer_playout.cpp | 64 +- tests/cpp/test_mixer_snapshot.cpp | 2 +- tests/cpp/test_pixel_layout.cpp | 54 + tests/cuda/_harness.py | 91 + tests/cuda/smoke_cut_latency.py | 2 +- tests/cuda/smoke_interop_matrix.py | 100 + tests/cuda/smoke_mixer_10bit.py | 214 ++ tests/cuda/smoke_output_qualification.py | 180 ++ tests/cuda/smoke_tonemap_transfers.py | 244 +++ tests/cuda/smoke_v210_to_cuda.py | 113 + tests/cuda/test_rect_scale.cu | 73 +- tests/cuda/tonemap_reference.py | 245 +++ tests/cuda/v210_fixture.py | 120 ++ tests/downstream/eka-recorder-smoke.sh | 96 + tests/python/smoke_shutdown.py | 62 + tests/test_compositor_color.py | 6 + tests/test_compositor_geometry.py | 12 +- tests/test_cut_latency.py | 12 +- tests/test_hdr_metadata.py | 6 + tests/test_mixer_color.py | 182 ++ tests/test_mixer_playout.py | 12 +- tests/test_mixer_snapshot.py | 12 +- tests/test_pixel_layout.py | 6 + 132 files changed, 8463 insertions(+), 3409 deletions(-) delete mode 100644 avpmixer/__init__.py delete mode 100644 avpmixer/janus.py create mode 100644 demos/mixer/config.example.hdr.json create mode 100644 demos/mixer/tests/sdr_patterns.py create mode 100644 demos/mixer/tests/smoke_live_takes.py create mode 100644 deps/ffmpeg/8/0008-avfilter-transition-cuda-10bit.patch create mode 100644 deps/ffmpeg/8/0009-avfilter-tonemap-cuda.patch rename doc/{mixer_orchestrator.md => mixer.md} (73%) delete mode 100644 pyplumber/mixer.py create mode 100644 pyplumber/mixer/__init__.py rename {avpmixer => pyplumber/mixer}/clipcache.py (100%) create mode 100644 pyplumber/mixer/color.py rename {avpmixer => pyplumber/mixer}/config.py (57%) rename {avpmixer => pyplumber/mixer}/control.py (100%) rename {avpmixer => pyplumber/mixer}/dmabuf_inputs.py (99%) rename {avpmixer => pyplumber/mixer}/graph.py (84%) rename {avpmixer => pyplumber/mixer}/inputs.py (53%) create mode 100644 pyplumber/mixer/janus.py rename {avpmixer => pyplumber/mixer}/models.py (90%) rename {avpmixer => pyplumber/mixer}/prewarm.py (100%) create mode 100644 src/hdr_metadata.hpp create mode 100644 src/mixer/TransitionScheduler.cpp create mode 100644 src/mixer/TransitionScheduler.hpp create mode 100644 src/mixer/graph_ops.cpp create mode 100644 src/mixer/graph_ops.hpp delete mode 100644 src/mixer/mixer_orchestrator.cpp rename src/mixer/{mixer_orchestrator.hpp => orchestrator/MixerOrchestrator.hpp} (88%) create mode 100644 src/mixer/orchestrator/core.cpp create mode 100644 src/mixer/orchestrator/cut.cpp create mode 100644 src/mixer/orchestrator/fade.cpp create mode 100644 src/mixer/orchestrator/internal.hpp create mode 100644 src/mixer/orchestrator/overlay.cpp create mode 100644 src/mixer/orchestrator/scene.cpp create mode 100644 src/mixer/orchestrator/wipe.cpp rename src/mixer/{ => primitives}/Cadence.hpp (96%) rename src/mixer/{ => primitives}/CutLatency.hpp (100%) rename src/mixer/{ => primitives}/CutLatencyProbe.hpp (82%) rename src/mixer/{ => primitives}/MixerState.hpp (95%) rename src/mixer/{ => primitives}/MonotonicClock.hpp (100%) rename src/mixer/{ => primitives}/OutputSnapshot.hpp (79%) rename src/mixer/{ => primitives}/Snapshot.hpp (100%) rename src/mixer/{FrameRate.hpp => primitives/TickGrid.hpp} (69%) create mode 100644 src/mixer/primitives/TransitionGuard.hpp create mode 100644 src/mixer/primitives/compositor_color.hpp rename src/{hwaccel/CompositorGeometry.hpp => mixer/primitives/compositor_geometry.hpp} (91%) create mode 100644 src/mixer/primitives/compositor_layers.hpp create mode 100644 src/mixer/primitives/pixel_layout.hpp create mode 100644 src/mixer/routing.hpp create mode 100644 src/nodes/hwaccel/cuda_rect_draw.cpp create mode 100644 src/nodes/hwaccel/cuda_rect_draw.hpp create mode 100644 src/nodes/hwaccel/v210_to_cuda.cpp create mode 100644 src/nodes/hwaccel/v210_unpack.cu create mode 100644 tests/conftest.py create mode 100644 tests/cpp/test_compositor_color.cpp create mode 100644 tests/cpp/test_hdr_metadata.cpp create mode 100644 tests/cpp/test_pixel_layout.cpp create mode 100644 tests/cuda/_harness.py create mode 100644 tests/cuda/smoke_interop_matrix.py create mode 100644 tests/cuda/smoke_mixer_10bit.py create mode 100644 tests/cuda/smoke_output_qualification.py create mode 100644 tests/cuda/smoke_tonemap_transfers.py create mode 100644 tests/cuda/smoke_v210_to_cuda.py create mode 100644 tests/cuda/tonemap_reference.py create mode 100644 tests/cuda/v210_fixture.py create mode 100755 tests/downstream/eka-recorder-smoke.sh create mode 100644 tests/python/smoke_shutdown.py create mode 100644 tests/test_compositor_color.py create mode 100644 tests/test_hdr_metadata.py create mode 100644 tests/test_mixer_color.py create mode 100644 tests/test_pixel_layout.py diff --git a/.gitattributes b/.gitattributes index 8316aaa1..3308b6db 100644 --- a/.gitattributes +++ b/.gitattributes @@ -4,5 +4,5 @@ # Preserve whitespace in imported patch payloads and vendored SDK sources. demos/dmabuf-browser/chromium/*.patch whitespace=-trailing-space -deps/ffmpeg-patches/*.patch whitespace=-trailing-space +deps/ffmpeg/8/*.patch whitespace=-trailing-space deps/Optical_Flow_SDK_5.0.7/** whitespace=-trailing-space diff --git a/AGENTS.md b/AGENTS.md index 2f4a11f8..434128e2 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -18,6 +18,10 @@ ## Code & logic style * No copy-paste between nodes or within a file; extract shared base classes, utilities, or helper functions/lambdas. * C++/Python project: prefer C++ idioms (RAII, exceptions) over C patterns, but don't over-apply OOP. +* Reuse libav* and avcpp before writing a helper: `av::Rational`/`av_inv_q`/`av_rescale*` for rates and timestamps, + `av::Timestamp` comparisons, `av_pix_fmt_*`/`av_image_*` for pixel-format geometry, `av_color_*_from_name` for colour + tags, `av_frame_*`/`av_*_metadata_alloc*` for frames and side data. A local reimplementation needs a comment saying + what libav lacks (e.g. the mixer's `TickGrid` rounding rule). ## Framework changes * Don't modify framework source (graph management, control protocol, main, sentinel) unless explicitly asked or the change is necessary, generally useful, and future-proof. diff --git a/Makefile b/Makefile index 1a9a1408..20da8691 100644 --- a/Makefile +++ b/Makefile @@ -81,7 +81,7 @@ override CXXFLAGS += -DSYNCMETER=1 endif nodes_list_file = graph_factory.generated.cpp -CPPSRC = avplumber.cpp util.cpp avutils.cpp graph_core.cpp graph_mgmt.cpp stats.cpp output_control.cpp instance_shared.cpp hwaccel_mgmt.cpp EventLoop.cpp TickSource.cpp rest_client.cpp mixer/mixer_orchestrator.cpp +CPPSRC = avplumber.cpp util.cpp avutils.cpp graph_core.cpp graph_mgmt.cpp stats.cpp output_control.cpp instance_shared.cpp hwaccel_mgmt.cpp EventLoop.cpp TickSource.cpp rest_client.cpp mixer/TransitionScheduler.cpp mixer/graph_ops.cpp mixer/orchestrator/core.cpp mixer/orchestrator/scene.cpp mixer/orchestrator/cut.cpp mixer/orchestrator/fade.cpp mixer/orchestrator/wipe.cpp mixer/orchestrator/overlay.cpp DEPS_LIBS = deps/cpr/build/lib/libcpr.a deps/avcpp/build/src/libavcpp.a # Python extension links via PYTHON_MODULE_EXTRA_LFLAGS (python3-config; -lpython3 is not a valid soname on many distros). LIBS_FLAGS = -lpthread -lcurl -lssl -lcrypto -lboost_thread -lboost_system -lavcodec -lavfilter -lavutil -lavformat -lavdevice -lswscale -lswresample -ldl -lz @@ -162,8 +162,10 @@ $(eval $(call ptx_kernel,$(SRCDIR)/nodes/neural_net/preprocess/mask_assemble.cu, endif ifeq ($(HAVE_CUDA)$(HAVE_NVCC),11) +NODES_SRC += $(SRCDIR)/nodes/hwaccel/v210_to_cuda.cpp +$(eval $(call ptx_kernel,$(SRCDIR)/nodes/hwaccel/v210_unpack.cu,avpl_v210_unpack_ptx,objs/src/nodes/hwaccel/v210_to_cuda.o)) override CXXFLAGS += -DHAVE_CUDA_RECT_SCALE=1 -$(eval $(call ptx_kernel,$(SRCDIR)/nodes/hwaccel/cuda_rect_scale.cu,avpl_rect_scale_ptx,objs/src/nodes/hwaccel/cuda_rect_overlay.o)) +$(eval $(call ptx_kernel,$(SRCDIR)/nodes/hwaccel/cuda_rect_scale.cu,avpl_rect_scale_ptx,objs/src/nodes/hwaccel/cuda_rect_draw.o)) NODES_SRC += $(SRCDIR)/nodes/scene_cut/luma_diff.cpp NODES_SRC += $(SRCDIR)/nodes/scene_cut/hog_diff.cpp $(eval $(call ptx_kernel,$(SRCDIR)/nodes/scene_cut/luma_diff.cu,avpl_luma_diff_ptx,objs/src/nodes/scene_cut/luma_diff.o)) @@ -174,6 +176,7 @@ CUDA_ROOT ?= /usr/local/cuda ifeq ($(HAVE_CUDA),1) NODES_SRC += $(IPC_CUDA_SOURCE_SRC) NODES_SRC += $(SRCDIR)/nodes/hwaccel/cuda_rect_overlay.cpp +NODES_SRC += $(SRCDIR)/nodes/hwaccel/cuda_rect_draw.cpp override CPPSRC += cuda.cpp override CXXFLAGS += -DHAVE_CUDA=1 -Iobjs -I$(CUDA_ROOT)/include -I$(CUDA_ROOT)/targets/x86_64-linux/include override LFLAGS += -L$(CUDA_ROOT)/targets/x86_64-linux/lib -Wl,-rpath,$(CUDA_ROOT)/targets/x86_64-linux/lib diff --git a/avpmixer/__init__.py b/avpmixer/__init__.py deleted file mode 100644 index 081647de..00000000 --- a/avpmixer/__init__.py +++ /dev/null @@ -1,16 +0,0 @@ -"""Mixer graph construction and control, separate from generic pyplumber bindings. - -``MixerGraphBuilder`` needs the native ``pyplumber`` module; it is imported lazily -so that ``avpmixer.control`` and other pure-Python helpers work without it. -""" - -from .models import MixerScene, MixerSource - -__all__ = ["MixerGraphBuilder", "MixerScene", "MixerSource"] - - -def __getattr__(name): - if name == "MixerGraphBuilder": - from .graph import MixerGraphBuilder - return MixerGraphBuilder - raise AttributeError(name) diff --git a/avpmixer/janus.py b/avpmixer/janus.py deleted file mode 100644 index d755ba0d..00000000 --- a/avpmixer/janus.py +++ /dev/null @@ -1,106 +0,0 @@ -"""Video-only H.264 RTP output to a Janus Streaming mountpoint, shared by demos.""" - -from __future__ import annotations - -from dataclasses import dataclass - -RTP_PACKET_SIZE = 1_200 -DEFAULT_KEYFRAME_MIN_INTERVAL_MS = 150 - - -@dataclass(frozen=True) -class JanusVideoConfig: - host: str = "127.0.0.1" - video_port: int = 5004 - payload_type: int = 96 - ssrc: int = 0x41565001 - bitrate_kbps: int = 4_500 - rtcp_bind: str = "0.0.0.0" - rtcp_port: int = 0 - keyframe_min_interval_ms: int = DEFAULT_KEYFRAME_MIN_INTERVAL_MS - - def __post_init__(self) -> None: - if not self.host: - raise ValueError("Janus host is required") - if not 1 <= self.video_port < 65535: - raise ValueError("Janus video port and its RTCP pair must be valid") - if not 0 <= self.payload_type <= 127: - raise ValueError("RTP payload type must be between 0 and 127") - if not 0 <= self.ssrc <= 0xFFFFFFFF: - raise ValueError("RTP SSRC must be a 32-bit unsigned integer") - if self.bitrate_kbps <= 0: - raise ValueError("Janus bitrate must be positive") - if not 0 <= self.rtcp_port <= 65535: - raise ValueError("RTCP port must be between 0 and 65535") - if (type(self.keyframe_min_interval_ms) is not int - or not 0 <= self.keyframe_min_interval_ms <= 2_147_483_647): - raise ValueError("keyframe_min_interval_ms must be a non-negative integer") - - @property - def rtcp_port_remote(self) -> int: - return self.video_port + 1 - - @property - def rtp_url(self) -> str: - return (f"rtp://{self.host}:{self.video_port}?pkt_size={RTP_PACKET_SIZE}" - f"&rtcp_port={self.rtcp_port_remote}") - - -JANUS_KEYFRAME_NODE = "janus_force_keyframe" -KEYFRAME_COMMAND = f"node.object.set {JANUS_KEYFRAME_NODE} trigger true" - - -def build_janus_output(avp, api, src_edge: str, janus: JanusVideoConfig, *, fps: int, - width: int, height: int, hwaccel: str = "@gpu", fps_den: int = 1, - group: str = "output", profile: str = "baseline", preset: str = "p7", - prefix: str = "janus"): - """Add ``force_fps -> keyframe -> nvenc -> bsf -> rtp mux -> output``; return the RTCP listener.""" - bitrate = f"{janus.bitrate_kbps}k" - avp.addNode(api.ForceFPS({ - "name": "janus_fps", "src": src_edge, "dst": "janus_fps", "fps": f"{fps}/{fps_den}", - "group": group, - })) - avp.addNode(api.ForceKeyFrame({ - "name": JANUS_KEYFRAME_NODE, "src": "janus_fps", "dst": "janus_keyframed", - "interval_sec": "1/1", "auto_restart": "panic", "group": group, - "min_interval_ms": janus.keyframe_min_interval_ms, - })) - avp.addNode(api.AssumeVideoFormat({ - "name": "janus_format", "src": "janus_keyframed", "dst": "janus_video", - "width": width, "height": height, "pixel_format": "cuda", "real_pixel_format": "nv12", - "auto_restart": "panic", "group": group, - })) - avp.addNode(api.EncVideo({ - "name": "janus_encoder", "src": "janus_video", "dst": "janus_encoded", - "codec": "h264_nvenc", "hwaccel": hwaccel, - "options": { - "b": bitrate, "maxrate": bitrate, "bufsize": bitrate, "g": fps, "bf": 0, - # p7 is NVENC's highest-quality preset; with tune=ull it stays a - # one-pass, no-lookahead, no-reordering encode, so the extra quality - # costs GPU time rather than latency. B-frames stay off: they need - # reordering, and WebRTC negotiates constrained baseline anyway. - "preset": preset, "profile": profile, "tune": "ull", "rc": "cbr", - "rc-lookahead": 0, "zerolatency": 1, "delay": 0, "forced-idr": 1, - "no-scenecut": 1, "strict_gop": 1, "aud": 1, "spatial-aq": 1, "temporal-aq": 0, - }, - "auto_restart": "panic", "group": group, - })) - avp.addNode(api.Bsf({ - "name": "janus_repeat_headers", "src": "janus_encoded", "dst": "janus_repeat_headers", - "bsf": "dump_extra=freq=keyframe", "auto_restart": "panic", "group": group, - })) - avp.addNode(api.Mux({ - "name": "janus_mux", "src": ["janus_repeat_headers"], "dst": "janus_video_rtp_mux", - "ts_sort_wait": 0, "auto_restart": "on", "on_error": "panic", "group": group, - })) - avp.addNode(api.Output({ - "name": "janus_rtp_output", "src": "janus_video_rtp_mux", "url": janus.rtp_url, - "format": "rtp", - "options": {"payload_type": janus.payload_type, "rtpflags": "skip_rtcp", "ssrc": janus.ssrc}, - "auto_restart": "on", "on_error": "panic", "group": group, - })) - return api.RtcpFeedbackListener( - bind_host=janus.rtcp_bind, bind_port=janus.rtcp_port, janus_host=janus.host, - janus_rtcp_port=janus.rtcp_port_remote, media_ssrc=janus.ssrc, - on_keyframe_request=lambda _request: avp.executeCommandsFromString(KEYFRAME_COMMAND), - ) diff --git a/demos/dmabuf-browser/compose.mixer.yaml b/demos/dmabuf-browser/compose.mixer.yaml index a5a2762c..9c29b05a 100644 --- a/demos/dmabuf-browser/compose.mixer.yaml +++ b/demos/dmabuf-browser/compose.mixer.yaml @@ -26,7 +26,7 @@ services: MIXER_WIPE_FILE: ${MIXER_WIPE_FILE:-} # alpha clip inside the container to warm the wipe up with volumes: - dma-browser-sockets:/tmp/dma-page - - ../../avpmixer:/opt/avplumber/avpmixer:ro,z + - ../../pyplumber/mixer:/opt/avplumber/pyplumber/mixer:ro,z - ../mixer:/opt/avplumber/demos/mixer:ro,z command: - sh diff --git a/demos/dmabuf-browser/consumer/Dockerfile.cuda b/demos/dmabuf-browser/consumer/Dockerfile.cuda index 0d0036ae..e961e560 100644 --- a/demos/dmabuf-browser/consumer/Dockerfile.cuda +++ b/demos/dmabuf-browser/consumer/Dockerfile.cuda @@ -17,7 +17,7 @@ FROM nvidia/cuda:${CUDA_IMAGE_VERSION}-devel-ubuntu${UBUNTU_VERSION} AS builder ARG DEBIAN_FRONTEND=noninteractive ARG FFMPEG_TAG=n8.1 -ARG NV_CODEC_HEADERS_TAG=n12.1.14.0 +ARG NV_CODEC_HEADERS_TAG=n13.0.19.0 # SDK 13: HEVC/AV1 HDR10 mastering-display and MaxCLL SEIs (driver >= 570) ARG CUDA_NVCC_FLAGS="-gencode arch=compute_70,code=compute_70 -O2" RUN apt-get update \ diff --git a/demos/mixer/Dockerfile b/demos/mixer/Dockerfile index 69bf8341..23e95dc7 100644 --- a/demos/mixer/Dockerfile +++ b/demos/mixer/Dockerfile @@ -3,7 +3,7 @@ FROM nvidia/cuda:11.7.1-devel-ubuntu22.04 ARG DEBIAN_FRONTEND=noninteractive ARG AVPLUMBER_REVISION=workspace ARG FFMPEG_TAG=n8.1 -ARG NV_CODEC_HEADERS_TAG=n12.1.14.0 +ARG NV_CODEC_HEADERS_TAG=n13.0.19.0 RUN apt-get update \ && apt-get install -y --no-install-recommends \ @@ -86,7 +86,6 @@ COPY deps /build/deps COPY Makefile generate_node_list /build/ COPY src /build/src COPY pyplumber /build/pyplumber -COPY avpmixer /build/avpmixer RUN git init --quiet /build \ && git -C /build config user.name "mixer-demo builder" \ diff --git a/demos/mixer/README.md b/demos/mixer/README.md index 21eba91b..1762e6b2 100644 --- a/demos/mixer/README.md +++ b/demos/mixer/README.md @@ -147,6 +147,36 @@ Sixteen generated clips with visible frame IDs (needs NumPy and FFmpeg): python3 demos/mixer/tests/frame_codes.py media --sources 16 --width 1920 --height 1080 --fps 60 --seconds 30 ``` +Eight colorful SDR patterns (bars, mandelbrot, life, ...) as NVENC clips: + +```sh +python3 demos/mixer/tests/sdr_patterns.py media/patterns --fps 60 --seconds 20 +``` + +## An HDR show + +`make_config.py` also writes HDR shows: an HLG (or PQ) 10-bit canvas, sources +tagged with their color, HEVC Main10 on the program mountpoint and a +tone-mapped H.264 rendition for SDR viewers on a second one. A 16-source +1080p60 example, mixing an NVDEC HDR movie, HLG patterns, an SDR clip, browser +pages and the patterns above: + +```sh +python3 demos/mixer/make_config.py --canvas 1920x1080 --fps 60 --color hlg --working-format p210le \ + --sdr-port 5004 --bitrate-kbps 8000 --preset p5 \ + hdr_movie=/media/hdr-movie.mp4 hlg_clip=/media/hlg-clip.mp4:hlg \ + hdr_pattern0=/fixtures/hlg0.v210@1920x1080:hlg hdr_pattern1=/fixtures/hlg1.v210@1920x1080:hlg \ + bunny=/media/bunny.mp4:sdr \ + web_a=https://example.org/a@1920x1080 web_b=https://example.org/b@1920x1080 web_c=https://example.org/c@1920x1080 \ + $(for p in testsrc2 bars rgbtest mandelbrot gradients life sierpinski cellauto; do echo sdr_$p=/media/patterns/$p.mp4:sdr; done) \ + > hdr16.json +``` + +Untagged clips (`hdr_movie` here) are typed by their decoded frames, so a PQ +movie lands on the HLG canvas through `tonemap_cuda`; the `grid_16_page_0` +scene shows all sixteen. Browser sources need the `dma-page` service and the +DMA-BUF container flags from [docs/guide.md](docs/guide.md#docker). + ## Under the hood Janus output limits forced keyframes to one per 150 ms by default (9 frames at @@ -159,10 +189,13 @@ next eligible frame. Periodic keyframes share the same limit. Two compositor slots draw every scene; a transition filter blends them and the wipe is one more layer in the same kernel, so nothing round-trips through the -CPU. Browser frames are converted from RGB to the NV12 canvas inside the draw -pass. The program is composited once and each rendition re-times and rescales +CPU. Browser frames are converted from RGB to the NV12, P010 or P210 canvas inside the +draw pass. The program is composited once and each rendition re-times and rescales it, so a second output costs an encode, not another composite. +Limits: 32 sources per show, no runtime source changes, 16 boxes in the built-in layouts; +see [docs/config.md](docs/config.md#known-limitations). + Output files, Janus settings, layouts and tests: [docs/guide.md](docs/guide.md). Measured samples and conditions: [runtime-load-1080p.json](docs/runtime-load-1080p.json), [latency.md](docs/latency.md). diff --git a/demos/mixer/config.example.hdr.json b/demos/mixer/config.example.hdr.json new file mode 100644 index 00000000..d0ad7b7c --- /dev/null +++ b/demos/mixer/config.example.hdr.json @@ -0,0 +1,46 @@ +{ + "canvas": {"width": 1920, "height": 1080, "fps": 60, "working_format": "p210le", "color": "hlg"}, + "sources": [ + {"id": "gen0", "kind": "v210", "path": "/fixtures/hlg0_1920x1080.v210", "width": 1920, "height": 1080, "color": "hlg"}, + {"id": "cam", "kind": "video", "path": "/media/camera_sdr.mp4", "width": 1920, "height": 1080}, + {"id": "page", "kind": "browser", "color": "sdr", "url": "https:///lower-third", "width": 1920, "height": 1080} + ], + "scenes": [ + { + "id": "gen_full", + "items": [ + { + "source": "gen0", + "dst": {"x": 0, "y": 0, "w": 1920, "h": 1080}, + "fit": "cover" + } + ] + }, + { + "id": "cam_with_graphics", + "items": [ + { + "source": "cam", + "dst": {"x": 0, "y": 0, "w": 1920, "h": 1080}, + "fit": "cover" + }, + { + "source": "gen0", + "dst": {"x": 1180, "y": 60, "w": 680, "h": 382}, + "fit": "cover" + }, + { + "source": "page", + "dst": {"x": 0, "y": 0, "w": 1920, "h": 1080} + } + ] + } + ], + "initial_scene": "cam_with_graphics", + "control": {"direct": true, "transition": "fade", "fade_seconds": 0.5}, + "renditions": [ + {"id": "hdr", "target": "janus", "port": 5006, "codec": "hevc_nvenc", "bitrate_kbps": 6000}, + {"id": "sdr", "target": "janus", "port": 5004, "codec": "h264_nvenc", "bitrate_kbps": 4500, "tonemap": "mobius", "tonemap_param": 0.9}, + {"id": "archive", "target": "/recordings/program_pq.ts", "codec": "hevc_nvenc", "color": "pq", "tonemap_peak": 10, "max_fall": 400, "bitrate_kbps": 12000} + ] +} diff --git a/demos/mixer/config.example.json b/demos/mixer/config.example.json index ba322e31..2f3581e9 100644 --- a/demos/mixer/config.example.json +++ b/demos/mixer/config.example.json @@ -21,6 +21,7 @@ { "id": "page0", "kind": "browser", + "color": "sdr", "url": "https:///page0", "width": 1920, "height": 1080 @@ -28,6 +29,7 @@ { "id": "page1", "kind": "browser", + "color": "sdr", "url": "https:///page1", "width": 1920, "height": 1080 @@ -35,6 +37,7 @@ { "id": "page2", "kind": "browser", + "color": "sdr", "url": "https:///page2", "width": 1920, "height": 1080 @@ -42,6 +45,7 @@ { "id": "page3", "kind": "browser", + "color": "sdr", "url": "https:///page3", "width": 1920, "height": 1080 @@ -49,6 +53,7 @@ { "id": "page4", "kind": "browser", + "color": "sdr", "url": "https:///page4", "width": 1920, "height": 1080 @@ -56,6 +61,7 @@ { "id": "page5", "kind": "browser", + "color": "sdr", "url": "https:///page5", "width": 1920, "height": 1080 diff --git a/demos/mixer/docs/config.md b/demos/mixer/docs/config.md index e803b667..99692ae0 100644 --- a/demos/mixer/docs/config.md +++ b/demos/mixer/docs/config.md @@ -35,12 +35,62 @@ supplied by that file; the runtime does not depend on the recorded demo's inputs `canvas`, `sources` and `scenes` are required; everything else has a default. +## Complete example + +Every meaningful option in one HDR show: three source kinds, two outputs, a wipe +library and a scene with each fit mode. Paths and URLs are placeholders. + +```json +{ + "canvas": {"width": 1920, "height": 1080, "fps": 60, + "working_format": "p210le", "color": "hlg", "latency_ms": 50}, + "sources": [ + {"id": "movie", "kind": "video", "path": "/media/hdr-movie.mp4", "loop": true}, + {"id": "cam", "kind": "video", "path": "srt://10.0.0.5:9000?mode=caller", "loop": false, + "color": "sdr", "width": 1920, "height": 1080}, + {"id": "graded", "kind": "video", "path": "/media/graded.mp4", + "filter": "tonemap_cuda=transfer_in=pq:transfer_out=hlg,scale_cuda=format=p210le", + "filter_output_format": "p210le"}, + {"id": "bars", "kind": "v210", "path": "/media/bars.v210", "width": 1920, "height": 1080, + "color_trc": "arib-std-b67", "color_primaries": "bt2020", "colorspace": "bt2020nc", "color_range": "tv"}, + {"id": "page", "kind": "browser", "url": "https://example.org/lower-third", + "width": 1920, "height": 1080, "fps": 60, "color": "sdr"} + ], + "wipes": [{"id": "ribbons", "path": "/media/wipes/ribbons.mov", "name": "Ribbons", "duration_seconds": 2.0}], + "wipe_dir": "/media/wipes", + "wipe_color": "sdr", + "control": {"direct": true, "transition": "wipe", "fade_seconds": 0.5, "default_wipe": "ribbons"}, + "renditions": [ + {"id": "program", "target": "janus", "port": 5006, "width": 1920, "height": 1080, "aspect": "16:9", + "fps": 60, "bitrate_kbps": 8000, "codec": "hevc_nvenc", "profile": "main10", "preset": "p5"}, + {"id": "sdr", "target": "janus", "port": 5004, "fps": 60, "bitrate_kbps": 6000, + "codec": "h264_nvenc", "profile": "baseline", "preset": "p5", + "tonemap": "mobius", "tonemap_param": 0.9, "tonemap_peak": 10, "tonemap_desat": 0}, + {"id": "archive", "target": "/recordings/program.ts", "fps": 30, "bitrate_kbps": 12000, + "codec": "hevc_nvenc", "profile": "main10", "color": "pq", "max_cll": 1000, "max_fall": 400} + ], + "scenes": [ + {"id": "pip", "items": [ + {"source": "movie", "dst": {"x": 0, "y": 0, "w": 1920, "h": 1080}, "fit": "cover"}, + {"source": "cam", "dst": {"x": 1180, "y": 620, "w": 680, "h": 400}, "fit": "contain", + "crop": {"x": 240, "y": 0, "w": 1440, "h": 1080}}, + {"source": "page", "dst": {"x": 0, "y": 0, "w": 1920, "h": 1080}, "fit": "stretch"} + ]}, + {"id": "bars_full", "items": [{"source": "bars", "dst": {"x": 0, "y": 0, "w": 1920, "h": 1080}}]} + ], + "initial_scene": "pip" +} +``` + ## canvas | field | default | meaning | | --- | --- | --- | | `width`, `height` | — | the program raster the compositor draws into | | `fps` | `30` | **how often the compositor renders**, and the clock the whole mixer runs on: inputs are re-timed to it and browser pages are asked to paint at it | +| `working_format` | `nv12` | compositor and transition pixel storage: `nv12` (8-bit 4:2:0), `p010le` (10-bit 4:2:0) or `p210le` (10-bit 4:2:2). 8-bit sources are promoted onto a 10-bit canvas; `p210le` keeps 4:2:2 through conversion and compositing. Renditions are 4:2:0 for NVENC, subsampled once | +| `color` | `sdr` | canvas color contract: `sdr` (BT.709), `hlg` or `pq` (BT.2020). HLG/PQ need a 10-bit `working_format`. Every source is converted to it on the GPU; renditions convert from it | +| `latency_ms` | two frames | playout buffer between a source frame's arrival and its tick (33 ms at 60 fps). A frame later than that is skipped and the previous one repeated, so set it just above the worst source jitter; must stay below six frames. `--mixer-latency-ms` on the command line overrides it | The canvas rate is the single biggest load knob. Halving it from 60 to 30 on the sixteen-source demo took the T4 from ~33% to ~17% GPU. @@ -59,10 +109,16 @@ output costs an encode, not another composite. | `aspect` | — | optional, e.g. `"9:16"`; checked against `width:height` | | `fps` | canvas fps | may only re-time **downwards**; a higher rate is rejected | | `bitrate_kbps` | `3000` | CBR target, also the `maxrate` and `bufsize` | -| `codec` | `"h264_nvenc"` | file targets only; the Janus target is always H.264 | -| `profile` | `"baseline"` | WebRTC negotiates constrained baseline; B-frames stay off | +| `codec` | from canvas depth | `h264_nvenc` for 8-bit or `hevc_nvenc` for 10-bit by default; applies to Janus and file targets | +| `profile` | from codec/depth | HEVC `main`/`main10`; H.264 `baseline` for Janus, `high` for files. No B-frames | | `preset` | `"p7"` | NVENC quality preset | | `port` | — | Janus target: overrides the RTP port from the command line | +| `color` | automatic | `sdr`, `hlg`, or `pq`; H.264 always requires SDR, HEVC otherwise inherits the canvas | +| `tonemap` | none | requests an SDR output from an HDR canvas: `clip` (exact SDR, hard-clipped highlights), `mobius` (see `tonemap_param`), `hable`, `reinhard`, `gamma`, `linear`, `none` | +| `tonemap_peak` | `10` | HDR peak in units of 100 nits, minimum `2.03` | +| `tonemap_desat` | `0` | highlight desaturation; `0` keeps saturation | +| `max_cll`, `max_fall` | derived | HDR10 static metadata for **PQ** outputs, nits. Defaults: MaxCLL = `tonemap_peak`×100, MaxFALL = 40% of it; `max_fall` may not exceed MaxCLL. Needs nv-codec-headers 13 and driver ≥ 570 (`Dockerfile.cuda`), else the SEIs are silently absent. HLG needs none | +| `tonemap_param` | `0` | operator knee in reference-white units; `0` = operator default (0.3 mobius/reinhard, 1.8 gamma). mobius must be below 1.0; `0.9` keeps 90% of SDR white untouched | With no `renditions` the demo builds its usual single output from the command line flags. @@ -75,6 +131,27 @@ line flags. ] ``` +Multiple Janus renditions need separate RTP ports and matching Janus +mountpoints. Each has its own encoder and RTCP feedback. The first keeps the +`janus_encoder` name used by cut-latency measurement; additional outputs use +`janus__encoder`. + +For a P010 HLG canvas, these outputs provide simultaneous HDR and SDR previews: + +```json +"renditions": [ + {"id": "hdr", "target": "janus", "port": 5006, + "codec": "hevc_nvenc", "profile": "main10"}, + {"id": "sdr", "target": "janus", "port": 5004, + "codec": "h264_nvenc", "profile": "baseline", "tonemap": "clip"} +] +``` + +`clip` preserves SDR content embedded into HLG with the same 203-nit white; +HDR highlights above that white are clipped. Operators such as `hable` compress +highlights and also change midtone brightness. A preview selector chooses +between the two continuously running outputs; the canvas remains HDR. + ## sources One entry per **unique** clip or page: two entries with the same `url` or @@ -86,22 +163,63 @@ decoder. ```json {"id": "cam0", "kind": "video", "path": "/media/camera-0.mp4", "loop": true} {"id": "page", "kind": "browser", "url": "https://example.org/live", - "width": 1920, "height": 1080, "fps": 30} + "width": 1920, "height": 1080, "fps": 30, "color": "sdr"} ``` | field | applies to | meaning | | --- | --- | --- | | `id` | both | referenced from scenes; no `#` | -| `kind` | both | `"video"` or `"browser"` | -| `path` | video | file or stream | +| `kind` | all | `"video"`, `"browser"` or `"v210"` (headerless packed 10-bit 4:2:2, e.g. generated HDR test content) | +| `path` | video, v210 | file or stream | +| `width`, `height` | v210 | required: the packed bytes carry no header | +| `color` | browser, v210 | required color contract (`sdr`, `hlg`, `pq`); browser pages must be `sdr`. Optional for `video`: by default the decoded frame tags decide and untagged files are treated as BT.709 SDR | | `url`, `width`, `height` | browser | page and the window it is rendered in (all three required) | | `fps` | browser | paint rate; defaults to the canvas rate | -| `width`, `height` | video | optional; probed with ffprobe/ffmpeg when a `cover` item needs them | +| `width`, `height` | video | optional; probed with ffprobe at load when absent | | `loop` | both | default `true` | +| `filter` | video/v210 | optional CUDA source graph, before automatic normalization; preserve dimensions and correct output metadata | +| `filter_output_format` | custom filters | required CUDA YUV output storage, e.g. `p010le` or `p210le` | Browser sources arrive over DMA-BUF from the `dma-page` service and are imported straight into CUDA; they mix with video sources on the same canvas. +Color normalization is automatic in `MixerGraphBuilder`, before alias and scene +fan-out. Video sources use decoded frame metadata, including live SRT streams. +A `video` source without frame color tags is treated as BT.709 SDR. A declared +contract is either the `color` preset (which supplies all four fields) or a complete, +consistent set of `color_trc`, `color_primaries`, `colorspace` and `color_range`. +Raw `v210` and browser inputs must declare one. + +For SDR Bunny on an HLG canvas, no manual tone-map filter is needed: + +```json +"canvas": {"width": 1920, "height": 1080, "fps": 60, + "working_format": "p010le", "color": "hlg"}, +"sources": [{"id": "bunny", "kind": "video", "path": "", "color": "sdr"}] +``` + +Supported video contracts are limited-range BT.709/BT.1886 SDR and BT.2020 +non-constant-luminance HLG/PQ. Full-range YUV, other gamuts/matrices and missing +metadata fail explicitly. SDR uses a 203-nit reference white and HLG a 1000-nit +peak. Frames already in the canvas storage (NV12, NV16, P010, P210) pass without GPU copies. Other declared CUDA YUV +storage uses `scale_cuda` around conversion. Custom source filters receive the +explicit source override first; their output metadata drives normalization, so +a manually converted source is not converted from its original transfer again. + +Packed SDR RGB graphics retain alpha and convert in the compositor to its target +transfer/gamut. This path requires NV12, P010 or P210 canvas storage. HDR alpha +sources are unsupported. Missing transfer and primaries tags on RGB(A) wipes +default to SDR BT.709; explicit tags are preserved and validated. Set top-level +`wipe_color: "sdr"` to override the wipe library's color tags. The CLI option +`--wipe-color sdr` provides the same override and takes precedence over the JSON +setting. Neither the fallback nor the override changes alpha. + +H.264 renditions and the default output path automatically convert HDR to SDR. +HEVC renditions inherit the canvas unless `color` requests another supported +contract. HDR outputs require 10-bit storage. The `clip` default preserves the +brightness of SDR embedded in HDR; it clips highlights above SDR white. Select +another operator explicitly when highlight compression is preferred. + ## wipes The media-wipe library. Every declared clip is decoded once at start and held @@ -164,6 +282,25 @@ Padding is always black. `cover` needs the source size, which is declared or probed. `initial_scene` names the scene on program at start (default: the first one). +## Known limitations + +- **32 sources per show.** Every source is a pad on the compositor, and the + active-pad mask is a 32-bit word. 8-bit or 10-bit makes no difference; + scenes and aliases are free. Sources cannot be added while running: the + pads are wired at build time. A document with more sources is rejected at load. +- **16 boxes per scene** in the built-in `--input` layouts; a `--config` scene has no box limit. +- **All sources run all the time.** A source not on screen is still decoded + and converted. Budget GPU for the whole catalogue, not the scene. +- **`--input` with more than 32 files** switches to a router that converts + only the 16 on-screen sources. It needs all inputs in one size, format and + frame rate, and it does not take `v210` or browser sources. +- **NVENC encodes 4:2:0.** A `p210le` canvas keeps 4:2:2 through compositing; + every rendition is 4:2:0. +- **HLG and PQ need a 10-bit canvas.** Browser pages are SDR only. + +Lifting the 32-source and runtime-add limits means routing every show +through the router with per-slot conversion; that is planned, not done. + ## Generating one `make_config.py` writes the demo's own fullscreen and 2/4/8/16-box layouts out @@ -178,3 +315,9 @@ python3 demos/mixer/make_config.py --fps 30 --bitrate-kbps 2700 \ Repeated locations collapse to one source, and grid slots take every distinct source before any repeat — a 16-box of sixteen unique sources shows all sixteen rather than one of them three times. + +`--color hlg|pq` with `--working-format p010le|p210le` emits an HDR canvas, a +HEVC Main10 program rendition and, with `--sdr-port`, a tone-mapped H.264 +rendition (`--sdr-tonemap`, `--sdr-knee`). Source arguments take a color +suffix (`clip=/m/c.mp4:hlg`) and raw v210 a size (`bars=/m/b.v210@1920x1080:hlg`); +see the README for a 16-source HDR example. diff --git a/demos/mixer/docs/guide.md b/demos/mixer/docs/guide.md index ad89fa60..de08faf6 100644 --- a/demos/mixer/docs/guide.md +++ b/demos/mixer/docs/guide.md @@ -147,7 +147,7 @@ connection fails or is lost, it shows the error in the connection bar; use ## Janus output -`--janus-output` publishes the video-only Program as H.264 RTP. It may be used +`--janus-output` publishes the video-only Program over RTP: H.264 from an 8-bit canvas, HEVC Main10 from a 10-bit one. It may be used alone or together with `--output`: ```sh diff --git a/demos/mixer/make_config.py b/demos/mixer/make_config.py index 6b0c0e9d..def0c853 100644 --- a/demos/mixer/make_config.py +++ b/demos/mixer/make_config.py @@ -5,8 +5,13 @@ cam=/media/cam.mp4 page=https://example.org/page@1920x1080 > mixer.json Each positional argument is ``id=path`` for a clip or ``id=url@WxH`` for a -browser page. The scenes are the same fullscreen and 2/4/8/16-box pages the -demo builds without a config, written out as plain data. +browser page. A ``:sdr``, ``:hlg`` or ``:pq`` suffix declares the clip's color +(``id=/m/clip.mp4:hlg``); raw v210 takes its size too (``id=/m/bars.v210@1920x1080:hlg``). +``--color hlg`` (or ``pq``) with ``--working-format p210le`` makes an HDR show: +the program rendition becomes HEVC Main10 and ``--sdr-port`` adds a tone-mapped +H.264 rendition, so one show feeds an HDR and an SDR mountpoint at once. +The scenes are the same fullscreen and 2/4/8/16-box pages the demo builds +without a config, written out as plain data. """ from __future__ import annotations @@ -19,18 +24,35 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) sys.path.insert(0, str(Path(__file__).resolve().parents[2])) import layouts # noqa: E402 -from avpmixer import config as mixer_config # noqa: E402 +from pyplumber.mixer import config as mixer_config # noqa: E402 + + +COLORS = ("sdr", "hlg", "pq") def source_spec(arg: str) -> dict: sid, _, rest = arg.partition("=") if not sid or not rest: - raise SystemExit(f"expected id=path or id=url@WxH, got {arg!r}") + raise SystemExit(f"expected id=path[:color] or id=url@WxH, got {arg!r}") if "://" in rest and "@" in rest.rsplit("/", 1)[-1]: url, size = rest.rsplit("@", 1) w, h = (int(v) for v in size.lower().split("x")) - return {"id": sid, "kind": "browser", "url": url, "width": w, "height": h} - return {"id": sid, "kind": "video", "path": rest} + return {"id": sid, "kind": "browser", "url": url, "width": w, "height": h, "color": "sdr"} + color = None + if rest.rsplit(":", 1)[-1] in COLORS: + rest, color = rest.rsplit(":", 1) + spec = {"id": sid, "kind": "video", "path": rest} + if "@" in rest.rsplit("/", 1)[-1]: + path, size = rest.rsplit("@", 1) + if not path.lower().endswith(".v210"): + raise SystemExit(f"only .v210 clips take a size, got {arg!r}") + if color is None: + raise SystemExit(f"v210 clips need a :sdr/:hlg/:pq color, got {arg!r}") + w, h = (int(v) for v in size.lower().split("x")) + spec = {"id": sid, "kind": "v210", "path": path, "width": w, "height": h} + if color: + spec["color"] = color + return spec def main(argv=None) -> None: @@ -50,7 +72,16 @@ def main(argv=None) -> None: help="program rendition rate; 0 keeps the canvas rate") parser.add_argument("--profile", default="baseline", help="NVENC profile of the program rendition") parser.add_argument("--preset", default="p7", help="NVENC preset of the program rendition") + parser.add_argument("--color", choices=COLORS, help="canvas color; hlg/pq need a 10-bit --working-format") + parser.add_argument("--working-format", choices=mixer_config.WORKING_FORMATS, help="canvas storage") + parser.add_argument("--sdr-port", type=int, metavar="PORT", + help="add a tone-mapped H.264 rendition on this RTP port (HDR canvas only)") + parser.add_argument("--sdr-tonemap", default="mobius", choices=mixer_config.OPERATORS) + parser.add_argument("--sdr-knee", type=float, default=0.9, help="tonemap_param of the SDR rendition") args = parser.parse_args(argv) + hdr = args.color in ("hlg", "pq") + if args.sdr_port and not hdr: + raise SystemExit("--sdr-port needs an HDR canvas (--color hlg or pq)") width, height = (int(v) for v in args.canvas.lower().split("x")) # One source per unique clip or page: repeated locations become references # to the first declaration, so a page shown eight times is one window. @@ -77,13 +108,21 @@ def main(argv=None) -> None: for p in scene.placements] scenes.append({"id": scene.name, "items": items}) rendition_fps = args.rendition_fps or args.fps + program = {"id": "program", "target": "janus", "width": width, "height": height, + "aspect": mixer_config.Rendition("program", width=width, height=height).aspect, + "fps": rendition_fps, "bitrate_kbps": args.bitrate_kbps, + "profile": "main10" if hdr else args.profile, "preset": args.preset, + **({"codec": "hevc_nvenc"} if hdr else {})} + renditions = [program] + if args.sdr_port: + renditions.append({**program, "id": "sdr", "port": args.sdr_port, "codec": "h264_nvenc", + "profile": args.profile, "tonemap": args.sdr_tonemap, "tonemap_param": args.sdr_knee}) + canvas = {"width": width, "height": height, "fps": args.fps, + **({"working_format": args.working_format} if args.working_format else {}), + **({"color": args.color} if args.color else {})} doc = { - "canvas": {"width": width, "height": height, "fps": args.fps}, - "renditions": [{"id": "program", "target": "janus", - "width": width, "height": height, - "aspect": mixer_config.Rendition("program", width=width, height=height).aspect, - "fps": rendition_fps, "bitrate_kbps": args.bitrate_kbps, - "profile": args.profile, "preset": args.preset}], + "canvas": canvas, + "renditions": renditions, "sources": sources, "wipes": [{"id": Path(p).stem, "path": p} for p in args.wipe], "control": {"direct": True, "fade_seconds": args.fade_seconds, diff --git a/demos/mixer/mixer.py b/demos/mixer/mixer.py index b566a9e6..bf0dc7e5 100644 --- a/demos/mixer/mixer.py +++ b/demos/mixer/mixer.py @@ -13,13 +13,14 @@ from dataclasses import dataclass, replace from types import SimpleNamespace -from avpmixer import clipcache -from avpmixer import config as mixer_config -from avpmixer.dmabuf_inputs import (dmabuf_cuda_input_nodes, is_dmabuf_url, open_browser_windows, +from pyplumber.mixer.color import TEN_BIT_FORMATS, TRANSFER_TAGS, conversion_graph, hdr_metadata, rendition_color +from pyplumber.mixer import clipcache +from pyplumber.mixer import config as mixer_config +from pyplumber.mixer.dmabuf_inputs import (dmabuf_cuda_input_nodes, is_dmabuf_url, open_browser_windows, open_windows, refresh_windows, wait_for_sockets, window_id) -from avpmixer.inputs import build_input -from avpmixer.janus import (DEFAULT_KEYFRAME_MIN_INTERVAL_MS, JANUS_KEYFRAME_NODE, - JanusVideoConfig, build_janus_output) +from pyplumber.mixer.inputs import build_input, build_v210_input +from pyplumber.mixer.janus import (DEFAULT_KEYFRAME_MIN_INTERVAL_MS, JANUS_KEYFRAME_NODE, + JanusVideoConfig, RtcpFeedbackGroup, add_nodes, build_janus_output) try: from .layouts import ( @@ -60,10 +61,12 @@ class GraphOptions: output_format: str | None = None remote_control_port: int = 7777 codec: str = "h264_nvenc" + working_format: str = "nv12" # compositor/transition sw_format; p210le keeps 10-bit 4:2:2 bitrate: str = "8M" fps: int = DEFAULT_FPS mixer_latency_ms: float | None = None loop_inputs: bool = False + input_color: str = "" # declared contract for every --input (sdr/hlg/pq); "" = frame tags janus_output: bool = False janus_host: str = JANUS_DEFAULT_HOST janus_video_port: int = JANUS_DEFAULT_VIDEO_PORT @@ -75,6 +78,7 @@ class GraphOptions: janus_rtcp_port: int = 0 preheat_timeout_sec: float = 60.0 wipe_file: str | None = None # warm the media wipe chain up with this clip at start + wipe_color: str = "" # explicit SDR override; empty preserves tags with an SDR fallback config: str | None = None # JSON document (sources, wipes, scenes) instead of --input webui_url: str = "" # AVPlumber web UI to register the graph with cut_latency_encoder: str = "" # opt-in cut-to-output observer on this encoder @@ -99,6 +103,12 @@ def validate(self) -> None: raise ValueError("at least one input is required") if not self.output and not self.janus_output: raise ValueError("--output or --janus-output is required") + if self.working_format not in mixer_config.WORKING_FORMATS: + raise ValueError(f"--working-format must be one of {mixer_config.WORKING_FORMATS}") + if self.input_color and self.input_color not in TRANSFER_TAGS: + raise ValueError("--input-color must be sdr, hlg or pq") + if self.wipe_color not in ("", "sdr"): + raise ValueError("--wipe-color supports sdr only; HDR alpha wipes are unsupported") if not self.codec.endswith("_nvenc"): raise ValueError("--codec must be an NVENC encoder for zero-copy output") if not 1 <= self.fps <= 240: @@ -148,7 +158,7 @@ class MixerApplication: prewarm_cut_scenes: tuple[str, ...] = () def _preload_wipes(self) -> None: - """Decode every wipe once into GPU memory (see avpmixer.clipcache). + """Decode every wipe once into GPU memory (see pyplumber.mixer.clipcache). The loader group is started only here; a take starts the player group alone and replays what this left behind. @@ -264,7 +274,7 @@ def stop(self) -> None: def load_avp_api(): from pyplumber import AVPlumber - from avpmixer import MixerGraphBuilder + from pyplumber.mixer import MixerGraphBuilder from pyplumber.node import ( AssumeVideoFormat, Bsf, @@ -284,6 +294,7 @@ def load_avp_api(): Realtime, RepeatLastFrame, Split, + V210ToCuda, ) from pyplumber.rtcp_feedback import RtcpFeedbackListener @@ -309,6 +320,7 @@ def load_avp_api(): RepeatLastFrame=RepeatLastFrame, RtcpFeedbackListener=RtcpFeedbackListener, Split=Split, + V210ToCuda=V210ToCuda, ) @@ -331,6 +343,29 @@ def _input_group(index: int) -> str: return f"input_{index}" +def _init_avp(avp_options, api): + """AVPlumber instance + control server + CUDA hwaccel — shared by both the + --input and --config build paths.""" + avp = api.AVPlumber() + if avp_options.remote_control_port: + avp.enableControlServer(avp_options.remote_control_port) + avp.executeCommandsFromString(f'hwaccel.init {{ "name": "{HWACCEL}", "type": "cuda" }}') + avp.edges.planCapacity("*", 4) + return avp + + +def _make_builder(avp, api, options, *, canvas, fps, working_format, color="sdr", wipe_color=None): + """The mixer builder, configured identically for both build paths (canvas, + rate and working_format are the only per-path differences).""" + return api.MixerGraphBuilder( + avp, name=MIXER_NAME, canvas=canvas, fps=(fps, FPS_DEN), + latency_ms=options.mixer_latency_ms, hwaccel=HWACCEL, enable_wipe=True, + defer_initial_routes=True, defer_output=True, + keyframe_node=JANUS_KEYFRAME_NODE if options.janus_output else None, + cache_wipes_mb=options.wipe_cache_mb or None, working_format=working_format, color=color, + wipe_color=options.wipe_color or wipe_color) + + def _build_input( avp, api, index: int, url: str, *, loop: bool, fps: int, normalize: bool, options: "GraphOptions | None" = None, @@ -374,13 +409,14 @@ def _build_input( return normalized_edge -def _register_sources(avp, api, mixer, input_edges: list[str], *, fps: int) -> bool: +def _register_sources(avp, api, mixer, input_edges: list[str], urls, *, fps: int, color: str = "") -> bool: # Compositor masks have 32 bits. Larger catalogues retain a small router # selecting the 16 visible positions, without per-layout filter branches. if len(input_edges) <= 32: - for index, edge in enumerate(input_edges): - mixer.add_source(f"source_{index}", pre_otm_edge=edge, - input_group=_input_group(index), default_graph="") + for index, (edge, url) in enumerate(zip(input_edges, urls)): + browser = is_dmabuf_url(url) # packed RGB, always SDR; decoded files follow --input-color + mixer.add_source(f"source_{index}", pre_otm_edge=edge, input_group=_input_group(index), + default_graph="", packed_rgb=browser, color="sdr" if browser else color or None) return False labels = [f"slot_{i}_{slot}" for i in range(16) for slot in ("a", "b")] edges = [f"route_{label}" for label in labels] @@ -397,7 +433,7 @@ def _register_sources(avp, api, mixer, input_edges: list[str], *, fps: int) -> b f"source_{index}", pre_filter_edge_a=edges[2 * index], pre_filter_edge_b=edges[2 * index + 1], input_group=ROUTER_GROUP, route_router="layout_preheat_router", route_output_label_a=labels[2 * index], - route_output_label_b=labels[2 * index + 1], default_graph="", + route_output_label_b=labels[2 * index + 1], default_graph="", color=color or None, ) return True @@ -421,98 +457,103 @@ def _define_scenes(mixer, input_count: int, routed: bool) -> None: mixer.add_scene(scene.name, sources, routes=routes) -def _build_record_output(avp, api, options: GraphOptions, mixer_edge: str, *, - width: int = CANVAS_WIDTH, height: int = CANVAS_HEIGHT) -> None: - if options.output is None: - raise ValueError("record output needs an output URL or path") - fps_edge = "program_fps" - assumed_edge = "program_video" - encoded_edge = "program_encoded" - muxed_edge = "program_muxed" - avp.addNode(api.ForceFPS({ - "name": "program_fps", - "src": mixer_edge, - "dst": fps_edge, - "fps": f"{options.fps}/{FPS_DEN}", - "group": OUTPUT_GROUP, - })) - avp.addNode(api.AssumeVideoFormat({ - "name": "program_format", - "src": fps_edge, - "dst": assumed_edge, - "width": width, - "height": height, - "pixel_format": "cuda", - "real_pixel_format": "nv12", - "group": OUTPUT_GROUP, - })) - avp.addNode(api.EncVideo({ - "name": "program_encoder", - "src": assumed_edge, - "dst": encoded_edge, - "codec": options.codec, - "hwaccel": HWACCEL, - "options": { - "b": options.bitrate, - "maxrate": options.bitrate, - "bufsize": options.bitrate, - "g": options.fps * 2, - "bf": 0, - "preset": "p3", - "tune": "ll", - "profile": "high", - }, - "group": OUTPUT_GROUP, - })) - avp.addNode(api.Mux({ - "name": "program_mux", - "src": [encoded_edge], - "dst": muxed_edge, - "ts_sort_wait": 0, - "group": OUTPUT_GROUP, - })) - avp.addNode(api.Output({ - "name": "program_output", - "src": muxed_edge, - "url": options.output, - "format": infer_output_format(options.output, options.output_format), - "auto_restart": "panic", - "group": OUTPUT_GROUP, - })) - +def _kbps(bitrate: str) -> int: + """``--bitrate`` in FFmpeg notation (``8M``, ``6000k`` or bit/s) as kbit/s.""" + scale = {"k": 1, "K": 1, "M": 1000}.get(bitrate[-1]) + return int(float(bitrate[:-1]) * scale) if scale else int(bitrate) // 1000 + + +def _flag_renditions(options: GraphOptions, width: int, height: int) -> tuple: + """``--output`` / ``--janus-output`` as renditions: a ``program`` record file + at the CLI codec and bitrate, and a ``janus`` stream whose codec follows depth.""" + record = mixer_config.Rendition("program", options.output or "", width, height, options.fps, + _kbps(options.bitrate), options.codec, preset="p3") + janus = mixer_config.Rendition("janus", "janus", width, height, options.fps, options.janus_video_bitrate_kbps) + return tuple(r for r, wanted in ((record, options.output), (janus, options.janus_output)) if wanted) + + +def _hdr_metadata(r: "mixer_config.Rendition", target) -> dict | None: + """HDR10 static metadata for a PQ output; nvenc emits the SEIs from it. HLG carries none.""" + return hdr_metadata(r.tonemap_peak * 100, max_cll=r.max_cll, max_fall=r.max_fall) if target.transfer == "pq" else None + + +def _build_record_output(avp, api, edge: str, r: "mixer_config.Rendition", *, codec: str, + enc_format: str, color: dict, output_format=None, hdr_metadata=None) -> None: + """``force_fps -> assume_format -> nvenc -> mux -> output``, nodes named ``_*``.""" + name = lambda suffix: f"{r.id}_{suffix}" # noqa: E731 + bitrate = f"{r.bitrate_kbps}k" + profile = r.profile or (("main10" if enc_format in TEN_BIT_FORMATS else "main") if "hevc" in codec else "high") + add_nodes(avp, api, [ + ("ForceFPS", {"name": name("fps"), "src": edge, "dst": name("fps"), "fps": f"{r.fps}/{FPS_DEN}"}), + ("AssumeVideoFormat", {"name": name("format"), "src": name("fps"), "dst": name("video"), + "width": r.width, "height": r.height, "pixel_format": "cuda", + "real_pixel_format": enc_format}), + ("EncVideo", {"name": name("encoder"), "src": name("video"), "dst": name("encoded"), + "codec": codec, "hwaccel": HWACCEL, + **({"hdr_metadata": hdr_metadata} if hdr_metadata else {}), + "options": {"b": bitrate, "maxrate": bitrate, "bufsize": bitrate, + "g": max(1, round(r.fps / FPS_DEN)) * 2, "bf": 0, "preset": r.preset, + "tune": "ll", "profile": profile, **color}}), + ("Mux", {"name": name("mux"), "src": [name("encoded")], "dst": name("muxed"), "ts_sort_wait": 0}), + ("Output", {"name": name("output"), "src": name("muxed"), "url": r.target, + "format": infer_output_format(r.target, output_format), "auto_restart": "panic"}), + ], group=OUTPUT_GROUP) + + +def _build_renditions(avp, api, options: GraphOptions, renditions, mixer_edge: str, *, + canvas, working_format: str, color="sdr"): + """One encoder per rendition, all fed from the single composited program. -def _build_outputs(avp, api, options: GraphOptions, mixer_edge: str, *, - width: int = CANVAS_WIDTH, height: int = CANVAS_HEIGHT): - record_edge = mixer_edge - janus_edge = mixer_edge - if options.output and options.janus_output: - record_edge = "program_video_record" - janus_edge = "program_video_janus" - avp.addNode(api.Split({ - "name": "split_program_video_output", - "src": mixer_edge, - "dst": [record_edge, janus_edge], - "group": OUTPUT_GROUP, - "on_error": "panic", + The compositor renders once at the canvas rate; a rendition converts, + re-times and rescales that picture for its own target, so extra renditions + cost an encode, not another composite. + """ + edges = [mixer_edge] + if len(renditions) > 1: + edges = [f"program_rendition_{r.id}" for r in renditions] + avp.addNode(api.Split({"name": "split_renditions", "src": mixer_edge, "dst": edges, + "group": OUTPUT_GROUP, "on_error": "panic"})) + listeners = [] + for r, edge in zip(renditions, edges): + codec = r.codec or ("hevc_nvenc" if working_format in TEN_BIT_FORMATS else "h264_nvenc") + target = rendition_color(color, codec, r.color or None, r.tonemap) + # 10-bit stays P010 for HEVC (Main10 carries depth and HDR); H.264 and 8-bit encode NV12. + ten_bit = target.transfer != "sdr" or (working_format in TEN_BIT_FORMATS and "hevc" in codec) + enc_format = "p010le" if ten_bit else "nv12" + scale = f"scale_cuda=w={r.width}:h={r.height}," if (r.width, r.height) != canvas else "" + scaled = f"program_scaled_{r.id}" + avp.addNode(api.FilterVideo({ + "name": f"scale_{r.id}", "src": edge, "dst": scaled, "hwaccel": HWACCEL, "group": OUTPUT_GROUP, + "graph": scale + conversion_graph(target, enc_format, source=color, source_format=working_format, + tonemap=r.tonemap or "clip", hdr_peak=r.tonemap_peak * 100, + desat=r.tonemap_desat, param=r.tonemap_param), })) - - if options.output: - _build_record_output(avp, api, options, record_edge, width=width, height=height) - if not options.janus_output: - return None - - return build_janus_output( - avp, api, janus_edge, - JanusVideoConfig( - host=options.janus_host, video_port=options.janus_video_port, - payload_type=options.janus_video_pt, ssrc=options.janus_video_ssrc, - bitrate_kbps=options.janus_video_bitrate_kbps, - keyframe_min_interval_ms=options.keyframe_min_interval_ms, - rtcp_bind=options.janus_rtcp_bind, rtcp_port=options.janus_rtcp_port, - ), - fps=options.fps, fps_den=FPS_DEN, width=width, height=height, - hwaccel=HWACCEL, group=OUTPUT_GROUP, - ) + if r.target != "janus": + _build_record_output(avp, api, scaled, r, codec=codec, enc_format=enc_format, + color=target.tags, output_format=options.output_format, + hdr_metadata=_hdr_metadata(r, target)) + continue + listeners.append(build_janus_output( + avp, api, scaled, + JanusVideoConfig( + host=options.janus_host, video_port=r.port or options.janus_video_port, + payload_type=options.janus_video_pt, ssrc=options.janus_video_ssrc, + bitrate_kbps=r.bitrate_kbps, keyframe_min_interval_ms=options.keyframe_min_interval_ms, + rtcp_bind=options.janus_rtcp_bind, rtcp_port=options.janus_rtcp_port if not listeners else 0, + ), + fps=r.fps, fps_den=FPS_DEN, width=r.width, height=r.height, hwaccel=HWACCEL, group=OUTPUT_GROUP, + codec=codec, profile=r.profile, preset=r.preset, enc_format=enc_format, color=target.tags, + hdr_metadata=_hdr_metadata(r, target), prefix="janus" if not listeners else f"janus_{r.id}")) + return RtcpFeedbackGroup(listeners) if len(listeners) > 1 else next(iter(listeners), None) + + +def _application(avp, mixer, options: GraphOptions, input_edges, listener, **extra) -> MixerApplication: + return MixerApplication( + avp=avp, mixer=mixer, input_edges=tuple(input_edges), + input_groups=tuple(_input_group(index) for index in range(len(input_edges))), + rtcp_feedback_listener=listener, preheat_timeout_sec=options.preheat_timeout_sec, + wipe_file=options.wipe_file, dmabuf_rest=options.dmabuf_rest, wipe_cache_mb=options.wipe_cache_mb, + cut_latency_encoder=options.cut_latency_encoder, prewarm_cut_scenes=options.prewarm_cut_scenes, **extra) def build_application(options: GraphOptions, api=None) -> MixerApplication: @@ -520,12 +561,7 @@ def build_application(options: GraphOptions, api=None) -> MixerApplication: api = api or load_avp_api() if options.config: return _build_from_config(options, mixer_config.with_probed_sizes(mixer_config.load(options.config)), api) - avp = api.AVPlumber() - if options.remote_control_port: - avp.enableControlServer(options.remote_control_port) - avp.executeCommandsFromString( - f'hwaccel.init {{ "name": "{HWACCEL}", "type": "cuda" }}' - ) + avp = _init_avp(options, api) dmabuf_ids = options.dmabuf_inputs if dmabuf_ids: if options.dmabuf_open: @@ -534,123 +570,40 @@ def build_application(options: GraphOptions, api=None) -> MixerApplication: width, height, options.fps) wait_for_sockets([f"{options.dmabuf_socket_dir}/{name}.sock" for name in dmabuf_ids], options.preheat_timeout_sec) - avp.edges.planCapacity("*", 4) input_edges = [ - _build_input( - avp, - api, - index, - url, - loop=options.loop_inputs, - fps=options.fps, - normalize=len(options.inputs) > 32, - options=options, - ) + _build_input(avp, api, index, url, loop=options.loop_inputs, fps=options.fps, + normalize=len(options.inputs) > 32, options=options) for index, url in enumerate(options.inputs) ] - mixer = api.MixerGraphBuilder( - avp, - name=MIXER_NAME, - canvas=(CANVAS_WIDTH, CANVAS_HEIGHT), - fps=(options.fps, FPS_DEN), - latency_ms=options.mixer_latency_ms, - hwaccel=HWACCEL, - enable_wipe=True, - defer_initial_routes=True, - defer_output=True, - keyframe_node=JANUS_KEYFRAME_NODE if options.janus_output else None, - cache_wipes_mb=options.wipe_cache_mb or None, - ) - routed_inputs = _register_sources( - avp, api, mixer, input_edges, fps=options.fps - ) + canvas = (CANVAS_WIDTH, CANVAS_HEIGHT) + mixer = _make_builder(avp, api, options, canvas=canvas, fps=options.fps, working_format=options.working_format) + routed_inputs = _register_sources(avp, api, mixer, input_edges, options.inputs, fps=options.fps, + color=options.input_color) _define_scenes(mixer, len(input_edges), routed_inputs) mixer.set_initial_scene("fullscreen_0", slot="A") - mixer_edge = mixer.build() - rtcp_feedback_listener = _build_outputs(avp, api, options, mixer_edge) - return MixerApplication( - avp=avp, - mixer=mixer, - input_groups=tuple(_input_group(index) for index in range(len(input_edges))), - input_edges=tuple(input_edges), - routed_inputs=routed_inputs, - preheat_timeout_sec=options.preheat_timeout_sec, - rtcp_feedback_listener=rtcp_feedback_listener, - wipe_file=options.wipe_file, - browser_windows=tuple(options.dmabuf_inputs), dmabuf_rest=options.dmabuf_rest, - wipe_cache_mb=options.wipe_cache_mb, - cut_latency_encoder=options.cut_latency_encoder, - prewarm_cut_scenes=options.prewarm_cut_scenes, - ) - - -def _build_renditions(avp, api, options: GraphOptions, cfg, mixer_edge: str): - """One encoder per rendition, all fed from the single composited program. - - The compositor renders once at the canvas rate; a rendition re-times and - rescales that picture for its own target, so extra renditions cost an - encode, not another composite. - """ - edges = [mixer_edge] - if len(cfg.renditions) > 1: - edges = [f"program_rendition_{r.id}" for r in cfg.renditions] - avp.addNode(api.Split({"name": "split_renditions", "src": mixer_edge, "dst": edges, - "group": OUTPUT_GROUP, "on_error": "panic"})) - listener = None - for rendition, edge in zip(cfg.renditions, edges): - scaled = edge - if (rendition.width, rendition.height) != (cfg.canvas_w, cfg.canvas_h): - scaled = f"program_scaled_{rendition.id}" - avp.addNode(api.FilterVideo({ - "name": f"scale_{rendition.id}", "src": edge, "dst": scaled, - "graph": f"scale_cuda=w={rendition.width}:h={rendition.height}", - "hwaccel": HWACCEL, "group": OUTPUT_GROUP, - })) - if rendition.target == "janus": - listener = build_janus_output( - avp, api, scaled, - JanusVideoConfig( - host=options.janus_host, - video_port=rendition.port or options.janus_video_port, - payload_type=options.janus_video_pt, ssrc=options.janus_video_ssrc, - bitrate_kbps=rendition.bitrate_kbps, - keyframe_min_interval_ms=options.keyframe_min_interval_ms, - rtcp_bind=options.janus_rtcp_bind, rtcp_port=options.janus_rtcp_port, - ), - fps=rendition.fps, fps_den=FPS_DEN, width=rendition.width, height=rendition.height, - hwaccel=HWACCEL, group=OUTPUT_GROUP, - profile=rendition.profile, preset=rendition.preset, - ) - else: - _build_record_output(avp, api, replace(options, output=rendition.target, - codec=rendition.codec, fps=rendition.fps), - scaled, width=rendition.width, height=rendition.height) - return listener + listener = _build_renditions(avp, api, options, _flag_renditions(options, *canvas), mixer.build(), + canvas=canvas, working_format=options.working_format) + return _application(avp, mixer, options, input_edges, listener, routed_inputs=routed_inputs, + browser_windows=tuple(options.dmabuf_inputs)) def _build_from_config(options: GraphOptions, cfg: "mixer_config.MixerConfig", api) -> MixerApplication: """Sources, wipes and scenes from a JSON document; one chain per source.""" options = replace(options, fps=cfg.fps) # the document owns the frame rate, outputs included - avp = api.AVPlumber() - if options.remote_control_port: - avp.enableControlServer(options.remote_control_port) - avp.executeCommandsFromString(f'hwaccel.init {{ "name": "{HWACCEL}", "type": "cuda" }}') + if options.mixer_latency_ms is None and cfg.latency_ms is not None: + options = replace(options, mixer_latency_ms=cfg.latency_ms) + avp = _init_avp(options, api) browsers = [s for s in cfg.sources if s.kind == "browser"] if browsers: open_windows(options.dmabuf_rest, [{"id": s.id, "url": s.location, "width": s.width, "height": s.height, "fps": s.fps or cfg.fps} for s in browsers]) wait_for_sockets([f"{options.dmabuf_socket_dir}/{s.id}.sock" for s in browsers], options.preheat_timeout_sec) - avp.edges.planCapacity("*", 4) - mixer = api.MixerGraphBuilder( - avp, name=MIXER_NAME, canvas=(cfg.canvas_w, cfg.canvas_h), fps=(cfg.fps, FPS_DEN), - latency_ms=options.mixer_latency_ms, hwaccel=HWACCEL, enable_wipe=True, - defer_initial_routes=True, defer_output=True, - keyframe_node=JANUS_KEYFRAME_NODE if options.janus_output else None, - cache_wipes_mb=options.wipe_cache_mb or None, - ) + canvas = (cfg.canvas_w, cfg.canvas_h) + mixer = _make_builder(avp, api, options, canvas=canvas, fps=cfg.fps, working_format=cfg.working_format, + color=cfg.out_color, wipe_color=cfg.wipe_color or None) aliases = cfg.alias_counts input_edges: list[str] = [] for index, source in enumerate(cfg.sources): @@ -662,158 +615,106 @@ def _build_from_config(options: GraphOptions, cfg: "mixer_config.MixerConfig", a cuda_hwaccel=HWACCEL, source_group=group, processing_group=group, hold=True) for node in nodes: avp.addNode(node) + elif source.kind == "v210": + # True 10-bit 4:2:2 sources: packed v210 unpacked to P210 on the GPU + # and stamped with their declared color contract. NVDEC only yields + # 4:2:0, so this is the one path that keeps 4:2:2 through the canvas. + edge = build_v210_input( + avp, api, str(index), source.location, width=source.width, height=source.height, + group=group, fps=cfg.fps, fps_den=FPS_DEN, hwaccel=HWACCEL, loop=source.loop, + color=source.color.tags) else: edge = build_input(avp, api, str(index), source.location, group=group, fps=cfg.fps, fps_den=FPS_DEN, hwaccel=HWACCEL, loop=source.loop) - input_edges.append(edge) - count = aliases[source.id] - edges = [edge] - if count > 1: - # The same frames under several names: one fan-out, no second decoder. - edges = [f"{edge}_alias{k}" for k in range(1, count + 1)] - avp.addNode(api.OneToMany({ - "type": "one_to_many", "name": f"alias_{index}", "src": edge, "dst": edges, - "outputs": (1 << count) - 1, "group": group, + if source.filter_graph: + filtered_edge = f"input_{index}_filtered" + avp.addNode(api.FilterVideo({ + "name": f"source_filter_{index}", "src": edge, "dst": filtered_edge, + "graph": (source.color.setparams + "," if source.color else "") + source.filter_graph, "hwaccel": HWACCEL, + "group": group, "auto_restart": "group", })) - for k, alias_edge in enumerate(edges, start=1): - mixer.add_source(mixer_config.alias_name(source.id, k), pre_otm_edge=alias_edge, - input_group=group, default_graph="") + edge = filtered_edge + input_edges.append(edge) + for k in range(1, aliases[source.id] + 1): + # The reusable builder shares conversion and fan-out for identical edges. + mixer.add_source(mixer_config.alias_name(source.id, k), pre_otm_edge=edge, + input_group=group, default_graph="", + color=None if source.filter_graph else source.color, + packed_rgb=source.kind == "browser", + pixel_format=source.filter_output_format or + ("p210le" if source.kind == "v210" else None)) for scene in cfg.scenes: mixer.add_scene(scene.id, mixer_config.scene_layers(cfg, scene)) mixer.set_initial_scene(cfg.initial_scene, slot="A") settings = json.dumps(cfg.settings(), separators=(",", ":")) + "\n" avp.registerControlCommand("mixer.settings", lambda _arg: settings, True) - mixer_edge = mixer.build() - if cfg.renditions: - rtcp_feedback_listener = _build_renditions(avp, api, options, cfg, mixer_edge) - else: - rtcp_feedback_listener = _build_outputs(avp, api, options, mixer_edge, - width=cfg.canvas_w, height=cfg.canvas_h) - return MixerApplication( - avp=avp, mixer=mixer, - input_groups=tuple(_input_group(index) for index in range(len(cfg.sources))), - input_edges=tuple(input_edges), routed_inputs=False, - preheat_timeout_sec=options.preheat_timeout_sec, - rtcp_feedback_listener=rtcp_feedback_listener, - wipe_file=options.wipe_file, wipe_files=tuple(w.path for w in cfg.wipes), - browser_windows=tuple(s.id for s in browsers), dmabuf_rest=options.dmabuf_rest, - wipe_cache_mb=options.wipe_cache_mb, - cut_latency_encoder=options.cut_latency_encoder, - prewarm_cut_scenes=options.prewarm_cut_scenes, - ) + listener = _build_renditions(avp, api, options, cfg.renditions or _flag_renditions(options, *canvas), + mixer.build(), canvas=canvas, working_format=cfg.working_format, color=cfg.out_color) + return _application(avp, mixer, options, input_edges, listener, routed_inputs=False, + wipe_files=tuple(w.path for w in cfg.wipes), browser_windows=tuple(s.id for s in browsers)) def parse_args(argv: list[str] | None = None) -> GraphOptions: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument( - "--input", - dest="inputs", - action="append", - default=[], - metavar="PATH", - help="Input media file or URL; repeat for each mixer input", - ) - parser.add_argument("--config", metavar="FILE", - help="JSON document with sources, wipes and scenes (replaces --input and the " - "built-in layouts; see doc/research/2026-09-08-mixer-config-schema.md)") - parser.add_argument("--output", help="Optional video-only output URL or path") - parser.add_argument("--dmabuf-socket-dir", default="/tmp/dma-page", - help="dma-browser socket directory for dmabuf:// inputs") - parser.add_argument("--dmabuf-size", default="1280x720", metavar="WxH", - help="browser window size for dmabuf:// inputs") - parser.add_argument("--dmabuf-open", metavar="URL", - help="open the dmabuf:// windows with this page through the dma-browser REST API") - parser.add_argument("--dmabuf-rest", default="http://127.0.0.1:9009") - parser.add_argument( - "--output-format", - help="Muxer format when it cannot be inferred from the output", - ) - parser.add_argument("--remote-control-port", type=int, default=7777) - parser.add_argument("--codec", default="h264_nvenc") - parser.add_argument("--bitrate", default="8M") - parser.add_argument( - "--fps", - type=int, - default=DEFAULT_FPS, - help=f"Mixer and output frame rate (default: {DEFAULT_FPS})", - ) - parser.add_argument("--mixer-latency-ms", type=float, help="Native playout buffer (default: two output frames)") - parser.add_argument("--loop-inputs", action="store_true") - parser.add_argument( - "--janus-output", - action="store_true", - help="Publish the video-only program to Janus over RTP", - ) - parser.add_argument("--janus-host", default=JANUS_DEFAULT_HOST) - parser.add_argument( - "--janus-video-port", type=int, default=JANUS_DEFAULT_VIDEO_PORT - ) - parser.add_argument( - "--janus-video-pt", type=int, default=JANUS_DEFAULT_VIDEO_PT - ) - parser.add_argument( - "--janus-video-ssrc", - type=lambda value: int(value, 0), - default=JANUS_DEFAULT_VIDEO_SSRC, - ) - parser.add_argument( - "--janus-video-bitrate-kbps", - type=int, - default=JANUS_DEFAULT_VIDEO_BITRATE_KBPS, - ) - parser.add_argument("--janus-rtcp-bind", default="0.0.0.0") - parser.add_argument("--keyframe-min-interval-ms", type=int, - default=DEFAULT_KEYFRAME_MIN_INTERVAL_MS, - help="Minimum forced-keyframe spacing for Janus output in media time " - "(default: 150 ms; 0 disables rate limiting)") - parser.add_argument("--janus-rtcp-port", type=int, default=0) - parser.add_argument("--preheat-timeout", type=float, default=60.0) - parser.add_argument("--wipe-file", help="Alpha wipe clip to warm the media-wipe chain up with at start " - "(the TUI still selects the clip for each wipe)") - parser.add_argument("--wipe-cache-mb", type=float, default=768.0, - help="Hold decoded wipe clips in GPU memory, up to this many MiB " - "(0 decodes each wipe on every take)") - parser.add_argument("--webui-url", default="", - help="Register the graph with an AVPlumber web UI, e.g. http://127.0.0.1:22222") - parser.add_argument("--cut-latency-encoder", default="", metavar="NODE", - help="Measure CUT receipt to matching encoded frame at NODE (e.g. janus_encoder)") - parser.add_argument("--prewarm-cut-scene", action="append", default=[], metavar="SCENE", - help="Keep source buffers warm for direct cuts (repeat; '*' selects all scenes)") - args = parser.parse_args(argv) - if not args.inputs and not args.config: - parser.error("pass --input (repeatable) or --config FILE") - return GraphOptions( - inputs=tuple(args.inputs), - output=args.output, - output_format=args.output_format, - remote_control_port=args.remote_control_port, - codec=args.codec, - bitrate=args.bitrate, - fps=args.fps, - mixer_latency_ms=args.mixer_latency_ms, - loop_inputs=args.loop_inputs, - janus_output=args.janus_output, - janus_host=args.janus_host, - janus_video_port=args.janus_video_port, - janus_video_pt=args.janus_video_pt, - janus_video_ssrc=args.janus_video_ssrc, - janus_video_bitrate_kbps=args.janus_video_bitrate_kbps, - keyframe_min_interval_ms=args.keyframe_min_interval_ms, - janus_rtcp_bind=args.janus_rtcp_bind, - janus_rtcp_port=args.janus_rtcp_port, - preheat_timeout_sec=args.preheat_timeout, - wipe_file=args.wipe_file, - dmabuf_socket_dir=args.dmabuf_socket_dir, - dmabuf_size=parse_size(args.dmabuf_size), - dmabuf_open=args.dmabuf_open, - dmabuf_rest=args.dmabuf_rest, - config=args.config, - webui_url=args.webui_url, - cut_latency_encoder=args.cut_latency_encoder, - prewarm_cut_scenes=tuple(args.prewarm_cut_scene), - wipe_cache_mb=args.wipe_cache_mb, - ) - + """Every option's ``dest`` is a GraphOptions field, so the parsed namespace + maps onto the dataclass directly; an unmapped option fails loudly instead + of being silently dropped.""" + p = argparse.ArgumentParser(description=__doc__) + add = p.add_argument + add("--input", dest="inputs", action="append", default=[], metavar="PATH", + help="Input media file or URL; repeat for each mixer input") + add("--config", metavar="FILE", + help="JSON document with sources, wipes and scenes (replaces --input and the built-in " + "layouts; see doc/research/2026-09-08-mixer-config-schema.md)") + add("--output", help="Optional video-only output URL or path") + add("--output-format", help="Muxer format when it cannot be inferred from the output") + add("--codec", default="h264_nvenc") + add("--bitrate", default="8M") + add("--fps", type=int, default=DEFAULT_FPS, help=f"Mixer and output frame rate (default: {DEFAULT_FPS})") + add("--wipe-color", default="", choices=("sdr",), + help="Override wipe color tags as SDR (default: preserve tags, assume SDR for missing tags)") + add("--input-color", default="", choices=("", "sdr", "hlg", "pq"), + help="color contract declared for every --input file (default: trust the decoded frame tags)") + add("--working-format", default="nv12", + help="compositor/transition sw_format; p210le keeps 10-bit 4:2:2 on the canvas " + "(renditions subsample to P010/NV12 for NVENC automatically)") + add("--mixer-latency-ms", type=float, help="Native playout buffer (default: two output frames)") + add("--loop-inputs", action="store_true") + add("--remote-control-port", type=int, default=7777) + add("--dmabuf-socket-dir", default="/tmp/dma-page", + help="dma-browser socket directory for dmabuf:// inputs") + add("--dmabuf-size", default="1280x720", metavar="WxH", + help="browser window size for dmabuf:// inputs") + add("--dmabuf-open", metavar="URL", + help="open the dmabuf:// windows with this page through the dma-browser REST API") + add("--dmabuf-rest", default="http://127.0.0.1:9009") + add("--janus-output", action="store_true", help="Publish the video-only program to Janus over RTP") + add("--janus-host", default=JANUS_DEFAULT_HOST) + add("--janus-video-port", type=int, default=JANUS_DEFAULT_VIDEO_PORT) + add("--janus-video-pt", type=int, default=JANUS_DEFAULT_VIDEO_PT) + add("--janus-video-ssrc", type=lambda v: int(v, 0), default=JANUS_DEFAULT_VIDEO_SSRC) + add("--janus-video-bitrate-kbps", type=int, default=JANUS_DEFAULT_VIDEO_BITRATE_KBPS) + add("--janus-rtcp-bind", default="0.0.0.0") + add("--janus-rtcp-port", type=int, default=0) + add("--keyframe-min-interval-ms", type=int, default=DEFAULT_KEYFRAME_MIN_INTERVAL_MS, + help="Minimum forced-keyframe spacing for Janus output in media time " + "(default: 150 ms; 0 disables rate limiting)") + add("--preheat-timeout", dest="preheat_timeout_sec", type=float, default=60.0) + add("--wipe-file", help="Alpha wipe clip to warm the media-wipe chain up with at start " + "(the TUI still selects the clip for each wipe)") + add("--wipe-cache-mb", type=float, default=768.0, + help="Hold decoded wipe clips in GPU memory, up to this many MiB (0 decodes each wipe on every take)") + add("--webui-url", default="", help="Register the graph with an AVPlumber web UI, e.g. http://127.0.0.1:22222") + add("--cut-latency-encoder", default="", metavar="NODE", + help="Measure CUT receipt to matching encoded frame at NODE (e.g. janus_encoder)") + add("--prewarm-cut-scene", dest="prewarm_cut_scenes", action="append", default=[], metavar="SCENE", + help="Keep source buffers warm for direct cuts (repeat; '*' selects all scenes)") + args = vars(p.parse_args(argv)) + if not args["inputs"] and not args["config"]: + p.error("pass --input (repeatable) or --config FILE") + args["inputs"] = tuple(args["inputs"]) + args["prewarm_cut_scenes"] = tuple(args["prewarm_cut_scenes"]) + args["dmabuf_size"] = parse_size(args["dmabuf_size"]) # ValueError on bad WxH, not a usage exit + return GraphOptions(**args) def parse_size(text: str) -> tuple[int, int]: try: diff --git a/demos/mixer/smoke_test.py b/demos/mixer/smoke_test.py index ba7d8f03..b5f2883b 100644 --- a/demos/mixer/smoke_test.py +++ b/demos/mixer/smoke_test.py @@ -7,9 +7,9 @@ import json import time -from avpmixer.control import AvpConnection +from pyplumber.mixer.control import AvpConnection -from avpmixer.control import mixer_command, parse_mixer_status, parse_scene_list +from pyplumber.mixer.control import mixer_command, parse_mixer_status, parse_scene_list async def _command( diff --git a/demos/mixer/tests/check_interruptions.py b/demos/mixer/tests/check_interruptions.py index f50a344d..247135e8 100644 --- a/demos/mixer/tests/check_interruptions.py +++ b/demos/mixer/tests/check_interruptions.py @@ -6,8 +6,8 @@ import sys import time sys.path.insert(0, str(pathlib.Path(__file__).resolve().parents[1])) -from avpmixer.control import AvpConnection -from avpmixer.control import mixer_command +from pyplumber.mixer.control import AvpConnection +from pyplumber.mixer.control import mixer_command async def run(args): c = AvpConnection(args.host, args.port) diff --git a/demos/mixer/tests/check_transition_recovery.py b/demos/mixer/tests/check_transition_recovery.py index 381b545c..782add48 100644 --- a/demos/mixer/tests/check_transition_recovery.py +++ b/demos/mixer/tests/check_transition_recovery.py @@ -14,7 +14,7 @@ import numpy as np -from avpmixer import MixerGraphBuilder +from pyplumber.mixer import MixerGraphBuilder from pyplumber import AVPlumber from pyplumber.node import DecVideo, Demux, FilterVideo, ForceFPS, InputRec, Realtime, SourceSwitcher from frame_codes import read_code diff --git a/demos/mixer/tests/conftest.py b/demos/mixer/tests/conftest.py index 200ffa5f..14a5b5ba 100644 --- a/demos/mixer/tests/conftest.py +++ b/demos/mixer/tests/conftest.py @@ -4,4 +4,4 @@ MIXER_DIR = Path(__file__).resolve().parents[1] sys.path.insert(0, str(MIXER_DIR)) -sys.path.insert(0, str(MIXER_DIR.parents[1])) # repo root: avpmixer package +sys.path.insert(0, str(MIXER_DIR.parents[1])) # repo root: pyplumber package diff --git a/demos/mixer/tests/sdr_patterns.py b/demos/mixer/tests/sdr_patterns.py new file mode 100644 index 00000000..55770961 --- /dev/null +++ b/demos/mixer/tests/sdr_patterns.py @@ -0,0 +1,53 @@ +#!/usr/bin/env python3 +"""Render colorful SDR pattern clips for a many-source show (needs FFmpeg with lavfi). + + sdr_patterns.py media/patterns --fps 60 --seconds 20 --encoder h264_nvenc + +One clip per generator below; each is a distinct NVDEC-decodable source, far +cheaper at run time than raw v210 (which streams ~330 MB/s per 1080p60 input). +""" +import argparse +import pathlib +import subprocess + +GENERATORS = { + "testsrc2": "testsrc2", + "bars": "smptehdbars", + "rgbtest": "rgbtestsrc", + "mandelbrot": "mandelbrot", + "gradients": "gradients=speed=0.05:nb_colors=6", + "life": "life=mold=10:life_color=#ffcc00:death_color=#3300aa:rule=B3/S23", + "sierpinski": "sierpinski=type=triangle:seed=7", + "cellauto": "cellauto=rule=30:scroll=1", +} + + +def render(directory: pathlib.Path, name: str, graph: str, size: str, fps: int, seconds: int, + encoder: str, ffmpeg: str) -> pathlib.Path: + out = directory / f"{name}.mp4" + source = f"{graph}{':' if '=' in graph else '='}size={size}:rate={fps}" + subprocess.run([ffmpeg, "-v", "error", "-nostdin", "-y", "-f", "lavfi", "-i", source, "-t", str(seconds), + "-c:v", encoder, "-b:v", "12M", "-maxrate", "16M", "-g", str(fps), "-pix_fmt", "yuv420p", str(out)], + check=True) + return out + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("directory", type=pathlib.Path) + parser.add_argument("--size", default="1920x1080") + parser.add_argument("--fps", type=int, default=60) + parser.add_argument("--seconds", type=int, default=20) + parser.add_argument("--encoder", default="h264_nvenc", help="h264_nvenc, or libx264 without a GPU") + parser.add_argument("--ffmpeg", default="ffmpeg") + parser.add_argument("--only", nargs="*", choices=sorted(GENERATORS), help="subset of generators") + args = parser.parse_args() + args.directory.mkdir(parents=True, exist_ok=True) + for name, graph in GENERATORS.items(): + if args.only and name not in args.only: + continue + print(render(args.directory, name, graph, args.size, args.fps, args.seconds, args.encoder, args.ffmpeg)) + + +if __name__ == "__main__": + main() diff --git a/demos/mixer/tests/smoke_live_takes.py b/demos/mixer/tests/smoke_live_takes.py new file mode 100644 index 00000000..3f72d231 --- /dev/null +++ b/demos/mixer/tests/smoke_live_takes.py @@ -0,0 +1,84 @@ +"""Drive a running mixer through every scene with cuts and fades and check that its +Janus outputs keep flowing. Run against a live deployment, not a fixture:: + + python3 smoke_live_takes.py --host 127.0.0.1 --port 7777 --janus http://127.0.0.1:8088/janus --mountpoints 1 2 + +Every take must leave the requested scene on program with the transition idle and +every listed Janus mountpoint receiving packets younger than --max-age-ms. +""" + +import argparse +import asyncio +import json +import sys +import time +import urllib.request +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) +from pyplumber.mixer.control import AvpConnection, mixer_command, parse_mixer_status, parse_scene_list # noqa: E402 + + +def janus_ages(url, mountpoints, secret): + def post(path, body): + req = urllib.request.Request(path, data=json.dumps(body).encode(), headers={"Content-Type": "application/json"}) + with urllib.request.urlopen(req, timeout=5) as r: + return json.load(r) + sid = post(url, {"janus": "create", "transaction": "t"})["data"]["id"] + hid = post(f"{url}/{sid}", {"janus": "attach", "plugin": "janus.plugin.streaming", "transaction": "t"})["data"]["id"] + ages = {} + for mp in mountpoints: + info = post(f"{url}/{sid}/{hid}", {"janus": "message", "transaction": "t", + "body": {"request": "info", "id": mp, "secret": secret}}) + ages[mp] = info["plugindata"]["data"]["info"]["media"][0].get("age_ms") + return ages + + +async def run(args): + c = AvpConnection(args.host, args.port) + await c.connect() + scenes = parse_scene_list(await c.command(f"mixer.scenes {args.mixer}")) + print("scenes:", scenes, flush=True) + failures = 0 + + async def take(kind, scene, **payload): + nonlocal failures + await c.command(mixer_command("preview", args.mixer, scene=scene)) + await c.command(mixer_command(kind, args.mixer, scene=scene, **payload)) + await asyncio.sleep(payload.get("duration_sec", 0) + args.settle) + status = parse_mixer_status(await c.command(f"mixer.status {args.mixer}")) + ages = janus_ages(args.janus, args.mountpoints, args.secret) if args.janus else {} + ok = status.pgm_scene == scene and status.transition == "idle" and \ + all(a is not None and a < args.max_age_ms for a in ages.values()) + failures += not ok + print(f"{'OK ' if ok else 'BAD'} {kind:<4} -> {scene:<18} pgm={status.pgm_scene:<18} " + f"transition={status.transition:<5} janus age ms={ages}", flush=True) + + for scene in scenes: + await take("cut", scene) + for scene in scenes: + await take("fade", scene, duration_sec=args.fade) + for i in range(args.rapid): + await take("cut", scenes[i % len(scenes)]) + await c.disconnect() + print("RESULT:", "PASS" if not failures else f"FAIL ({failures} takes)", flush=True) + return 0 if not failures else 1 + + +def main(): + p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("--host", default="127.0.0.1") + p.add_argument("--port", type=int, default=7777) + p.add_argument("--mixer", default="mixer") + p.add_argument("--janus", default="", help="Janus HTTP API base URL; empty skips the output check") + p.add_argument("--mountpoints", type=int, nargs="*", default=[1]) + p.add_argument("--secret", default="avpsecret") + p.add_argument("--fade", type=float, default=0.5) + p.add_argument("--rapid", type=int, default=10, help="back-to-back cuts after the scene sweep") + p.add_argument("--settle", type=float, default=1.2, help="seconds to wait after each take before checking") + p.add_argument("--max-age-ms", type=int, default=400) + sys.exit(asyncio.run(run(p.parse_args()))) + + +if __name__ == "__main__": + main() diff --git a/demos/mixer/tests/test_control.py b/demos/mixer/tests/test_control.py index ca26c1b2..28effbe9 100644 --- a/demos/mixer/tests/test_control.py +++ b/demos/mixer/tests/test_control.py @@ -3,7 +3,7 @@ import pytest -from avpmixer.control import (AvpConnection, mixer_command, parse_mixer_status, +from pyplumber.mixer.control import (AvpConnection, mixer_command, parse_mixer_status, parse_scene_list) diff --git a/demos/mixer/tests/test_graph.py b/demos/mixer/tests/test_graph.py index 3c424788..251ebcb3 100644 --- a/demos/mixer/tests/test_graph.py +++ b/demos/mixer/tests/test_graph.py @@ -4,6 +4,8 @@ import pytest +from pyplumber.mixer import config as mixer_config +from pyplumber.mixer.color import Color from mixer import GraphOptions, build_application, infer_output_format, parse_args @@ -155,6 +157,7 @@ def fake_api(): "realtime", "repeat_last_frame", "split", + "v210_to_cuda", ) api = { "AVPlumber": FakeAvp, @@ -182,6 +185,7 @@ def fake_api(): "realtime": "Realtime", "repeat_last_frame": "RepeatLastFrame", "split": "Split", + "v210_to_cuda": "V210ToCuda", }[name]: node_type(name) for name in names }) @@ -312,10 +316,9 @@ def test_record_and_janus_outputs_split_program_video(): ) nodes = {node.parameters["name"]: node.parameters for node in application.avp.nodes} - assert nodes["split_program_video_output"]["dst"] == [ - "program_video_record", - "program_video_janus", - ] + # Both flags become renditions of the one program, like a config document's. + assert nodes["split_renditions"]["dst"] == ["program_rendition_program", "program_rendition_janus"] + assert nodes["program_encoder"]["codec"] == "h264_nvenc" and nodes["program_encoder"]["options"]["b"] == "8000k" def test_output_target_is_required(): @@ -393,7 +396,7 @@ def test_cli_parses_dmabuf_options(): def test_dmabuf_windows_are_closed_before_reopening(monkeypatch): - from avpmixer import dmabuf_inputs + from pyplumber.mixer import dmabuf_inputs calls = [] @@ -440,7 +443,7 @@ def test_wipe_file_preloads_into_the_clip_cache_at_start(monkeypatch): "canvas": {"width": 1920, "height": 1080, "fps": 60}, "sources": [ {"id": "cam", "kind": "video", "path": "/media/cam.mp4", "width": 1920, "height": 1080}, - {"id": "page", "kind": "browser", "url": "https://example.org/", "width": 1280, "height": 720}, + {"id": "page", "kind": "browser", "color": "sdr", "url": "https://example.org/", "width": 1280, "height": 720}, ], "wipes": [{"id": "swoosh", "path": "/media/swoosh.mov"}], "control": {"direct": False, "fade_seconds": 0.8}, @@ -457,7 +460,7 @@ def test_wipe_file_preloads_into_the_clip_cache_at_start(monkeypatch): def test_config_scene_layers_carry_z_cover_and_aliases(): - from avpmixer import config as mc + from pyplumber.mixer import config as mc cfg = mc.parse(CONFIG) assert cfg.alias_counts == {"cam": 2, "page": 1} full = mc.scene_layers(cfg, cfg.scenes[0]) @@ -473,7 +476,7 @@ def test_config_scene_layers_carry_z_cover_and_aliases(): def test_config_rejects_duplicate_locations_and_bad_references(): - from avpmixer import config as mc + from pyplumber.mixer import config as mc import copy dup = copy.deepcopy(CONFIG) dup["sources"].append({"id": "cam2", "kind": "video", "path": "/media/cam.mp4"}) @@ -483,6 +486,10 @@ def test_config_rejects_duplicate_locations_and_bad_references(): bad["scenes"][0]["items"][0]["source"] = "nope" with pytest.raises(mc.ConfigError, match="unknown source"): mc.parse(bad) + bad_filter = copy.deepcopy(CONFIG) + bad_filter["sources"][0]["filter"] = {"transfer_in": "sdr"} + with pytest.raises(mc.ConfigError, match="filter must be a CUDA filter graph string"): + mc.parse(bad_filter) nocover = copy.deepcopy(CONFIG) del nocover["sources"][0]["width"] cfg = mc.parse(nocover) @@ -495,16 +502,19 @@ def test_config_rejects_duplicate_locations_and_bad_references(): assert mc.scene_layers(probed, probed.scenes[0])["cam"]["crop"] == {"x": 0, "y": 0, "w": 640, "h": 360} -def test_config_builds_one_chain_per_source_with_alias_fanout(tmp_path, monkeypatch): +@pytest.mark.parametrize("source_filter", ["", "tonemap_cuda=transfer_in=sdr:transfer_out=hlg,scale_cuda=format=p210le"]) +def test_config_builds_one_chain_per_source_with_alias_fanout(tmp_path, monkeypatch, source_filter): import json as _json - from avpmixer import dmabuf_inputs + from pyplumber.mixer import dmabuf_inputs (tmp_path / "page.sock").touch() path = tmp_path / "mixer.json" - path.write_text(_json.dumps(CONFIG)) + doc = {**CONFIG, "sources": [{**CONFIG["sources"][0], "filter": source_filter, "filter_output_format": "p210le" if source_filter else ""}, + *CONFIG["sources"][1:]]} + path.write_text(_json.dumps(doc)) opened = [] monkeypatch.setattr(dmabuf_inputs, "rest_request", lambda base, method, p, body=None: opened.append((method, p, body)) or {"windows": []}) - from avpmixer import config as mc + from pyplumber.mixer import config as mc monkeypatch.setattr(mc, "probe_video_size", lambda path: (_ for _ in ()).throw(AssertionError("declared sizes must not be probed"))) FakeMixer.instances.clear() application = build_application( @@ -513,8 +523,18 @@ def test_config_builds_one_chain_per_source_with_alias_fanout(tmp_path, monkeypa mixer = FakeMixer.instances[-1] assert [name for name in nodes if name and name.startswith("decode_")] == ["decode_0"] - assert nodes["alias_0"]["dst"] == ["input_0_fps_alias1", "input_0_fps_alias2"] - assert nodes["alias_0"]["outputs"] == 3 + source_edge = "input_0_filtered" if source_filter else "input_0_fps" + assert dict(mixer.sources)["cam"]["pre_otm_edge"] == source_edge + assert dict(mixer.sources)["cam#2"]["pre_otm_edge"] == source_edge + if source_filter: + assert nodes["source_filter_0"]["graph"] == source_filter + assert nodes["source_filter_0"]["src"] == "input_0_fps" + assert nodes["source_filter_0"]["hwaccel"] == "@gpu" + assert nodes["source_filter_0"]["group"] == "input_0" + assert [n for n in nodes if n and n.startswith("source_filter_")] == ["source_filter_0"] + else: + assert "source_filter_0" not in nodes + assert "alias_0" not in nodes # shared fan-out belongs to the reusable builder assert [name for name, _ in mixer.sources] == ["cam", "cam#2", "page"] assert dict(mixer.sources)["page"]["pre_otm_edge"] == "input_1_held" assert opened[-1][2] == {"id": "page", "url": "https://example.org/", "width": 1280, "height": 720, @@ -539,7 +559,7 @@ def test_config_builds_one_chain_per_source_with_alias_fanout(tmp_path, monkeypa def test_example_configuration_uses_placeholder_locations_and_valid_layers(): - from avpmixer import config as mc + from pyplumber.mixer import config as mc path = Path(__file__).resolve().parents[1] / "config.example.json" cfg = mc.load(path) @@ -581,7 +601,7 @@ def test_transitions_trigger_a_keyframe_only_when_streaming(): @pytest.mark.parametrize("minimum", [0, 100, 150, 200, 500]) def test_janus_keyframe_limit_is_configurable(minimum): - from avpmixer.janus import JanusVideoConfig, build_janus_output + from pyplumber.mixer.janus import JanusVideoConfig, build_janus_output avp = FakeAvp() build_janus_output(avp, fake_api(), "program", JanusVideoConfig(keyframe_min_interval_ms=minimum), fps=60, width=1920, height=1080) @@ -591,7 +611,7 @@ def test_janus_keyframe_limit_is_configurable(minimum): @pytest.mark.parametrize("minimum", [-1, 0.2, True, "200", 2**31]) def test_janus_rejects_invalid_keyframe_limit(minimum): - from avpmixer.janus import JanusVideoConfig + from pyplumber.mixer.janus import JanusVideoConfig with pytest.raises(ValueError, match="keyframe_min_interval_ms"): JanusVideoConfig(keyframe_min_interval_ms=minimum) @@ -615,7 +635,7 @@ def test_mixer_keyframe_option_reaches_each_output_path(tmp_path, configured): def test_wipe_dir_scans_a_library_and_explicit_entries_win(tmp_path): - from avpmixer import config as mc + from pyplumber.mixer import config as mc for name in ("b_swoosh.mov", "a_dip.webm", "notes.txt", "c_star.mp4"): (tmp_path / name).write_bytes(b"x") doc = {**CONFIG, "wipe_dir": str(tmp_path), @@ -632,7 +652,7 @@ def test_wipe_dir_scans_a_library_and_explicit_entries_win(tmp_path): def test_wipe_cache_is_optional_and_splits_the_chain(tmp_path): - from avpmixer import clipcache + from pyplumber.mixer import clipcache FakeMixer.instances.clear() default = build_application(GraphOptions(inputs=("a.mp4",), output="p.mp4"), api=fake_api()) @@ -651,7 +671,7 @@ def test_wipe_cache_is_optional_and_splits_the_chain(tmp_path): def test_clip_cache_node_parameters_name_the_clip_by_url(): - from avpmixer import clipcache + from pyplumber.mixer import clipcache node = clipcache.cache_node(name="mixer_wipe_cache", src="in", dst="out", group="g", fps="60/1", budget_mb=256, url="/media/w.mov") @@ -668,7 +688,8 @@ def test_clip_cache_node_parameters_name_the_clip_by_url(): def test_renditions_encode_the_one_composited_program(tmp_path, monkeypatch): import json as _json - from avpmixer import dmabuf_inputs + from pyplumber.mixer import dmabuf_inputs + from pyplumber.mixer import dmabuf_inputs (tmp_path / "page.sock").touch() monkeypatch.setattr(dmabuf_inputs, "rest_request", lambda *a, **k: {"windows": []}) doc = {**RENDITION_CONFIG, @@ -683,8 +704,8 @@ def test_renditions_encode_the_one_composited_program(tmp_path, monkeypatch): nodes = {n.parameters.get("name"): n.parameters for n in app.avp.nodes} assert nodes["split_renditions"]["dst"] == ["program_rendition_program", "program_rendition_square"] - assert "scale_program" not in nodes # already the canvas size - assert nodes["scale_square"]["graph"] == "scale_cuda=w=1080:h=1080" + assert nodes["scale_program"]["graph"] == Color().setparams # SDR canvas to SDR output: tags only + assert nodes["scale_square"]["graph"].startswith("scale_cuda=w=1080:h=1080,") assert nodes["janus_encoder"]["options"]["preset"] == "p7" assert nodes["janus_encoder"]["options"]["profile"] == "baseline" assert nodes["janus_fps"]["fps"] == "30/1" @@ -692,7 +713,7 @@ def test_renditions_encode_the_one_composited_program(tmp_path, monkeypatch): def test_a_rendition_may_not_ask_for_more_than_the_composer_renders(): - from avpmixer import config as mc + from pyplumber.mixer import config as mc with pytest.raises(mc.ConfigError, match="exceeds the canvas rate"): mc.parse({**RENDITION_CONFIG, "renditions": [{**RENDITION_CONFIG["renditions"][0], "fps": 60}]}) @@ -703,8 +724,242 @@ def test_a_rendition_may_not_ask_for_more_than_the_composer_renders(): assert cfg.renditions[0].aspect == "9:16" and cfg.renditions[0].fps == 30 +def test_hdr_and_sdr_janus_renditions_have_independent_feedback(tmp_path): + doc = {**CONFIG, "sources": CONFIG["sources"][:1], "scenes": CONFIG["scenes"][:1], + "initial_scene": "full", "wipes": [], + "canvas": {"width": 1920, "height": 1080, "fps": 60, "working_format": "p010le", + "color": "hlg"}, + "renditions": [ + {"id": "hdr", "target": "janus", "port": 5006, "codec": "hevc_nvenc", "profile": "main10"}, + {"id": "sdr", "target": "janus", "port": 5004, "codec": "h264_nvenc", + "profile": "baseline", "tonemap": "mobius", "tonemap_param": 0.9}]} + path = tmp_path / "dual.json" + path.write_text(json.dumps(doc)) + app = build_application(GraphOptions(config=str(path), janus_output=True), api=fake_api()) + nodes = {n.parameters["name"]: n.parameters for n in app.avp.nodes} + assert len(nodes) == len(app.avp.nodes) + assert nodes["janus_encoder"]["options"]["profile"] == "main10" + assert nodes["janus_sdr_encoder"]["options"]["profile"] == "baseline" + assert nodes["janus_sdr_encoder"]["options"]["color_trc"] == "bt709" + assert nodes["janus_sdr_format"]["real_pixel_format"] == "nv12" + assert "tonemap_cuda=transfer_in=auto:transfer_out=sdr" in nodes["scale_sdr"]["graph"] + assert ":tonemap=mobius:sdr_white=203:hdr_peak=1000:desat=0:param=0.9" in nodes["scale_sdr"]["graph"] + assert ":5006?" in nodes["janus_rtp_output"]["url"] + assert ":5004?" in nodes["janus_sdr_rtp_output"]["url"] + feedback = app.rtcp_feedback_listener + feedback.start() + assert all(listener.started for listener in feedback.listeners) + for listener in feedback.listeners: + listener.parameters["on_keyframe_request"](None) + assert app.avp.commands[-2:] == [ + "node.object.set janus_force_keyframe trigger true", + "node.object.set janus_sdr_force_keyframe trigger true"] + feedback.stop() + assert not any(listener.started for listener in feedback.listeners) + + +def test_v210_sources_keep_422_through_a_p210_canvas(tmp_path): + """NVDEC only yields 4:2:0; generated v210 content is the 4:2:2 path. It must + reach the P210 canvas untouched and be subsampled once, at the encoder.""" + doc = {**CONFIG, "wipes": [], "initial_scene": "full", + "sources": [{"id": "gen", "kind": "v210", "path": "/media/gen.v210", "width": 1920, + "height": 1080, "color": "hlg"}, CONFIG["sources"][0]], + "scenes": [{"id": "full", "items": [ + {"source": "gen", "dst": {"x": 0, "y": 0, "w": 1920, "h": 1080}}, + {"source": "cam", "dst": {"x": 0, "y": 0, "w": 960, "h": 540}}]}], + "canvas": {"width": 1920, "height": 1080, "fps": 60, "working_format": "p210le", "color": "hlg"}, + "renditions": [{"id": "hdr", "target": "janus"}, + {"id": "sdr", "target": "/rec/sdr.mp4", "codec": "h264_nvenc", "tonemap": "hable"}]} + path = tmp_path / "gen.json" + path.write_text(json.dumps(doc)) + FakeMixer.instances.clear() + app = build_application(GraphOptions(config=str(path), janus_output=True), api=fake_api()) + nodes = {n.parameters["name"]: n.parameters for n in app.avp.nodes} + sources = dict(FakeMixer.instances[-1].sources) + assert nodes["unpack_0"]["sw_format"] == "p210le" and nodes["unpack_0"]["color_trc"] == "arib-std-b67" + assert sources["gen"]["pixel_format"] == "p210le" and sources["gen"]["color"] == Color("hlg") + assert sources["cam"]["pixel_format"] is None and sources["cam"]["color"] is None # NVDEC, tags from the frames + # HDR out: one chroma subsample to P010 for NVENC, no tone-map pass. + assert nodes["scale_hdr"]["graph"] == Color("hlg").setparams + ",scale_cuda=format=p010le" + assert nodes["janus_format"]["real_pixel_format"] == "p010le" + assert nodes["scale_sdr"]["graph"].startswith(Color("hlg").setparams + ",tonemap_cuda=transfer_in=auto:transfer_out=sdr:format=nv12") + + +def test_cli_inputs_declare_browser_rgb_and_optional_file_color(tmp_path): + (tmp_path / "page_00.sock").touch() + FakeMixer.instances.clear() + build_application(GraphOptions(inputs=("camera.mp4", "dmabuf://page_00"), output="program.ts", + dmabuf_socket_dir=str(tmp_path), input_color="hlg"), api=fake_api()) + sources = dict(FakeMixer.instances[-1].sources) + assert sources["source_0"]["color"] == "hlg" and not sources["source_0"]["packed_rgb"] + assert sources["source_1"]["color"] == "sdr" and sources["source_1"]["packed_rgb"] + FakeMixer.instances.clear() + build_application(GraphOptions(inputs=("camera.mp4",), output="program.ts"), api=fake_api()) + assert dict(FakeMixer.instances[-1].sources)["source_0"]["color"] is None # frame tags decide + with pytest.raises(ValueError, match="input-color"): + GraphOptions(inputs=("a.mp4",), output="o.ts", input_color="rec2020").validate() + + +@pytest.mark.parametrize("input_count", (32, 33)) +@pytest.mark.parametrize("color", ("", "sdr", "hlg", "pq")) +def test_cli_input_color_survives_routing_threshold(input_count, color): + app = build_application(GraphOptions(inputs=tuple(f"camera-{i}.mp4" for i in range(input_count)), + output="program.ts", input_color=color), api=fake_api()) + sources = app.mixer.routed_sources if app.routed_inputs else app.mixer.sources + assert app.routed_inputs == (input_count > 32) + assert all(params["color"] == (color or None) for _, params in sources) + + +@pytest.mark.parametrize("wipe_color", ("", "sdr")) +def test_cli_wipe_color_reaches_builder(wipe_color): + argv = ["--input", "camera.mp4", "--output", "program.ts", "--wipe-file", "wipe.mov"] + if wipe_color: + argv += ["--wipe-color", wipe_color] + app = build_application(parse_args(argv), api=fake_api()) + assert app.mixer.parameters["wipe_color"] == (wipe_color or None) + with pytest.raises(ValueError, match="HDR alpha"): + GraphOptions(inputs=("camera.mp4",), output="program.ts", wipe_color="hlg").validate() + + +def test_cli_wipe_color_overrides_config(tmp_path): + doc = {**CONFIG, "sources": CONFIG["sources"][:1], "scenes": CONFIG["scenes"][:1], + "initial_scene": "full", "wipe_color": "hlg"} + path = tmp_path / "mixer.json" + path.write_text(json.dumps(doc)) + app = build_application(GraphOptions(config=str(path), output="program.ts", wipe_color="sdr"), api=fake_api()) + assert app.mixer.parameters["wipe_color"] == "sdr" + + +def test_pq_renditions_carry_hdr10_static_metadata(tmp_path): + doc = {**CONFIG, "sources": CONFIG["sources"][:1], "scenes": CONFIG["scenes"][:1], "wipes": [], + "initial_scene": "full", + "canvas": {"width": 1920, "height": 1080, "fps": 60, "working_format": "p010le", "color": "hlg"}, + "renditions": [ + {"id": "pq", "target": "janus", "codec": "hevc_nvenc", "color": "pq", "max_fall": 300}, + {"id": "hlg", "target": "/rec/hlg.ts", "codec": "hevc_nvenc"}, + {"id": "pqfile", "target": "/rec/pq.ts", "codec": "hevc_nvenc", "color": "pq", "tonemap_peak": 40}]} + path = tmp_path / "pq.json" + path.write_text(json.dumps(doc)) + app = build_application(GraphOptions(config=str(path), janus_output=True), api=fake_api()) + nodes = {n.parameters["name"]: n.parameters for n in app.avp.nodes} + md = nodes["janus_encoder"]["hdr_metadata"] + assert md["primaries"][0] == [0.708, 0.292] and md["white_point"] == [0.3127, 0.3290] # BT.2020 / D65 + assert (md["max_luminance"], md["min_luminance"], md["max_cll"], md["max_fall"]) == (1000, 0.0001, 1000, 300) + assert nodes["janus_encoder"]["options"]["color_trc"] == "smpte2084" + assert "hdr_metadata" not in nodes["hlg_encoder"] # HLG signals nothing static + assert nodes["pqfile_encoder"]["hdr_metadata"]["max_luminance"] == 4000 + assert nodes["pqfile_encoder"]["hdr_metadata"]["max_fall"] == 1600 + with pytest.raises(mixer_config.ConfigError, match="max_fall"): + mixer_config.parse({**doc, "renditions": [{"id": "x", "codec": "hevc_nvenc", "color": "pq", "max_fall": 5000}]}) + with pytest.raises(mixer_config.ConfigError, match="wipe_color"): + mixer_config.parse({**CONFIG, "wipe_color": "rec2020"}) + + +def test_hdr_example_config_parses_and_builds(tmp_path, monkeypatch): + """The shipped HDR example must stay valid: P210 HLG canvas, v210 + video + browser sources, + HLG Janus, mobius SDR Janus and a PQ archive with HDR10 metadata.""" + path = Path(__file__).resolve().parents[1] / "config.example.hdr.json" + cfg = mixer_config.parse(json.loads(path.read_text())) + assert cfg.working_format == "p210le" and cfg.out_color == Color("hlg") + assert [r.id for r in cfg.renditions] == ["hdr", "sdr", "archive"] + (tmp_path / "page.sock").touch() + from pyplumber.mixer import dmabuf_inputs + monkeypatch.setattr(dmabuf_inputs, "rest_request", lambda *a, **k: {"windows": []}) + FakeMixer.instances.clear() + app = build_application(GraphOptions(config=str(path), janus_output=True, dmabuf_socket_dir=str(tmp_path)), + api=fake_api()) + nodes = {n.parameters.get("name"): n.parameters for n in app.avp.nodes} + assert nodes["unpack_0"]["sw_format"] == "p210le" + assert nodes["scale_hdr"]["graph"] == Color("hlg").setparams + ",scale_cuda=format=p010le" + assert ":tonemap=mobius:" in nodes["scale_sdr"]["graph"] and ":param=0.9" in nodes["scale_sdr"]["graph"] + assert nodes["archive_encoder"]["hdr_metadata"]["max_cll"] == 1000 + assert nodes["archive_encoder"]["options"]["color_trc"] == "smpte2084" + + +def test_generated_hdr_show_has_hdr_and_sdr_renditions(): + import make_config + import io + import contextlib + args = ["--canvas", "1920x1080", "--fps", "60", "--color", "hlg", "--working-format", "p210le", + "--sdr-port", "5004", "movie=/m/hdr.mp4", "clip=/m/hlg.mp4:hlg", "pat=/f/p.v210@1920x1080:hlg", + "bunny=/m/bunny.mp4:sdr", "page=https://example.org/a@1920x1080"] + out = io.StringIO() + with contextlib.redirect_stdout(out): + make_config.main(args) + doc = json.loads(out.getvalue()) + cfg = mixer_config.parse(doc) + assert cfg.working_format == "p210le" and cfg.out_color == Color("hlg") + assert doc["canvas"]["color"] == "hlg" + program, sdr = doc["renditions"] + assert (program["codec"], program["profile"]) == ("hevc_nvenc", "main10") and "tonemap" not in program + assert (sdr["codec"], sdr["port"], sdr["tonemap"], sdr["tonemap_param"]) == ("h264_nvenc", 5004, "mobius", 0.9) + kinds = {s["id"]: (s["kind"], s.get("color")) for s in doc["sources"]} + assert kinds == {"movie": ("video", None), "clip": ("video", "hlg"), "pat": ("v210", "hlg"), + "bunny": ("video", "sdr"), "page": ("browser", "sdr")} + assert next(s for s in doc["sources"] if s["id"] == "pat")["width"] == 1920 + with pytest.raises(SystemExit, match="v210 clips need"): + make_config.source_spec("raw=/f/p.v210@1920x1080") + with pytest.raises(SystemExit, match="--sdr-port needs"): + make_config.main(["--sdr-port", "5004", "clip=/m/c.mp4"]) + + +def test_canvas_latency_reaches_the_builder_unless_the_cli_overrides_it(tmp_path, monkeypatch): + from pyplumber.mixer import dmabuf_inputs + monkeypatch.setattr(dmabuf_inputs, "rest_request", lambda *a, **k: {"windows": []}) + (tmp_path / "page.sock").touch() + doc = {**CONFIG, "canvas": {**CONFIG["canvas"], "latency_ms": 50}} + assert mixer_config.parse(doc).latency_ms == 50.0 + assert mixer_config.parse(CONFIG).latency_ms is None + with pytest.raises(mixer_config.ConfigError, match="six frames"): + mixer_config.parse({**CONFIG, "canvas": {**CONFIG["canvas"], "fps": 60, "latency_ms": 100}}) + with pytest.raises(mixer_config.ConfigError, match="latency_ms"): + mixer_config.parse({**CONFIG, "canvas": {**CONFIG["canvas"], "latency_ms": "soon"}}) + path = tmp_path / "show.json" + path.write_text(json.dumps(doc)) + FakeMixer.instances.clear() + build_application(GraphOptions(config=str(path), janus_output=True, dmabuf_socket_dir=str(tmp_path)), + api=fake_api()) + assert FakeMixer.instances[-1].parameters["latency_ms"] == 50.0 + FakeMixer.instances.clear() + build_application(GraphOptions(config=str(path), janus_output=True, dmabuf_socket_dir=str(tmp_path), + mixer_latency_ms=20.0), api=fake_api()) + assert FakeMixer.instances[-1].parameters["latency_ms"] == 20.0 + + +def test_more_than_32_sources_is_rejected_at_load(): + sources = [{"id": f"s{i}", "kind": "video", "path": f"/m/{i}.mp4", "width": 16, "height": 16} for i in range(33)] + doc = {**CONFIG, "sources": sources, "scenes": [{"id": "s", "items": [{"source": "s0", "dst": {"x": 0, "y": 0, "w": 16, "h": 16}}]}]} + with pytest.raises(mixer_config.ConfigError, match="32 sources"): + mixer_config.parse(doc) + + +def test_mobius_knee_must_leave_shoulder_room(): + doc = {**CONFIG, "renditions": [{"id": "sdr", "codec": "h264_nvenc", "tonemap": "mobius", "tonemap_param": 1.0}]} + with pytest.raises(mixer_config.ConfigError, match="below 1.0"): + mixer_config.parse(doc) + + +@pytest.mark.parametrize("fmt", ["yuv420p", "yuv444p10le", "yuv422p10le"]) +def test_planar_working_formats_are_rejected_up_front(fmt): + # The compositor cannot promote 8-bit sources or draw the RGBA wipe onto planar canvases. + with pytest.raises(ValueError, match="working-format"): + GraphOptions(inputs=("a.mp4",), output="o.ts", working_format=fmt).validate() + with pytest.raises(mixer_config.ConfigError, match="working_format"): + mixer_config.parse({**CONFIG, "canvas": {**CONFIG["canvas"], "working_format": fmt}}) + + +def test_fractional_janus_rate_uses_one_second_gop(): + from pyplumber.mixer.janus import JanusVideoConfig, build_janus_output + avp = FakeAvp() + build_janus_output(avp, fake_api(), "program", JanusVideoConfig(), fps=60000, fps_den=1001, + width=1920, height=1080) + nodes = {n.parameters["name"]: n.parameters for n in avp.nodes} + assert nodes["janus_fps"]["fps"] == "60000/1001" + assert nodes["janus_encoder"]["options"]["g"] == 60 + + def test_control_section_carries_the_defaults_the_surfaces_start_from(): - from avpmixer import config as mc + from pyplumber.mixer import config as mc import copy doc = copy.deepcopy(CONFIG) del doc["control"] diff --git a/demos/mixer/tests/test_prewarm.py b/demos/mixer/tests/test_prewarm.py index f2efff4b..00b1d51f 100644 --- a/demos/mixer/tests/test_prewarm.py +++ b/demos/mixer/tests/test_prewarm.py @@ -18,10 +18,10 @@ def native_boundary(monkeypatch): before = set(sys.modules) monkeypatch.setitem(sys.modules, "_avplumber", SimpleNamespace(AVPlumber=object)) nodes = importlib.import_module("pyplumber.node") - builder = importlib.import_module("avpmixer").MixerGraphBuilder + builder = importlib.import_module("pyplumber.mixer").MixerGraphBuilder yield nodes, builder for name in set(sys.modules) - before: - if name == "avpmixer" or name.startswith(("avpmixer.", "pyplumber")): + if name == "pyplumber.mixer" or name.startswith(("pyplumber.mixer.", "pyplumber")): sys.modules.pop(name, None) @@ -189,7 +189,7 @@ def test_media_wipe_path_is_registered_without_starting_an_empty_clip(native_bou config = json.loads(init.split(" ", 2)[2]) assert config["wipe_group"] == "mixer_wipe" # With clips cached, the take arms the cache node and the decode chain sits - # in a group a take never starts (see avpmixer.clipcache). + # in a group a take never starts (see pyplumber.mixer.clipcache). assert config["wipe_input_node"] == "mixer_wipe_cache" assert engine.nodes["mixer_wipe_cache"]["group"] == "mixer_wipe" assert engine.nodes["mixer_wipe_input"]["group"] == "mixer_wipe_load" @@ -258,5 +258,5 @@ def test_existing_explicit_filter_sources_still_create_slot_filters(native_bound mixer.set_initial_scene("program") mixer.build() for slot in ("a", "b"): - assert engine.nodes[f"mixer_cs_camera_{slot}"]["graph"] == graph + assert engine.nodes[f"mixer_cs_camera_{slot}"]["graph"] == graph + ",scale_cuda=format=nv12" assert engine.nodes[f"mixer_comp_{slot}"]["src"] == [f"mixer_camera_scaled_{slot}"] diff --git a/demos/mixer/tui.py b/demos/mixer/tui.py index 15f21a85..93d25e43 100644 --- a/demos/mixer/tui.py +++ b/demos/mixer/tui.py @@ -14,9 +14,9 @@ from textual.scrollbar import ScrollBarRender from textual.widgets import Button, Footer, Header, Input, Label, Select, Static -from avpmixer.control import AvpConnection +from pyplumber.mixer.control import AvpConnection -from avpmixer.control import mixer_command, parse_mixer_status, parse_scene_list +from pyplumber.mixer.control import mixer_command, parse_mixer_status, parse_scene_list class SceneScrollBarRender(ScrollBarRender): diff --git a/demos/mixer/webui.py b/demos/mixer/webui.py index 94c19b76..3d64df81 100644 --- a/demos/mixer/webui.py +++ b/demos/mixer/webui.py @@ -22,7 +22,7 @@ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from pathlib import Path -from avpmixer.control import AvpConnection, mixer_command +from pyplumber.mixer.control import AvpConnection, mixer_command PAGE = Path(__file__).with_name("webui") / "index.html" TAKE_COMMANDS = ("cut", "fade", "wipe", "preview", "interrupt") diff --git a/demos/playlist/Dockerfile b/demos/playlist/Dockerfile index 03562528..e015680d 100644 --- a/demos/playlist/Dockerfile +++ b/demos/playlist/Dockerfile @@ -29,7 +29,7 @@ COPY server.py player.py playlist.py engine.py control.py regression.py requirem COPY test-media/generate.sh ./test-media/generate.sh COPY --from=fixtures /fixtures/*.mp4 ./test-media/ -RUN python3 -c 'from importlib.util import find_spec; from pathlib import Path; import textual; from server import FIXTURE_NAMES; root = Path("test-media"); paths = [root / name for name in FIXTURE_NAMES]; assert find_spec("pyplumber") is not None; assert find_spec("avpmixer") is not None; assert len(paths) == 5 and all(path.is_file() for path in paths), paths' +RUN python3 -c 'from importlib.util import find_spec; from pathlib import Path; import textual; from server import FIXTURE_NAMES; root = Path("test-media"); paths = [root / name for name in FIXTURE_NAMES]; assert find_spec("pyplumber") is not None; assert find_spec("pyplumber.mixer") is not None; assert len(paths) == 5 and all(path.is_file() for path in paths), paths' # Backend by default; run `player.py` in a second container or on the host to attach the TUI. ENTRYPOINT ["python3", "server.py"] diff --git a/demos/playlist/control.py b/demos/playlist/control.py index 220807d2..0868e66e 100644 --- a/demos/playlist/control.py +++ b/demos/playlist/control.py @@ -11,7 +11,7 @@ from dataclasses import dataclass from typing import Any, Dict, List, Optional -from avpmixer.control import AvpConnection +from pyplumber.mixer.control import AvpConnection from playlist import (Clip, ElementMode, PlaylistController, PlaylistMode, Transition, TransportState, now_ms) diff --git a/demos/playlist/docs/guide.md b/demos/playlist/docs/guide.md index 838a52c2..d452d120 100644 --- a/demos/playlist/docs/guide.md +++ b/demos/playlist/docs/guide.md @@ -59,7 +59,7 @@ element, and the same actions on keys: The same as the mixer demo: a Linux NVIDIA host with hardware decode and NVENC, `pyplumber` built with CUDA and NVCC, FFmpeg with CUDA decoding and -`h264_nvenc`, the `avpmixer` package on `PYTHONPATH`, and a video-only Janus +`h264_nvenc`, the `pyplumber.mixer` package on `PYTHONPATH`, and a video-only Janus Streaming mountpoint that accepts H.264 RTP (the shared Janus preview from the [demo setup guide](../../README.md) provides one on port 5004). Neural models and TensorRT are not needed. @@ -122,7 +122,7 @@ docker run --rm -it --network host --entrypoint python3 avplumber-playlist:local ``` Set `AVP_BASE_IMAGE=` to any image that provides -`pyplumber` and `avpmixer`. The image generates and validates the fixtures +`pyplumber` and `pyplumber.mixer`. The image generates and validates the fixtures while building. ## Verifying frame continuity @@ -168,7 +168,7 @@ transition changes. `engine.py` binds elements to sixteen fixed mixer sources (group `pl_item_`, fullscreen scene `item_`). The decode chain is -`avpmixer.inputs.build_input` with the replay demo's playback controls: a +`pyplumber.mixer.inputs.build_input` with the replay demo's playback controls: a pause team and pause node, a realtime sync team, and `h264_cuvid` with `flush_magic` so a seek lands on the exact frame. A parked element sits on its cue-in frame with the decoder resident; every chain loops between its cue diff --git a/demos/playlist/engine.py b/demos/playlist/engine.py index 49c0dd6e..4ce36440 100644 --- a/demos/playlist/engine.py +++ b/demos/playlist/engine.py @@ -2,7 +2,7 @@ """Mixer-backed playlist engine. Every playlist element owns one of sixteen fixed mixer sources. Its decode -chain (``avpmixer.inputs.build_input`` with a pause node and a realtime sync +chain (``pyplumber.mixer.inputs.build_input`` with a pause node and a realtime sync team, the replay demo's wiring) lives in group ``pl_item_`` and feeds scene ``item_``, a fullscreen layout on the native two-slot mixer. @@ -29,8 +29,8 @@ from pathlib import Path from typing import Dict, List, Optional, Tuple -from avpmixer.inputs import build_input -from avpmixer.janus import JanusVideoConfig, build_janus_output +from pyplumber.mixer.inputs import build_input +from pyplumber.mixer.janus import JanusVideoConfig, build_janus_output from playlist import SLOT_CAPACITY, BackendEvent, Clip, Transition, now_ms MIXER = "mixer" @@ -104,7 +104,7 @@ def load_avp_api(): ForceKeyFrame, InputRec, Mux, Output, Pause, Realtime, SpeedVideo, Split) from pyplumber.rtcp_feedback import RtcpFeedbackListener - from avpmixer import MixerGraphBuilder + from pyplumber.mixer import MixerGraphBuilder from types import SimpleNamespace return SimpleNamespace( AVPlumber=AVPlumber, MixerGraphBuilder=MixerGraphBuilder, diff --git a/demos/playlist/server.py b/demos/playlist/server.py index 28f8a241..0e3b0a6e 100755 --- a/demos/playlist/server.py +++ b/demos/playlist/server.py @@ -15,7 +15,7 @@ from pathlib import Path from typing import List -from avpmixer.janus import JanusVideoConfig +from pyplumber.mixer.janus import JanusVideoConfig from control import VERBS, apply_verb, now_ms, status_json from engine import PlaylistConfig, PlaylistEngine, load_avp_api from playlist import Clip, ElementMode, PlaylistController, PlaylistMode, Transition diff --git a/demos/playlist/tests/conftest.py b/demos/playlist/tests/conftest.py index 4bd1eae0..29bb3abd 100644 --- a/demos/playlist/tests/conftest.py +++ b/demos/playlist/tests/conftest.py @@ -3,4 +3,4 @@ HERE = os.path.dirname(os.path.abspath(__file__)) sys.path.insert(0, os.path.dirname(HERE)) # demos/playlist -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.dirname(HERE)))) # repo root: avpmixer +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.dirname(HERE)))) # repo root: pyplumber diff --git a/demos/playlist/tests/test_container.py b/demos/playlist/tests/test_container.py index ac688960..9fa8eeed 100644 --- a/demos/playlist/tests/test_container.py +++ b/demos/playlist/tests/test_container.py @@ -22,7 +22,7 @@ def test_runtime_uses_configurable_avplumber_base_and_packaged_dependencies(): assert "FROM ${AVP_BASE_IMAGE} AS runtime" in source assert "COPY --from=python-dependencies /opt/playlist-python" in source assert "ENV PYTHONPATH=/opt/playlist-python:${PYTHONPATH}" in source - assert 'find_spec("pyplumber")' in source and 'find_spec("avpmixer")' in source + assert 'find_spec("pyplumber")' in source and 'find_spec("pyplumber.mixer")' in source assert "import pyplumber" not in source assert 'ENTRYPOINT ["python3", "server.py"]' in source for module in ("server.py", "player.py", "playlist.py", "engine.py", "control.py"): diff --git a/deps/ffmpeg/8/0008-avfilter-transition-cuda-10bit.patch b/deps/ffmpeg/8/0008-avfilter-transition-cuda-10bit.patch new file mode 100644 index 00000000..0d2c1a62 --- /dev/null +++ b/deps/ffmpeg/8/0008-avfilter-transition-cuda-10bit.patch @@ -0,0 +1,235 @@ +From 9182da7189636d07c12b4ab89f8aa07d5f85c638 Mon Sep 17 00:00:00 2001 +From: avplumber patches +Date: Thu, 17 Sep 2026 13:16:00 +0200 +Subject: [PATCH 8/9] avfilter/transition_cuda: 10-bit and 4:2:2/4:4:4 layouts + in a word-sample path + +Add YUV420P10/422P10/444P10, P010 and P210 (plus 8-bit YUV422P/YUV444P) +to transition_cuda. Formats above 8 bits run a 16-bit word kernel whose +lanes group interleaved Cb/Cr so a wipe boundary moves per chroma pixel. +Both inputs must arrive in the identical layout and depth; no alpha- +carrying CUDA formats exist for these yet. +--- + libavfilter/vf_transition_cuda.c | 106 ++++++++++++++++++++++++++++++ + libavfilter/vf_transition_cuda.cu | 41 ++++++++++++ + 2 files changed, 147 insertions(+) + +diff --git a/libavfilter/vf_transition_cuda.c b/libavfilter/vf_transition_cuda.c +index 0c055a7..77132f8 100644 +--- a/libavfilter/vf_transition_cuda.c ++++ b/libavfilter/vf_transition_cuda.c +@@ -50,6 +50,13 @@ + static const enum AVPixelFormat supported_main_formats[] = { + AV_PIX_FMT_NV12, + AV_PIX_FMT_YUV420P, ++ AV_PIX_FMT_YUV422P, ++ AV_PIX_FMT_YUV444P, ++ AV_PIX_FMT_YUV420P10, ++ AV_PIX_FMT_YUV422P10, ++ AV_PIX_FMT_YUV444P10, ++ AV_PIX_FMT_P010, ++ AV_PIX_FMT_P210, + AV_PIX_FMT_NONE, + }; + +@@ -57,6 +64,13 @@ static const enum AVPixelFormat supported_overlay_formats[] = { + AV_PIX_FMT_NV12, + AV_PIX_FMT_YUV420P, + AV_PIX_FMT_YUVA420P, ++ AV_PIX_FMT_YUV422P, ++ AV_PIX_FMT_YUV444P, ++ AV_PIX_FMT_YUV420P10, ++ AV_PIX_FMT_YUV422P10, ++ AV_PIX_FMT_YUV444P10, ++ AV_PIX_FMT_P010, ++ AV_PIX_FMT_P210, + AV_PIX_FMT_NONE, + }; + +@@ -114,6 +128,7 @@ typedef struct TransitionCUDAContext { + CUcontext cu_ctx; + CUmodule cu_module; + CUfunction cu_func; ++ CUfunction cu_func_word; + CUstream cu_stream; + + FFFrameSync fs; +@@ -184,6 +199,16 @@ static int formats_match(const enum AVPixelFormat format_main, const enum AVPixe + case AV_PIX_FMT_YUV420P: + return format_overlay == AV_PIX_FMT_YUV420P || + format_overlay == AV_PIX_FMT_YUVA420P; ++ case AV_PIX_FMT_YUV422P: ++ case AV_PIX_FMT_YUV444P: ++ case AV_PIX_FMT_YUV420P10: ++ case AV_PIX_FMT_YUV422P10: ++ case AV_PIX_FMT_YUV444P10: ++ case AV_PIX_FMT_P010: ++ case AV_PIX_FMT_P210: ++ /* No alpha-carrying CUDA frame formats exist for these yet: both ++ * inputs must arrive in the identical layout and depth. */ ++ return format_overlay == format_main; + default: + return 0; + } +@@ -222,6 +247,40 @@ static int transition_cuda_call_kernel( + 0, ctx->cu_stream, kernel_args, NULL)); + } + ++/** ++ * Call the word transition kernel for one 16-bit-stored plane. ++ * Widths, coordinates and linesizes count 16-bit elements; `lanes` groups ++ * interleaved Cb/Cr so wipes move per chroma pixel. No alpha plane: word ++ * formats have no alpha-capable CUDA frame layout yet. ++ */ ++static int transition_cuda_call_kernel_word( ++ TransitionCUDAContext *ctx, ++ uint16_t* main_data, int main_linesize, ++ int main_width, int main_height, ++ uint16_t* overlay_data, int overlay_linesize, ++ int overlay_width, int overlay_height, ++ int lanes, int shift, ++ float alpha_coef, ++ int transition_mode) { ++ ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ ++ void* kernel_args[] = { ++ &main_data, &main_linesize, ++ &overlay_data, &overlay_linesize, ++ &overlay_width, &overlay_height, ++ &lanes, &shift, ++ &alpha_coef, ++ &transition_mode ++ }; ++ ++ return CHECK_CU(cu->cuLaunchKernel( ++ ctx->cu_func_word, ++ DIV_UP(main_width, BLOCK_X), DIV_UP(main_height, BLOCK_Y), 1, ++ BLOCK_X, BLOCK_Y, 1, ++ 0, ctx->cu_stream, kernel_args, NULL)); ++} ++ + /** + * Perform blend overlay picture over main picture + */ +@@ -285,6 +344,33 @@ static int transition_cuda_blend(FFFrameSync *fs) + ctx->alpha_coef); + } + ++ // Word-stored formats (P010/P210, planar 10-bit) arrive as identical ++ // main/overlay layouts with no alpha plane; every plane blends through ++ // the sample kernel with its own geometry. ++ ++ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(ctx->in_format_main); ++ if (desc->comp[0].depth > 8) { ++ const int shift = desc->comp[0].shift; ++ for (int p = 0; p < av_pix_fmt_count_planes(ctx->in_format_main); p++) { ++ const int sub_x = (p == 1 || p == 2) ? desc->log2_chroma_w : 0; ++ const int sub_y = (p == 1 || p == 2) ? desc->log2_chroma_h : 0; ++ int lanes = 1; ++ for (int c = 0; c < desc->nb_components; c++) ++ if (desc->comp[c].plane == p && desc->comp[c].step / 2 > lanes) ++ lanes = desc->comp[c].step / 2; ++ transition_cuda_call_kernel_word(ctx, ++ (uint16_t*)input_main->data[p], input_main->linesize[p] / 2, ++ AV_CEIL_RSHIFT(input_main->width, sub_x) * lanes, ++ AV_CEIL_RSHIFT(input_main->height, sub_y), ++ (uint16_t*)input_overlay->data[p], input_overlay->linesize[p] / 2, ++ AV_CEIL_RSHIFT(input_overlay->width, sub_x) * lanes, ++ AV_CEIL_RSHIFT(input_overlay->height, sub_y), ++ lanes, shift, ctx->alpha_coef, ctx->transition_mode); ++ } ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return ff_filter_frame(outlink, input_main); ++ } ++ + // overlay first plane + + transition_cuda_call_kernel(ctx, +@@ -325,6 +411,20 @@ static int transition_cuda_blend(FFFrameSync *fs) + input_overlay->data[3], input_overlay->linesize[3], 2, 2, + ctx->alpha_coef, ctx->transition_mode); + break; ++ case AV_PIX_FMT_YUV422P: ++ case AV_PIX_FMT_YUV444P: ++ // 8-bit planar chroma at its own subsampling, no alpha plane. ++ for (int p = 1; p <= 2; p++) ++ transition_cuda_call_kernel(ctx, ++ input_main->data[p], input_main->linesize[p], ++ AV_CEIL_RSHIFT(input_main->width, desc->log2_chroma_w), ++ AV_CEIL_RSHIFT(input_main->height, desc->log2_chroma_h), ++ input_overlay->data[p], input_overlay->linesize[p], ++ AV_CEIL_RSHIFT(input_overlay->width, desc->log2_chroma_w), ++ AV_CEIL_RSHIFT(input_overlay->height, desc->log2_chroma_h), ++ 0, 0, 0, 0, ++ ctx->alpha_coef, ctx->transition_mode); ++ break; + default: + av_log(ctx, AV_LOG_ERROR, "Passed unsupported overlay pixel format\n"); + av_frame_free(&input_main); +@@ -504,6 +604,12 @@ static int transition_cuda_config_output(AVFilterLink *outlink) + return err; + } + ++ err = CHECK_CU(cu->cuModuleGetFunction(&ctx->cu_func_word, ctx->cu_module, "Transition_Cuda_Word")); ++ if (err < 0) { ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return err; ++ } ++ + CHECK_CU(cu->cuCtxPopCurrent(&dummy)); + + // init dual input +diff --git a/libavfilter/vf_transition_cuda.cu b/libavfilter/vf_transition_cuda.cu +index c688e66..86a2eaf 100644 +--- a/libavfilter/vf_transition_cuda.cu ++++ b/libavfilter/vf_transition_cuda.cu +@@ -53,4 +53,45 @@ __global__ void Transition_Cuda( + main[x + y*main_linesize] = alpha * overlay[x + y*overlay_linesize] + (1.f - alpha) * main[x + y*main_linesize]; + } + ++/* ++ * Word-stored planes (P210/P010 use shift 6, planar 10-bit shift 0): logical ++ * samples blend like the byte kernel but store round-to-nearest, keeping the ++ * padding bits zero. Interleaved Cb/Cr passes lanes=2 so a wipe boundary ++ * moves per chroma pixel and can never split one pixel's U from its V. ++ * Coordinates and linesizes count 16-bit elements. ++ */ ++__global__ void Transition_Cuda_Word( ++ unsigned short* main, int main_linesize, ++ unsigned short* overlay, int overlay_linesize, ++ int overlay_w, int overlay_h, ++ int lanes, int shift, ++ float alpha_coef, int transition_mode) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ ++ if (x >= overlay_w || ++ y >= overlay_h) { ++ return; ++ } ++ ++ float alpha = alpha_coef; ++ int px = x / lanes; ++ int pw = overlay_w / lanes; ++ if (transition_mode == 1) { ++ alpha = (px + 0.5f) / pw <= alpha_coef; ++ } else if (transition_mode == 2) { ++ alpha = (px + 0.5f) / pw >= 1.f - alpha_coef; ++ } else if (transition_mode == 3) { ++ alpha = (y + 0.5f) / overlay_h <= alpha_coef; ++ } else if (transition_mode == 4) { ++ alpha = (y + 0.5f) / overlay_h >= 1.f - alpha_coef; ++ } ++ ++ float m = float(main[x + y*main_linesize] >> shift); ++ float o = float(overlay[x + y*overlay_linesize] >> shift); ++ main[x + y*main_linesize] = ++ (unsigned short)((unsigned short)(alpha * o + (1.f - alpha) * m + 0.5f) << shift); ++} ++ + } +-- +2.55.0 + diff --git a/deps/ffmpeg/8/0009-avfilter-tonemap-cuda.patch b/deps/ffmpeg/8/0009-avfilter-tonemap-cuda.patch new file mode 100644 index 00000000..c70ab801 --- /dev/null +++ b/deps/ffmpeg/8/0009-avfilter-tonemap-cuda.patch @@ -0,0 +1,1000 @@ +From 10debd99c4f1cdea237221e44c25dc1506ddf504 Mon Sep 17 00:00:00 2001 +From: avplumber patches +Date: Thu, 17 Sep 2026 13:16:00 +0200 +Subject: [PATCH] avfilter: add tonemap_cuda, SDR/HLG/PQ conversion on CUDA + NV12/P010 frames + +A CUDA filter converting limited-range BT.709 SDR and BT.2020 HLG/PQ in +both directions (transfer_in=auto|sdr|hlg|pq, transfer_out=sdr|hlg|pq): +display-light conversion with sdr_white (default 203 nits) and hdr_peak +(default 1000 nits); HDR->SDR tone mapping with the tonemap_opencl +operators (clip, reinhard, hable, mobius, gamma, linear), `param` knee and +`desat`; `format=nv12|p010le` selects the output storage. transfer_in=auto +resolves the contract from every frame's tags, treats untagged frames as +BT.709 SDR and rejects contradictory tags; identity frames are forwarded +without a device copy, stamped with the resolved contract. Colour- +dependent side data is dropped on conversion. The math is ported from +libavfilter/opencl/tonemap.cl and colorspace_common.cl. +--- + configure | 1 + + doc/filters.texi | 61 ++++ + libavfilter/Makefile | 2 + + libavfilter/allfilters.c | 1 + + libavfilter/vf_tonemap_cuda.c | 544 +++++++++++++++++++++++++++++++++ + libavfilter/vf_tonemap_cuda.cu | 304 ++++++++++++++++++ + 6 files changed, 913 insertions(+) + create mode 100644 libavfilter/vf_tonemap_cuda.c + create mode 100644 libavfilter/vf_tonemap_cuda.cu + +diff --git a/configure b/configure +index f086594..82cdbd2 100755 +--- a/configure ++++ b/configure +@@ -3526,6 +3526,7 @@ scale_cuda_filter_deps="ffnvcodec" + scale_cuda_filter_deps_any="cuda_nvcc cuda_llvm" + thumbnail_cuda_filter_deps="ffnvcodec" + thumbnail_cuda_filter_deps_any="cuda_nvcc cuda_llvm" ++tonemap_cuda_filter_deps="ffnvcodec cuda_nvcc" + transpose_npp_filter_deps="ffnvcodec libnpp" + overlay_cuda_filter_deps="ffnvcodec" + overlay_cuda_filter_deps_any="cuda_nvcc cuda_llvm" +diff --git a/doc/filters.texi b/doc/filters.texi +index 5ae2fc5..82baec9 100644 +--- a/doc/filters.texi ++++ b/doc/filters.texi +@@ -27409,6 +27409,67 @@ scale_cuda=passthrough=0 + @end example + @end itemize + ++@subsection tonemap_cuda ++ ++Convert limited-range CUDA semiplanar frames (NV12/P010 4:2:0, NV16/P210 ++4:2:2) between SDR BT.709, HLG BT.2020 and PQ BT.2020. Width and height must ++be even. By default the output keeps the input chroma subsampling, 8 bits for ++SDR and 10 bits for HDR; @option{format} selects @code{nv12}, @code{p010le}, ++@code{nv16} or @code{p210le} explicitly and resamples 4:2:2 and 4:2:0 in the ++same pass. Equal input and output transfers, depth and subsampling forward ++the original frames and metadata without allocating or converting pixels. ++ ++@table @option ++@item transfer_in ++@item transfer_out ++Set both options to @code{sdr}, @code{hlg} or @code{pq}. The choice includes ++the gamut: BT.709 for SDR, BT.2020 non-constant-luminance YCbCr for HDR. ++Explicit choices also define the interpretation of untagged input. Full-range ++input must first be converted to limited range (e.g. with colorspace_cuda). ++ ++SDR uses the BT.1886 display response with ideal black (gamma 2.4). SDR-to-HDR ++conversion follows display-light mapping: decode SDR, convert primaries, ++scale reference white, then encode PQ or the inverse HLG EOTF. It preserves ++the SDR appearance without expanding highlights. HLG/PQ conversion preserves ++display luminance, clipping values outside the destination signal range. ++ ++@item sdr_white ++SDR reference white in cd/m2. Default 203. Range 1 to 10000; must not exceed ++@option{hdr_peak}. This sets SDR's brightness inside an HDR programme. ++ ++@item hdr_peak ++HLG nominal display peak and HDR-to-SDR tone-mapping input peak in cd/m2. ++Default 1000, range 100 to 10000. PQ values always represent absolute light; ++this option does not rescale PQ. For HLG, the system gamma is derived from ++this peak, with a minimum gamma of 1.0. ++ ++@item tonemap ++HDR-to-SDR operator: @code{none}, @code{linear}, @code{gamma}, @code{clip}, ++@code{reinhard}, @code{hable} (default) or @code{mobius}. Used only when ++converting HDR to SDR. @code{none:desat=0} preserves display light up to SDR ++white, then clips. @option{param} overrides the operator parameter; @option{desat} ++controls highlight desaturation (default 0.5; 0 disables it). ++ ++@end table ++ ++Converted frames carry the output color tags; obsolete mastering, content ++light and dynamic HDR metadata is removed. Pass-through preserves all metadata. ++ ++For an SDR source entering an HLG P210 mixer, use this GPU-only source chain: ++@example ++tonemap_cuda=transfer_in=sdr:transfer_out=hlg:sdr_white=203:hdr_peak=1000,scale_cuda=format=p210le ++@end example ++ ++To convert HLG to PQ without tone mapping: ++@example ++tonemap_cuda=transfer_in=hlg:transfer_out=pq:hdr_peak=1000 ++@end example ++ ++To produce SDR from PQ using the Hable operator: ++@example ++tonemap_cuda=transfer_in=pq:transfer_out=sdr:tonemap=hable:sdr_white=203:hdr_peak=1000 ++@end example ++ + @subsection thumbnail_cuda + + Select the most representative frame in a given sequence of consecutive frames using CUDA. +diff --git a/libavfilter/Makefile b/libavfilter/Makefile +index 2fae985..586d43e 100644 +--- a/libavfilter/Makefile ++++ b/libavfilter/Makefile +@@ -547,6 +547,8 @@ OBJS-$(CONFIG_TMEDIAN_FILTER) += vf_xmedian.o framesync.o + OBJS-$(CONFIG_TMIDEQUALIZER_FILTER) += vf_tmidequalizer.o + OBJS-$(CONFIG_TMIX_FILTER) += vf_mix.o framesync.o + OBJS-$(CONFIG_TONEMAP_FILTER) += vf_tonemap.o ++OBJS-$(CONFIG_TONEMAP_CUDA_FILTER) += vf_tonemap_cuda.o vf_tonemap_cuda.ptx.o \ ++ cuda/load_helper.o + OBJS-$(CONFIG_TONEMAP_OPENCL_FILTER) += vf_tonemap_opencl.o opencl.o \ + opencl/tonemap.o opencl/colorspace_common.o + OBJS-$(CONFIG_TONEMAP_VAAPI_FILTER) += vf_tonemap_vaapi.o vaapi_vpp.o +diff --git a/libavfilter/allfilters.c b/libavfilter/allfilters.c +index 01c75b2..64b5cd7 100644 +--- a/libavfilter/allfilters.c ++++ b/libavfilter/allfilters.c +@@ -512,6 +512,7 @@ extern const FFFilter ff_vf_tmedian; + extern const FFFilter ff_vf_tmidequalizer; + extern const FFFilter ff_vf_tmix; + extern const FFFilter ff_vf_tonemap; ++extern const FFFilter ff_vf_tonemap_cuda; + extern const FFFilter ff_vf_tonemap_opencl; + extern const FFFilter ff_vf_tonemap_vaapi; + extern const FFFilter ff_vf_tpad; +diff --git a/libavfilter/vf_tonemap_cuda.c b/libavfilter/vf_tonemap_cuda.c +new file mode 100644 +index 0000000..0c58775 +--- /dev/null ++++ b/libavfilter/vf_tonemap_cuda.c +@@ -0,0 +1,544 @@ ++/* ++ * This file is part of FFmpeg. ++ * ++ * FFmpeg is free software; you can redistribute it and/or ++ * modify it under the terms of the GNU Lesser General Public ++ * License as published by the Free Software Foundation; either ++ * version 2.1 of the License, or (at your option) any later version. ++ * ++ * FFmpeg is distributed in the hope that it will be useful, ++ * but WITHOUT ANY WARRANTY; without even the implied warranty of ++ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ * Lesser General Public License for more details. ++ * ++ * You should have received a copy of the GNU Lesser General Public ++ * License along with FFmpeg; if not, write to the Free Software ++ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA ++ */ ++ ++/** ++ * @file ++ * SDR BT.709 / HDR BT.2020 HLG and PQ conversion on CUDA semiplanar frames ++ * (NV12/P010 4:2:0, NV16/P210 4:2:2), with chroma resampling between them. ++ */ ++ ++#include ++#include ++ ++#include "libavutil/common.h" ++#include "libavutil/cuda_check.h" ++#include "libavutil/hwcontext.h" ++#include "libavutil/hwcontext_cuda_internal.h" ++#include "libavutil/internal.h" ++#include "libavutil/opt.h" ++#include "libavutil/pixdesc.h" ++ ++#include "avfilter.h" ++#include "filters.h" ++ ++#include "cuda/load_helper.h" ++ ++#define DIV_UP(a, b) (((a) + (b) - 1) / (b)) ++#define BLOCK_X 16 ++#define BLOCK_Y 16 ++ ++#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, s->hwctx->internal->cuda_dl, x) ++ ++enum TonemapAlgorithm { ++ TONEMAP_NONE, ++ TONEMAP_LINEAR, ++ TONEMAP_GAMMA, ++ TONEMAP_CLIP, ++ TONEMAP_REINHARD, ++ TONEMAP_HABLE, ++ TONEMAP_MOBIUS, ++ TONEMAP_MAX, ++}; ++ ++enum TonemapTransfer { ++ TONEMAP_TRANSFER_HLG, ++ TONEMAP_TRANSFER_PQ, ++ TONEMAP_TRANSFER_SDR, ++ TONEMAP_TRANSFER_AUTO, ++}; ++ ++/* Maps the public TonemapAlgorithm enum onto the operator codes understood by ++ * the Tonemap_Cuda kernel (0 direct, 1 linear, 2 clip, 3 reinhard, 4 hable, ++ * 5 mobius, 6 gamma). */ ++static const int tonemap_op_map[TONEMAP_MAX] = { ++ [TONEMAP_NONE] = 0, ++ [TONEMAP_LINEAR] = 1, ++ [TONEMAP_GAMMA] = 6, ++ [TONEMAP_CLIP] = 2, ++ [TONEMAP_REINHARD] = 3, ++ [TONEMAP_HABLE] = 4, ++ [TONEMAP_MOBIUS] = 5, ++}; ++ ++typedef struct TonemapCUDAContext { ++ const AVClass *class; ++ ++ AVCUDADeviceContext *hwctx; ++ AVBufferRef *frames_ctx; ++ AVFrame *own_frame; ++ AVFrame *tmp_frame; ++ ++ CUstream cu_stream; ++ CUmodule cu_module; ++ CUfunction cu_func; ++ ++ /* enum TonemapAlgorithm */ ++ int tonemap; ++ /* enum TonemapTransfer */ ++ int transfer_in; ++ int transfer_out; ++ int passthrough; ++ int current_transfer; ++ int warned_untagged; ++ int output_format; ++ int input_depth; ++ int output_depth; ++ int input_422; ++ int output_422; ++ double sdr_white; ++ double hdr_peak; ++ double param; ++ double desat; ++} TonemapCUDAContext; ++ ++static av_cold int tonemap_cuda_init(AVFilterContext *ctx) ++{ ++ TonemapCUDAContext *s = ctx->priv; ++ ++ if (s->transfer_in < 0 || s->transfer_out < 0 || s->transfer_out == TONEMAP_TRANSFER_AUTO) { ++ av_log(ctx, AV_LOG_ERROR, "Set both transfer_in and transfer_out (transfer_out cannot be auto)\n"); ++ return AVERROR(EINVAL); ++ } ++ ++ s->own_frame = av_frame_alloc(); ++ if (!s->own_frame) ++ return AVERROR(ENOMEM); ++ ++ s->tmp_frame = av_frame_alloc(); ++ if (!s->tmp_frame) ++ return AVERROR(ENOMEM); ++ ++ return 0; ++} ++ ++static av_cold void tonemap_cuda_uninit(AVFilterContext *ctx) ++{ ++ TonemapCUDAContext *s = ctx->priv; ++ ++ if (s->hwctx && s->cu_module) { ++ CudaFunctions *cu = s->hwctx->internal->cuda_dl; ++ CUcontext dummy; ++ ++ CHECK_CU(cu->cuCtxPushCurrent(s->hwctx->cuda_ctx)); ++ CHECK_CU(cu->cuModuleUnload(s->cu_module)); ++ s->cu_module = NULL; ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ } ++ ++ av_frame_free(&s->own_frame); ++ av_buffer_unref(&s->frames_ctx); ++ av_frame_free(&s->tmp_frame); ++} ++ ++static av_cold int init_hwframe_ctx(TonemapCUDAContext *s, AVBufferRef *device_ctx, ++ int width, int height) ++{ ++ AVBufferRef *out_ref = NULL; ++ AVHWFramesContext *out_ctx; ++ int ret; ++ ++ out_ref = av_hwframe_ctx_alloc(device_ctx); ++ if (!out_ref) ++ return AVERROR(ENOMEM); ++ ++ out_ctx = (AVHWFramesContext*)out_ref->data; ++ ++ out_ctx->format = AV_PIX_FMT_CUDA; ++ out_ctx->sw_format = s->output_depth == 10 ? (s->output_422 ? AV_PIX_FMT_P210 : AV_PIX_FMT_P010) ++ : (s->output_422 ? AV_PIX_FMT_NV16 : AV_PIX_FMT_NV12); ++ out_ctx->width = FFALIGN(width, 32); ++ out_ctx->height = FFALIGN(height, 32); ++ ++ ret = av_hwframe_ctx_init(out_ref); ++ if (ret < 0) ++ goto fail; ++ ++ av_frame_unref(s->own_frame); ++ ret = av_hwframe_get_buffer(out_ref, s->own_frame, 0); ++ if (ret < 0) ++ goto fail; ++ ++ s->own_frame->width = width; ++ s->own_frame->height = height; ++ ++ av_buffer_unref(&s->frames_ctx); ++ s->frames_ctx = out_ref; ++ ++ return 0; ++fail: ++ av_buffer_unref(&out_ref); ++ return ret; ++} ++ ++static av_cold int tonemap_cuda_load_functions(AVFilterContext *ctx) ++{ ++ TonemapCUDAContext *s = ctx->priv; ++ CUcontext dummy, cuda_ctx = s->hwctx->cuda_ctx; ++ CudaFunctions *cu = s->hwctx->internal->cuda_dl; ++ int ret; ++ ++ extern const unsigned char ff_vf_tonemap_cuda_ptx_data[]; ++ extern const unsigned int ff_vf_tonemap_cuda_ptx_len; ++ ++ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); ++ if (ret < 0) ++ return ret; ++ ++ ret = ff_cuda_load_module(ctx, s->hwctx, &s->cu_module, ++ ff_vf_tonemap_cuda_ptx_data, ++ ff_vf_tonemap_cuda_ptx_len); ++ if (ret < 0) ++ goto fail; ++ ++ ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func, s->cu_module, "Tonemap_Cuda")); ++ if (ret < 0) ++ goto fail; ++ ++fail: ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return ret; ++} ++ ++static av_cold int tonemap_cuda_config_output(AVFilterLink *outlink) ++{ ++ AVFilterContext *ctx = outlink->src; ++ AVFilterLink *inlink = ctx->inputs[0]; ++ FilterLink *inl = ff_filter_link(inlink); ++ FilterLink *outl = ff_filter_link(outlink); ++ TonemapCUDAContext *s = ctx->priv; ++ AVHWFramesContext *in_frames_ctx; ++ int ret; ++ ++ if (!inl->hw_frames_ctx) { ++ av_log(ctx, AV_LOG_ERROR, "No hw context provided on input\n"); ++ return AVERROR(EINVAL); ++ } ++ ++ in_frames_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; ++ ++ if (in_frames_ctx->sw_format != AV_PIX_FMT_P010 && in_frames_ctx->sw_format != AV_PIX_FMT_NV12 && ++ in_frames_ctx->sw_format != AV_PIX_FMT_P210 && in_frames_ctx->sw_format != AV_PIX_FMT_NV16) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported input format: %s (NV12, P010, NV16 or P210 required)\n", ++ av_get_pix_fmt_name(in_frames_ctx->sw_format)); ++ return AVERROR(ENOSYS); ++ } ++ if ((inlink->w & 1) || (inlink->h & 1)) { ++ av_log(ctx, AV_LOG_ERROR, "CUDA 4:2:0 conversion requires even dimensions\n"); ++ return AVERROR(EINVAL); ++ } ++ if (s->sdr_white > s->hdr_peak) { ++ av_log(ctx, AV_LOG_ERROR, "sdr_white must not exceed hdr_peak\n"); ++ return AVERROR(EINVAL); ++ } ++ s->input_depth = in_frames_ctx->sw_format == AV_PIX_FMT_P010 || in_frames_ctx->sw_format == AV_PIX_FMT_P210 ? 10 : 8; ++ s->input_422 = in_frames_ctx->sw_format == AV_PIX_FMT_NV16 || in_frames_ctx->sw_format == AV_PIX_FMT_P210; ++ /* Default output: depth from the transfer, chroma subsampling from the input. */ ++ s->output_depth = s->transfer_out == TONEMAP_TRANSFER_SDR ? 8 : 10; ++ s->output_422 = s->input_422; ++ if (s->output_format != AV_PIX_FMT_NONE) { ++ const int ten_bit = s->output_format == AV_PIX_FMT_P010 || s->output_format == AV_PIX_FMT_P210; ++ const int eight_bit = s->output_format == AV_PIX_FMT_NV12 || s->output_format == AV_PIX_FMT_NV16; ++ if (!(ten_bit || eight_bit) || (s->transfer_out != TONEMAP_TRANSFER_SDR && !ten_bit)) { ++ av_log(ctx, AV_LOG_ERROR, "format must be nv12, p010le, nv16 or p210le; HDR requires 10 bits\n"); ++ return AVERROR(EINVAL); ++ } ++ s->output_depth = ten_bit ? 10 : 8; ++ s->output_422 = s->output_format == AV_PIX_FMT_NV16 || s->output_format == AV_PIX_FMT_P210; ++ } ++ /* Same transfer with no explicit format: forward frames at their own storage. */ ++ s->passthrough = s->transfer_in == s->transfer_out && ++ (s->output_format == AV_PIX_FMT_NONE || ++ (s->input_depth == s->output_depth && s->input_422 == s->output_422)); ++ ++ /* Resolve the per-operator default tone mapping parameter. */ ++ if (isnan(s->param)) { ++ switch (s->tonemap) { ++ case TONEMAP_REINHARD: ++ case TONEMAP_MOBIUS: ++ s->param = 0.3; ++ break; ++ case TONEMAP_GAMMA: ++ s->param = 1.8; ++ break; ++ default: ++ s->param = 1.0; ++ break; ++ } ++ } ++ ++ s->hwctx = in_frames_ctx->device_ctx->hwctx; ++ s->cu_stream = s->hwctx->stream; ++ ++ outlink->w = inlink->w; ++ outlink->h = inlink->h; ++ outlink->sample_aspect_ratio = inlink->sample_aspect_ratio; ++ ++ if (s->passthrough) { ++ outl->hw_frames_ctx = av_buffer_ref(inl->hw_frames_ctx); ++ return outl->hw_frames_ctx ? 0 : AVERROR(ENOMEM); ++ } ++ ++ ret = init_hwframe_ctx(s, in_frames_ctx->device_ref, inlink->w, inlink->h); ++ if (ret < 0) ++ return ret; ++ ++ outl->hw_frames_ctx = av_buffer_ref(s->frames_ctx); ++ if (!outl->hw_frames_ctx) ++ return AVERROR(ENOMEM); ++ ++ ret = tonemap_cuda_load_functions(ctx); ++ if (ret < 0) ++ return ret; ++ ++ return 0; ++} ++ ++static int tonemap_cuda_conv(AVFilterContext *ctx, AVFrame *out, AVFrame *in) ++{ ++ TonemapCUDAContext *s = ctx->priv; ++ AVFilterLink *outlink = ctx->outputs[0]; ++ CudaFunctions *cu = s->hwctx->internal->cuda_dl; ++ int ret; ++ ++ int width = in->width; ++ int height = in->height; ++ int cw = width / 2; ++ int ch = height / 2; ++ ++ int transfer = s->current_transfer; ++ int transfer_out = s->transfer_out; ++ int input_depth = s->input_depth; ++ int output_depth = s->output_depth; ++ int input_422 = s->input_422; ++ int output_422 = s->output_422; ++ float sdr_white = s->sdr_white; ++ float hdr_peak = s->hdr_peak; ++ int op = tonemap_op_map[s->tonemap]; ++ float param = s->param; ++ float desat = s->desat; ++ ++ void *args[] = { ++ &in->data[0], &in->linesize[0], ++ &in->data[1], &in->linesize[1], ++ &s->own_frame->data[0], &s->own_frame->linesize[0], ++ &s->own_frame->data[1], &s->own_frame->linesize[1], ++ &width, &height, ++ &transfer, &op, ¶m, &desat, ++ &transfer_out, &input_depth, &output_depth, &input_422, &output_422, &sdr_white, &hdr_peak ++ }; ++ ++ ret = CHECK_CU(cu->cuLaunchKernel(s->cu_func, ++ DIV_UP(cw, BLOCK_X), DIV_UP(ch, BLOCK_Y), 1, ++ BLOCK_X, BLOCK_Y, 1, ++ 0, s->cu_stream, args, NULL)); ++ if (ret < 0) ++ return ret; ++ ++ ret = av_hwframe_get_buffer(s->own_frame->hw_frames_ctx, s->tmp_frame, 0); ++ if (ret < 0) ++ return ret; ++ ++ av_frame_move_ref(out, s->own_frame); ++ av_frame_move_ref(s->own_frame, s->tmp_frame); ++ ++ s->own_frame->width = outlink->w; ++ s->own_frame->height = outlink->h; ++ ++ ret = av_frame_copy_props(out, in); ++ if (ret < 0) ++ return ret; ++ ++ out->color_trc = s->transfer_out == TONEMAP_TRANSFER_SDR ? AVCOL_TRC_BT709 : ++ s->transfer_out == TONEMAP_TRANSFER_HLG ? AVCOL_TRC_ARIB_STD_B67 : ++ AVCOL_TRC_SMPTE2084; ++ out->color_primaries = s->transfer_out == TONEMAP_TRANSFER_SDR ? AVCOL_PRI_BT709 : AVCOL_PRI_BT2020; ++ out->colorspace = s->transfer_out == TONEMAP_TRANSFER_SDR ? AVCOL_SPC_BT709 : AVCOL_SPC_BT2020_NCL; ++ out->color_range = AVCOL_RANGE_MPEG; ++ /* Source mastering / dynamic HDR metadata no longer describes these pixels. */ ++ av_frame_side_data_remove_by_props(&out->side_data, &out->nb_side_data, ++ AV_SIDE_DATA_PROP_COLOR_DEPENDENT); ++ ++ return 0; ++} ++ ++/* Resolve every frame: live streams may change color without changing format. ++ * Never infer a transfer, matrix, primaries or range from the pixel depth. */ ++static int resolve_transfer(AVFilterContext *ctx, const AVFrame *in) ++{ ++ TonemapCUDAContext *s = ctx->priv; ++ int transfer, hdr; ++ if (s->transfer_in != TONEMAP_TRANSFER_AUTO) ++ return s->transfer_in; ++ switch (in->color_trc) { ++ case AVCOL_TRC_BT709: transfer = TONEMAP_TRANSFER_SDR; break; ++ case AVCOL_TRC_ARIB_STD_B67: transfer = TONEMAP_TRANSFER_HLG; break; ++ case AVCOL_TRC_SMPTE2084: transfer = TONEMAP_TRANSFER_PQ; break; ++ case AVCOL_TRC_UNSPECIFIED: ++ /* Untagged decodes (most MP4/TS files) are SDR BT.709 in practice, and ++ * that is what every player assumes. HDR content has to be tagged or ++ * declared explicitly. */ ++ if (!s->warned_untagged) { ++ av_log(ctx, AV_LOG_WARNING, "Input carries no color_trc; assuming limited-range BT.709 SDR\n"); ++ s->warned_untagged = 1; ++ } ++ transfer = TONEMAP_TRANSFER_SDR; ++ break; ++ default: ++ av_log(ctx, AV_LOG_ERROR, "Unsupported input color_trc (%d); declare an explicit source color setting\n", in->color_trc); ++ return AVERROR(EINVAL); ++ } ++ hdr = transfer != TONEMAP_TRANSFER_SDR; ++ /* Unspecified companions follow the transfer; only explicit contradictions fail. */ ++ if ((in->color_primaries != AVCOL_PRI_UNSPECIFIED && in->color_primaries != (hdr ? AVCOL_PRI_BT2020 : AVCOL_PRI_BT709)) || ++ (in->colorspace != AVCOL_SPC_UNSPECIFIED && in->colorspace != (hdr ? AVCOL_SPC_BT2020_NCL : AVCOL_SPC_BT709)) || ++ (in->color_range != AVCOL_RANGE_UNSPECIFIED && in->color_range != AVCOL_RANGE_MPEG)) { ++ av_log(ctx, AV_LOG_ERROR, ++ "Unsupported or contradictory input color metadata (trc=%d primaries=%d matrix=%d range=%d); declare a supported source color setting\n", ++ in->color_trc, in->color_primaries, in->colorspace, in->color_range); ++ return AVERROR(EINVAL); ++ } ++ return transfer; ++} ++ ++static int tonemap_cuda_filter_frame(AVFilterLink *link, AVFrame *in) ++{ ++ AVFilterContext *ctx = link->dst; ++ TonemapCUDAContext *s = ctx->priv; ++ AVFilterLink *outlink = ctx->outputs[0]; ++ CudaFunctions *cu = s->hwctx->internal->cuda_dl; ++ ++ AVFrame *out = NULL; ++ CUcontext dummy; ++ int ret = 0; ++ ++ ret = resolve_transfer(ctx, in); ++ if (ret < 0) ++ goto fail; ++ s->current_transfer = ret; ++ if (s->passthrough || (s->current_transfer == s->transfer_out && s->input_depth == s->output_depth && ++ s->input_422 == s->output_422)) { ++ // Identity frames leave with the resolved contract stamped: an assumed ++ // (untagged) SDR input must not stay untagged for the encoder VUI or a ++ // compositor colour check downstream. ++ { ++ int hdr = s->current_transfer != TONEMAP_TRANSFER_SDR; ++ in->color_trc = s->current_transfer == TONEMAP_TRANSFER_SDR ? AVCOL_TRC_BT709 : ++ s->current_transfer == TONEMAP_TRANSFER_HLG ? AVCOL_TRC_ARIB_STD_B67 : AVCOL_TRC_SMPTE2084; ++ in->color_primaries = hdr ? AVCOL_PRI_BT2020 : AVCOL_PRI_BT709; ++ in->colorspace = hdr ? AVCOL_SPC_BT2020_NCL : AVCOL_SPC_BT709; ++ in->color_range = AVCOL_RANGE_MPEG; ++ } ++ // Auto mode has a stable output pool even when frame transfers change. ++ // CUDA buffers remain owned by in->buf; only the compatible pool tag ++ // changes, so identity frames still require no device copy. ++ if (s->frames_ctx) { ++ av_buffer_unref(&in->hw_frames_ctx); ++ in->hw_frames_ctx = av_buffer_ref(s->frames_ctx); ++ if (!in->hw_frames_ctx) { ++ ret = AVERROR(ENOMEM); ++ goto fail; ++ } ++ } ++ return ff_filter_frame(outlink, in); ++ } ++ ++ if (in->color_range == AVCOL_RANGE_JPEG) { ++ av_log(ctx, AV_LOG_ERROR, "Full-range input is unsupported; convert to limited range first\n"); ++ ret = AVERROR(EINVAL); ++ goto fail; ++ } ++ ++ out = av_frame_alloc(); ++ if (!out) { ++ ret = AVERROR(ENOMEM); ++ goto fail; ++ } ++ ++ ret = CHECK_CU(cu->cuCtxPushCurrent(s->hwctx->cuda_ctx)); ++ if (ret < 0) ++ goto fail; ++ ++ ret = tonemap_cuda_conv(ctx, out, in); ++ ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ if (ret < 0) ++ goto fail; ++ ++ av_frame_free(&in); ++ return ff_filter_frame(outlink, out); ++fail: ++ av_frame_free(&in); ++ av_frame_free(&out); ++ return ret; ++} ++ ++#define OFFSET(x) offsetof(TonemapCUDAContext, x) ++#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM) ++static const AVOption tonemap_cuda_options[] = { ++ { "tonemap", "tonemap algorithm selection", OFFSET(tonemap), AV_OPT_TYPE_INT, { .i64 = TONEMAP_HABLE }, TONEMAP_NONE, TONEMAP_MAX - 1, FLAGS, .unit = "tonemap" }, ++ { "none", 0, 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_NONE }, 0, 0, FLAGS, .unit = "tonemap" }, ++ { "linear", 0, 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_LINEAR }, 0, 0, FLAGS, .unit = "tonemap" }, ++ { "gamma", 0, 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_GAMMA }, 0, 0, FLAGS, .unit = "tonemap" }, ++ { "clip", 0, 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_CLIP }, 0, 0, FLAGS, .unit = "tonemap" }, ++ { "reinhard", 0, 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_REINHARD }, 0, 0, FLAGS, .unit = "tonemap" }, ++ { "hable", 0, 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_HABLE }, 0, 0, FLAGS, .unit = "tonemap" }, ++ { "mobius", 0, 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_MOBIUS }, 0, 0, FLAGS, .unit = "tonemap" }, ++ { "transfer_in", "input: auto (strict frame metadata), sdr, hlg or pq", OFFSET(transfer_in), AV_OPT_TYPE_INT, { .i64 = -1 }, -1, TONEMAP_TRANSFER_AUTO, FLAGS, .unit = "transfer" }, ++ { "transfer_out", "output: sdr (NV12), hlg or pq (P010); same transfer passes through", OFFSET(transfer_out), AV_OPT_TYPE_INT, { .i64 = -1 }, -1, TONEMAP_TRANSFER_SDR, FLAGS, .unit = "transfer" }, ++ { "auto", "Require supported color metadata on every input frame", 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_TRANSFER_AUTO }, 0, 0, FLAGS, .unit = "transfer" }, ++ { "hlg", "ARIB STD-B67 (Hybrid Log-Gamma)", 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_TRANSFER_HLG }, 0, 0, FLAGS, .unit = "transfer" }, ++ { "pq", "SMPTE ST 2084 (Perceptual Quantizer)", 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_TRANSFER_PQ }, 0, 0, FLAGS, .unit = "transfer" }, ++ { "sdr", "BT.709 gamut with BT.1886 display response", 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_TRANSFER_SDR }, 0, 0, FLAGS, .unit = "transfer" }, ++ { "format", "Explicit output storage (nv12, p010le, nv16 or p210le)", OFFSET(output_format), AV_OPT_TYPE_PIXEL_FMT, { .i64 = AV_PIX_FMT_NONE }, -1, INT_MAX, FLAGS }, ++ { "sdr_white", "SDR reference white in nits", OFFSET(sdr_white), AV_OPT_TYPE_DOUBLE, { .dbl = 203.0 }, 1, 10000, FLAGS }, ++ { "hdr_peak", "HLG display peak / HDR tone mapping peak in nits", OFFSET(hdr_peak), AV_OPT_TYPE_DOUBLE, { .dbl = 1000.0 }, 100, 10000, FLAGS }, ++ { "param", "tonemap parameter", OFFSET(param), AV_OPT_TYPE_DOUBLE, { .dbl = NAN }, DBL_MIN, DBL_MAX, FLAGS }, ++ { "desat", "desaturation parameter", OFFSET(desat), AV_OPT_TYPE_DOUBLE, { .dbl = 0.5 }, 0, DBL_MAX, FLAGS }, ++ { NULL } ++}; ++ ++AVFILTER_DEFINE_CLASS(tonemap_cuda); ++ ++static const AVFilterPad tonemap_cuda_inputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .filter_frame = tonemap_cuda_filter_frame, ++ }, ++}; ++ ++static const AVFilterPad tonemap_cuda_outputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .config_props = tonemap_cuda_config_output, ++ }, ++}; ++ ++const FFFilter ff_vf_tonemap_cuda = { ++ .p.name = "tonemap_cuda", ++ .p.description = NULL_IF_CONFIG_SMALL("Convert SDR, HLG and PQ with tonemapping using CUDA"), ++ .p.priv_class = &tonemap_cuda_class, ++ .priv_size = sizeof(TonemapCUDAContext), ++ .init = tonemap_cuda_init, ++ .uninit = tonemap_cuda_uninit, ++ FILTER_INPUTS(tonemap_cuda_inputs), ++ FILTER_OUTPUTS(tonemap_cuda_outputs), ++ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), ++ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, ++}; +diff --git a/libavfilter/vf_tonemap_cuda.cu b/libavfilter/vf_tonemap_cuda.cu +new file mode 100644 +index 0000000..ccfc54d +--- /dev/null ++++ b/libavfilter/vf_tonemap_cuda.cu +@@ -0,0 +1,304 @@ ++/* ++ * This file is part of FFmpeg. ++ * ++ * FFmpeg is free software; you can redistribute it and/or ++ * modify it under the terms of the GNU Lesser General Public ++ * License as published by the Free Software Foundation; either ++ * version 2.1 of the License, or (at your option) any later version. ++ * ++ * FFmpeg is distributed in the hope that it will be useful, ++ * but WITHOUT ANY WARRANTY; without even the implied warranty of ++ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ * Lesser General Public License for more details. ++ * ++ * You should have received a copy of the GNU Lesser General Public ++ * License along with FFmpeg; if not, write to the Free Software ++ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA ++ */ ++ ++/* ++ * CUDA HDR->SDR tone mapper, ported from FFmpeg libavfilter/opencl/tonemap.cl ++ * and colorspace_common.cl, with display-light SDR/HLG/PQ conversion on ++ * limited-range NV12/P010 and SDR BT.1886 with an ideal black display. ++ */ ++ ++#define REFERENCE_WHITE 100.0f ++#define ST2084_MAX_LUMINANCE 10000.0f ++__device__ __constant__ float ST2084_M1 = 0.1593017578125f; ++__device__ __constant__ float ST2084_M2 = 78.84375f; ++__device__ __constant__ float ST2084_C1 = 0.8359375f; ++__device__ __constant__ float ST2084_C2 = 18.8515625f; ++__device__ __constant__ float ST2084_C3 = 18.6875f; ++__device__ __constant__ float HLG_A = 0.17883277f; ++__device__ __constant__ float HLG_B = 0.28466892f; ++__device__ __constant__ float HLG_C = 0.55991073f; ++#define SDR_AVG 0.25f ++ ++// Tone-map operators (op=0 direct/none, 1 linear, 2 clip, 3 reinhard, ++// 4 hable, 5 mobius, 6 gamma). ++__device__ static inline float hable_f(float in) { ++ float a = 0.15f, b = 0.50f, c = 0.10f, d = 0.20f, e = 0.02f, f = 0.30f; ++ return (in * (in * a + b * c) + d * e) / (in * (in * a + b) + d * f) - e / f; ++} ++// fast-math intrinsics: ~precision for speed on the SM-bound tonemap path ++__device__ static float tone_op(int op, float s, float peak, float p) { ++ switch (op) { ++ case 1: return s * p / peak; // linear ++ case 2: return fminf(fmaxf(s * p, 0.0f), 1.0f); // clip ++ case 3: return s / (s + p) * (peak + p) / peak; // reinhard ++ case 4: return hable_f(s) / hable_f(peak); // hable ++ case 5: { // mobius ++ float j = p; ++ if (s <= j) return s; ++ float a = -j * j * (peak - 1.0f) / (j * j - 2.0f * j + peak); ++ float b = (j * j - 2.0f * j * peak + peak) / fmaxf(peak - 1.0f, 1e-6f); ++ return (b * b + 2.0f * b * j + j * j) / (b - a) * (s + a) / (s + b); ++ } ++ case 6: { // gamma ++ float pv = s > 0.05f ? s / peak : 0.05f / peak; ++ float v = __powf(pv, 1.0f / p); ++ return s > 0.05f ? v : (s * v / 0.05f); ++ } ++ default: return s; // direct ++ } ++} ++ ++// fast-math intrinsics: ~precision for speed on the SM-bound tonemap path ++__device__ static inline float eotf_st2084(float x) { ++ float p = __powf(fmaxf(x, 0.0f), 1.0f / ST2084_M2); ++ float a = fmaxf(p - ST2084_C1, 0.0f); ++ float b = fmaxf(ST2084_C2 - ST2084_C3 * p, 1e-6f); ++ float c = __powf(a / b, 1.0f / ST2084_M1); ++ return x > 0.0f ? c * ST2084_MAX_LUMINANCE / REFERENCE_WHITE : 0.0f; ++} ++// fast-math intrinsics: ~precision for speed on the SM-bound tonemap path ++__device__ static inline float inverse_oetf_hlg(float x) { ++ float a = 4.0f * x * x; ++ float b = __expf((x - HLG_C) / HLG_A) + HLG_B; ++ return x < 0.5f ? a : b; ++} ++// BT.2020 luma for the HLG OOTF and BT.709 luma for desaturation. ++__device__ static inline float luma_2020(float3 c) { return 0.2627f * c.x + 0.6780f * c.y + 0.0593f * c.z; } ++__device__ static inline float luma_709(float3 c) { return 0.2126f * c.x + 0.7152f * c.y + 0.0722f * c.z; } ++ ++// fast-math intrinsics: ~precision for speed on the SM-bound tonemap path ++__device__ static inline float3 ootf_hlg(float3 c, float peak) { ++ float luma = luma_2020(c); ++ float gamma = fmaxf(1.0f, 1.2f + 0.42f * __log10f(peak * REFERENCE_WHITE / 1000.0f)); ++ float factor = peak * __powf(fmaxf(luma, 1e-6f), gamma - 1.0f) / __powf(12.0f, gamma); ++ return make_float3(c.x * factor, c.y * factor, c.z * factor); ++} ++ ++// Linear BT.2020 -> linear BT.709 primaries. ++__device__ static inline float3 lrgb2020_to_709(float3 c) { ++ return make_float3( ++ 1.660491f * c.x - 0.587641f * c.y - 0.072850f * c.z, ++ -0.124550f * c.x + 1.132900f * c.y - 0.008349f * c.z, ++ -0.018151f * c.x - 0.100579f * c.y + 1.118730f * c.z); ++} ++ ++// fast-math intrinsics: ~precision for speed on the SM-bound tonemap path ++__device__ static inline float3 map_one_pixel_rgb(float3 rgb, float peak, float average, ++ int op, float param, float desat) { ++ float sig = fmaxf(fmaxf(rgb.x, fmaxf(rgb.y, rgb.z)), 1e-6f); ++ float sig_old = sig; ++ float slope = fminf(1.0f, SDR_AVG / average); ++ sig *= slope; peak *= slope; ++ if (desat > 0.0f) { ++ float luma = luma_709(rgb); ++ float coeff = fmaxf(sig - 0.18f, 1e-6f) / fmaxf(sig, 1e-6f); ++ coeff = __powf(coeff, 10.0f / desat); ++ rgb = make_float3(rgb.x * (1 - coeff) + luma * coeff, ++ rgb.y * (1 - coeff) + luma * coeff, ++ rgb.z * (1 - coeff) + luma * coeff); ++ sig = sig * (1 - coeff) + luma * slope * coeff; ++ } ++ sig = tone_op(op, sig, peak, param); ++ sig = fminf(sig, 1.0f); ++ float g = sig / sig_old; ++ return make_float3(rgb.x * g, rgb.y * g, rgb.z * g); ++} ++ ++// BT.709 non-linear RGB in [0,1] -> limited-range YCbCr (normalized, before ++// 8-bit quantization). ++__device__ static inline float3 rgb2yuv_709(float3 c) { ++ float y = 0.2126f * c.x + 0.7152f * c.y + 0.0722f * c.z; ++ float u = (c.z - y) / 1.8556f; ++ float v = (c.x - y) / 1.5748f; ++ return make_float3(y, u, v); ++} ++ ++__device__ static inline float3 mul_rgb(float3 c, float scale) { ++ return make_float3(c.x * scale, c.y * scale, c.z * scale); ++} ++ ++// Display-light mapping: SDR BT.1886 (ideal black), PQ ST 2084, HLG BT.2100. ++// Linear RGB is in units of 100 nits. ++__device__ static inline float3 to_display_light(float3 c, int transfer, ++ float sdr_white, float hdr_peak) { ++ c = make_float3(fmaxf(c.x, 0.0f), fmaxf(c.y, 0.0f), fmaxf(c.z, 0.0f)); ++ if (transfer == 2) ++ return mul_rgb(make_float3(__powf(c.x, 2.4f), __powf(c.y, 2.4f), __powf(c.z, 2.4f)), ++ sdr_white / REFERENCE_WHITE); ++ if (transfer == 1) ++ return make_float3(eotf_st2084(c.x), eotf_st2084(c.y), eotf_st2084(c.z)); ++ return ootf_hlg(make_float3(inverse_oetf_hlg(c.x), inverse_oetf_hlg(c.y), ++ inverse_oetf_hlg(c.z)), hdr_peak / REFERENCE_WHITE); ++} ++ ++__device__ static inline float oetf_pq(float c) { ++ float p = __powf(fmaxf(c, 0.0f) * REFERENCE_WHITE / ST2084_MAX_LUMINANCE, ST2084_M1); ++ return __powf((ST2084_C1 + ST2084_C2 * p) / (1.0f + ST2084_C3 * p), ST2084_M2); ++} ++ ++// Inverse of inverse_oetf_hlg: scene RGB here is normalized to [0,12]. ++__device__ static inline float oetf_hlg(float c) { ++ c = fmaxf(c, 0.0f); ++ return c <= 1.0f ? 0.5f * sqrtf(c) : HLG_A * __logf(c - HLG_B) + HLG_C; ++} ++ ++__device__ static inline float3 from_display_light(float3 c, int transfer, ++ float sdr_white, float hdr_peak) { ++ c = make_float3(fmaxf(c.x, 0.0f), fmaxf(c.y, 0.0f), fmaxf(c.z, 0.0f)); ++ if (transfer == 2) { ++ c = mul_rgb(c, REFERENCE_WHITE / sdr_white); ++ return make_float3(__powf(c.x, 1.0f / 2.4f), __powf(c.y, 1.0f / 2.4f), ++ __powf(c.z, 1.0f / 2.4f)); ++ } ++ if (transfer == 1) ++ return make_float3(oetf_pq(c.x), oetf_pq(c.y), oetf_pq(c.z)); ++ float gamma = fmaxf(1.0f, 1.2f + 0.42f * __log10f(hdr_peak / 1000.0f)); ++ float luma = luma_2020(c); ++ if (luma <= 0.0f) ++ return make_float3(0.0f, 0.0f, 0.0f); ++ // BT.2100 inverse OOTF, preserving chromaticity in display light. ++ float scene_luma = 12.0f * __powf(luma * REFERENCE_WHITE / hdr_peak, 1.0f / gamma); ++ c = mul_rgb(c, scene_luma / luma); ++ return make_float3(oetf_hlg(c.x), oetf_hlg(c.y), oetf_hlg(c.z)); ++} ++ ++__device__ static inline float3 lrgb709_to_2020(float3 c) { ++ return make_float3(0.627404f * c.x + 0.329283f * c.y + 0.043313f * c.z, ++ 0.069097f * c.x + 0.919540f * c.y + 0.011362f * c.z, ++ 0.016391f * c.x + 0.088013f * c.y + 0.895595f * c.z); ++} ++ ++__device__ static inline float3 convert_rgb(float3 c, int transfer_in, int transfer_out, ++ float sdr_white, float hdr_peak, ++ int op, float param, float desat) { ++ c = to_display_light(c, transfer_in, sdr_white, hdr_peak); ++ if (transfer_in == 2 && transfer_out != 2) ++ c = lrgb709_to_2020(c); ++ else if (transfer_in != 2 && transfer_out == 2) { ++ c = lrgb2020_to_709(c); ++ // Only HDR->SDR compresses highlights. SDR embedding and HLG/PQ ++ // conversion preserve display luminance, without inventing highlights. ++ c = mul_rgb(c, REFERENCE_WHITE / sdr_white); ++ c = map_one_pixel_rgb(c, hdr_peak / sdr_white, SDR_AVG, op, param, desat); ++ c = mul_rgb(c, sdr_white / REFERENCE_WHITE); ++ } ++ c = from_display_light(c, transfer_out, sdr_white, hdr_peak); ++ // Clip outside the target signal range before forming YCbCr. ++ return make_float3(fminf(c.x, 1.0f), fminf(c.y, 1.0f), fminf(c.z, 1.0f)); ++} ++ ++__device__ static inline float3 yuv_to_rgb(float y, float u, float v, int depth, int transfer) { ++ float scale = depth == 10 ? 4.0f : 1.0f; ++ y = (y / scale - 16.0f) / 219.0f; ++ u = (u / scale - 128.0f) / 224.0f; ++ v = (v / scale - 128.0f) / 224.0f; ++ if (transfer == 2) ++ return make_float3(y + 1.5748f * v, y - 0.187324f * u - 0.468124f * v, y + 1.8556f * u); ++ return make_float3(y + 1.4746f * v, y - 0.164553f * u - 0.571353f * v, y + 1.8814f * u); ++} ++ ++__device__ static inline float3 rgb_to_yuv(float3 c, int transfer) { ++ if (transfer == 2) ++ return rgb2yuv_709(c); ++ float y = luma_2020(c); ++ return make_float3(y, (c.z - y) / 1.8814f, (c.x - y) / 1.4746f); ++} ++ ++__device__ static inline float read_code(const unsigned char *p, int pitch, int x, int y, int depth) { ++ const unsigned char *row = p + y * pitch; ++ return depth == 10 ? ((const unsigned short *)row)[x] >> 6 : row[x]; ++} ++ ++__device__ static inline void write_code(unsigned char *p, int pitch, int x, int y, int depth, float code) { ++ unsigned char *row = p + y * pitch; ++ if (depth == 10) ++ ((unsigned short *)row)[x] = (unsigned short)fminf(fmaxf(rintf(code * 4.0f), 0.0f), 1023.0f) << 6; ++ else ++ row[x] = (unsigned char)fminf(fmaxf(rintf(code), 0.0f), 255.0f); ++} ++ ++// A thread handles one 2x2 luma block. 4:2:0 input shares one chroma sample ++// across the block, 4:2:2 input has one per luma row; the output side likewise ++// writes one sample (average of the block) or one per row. All pitches are bytes. ++extern "C" __global__ void Tonemap_Cuda( ++ unsigned char *src_y, int src_y_linesize, ++ unsigned char *src_uv, int src_uv_linesize, ++ unsigned char *dst_y, int dst_y_linesize, ++ unsigned char *dst_uv, int dst_uv_linesize, ++ int width, int height, ++ int transfer, int op, float param, float desat, ++ int transfer_out, int input_depth, int output_depth, int input_422, int output_422, ++ float sdr_white, float hdr_peak) ++{ ++ int xi = blockIdx.x * blockDim.x + threadIdx.x; ++ int yi = blockIdx.y * blockDim.y + threadIdx.y; ++ ++ int cw = width >> 1; ++ int ch = height >> 1; ++ if (xi >= cw || yi >= ch) ++ return; ++ ++ int x = 2 * xi; ++ int y = 2 * yi; ++ ++ float cb[2], cr[2]; ++ for (int r = 0; r < 2; ++r) { ++ int crow = input_422 ? y + r : yi; ++ cb[r] = read_code(src_uv, src_uv_linesize, x, crow, input_depth); ++ cr[r] = read_code(src_uv, src_uv_linesize, x + 1, crow, input_depth); ++ } ++ if (transfer == transfer_out) { ++ // Depth-only conversion preserves code values, including foot/headroom. ++ float scale = input_depth == 10 ? 0.25f : 1.0f; ++ for (int i = 0; i < 4; ++i) { ++ int px = x + (i & 1), py = y + (i >> 1); ++ write_code(dst_y, dst_y_linesize, px, py, output_depth, ++ read_code(src_y, src_y_linesize, px, py, input_depth) * scale); ++ } ++ if (output_422) { ++ for (int r = 0; r < 2; ++r) { ++ write_code(dst_uv, dst_uv_linesize, x, y + r, output_depth, cb[r] * scale); ++ write_code(dst_uv, dst_uv_linesize, x + 1, y + r, output_depth, cr[r] * scale); ++ } ++ } else { ++ write_code(dst_uv, dst_uv_linesize, x, yi, output_depth, 0.5f * (cb[0] + cb[1]) * scale); ++ write_code(dst_uv, dst_uv_linesize, x + 1, yi, output_depth, 0.5f * (cr[0] + cr[1]) * scale); ++ } ++ return; ++ } ++ float u[2] = {0.0f, 0.0f}, v[2] = {0.0f, 0.0f}; ++ for (int i = 0; i < 4; ++i) { ++ int px = x + (i & 1), r = i >> 1, py = y + r; ++ float code = read_code(src_y, src_y_linesize, px, py, input_depth); ++ float3 c = rgb_to_yuv(convert_rgb(yuv_to_rgb(code, cb[r], cr[r], input_depth, transfer), ++ transfer, transfer_out, sdr_white, hdr_peak, op, param, desat), ++ transfer_out); ++ write_code(dst_y, dst_y_linesize, px, py, output_depth, 219.0f * c.x + 16.0f); ++ u[r] += c.y; ++ v[r] += c.z; ++ } ++ if (output_422) { ++ for (int r = 0; r < 2; ++r) { ++ write_code(dst_uv, dst_uv_linesize, x, y + r, output_depth, 224.0f * (u[r] * 0.5f) + 128.0f); ++ write_code(dst_uv, dst_uv_linesize, x + 1, y + r, output_depth, 224.0f * (v[r] * 0.5f) + 128.0f); ++ } ++ } else { ++ write_code(dst_uv, dst_uv_linesize, x, yi, output_depth, 224.0f * ((u[0] + u[1]) * 0.25f) + 128.0f); ++ write_code(dst_uv, dst_uv_linesize, x + 1, yi, output_depth, 224.0f * ((v[0] + v[1]) * 0.25f) + 128.0f); ++ } ++} +-- +2.55.0 + diff --git a/deps/ffmpeg/8/bases.env b/deps/ffmpeg/8/bases.env index ea60d972..ba0b5038 100644 --- a/deps/ffmpeg/8/bases.env +++ b/deps/ffmpeg/8/bases.env @@ -1,5 +1,5 @@ n80_commit=140fd653aed8cad774f991ba083e2d01e86420c7 -n80_tree=6ff5e4f32a3acb868c6fd4a2ae9e16c25115a6f2 +n80_tree=5558bae0fb3a3c36bb75335a6e06950469c4e598 n81_commit=9047fa1b084f76b1b4d065af2d743df1b40dfb56 -n81_tree=3f4f8ba98c493fe9c0e0bfae5a61cd8b83fc30ca -patch_count=7 +n81_tree=e281422964996db696f00c63ef0b287cf94e07de +patch_count=9 diff --git a/deps/ffmpeg/README.md b/deps/ffmpeg/README.md index 9ec7429f..27e47a86 100644 --- a/deps/ffmpeg/README.md +++ b/deps/ffmpeg/README.md @@ -39,6 +39,14 @@ and the filter changes) is base-independent. 6. **V4L2 source timestamps.** 7. **NDI v5 registration** — registers the optional integration only; no SDK or device implementation is supplied, and NDI stays disabled in demo builds. +8. **10-bit CUDA transitions** — YUV420P10/422P10/444P10, P010 and P210 (plus + 8-bit 4:2:2/4:4:4) in a word-sample `transition_cuda` kernel. +9. **`tonemap_cuda`** — SDR BT.709 / HLG / PQ conversion on CUDA semiplanar + frames (NV12/P010 4:2:0, NV16/P210 4:2:2, resampled in the same pass) in + both directions: display-light conversion with configurable SDR + white and HDR peak, HDR-to-SDR operators with a knee parameter, automatic + per-frame contract resolution (untagged frames are BT.709 SDR), zero-copy + identity frames and fixed NV12/P010 output storage. ## FFmpeg 8 notes diff --git a/doc/NODES.md b/doc/NODES.md index 6bfc408e..a5a454ac 100644 --- a/doc/NODES.md +++ b/doc/NODES.md @@ -477,6 +477,13 @@ Encodes video or audio frames. - `hwaccel` (string, name of instance-shared object) - optional (mandatory for some encoders), name of hwaccel previously created with `hwaccel.init` +- `hdr_metadata` (object, `enc_video` only; `enc_audio` rejects it) - static HDR metadata serialised + verbatim as side data before the encoder opens: `primaries` (three `[x, y]` + pairs), `white_point` (`[x, y]`), `max_luminance` / `min_luminance` and + `max_cll` / `max_fall` in nits. `hevc_nvenc` and `av1_nvenc` emit the HDR10 + SEIs from it when FFmpeg is built against nv-codec-headers 13 (driver >= 570); + older headers silently emit none. The mixer fills it for PQ renditions + (BT.2020 / D65) from `pyplumber.mixer.color.hdr_metadata`. - `timestamps_passthrough` (bool) - default `false`, intended for codecs that don't buffer data (otherwise bad things like repeated timestamps may happen), replace PTS & DTS in outgoing packet with @@ -777,6 +784,19 @@ Import DRM PRIME frames into CUDA frames via EGL/GL interop. Non-DRM PRIME frame Parameters: - `hwaccel` (string, required) - CUDA device created with `hwaccel.init` +### `v210_to_cuda` + +Unpacks headerless packed 10-bit 4:2:2 (`v210`) packets straight into CUDA frames, one packet per frame, so generated HDR test content and SDI-style feeds reach the GPU mixer without a CPU decode. + +1 input: `av::Packet` (`stride * height` bytes each), 1 output: `av::VideoFrame` (hardware "pixel format" `cuda`) + +Parameters: +- `hwaccel` (string, required) - CUDA device created with `hwaccel.init` +- `width`, `height` (int, required) - even width; the packed bytes carry no header +- `stride` (int) - row pitch in bytes, default the 128-byte aligned v210 pitch +- `sw_format` (string) - CUDA storage, `p210le` (default) or `yuv422p10le` +- `color_trc`, `color_primaries`, `colorspace`, `color_range`, `chroma_location` (string) - tags stamped on every output frame (libavutil names, e.g. `arib-std-b67`, `bt2020`, `bt2020nc`, `tv`) + ### `cuda_infer_yolo` Run YOLO object detection on preprocessed CUDA frames using a prebuilt TensorRT engine (`.plan` / `.engine`). diff --git a/doc/mixer_orchestrator.md b/doc/mixer.md similarity index 73% rename from doc/mixer_orchestrator.md rename to doc/mixer.md index 640036a8..ddddef45 100644 --- a/doc/mixer_orchestrator.md +++ b/doc/mixer.md @@ -1,8 +1,23 @@ # Video Mixer AVPlumber's mixer is a two-slot program/preview video switcher. The reusable -graph builder is `avpmixer/graph.py`; the native control implementation is in -`src/mixer/mixer_orchestrator.cpp`; the maintained example is `demos/mixer/`. +graph builder is `pyplumber/mixer/graph.py`; the native control implementation is in +`src/mixer/`; the maintained example is `demos/mixer/`. + +## Native code layout + +| Module | Role | +|---|---| +| `src/mixer/orchestrator/MixerOrchestrator.hpp`, `core.cpp` | `MixerOrchestrator` core: node adapters, interruption/abort, status | +| `src/mixer/orchestrator/{scene,cut,fade,wipe,overlay}.cpp` | one transition kind per file | +| `src/mixer/routing.hpp` | state → router route tables and compositor layer arrays (pure) | +| `src/mixer/graph_ops.{hpp,cpp}` | node/edge lookups, deferred `setObject`, readiness polls | +| `src/mixer/TransitionScheduler.{hpp,cpp}` | worker thread for scheduled transition steps | +| `src/mixer/Playout.hpp` | clocked playout: per-input queues, deadlines, frame pick (unit-tested) | +| `src/mixer/primitives/` | headers with no graph or CUDA dependency: `TickGrid`, `Cadence`, `MonotonicClock`, `CutLatency`, `CutLatencyProbe`, `Snapshot`, `OutputSnapshot`, `MixerState`, `TransitionGuard` (unit-tested where they carry logic), plus the compositor geometry headers below | +| `src/nodes/hwaccel/cuda_rect_overlay.cpp` | compositor node: scheduling, control, output plumbing | +| `src/nodes/hwaccel/cuda_rect_draw.{hpp,cpp}`, `cuda_rect_scale.cu` | kernel module, canvas clear, per-layer draw | +| `src/mixer/primitives/compositor_layers.hpp`, `pixel_layout.hpp`, `compositor_geometry.hpp` | layer parsing and draw-op resolution, format geometry, placement (pure; geometry and layout unit-tested) | The mixer carries video frames only. It has no audio routing, VAD, speaker selection, face tracking, or camera policy. diff --git a/doc/research/2026-09-07-playlist-on-mixer-plan.md b/doc/research/2026-09-07-playlist-on-mixer-plan.md index 5c732ae6..540e6c4b 100644 --- a/doc/research/2026-09-07-playlist-on-mixer-plan.md +++ b/doc/research/2026-09-07-playlist-on-mixer-plan.md @@ -136,7 +136,7 @@ worker with a monotonic timer. The native form is preferred for jitter. The Janus output is the mixer's `_build_janus_output` (force_fps, keyframe on PLI, assume format, NVENC, `dump_extra`, RTP mux, RTCP listener). Move it and -`_build_input` from `demos/mixer/mixer.py` into `avpmixer` so both demos import +`_build_input` from `demos/mixer/mixer.py` into `pyplumber.mixer` so both demos import one implementation instead of carrying copies. ### Code layout @@ -145,7 +145,7 @@ one implementation instead of carrying copies. | --- | --- | | `demos/playlist/playlist.py` | Keep policy and controller. Drop `plan_item_nodes`/`plan_switch_nodes`; add transition type and duration to `Clip`; add `scheduled_end_ms` to status. | | `demos/playlist/playlist_app.py` | Rewrite backend on `MixerGraphBuilder`. Slot pool, parking, native scheduling, orchestrator completion events. | -| `avpmixer/inputs.py`, `avpmixer/janus.py` | Extracted from the mixer demo, shared. | +| `pyplumber/mixer/inputs.py`, `pyplumber/mixer/janus.py` | Extracted from the mixer demo, shared. | | `src/...` pause team | Optional `resume at `. | | `demos/playlist/player.py` | TUI gains transition selector per item and a countdown to the scheduled cut; poll loop stays for status only. | | `demos/playlist/tests/` | Controller tests unchanged. Backend tests against a fake builder recording `preview`/`cut`/`fade` calls and their `start_pts_ms`. | @@ -188,7 +188,7 @@ Done locally, 74 unit tests green, not yet run on the NVIDIA host: - `src/PauseControlTeam.hpp`, `src/avplumber.cpp`: `resume at `; an explicit pause cancels a scheduled resume. -- `avpmixer/inputs.py`, `avpmixer/janus.py`, `avpmixer/control.py`: shared +- `pyplumber/mixer/inputs.py`, `pyplumber/mixer/janus.py`, `pyplumber/mixer/control.py`: shared decode chain, RTP output and TCP client (the mixer demo keeps its own copies). - `demos/playlist/playlist.py` policy with wallclock scheduling; `engine.py` mixer-backed backend; `control.py` JSON protocol; diff --git a/doc/research/2026-09-08-mixer-config-schema.md b/doc/research/2026-09-08-mixer-config-schema.md index 1d1815ae..88778b8d 100644 --- a/doc/research/2026-09-08-mixer-config-schema.md +++ b/doc/research/2026-09-08-mixer-config-schema.md @@ -141,8 +141,8 @@ of the engine. | Document | Engine today | Gap | | --- | --- | --- | -| `sources[]` video | `avpmixer.inputs.build_input` (NVDEC chain) | none | -| `sources[]` browser | `avpmixer.dmabuf_inputs` (window open, DMA-BUF import) | none | +| `sources[]` video | `pyplumber.mixer.inputs.build_input` (NVDEC chain) | none | +| `sources[]` browser | `pyplumber.mixer.dmabuf_inputs` (window open, DMA-BUF import) | none | | duplicate `url`/`path` | — | loader rejects; alias support (one chain, two ids) is a second `one_to_many` output, ~10 lines | | `item.dst`, `crop`, `fit: stretch\|contain` | `cuda_rect_overlay` layer: `dst_*`, `crop`, `fit` | none | | `fit: cover` | loader computes the crop from the aspect; clip sizes probed with ffprobe | none | diff --git a/doc/research/2026-09-08-mixer-dmabuf-plan.md b/doc/research/2026-09-08-mixer-dmabuf-plan.md index 33559c9c..07909e88 100644 --- a/doc/research/2026-09-08-mixer-dmabuf-plan.md +++ b/doc/research/2026-09-08-mixer-dmabuf-plan.md @@ -35,9 +35,9 @@ same 16-box. ## Code changes (small) 1. Put an api-parameterised `dmabuf_cuda_input_nodes` plus `rest_request`, - `open_browser_windows` and `wait_for_sockets` in `avpmixer/dmabuf_inputs.py`. + `open_browser_windows` and `wait_for_sockets` in `pyplumber/mixer/dmabuf_inputs.py`. The DMA-BUF demo keeps its own copy of the chain builder: its runtime image - has no `avpmixer` on the path, and re-pointing it would change that demo's + has no `pyplumber.mixer` on the path, and re-pointing it would change that demo's image or compose files, which this change must not do. 2. `demos/mixer/mixer.py`: the scheme dispatch in `_build_input`, the four options in `parse_args`/`GraphOptions`, and window opening in @@ -46,7 +46,7 @@ same 16-box. check for a browser source too. 3. `demos/dmabuf-browser/compose.mixer.yaml`: an override that runs the mixer demo in the existing consumer image (it is the one built with DRM/EGL and - NVIDIA), mounting `avpmixer/` and `demos/mixer/` read-only, sharing the + NVIDIA), mounting `pyplumber/mixer/` and `demos/mixer/` read-only, sharing the `dma-browser-sockets` volume, exposing the control port 7777 for the TUI. The base `compose.yaml` and `compose.scaling.yaml` are unchanged. 4. Tests: a fake-API graph test in `demos/mixer/tests` asserting that a diff --git a/doc/superpowers/specs/2026-09-07-mixer-graph-optimization-design.md b/doc/superpowers/specs/2026-09-07-mixer-graph-optimization-design.md index e305fd5d..54f351ba 100644 --- a/doc/superpowers/specs/2026-09-07-mixer-graph-optimization-design.md +++ b/doc/superpowers/specs/2026-09-07-mixer-graph-optimization-design.md @@ -6,7 +6,7 @@ Approved scope: reduce actual graph/code complexity and CPU/GPU consumption agai Real inputs fan out to the two permanent CUDA compositors. Each scene specifies per-source crop and destination rectangles. The compositor scales directly into the fixed output canvas, using each selected frame's dimensions and pitch. Equal-size copies retain the existing copy path. Dimension changes do not rebuild the graph; old queued frames retain their own references. Pixel formats remain explicitly validated. -Python graph construction stays in `avpmixer`; scene generation stays in the demo. C++ owns runtime routing, shared playout, preview readiness, Cut/Fade/media-Wipe and exact-picture interruption. Explicit FFmpeg preprocessing remains supported for existing callers. No framework graph-management or streaming decoder changes. +Python graph construction stays in `pyplumber.mixer`; scene generation stays in the demo. C++ owns runtime routing, shared playout, preview readiness, Cut/Fade/media-Wipe and exact-picture interruption. Explicit FFmpeg preprocessing remains supported for existing callers. No framework graph-management or streaming decoder changes. The first comparison retains input normalization and all FPS/timing stages. This isolates replacement of 62 geometry filters and their router with 16 input fanouts. Earlier estimates that also removed normalization are not the count for this first stage. Removing normalization requires a separate before/after validation within the same PR. All FPS stages remain. diff --git a/pyplumber/__init__.py b/pyplumber/__init__.py index aa23b9d6..fab58746 100644 --- a/pyplumber/__init__.py +++ b/pyplumber/__init__.py @@ -1 +1,16 @@ -from .core import AVPlumber \ No newline at end of file +"""Python side of avplumber: the engine API (``AVPlumber``, ``pyplumber.node``) and the +application libraries built on it (``pyplumber.mixer``). + +``AVPlumber`` needs the native ``_avplumber`` module and is imported on first use, so the +pure-Python parts (mixer configuration, control client, color contracts) load without it. +""" + + +def __getattr__(name): + if name == "AVPlumber": + from .core import AVPlumber + return AVPlumber + raise AttributeError(name) + + +__all__ = ["AVPlumber"] diff --git a/pyplumber/mixer.py b/pyplumber/mixer.py deleted file mode 100644 index 370c1585..00000000 --- a/pyplumber/mixer.py +++ /dev/null @@ -1,5 +0,0 @@ -"""Compatibility import; mixer implementation lives in avpmixer.""" - -from avpmixer import MixerGraphBuilder, MixerScene, MixerSource - -__all__ = ["MixerGraphBuilder", "MixerScene", "MixerSource"] diff --git a/pyplumber/mixer/__init__.py b/pyplumber/mixer/__init__.py new file mode 100644 index 00000000..fa575165 --- /dev/null +++ b/pyplumber/mixer/__init__.py @@ -0,0 +1,17 @@ +"""The mixer library: graph construction, show configuration, control client, color contracts +and Janus output, on top of the pyplumber engine API. + +``MixerGraphBuilder`` needs the native ``_avplumber`` module; it is imported lazily so that +``pyplumber.mixer.control`` and the other pure-Python helpers work without it. +""" + +from .models import MixerScene, MixerSource + +__all__ = ["MixerGraphBuilder", "MixerScene", "MixerSource"] + + +def __getattr__(name): + if name == "MixerGraphBuilder": + from .graph import MixerGraphBuilder + return MixerGraphBuilder + raise AttributeError(name) diff --git a/avpmixer/clipcache.py b/pyplumber/mixer/clipcache.py similarity index 100% rename from avpmixer/clipcache.py rename to pyplumber/mixer/clipcache.py diff --git a/pyplumber/mixer/color.py b/pyplumber/mixer/color.py new file mode 100644 index 00000000..b26fe325 --- /dev/null +++ b/pyplumber/mixer/color.py @@ -0,0 +1,133 @@ +"""GPU color contracts shared by source, canvas and rendition builders. + +Input metadata is resolved on every decoded frame by tonemap_cuda. A declared +contract overrides frame tags explicitly; pixel depth never selects a transfer. +""" + +from dataclasses import dataclass +from math import isfinite +from typing import Mapping + +TRANSFER_TAGS = {"sdr": "bt709", "hlg": "arib-std-b67", "pq": "smpte2084"} +OPERATORS = ("none", "linear", "gamma", "clip", "reinhard", "hable", "mobius") +COLOR_KEYS = ("color_trc", "color_primaries", "colorspace", "color_range") +TEN_BIT_FORMATS = ("p010le", "p210le", "yuv420p10le", "yuv422p10le", "yuv444p10le") +SEMIPLANAR_FORMATS = ("nv12", "nv16", "p010le", "p210le") # what tonemap_cuda reads and writes +# HDR10 static metadata for PQ outputs: the mixer masters on BT.2020 primaries with D65 white. +MASTERING_PRIMARIES = {"bt2020": {"primaries": [[0.708, 0.292], [0.170, 0.797], [0.131, 0.046]], + "white_point": [0.3127, 0.3290]}} + + +def hdr_metadata(peak_nits, *, max_cll=0, max_fall=0, min_nits=0.0001): + """Mastering display + content light level for a PQ output, in nits. MaxCLL defaults + to the peak and MaxFALL to 40% of MaxCLL, the usual 1000/400 pairing.""" + cll = max_cll or round(peak_nits) + return {**MASTERING_PRIMARIES["bt2020"], "max_luminance": round(peak_nits), "min_luminance": min_nits, + "max_cll": cll, "max_fall": max_fall or round(cll * 0.4)} +YUV_FORMATS = ("nv12", "nv16", "yuv420p", "yuv422p", "yuv444p", *TEN_BIT_FORMATS) + + +@dataclass(frozen=True) +class Color: + transfer: str = "sdr" + + def __post_init__(self): + if self.transfer not in TRANSFER_TAGS: + raise ValueError(f"unsupported color transfer {self.transfer!r}; use sdr, hlg or pq") + + @property + def tags(self): + hdr = self.transfer != "sdr" + return dict(zip(COLOR_KEYS, (TRANSFER_TAGS[self.transfer], + "bt2020" if hdr else "bt709", + "bt2020nc" if hdr else "bt709", "tv"))) + + @property + def setparams(self): + return "setparams=" + ":".join(f"{'range' if k == 'color_range' else k}={v}" + for k, v in self.tags.items()) + + def validate_format(self, pixel_format): + if pixel_format not in SEMIPLANAR_FORMATS: + raise ValueError(f"unsupported canvas/output pixel format {pixel_format!r}") + if self.transfer != "sdr" and pixel_format not in TEN_BIT_FORMATS: + raise ValueError("HLG/PQ canvas and outputs require a 10-bit pixel format") + + @classmethod + def parse(cls, value): + if isinstance(value, cls): + return value + if isinstance(value, str): + return cls(value) + if not isinstance(value, Mapping): + raise ValueError("color must be sdr, hlg, pq or a complete color metadata object") + missing = [k for k in COLOR_KEYS if not value.get(k)] + if missing: + raise ValueError(f"explicit source color setting is incomplete: missing {', '.join(missing)}") + tags = dict(value) + tags["color_range"] = {"limited": "tv", "mpeg": "tv"}.get(tags["color_range"], tags["color_range"]) + tags["colorspace"] = {"bt2020_ncl": "bt2020nc"}.get(tags["colorspace"], tags["colorspace"]) + for transfer in TRANSFER_TAGS: + color = cls(transfer) + if all(tags[k] == v for k, v in color.tags.items()): + return color + raise ValueError("unsupported or contradictory color metadata; supported contracts are limited-range " + "BT.709 SDR and BT.2020 non-constant-luminance HLG/PQ") + + +def declared_color(obj): + """None means require frame metadata; an override must be complete.""" + fields = {k: obj[k] for k in COLOR_KEYS if k in obj} + if "color" in obj: + color = Color.parse(obj["color"]) + if fields and any(color.tags[k] != v for k, v in fields.items()): + raise ValueError("color preset contradicts individual color metadata fields") + return color + return Color.parse(fields) if fields else None + + +def conversion_graph(target, pixel_format, *, source=None, source_format=None, + tonemap="clip", sdr_white=203.0, hdr_peak=1000.0, desat=0.0, param=0.0): + """Validate and normalize CUDA YUV frames, keeping identity frames zero-copy. + + NVDEC emits NV12/P010. Callers with other CUDA storage must declare it so + the required chroma conversion precedes tone mapping. Custom source filters + run before this graph: their output frame metadata is authoritative. + """ + target = Color.parse(target) + target.validate_format(pixel_format) + if tonemap not in OPERATORS: + raise ValueError(f"unsupported tone-map operator {tonemap!r}") + if not all(isfinite(v) for v in (sdr_white, hdr_peak, desat, param)) or not (1 <= sdr_white <= hdr_peak <= 10000 and hdr_peak >= 100 and desat >= 0 and param >= 0): + raise ValueError("require finite 1 <= sdr_white <= hdr_peak <= 10000, hdr_peak >= 100, desat >= 0 and param >= 0") + if source_format and source_format not in YUV_FORMATS: + raise ValueError(f"unsupported source pixel format {source_format!r}; color conversion requires CUDA YUV") + if source is not None and Color.parse(source) == target: + # Same contract: stamp it and only change storage, so 4:2:2 (P210) content + # never round-trips through the 4:2:0-only tone mapper. + parts = [target.setparams] + return ",".join(parts if source_format == pixel_format else parts + [f"scale_cuda=format={pixel_format}"]) + parts = [Color.parse(source).setparams] if source is not None else [] + if source_format and source_format not in SEMIPLANAR_FORMATS: + # tonemap_cuda works on semiplanar storage; planar sources are re-laid out at 10 bits. + parts.append("scale_cuda=format=p210le" if "422" in source_format else "scale_cuda=format=p010le") + # tonemap_cuda converts color and storage in one pass, 4:2:0 or 4:2:2 in and out. + # param is the operator knee in reference-white units (mobius/reinhard; 0 keeps the + # filter default 0.3). mobius at 0.9 keeps 0..90% of SDR white linear and folds + # everything brighter into the top 10% of the SDR range; 1.0 would be a plain clip. + parts.append(f"tonemap_cuda=transfer_in=auto:transfer_out={target.transfer}:format={pixel_format}" + f":tonemap={tonemap}:sdr_white={sdr_white:g}:hdr_peak={hdr_peak:g}:desat={desat:g}" + + (f":param={param:g}" if param else "")) + return ",".join(parts) + + +def rendition_color(canvas, codec, requested=None, tonemap=""): + canvas = Color.parse(canvas) + if codec not in ("h264_nvenc", "hevc_nvenc"): + raise ValueError(f"unsupported mixer output codec {codec!r}") + color = Color.parse(requested) if requested is not None else (Color() if tonemap or codec == "h264_nvenc" else canvas) + if codec == "h264_nvenc" and color.transfer != "sdr": + raise ValueError("H.264 renditions require SDR; use HEVC Main10 for HLG/PQ") + if tonemap and color.transfer != "sdr": + raise ValueError("rendition.tonemap requests SDR and contradicts the output color") + return color diff --git a/avpmixer/config.py b/pyplumber/mixer/config.py similarity index 57% rename from avpmixer/config.py rename to pyplumber/mixer/config.py index 4529d447..47baa416 100644 --- a/avpmixer/config.py +++ b/pyplumber/mixer/config.py @@ -12,13 +12,21 @@ import re import shutil import subprocess -from dataclasses import dataclass, replace +from dataclasses import dataclass, fields, replace from pathlib import Path from typing import Any, Dict, List, Optional, Tuple +from .color import Color, declared_color, OPERATORS, TRANSFER_TAGS, YUV_FORMATS + FITS = ("stretch", "contain", "cover") TRANSITIONS = ("cut", "fade", "wipe") +# Compositor/transition working formats: the semiplanar family only. NV12 is the +# 8-bit default; P010/P210 are 10-bit 4:2:0/4:2:2. Planar layouts are excluded on +# purpose: the compositor cannot promote 8-bit sources or draw the RGBA wipe +# onto them, so they only fail later. +WORKING_FORMATS = ("nv12", "p010le", "p210le") DEFAULT_FPS = 30 # canvas.fps when the document does not say +MAX_SOURCES = 32 # cuda_rect_overlay active_inputs is a 32-bit pad mask DEFAULT_FADE_SECONDS = 0.5 DEFAULT_TRANSITION = "cut" @@ -34,12 +42,25 @@ class Rect: @dataclass(frozen=True) class Source: id: str - kind: str # "browser" | "video" - location: str # url (browser) or path (video) + kind: str # "browser" | "video" | "v210" + location: str # url (browser) or path (video/v210 raw file) width: int = 0 height: int = 0 fps: int = 0 # browser paint rate; 0 = canvas fps loop: bool = True + # Empty fields require complete decoded frame metadata. Raw inputs must + # declare a contract because their bytes carry no color metadata. + color_trc: str = "" + color_primaries: str = "" + colorspace: str = "" + color_range: str = "" + filter_graph: str = "" # optional CUDA source filter, before scene/alias fan-out + filter_output_format: str = "" + + @property + def color(self): + tags = {k: getattr(self, k) for k in ("color_trc", "color_primaries", "colorspace", "color_range")} + return Color.parse(tags) if any(tags.values()) else None @dataclass(frozen=True) @@ -52,10 +73,22 @@ class Rendition: height: int = 0 fps: int = 0 # 0 keeps the canvas rate bitrate_kbps: int = 3000 - codec: str = "h264_nvenc" - profile: str = "baseline" # WebRTC negotiates constrained baseline + codec: str = "" # "" auto-selects: HEVC for a 10-bit program, else H.264 + profile: str = "" # "" lets the encoder pick main/main10/baseline preset: str = "p7" # NVENC quality preset port: int = 0 # janus target: 0 keeps the configured port + # An explicit operator requests SDR. Otherwise codec/color select the target + # and automatic HDR-to-SDR conversion uses clip to preserve SDR reference white. + tonemap: str = "" + tonemap_peak: float = 10.0 # source peak in REFERENCE_WHITE units (HLG 1000 nits) + tonemap_desat: float = 0.0 # 0 keeps saturation; FFmpeg's 0.5 default washes colors out + tonemap_param: float = 0.0 # operator knee in reference-white units; 1.0 = SDR range untouched + # HDR10 static metadata for PQ outputs (HLG needs none). 0 derives MaxCLL from + # tonemap_peak and MaxFALL as 40% of it, the usual 1000/400 pair. + max_cll: int = 0 + max_fall: int = 0 + + color: str = "" # empty: inherit canvas, except H.264/tonemap imply SDR @property def aspect(self) -> str: @@ -106,6 +139,10 @@ class MixerConfig: fade_seconds: float = DEFAULT_FADE_SECONDS transition: str = DEFAULT_TRANSITION # what a pick takes with in direct mode default_wipe: str = "" + working_format: str = "nv12" # canvas.working_format: compositor/transition sw_format + latency_ms: Optional[float] = None # canvas.latency_ms: playout buffer, default two output frames + out_color: Color = Color() # canvas color contract; renditions convert from it and signal it (VUI) + wipe_color: str = "" # optional explicit override for all alpha wipe clips def source(self, id: str) -> Source: return next(s for s in self.sources if s.id == id) @@ -150,6 +187,98 @@ def _rect(obj: Any, where: str) -> Rect: return r +_SOURCE_KEYS = {"browser": ("url", "width", "height"), "video": ("path",), "v210": ("path", "width", "height")} + + +def _parse_source(s: Dict[str, Any], where: str, fps: int) -> Source: + sid, kind = str(s.get("id", "")), s.get("kind") + if not sid or "#" in sid: + raise ConfigError(f"{where}: id required (no '#')") + if kind not in _SOURCE_KEYS: + raise ConfigError(f"{where}: kind must be browser, video or v210") + if not all(k in s for k in _SOURCE_KEYS[kind]): + raise ConfigError(f"{where}: {kind} source needs {', '.join(_SOURCE_KEYS[kind])}") + source_filter = s.get("filter", "") + if not isinstance(source_filter, str): + raise ConfigError(f"{where}: filter must be a CUDA filter graph string") + filter_format = str(s.get("filter_output_format", "")) + if filter_format and filter_format not in YUV_FORMATS or source_filter and not filter_format: + raise ConfigError(f"{where}: custom filter requires filter_output_format (CUDA YUV storage)") + if source_filter and kind == "browser": + raise ConfigError(f"{where}: browser source filters are unsupported; preserve packed RGB alpha") + try: + color = declared_color(s) + if kind != "video" and color is None: + raise ValueError("raw input has no color metadata; declare an explicit color setting") + if kind == "browser" and color != Color(): + raise ValueError("browser input supports SDR only") + except ValueError as e: + raise ConfigError(f"{where}: {e}") from e + return Source(sid, kind, str(s.get("url", s.get("path"))), width=int(s.get("width", 0)), + height=int(s.get("height", 0)), fps=int(s.get("fps", fps)) if kind == "browser" else 0, + loop=bool(s.get("loop", True)), filter_graph=source_filter, + filter_output_format=filter_format, **(color.tags if color else {})) + + +def _parse_rendition(r: Dict[str, Any], where: str, canvas_w: int, canvas_h: int, fps: int) -> Rendition: + rid = str(r.get("id", "")) + if not rid: + raise ConfigError(f"{where}: id required") + base = Rendition(rid, width=canvas_w, height=canvas_h, fps=fps) + # Every other key coerces to its field's type; unknown keys are ignored. + rendition = replace(base, **{f.name: type(getattr(base, f.name))(r[f.name]) + for f in fields(Rendition) if f.name in r and f.name != "id"}) + if rendition.tonemap and rendition.tonemap not in OPERATORS: + raise ConfigError(f"{where}: unsupported tone-map operator") + if rendition.color: + try: + Color.parse(rendition.color) + except ValueError as e: + raise ConfigError(f"{where}: {e}") from e + if rendition.width <= 0 or rendition.height <= 0: + raise ConfigError(f"{where}: width and height must be positive") + if rendition.fps <= 0 or rendition.bitrate_kbps <= 0: + raise ConfigError(f"{where}: fps and bitrate_kbps must be positive") + if rendition.tonemap_peak < 2.03 or rendition.tonemap_desat < 0 or rendition.tonemap_param < 0: + raise ConfigError(f"{where}: tonemap_peak >= 2.03 (203 nits), tonemap_desat >= 0 and tonemap_param >= 0") + if rendition.max_cll < 0 or rendition.max_fall < 0 or rendition.max_fall > max(rendition.max_cll, rendition.tonemap_peak * 100): + raise ConfigError(f"{where}: max_cll and max_fall must be non-negative nits, max_fall no higher than MaxCLL") + if rendition.tonemap == "mobius" and rendition.tonemap_param >= 1: + # The Möbius shoulder maps [knee, peak] onto [knee, 1]; at knee 1.0 it degenerates to clip. + raise ConfigError(f"{where}: mobius tonemap_param must be below 1.0 (0.9 keeps 90% of SDR white linear)") + if rendition.fps > fps: + raise ConfigError(f"{where}: fps {rendition.fps} exceeds the canvas rate {fps}; " + "a rendition can only re-time the program downwards") + wanted = str(r.get("aspect", "")) + if wanted and wanted != rendition.aspect: + raise ConfigError(f"{where}: {rendition.width}x{rendition.height} is " + f"{rendition.aspect}, not {wanted}") + return rendition + + +def _parse_control(control: Any, wipes: List[Wipe]) -> Dict[str, Any]: + if not isinstance(control, dict): + raise ConfigError("control must be an object") + default_wipe = str(control.get("default_wipe", wipes[0].id if wipes else "")) + if default_wipe and not any(w.id == default_wipe for w in wipes): + raise ConfigError(f"control.default_wipe '{default_wipe}' is not a wipe") + transition = str(control.get("transition", DEFAULT_TRANSITION)) + if transition not in TRANSITIONS: + raise ConfigError(f"control.transition must be one of {TRANSITIONS}") + if transition == "wipe" and not wipes: + raise ConfigError("control.transition 'wipe' needs a wipe library") + fade_seconds = float(control.get("fade_seconds", DEFAULT_FADE_SECONDS)) + if fade_seconds <= 0: + raise ConfigError("control.fade_seconds must be positive") + return {"direct": bool(control.get("direct", True)), "fade_seconds": fade_seconds, + "transition": transition, "default_wipe": default_wipe} + + +def _unique(items: List[Any], where: str, label: str = "id") -> None: + if any(x.id == items[-1].id for x in items[:-1]): + raise ConfigError(f"{where}: duplicate {label} '{items[-1].id}'") + + def parse(doc: Dict[str, Any]) -> MixerConfig: for key in ("canvas", "sources", "scenes"): if key not in doc: @@ -163,72 +292,51 @@ def parse(doc: Dict[str, Any]) -> MixerConfig: raise ConfigError("canvas needs integer width, height and fps") from None if canvas_w <= 0 or canvas_h <= 0 or fps <= 0: raise ConfigError("canvas width, height and fps must be positive") + working_format = str(canvas.get("working_format", "nv12")) + if working_format not in WORKING_FORMATS: + raise ConfigError(f"canvas.working_format must be one of {WORKING_FORMATS}") + latency_ms = canvas.get("latency_ms") + if latency_ms is not None: + try: + latency_ms = float(latency_ms) + except (TypeError, ValueError): + raise ConfigError("canvas.latency_ms must be a number of milliseconds") from None + if not 0 <= latency_ms < 6000 / fps: + raise ConfigError(f"canvas.latency_ms must be at least 0 and below six frames ({6000 / fps:.1f} ms at {fps} fps)") + try: + out_color = declared_color(canvas) or Color() + out_color.validate_format(working_format) + except ValueError as e: + raise ConfigError(f"canvas: {e}") from e sources: List[Source] = [] locations: Dict[Tuple[str, str], str] = {} for i, s in enumerate(doc["sources"]): where = f"sources[{i}]" - sid, kind = str(s.get("id", "")), s.get("kind") - if not sid or "#" in sid: - raise ConfigError(f"{where}: id required (no '#')") - if any(x.id == sid for x in sources): - raise ConfigError(f"{where}: duplicate id '{sid}'") - if kind == "browser": - if not all(k in s for k in ("url", "width", "height")): - raise ConfigError(f"{where}: browser source needs url, width, height") - src = Source(sid, kind, str(s["url"]), int(s["width"]), int(s["height"]), - int(s.get("fps", fps)), bool(s.get("loop", True))) - elif kind == "video": - if "path" not in s: - raise ConfigError(f"{where}: video source needs path") - src = Source(sid, kind, str(s["path"]), int(s.get("width", 0)), int(s.get("height", 0)), - 0, bool(s.get("loop", True))) - else: - raise ConfigError(f"{where}: kind must be browser or video") - key = (kind, src.location) + sources.append(_parse_source(s, where, fps)) + _unique(sources, where) + key = (sources[-1].kind, sources[-1].location) if key in locations: - raise ConfigError(f"{where}: '{src.location}' already declared as '{locations[key]}'; " + raise ConfigError(f"{where}: '{key[1]}' already declared as '{locations[key]}'; " "reference that id instead (one decode per unique source)") - locations[key] = sid - sources.append(src) + locations[key] = sources[-1].id if not sources: raise ConfigError("sources must not be empty") + if len(sources) > MAX_SOURCES: + raise ConfigError(f"at most {MAX_SOURCES} sources per show: each is a compositor pad and the pad mask is 32 bits") renditions: List[Rendition] = [] for i, r in enumerate(doc.get("renditions", [])): - where = f"renditions[{i}]" - rid = str(r.get("id", "")) - if not rid: - raise ConfigError(f"{where}: id required") - if any(x.id == rid for x in renditions): - raise ConfigError(f"{where}: duplicate id '{rid}'") - rendition = Rendition( - rid, str(r.get("target", "janus")), - int(r.get("width", canvas_w)), int(r.get("height", canvas_h)), - int(r.get("fps", fps)), int(r.get("bitrate_kbps", 3000)), - str(r.get("codec", "h264_nvenc")), str(r.get("profile", "baseline")), - str(r.get("preset", "p7")), int(r.get("port", 0))) - if rendition.width <= 0 or rendition.height <= 0: - raise ConfigError(f"{where}: width and height must be positive") - if rendition.fps <= 0 or rendition.bitrate_kbps <= 0: - raise ConfigError(f"{where}: fps and bitrate_kbps must be positive") - if rendition.fps > fps: - raise ConfigError(f"{where}: fps {rendition.fps} exceeds the canvas rate {fps}; " - "a rendition can only re-time the program downwards") - wanted = str(r.get("aspect", "")) - if wanted and wanted != rendition.aspect: - raise ConfigError(f"{where}: {rendition.width}x{rendition.height} is " - f"{rendition.aspect}, not {wanted}") - renditions.append(rendition) + renditions.append(_parse_rendition(r, f"renditions[{i}]", canvas_w, canvas_h, fps)) + _unique(renditions, f"renditions[{i}]") wipes: List[Wipe] = [] for i, w in enumerate(doc.get("wipes", [])): if "id" not in w or "path" not in w: raise ConfigError(f"wipes[{i}]: id and path required") - if any(x.id == w["id"] for x in wipes): - raise ConfigError(f"wipes[{i}]: duplicate id '{w['id']}'") wipes.append(Wipe(str(w["id"]), str(w["path"]), float(w.get("duration_seconds", 0)), str(w.get("name", "")))) + _unique(wipes, f"wipes[{i}]") if "wipe_dir" in doc: # Every clip in the directory joins the library under its file name. # Entries declared above keep their id, name and duration. @@ -241,8 +349,6 @@ def parse(doc: Dict[str, Any]) -> MixerConfig: where = f"scenes[{i}]" if "id" not in sc or not isinstance(sc.get("items"), list) or not sc["items"]: raise ConfigError(f"{where}: id and a non-empty items list required") - if any(x.id == sc["id"] for x in scenes): - raise ConfigError(f"{where}: duplicate scene id '{sc['id']}'") items: List[Item] = [] for j, it in enumerate(sc["items"]): iw = f"{where}.items[{j}]" @@ -254,29 +360,19 @@ def parse(doc: Dict[str, Any]) -> MixerConfig: crop = _rect(it["crop"], iw + ".crop") if "crop" in it else None items.append(Item(str(it["source"]), _rect(it["dst"], iw + ".dst"), fit, crop)) scenes.append(Scene(str(sc["id"]), tuple(items))) + _unique(scenes, where, "scene id") if not scenes: raise ConfigError("scenes must not be empty") initial = str(doc.get("initial_scene", scenes[0].id)) if not any(s.id == initial for s in scenes): raise ConfigError(f"initial_scene '{initial}' is not a scene") - control = doc.get("control", {}) - if not isinstance(control, dict): - raise ConfigError("control must be an object") - default_wipe = str(control.get("default_wipe", wipes[0].id if wipes else "")) - if default_wipe and not any(w.id == default_wipe for w in wipes): - raise ConfigError(f"control.default_wipe '{default_wipe}' is not a wipe") - transition = str(control.get("transition", DEFAULT_TRANSITION)) - if transition not in TRANSITIONS: - raise ConfigError(f"control.transition must be one of {TRANSITIONS}") - if transition == "wipe" and not wipes: - raise ConfigError("control.transition 'wipe' needs a wipe library") - fade_seconds = float(control.get("fade_seconds", DEFAULT_FADE_SECONDS)) - if fade_seconds <= 0: - raise ConfigError("control.fade_seconds must be positive") - return MixerConfig(canvas_w, canvas_h, fps, tuple(sources), tuple(scenes), tuple(wipes), - tuple(renditions), initial, - bool(control.get("direct", True)), fade_seconds, transition, default_wipe) + wipe_color = str(doc.get("wipe_color", "")) + if wipe_color and wipe_color not in TRANSFER_TAGS: + raise ConfigError("wipe_color must be sdr, hlg or pq") + return MixerConfig(canvas_w, canvas_h, fps, tuple(sources), tuple(scenes), tuple(wipes), tuple(renditions), + initial_scene=initial, working_format=working_format, latency_ms=latency_ms, out_color=out_color, + wipe_color=wipe_color, **_parse_control(doc.get("control", {}), wipes)) WIPE_SUFFIXES = (".mov", ".webm", ".mkv", ".mp4", ".avi", ".png", ".gif") diff --git a/avpmixer/control.py b/pyplumber/mixer/control.py similarity index 100% rename from avpmixer/control.py rename to pyplumber/mixer/control.py diff --git a/avpmixer/dmabuf_inputs.py b/pyplumber/mixer/dmabuf_inputs.py similarity index 99% rename from avpmixer/dmabuf_inputs.py rename to pyplumber/mixer/dmabuf_inputs.py index dc097bd5..b900654f 100644 --- a/avpmixer/dmabuf_inputs.py +++ b/pyplumber/mixer/dmabuf_inputs.py @@ -4,7 +4,7 @@ ``dmabuf_cuda_input_nodes`` turns it into a CUDA edge on the shared monotonic clock, snapped to the 1/fps grid, exactly as ``demos/dmabuf-browser/graph/dmabuf_browser_common.py`` does for that demo -(kept there unchanged because the demo's runtime image has no ``avpmixer``). +(kept there unchanged because the demo's runtime image has no ``pyplumber.mixer``). The REST helpers open the windows and wait for their sockets. """ diff --git a/avpmixer/graph.py b/pyplumber/mixer/graph.py similarity index 84% rename from avpmixer/graph.py rename to pyplumber/mixer/graph.py index a2124eb2..7e403a22 100644 --- a/avpmixer/graph.py +++ b/pyplumber/mixer/graph.py @@ -5,7 +5,7 @@ Typical usage ------------- from pyplumber import AVPlumber - from avpmixer import MixerGraphBuilder + from pyplumber.mixer import MixerGraphBuilder from pyplumber.node import InputRec, Demux, DecVideo, Realtime, ForceFPS avp = AVPlumber() @@ -66,6 +66,7 @@ ) +from .color import Color, conversion_graph from . import clipcache from .models import MixerScene, MixerSource from .prewarm import TransitionPrewarm @@ -100,6 +101,9 @@ def __init__( defer_output: bool = False, keyframe_node: Optional[str] = None, cache_wipes_mb: Optional[float] = None, # None keeps the decode-per-take chain + working_format: str = "nv12", # compositor/transition sw_format + color="sdr", + wipe_color=None, ): if switch_margin_ms < 0: raise ValueError("switch_margin_ms must be >= 0") @@ -114,11 +118,17 @@ def __init__( # Triggered when a transition reaches the output, so receivers do not wait # for the next periodic keyframe to see the new scene. self.keyframe_node = keyframe_node - # When set, wipe clips are held decoded in GPU memory (avpmixer.clipcache) + # When set, wipe clips are held decoded in GPU memory (pyplumber.mixer.clipcache) # and a take replays them instead of opening and decoding the file again. self.cache_wipes_mb = cache_wipes_mb self.defer_initial_routes = defer_initial_routes self.latency_ms = latency_ms + self.working_format = working_format + self.color = Color.parse(color) + self.color.validate_format(working_format) + self.wipe_color = Color.parse(wipe_color) if wipe_color is not None else None + if self.wipe_color is not None and self.wipe_color != Color(): + raise ValueError("Alpha wipes currently require SDR; HDR alpha decode is unsupported") self._output_started = not defer_output self._transition_prewarm = TransitionPrewarm(avp, self.name, self.timeline) @@ -141,9 +151,16 @@ def add_source( pre_otm_edge: str, input_group: str, default_graph: Optional[str] = None, + *, color=None, pixel_format=None, packed_rgb=False, ) -> "MixerGraphBuilder": """Register one camera source. + Color defaults to strict decoded-frame metadata. An explicit color + contract overrides source tags. Sources sharing pre_otm_edge are aliases: + the builder converts once, then fans out to their scene-slot routers. + pixel_format is required for CUDA layouts other than NV12/P010. + packed_rgb keeps alpha for fused compositing and requires explicit SDR. + The source's one_to_many and per-slot crop-scale nodes will be created in *input_group* during build() so that they restart together with the input decode chain. @@ -169,8 +186,13 @@ def add_source( raise RuntimeError("Cannot add sources after build()") if name in self._source_index: raise ValueError(f"Source '{name}' already registered") + if color is not None: + color = Color.parse(color) + if packed_rgb and color != Color(): + raise ValueError(f"Source '{name}': packed RGB requires an explicit SDR color setting") idx = len(self._sources) - self._sources.append(MixerSource(name, pre_otm_edge, input_group, default_graph)) + self._sources.append(MixerSource(name, pre_otm_edge, input_group, default_graph, + color=color, pixel_format=pixel_format, packed_rgb=packed_rgb)) self._source_index[name] = idx return self @@ -184,8 +206,13 @@ def add_routed_source( route_output_label_a: str, route_output_label_b: str, default_graph: Optional[str] = None, + *, color=None, ) -> "MixerGraphBuilder": - """Register a source whose slot filters are fed by a native preheat router.""" + """Register a source whose slot filters are fed by a native preheat router. + + An explicit color contract applies to every camera routed into this + source. Otherwise, normalization resolves each frame's color metadata. + """ if self._built: raise RuntimeError("Cannot add sources after build()") if name in self._source_index: @@ -201,6 +228,7 @@ def add_routed_source( route_router=route_router, route_output_label_a=route_output_label_a, route_output_label_b=route_output_label_b, + color=Color.parse(color) if color is not None else None, )) self._source_index[name] = idx return self @@ -241,6 +269,9 @@ def define_scene( After build, it also emits ``mixer.scene`` so runtime policies can reuse generic scene names with different source-slot assignments. """ + sources = {source: ({**spec, "graph": self._normalized_graph(spec["graph"])} + if spec.get("graph") else dict(spec)) + for source, spec in sources.items()} unknown = [s for s in sources if s not in self._source_index] if unknown: raise ValueError(f"Scene '{name}' references unknown source(s): {unknown}") @@ -254,6 +285,15 @@ def define_scene( self.avp.executeCommandsFromString(self._scene_command(name, scene)) return self + def _normalized_graph(self, graph: str) -> str: + """Keep scene geometry filters in the canvas storage format. + + Color conversions belong before source fan-out. The compositor checks + the contract again so a custom scene filter cannot silently retag pixels. + """ + suffix = f"scale_cuda=format={self.working_format}" + return graph if not graph or graph.endswith(suffix) else f"{graph},{suffix}" + def set_initial_scene(self, scene_name: str, slot: str = "A") -> "MixerGraphBuilder": """Declare which scene starts on PGM. @@ -431,8 +471,53 @@ def _active_inputs_mask(self, scene: MixerScene) -> int: mask |= 1 << idx return mask + def _prepare_source_edges(self): + shared = {} + result = {} + for source in self._sources: + if source.route_router is None: + shared.setdefault(source.pre_otm_edge, []).append(source) + else: + # Routed edges may change cameras at runtime. Resolve color from + # every frame after the router, before per-slot scene geometry. + for slot, edge in (("a", source.pre_filter_edge_a), ("b", source.pre_filter_edge_b)): + result[(source.name, slot)] = self._color_edge(source, edge, f"{source.name}_{slot}") + for edge, sources in shared.items(): + first = sources[0] + contract = lambda s: (s.input_group, s.color, s.pixel_format, s.packed_rgb) + if any(contract(s) != contract(first) for s in sources): + raise ValueError(f"Conflicting color contracts for shared input edge {edge!r}") + prepared = self._color_edge(first, edge, first.name) + outputs = [prepared] + if len(sources) > 1: + outputs = [self._e(f"{s.name}_color_alias") for s in sources] + self.avp.addNode(OneToMany({ + "name": self._n(f"color_alias_{first.name}"), "src": prepared, "dst": outputs, + "outputs": (1 << len(outputs)) - 1, "group": first.input_group, + })) + result.update((s.name, output) for s, output in zip(sources, outputs)) + return result + + def _color_edge(self, source, edge, label): + if source.packed_rgb: + if self.working_format not in ("nv12", "p010le", "p210le"): + raise ValueError("Packed RGB compositing requires a semiplanar canvas (NV12/P010/P210)") + graph = source.color.setparams + else: + graph = conversion_graph(self.color, self.working_format, + source=source.color, source_format=source.pixel_format) + output = self._e(f"{label}_color") + self.avp.addNode(FilterVideo({ + "name": self._n(f"color_{label}"), "src": edge, "dst": output, + "graph": graph, + "hwaccel": self.hwaccel, "group": source.input_group, + "defer_preliminary_init": True, + })) + return output + def _build_per_source_nodes(self) -> None: """Create one_to_many + per-slot filter_video nodes for every source.""" + prepared_edges = self._prepare_source_edges() initial_scene = self._initial_scene_def() pgm_slot_bit = 0 if self._initial_pgm_slot == "A" else 1 @@ -447,7 +532,7 @@ def _build_per_source_nodes(self) -> None: self.avp.addNode(OneToMany({ "type": "one_to_many", "name": self._n(f"otm_{src.name}"), - "src": src.pre_otm_edge, + "src": prepared_edges[src.name], "dst": [self._e(f"{src.name}_a"), self._e(f"{src.name}_b")], "outputs": outputs_init, "timeline": self.timeline, @@ -456,8 +541,10 @@ def _build_per_source_nodes(self) -> None: slot_a_edge = self._e(f"{src.name}_a") slot_b_edge = self._e(f"{src.name}_b") else: - slot_a_edge = src.pre_filter_edge_a - slot_b_edge = src.pre_filter_edge_b + slot_a_edge = prepared_edges[(src.name, "a")] + slot_b_edge = prepared_edges[(src.name, "b")] + src.pre_filter_edge_a = slot_a_edge + src.pre_filter_edge_b = slot_b_edge # Default scale: fit to canvas. MixerOrchestrator rewrites the # graph string on every scene switch via node.param.set + auto_restart. @@ -465,6 +552,7 @@ def _build_per_source_nodes(self) -> None: f"scale_cuda=w={self.canvas_w}:h={self.canvas_h}:interp_algo=lanczos" ) default_graph = fallback_graph if src.default_graph is None else src.default_graph + default_graph = self._normalized_graph(default_graph) if not default_graph: continue @@ -499,7 +587,8 @@ def _build_compositors(self) -> None: "hwaccel": self.hwaccel, "width": self.canvas_w, "height": self.canvas_h, - "sw_format": "nv12", + "sw_format": self.working_format, + "color": self.color.transfer, "fps": self._fps_str(), **timing, "scale": any(source.default_graph == "" for source in self._sources), @@ -631,7 +720,7 @@ def _build_wipe_subgraph(self) -> None: # Upload the clip at its own size and let the compositor scale it on # the GPU. Resizing to the canvas on a CPU thread cost two thirds of # this chain and made the compositor miss 60 Hz ticks during a wipe. - "graph": "format=rgba,hwupload", + "graph": (self.wipe_color.setparams + "," if self.wipe_color else "") + "format=rgba,hwupload", "hwaccel": self.hwaccel, "group": load_group, })) @@ -662,7 +751,8 @@ def _build_wipe_subgraph(self) -> None: "src": [self._e("final_wipe_in"), self._e("wipe_rt_fps_out")], "dst": self._e("wipe_overlay_out"), "hwaccel": self.hwaccel, - "width": W, "height": H, "sw_format": "nv12", "fps": fps_str, "scale": True, + "width": W, "height": H, "sw_format": self.working_format, + "color": self.color.transfer, "fps": fps_str, "scale": True, "layers": [{"dst_x": 0, "dst_y": 0, "dst_w": W, "dst_h": H}, {"dst_x": 0, "dst_y": 0, "dst_w": W, "dst_h": H, "z": 1, "blend": True}], "active_inputs": 3, diff --git a/avpmixer/inputs.py b/pyplumber/mixer/inputs.py similarity index 53% rename from avpmixer/inputs.py rename to pyplumber/mixer/inputs.py index 01727903..4cf92825 100644 --- a/avpmixer/inputs.py +++ b/pyplumber/mixer/inputs.py @@ -11,6 +11,10 @@ from typing import Optional +# Input watchdog in seconds: effectively never, so a looped or paused file input is not torn down. +INPUT_TIMEOUT_S = 3_942_000_000 + + def build_input(avp, api, tag: str, url: str, *, group: str, fps: int, fps_den: int = 1, hwaccel: str = "@gpu", loop: bool = False, input_params: Optional[dict] = None, @@ -32,7 +36,7 @@ def build_input(avp, api, tag: str, url: str, *, group: str, fps: int, fps_den: sync = {} if sync_team is None else {"sync_team": sync_team} avp.addNode(api.InputRec({ "name": f"input_{tag}", "url": url, "dst": edge("packets"), "loop": loop, - "initial_timeout": 20, "timeout": 3_942_000_000, "group": group, + "initial_timeout": 20, "timeout": INPUT_TIMEOUT_S, "group": group, **(input_params or {}), })) avp.addNode(api.Demux({ @@ -58,13 +62,57 @@ def build_input(avp, api, tag: str, url: str, *, group: str, fps: int, fps_den: "team": pause_team, "group": group, **sync, **(pause_params or {}), })) realtime_src = edge("paused") + return _pace(avp, api, tag, realtime_src, fps=fps, fps_den=fps_den, group=group, + sync_team=sync_team, realtime_params=realtime_params) + + +def _pace(avp, api, tag: str, src: str, *, fps: int, fps_den: int, group: str, + sync_team: Optional[str] = None, realtime_params: Optional[dict] = None) -> str: + """``realtime(set_pts) -> force_fps`` tail shared by every source chain: + rebase onto the host clock, then fix the rate. Returns ``input__fps``.""" avp.addNode(api.Realtime({ - "name": f"realtime_{tag}", "src": realtime_src, "dst": edge("realtime"), + "name": f"realtime_{tag}", "src": src, "dst": f"input_{tag}_realtime", "set_pts": True, "group": group, **({} if sync_team is None else {"team": sync_team}), **(realtime_params or {}), })) avp.addNode(api.ForceFPS({ - "name": f"fps_{tag}", "src": edge("realtime"), "dst": edge("fps"), + "name": f"fps_{tag}", "src": f"input_{tag}_realtime", "dst": f"input_{tag}_fps", "fps": f"{fps}/{fps_den}", "group": group, })) - return edge("fps") + return f"input_{tag}_fps" + + +def v210_row_stride(width: int) -> int: + """Standard v210 row stride: ceil(width/48) * 128 bytes.""" + return ((width + 47) // 48) * 128 + + +def build_v210_input(avp, api, tag: str, path: str, *, width: int, height: int, group: str, + fps: int, fps_den: int = 1, hwaccel: str = "@gpu", loop: bool = False, + color: Optional[dict] = None) -> str: + """Headerless packed v210 file -> GPU unpack (P210, 10-bit 4:2:2) -> paced output edge. + + ``input_rec -> demux -> v210_to_cuda -> realtime(set_pts) -> force_fps``. + The packed bytes carry no metadata, so the color contract (HLG/BT.2020 for + the HDR sources) is supplied here and stamped on the CUDA frames. + """ + edge = lambda suffix: f"input_{tag}_{suffix}" # noqa: E731 + restart = {} if loop else {"auto_restart": "group"} + stride = v210_row_stride(width) + avp.addNode(api.InputRec({ + "name": f"input_{tag}", "url": path, "dst": edge("packets"), "loop": loop, + "format": "rawvideo", "initial_timeout": 20, "timeout": INPUT_TIMEOUT_S, "group": group, + "options": {"pixel_format": "gray", "video_size": f"{stride}x{height}", + "framerate": f"{fps}/{fps_den}"}, + })) + avp.addNode(api.Demux({ + "name": f"demux_{tag}", "src": edge("packets"), "routing": {"v:0": edge("packed")}, + "wait_for_keyframe": False, "group": group, **restart, + })) + avp.addNode(api.V210ToCuda({ + "name": f"unpack_{tag}", "src": edge("packed"), "dst": edge("cuda"), + "hwaccel": hwaccel, "width": width, "height": height, "stride": stride, + "fps": f"{fps}/{fps_den}", "timebase": "1/90000", "sw_format": "p210le", + "group": group, **restart, **(color or {}), + })) + return _pace(avp, api, tag, edge("cuda"), fps=fps, fps_den=fps_den, group=group) diff --git a/pyplumber/mixer/janus.py b/pyplumber/mixer/janus.py new file mode 100644 index 00000000..a829a7dd --- /dev/null +++ b/pyplumber/mixer/janus.py @@ -0,0 +1,141 @@ +"""Video-only H.264/HEVC RTP output to a Janus Streaming mountpoint, shared by demos.""" + +from __future__ import annotations + +from dataclasses import dataclass + +from .color import TEN_BIT_FORMATS + +RTP_PACKET_SIZE = 1_200 +DEFAULT_KEYFRAME_MIN_INTERVAL_MS = 150 + + +@dataclass(frozen=True) +class JanusVideoConfig: + host: str = "127.0.0.1" + video_port: int = 5004 + payload_type: int = 96 + ssrc: int = 0x41565001 + bitrate_kbps: int = 4_500 + rtcp_bind: str = "0.0.0.0" + rtcp_port: int = 0 + keyframe_min_interval_ms: int = DEFAULT_KEYFRAME_MIN_INTERVAL_MS + + def __post_init__(self) -> None: + if not self.host: + raise ValueError("Janus host is required") + if not 1 <= self.video_port < 65535: + raise ValueError("Janus video port and its RTCP pair must be valid") + if not 0 <= self.payload_type <= 127: + raise ValueError("RTP payload type must be between 0 and 127") + if not 0 <= self.ssrc <= 0xFFFFFFFF: + raise ValueError("RTP SSRC must be a 32-bit unsigned integer") + if self.bitrate_kbps <= 0: + raise ValueError("Janus bitrate must be positive") + if not 0 <= self.rtcp_port <= 65535: + raise ValueError("RTCP port must be between 0 and 65535") + if (type(self.keyframe_min_interval_ms) is not int + or not 0 <= self.keyframe_min_interval_ms <= 2_147_483_647): + raise ValueError("keyframe_min_interval_ms must be a non-negative integer") + + @property + def rtcp_port_remote(self) -> int: + return self.video_port + 1 + + @property + def rtp_url(self) -> str: + return (f"rtp://{self.host}:{self.video_port}?pkt_size={RTP_PACKET_SIZE}" + f"&rtcp_port={self.rtcp_port_remote}") + + +JANUS_KEYFRAME_NODE = "janus_force_keyframe" +KEYFRAME_COMMAND = f"node.object.set {JANUS_KEYFRAME_NODE} trigger true" + + +class RtcpFeedbackGroup: + """Manage feedback for multiple independently encoded renditions.""" + + def __init__(self, listeners): + self.listeners = tuple(listeners) + + def start(self): + started = [] + try: + for listener in self.listeners: + listener.start() + started.append(listener) + except Exception: + for listener in reversed(started): + listener.stop() + raise + + def stop(self): + for listener in reversed(self.listeners): + listener.stop() + + +def add_nodes(avp, api, specs, **defaults): + """Build and register nodes from ``(node_type, params)`` pairs, filling in shared + parameters (``group``, ``auto_restart``) a node does not set itself.""" + for node_type, params in specs: + avp.addNode(getattr(api, node_type)({**defaults, **params})) + + +def build_janus_output(avp, api, src_edge: str, janus: JanusVideoConfig, *, fps: int, + width: int, height: int, hwaccel: str = "@gpu", fps_den: int = 1, + group: str = "output", codec: str = "", profile: str = "", + preset: str = "p7", enc_format: str = "nv12", color=None, + hdr_metadata=None, prefix: str = "janus"): + """Add ``force_fps -> keyframe -> nvenc -> bsf -> rtp mux -> output``; return the RTCP listener. + + Defaults to HEVC (Main/Main10), which current Safari and Chrome negotiate + over WebRTC on hardware-decode machines: ~2x the efficiency of H.264 at the + same bitrate and it carries 10-bit + HDR signaling. ``enc_format`` is the + encoder's CUDA input (``p010le`` keeps 10-bit, ``nv12`` is 8-bit) and + ``color`` supplies the VUI so an HLG program signals BT.2020/arib-std-b67. + """ + node_name = lambda suffix: f"{prefix}_{suffix}" + keyframe_node = node_name("force_keyframe") + bitrate = f"{janus.bitrate_kbps}k" + ten_bit = enc_format in TEN_BIT_FORMATS + if not codec: + codec = "hevc_nvenc" if ten_bit else "h264_nvenc" # HEVC only when the input is 10-bit + if not profile: + profile = ("main10" if ten_bit else "main") if "hevc" in codec else "baseline" + add_nodes(avp, api, [ + ("ForceFPS", {"name": node_name("fps"), "src": src_edge, "dst": node_name("fps"), + "fps": f"{fps}/{fps_den}"}), + ("ForceKeyFrame", {"name": keyframe_node, "src": node_name("fps"), "dst": node_name("keyframed"), + "interval_sec": "1/1", "min_interval_ms": janus.keyframe_min_interval_ms}), + ("AssumeVideoFormat", {"name": node_name("format"), "src": node_name("keyframed"), "dst": node_name("video"), + "width": width, "height": height, "pixel_format": "cuda", + "real_pixel_format": enc_format}), + ("EncVideo", { + "name": node_name("encoder"), "src": node_name("video"), "dst": node_name("encoded"), + "codec": codec, "hwaccel": hwaccel, + **({"hdr_metadata": hdr_metadata} if hdr_metadata else {}), + "options": { + "b": bitrate, "maxrate": bitrate, "bufsize": bitrate, "g": max(1, round(fps / fps_den)), "bf": 0, + # p5..p7 are NVENC's quality presets; with tune=ull it stays a + # one-pass, no-lookahead, no-reordering encode, so the extra quality + # costs GPU time rather than latency. B-frames stay off: they need + # reordering, which WebRTC's jitter budget will not absorb. + "preset": preset, "profile": profile, "tune": "ull", "rc": "cbr", + "rc-lookahead": 0, "zerolatency": 1, "delay": 0, "forced-idr": 1, + "no-scenecut": 1, "strict_gop": 1, "aud": 1, "spatial-aq": 1, "temporal-aq": 0, + **(color or {}), + }, + }), + ("Bsf", {"name": node_name("repeat_headers"), "src": node_name("encoded"), "dst": node_name("repeat_headers"), + "bsf": "dump_extra=freq=keyframe"}), + ("Mux", {"name": node_name("mux"), "src": [node_name("repeat_headers")], "dst": node_name("video_rtp_mux"), + "ts_sort_wait": 0, "auto_restart": "on", "on_error": "panic"}), + ("Output", {"name": node_name("rtp_output"), "src": node_name("video_rtp_mux"), "url": janus.rtp_url, + "format": "rtp", "auto_restart": "on", "on_error": "panic", + "options": {"payload_type": janus.payload_type, "rtpflags": "skip_rtcp", "ssrc": janus.ssrc}}), + ], group=group, auto_restart="panic") + return api.RtcpFeedbackListener( + bind_host=janus.rtcp_bind, bind_port=janus.rtcp_port, janus_host=janus.host, + janus_rtcp_port=janus.rtcp_port_remote, media_ssrc=janus.ssrc, + on_keyframe_request=lambda _request: avp.executeCommandsFromString(f"node.object.set {keyframe_node} trigger true"), + ) diff --git a/avpmixer/models.py b/pyplumber/mixer/models.py similarity index 90% rename from avpmixer/models.py rename to pyplumber/mixer/models.py index 5e548c18..7bd16685 100644 --- a/avpmixer/models.py +++ b/pyplumber/mixer/models.py @@ -15,6 +15,9 @@ class MixerSource: route_router: Optional[str] = None route_output_label_a: Optional[str] = None route_output_label_b: Optional[str] = None + color: Any = None + pixel_format: Optional[str] = None + packed_rgb: bool = False @dataclass diff --git a/avpmixer/prewarm.py b/pyplumber/mixer/prewarm.py similarity index 100% rename from avpmixer/prewarm.py rename to pyplumber/mixer/prewarm.py diff --git a/pyplumber/node.py b/pyplumber/node.py index 9463d1ab..3e76b8f3 100644 --- a/pyplumber/node.py +++ b/pyplumber/node.py @@ -125,6 +125,10 @@ class DecVideo(InternalNode): TYPE = "dec_video" +class V210ToCuda(InternalNode): + TYPE = "v210_to_cuda" + + class DecAudio(InternalNode): TYPE = "dec_audio" diff --git a/src/avplumber.cpp b/src/avplumber.cpp index 6e58100a..d80ebcc7 100644 --- a/src/avplumber.cpp +++ b/src/avplumber.cpp @@ -29,8 +29,14 @@ #include "PTSCorrectorCommon.hpp" #include "rest_client.hpp" #include "SharedTimeline.hpp" -#include "mixer/MixerState.hpp" -#include "mixer/mixer_orchestrator.hpp" +#include "mixer/primitives/MixerState.hpp" +#include "mixer/orchestrator/MixerOrchestrator.hpp" +using avp::mixer::MixerOrchestrator; +using avp::mixer::MixerState; +using avp::mixer::SceneControl; +using avp::mixer::SceneDefinition; +using avp::mixer::SourceLayout; +using avp::mixer::TransitionScheduler; #include "CommandTiming.hpp" #include #ifdef EMBED_IN_OBS @@ -344,6 +350,7 @@ class ControlImpl { detached_threads_.clear(); } if (manager_) { + InstanceSharedObjectsDestructors::callInstanceDestructors(&manager_->instanceData()); if (manager_.use_count() <= 1) { logstream << "Destroying NodeManager"; } else { @@ -815,7 +822,7 @@ class ControlImpl { auto mixerOrchestrator = [this](const std::string& mixer_name) { auto state = InstanceSharedObjects::get(manager_->instanceData(), mixer_name); auto tl = InstanceSharedObjects::get(manager_->instanceData(), state->timeline_name); - auto scheduler = InstanceSharedObjects::get(manager_->instanceData(), mixer_name); + auto scheduler = InstanceSharedObjects::get(manager_->instanceData(), mixer_name); return MixerOrchestrator(manager_->shared_from_this(), state, tl, scheduler); }; diff --git a/src/avplumber_pybind.cpp b/src/avplumber_pybind.cpp index 5dfebb45..f7843fe5 100644 --- a/src/avplumber_pybind.cpp +++ b/src/avplumber_pybind.cpp @@ -598,7 +598,13 @@ PYBIND11_MODULE(_avplumber, m) { }); }, py::arg("callback")) .def("setReady", &AVPlumber::setReady) - .def("shutdown", &AVPlumber::shutdown) + .def("shutdown", [](AVPlumber &avp) { + // Keep Python callbacks alive until the GIL is reacquired, while + // allowing worker process/doStop hooks to run during the joins. + auto manager = avp.manager(); + py::gil_scoped_release release; + avp.shutdown(); + }) .def("mainLoop", &AVPlumber::mainLoop) .def("stopMainLoop", &AVPlumber::stopMainLoop) .def("heartbeat", &AVPlumber::heartbeat) @@ -693,11 +699,15 @@ PYBIND11_MODULE(_avplumber, m) { }) .def_property_readonly("parameters", &NodeWrapper::parameters) .def("getObject", &NodeWrapper::getObject) - .def("start", &NodeWrapper::start) - .def("stop", &NodeWrapper::stop) - .def("interrupt", &NodeWrapper::interrupt, py::arg("optional") = false) - .def("stopAndWait", &NodeWrapper::stopAndWait) - .def("join", &NodeWrapper::join) + // These block on node threads (which may need the GIL for Python nodes + // or callbacks); release it like NodeGroup.startNodes/stopNodes do, so a + // slow stop cannot freeze the interpreter. + .def("start", &NodeWrapper::start, py::call_guard()) + .def("stop", &NodeWrapper::stop, py::call_guard()) + .def("interrupt", &NodeWrapper::interrupt, py::arg("optional") = false, + py::call_guard()) + .def("stopAndWait", &NodeWrapper::stopAndWait, py::call_guard()) + .def("join", &NodeWrapper::join, py::call_guard()) .def_property_readonly("isWorking", [](NodeWrapper &nw) { return nw.isWorking(); }) ; diff --git a/src/graph_core.hpp b/src/graph_core.hpp index 5e001591..93cf823d 100644 --- a/src/graph_core.hpp +++ b/src/graph_core.hpp @@ -301,7 +301,12 @@ class EdgeBase: public std::enable_shared_from_this { flushing_ = true; } void stopFlushing() { + // A seek flush ends with the same producer and consumer still attached, so + // the stop requests raised by flushAndSeek_start() must be cleared here; + // they stay latched only across a real stop (see waitDo / setNodePointer). flushing_ = false; + finish_producer_ = false; + finish_consumer_ = false; } bool isFlushed() { return flushed_; @@ -424,8 +429,8 @@ template class Edge: public EdgeBase { signal_event.signal(); return true; } else { - logstream << "resetting flag"; - alt_finish = false; // reset flag + // Stop remains latched through final process/flush calls. A newly + // connected producer or consumer resets it in setNodePointer(). return false; } } @@ -630,7 +635,7 @@ template class Edge: public EdgeBase { maybeFlush(); T* r = queue_.peek(); if (timeout_ms==0) return r; - if (r != nullptr) return r; + if (r != nullptr || finish_consumer_) return r; AVTS remaining = timeout_ms; bool wait_inf = timeout_ms < 0; AVTS wait_till; diff --git a/src/hdr_metadata.hpp b/src/hdr_metadata.hpp new file mode 100644 index 00000000..566bfbfe --- /dev/null +++ b/src/hdr_metadata.hpp @@ -0,0 +1,50 @@ +#pragma once +// Static HDR metadata (mastering display + content light level) as AVFrame side +// data on a codec context, so encoders such as hevc_nvenc emit the HDR10 SEIs. +// This only serialises the values it is given; the color numbers live with the +// color contracts in the Python layer (pyplumber.mixer.color), not here. +#include +#include "util.hpp" +extern "C" { +#include +#include +#include +} + +// hdr_metadata: {"primaries": [[rx,ry],[gx,gy],[bx,by]], "white_point": [wx,wy], +// "max_luminance": nits, "min_luminance": nits, "max_cll": nits, "max_fall": nits} +inline void attachHdrMetadata(AVCodecContext *ctx, const Parameters &md) { + // Fixed 1/50000 and 1/10000 grids as the HDR10 SEI requires; av_d2q would pick any denominator. + auto q = [](double v, int den) { return av_make_q((int)std::lround(v * den), den); }; + auto sideDataSize = [](void *(*alloc)(size_t *)) { + size_t size = 0; + av_free(alloc(&size)); + return size; + }; + for (const char *key : {"primaries", "white_point", "max_luminance", "min_luminance", "max_cll", "max_fall"}) + if (!md.contains(key)) throw Error(std::string("hdr_metadata: missing ") + key); + const auto &prim = md.at("primaries"); + const auto &white = md.at("white_point"); + if (!prim.is_array() || prim.size() != 3 || !white.is_array() || white.size() != 2) + throw Error("hdr_metadata: primaries must be three [x, y] pairs and white_point one [x, y] pair"); + AVFrameSideData *sd = av_frame_side_data_new(&ctx->decoded_side_data, &ctx->nb_decoded_side_data, + AV_FRAME_DATA_MASTERING_DISPLAY_METADATA, + sideDataSize([](size_t *s) -> void * { return av_mastering_display_metadata_alloc_size(s); }), 0); + if (!sd) throw Error("hdr_metadata: cannot allocate mastering display metadata"); + auto *mdm = reinterpret_cast(sd->data); + for (int i = 0; i < 3; i++) + for (int j = 0; j < 2; j++) + mdm->display_primaries[i][j] = q(prim[i][j].get(), 50000); // chromaticity in 1/50000 + mdm->white_point[0] = q(white[0].get(), 50000); + mdm->white_point[1] = q(white[1].get(), 50000); + mdm->max_luminance = q(md.at("max_luminance").get(), 10000); // nits in 1/10000 + mdm->min_luminance = q(md.at("min_luminance").get(), 10000); + mdm->has_primaries = mdm->has_luminance = 1; + sd = av_frame_side_data_new(&ctx->decoded_side_data, &ctx->nb_decoded_side_data, + AV_FRAME_DATA_CONTENT_LIGHT_LEVEL, + sideDataSize([](size_t *s) -> void * { return av_content_light_metadata_alloc(s); }), 0); + if (!sd) throw Error("hdr_metadata: cannot allocate content light level metadata"); + auto *cll = reinterpret_cast(sd->data); + cll->MaxCLL = md.at("max_cll").get(); + cll->MaxFALL = md.at("max_fall").get(); +} diff --git a/src/instance_shared.hpp b/src/instance_shared.hpp index 7abe8539..021dc295 100644 --- a/src/instance_shared.hpp +++ b/src/instance_shared.hpp @@ -44,6 +44,11 @@ class InstanceSharedObjectsDestructors { } } public: + // Release instance resources after workers join, even when Python wrappers + // keep InstanceData alive beyond shutdown (and CUDA library teardown). + static void callInstanceDestructors(const InstanceData* instance) { + callDestructors(instance); + } static void addDestructor(const InstanceData* instance, std::function destructor) { std::unique_lock lock(busy_); destructors_[instance].push_back(destructor); @@ -193,4 +198,3 @@ std::unordered_map std::mutex InstanceSharedObjects::busy_; - diff --git a/src/mixer/Playout.hpp b/src/mixer/Playout.hpp index f78e6503..7fa3f997 100644 --- a/src/mixer/Playout.hpp +++ b/src/mixer/Playout.hpp @@ -1,7 +1,7 @@ #pragma once -#include "FrameRate.hpp" -#include "Cadence.hpp" +#include "primitives/TickGrid.hpp" +#include "primitives/Cadence.hpp" #include #include #include @@ -31,7 +31,7 @@ template class Playout { }; private: - static constexpr size_t queue_capacity = 8; + static constexpr size_t kQueueCapacity = 8; struct Entry { int64_t index; Frame frame; }; struct Input { std::deque queue; @@ -45,7 +45,7 @@ template class Playout { bool prewarm = false; Stats stats; }; - FrameRate rate_; + TickGrid rate_; TimestampMode timestamp_mode_; int64_t latency_ns_; std::vector inputs_; @@ -57,7 +57,7 @@ template class Playout { uint64_t missed_deadlines_ = 0; public: - Playout(size_t inputs, FrameRate rate, std::optional latency_ms = {}, + Playout(size_t inputs, TickGrid rate, std::optional latency_ms = {}, TimestampMode timestamp_mode = TimestampMode::Cadence) : rate_(rate), timestamp_mode_(timestamp_mode), latency_ns_(rate.time(2)), inputs_(inputs), consume_(inputs) { @@ -70,7 +70,7 @@ template class Playout { } // Reserve two queue entries for phase alignment and an arriving frame; // larger delays would silently evict frames before their deadlines. - if (latency_ns_ > rate_.time(queue_capacity - 2)) + if (latency_ns_ > rate_.time(kQueueCapacity - 2)) throw std::invalid_argument("mixer latency_ms exceeds the six-frame buffer budget"); } @@ -115,7 +115,7 @@ template class Playout { slot = *index_; ++state.stats.phase_corrections; } - if (state.queue.size() == queue_capacity) { + if (state.queue.size() == kQueueCapacity) { state.queue.pop_front(); ++state.stats.overflow; ++state.stats.discarded; diff --git a/src/mixer/TransitionScheduler.cpp b/src/mixer/TransitionScheduler.cpp new file mode 100644 index 00000000..24e8f2ed --- /dev/null +++ b/src/mixer/TransitionScheduler.cpp @@ -0,0 +1,111 @@ +#include "TransitionScheduler.hpp" +#include +#include +#include +#include +#include +#include + +namespace avp::mixer { + +struct TransitionScheduler::Impl { + struct Task { + std::chrono::steady_clock::time_point when; + uint64_t seq = 0; + std::string label; + std::function run; + }; + + std::mutex mutex; + std::condition_variable cv; + std::vector tasks; + bool stopping = false; + uint64_t next_seq = 0; + std::thread worker; + + static bool before(const Task& a, const Task& b) { + if (a.when != b.when) + return a.when < b.when; + return a.seq < b.seq; + } + + void workerLoop() { + std::unique_lock lock(mutex); + while (true) { + if (tasks.empty()) { + if (stopping) + break; + cv.wait(lock, [&] { return stopping || !tasks.empty(); }); + continue; + } + + auto next = std::min_element(tasks.begin(), tasks.end(), before); + auto now = std::chrono::steady_clock::now(); + if (next->when > now) { + cv.wait_until(lock, next->when); + continue; + } + + Task task = std::move(*next); + tasks.erase(next); + lock.unlock(); + try { + task.run(); + } catch (const std::exception& e) { + logstream << "mixer transition task " << task.label << " failed: " << e.what(); + } catch (...) { + logstream << "mixer transition task " << task.label << " failed with unknown exception"; + } + lock.lock(); + } + } +}; + +TransitionScheduler::TransitionScheduler() + : impl_(std::make_unique()) { + impl_->worker = start_thread("mixer transitions", [impl = impl_.get()] { + impl->workerLoop(); + }); +} + +TransitionScheduler::~TransitionScheduler() { + shutdown(); +} + +void TransitionScheduler::post(std::string label, std::function task) { + postAfter(std::move(label), 0, std::move(task)); +} + +void TransitionScheduler::postAfter(std::string label, int64_t delay_ms, std::function task) { + if (!task) + throw Error("mixer transition scheduler: empty task"); + + std::lock_guard lock(impl_->mutex); + if (impl_->stopping) + throw Error("mixer transition scheduler is stopped"); + + Impl::Task queued; + queued.when = std::chrono::steady_clock::now() + + std::chrono::milliseconds(std::max(0, delay_ms)); + queued.seq = impl_->next_seq++; + queued.label = std::move(label); + queued.run = std::move(task); + impl_->tasks.push_back(std::move(queued)); + impl_->cv.notify_one(); +} + +void TransitionScheduler::shutdown() { + if (!impl_) + return; + + { + std::lock_guard lock(impl_->mutex); + impl_->stopping = true; + impl_->tasks.clear(); + } + impl_->cv.notify_all(); + if (impl_->worker.joinable()) + impl_->worker.join(); +} + +} // namespace avp::mixer diff --git a/src/mixer/TransitionScheduler.hpp b/src/mixer/TransitionScheduler.hpp new file mode 100644 index 00000000..ae3df514 --- /dev/null +++ b/src/mixer/TransitionScheduler.hpp @@ -0,0 +1,26 @@ +#pragma once +// Single worker thread that runs mixer transition steps at their scheduled +// wall-clock time, in submission order for equal times. +#include "../instance_shared.hpp" +#include "../graph_mgmt.hpp" +#include +#include +#include +#include + +namespace avp::mixer { + +class TransitionScheduler : public InstanceShared, public IShutdownable { + struct Impl; + std::unique_ptr impl_; + +public: + TransitionScheduler(); + ~TransitionScheduler(); + + void post(std::string label, std::function task); + void postAfter(std::string label, int64_t delay_ms, std::function task); + void shutdown(); +}; + +} // namespace avp::mixer diff --git a/src/mixer/graph_ops.cpp b/src/mixer/graph_ops.cpp new file mode 100644 index 00000000..395099db --- /dev/null +++ b/src/mixer/graph_ops.cpp @@ -0,0 +1,235 @@ +#include "graph_ops.hpp" +#include "../graph_interfaces.hpp" +#include +#include +#include +#include +#include + +namespace avp::mixer::graph { + +void resetInputIf(std::shared_ptr nodes, const std::string& name) { + if (name.empty()) + return; + auto w = nodes->node_if_exists(name); + if (!w || !w->node()) + return; + if (auto r = std::dynamic_pointer_cast(w->node())) + r->resetInput(); +} + +void resetSlotNormFps(std::shared_ptr nodes, const MixerState& st) { + resetInputIf(nodes, st.slot_a.norm_ts_name); + resetInputIf(nodes, st.slot_b.norm_ts_name); +} + +bool nodeWorkingIfExists(std::shared_ptr nodes, const std::string& name) { + if (name.empty()) + return false; + auto w = nodes->node_if_exists(name); + return w && w->isWorking(); +} + +static bool nodeConsumesEdge(const std::shared_ptr& node, const std::string& edge_name) { + if (!node) + return false; + const auto& params = node->parameters(); + if (!params.count("src")) + return false; + for (const auto& src_name : jsonToStringList(params["src"])) { + if (src_name == edge_name) + return true; + } + return false; +} + +std::shared_ptr workingConsumerForEdge(std::shared_ptr nodes, + const std::string& edge_name) { + for (const auto& [_, node] : nodes->allNodes()) { + if (node && node->isWorking() && nodeConsumesEdge(node, edge_name)) + return node; + } + return nullptr; +} + +bool setNodeObjectIfCreated(std::shared_ptr nodes, + const std::string& node_name, + const std::string& key, + const Parameters& value) { + auto wrapper = nodes->node_if_exists(node_name); + if (!wrapper) + throw Error("Node " + node_name + " doesn't exist."); + + wrapper->parameters()[key] = value; + if (!wrapper->node()) + return false; + + try { + wrapper->setObject(key, value); + return true; + } catch (const std::exception& e) { + if (std::string(e.what()) == "Node not created") + return false; + throw; + } +} + +int edgeOccupiedIfExists(std::shared_ptr nodes, const std::string& name) { + if (name.empty()) + return 0; + auto e = nodes->edges()->findAny(name); + return e ? e->occupied() : 0; +} + +std::string firstDstEdgeName(std::shared_ptr nodes, const std::string& node_name) { + auto node = nodes->node_if_exists(node_name); + if (!node) + return ""; + const auto& params = node->parameters(); + if (!params.count("dst")) + return ""; + auto names = jsonToStringList(params["dst"]); + return names.empty() ? "" : names.front(); +} + +std::string edgeNameAt(std::shared_ptr nodes, + const std::string& node_name, + const std::string& param_name, + size_t index) { + auto node = nodes->node_if_exists(node_name); + if (!node) + return ""; + const auto& params = node->parameters(); + if (!params.count(param_name)) + return ""; + auto names = jsonToStringList(params[param_name]); + if (index >= names.size()) + return ""; + auto it = names.begin(); + std::advance(it, index); + return *it; +} + +av::Timestamp edgeLastTsIfExists(std::shared_ptr nodes, const std::string& name) { + if (name.empty()) + return NOTS; + auto e = nodes->edges()->findAny(name); + return e ? e->lastTS() : NOTS; +} + +WipeReadyResult waitForWipeOverlayReady(std::shared_ptr nodes, + const std::string& edge_name, + av::Timestamp initial_ts, + int64_t earliest_visible_pts_ms, + const std::shared_ptr& state, uint64_t generation, + int64_t timeout_ms) { + WipeReadyResult result; + while (result.waited_ms < timeout_ms) { + if (state->transition_generation.load() != generation) return result; + const bool time_ready = wallclock.pts() >= earliest_visible_pts_ms; + av::Timestamp ts = edgeLastTsIfExists(nodes, edge_name); + const bool frame_ready = ts.isValid() && (!initial_ts.isValid() || ts > initial_ts); + if (time_ready && frame_ready) { + result.ready = true; + result.ready_ts = ts; + return result; + } + std::this_thread::sleep_for(std::chrono::milliseconds(kPollMs)); + result.waited_ms += kPollMs; + } + result.ready_ts = edgeLastTsIfExists(nodes, edge_name); + return result; +} + +bool overlayCommandCurrent(const std::shared_ptr& state, uint64_t generation) { + return state->overlay_generation.load(std::memory_order_acquire) == generation; +} + +OverlayReadyResult waitForOverlayBranchReady(std::shared_ptr nodes, + std::shared_ptr state, + uint64_t generation, + const std::string& edge_name, + av::Timestamp initial_ts, + av::Timestamp minimum_ts, + int64_t timeout_ms, + int64_t poll_ms) { + OverlayReadyResult result; + timeout_ms = std::max(0, timeout_ms); + poll_ms = std::max(1, poll_ms); + while (result.waited_ms < timeout_ms) { + if (!overlayCommandCurrent(state, generation)) { + result.cancelled = true; + return result; + } + av::Timestamp ts = edgeLastTsIfExists(nodes, edge_name); + const bool fresh = ts.isValid() && (!initial_ts.isValid() || ts > initial_ts); + const bool monotonic = !minimum_ts.isValid() || (ts.isValid() && !(ts < minimum_ts)); + if (fresh && monotonic) { + result.ready = true; + result.ready_ts = ts; + return result; + } + std::this_thread::sleep_for(std::chrono::milliseconds(poll_ms)); + result.waited_ms += poll_ms; + } + result.ready_ts = edgeLastTsIfExists(nodes, edge_name); + return result; +} + +static std::shared_ptr requireNodeWrapper(std::shared_ptr nodes, + const std::string& node_name, + const std::string& context) { + auto wrapper = nodes->node_if_exists(node_name); + if (!wrapper) + throw Error(context + ": node " + node_name + " doesn't exist"); + return wrapper; +} + +static std::vector routerLabels(std::shared_ptr nodes, const std::string& router_name) { + auto wrapper = requireNodeWrapper(nodes, router_name, "mixer.routed_source"); + const auto& params = wrapper->parameters(); + if (!params.count("dst")) + throw Error("mixer.routed_source: router " + router_name + " has no dst parameter"); + if (!params.count("labels")) + throw Error("mixer.routed_source: router " + router_name + " has no labels parameter"); + + auto dst = jsonToStringList(params["dst"]); + auto labels_list = jsonToStringList(params["labels"]); + std::vector labels(labels_list.begin(), labels_list.end()); + if (dst.size() != labels.size()) { + throw Error("mixer.routed_source: router " + router_name + " labels size " + + std::to_string(labels.size()) + " does not match dst size " + + std::to_string(dst.size())); + } + return labels; +} + +int routerOutputCount(std::shared_ptr nodes, const std::string& router_name) { + auto wrapper = requireNodeWrapper(nodes, router_name, "mixer.init_routes"); + const auto& params = wrapper->parameters(); + if (!params.count("dst")) + throw Error("mixer.init_routes: router " + router_name + " has no dst parameter"); + return static_cast(jsonToStringList(params["dst"]).size()); +} + +int routerOutputIndexFromLabel(std::shared_ptr nodes, + const std::string& router_name, + const std::string& label) { + auto labels = routerLabels(nodes, router_name); + if (!label.empty() && std::all_of(label.begin(), label.end(), [](unsigned char ch) { return std::isdigit(ch); })) { + int output_index = std::stoi(label); + if (output_index < 0 || output_index >= (int)labels.size()) { + throw Error("mixer.routed_source: router " + router_name + + " output index " + label + " is out of range"); + } + return output_index; + } + auto it = std::find(labels.begin(), labels.end(), label); + if (it == labels.end()) { + throw Error("mixer.routed_source: router " + router_name + + " has no output label " + label); + } + return static_cast(std::distance(labels.begin(), it)); +} + +} diff --git a/src/mixer/graph_ops.hpp b/src/mixer/graph_ops.hpp new file mode 100644 index 00000000..2bddc618 --- /dev/null +++ b/src/mixer/graph_ops.hpp @@ -0,0 +1,70 @@ +#pragma once +// Thin helpers over NodeManager/edges used by the orchestrator: lookups that +// tolerate missing nodes, setObject that queues for not-yet-created nodes, +// router label resolution, and the readiness polls for wipe/overlay branches. +#include "primitives/MixerState.hpp" +#include "../avutils.hpp" +#include "../graph_mgmt.hpp" +#include +#include +#include +#include + +namespace avp::mixer::graph { + +/// Poll period of every readiness wait in the orchestrator. +constexpr int64_t kPollMs = 5; +constexpr int64_t kWipeReadyTimeoutMs = 5000; + +void resetInputIf(std::shared_ptr nodes, const std::string& name); + +/// After a PGM path change, slot compositors may have been idle (`active_inputs=0`) while cameras +/// advanced in PTS; `force_fps` on `norm_*` would otherwise compare the next real frame to a stale +/// grid and produce a big `Discontinuity` jump with a duplicate burst across the cut. +/// +/// Deliberately NOT resetting the final post-`out_sel` `force_fps` (e.g. `mixer_norm_fps`): it is +/// the last VFR guard before the encoder and resetting it would drop that guarantee. +void resetSlotNormFps(std::shared_ptr nodes, const MixerState& st); + +bool nodeWorkingIfExists(std::shared_ptr nodes, const std::string& name); +std::shared_ptr workingConsumerForEdge(std::shared_ptr nodes, + const std::string& edge_name); +/// Stores the value in the wrapper's parameters and applies it when the node exists; false when +/// the node is not created yet (the value is picked up at creation). +bool setNodeObjectIfCreated(std::shared_ptr nodes, const std::string& node_name, + const std::string& key, const Parameters& value); +int edgeOccupiedIfExists(std::shared_ptr nodes, const std::string& name); +std::string firstDstEdgeName(std::shared_ptr nodes, const std::string& node_name); +std::string edgeNameAt(std::shared_ptr nodes, const std::string& node_name, + const std::string& param_name, size_t index); +av::Timestamp edgeLastTsIfExists(std::shared_ptr nodes, const std::string& name); +int routerOutputCount(std::shared_ptr nodes, const std::string& router_name); +int routerOutputIndexFromLabel(std::shared_ptr nodes, const std::string& router_name, + const std::string& label); + +struct WipeReadyResult { + bool ready = false; + int64_t waited_ms = 0; + av::Timestamp ready_ts = NOTS; +}; + +struct OverlayReadyResult { + bool ready = false; + bool cancelled = false; + int64_t waited_ms = 0; + av::Timestamp ready_ts = NOTS; +}; + +/// Poll until `edge_name` carries a frame newer than `initial_ts` and wallclock reached +/// `earliest_visible_pts_ms`, or the transition generation moved on, or `timeout_ms` passed. +WipeReadyResult waitForWipeOverlayReady(std::shared_ptr nodes, const std::string& edge_name, + av::Timestamp initial_ts, int64_t earliest_visible_pts_ms, + const std::shared_ptr& state, uint64_t generation, + int64_t timeout_ms = kWipeReadyTimeoutMs); +bool overlayCommandCurrent(const std::shared_ptr& state, uint64_t generation); +OverlayReadyResult waitForOverlayBranchReady(std::shared_ptr nodes, + std::shared_ptr state, uint64_t generation, + const std::string& edge_name, av::Timestamp initial_ts, + av::Timestamp minimum_ts, int64_t timeout_ms, int64_t poll_ms); + +} diff --git a/src/mixer/mixer_orchestrator.cpp b/src/mixer/mixer_orchestrator.cpp deleted file mode 100644 index e4a017af..00000000 --- a/src/mixer/mixer_orchestrator.cpp +++ /dev/null @@ -1,1898 +0,0 @@ -#include "mixer_orchestrator.hpp" -#include "../avutils.hpp" -#include "../graph_interfaces.hpp" -#include "OutputSnapshot.hpp" -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include - -struct MixerTransitionScheduler::Impl { - struct Task { - std::chrono::steady_clock::time_point when; - uint64_t seq = 0; - std::string label; - std::function run; - }; - - std::mutex mutex; - std::condition_variable cv; - std::vector tasks; - bool stopping = false; - uint64_t next_seq = 0; - std::thread worker; - - static bool before(const Task& a, const Task& b) { - if (a.when != b.when) - return a.when < b.when; - return a.seq < b.seq; - } - - void workerLoop() { - std::unique_lock lock(mutex); - while (true) { - if (tasks.empty()) { - if (stopping) - break; - cv.wait(lock, [&] { return stopping || !tasks.empty(); }); - continue; - } - - auto next = std::min_element(tasks.begin(), tasks.end(), before); - auto now = std::chrono::steady_clock::now(); - if (next->when > now) { - cv.wait_until(lock, next->when); - continue; - } - - Task task = std::move(*next); - tasks.erase(next); - lock.unlock(); - try { - task.run(); - } catch (const std::exception& e) { - logstream << "mixer transition task " << task.label << " failed: " << e.what(); - } catch (...) { - logstream << "mixer transition task " << task.label << " failed with unknown exception"; - } - lock.lock(); - } - } -}; - -MixerTransitionScheduler::MixerTransitionScheduler() - : impl_(std::make_unique()) { - impl_->worker = start_thread("mixer transitions", [impl = impl_.get()] { - impl->workerLoop(); - }); -} - -MixerTransitionScheduler::~MixerTransitionScheduler() { - shutdown(); -} - -void MixerTransitionScheduler::post(std::string label, std::function task) { - postAfter(std::move(label), 0, std::move(task)); -} - -void MixerTransitionScheduler::postAfter(std::string label, int64_t delay_ms, std::function task) { - if (!task) - throw Error("mixer transition scheduler: empty task"); - - std::lock_guard lock(impl_->mutex); - if (impl_->stopping) - throw Error("mixer transition scheduler is stopped"); - - Impl::Task queued; - queued.when = std::chrono::steady_clock::now() + - std::chrono::milliseconds(std::max(0, delay_ms)); - queued.seq = impl_->next_seq++; - queued.label = std::move(label); - queued.run = std::move(task); - impl_->tasks.push_back(std::move(queued)); - impl_->cv.notify_one(); -} - -void MixerTransitionScheduler::shutdown() { - if (!impl_) - return; - - { - std::lock_guard lock(impl_->mutex); - impl_->stopping = true; - impl_->tasks.clear(); - } - impl_->cv.notify_all(); - if (impl_->worker.joinable()) - impl_->worker.join(); -} - -namespace { - -constexpr int64_t kWipeSwitchGraceMs = 500; -constexpr int64_t kWipeReadyPollMs = 5; -constexpr int64_t kWipeReadyTimeoutMs = 5000; -constexpr int kOverlayDirectInput = 0; -constexpr int kOverlayCompositedInput = 1; - -void resetInputIf(std::shared_ptr nodes, const std::string& name) { - if (name.empty()) - return; - auto w = nodes->node_if_exists(name); - if (!w || !w->node()) - return; - if (auto r = std::dynamic_pointer_cast(w->node())) - r->resetInput(); -} - -/// After a PGM path change, slot compositors may have been idle (`active_inputs=0`) while cameras -/// advanced in PTS; `force_fps` on `norm_*` would otherwise compare the next real frame to a stale -/// grid and produce a big `Discontinuity` jump with a duplicate burst across the cut. -/// -/// Deliberately NOT resetting the final post-`out_sel` `force_fps` (e.g. `mixer_norm_fps`): it is -/// the last VFR guard before the encoder and resetting it would drop that guarantee. -void resetSlotNormFps(std::shared_ptr nodes, const MixerState& st) { - resetInputIf(nodes, st.slot_a.norm_ts_name); - resetInputIf(nodes, st.slot_b.norm_ts_name); -} - -bool nodeWorkingIfExists(std::shared_ptr nodes, const std::string& name) { - if (name.empty()) - return false; - auto w = nodes->node_if_exists(name); - return w && w->isWorking(); -} - -bool nodeConsumesEdge(const std::shared_ptr& node, const std::string& edge_name) { - if (!node) - return false; - const auto& params = node->parameters(); - if (!params.count("src")) - return false; - for (const auto& src_name : jsonToStringList(params["src"])) { - if (src_name == edge_name) - return true; - } - return false; -} - -std::shared_ptr workingConsumerForEdge(std::shared_ptr nodes, - const std::string& edge_name) { - for (const auto& [_, node] : nodes->allNodes()) { - if (node && node->isWorking() && nodeConsumesEdge(node, edge_name)) - return node; - } - return nullptr; -} - -bool setNodeObjectIfCreated(std::shared_ptr nodes, - const std::string& node_name, - const std::string& key, - const Parameters& value) { - auto wrapper = nodes->node_if_exists(node_name); - if (!wrapper) - throw Error("Node " + node_name + " doesn't exist."); - - wrapper->parameters()[key] = value; - if (!wrapper->node()) - return false; - - try { - wrapper->setObject(key, value); - return true; - } catch (const std::exception& e) { - if (std::string(e.what()) == "Node not created") - return false; - throw; - } -} - -int edgeOccupiedIfExists(std::shared_ptr nodes, const std::string& name) { - if (name.empty()) - return 0; - auto e = nodes->edges()->findAny(name); - return e ? e->occupied() : 0; -} - -std::string firstDstEdgeName(std::shared_ptr nodes, const std::string& node_name) { - auto node = nodes->node_if_exists(node_name); - if (!node) - return ""; - const auto& params = node->parameters(); - if (!params.count("dst")) - return ""; - auto names = jsonToStringList(params["dst"]); - return names.empty() ? "" : names.front(); -} - -std::string edgeNameAt(std::shared_ptr nodes, - const std::string& node_name, - const std::string& param_name, - size_t index) { - auto node = nodes->node_if_exists(node_name); - if (!node) - return ""; - const auto& params = node->parameters(); - if (!params.count(param_name)) - return ""; - auto names = jsonToStringList(params[param_name]); - if (index >= names.size()) - return ""; - auto it = names.begin(); - std::advance(it, index); - return *it; -} - -av::Timestamp edgeLastTsIfExists(std::shared_ptr nodes, const std::string& name) { - if (name.empty()) - return NOTS; - auto e = nodes->edges()->findAny(name); - return e ? e->lastTS() : NOTS; -} - -struct WipeReadyResult { - bool ready = false; - int64_t waited_ms = 0; - av::Timestamp ready_ts = NOTS; -}; - -struct OverlayReadyResult { - bool ready = false; - bool cancelled = false; - int64_t waited_ms = 0; - av::Timestamp ready_ts = NOTS; -}; - -WipeReadyResult waitForWipeOverlayReady(std::shared_ptr nodes, - const std::string& edge_name, - av::Timestamp initial_ts, - int64_t earliest_visible_pts_ms, - const std::shared_ptr& state, uint64_t generation, - int64_t timeout_ms = kWipeReadyTimeoutMs) { - WipeReadyResult result; - while (result.waited_ms < timeout_ms) { - if (state->transition_generation.load() != generation) return result; - const bool time_ready = wallclock.pts() >= earliest_visible_pts_ms; - av::Timestamp ts = edgeLastTsIfExists(nodes, edge_name); - const bool frame_ready = ts.isValid() && (!initial_ts.isValid() || ts > initial_ts); - if (time_ready && frame_ready) { - result.ready = true; - result.ready_ts = ts; - return result; - } - std::this_thread::sleep_for(std::chrono::milliseconds(kWipeReadyPollMs)); - result.waited_ms += kWipeReadyPollMs; - } - result.ready_ts = edgeLastTsIfExists(nodes, edge_name); - return result; -} - -bool overlayCommandCurrent(const std::shared_ptr& state, uint64_t generation) { - return state->overlay_generation.load(std::memory_order_acquire) == generation; -} - -OverlayReadyResult waitForOverlayBranchReady(std::shared_ptr nodes, - std::shared_ptr state, - uint64_t generation, - const std::string& edge_name, - av::Timestamp initial_ts, - av::Timestamp minimum_ts, - int64_t timeout_ms, - int64_t poll_ms) { - OverlayReadyResult result; - timeout_ms = std::max(0, timeout_ms); - poll_ms = std::max(1, poll_ms); - while (result.waited_ms < timeout_ms) { - if (!overlayCommandCurrent(state, generation)) { - result.cancelled = true; - return result; - } - av::Timestamp ts = edgeLastTsIfExists(nodes, edge_name); - const bool fresh = ts.isValid() && (!initial_ts.isValid() || ts > initial_ts); - const bool monotonic = !minimum_ts.isValid() || (ts.isValid() && !(ts < minimum_ts)); - if (fresh && monotonic) { - result.ready = true; - result.ready_ts = ts; - return result; - } - std::this_thread::sleep_for(std::chrono::milliseconds(poll_ms)); - result.waited_ms += poll_ms; - } - result.ready_ts = edgeLastTsIfExists(nodes, edge_name); - return result; -} - -std::shared_ptr requireNodeWrapper(std::shared_ptr nodes, - const std::string& node_name, - const std::string& context) { - auto wrapper = nodes->node_if_exists(node_name); - if (!wrapper) - throw Error(context + ": node " + node_name + " doesn't exist"); - return wrapper; -} - -std::vector routerLabels(std::shared_ptr nodes, const std::string& router_name) { - auto wrapper = requireNodeWrapper(nodes, router_name, "mixer.routed_source"); - const auto& params = wrapper->parameters(); - if (!params.count("dst")) - throw Error("mixer.routed_source: router " + router_name + " has no dst parameter"); - if (!params.count("labels")) - throw Error("mixer.routed_source: router " + router_name + " has no labels parameter"); - - auto dst = jsonToStringList(params["dst"]); - auto labels_list = jsonToStringList(params["labels"]); - std::vector labels(labels_list.begin(), labels_list.end()); - if (dst.size() != labels.size()) { - throw Error("mixer.routed_source: router " + router_name + " labels size " + - std::to_string(labels.size()) + " does not match dst size " + - std::to_string(dst.size())); - } - return labels; -} - -int routerOutputCount(std::shared_ptr nodes, const std::string& router_name) { - auto wrapper = requireNodeWrapper(nodes, router_name, "mixer.init_routes"); - const auto& params = wrapper->parameters(); - if (!params.count("dst")) - throw Error("mixer.init_routes: router " + router_name + " has no dst parameter"); - return static_cast(jsonToStringList(params["dst"]).size()); -} - -int routerOutputIndexFromLabel(std::shared_ptr nodes, - const std::string& router_name, - const std::string& label) { - auto labels = routerLabels(nodes, router_name); - if (!label.empty() && std::all_of(label.begin(), label.end(), [](unsigned char ch) { return std::isdigit(ch); })) { - int output_index = std::stoi(label); - if (output_index < 0 || output_index >= (int)labels.size()) { - throw Error("mixer.routed_source: router " + router_name + - " output index " + label + " is out of range"); - } - return output_index; - } - auto it = std::find(labels.begin(), labels.end(), label); - if (it == labels.end()) { - throw Error("mixer.routed_source: router " + router_name + - " has no output label " + label); - } - return static_cast(std::distance(labels.begin(), it)); -} - -Parameters routesToParameters(const std::vector& routes) { - Parameters arr = Parameters::array(); - for (int input_index : routes) - arr.push_back(input_index); - return arr; -} - -void ensureRouteTableSize(MixerState& st, const std::string& router_name) { - int count = st.router_output_counts[router_name]; - auto& routes = st.router_routes[router_name]; - if ((int)routes.size() != count) - routes.assign(count, -1); -} - -std::unordered_map> currentRouterTables(MixerState& st) { - std::unordered_map> tables; - for (const auto& [router_name, count] : st.router_output_counts) { - ensureRouteTableSize(st, router_name); - tables[router_name] = st.router_routes[router_name]; - if ((int)tables[router_name].size() != count) - tables[router_name].assign(count, -1); - } - return tables; -} - -int routeOutputForSlot(const MixerState::SourceInfo& info, bool is_slot_a) { - return is_slot_a ? info.route_output_a : info.route_output_b; -} - -std::string routeOutputLabelForSlot(const MixerState::SourceInfo& info, bool is_slot_a) { - return is_slot_a ? info.route_output_label_a : info.route_output_label_b; -} - -void setRoutedSlotInTables(MixerState& st, - std::unordered_map>& tables, - bool is_slot_a, - const SceneDefinition* scene) { - for (const auto& [src_name, info] : st.sources) { - if (!info.routed) - continue; - const int output_index = routeOutputForSlot(info, is_slot_a); - if (output_index < 0) - continue; - - auto& routes = tables[info.router_node_name]; - const int count = st.router_output_counts[info.router_node_name]; - if ((int)routes.size() != count) - routes.assign(count, -1); - if (output_index >= (int)routes.size()) - throw Error("mixer: routed source " + src_name + " output index " + - std::to_string(output_index) + " (" + - routeOutputLabelForSlot(info, is_slot_a) + ") exceeds router " + - info.router_node_name + " route table"); - - routes[output_index] = -1; - if (!scene || !scene->sources.count(src_name)) - continue; - - auto route_it = scene->routes.find(src_name); - if (route_it == scene->routes.end()) { - throw Error("mixer: scene " + scene->name + " uses routed source " + - src_name + " without an explicit route"); - } - routes[output_index] = route_it->second; - } -} - - -/// One cuda_rect_overlay layer per compositor src index (see mixer.source). Omitted sources use a dummy rect. -Parameters compositorLayersFromScene(const MixerState& st, const SceneDefinition& scene) { - static const Parameters kUnusedLayer = Parameters({{"dst_x", 0}, {"dst_y", 0}}); - - int max_idx = -1; - for (const auto& [_, info] : st.sources) - max_idx = std::max(max_idx, info.input_index); - - Parameters arr = Parameters::array(); - for (int i = 0; i <= max_idx; ++i) { - std::string name_at; - for (const auto& [name, info] : st.sources) { - if (info.input_index == i) { - name_at = name; - break; - } - } - if (name_at.empty()) { - arr.push_back(kUnusedLayer); - continue; - } - auto it = scene.sources.find(name_at); - if (it == scene.sources.end()) - arr.push_back(kUnusedLayer); - else - arr.push_back(it->second.layer); - } - return arr; -} - -class TransitionPrepGuard { - std::function abort_; - bool active_ = true; - -public: - explicit TransitionPrepGuard(std::function abort) - : abort_(std::move(abort)) {} - - ~TransitionPrepGuard() { - if (active_) abort_(); - } - - void release() { - active_ = false; - } -}; - -bool transitionIsCurrent(const std::shared_ptr& state, - uint64_t generation, - MixerState::TransitionMode mode) { - return state->transition_generation.load(std::memory_order_acquire) == generation && - state->transition_mode.load(std::memory_order_acquire) == mode; -} - -} // namespace - -MixerOrchestrator::MixerOrchestrator( - std::shared_ptr nodes, - std::shared_ptr state, - std::shared_ptr timeline, - std::shared_ptr scheduler) - : nodes_(std::move(nodes)), - state_(std::move(state)), - timeline_(std::move(timeline)), - scheduler_(std::move(scheduler)) {} - -void MixerOrchestrator::postTransitionTask(std::string label, int64_t delay_ms, std::function task) { - if (!scheduler_) - throw Error("mixer: transition scheduler is not configured"); - scheduler_->postAfter(std::move(label), delay_ms, std::move(task)); -} - -void MixerOrchestrator::setNodeObject(const std::string& node_name, const std::string& key, const Parameters& value) { - try { - auto node = nodes_->node(node_name); - node->setObject(key, value); - } catch (const std::exception& e) { - throw Error("mixer: set " + node_name + "." + key + " failed: " + e.what()); - } -} - -void MixerOrchestrator::publishRuntimeObject(const std::string& node_name, - const std::string& key, - const Parameters& value) { - timeline_->clearKey(node_name, key); - if (!setNodeObjectIfCreated(nodes_, node_name, key, value)) { - logstream << "mixer: queued " << node_name << "." << key - << " for node not created yet"; - } - timeline_->set(node_name, key, wallclock.pts(), value); -} - -void MixerOrchestrator::publishCameraOtmOutputs(const std::string& otm_name, uint32_t mask) { - // `one_to_many` with `timeline` uses tlGetRaw("outputs") whenever any entry matches; stale - // rows (e.g. an old T_cleanup) would override setObject. We drop only the "outputs" key on - // this OTM channel — not post-scene otms, not source_switcher, not other keys here. - // cut/fade append new `outputs` rows at T_cleanup *after* loadSceneIntoSlot returns, so - // those are not cleared by this call. Overlapping mixer commands are rejected by ensureIdle(). - timeline_->clearKey(otm_name, "outputs"); - if (!setNodeObjectIfCreated(nodes_, otm_name, "outputs", Parameters(mask))) { - logstream << "mixer: queued " << otm_name << ".outputs=" << mask - << " for node not created yet"; - } - timeline_->set(otm_name, "outputs", wallclock.pts(), Parameters(mask)); -} - -void MixerOrchestrator::setNodeParam(const std::string& node_name, const std::string& param, const std::string& value) { - auto node = nodes_->node(node_name); - auto& params = node->parameters(); - params[param] = value; -} - -void MixerOrchestrator::autoRestartNode(const std::string& node_name) { - auto node = nodes_->node(node_name); - node->stop(false); -} - -void MixerOrchestrator::createAndStartNode(const Parameters& params) { - Parameters p = params; - nodes_->createNode(p, true, true); -} - -void MixerOrchestrator::deleteNodeIfExists(const std::string& name) { - auto node = nodes_->node_if_exists(name); - if (node) { - nodes_->deleteNode(name); - } -} - -void MixerOrchestrator::startGroup(const std::string& group_name) { - nodes_->group(group_name)->startNodes(); -} - -void MixerOrchestrator::stopGroup(const std::string& group_name) { - nodes_->group(group_name)->stopNodes(); -} - -void MixerOrchestrator::flushWipeEdges() { - for (const auto& name : state_->wipe_flush_edges) { - auto edge = nodes_->edges()->findAny(name); - if (!edge) - continue; - - int occupied = edge->occupied(); - if (occupied <= 0) - continue; - - auto active_consumer = workingConsumerForEdge(nodes_, name); - if (active_consumer) { - logstream << "mixer: leaving active wipe edge " << name - << " unflushed (" << occupied << " queued, consumer=" - << active_consumer->name() << ")"; - continue; - } - - logstream << "mixer: flushing stale wipe edge " << name - << " (" << occupied << " queued)"; - edge->clear(); - } -} - -void MixerOrchestrator::flushSlotEdges(bool is_slot_a) { - const auto& slot = is_slot_a ? state_->slot_a : state_->slot_b; - - auto clearEdge = [this](const std::string& name) { - if (name.empty()) - return; - auto edge = nodes_->edges()->findAny(name); - if (edge && edge->occupied() > 0) { - // readerwriterqueue is SPSC: clearing from this control thread would - // consume the queue concurrently with the node that owns the edge. - if (!edge->consumer().expired()) { - logstream << "mixer: leaving live slot edge " << name - << " unflushed (" << edge->occupied() << " queued)"; - return; - } - logstream << "mixer: flushing stale slot edge " << name - << " (" << edge->occupied() << " queued)"; - edge->clear(); - } - }; - - for (const auto& [_, info] : state_->sources) { - const std::string& cs_node = is_slot_a ? info.cs_node_a : info.cs_node_b; - auto node = nodes_->node_if_exists(cs_node); - if (!node) - continue; - const auto& params = node->parameters(); - if (params.count("src")) { - for (const auto& edge_name : jsonToStringList(params["src"])) - clearEdge(edge_name); - } - if (params.count("dst")) { - for (const auto& edge_name : jsonToStringList(params["dst"])) - clearEdge(edge_name); - } - } - - for (const std::string& node_name : {slot.compositor_name, slot.norm_ts_name, slot.post_otm_name}) { - auto node = nodes_->node_if_exists(node_name); - if (!node) - continue; - const auto& params = node->parameters(); - if (params.count("dst")) { - for (const auto& edge_name : jsonToStringList(params["dst"])) - clearEdge(edge_name); - } - } -} - -void MixerOrchestrator::ensureIdle() const { - auto mode = state_->transition_mode.load(); - if (mode != MixerState::TransitionMode::Idle) - throw Error("mixer: transition already in progress"); -} - -void MixerOrchestrator::interruptTransition() { - if (state_->cut_latency) state_->cut_latency->timing.cancel(); - if (state_->transition_mode == MixerState::TransitionMode::Idle) return; - const auto previous_mode = state_->transition_mode.load(); - // A crossfade's blended picture exists only inside the transition compositor, - // so it has to be frozen to survive the interruption. A wipe's does not: the - // output carries the wipe graphic, and freezing that would paint the graphic - // into the program, where the next wipe would composite over it and the two - // would stack. In every other mode the program slot keeps rendering, so the - // direct branch restored below is already the right picture. - const bool freeze = previous_mode == MixerState::TransitionMode::Crossfade; - auto snapshot = InstanceSharedObjects::get( - nodes_->instanceData(), state_->source_switcher_name + "_snapshot"); - if (freeze) { - std::lock_guard lock(snapshot->mutex); - if (!snapshot->output_connected) - throw Error("mixer: interruption requires mixer_snapshot output and slot nodes"); - snapshot->frames.capture(state_->pgmSourceSwitcherIndex()); - } - const auto generation = ++state_->transition_generation; - TransitionPrepGuard guard([&] { abortTransition(generation); }); - restoreProgramRouting(); - if (previous_mode == MixerState::TransitionMode::Wipe && !state_->wipe_group_name.empty()) { - // Group management retires the old decoder independently. Waiting here - // would add teardown time to every correction, including a hard cut. - stopGroup(state_->wipe_group_name); - } - state_->pvw_scene_name.clear(); - state_->transition_mode = MixerState::TransitionMode::Idle; - { - std::lock_guard lock(snapshot->mutex); - if (freeze) { - snapshot->frames.arm(wallclock.pts() * 1000000); - } else { - // Drop any substitution an earlier interruption left in the slot. - snapshot->frames.finish(); - snapshot->frames.arm(wallclock.pts() * 1000000, false); - } - } - guard.release(); - logstream << "mixer: interrupted transition; " << (freeze ? "retained the blended picture" - : "returned to the program picture"); -} - -void MixerOrchestrator::restoreProgramRouting() { - // Remove scheduled controls as well as routes; cancelling a worker alone - // cannot cancel a future selector flip. - if (auto scene = state_->scenes.find(state_->transition_scene_name); scene != state_->scenes.end()) { - for (const auto& control : scene->second.controls) - timeline_->clearKey(control.node_name, control.key); - } - applyPostTransitionRouting(state_->pgm_is_slot_a, state_->pgm_scene_name); - scheduleSceneControls(state_->scenes.at(state_->pgm_scene_name), wallclock.pts()); - for (const auto& [name, key, value] : std::vector>{ - {state_->wipe_selector_name, "active", 0}, {state_->wipe_otm_name, "outputs", 1}}) { - if (name.empty()) continue; - timeline_->clearKey(name, key); - setNodeObject(name, key, Parameters(value)); - } -} - -void MixerOrchestrator::abortTransition(uint64_t generation) noexcept { - // All callers hold the control mutex. A cancelled worker must never undo - // the routing of its replacement transition. - if (state_->transition_generation != generation) return; - if (state_->cut_latency) state_->cut_latency->timing.cancel("failed"); - ++state_->transition_generation; - const auto mode = state_->transition_mode.load(); - auto cleanup = [](auto action) { - try { action(); } - catch (const std::exception& e) { - logstream << "mixer: transition abort cleanup failed: " << e.what(); - } - }; - cleanup([&] { restoreProgramRouting(); }); - if (mode == MixerState::TransitionMode::Wipe && !state_->wipe_group_name.empty()) - cleanup([&] { stopGroup(state_->wipe_group_name); }); - // Remove slot substitution even when the target never produced a frame. - // The output gate still waits for a fresh program frame before releasing. - cleanup([&] { finishSnapshot(); }); - state_->pvw_scene_name.clear(); - state_->transition_mode = MixerState::TransitionMode::Idle; -} - -void MixerOrchestrator::finishSnapshot() { - auto snapshot = InstanceSharedObjects::get( - nodes_->instanceData(), state_->source_switcher_name + "_snapshot"); - std::lock_guard lock(snapshot->mutex); - snapshot->frames.finish(); - snapshot->frames.arm(wallclock.pts() * 1000000, false); -} - -int64_t MixerOrchestrator::resolveTransitionStartPts(int64_t requested_start_pts_ms) const { - int64_t now = wallclock.pts(); - if (requested_start_pts_ms < 0) - return now; - int64_t earliest = now + state_->switch_margin_ms; - if (requested_start_pts_ms < earliest) - throw Error("mixer: start_pts_ms must be at least " + std::to_string(state_->switch_margin_ms) + - "ms in the future"); - return requested_start_pts_ms; -} - -void MixerOrchestrator::defineSource(const std::string& name, const std::string& otm_node, int input_index, - const std::string& cs_node_a, const std::string& cs_node_b) { - std::lock_guard lock(state_->mutex); - MixerState::SourceInfo info; - info.otm_node_name = otm_node; - info.input_index = input_index; - info.cs_node_a = cs_node_a; - info.cs_node_b = cs_node_b; - state_->sources[name] = std::move(info); -} - -void MixerOrchestrator::defineRoutedSource(const std::string& name, const std::string& router_node, - int input_index, - const std::string& route_output_label_a, - const std::string& route_output_label_b, - const std::string& cs_node_a, const std::string& cs_node_b) { - const int output_count = routerOutputCount(nodes_, router_node); - const int route_output_a = routerOutputIndexFromLabel(nodes_, router_node, route_output_label_a); - const int route_output_b = routerOutputIndexFromLabel(nodes_, router_node, route_output_label_b); - - std::lock_guard lock(state_->mutex); - MixerState::SourceInfo info; - info.input_index = input_index; - info.cs_node_a = cs_node_a; - info.cs_node_b = cs_node_b; - info.routed = true; - info.router_node_name = router_node; - info.route_output_label_a = route_output_label_a; - info.route_output_label_b = route_output_label_b; - info.route_output_a = route_output_a; - info.route_output_b = route_output_b; - state_->sources[name] = std::move(info); - - int& stored_output_count = state_->router_output_counts[router_node]; - if (stored_output_count != 0 && stored_output_count != output_count) { - throw Error("mixer.routed_source: router " + router_node + " output count changed from " + - std::to_string(stored_output_count) + " to " + - std::to_string(output_count)); - } - stored_output_count = output_count; - ensureRouteTableSize(*state_, router_node); -} - -void MixerOrchestrator::defineScene(const std::string& name, const SceneDefinition& def) { - std::lock_guard lock(state_->mutex); - if (state_->prewarm_cut_scenes.count(name) && - (!canPrewarmScene(def) || (state_->computeActiveInputsMask(def) & ~state_->prewarm_source_mask))) - state_->prewarm_cut_scenes.erase(name); // Edited source identity takes the ordinary cold path. - state_->scenes[name] = def; -} - -bool MixerOrchestrator::canPrewarmScene(const SceneDefinition& scene) const { - if (!scene.routes.empty() || !scene.controls.empty()) return false; - for (const auto& [name, layout] : scene.sources) { - const auto source = state_->sources.find(name); - if (source == state_->sources.end() || source->second.routed || - !source->second.cs_node_a.empty() || !source->second.cs_node_b.empty() || - !layout.crop_scale_graph.empty()) return false; - } - return !scene.sources.empty(); -} - -void MixerOrchestrator::prewarmCuts(const std::vector& scenes) { - std::lock_guard lock(state_->mutex); - ensureIdle(); - uint32_t mask = 0; - for (const auto& name : scenes) { - const auto scene = state_->scenes.find(name); - if (scene == state_->scenes.end() || !canPrewarmScene(scene->second)) - throw Error("mixer.prewarm: scenes require fixed, filter-free sources without routes or controls: " + name); - mask |= state_->computeActiveInputsMask(scene->second); - } - for (const auto& slot : {state_->slot_a, state_->slot_b}) { - const auto node = nodes_->node(slot.compositor_name); - if (!node->node() || !node->parameters().contains("fps")) - throw Error("mixer.prewarm: requires created clocked compositors"); - } - for (const auto& slot : {state_->slot_a, state_->slot_b}) - setNodeObject(slot.compositor_name, "prewarm_inputs", Parameters(mask)); - state_->prewarm_cut_scenes = {scenes.begin(), scenes.end()}; - state_->prewarm_source_mask = mask; - for (const auto& [name, source] : state_->sources) { - if (source.routed) continue; - uint32_t outputs = state_->scenes.at(state_->pgm_scene_name).sources.count(name) ? state_->pgmOutputBit() : 0u; - if (!state_->pvw_scene_name.empty() && state_->scenes.at(state_->pvw_scene_name).sources.count(name)) - outputs |= state_->pvwOutputBit(); - publishCameraOtmOutputs(source.otm_node_name, state_->sourceOutputMask(source, outputs)); - } -} - -void MixerOrchestrator::applyRoutedSceneRoutesForSlot(bool is_slot_a, const SceneDefinition& scene, - int64_t at_pts_ms, bool immediate) { - auto tables = currentRouterTables(*state_); - setRoutedSlotInTables(*state_, tables, is_slot_a, &scene); - - for (const auto& [router_name, routes] : tables) { - Parameters value = routesToParameters(routes); - if (immediate) { - timeline_->clearKey(router_name, "routes"); - if (!setNodeObjectIfCreated(nodes_, router_name, "routes", value)) { - logstream << "mixer: queued " << router_name - << ".routes for router node not created yet"; - } - state_->router_routes[router_name] = routes; - } - timeline_->set(router_name, "routes", at_pts_ms, value); - } -} - -void MixerOrchestrator::publishRoutedRoutesForProgramOnly(bool pgm_is_slot_a, - const SceneDefinition& scene, - int64_t at_pts_ms, - bool immediate) { - auto tables = currentRouterTables(*state_); - setRoutedSlotInTables(*state_, tables, true, nullptr); - setRoutedSlotInTables(*state_, tables, false, nullptr); - setRoutedSlotInTables(*state_, tables, pgm_is_slot_a, &scene); - - for (const auto& [router_name, routes] : tables) { - Parameters value = routesToParameters(routes); - if (immediate) { - timeline_->clearKey(router_name, "routes"); - if (!setNodeObjectIfCreated(nodes_, router_name, "routes", value)) { - logstream << "mixer: queued " << router_name - << ".routes for router node not created yet"; - } - state_->router_routes[router_name] = routes; - } - timeline_->set(router_name, "routes", at_pts_ms, value); - } -} - -void MixerOrchestrator::initializeRoutedRoutes() { - std::lock_guard lock(state_->mutex); - for (const auto& [router_name, expected_count] : state_->router_output_counts) { - const int actual_count = routerOutputCount(nodes_, router_name); - if (expected_count != actual_count) { - throw Error("mixer.init_routes: router " + router_name + " expected output count " + - std::to_string(expected_count) + " but node dst has " + - std::to_string(actual_count)); - } - ensureRouteTableSize(*state_, router_name); - } - if (state_->pgm_scene_name.empty() || !state_->scenes.count(state_->pgm_scene_name)) - return; - publishRoutedRoutesForProgramOnly( - state_->pgm_is_slot_a, - state_->scenes.at(state_->pgm_scene_name), - wallclock.pts(), - true); -} - -void MixerOrchestrator::applyPostTransitionRouting(bool new_pgm_is_slot_a, - const std::string& new_pgm_scene) { - const auto scene_it = state_->scenes.find(new_pgm_scene); - if (scene_it == state_->scenes.end()) - return; - - const SceneDefinition& scene = scene_it->second; - const uint32_t pgm_bit = new_pgm_is_slot_a ? 1u : 2u; - const uint32_t active = state_->computeActiveInputsMask(scene); - const auto& new_slot = new_pgm_is_slot_a ? state_->slot_a : state_->slot_b; - const auto& old_slot = new_pgm_is_slot_a ? state_->slot_b : state_->slot_a; - - // Source_switcher first: this is the only setting visible at the SDI output. - // Any short window between this and the OTM/compositor flips below would only - // surface if the new direct path were not already producing frames; in both - // callers (ready cut and deferred fade cleanup) it is. - timeline_->clearKey(state_->source_switcher_name, "active"); - setNodeObject(state_->source_switcher_name, "active", - Parameters(new_pgm_is_slot_a ? 0 : 1)); - - // The encoder must not make the receiver wait for the next periodic keyframe: - // a cut changes the whole picture, and a P-frame carrying it can exceed what - // the receiver can recover from. The node coalesces bursts into one keyframe. - if (!state_->keyframe_node_name.empty()) { - try { - setNodeObject(state_->keyframe_node_name, "trigger", Parameters(true)); - } catch (const std::exception& e) { - logstream << "mixer: keyframe trigger failed: " << e.what(); - } - } - - for (const auto& [src_name, info] : state_->sources) { - if (info.routed) - continue; - const bool in_scene = scene.sources.count(src_name) > 0; - const bool active_input = (active & (1u << (unsigned)info.input_index)) != 0; - const uint32_t mask = state_->sourceOutputMask(info, (in_scene && active_input) ? pgm_bit : 0u); - timeline_->clearKey(info.otm_node_name, "outputs"); - setNodeObjectIfCreated(nodes_, info.otm_node_name, "outputs", Parameters(mask)); - } - publishRoutedRoutesForProgramOnly(new_pgm_is_slot_a, scene, wallclock.pts(), true); - - timeline_->clearKey(new_slot.post_otm_name, "outputs"); - timeline_->clearKey(old_slot.post_otm_name, "outputs"); - timeline_->clearKey(new_slot.compositor_name, "active_inputs"); - timeline_->clearKey(old_slot.compositor_name, "active_inputs"); - nodes_->node(new_slot.post_otm_name)->setObject("outputs", Parameters(1u)); - nodes_->node(old_slot.post_otm_name)->setObject("outputs", Parameters(0u)); - nodes_->node(new_slot.compositor_name)->setObject("active_inputs", Parameters(active)); - nodes_->node(old_slot.compositor_name)->setObject("active_inputs", Parameters(0u)); -} - -void MixerOrchestrator::rewriteCameraOutputsForSlot(uint32_t slot_bit, const SceneDefinition& scene) { - uint32_t active = state_->computeActiveInputsMask(scene); - for (const auto& [src_name, info] : state_->sources) { - if (info.routed) - continue; - Parameters current_val; - uint32_t mask = nodes_->node(info.otm_node_name)->getObjectTry("outputs", current_val) - ? current_val.get() - : 0u; - mask &= ~slot_bit; - if (scene.sources.count(src_name) && (active & (1u << (unsigned)info.input_index))) - mask |= slot_bit; - publishCameraOtmOutputs(info.otm_node_name, state_->sourceOutputMask(info, mask)); - } -} - -void MixerOrchestrator::loadSceneIntoSlot(bool is_slot_a, const std::string& scene_name, bool warm_cut) { - auto& scene = state_->scenes.at(scene_name); - const auto& slot = is_slot_a ? state_->slot_a : state_->slot_b; - - // The slot being loaded is the broadcast-inactive PVW slot. Its compositor - // was previously idled with active_inputs=0, so it may still hold frames on - // its input edges. If left there, the next activation starts by rendering - // stale frames and appears to lag behind the scene switch. - flushSlotEdges(is_slot_a); - // Reset on the compositor's worker thread and reject frames from before - // this load, including those still travelling through live upstream edges. - if (warm_cut && state_->prewarm_cut_scenes.count(scene_name) && canPrewarmScene(scene)) - setNodeObject(slot.compositor_name, "warm_reset", Parameters(true)); - else - resetInputIf(nodes_, slot.compositor_name); - - for (const auto& [src_name, layout] : scene.sources) { - auto src_it = state_->sources.find(src_name); - if (src_it == state_->sources.end()) continue; - const auto& info = src_it->second; - const std::string& cs_node = is_slot_a ? info.cs_node_a : info.cs_node_b; - - if (cs_node.empty()) { - if (!layout.crop_scale_graph.empty()) - throw Error("mixer: source " + src_name + " has no filter node for its scene graph"); - continue; - } - - // Only restart the crop/scale node when the graph string actually changed. - // Restarting a filter_video node tears down and rebuilds its FFmpeg filter - // graph, which briefly stops producing frames and allocates a new - // hw_frames_ctx pool. Downstream filter_video nodes now absorb pool - // rotations via a semantic hw_frames_ctx comparison so this no longer - // causes a mid-wipe EXT_NULL gap, but the restart is still a wasted - // stall and a frame-timing hiccup when the graph string is unchanged. - const auto& node_params = nodes_->node(cs_node)->parameters(); - const std::string old_graph = node_params.value("graph", std::string("")); - if (old_graph == layout.crop_scale_graph) { - logstream << "loadSceneIntoSlot: " << cs_node - << " graph unchanged – skipping restart to avoid spurious hw_frames_ctx change"; - } else { - logstream << "loadSceneIntoSlot: " << cs_node - << " graph changed (\"" << old_graph << "\" -> \"" - << layout.crop_scale_graph << "\") – restarting"; - setNodeParam(cs_node, "graph", layout.crop_scale_graph); - autoRestartNode(cs_node); - } - } - - setNodeObject(slot.compositor_name, "layers", compositorLayersFromScene(*state_, scene)); - - uint32_t active_mask = state_->computeActiveInputsMask(scene); - // Same pattern as camera otms: cuda_rect_overlay reads "active_inputs" from timeline only. - // clearKey does not touch "layers" or other keys on this compositor channel. - timeline_->clearKey(slot.compositor_name, "active_inputs"); - setNodeObject(slot.compositor_name, "active_inputs", Parameters(active_mask)); - timeline_->set(slot.compositor_name, "active_inputs", wallclock.pts(), Parameters(active_mask)); - - // Drop slot bit for every camera, then enable only sources in scene with active_inputs set. - // Keeps `outputs` consistent with compositor consumption (no frames into unused inputs). - const uint32_t slot_bit = is_slot_a ? 1u : 2u; - rewriteCameraOutputsForSlot(slot_bit, scene); - applyRoutedSceneRoutesForSlot(is_slot_a, scene, wallclock.pts(), true); - - state_->pvw_scene_name = scene_name; -} - -void MixerOrchestrator::scheduleSceneControls(const SceneDefinition& scene, int64_t at_pts_ms) { - for (const auto& control : scene.controls) { - timeline_->set(control.node_name, control.key, at_pts_ms, control.value); - logstream << "mixer scene control: " << control.node_name << "." << control.key - << " at " << at_pts_ms << " -> " << control.value; - } -} - -void MixerOrchestrator::preview(const std::string& scene_name) { - std::lock_guard lock(state_->mutex); - ensureIdle(); - if (!state_->scenes.count(scene_name)) - throw Error("mixer.preview: unknown scene: " + scene_name); - - bool pvw_is_slot_a = !state_->pgm_is_slot_a; - const auto& slot = state_->pvwSlot(); - - if (state_->pvw_scene_name == scene_name) { - logstream << "mixer preview: scene already loaded in PVW: " << scene_name; - } else { - loadSceneIntoSlot(pvw_is_slot_a, scene_name); - resetInputIf(nodes_, slot.norm_ts_name); - } - - int64_t T_prep = wallclock.pts(); - timeline_->clearKey(slot.post_otm_name, "outputs"); - setNodeObject(slot.post_otm_name, "outputs", Parameters(1u)); - timeline_->set(slot.post_otm_name, "outputs", T_prep, Parameters(1u)); - logstream << "mixer preview armed: scene=" << scene_name - << " slot=" << (pvw_is_slot_a ? 'A' : 'B') - << " post_otm " << slot.post_otm_name << "->1"; -} - -// --------------------------------------------------------------------------- -// cutInternal: the graph-level work for a hard cut, without touching -// transition_mode or pgm_is_slot_a. All values are read from pre-flip state. -// Caller must hold state_->mutex. -// Returns the earliest cut PTS (wallclock ms). Cold cuts are gated until the -// incoming direct edge has produced a fresh frame; preloaded PVW cuts only wait -// for the scheduled PTS. -// --------------------------------------------------------------------------- -int64_t MixerOrchestrator::cutInternal(const std::string& scene_name, int64_t start_pts_ms, bool warm_cut) { - bool pvw_is_slot_a = !state_->pgm_is_slot_a; - - if (state_->pvw_scene_name == scene_name) { - logstream << "mixer cut: reusing preloaded PVW scene=" << scene_name; - } else { - loadSceneIntoSlot(pvw_is_slot_a, scene_name, warm_cut); - } - - int64_t T_prep = wallclock.pts(); - int64_t T_cut = start_pts_ms; - const auto& new_slot = state_->pvwSlot(); - - // Pre-warm the hidden direct branch before the visible `out_sel` switch. Without this, - // enabling `post_otm` and switching `out_sel` at the same PTS leaves the newly selected - // path one pipeline-latency late, so the final encoder-side force_fps repeats the last - // visible frame for a few ticks across the cut. `one_to_many` with timeline runs in - // drop_dynamic_ mode, so feeding an inactive direct branch here is safe: `source_switcher` - // drains and drops those pre-roll frames instead of back-pressuring the slot. - timeline_->clearKey(new_slot.post_otm_name, "outputs"); - setNodeObject(new_slot.post_otm_name, "outputs", Parameters(1u)); - timeline_->set(new_slot.post_otm_name, "outputs", T_prep, Parameters(1u)); - - logstream << "mixer cut armed: scene=" << scene_name << " earliest T_cut=" << T_cut - << " post_otm prep " << new_slot.post_otm_name << "->1"; - - resetSlotNormFps(nodes_, *state_); - - return T_cut; -} - -// --------------------------------------------------------------------------- -// deferredCleanup: flips internal bookkeeping and deletes nodes that can't be -// removed via timeline. The scheduler decides when this runs. -// --------------------------------------------------------------------------- -void MixerOrchestrator::deferredCleanup( - std::shared_ptr nodes, - std::shared_ptr state, - std::shared_ptr timeline, - std::shared_ptr scheduler, - uint64_t transition_generation, - bool new_pgm_is_slot_a, - std::string new_pgm_scene, - int64_t end_pts_ms) { - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Crossfade)) - return; - - // Media time can lag wall time by the configured playout budget. Finish - // only after the selector has produced the first frame at/after the end, - // then release the controls immediately. Do not park the scheduler worker - // while waiting: other commands and shutdown must remain responsive. - const auto presented = edgeLastTsIfExists(nodes, - firstDstEdgeName(nodes, state->source_switcher_name)); - if (!presented.isValid() || presented < av::Timestamp(end_pts_ms, {1, 1000})) { - scheduler->postAfter("mixer.fade.presented", 2, - [nodes, state, timeline, scheduler, transition_generation, - new_pgm_is_slot_a, new_pgm_scene, end_pts_ms] { - deferredCleanup(nodes, state, timeline, scheduler, transition_generation, - new_pgm_is_slot_a, new_pgm_scene, end_pts_ms); - }); - return; - } - - std::lock_guard lock(state->mutex); - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Crossfade)) - return; - try { - MixerOrchestrator orch(nodes, state, timeline, scheduler); - orch.applyPostTransitionRouting(new_pgm_is_slot_a, new_pgm_scene); - orch.finishSnapshot(); - } catch (const std::exception& e) { - logstream << "mixer: deferred cleanup error restoring routing: " << e.what(); - } - state->pgm_is_slot_a = new_pgm_is_slot_a; - state->pgm_scene_name = std::move(new_pgm_scene); - state->pvw_scene_name = ""; - state->transition_mode = MixerState::TransitionMode::Idle; -} - -void MixerOrchestrator::readyCutTask( - std::shared_ptr nodes, - std::shared_ptr state, - std::shared_ptr timeline, - std::shared_ptr scheduler, - uint64_t transition_generation, - bool new_pgm_is_slot_a, - std::string new_pgm_scene, - std::string ready_edge_name, - av::Timestamp ready_edge_initial_ts, - int64_t earliest_switch_pts_ms, - bool require_new_ready_frame) { - constexpr int64_t kPollMs = 5; - int64_t waited_ms = 0; - while (true) { - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Cut)) - return; - const bool time_ready = wallclock.pts() >= earliest_switch_pts_ms; - bool edge_ready = !require_new_ready_frame; - if (require_new_ready_frame) { - auto edge = nodes->edges()->findAny(ready_edge_name); - if (!edge) { - logstream << "mixer ready cut: missing ready edge " << ready_edge_name - << " for scene=" << new_pgm_scene; - std::lock_guard lock(state->mutex); - MixerOrchestrator(nodes, state, timeline, scheduler).abortTransition(transition_generation); - return; - } - av::Timestamp ts = edge->lastTS(); - edge_ready = ts.isValid() && (!ready_edge_initial_ts.isValid() || ts > ready_edge_initial_ts); - } - if (edge_ready && time_ready) - break; - std::this_thread::sleep_for(std::chrono::milliseconds(kPollMs)); - waited_ms += kPollMs; - } - if (waited_ms > 0 || require_new_ready_frame) { - logstream << "mixer ready cut: scene=" << new_pgm_scene - << " waited_ms=" << waited_ms - << " require_new_ready_frame=" << (require_new_ready_frame ? "true" : "false"); - } - - std::lock_guard lock(state->mutex); - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Cut)) - return; - try { - MixerOrchestrator orch(nodes, state, timeline, scheduler); - if (state->cut_latency) state->cut_latency->timing.arm(); - orch.applyPostTransitionRouting(new_pgm_is_slot_a, new_pgm_scene); - orch.finishSnapshot(); - } catch (const std::exception& e) { - if (state->cut_latency) state->cut_latency->timing.cancel("failed"); - logstream << "mixer: ready cut error restoring routing: " << e.what(); - } - state->pgm_is_slot_a = new_pgm_is_slot_a; - state->pgm_scene_name = std::move(new_pgm_scene); - state->pvw_scene_name = ""; - state->transition_mode = MixerState::TransitionMode::Idle; -} - -// --------------------------------------------------------------------------- -// cut: PTS-scheduled hard cut. Graph work + timeline entries happen now; -// state flip is deferred until the timeline entries have taken effect. -// --------------------------------------------------------------------------- -void MixerOrchestrator::cut(const std::string& scene_name, int64_t start_pts_ms, - avp::mixer::CutLatency::Clock::time_point received) { - std::lock_guard lock(state_->mutex); - if (!state_->scenes.count(scene_name)) - throw Error("mixer: unknown scene: " + scene_name); - int64_t T_cut = resolveTransitionStartPts(start_pts_ms); - interruptTransition(); - state_->transition_mode = MixerState::TransitionMode::Cut; - uint64_t transition_generation = ++state_->transition_generation; - state_->transition_scene_name = scene_name; - TransitionPrepGuard prep_guard([&] { abortTransition(transition_generation); }); - bool pvw_is_slot_a = !state_->pgm_is_slot_a; - bool was_preloaded = state_->pvw_scene_name == scene_name; - if (state_->cut_latency) - state_->cut_latency->timing.begin(scene_name, was_preloaded, pvw_is_slot_a ? 0 : 1, received); - - scheduleSceneControls(state_->scenes.at(scene_name), T_cut); - cutInternal(scene_name, T_cut, true); - - const auto& new_slot = state_->pvwSlot(); - std::string ready_edge_name = firstDstEdgeName(nodes_, new_slot.post_otm_name); - auto ready_edge = nodes_->edges()->findAny(ready_edge_name); - av::Timestamp ready_edge_initial_ts = ready_edge ? ready_edge->lastTS() : NOTS; - postTransitionTask("mixer.cut.ready", T_cut - wallclock.pts(), - [nodes = nodes_, state = state_, timeline = timeline_, scheduler = scheduler_, - transition_generation, pvw_is_slot_a, scene_name, ready_edge_name, - ready_edge_initial_ts, T_cut, was_preloaded] { - readyCutTask(nodes, state, timeline, scheduler, transition_generation, - pvw_is_slot_a, scene_name, ready_edge_name, ready_edge_initial_ts, - T_cut, !was_preloaded); - }); - prep_guard.release(); -} - -// --------------------------------------------------------------------------- -// fade: crossfade transition through the permanent preheated CUDA filter. -// All timeline values are computed from the pre-flip state. -// --------------------------------------------------------------------------- -void MixerOrchestrator::fade(const std::string& scene_name, double duration_sec, - int64_t start_pts_ms) { - std::lock_guard lock(state_->mutex); - if (!state_->scenes.count(scene_name)) throw Error("mixer: unknown scene: " + scene_name); - if (!std::isfinite(duration_sec) || duration_sec <= 0) throw Error("mixer: invalid fade duration"); - const auto start = resolveTransitionStartPts(start_pts_ms); - interruptTransition(); - state_->transition_mode = MixerState::TransitionMode::Crossfade; - const auto generation = ++state_->transition_generation; - state_->transition_scene_name = scene_name; - TransitionPrepGuard guard([&] { abortTransition(generation); }); - cutInternal(scene_name, start); - const auto initial = edgeLastTsIfExists(nodes_, firstDstEdgeName(nodes_, state_->pvwSlot().post_otm_name)); - postTransitionTask("mixer.fade.ready", 0, - [orch = *this, scene_name, duration_sec, start, generation, initial]() mutable { - orch.startFadeWhenReady(scene_name, duration_sec, start, generation, initial, wallclock.pts() + 2000); - }); - guard.release(); -} - -void MixerOrchestrator::startFadeWhenReady(std::string scene_name, double duration_sec, - int64_t requested_pts, uint64_t generation, av::Timestamp initial_ts, int64_t deadline_ms) { - std::lock_guard lock(state_->mutex); - if (!transitionIsCurrent(state_, generation, MixerState::TransitionMode::Crossfade)) return; - TransitionPrepGuard guard([&] { abortTransition(generation); }); - const auto ready = edgeLastTsIfExists(nodes_, firstDstEdgeName(nodes_, state_->pvwSlot().post_otm_name)); - auto snapshot = InstanceSharedObjects::get( - nodes_->instanceData(), state_->source_switcher_name + "_snapshot"); - bool output_held; - { - std::lock_guard snapshot_lock(snapshot->mutex); - output_held = snapshot->frames.holding(); - } - if (output_held || !ready.isValid() || (initial_ts.isValid() && ready <= initial_ts)) { - if (wallclock.pts() >= deadline_ms) { - throw Error("mixer.fade: target scene did not produce a fresh frame within 2 seconds"); - } - postTransitionTask("mixer.fade.ready", 2, - [orch = *this, scene_name, duration_sec, requested_pts, generation, initial_ts, deadline_ms]() mutable { - orch.startFadeWhenReady(scene_name, duration_sec, requested_pts, generation, initial_ts, deadline_ms); - }); - guard.release(); - return; - } - startFade(scene_name, duration_sec, std::max(requested_pts, wallclock.pts()), generation); - guard.release(); -} - -void MixerOrchestrator::startFade(const std::string& scene_name, double duration_sec, - int64_t T_start, uint64_t transition_generation) { - // Capture all needed values from pre-flip state - bool pvw_is_slot_a = !state_->pgm_is_slot_a; - uint32_t pvw_bit = state_->pvwOutputBit(); - const auto& target_slot = pvw_is_slot_a ? state_->slot_a : state_->slot_b; - const auto& old_slot = pvw_is_slot_a ? state_->slot_b : state_->slot_a; - int pvw_sw_idx = state_->pvwSourceSwitcherIndex(); - - auto& target_scene = state_->scenes.at(scene_name); - scheduleSceneControls(target_scene, T_start); - - // 2. Update the preheated transition_cuda expression while its input - // branches are idle. The filter graph itself remains running. - std::string progress_expr = "clip((t-" + std::to_string(T_start / 1000.0) + - ")/" + std::to_string(duration_sec) + ",0,1)"; - std::string alpha_expr = pvw_is_slot_a ? "1-" + progress_expr : progress_expr; - - std::string transition_node_name = state_->source_switcher_name.empty() - ? transition_node_name_ - : state_->source_switcher_name + "_transition"; - Parameters alpha_command = { - {"target", "transition_cuda"}, - {"command", "alpha"}, - {"argument", alpha_expr}, - }; - setNodeObject(transition_node_name, "filter_command", alpha_command); - - // 3. Camera routing: applied in loadSceneIntoSlot via rewriteCameraOutputsForSlot - - // 4–5. Timeline: priming post-scene otms (direct+trans) then visible-path switches - int64_t T_prep = wallclock.pts(); - timeline_->set(state_->slot_a.post_otm_name, "outputs", T_prep, Parameters(3u)); // 0b11 warmup - timeline_->set(state_->slot_b.post_otm_name, "outputs", T_prep, Parameters(3u)); - - int64_t T_end = T_start + (int64_t)(duration_sec * 1000); - - // At T_start: switch output to transition - timeline_->set(state_->source_switcher_name, "active", T_start, - Parameters(MixerState::transSourceSwitcherIndex())); - timeline_->set(state_->slot_a.post_otm_name, "outputs", T_start, Parameters(2u)); // 0b10 trans only - timeline_->set(state_->slot_b.post_otm_name, "outputs", T_start, Parameters(2u)); - - // At T_end: switch output to new PGM direct - timeline_->set(state_->source_switcher_name, "active", T_end, Parameters(pvw_sw_idx)); - timeline_->set(target_slot.post_otm_name, "outputs", T_end, Parameters(1u)); // 0b01 direct only - timeline_->set(old_slot.post_otm_name, "outputs", T_end, Parameters(0u)); // idle - - // Camera cleanup at T_cleanup: converge to new-PGM-only bitmasks - // pvw_bit == post-flip PGM bit (the PVW slot becomes the new PGM) - int64_t T_cleanup = T_end + 100; - for (const auto& [src_name, info] : state_->sources) { - if (info.routed) - continue; - uint32_t new_mask = target_scene.sources.count(src_name) ? pvw_bit : 0u; - timeline_->set(info.otm_node_name, "outputs", T_cleanup, Parameters(state_->sourceOutputMask(info, new_mask))); - } - publishRoutedRoutesForProgramOnly(pvw_is_slot_a, target_scene, T_cleanup, false); - timeline_->set(old_slot.compositor_name, "active_inputs", T_cleanup, Parameters(0u)); - - // 6. Deferred state/routing cleanup. The transition node stays hot. - int64_t flip_delay = T_end - wallclock.pts(); - postTransitionTask("mixer.fade.cleanup", flip_delay, - [nodes = nodes_, state = state_, timeline = timeline_, scheduler = scheduler_, - transition_generation, pvw_is_slot_a, scene_name, T_end] { - deferredCleanup(nodes, state, timeline, scheduler, transition_generation, - pvw_is_slot_a, scene_name, T_end); - }); -} - -// --------------------------------------------------------------------------- -// runWipeMidpointAndCleanup: -// Phase 1 (midpoint): PVW slot prep + timeline source_switcher (hidden under opaque wipe). -// Phase 2 (end): routing cleanup, tear down wipe chain, flip state. -// -// Generation checks prevent an interrupted wipe from changing new routing. -// --------------------------------------------------------------------------- -void MixerOrchestrator::runWipeMidpointAndCleanup( - std::shared_ptr nodes, - std::shared_ptr state, - std::shared_ptr timeline, - std::shared_ptr scheduler, - uint64_t transition_generation, - std::string scene_name, - bool new_pgm_is_slot_a, - int64_t remaining_ms) { - - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return; - - // --- Phase 1: midpoint - do invisible scene switch under the fully-opaque wipe --- - try { - std::lock_guard lock(state->mutex); - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return; - MixerOrchestrator orch(nodes, state, timeline, scheduler); - - // Keep a prewarmed scene intact at the wipe midpoint. - if (state->pvw_scene_name != scene_name) - orch.loadSceneIntoSlot(new_pgm_is_slot_a, scene_name); - - // Same post-scene OTM flip as cutInternal: out_sel will read PVW `sc*_direct`, so that slot's - // `one_to_many` must have outputs=1. If it stays 0 (idle default), frames are popped from - // norm_* with nowhere to go and the wipe path freezes. Stop the old PGM branch to avoid backup. - int64_t Tw = wallclock.pts(); - const auto& new_slot = state->pvwSlot(); - const auto& old_slot = state->pgmSlot(); - timeline->clearKey(old_slot.post_otm_name, "outputs"); - orch.setNodeObject(old_slot.post_otm_name, "outputs", Parameters(0u)); - timeline->set(old_slot.post_otm_name, "outputs", Tw, Parameters(0u)); - timeline->clearKey(new_slot.post_otm_name, "outputs"); - orch.setNodeObject(new_slot.post_otm_name, "outputs", Parameters(1u)); - timeline->set(new_slot.post_otm_name, "outputs", Tw, Parameters(1u)); - - // Switch source_switcher (invisible behind wipe overlay); timeline for consistency with other switches - int sw = state->pvwSourceSwitcherIndex(); - timeline->set(state->source_switcher_name, "active", Tw, Parameters(sw)); - logstream << "mixer wipe midpoint: Tw=" << Tw << " scene=" << scene_name << " out_sel.active=" << sw - << " new_slot post_otm=" << new_slot.post_otm_name << " old_slot post_otm=" - << old_slot.post_otm_name; - resetSlotNormFps(nodes, *state); - } catch (const std::exception& e) { - logstream << "mixer: wipe midpoint error: " << e.what(); - } - - // --- Phase 2: wipe end – tear down and flip --- - // Stage 2a: wait for the wipe source to EOF (or until the planned duration elapses). - bool hit_input_eof = false; - if (remaining_ms > 0) { - int64_t waited_ms = 0; - constexpr int64_t kPollMs = 10; - while (waited_ms < remaining_ms) { - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return; - if (!nodeWorkingIfExists(nodes, state->wipe_input_node_name)) { - logstream << "mixer wipe: cleanup pulled to wipe EOF after " << waited_ms - << "ms of remaining tail"; - hit_input_eof = true; - break; - } - int64_t step_ms = std::min(kPollMs, remaining_ms - waited_ms); - std::this_thread::sleep_for(std::chrono::milliseconds(step_ms)); - waited_ms += step_ms; - } - } - - // Stage 2b: after source EOF, the tail of the wipe is still propagating through - // wipe_demux → wipe_dec → wipe_fmt → wipe_rt → wipe_rt_fps → wipe_overlay (six - // queues + filter internal buffering). Flipping `wipe_sel` now would cut those - // tail frames. Wait until the last pre-overlay edge (`wipe_tail_edge`) has been - // drained by the overlay, then give the overlay a short grace to emit the final - // blended frames through `wipe_overlay_out` to `wipe_sel`. - if (hit_input_eof && !state->wipe_tail_edge.empty()) { - constexpr int64_t kWipeDrainTimeoutMs = 1000; - constexpr int64_t kWipeDrainPollMs = 10; - // Grace period for the overlay filter to emit any frames already buffered - // in its filter graph after `wipe_tail_edge` drained. 120ms is ~3-4 frames - // at 30fps and ~7-8 frames at 60fps; both are within the typical libavfilter - // internal queue depth. If a future wipe overlay graph buffers more (e.g. - // a multi-stage temporal filter), bump this together with kWipeDrainTimeoutMs. - // Going below ~80ms risks cutting tail blended frames at 30fps. - constexpr int64_t kWipeOverlayTailMs = 120; - int64_t waited = 0; - while (waited < kWipeDrainTimeoutMs) { - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return; - if (edgeOccupiedIfExists(nodes, state->wipe_tail_edge) == 0) - break; - std::this_thread::sleep_for(std::chrono::milliseconds(kWipeDrainPollMs)); - waited += kWipeDrainPollMs; - } - for (int64_t tail = 0; tail < kWipeOverlayTailMs; tail += 5) { - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) return; - std::this_thread::sleep_for(std::chrono::milliseconds(5)); - } - logstream << "mixer wipe: drained " << state->wipe_tail_edge << " in " << waited - << "ms + " << kWipeOverlayTailMs << "ms overlay tail"; - } - - try { - std::lock_guard lock(state->mutex); - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return; - MixerOrchestrator orch(nodes, state, timeline, scheduler); - - int64_t Tw = wallclock.pts(); - if (!state->wipe_otm_name.empty()) { - timeline->clearKey(state->wipe_otm_name, "outputs"); - orch.setNodeObject(state->wipe_otm_name, "outputs", Parameters(1u)); - timeline->set(state->wipe_otm_name, "outputs", Tw, Parameters(1u)); - } - if (!state->wipe_selector_name.empty()) { - timeline->clearKey(state->wipe_selector_name, "active"); - orch.setNodeObject(state->wipe_selector_name, "active", Parameters(0)); - timeline->set(state->wipe_selector_name, "active", Tw, Parameters(0)); - } - logstream << "mixer wipe cleanup: Tw=" << Tw << " otm_final.outputs=1 wipe_sel.active=0" - << " stop_wipe_group_in_ms=" << kWipeSwitchGraceMs; - - uint32_t new_pgm_bit = state->pvwOutputBit(); - auto& scene = state->scenes.at(scene_name); - for (const auto& [src_name, info] : state->sources) { - if (info.routed) - continue; - uint32_t mask = scene.sources.count(src_name) ? new_pgm_bit : 0u; - timeline->set(info.otm_node_name, "outputs", Tw, Parameters(state->sourceOutputMask(info, mask))); - } - orch.publishRoutedRoutesForProgramOnly(new_pgm_is_slot_a, scene, Tw, true); - - const auto& old_slot = state->pgmSlot(); - timeline->set(old_slot.compositor_name, "active_inputs", Tw, Parameters(0u)); - } catch (const std::exception& e) { - logstream << "mixer: wipe cleanup error: " << e.what(); - if (transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - state->transition_mode = MixerState::TransitionMode::Idle; - return; - } - - for (int64_t waited = 0; waited < kWipeSwitchGraceMs; waited += 5) { - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) return; - std::this_thread::sleep_for(std::chrono::milliseconds(5)); - } - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return; - - try { - std::lock_guard lock(state->mutex); - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return; - MixerOrchestrator orch(nodes, state, timeline, scheduler); - - // Stop the pre-created wipe subgraph only after the direct-path switch - // has had time to land on the frame timeline. Tearing it down at the - // same wallclock instant as the switch starves wipe_sel/final_out for a - // few ticks and the encoder-side force_fps visibly repeats the last wipe frame. - if (!state->wipe_group_name.empty()) { - orch.stopGroup(state->wipe_group_name); - // Release any frames still sitting in wipe pipeline edges so they - // don't replay at the start of the next wipe. - orch.flushWipeEdges(); - } - - orch.finishSnapshot(); - state->pgm_is_slot_a = new_pgm_is_slot_a; - state->pgm_scene_name = scene_name; - state->pvw_scene_name = ""; - state->transition_mode = MixerState::TransitionMode::Idle; - } catch (const std::exception& e) { - logstream << "mixer: wipe teardown error: " << e.what(); - if (transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - state->transition_mode = MixerState::TransitionMode::Idle; - } -} - -// --------------------------------------------------------------------------- -// wipe: media wipe transition. Uses the static wipe_otm + wipe_selector -// nodes to route through the overlay without edge rewiring. -// The wipe subgraph (group wipe_group_name) is pre-created but not running -// in steady state; it is started here and stopped at the end of the wipe. -// --------------------------------------------------------------------------- -int64_t MixerOrchestrator::prepareWipe( - std::shared_ptr nodes, - std::shared_ptr state, - std::shared_ptr timeline, - std::shared_ptr scheduler, - uint64_t transition_generation, - std::string scene_name, - std::string wipe_file, - double duration_sec, - bool new_pgm_is_slot_a, - int64_t earliest_visible_pts_ms) { - std::string overlay_edge_name; - av::Timestamp overlay_initial_ts = NOTS; - int64_t T_prep = wallclock.pts(); - - // Only the serialized transition worker reuses wipe nodes. Finish retiring - // the previous clip outside the control mutex, then recheck cancellation. - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) return -1; - try { - nodes->group(state->wipe_group_name)->stopNodesAndWait(); - std::lock_guard lock(state->mutex); - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return -1; - MixerOrchestrator orch(nodes, state, timeline, scheduler); - - overlay_edge_name = edgeNameAt(nodes, state->wipe_selector_name, "src", 1); - overlay_initial_ts = edgeLastTsIfExists(nodes, overlay_edge_name); - - nodes->node(state->wipe_input_node_name)->stop(true); - orch.setNodeParam(state->wipe_input_node_name, "url", wipe_file); - orch.flushWipeEdges(); - resetInputIf(nodes, state->wipe_base_fps_name); - orch.startGroup(state->wipe_group_name); - - T_prep = wallclock.pts(); - timeline->clearKey(state->wipe_otm_name, "outputs"); - orch.setNodeObject(state->wipe_otm_name, "outputs", Parameters(3u)); // 0b11 both direct + wipe_in - timeline->set(state->wipe_otm_name, "outputs", T_prep, Parameters(3u)); - timeline->clearKey(state->wipe_selector_name, "active"); - orch.setNodeObject(state->wipe_selector_name, "active", Parameters(0)); // direct branch while prerolling - timeline->set(state->wipe_selector_name, "active", T_prep, Parameters(0)); - - int64_t total_ms = (int64_t)(duration_sec * 1000); - int64_t midpoint_ms = total_ms / 2; - logstream << "mixer wipe: scene=" << scene_name << " file=" << wipe_file << " T_prep=" << T_prep - << " requested_visible=" << earliest_visible_pts_ms << " total_ms=" << total_ms << " midpoint_ms=" << midpoint_ms - << " new_pgm_slot_" << (new_pgm_is_slot_a ? 'A' : 'B'); - } catch (const std::exception& e) { - logstream << "mixer: wipe prep error: " << e.what(); - std::lock_guard lock(state->mutex); - MixerOrchestrator(nodes, state, timeline, scheduler).abortTransition(transition_generation); - return -1; - } - - WipeReadyResult ready = waitForWipeOverlayReady( - nodes, overlay_edge_name, overlay_initial_ts, earliest_visible_pts_ms, state, transition_generation); - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) return -1; - if (!ready.ready) { - logstream << "mixer: wipe overlay did not become ready within " << kWipeReadyTimeoutMs - << "ms: edge=" << overlay_edge_name << " last_ts=" << ready.ready_ts; - std::lock_guard lock(state->mutex); - MixerOrchestrator(nodes, state, timeline, scheduler).abortTransition(transition_generation); - return -1; - } - - try { - std::lock_guard lock(state->mutex); - if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) - return -1; - MixerOrchestrator orch(nodes, state, timeline, scheduler); - int64_t T_visible = wallclock.pts(); - orch.setNodeObject(state->wipe_selector_name, "active", Parameters(1)); // wipe_overlay_out - timeline->set(state->wipe_selector_name, "active", T_visible, Parameters(1)); - logstream << "mixer wipe overlay ready: edge=" << overlay_edge_name - << " waited_ms=" << ready.waited_ms - << " ready_ts=" << ready.ready_ts - << " T_visible=" << T_visible - << " requested_visible=" << earliest_visible_pts_ms; - return T_visible; - } catch (const std::exception& e) { - logstream << "mixer: wipe visible switch error: " << e.what(); - std::lock_guard lock(state->mutex); - MixerOrchestrator(nodes, state, timeline, scheduler).abortTransition(transition_generation); - return -1; - } -} - -void MixerOrchestrator::warmupWipe(const std::string& wipe_file, int64_t timeout_ms) { - std::string overlay_edge_name; - av::Timestamp overlay_initial_ts = NOTS; - uint64_t generation; - const int64_t t0 = wallclock.pts(); - { - std::lock_guard lock(state_->mutex); - if (state_->wipe_otm_name.empty() || state_->wipe_selector_name.empty() || - state_->wipe_group_name.empty() || state_->wipe_input_node_name.empty()) - throw Error("mixer: wipe warm-up requires the wipe subgraph (see mixer.init)"); - if (state_->transition_mode.load() != MixerState::TransitionMode::Idle) - throw Error("mixer: cannot warm up the wipe during a transition"); - generation = state_->transition_generation.load(); - overlay_edge_name = edgeNameAt(nodes_, state_->wipe_selector_name, "src", 1); - overlay_initial_ts = edgeLastTsIfExists(nodes_, overlay_edge_name); - nodes_->group(state_->wipe_group_name)->stopNodesAndWait(); - nodes_->node(state_->wipe_input_node_name)->stop(true); - setNodeParam(state_->wipe_input_node_name, "url", wipe_file); - flushWipeEdges(); - resetInputIf(nodes_, state_->wipe_base_fps_name); - startGroup(state_->wipe_group_name); - // Feed the overlay's program input as a real wipe would; the selector - // stays on the direct branch so nothing of this reaches the output. - timeline_->clearKey(state_->wipe_otm_name, "outputs"); - setNodeObject(state_->wipe_otm_name, "outputs", Parameters(3u)); - } - WipeReadyResult ready = waitForWipeOverlayReady(nodes_, overlay_edge_name, overlay_initial_ts, 0, - state_, generation, timeout_ms); - { - std::lock_guard lock(state_->mutex); - timeline_->clearKey(state_->wipe_otm_name, "outputs"); - setNodeObject(state_->wipe_otm_name, "outputs", Parameters(1u)); - stopGroup(state_->wipe_group_name); - flushWipeEdges(); - } - logstream << "mixer wipe warm-up: file=" << wipe_file << (ready.ready ? " ready" : " NOT ready") - << " after " << (wallclock.pts() - t0) << "ms (overlay waited " << ready.waited_ms << "ms)"; - if (!ready.ready) - throw Error("mixer: wipe warm-up did not produce an overlay frame within " + std::to_string(timeout_ms) + "ms"); -} - -void MixerOrchestrator::wipe(const std::string& scene_name, const std::string& wipe_file, double duration_sec, - int64_t start_pts_ms) { - std::lock_guard lock(state_->mutex); - if (!state_->scenes.count(scene_name)) - throw Error("mixer: unknown scene: " + scene_name); - - if (state_->wipe_otm_name.empty() || state_->wipe_selector_name.empty()) - throw Error("mixer: wipe requires wipe_otm and wipe_selector nodes (see mixer.init)"); - if (state_->wipe_group_name.empty() || state_->wipe_input_node_name.empty()) - throw Error("mixer: wipe requires wipe_group and wipe_input_node (see mixer.init)"); - - if (!std::isfinite(duration_sec) || duration_sec <= 0) throw Error("mixer: invalid wipe duration"); - int64_t T_start = resolveTransitionStartPts(start_pts_ms); - interruptTransition(); - state_->transition_mode = MixerState::TransitionMode::Wipe; - uint64_t transition_generation = ++state_->transition_generation; - state_->transition_scene_name = scene_name; - TransitionPrepGuard prep_guard([&] { abortTransition(transition_generation); }); - bool pvw_is_slot_a = !state_->pgm_is_slot_a; - cutInternal(scene_name, T_start); - scheduleSceneControls(state_->scenes.at(scene_name), T_start); - - int64_t total_ms = (int64_t)(duration_sec * 1000); - int64_t midpoint_ms = total_ms / 2; - int64_t remaining_ms = total_ms - midpoint_ms; - int64_t now_ms = wallclock.pts(); - postTransitionTask("mixer.wipe.prepare", T_start - now_ms, - [scheduler = scheduler_, nodes = nodes_, state = state_, timeline = timeline_, transition_generation, - scene_name, wipe_file, duration_sec, pvw_is_slot_a, T_start, midpoint_ms, remaining_ms] { - int64_t T_visible = prepareWipe(nodes, state, timeline, scheduler, transition_generation, - scene_name, wipe_file, duration_sec, pvw_is_slot_a, - T_start); - if (T_visible < 0) - return; - - scheduler->postAfter("mixer.wipe.midpoint", midpoint_ms, - [nodes, state, timeline, scheduler, transition_generation, scene_name, pvw_is_slot_a, remaining_ms] { - runWipeMidpointAndCleanup(nodes, state, timeline, scheduler, transition_generation, - scene_name, pvw_is_slot_a, remaining_ms); - }); - }); - prep_guard.release(); -} - -void MixerOrchestrator::setOverlayEnabled(bool enabled, int64_t ready_timeout_ms) { - std::string source_otm_name; - std::string overlay_otm_name; - std::string selector_name; - std::string candidate_edge_name; - std::string selector_output_edge_name; - av::Timestamp visible_ts = NOTS; - av::Timestamp initial_candidate_ts = NOTS; - uint64_t generation = 0; - int64_t timeout_ms = 0; - int64_t poll_ms = 0; - const int candidate_input = enabled ? kOverlayCompositedInput : kOverlayDirectInput; - - { - std::lock_guard lock(state_->mutex); - if (state_->overlay_source_otm_name.empty() || - state_->overlay_otm_name.empty() || - state_->overlay_selector_name.empty()) { - throw Error("mixer.overlay: overlay nodes are not configured"); - } - source_otm_name = state_->overlay_source_otm_name; - overlay_otm_name = state_->overlay_otm_name; - selector_name = state_->overlay_selector_name; - timeout_ms = ready_timeout_ms >= 0 ? ready_timeout_ms : state_->overlay_ready_timeout_ms; - poll_ms = state_->overlay_ready_poll_ms; - generation = ++state_->overlay_generation; - - selector_output_edge_name = firstDstEdgeName(nodes_, selector_name); - candidate_edge_name = edgeNameAt(nodes_, selector_name, "src", candidate_input); - visible_ts = edgeLastTsIfExists(nodes_, selector_output_edge_name); - initial_candidate_ts = edgeLastTsIfExists(nodes_, candidate_edge_name); - - if (candidate_edge_name.empty()) { - throw Error("mixer.overlay: selector " + selector_name + " does not expose input " + - std::to_string(candidate_input)); - } - - if (!setNodeObjectIfCreated(nodes_, selector_name, "drop_non_monotonic", Parameters(true))) { - logstream << "mixer.overlay: queued " << selector_name - << ".drop_non_monotonic for node not created yet"; - } - - if (enabled) { - publishRuntimeObject(selector_name, "active", Parameters(kOverlayDirectInput)); - publishRuntimeObject(source_otm_name, "outputs", Parameters(1u)); - publishRuntimeObject(overlay_otm_name, "outputs", Parameters(3u)); - } else { - // Keep both legs fed until the direct leg has caught up with the last - // visible frame. The selector flips only after the wait below. - publishRuntimeObject(overlay_otm_name, "outputs", Parameters(3u)); - } - - logstream << "mixer.overlay: armed " << (enabled ? "enable" : "disable") - << " selector=" << selector_name - << " candidate_edge=" << candidate_edge_name - << " visible_ts=" << visible_ts - << " initial_candidate_ts=" << initial_candidate_ts - << " timeout_ms=" << timeout_ms; - } - - OverlayReadyResult ready = waitForOverlayBranchReady( - nodes_, state_, generation, candidate_edge_name, initial_candidate_ts, - visible_ts, timeout_ms, poll_ms); - if (ready.cancelled) { - logstream << "mixer.overlay: " << (enabled ? "enable" : "disable") - << " superseded before visible switch"; - return; - } - - { - std::lock_guard lock(state_->mutex); - if (!overlayCommandCurrent(state_, generation)) { - logstream << "mixer.overlay: " << (enabled ? "enable" : "disable") - << " superseded before finalizing"; - return; - } - - if (!ready.ready) { - publishRuntimeObject(selector_name, "active", Parameters(kOverlayDirectInput)); - publishRuntimeObject(overlay_otm_name, "outputs", Parameters(1u)); - publishRuntimeObject(source_otm_name, "outputs", Parameters(0u)); - state_->overlay_enabled = false; - std::ostringstream msg; - msg << "mixer.overlay: " << (enabled ? "overlay" : "direct") - << " branch did not reach monotonic PTS before timeout; edge=" - << candidate_edge_name << " last_ts=" << ready.ready_ts; - throw Error(msg.str()); - } - - publishRuntimeObject(selector_name, "active", Parameters(candidate_input)); - if (!enabled) { - publishRuntimeObject(overlay_otm_name, "outputs", Parameters(1u)); - publishRuntimeObject(source_otm_name, "outputs", Parameters(0u)); - } - state_->overlay_enabled = enabled; - logstream << "mixer.overlay: " << (enabled ? "enabled" : "disabled") - << " waited_ms=" << ready.waited_ms - << " ready_ts=" << ready.ready_ts - << " visible_ts=" << visible_ts; - } -} - -std::vector MixerOrchestrator::sceneNames() const { - std::lock_guard lock(state_->mutex); - std::vector names; - names.reserve(state_->scenes.size()); - for (const auto& [name, _] : state_->scenes) - names.push_back(name); - std::sort(names.begin(), names.end()); - return names; -} - -Parameters MixerOrchestrator::status() const { - std::lock_guard lock(state_->mutex); - Parameters s; - s["pgm_scene"] = state_->pgm_scene_name; - s["pvw_scene"] = state_->pvw_scene_name; - s["pgm_slot"] = state_->pgm_is_slot_a ? "A" : "B"; - s["switch_margin_ms"] = state_->switch_margin_ms; - s["now_pts_ms"] = wallclock.pts(); - s["cut_latency"] = state_->cut_latency ? state_->cut_latency->status() : Parameters(nullptr); - s["prewarm_cut_scenes"] = state_->prewarm_cut_scenes; - s["prewarm_source_mask"] = state_->prewarm_source_mask; - if (!state_->overlay_selector_name.empty()) { - s["overlay_enabled"] = state_->overlay_enabled; - s["overlay_selector"] = state_->overlay_selector_name; - } - auto mode = state_->transition_mode.load(); - switch (mode) { - case MixerState::TransitionMode::Idle: s["transition"] = "idle"; break; - case MixerState::TransitionMode::Cut: s["transition"] = "cut"; break; - case MixerState::TransitionMode::Crossfade: s["transition"] = "crossfade"; break; - case MixerState::TransitionMode::Wipe: s["transition"] = "wipe"; break; - } - return s; -} - -void MixerOrchestrator::enableCutMeasurements(const std::string& mixer_name, const std::string& encoder_name) { - std::lock_guard lock(state_->mutex); - auto selector = std::dynamic_pointer_cast(nodes_->node(state_->source_switcher_name)->node()); - auto encoder = std::dynamic_pointer_cast(nodes_->node(encoder_name)->node()); - if (!selector || !encoder || nodes_->node(encoder_name)->parameters().value("type", std::string()) != "enc_video") - throw Error("mixer.measurements requires created source_switcher and enc_video nodes"); - if (state_->cut_latency) { - if (state_->cut_latency->encoder_name != encoder_name) - throw Error("mixer.measurements is already bound to another encoder"); - return; - } - if (selector->cutLatencyProbe() || encoder->cutLatencyProbe()) - throw Error("mixer.measurements node already belongs to another probe"); - auto probe = std::make_shared(mixer_name, encoder_name); - selector->setCutLatencyProbe(probe); - encoder->setCutLatencyProbe(probe); - state_->cut_latency = std::move(probe); -} diff --git a/src/mixer/mixer_orchestrator.hpp b/src/mixer/orchestrator/MixerOrchestrator.hpp similarity index 88% rename from src/mixer/mixer_orchestrator.hpp rename to src/mixer/orchestrator/MixerOrchestrator.hpp index 21bf0b85..e786cff4 100644 --- a/src/mixer/mixer_orchestrator.hpp +++ b/src/mixer/orchestrator/MixerOrchestrator.hpp @@ -1,32 +1,22 @@ #pragma once -#include "MixerState.hpp" -#include "../SharedTimeline.hpp" -#include "../graph_mgmt.hpp" -#include "../instance_shared.hpp" +#include "../primitives/MixerState.hpp" +#include "../TransitionScheduler.hpp" +#include "../../SharedTimeline.hpp" +#include "../../graph_mgmt.hpp" +#include "../../instance_shared.hpp" #include #include #include #include #include -class MixerTransitionScheduler : public InstanceShared, public IShutdownable { - struct Impl; - std::unique_ptr impl_; - -public: - MixerTransitionScheduler(); - ~MixerTransitionScheduler(); - - void post(std::string label, std::function task); - void postAfter(std::string label, int64_t delay_ms, std::function task); - void shutdown(); -}; +namespace avp::mixer { class MixerOrchestrator { std::shared_ptr nodes_; std::shared_ptr state_; std::shared_ptr timeline_; - std::shared_ptr scheduler_; + std::shared_ptr scheduler_; void setNodeObject(const std::string& node_name, const std::string& key, const Parameters& value); void postTransitionTask(std::string label, int64_t delay_ms, std::function task); @@ -76,13 +66,13 @@ class MixerOrchestrator { void abortTransition(uint64_t generation) noexcept; void startFadeWhenReady(std::string scene_name, double duration_sec, int64_t requested_pts, uint64_t generation, av::Timestamp initial_ts, int64_t deadline_ms); - void startFade(const std::string& scene_name, double duration_sec, int64_t T_start, + void startFade(const std::string& scene_name, double duration_sec, int64_t start_ms, uint64_t transition_generation); int64_t resolveTransitionStartPts(int64_t requested_start_pts_ms) const; // Core hard-cut logic: ensure PVW is configured, enable cameras, write timeline entries. // Does NOT modify pgm_is_slot_a, pgm_scene_name, or transition_mode. - // Caller must hold state_->mutex. Returns T_cleanup timestamp. + // Caller must hold state_->mutex. Returns cleanup_ms timestamp. int64_t cutInternal(const std::string& scene_name, int64_t start_pts_ms, bool warm_cut = false); // Complete crossfade routing and state once the final frame is presented. @@ -92,7 +82,7 @@ class MixerOrchestrator { static void deferredCleanup(std::shared_ptr nodes, std::shared_ptr state, std::shared_ptr timeline, - std::shared_ptr scheduler, + std::shared_ptr scheduler, uint64_t transition_generation, bool new_pgm_is_slot_a, std::string new_pgm_scene, @@ -100,7 +90,7 @@ class MixerOrchestrator { static void readyCutTask(std::shared_ptr nodes, std::shared_ptr state, std::shared_ptr timeline, - std::shared_ptr scheduler, + std::shared_ptr scheduler, uint64_t transition_generation, bool new_pgm_is_slot_a, std::string new_pgm_scene, @@ -114,7 +104,7 @@ class MixerOrchestrator { static void runWipeMidpointAndCleanup(std::shared_ptr nodes, std::shared_ptr state, std::shared_ptr timeline, - std::shared_ptr scheduler, + std::shared_ptr scheduler, uint64_t transition_generation, std::string scene_name, bool new_pgm_is_slot_a, @@ -122,7 +112,7 @@ class MixerOrchestrator { static int64_t prepareWipe(std::shared_ptr nodes, std::shared_ptr state, std::shared_ptr timeline, - std::shared_ptr scheduler, + std::shared_ptr scheduler, uint64_t transition_generation, std::string scene_name, std::string wipe_file, @@ -139,7 +129,7 @@ class MixerOrchestrator { MixerOrchestrator(std::shared_ptr nodes, std::shared_ptr state, std::shared_ptr timeline, - std::shared_ptr scheduler = nullptr); + std::shared_ptr scheduler = nullptr); void defineSource(const std::string& name, const std::string& otm_node, int input_index, const std::string& cs_node_a, const std::string& cs_node_b); @@ -170,3 +160,5 @@ class MixerOrchestrator { std::vector sceneNames() const; Parameters status() const; }; + +} // namespace avp::mixer diff --git a/src/mixer/orchestrator/core.cpp b/src/mixer/orchestrator/core.cpp new file mode 100644 index 00000000..314aee4f --- /dev/null +++ b/src/mixer/orchestrator/core.cpp @@ -0,0 +1,325 @@ +// MixerOrchestrator core: construction, node/edge adapters, transition +// interruption and abort, and status. Scene loading, cut, fade, wipe and +// overlay live in the sibling scene/cut/fade/wipe/overlay.cpp files. +#include "internal.hpp" + +namespace avp::mixer { + +MixerOrchestrator::MixerOrchestrator( + std::shared_ptr nodes, + std::shared_ptr state, + std::shared_ptr timeline, + std::shared_ptr scheduler) + : nodes_(std::move(nodes)), + state_(std::move(state)), + timeline_(std::move(timeline)), + scheduler_(std::move(scheduler)) {} + +void MixerOrchestrator::postTransitionTask(std::string label, int64_t delay_ms, std::function task) { + if (!scheduler_) + throw Error("mixer: transition scheduler is not configured"); + scheduler_->postAfter(std::move(label), delay_ms, std::move(task)); +} + +void MixerOrchestrator::setNodeObject(const std::string& node_name, const std::string& key, const Parameters& value) { + try { + auto node = nodes_->node(node_name); + node->setObject(key, value); + } catch (const std::exception& e) { + throw Error("mixer: set " + node_name + "." + key + " failed: " + e.what()); + } +} + +void MixerOrchestrator::publishRuntimeObject(const std::string& node_name, + const std::string& key, + const Parameters& value) { + timeline_->clearKey(node_name, key); + if (!setNodeObjectIfCreated(nodes_, node_name, key, value)) { + logstream << "mixer: queued " << node_name << "." << key + << " for node not created yet"; + } + timeline_->set(node_name, key, wallclock.pts(), value); +} + +void MixerOrchestrator::publishCameraOtmOutputs(const std::string& otm_name, uint32_t mask) { + // `one_to_many` with `timeline` uses tlGetRaw("outputs") whenever any entry matches; stale + // rows (e.g. an old cleanup_ms) would override setObject. We drop only the "outputs" key on + // this OTM channel — not post-scene otms, not source_switcher, not other keys here. + // cut/fade append new `outputs` rows at cleanup_ms *after* loadSceneIntoSlot returns, so + // those are not cleared by this call. Overlapping mixer commands are rejected by ensureIdle(). + timeline_->clearKey(otm_name, "outputs"); + if (!setNodeObjectIfCreated(nodes_, otm_name, "outputs", Parameters(mask))) { + logstream << "mixer: queued " << otm_name << ".outputs=" << mask + << " for node not created yet"; + } + timeline_->set(otm_name, "outputs", wallclock.pts(), Parameters(mask)); +} + +void MixerOrchestrator::setNodeParam(const std::string& node_name, const std::string& param, const std::string& value) { + auto node = nodes_->node(node_name); + auto& params = node->parameters(); + params[param] = value; +} + +void MixerOrchestrator::autoRestartNode(const std::string& node_name) { + auto node = nodes_->node(node_name); + node->stop(false); +} + +void MixerOrchestrator::createAndStartNode(const Parameters& params) { + Parameters p = params; + nodes_->createNode(p, true, true); +} + +void MixerOrchestrator::deleteNodeIfExists(const std::string& name) { + auto node = nodes_->node_if_exists(name); + if (node) { + nodes_->deleteNode(name); + } +} + +void MixerOrchestrator::startGroup(const std::string& group_name) { + nodes_->group(group_name)->startNodes(); +} + +void MixerOrchestrator::stopGroup(const std::string& group_name) { + nodes_->group(group_name)->stopNodes(); +} + +void MixerOrchestrator::flushWipeEdges() { + for (const auto& name : state_->wipe_flush_edges) { + auto edge = nodes_->edges()->findAny(name); + if (!edge) + continue; + + int occupied = edge->occupied(); + if (occupied <= 0) + continue; + + auto active_consumer = workingConsumerForEdge(nodes_, name); + if (active_consumer) { + logstream << "mixer: leaving active wipe edge " << name + << " unflushed (" << occupied << " queued, consumer=" + << active_consumer->name() << ")"; + continue; + } + + logstream << "mixer: flushing stale wipe edge " << name + << " (" << occupied << " queued)"; + edge->clear(); + } +} + +void MixerOrchestrator::flushSlotEdges(bool is_slot_a) { + const auto& slot = is_slot_a ? state_->slot_a : state_->slot_b; + + auto clearEdge = [this](const std::string& name) { + if (name.empty()) + return; + auto edge = nodes_->edges()->findAny(name); + if (edge && edge->occupied() > 0) { + // readerwriterqueue is SPSC: clearing from this control thread would + // consume the queue concurrently with the node that owns the edge. + if (!edge->consumer().expired()) { + logstream << "mixer: leaving live slot edge " << name + << " unflushed (" << edge->occupied() << " queued)"; + return; + } + logstream << "mixer: flushing stale slot edge " << name + << " (" << edge->occupied() << " queued)"; + edge->clear(); + } + }; + + for (const auto& [_, info] : state_->sources) { + const std::string& cs_node = is_slot_a ? info.cs_node_a : info.cs_node_b; + auto node = nodes_->node_if_exists(cs_node); + if (!node) + continue; + const auto& params = node->parameters(); + if (params.count("src")) { + for (const auto& edge_name : jsonToStringList(params["src"])) + clearEdge(edge_name); + } + if (params.count("dst")) { + for (const auto& edge_name : jsonToStringList(params["dst"])) + clearEdge(edge_name); + } + } + + for (const std::string& node_name : {slot.compositor_name, slot.norm_ts_name, slot.post_otm_name}) { + auto node = nodes_->node_if_exists(node_name); + if (!node) + continue; + const auto& params = node->parameters(); + if (params.count("dst")) { + for (const auto& edge_name : jsonToStringList(params["dst"])) + clearEdge(edge_name); + } + } +} + +void MixerOrchestrator::ensureIdle() const { + auto mode = state_->transition_mode.load(); + if (mode != MixerState::TransitionMode::Idle) + throw Error("mixer: transition already in progress"); +} + +void MixerOrchestrator::interruptTransition() { + if (state_->cut_latency) state_->cut_latency->timing.cancel(); + if (state_->transition_mode == MixerState::TransitionMode::Idle) return; + const auto previous_mode = state_->transition_mode.load(); + // A crossfade's blended picture exists only inside the transition compositor, + // so it has to be frozen to survive the interruption. A wipe's does not: the + // output carries the wipe graphic, and freezing that would paint the graphic + // into the program, where the next wipe would composite over it and the two + // would stack. In every other mode the program slot keeps rendering, so the + // direct branch restored below is already the right picture. + const bool freeze = previous_mode == MixerState::TransitionMode::Crossfade; + auto snapshot = InstanceSharedObjects::get( + nodes_->instanceData(), state_->source_switcher_name + "_snapshot"); + if (freeze) { + std::lock_guard lock(snapshot->mutex); + if (!snapshot->output_connected) + throw Error("mixer: interruption requires mixer_snapshot output and slot nodes"); + snapshot->frames.capture(state_->pgmSourceSwitcherIndex()); + } + const auto generation = ++state_->transition_generation; + TransitionGuard guard([&] { abortTransition(generation); }); + restoreProgramRouting(); + if (previous_mode == MixerState::TransitionMode::Wipe && !state_->wipe_group_name.empty()) { + // Group management retires the old decoder independently. Waiting here + // would add teardown time to every correction, including a hard cut. + stopGroup(state_->wipe_group_name); + } + state_->pvw_scene_name.clear(); + state_->transition_mode = MixerState::TransitionMode::Idle; + { + std::lock_guard lock(snapshot->mutex); + if (freeze) { + snapshot->frames.arm(wallclock.pts() * 1000000); + } else { + // Drop any substitution an earlier interruption left in the slot. + snapshot->frames.finish(); + snapshot->frames.arm(wallclock.pts() * 1000000, false); + } + } + guard.release(); + logstream << "mixer: interrupted transition; " << (freeze ? "retained the blended picture" + : "returned to the program picture"); +} + +void MixerOrchestrator::restoreProgramRouting() { + // Remove scheduled controls as well as routes; cancelling a worker alone + // cannot cancel a future selector flip. + if (auto scene = state_->scenes.find(state_->transition_scene_name); scene != state_->scenes.end()) { + for (const auto& control : scene->second.controls) + timeline_->clearKey(control.node_name, control.key); + } + applyPostTransitionRouting(state_->pgm_is_slot_a, state_->pgm_scene_name); + scheduleSceneControls(state_->scenes.at(state_->pgm_scene_name), wallclock.pts()); + for (const auto& [name, key, value] : std::vector>{ + {state_->wipe_selector_name, "active", 0}, {state_->wipe_otm_name, "outputs", 1}}) { + if (name.empty()) continue; + timeline_->clearKey(name, key); + setNodeObject(name, key, Parameters(value)); + } +} + +void MixerOrchestrator::abortTransition(uint64_t generation) noexcept { + // All callers hold the control mutex. A cancelled worker must never undo + // the routing of its replacement transition. + if (state_->transition_generation != generation) return; + if (state_->cut_latency) state_->cut_latency->timing.cancel("failed"); + ++state_->transition_generation; + const auto mode = state_->transition_mode.load(); + auto cleanup = [](auto action) { + try { action(); } + catch (const std::exception& e) { + logstream << "mixer: transition abort cleanup failed: " << e.what(); + } + }; + cleanup([&] { restoreProgramRouting(); }); + if (mode == MixerState::TransitionMode::Wipe && !state_->wipe_group_name.empty()) + cleanup([&] { stopGroup(state_->wipe_group_name); }); + // Remove slot substitution even when the target never produced a frame. + // The output gate still waits for a fresh program frame before releasing. + cleanup([&] { finishSnapshot(); }); + state_->pvw_scene_name.clear(); + state_->transition_mode = MixerState::TransitionMode::Idle; +} + +void MixerOrchestrator::finishSnapshot() { + auto snapshot = InstanceSharedObjects::get( + nodes_->instanceData(), state_->source_switcher_name + "_snapshot"); + std::lock_guard lock(snapshot->mutex); + snapshot->frames.finish(); + snapshot->frames.arm(wallclock.pts() * 1000000, false); +} + +int64_t MixerOrchestrator::resolveTransitionStartPts(int64_t requested_start_pts_ms) const { + int64_t now = wallclock.pts(); + if (requested_start_pts_ms < 0) + return now; + int64_t earliest = now + state_->switch_margin_ms; + if (requested_start_pts_ms < earliest) + throw Error("mixer: start_pts_ms must be at least " + std::to_string(state_->switch_margin_ms) + + "ms in the future"); + return requested_start_pts_ms; +} + +std::vector MixerOrchestrator::sceneNames() const { + std::lock_guard lock(state_->mutex); + std::vector names; + names.reserve(state_->scenes.size()); + for (const auto& [name, _] : state_->scenes) + names.push_back(name); + std::sort(names.begin(), names.end()); + return names; +} + +Parameters MixerOrchestrator::status() const { + std::lock_guard lock(state_->mutex); + Parameters s; + s["pgm_scene"] = state_->pgm_scene_name; + s["pvw_scene"] = state_->pvw_scene_name; + s["pgm_slot"] = state_->pgm_is_slot_a ? "A" : "B"; + s["switch_margin_ms"] = state_->switch_margin_ms; + s["now_pts_ms"] = wallclock.pts(); + s["cut_latency"] = state_->cut_latency ? state_->cut_latency->status() : Parameters(nullptr); + s["prewarm_cut_scenes"] = state_->prewarm_cut_scenes; + s["prewarm_source_mask"] = state_->prewarm_source_mask; + if (!state_->overlay_selector_name.empty()) { + s["overlay_enabled"] = state_->overlay_enabled; + s["overlay_selector"] = state_->overlay_selector_name; + } + auto mode = state_->transition_mode.load(); + switch (mode) { + case MixerState::TransitionMode::Idle: s["transition"] = "idle"; break; + case MixerState::TransitionMode::Cut: s["transition"] = "cut"; break; + case MixerState::TransitionMode::Crossfade: s["transition"] = "crossfade"; break; + case MixerState::TransitionMode::Wipe: s["transition"] = "wipe"; break; + } + return s; +} + +void MixerOrchestrator::enableCutMeasurements(const std::string& mixer_name, const std::string& encoder_name) { + std::lock_guard lock(state_->mutex); + auto selector = std::dynamic_pointer_cast(nodes_->node(state_->source_switcher_name)->node()); + auto encoder = std::dynamic_pointer_cast(nodes_->node(encoder_name)->node()); + if (!selector || !encoder || nodes_->node(encoder_name)->parameters().value("type", std::string()) != "enc_video") + throw Error("mixer.measurements requires created source_switcher and enc_video nodes"); + if (state_->cut_latency) { + if (state_->cut_latency->encoder_name != encoder_name) + throw Error("mixer.measurements is already bound to another encoder"); + return; + } + if (selector->cutLatencyProbe() || encoder->cutLatencyProbe()) + throw Error("mixer.measurements node already belongs to another probe"); + auto probe = std::make_shared(mixer_name, encoder_name); + selector->setCutLatencyProbe(probe); + encoder->setCutLatencyProbe(probe); + state_->cut_latency = std::move(probe); +} + +} // namespace avp::mixer diff --git a/src/mixer/orchestrator/cut.cpp b/src/mixer/orchestrator/cut.cpp new file mode 100644 index 00000000..d2ca105b --- /dev/null +++ b/src/mixer/orchestrator/cut.cpp @@ -0,0 +1,143 @@ +// Hard cut: arm the PVW direct branch, then flip routing once the ready edge +// has produced a fresh frame at or after the scheduled PTS. +#include "internal.hpp" + +namespace avp::mixer { + +// --------------------------------------------------------------------------- +// cutInternal: the graph-level work for a hard cut, without touching +// transition_mode or pgm_is_slot_a. All values are read from pre-flip state. +// Caller must hold state_->mutex. +// Returns the earliest cut PTS (wallclock ms). Cold cuts are gated until the +// incoming direct edge has produced a fresh frame; preloaded PVW cuts only wait +// for the scheduled PTS. +// --------------------------------------------------------------------------- +int64_t MixerOrchestrator::cutInternal(const std::string& scene_name, int64_t start_pts_ms, bool warm_cut) { + bool pvw_is_slot_a = !state_->pgm_is_slot_a; + + if (state_->pvw_scene_name == scene_name) { + logstream << "mixer cut: reusing preloaded PVW scene=" << scene_name; + } else { + loadSceneIntoSlot(pvw_is_slot_a, scene_name, warm_cut); + } + + int64_t prep_ms = wallclock.pts(); + int64_t cut_ms = start_pts_ms; + const auto& new_slot = state_->pvwSlot(); + + // Pre-warm the hidden direct branch before the visible `out_sel` switch. Without this, + // enabling `post_otm` and switching `out_sel` at the same PTS leaves the newly selected + // path one pipeline-latency late, so the final encoder-side force_fps repeats the last + // visible frame for a few ticks across the cut. `one_to_many` with timeline runs in + // drop_dynamic_ mode, so feeding an inactive direct branch here is safe: `source_switcher` + // drains and drops those pre-roll frames instead of back-pressuring the slot. + timeline_->clearKey(new_slot.post_otm_name, "outputs"); + setNodeObject(new_slot.post_otm_name, "outputs", Parameters(1u)); + timeline_->set(new_slot.post_otm_name, "outputs", prep_ms, Parameters(1u)); + + logstream << "mixer cut armed: scene=" << scene_name << " earliest cut_ms=" << cut_ms + << " post_otm prep " << new_slot.post_otm_name << "->1"; + + resetSlotNormFps(nodes_, *state_); + + return cut_ms; +} + +void MixerOrchestrator::readyCutTask( + std::shared_ptr nodes, + std::shared_ptr state, + std::shared_ptr timeline, + std::shared_ptr scheduler, + uint64_t transition_generation, + bool new_pgm_is_slot_a, + std::string new_pgm_scene, + std::string ready_edge_name, + av::Timestamp ready_edge_initial_ts, + int64_t earliest_switch_pts_ms, + bool require_new_ready_frame) { + int64_t waited_ms = 0; + while (true) { + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Cut)) + return; + const bool time_ready = wallclock.pts() >= earliest_switch_pts_ms; + bool edge_ready = !require_new_ready_frame; + if (require_new_ready_frame) { + auto edge = nodes->edges()->findAny(ready_edge_name); + if (!edge) { + logstream << "mixer ready cut: missing ready edge " << ready_edge_name + << " for scene=" << new_pgm_scene; + std::lock_guard lock(state->mutex); + MixerOrchestrator(nodes, state, timeline, scheduler).abortTransition(transition_generation); + return; + } + av::Timestamp ts = edge->lastTS(); + edge_ready = ts.isValid() && (!ready_edge_initial_ts.isValid() || ts > ready_edge_initial_ts); + } + if (edge_ready && time_ready) + break; + std::this_thread::sleep_for(std::chrono::milliseconds(kPollMs)); + waited_ms += kPollMs; + } + if (waited_ms > 0 || require_new_ready_frame) { + logstream << "mixer ready cut: scene=" << new_pgm_scene + << " waited_ms=" << waited_ms + << " require_new_ready_frame=" << (require_new_ready_frame ? "true" : "false"); + } + + std::lock_guard lock(state->mutex); + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Cut)) + return; + try { + MixerOrchestrator orch(nodes, state, timeline, scheduler); + if (state->cut_latency) state->cut_latency->timing.arm(); + orch.applyPostTransitionRouting(new_pgm_is_slot_a, new_pgm_scene); + orch.finishSnapshot(); + } catch (const std::exception& e) { + if (state->cut_latency) state->cut_latency->timing.cancel("failed"); + logstream << "mixer: ready cut error restoring routing: " << e.what(); + } + state->pgm_is_slot_a = new_pgm_is_slot_a; + state->pgm_scene_name = std::move(new_pgm_scene); + state->pvw_scene_name = ""; + state->transition_mode = MixerState::TransitionMode::Idle; +} + +// --------------------------------------------------------------------------- +// cut: PTS-scheduled hard cut. Graph work + timeline entries happen now; +// state flip is deferred until the timeline entries have taken effect. +// --------------------------------------------------------------------------- +void MixerOrchestrator::cut(const std::string& scene_name, int64_t start_pts_ms, + avp::mixer::CutLatency::Clock::time_point received) { + std::lock_guard lock(state_->mutex); + if (!state_->scenes.count(scene_name)) + throw Error("mixer: unknown scene: " + scene_name); + int64_t cut_ms = resolveTransitionStartPts(start_pts_ms); + interruptTransition(); + state_->transition_mode = MixerState::TransitionMode::Cut; + uint64_t transition_generation = ++state_->transition_generation; + state_->transition_scene_name = scene_name; + TransitionGuard prep_guard([&] { abortTransition(transition_generation); }); + bool pvw_is_slot_a = !state_->pgm_is_slot_a; + bool was_preloaded = state_->pvw_scene_name == scene_name; + if (state_->cut_latency) + state_->cut_latency->timing.begin(scene_name, was_preloaded, pvw_is_slot_a ? 0 : 1, received); + + scheduleSceneControls(state_->scenes.at(scene_name), cut_ms); + cutInternal(scene_name, cut_ms, true); + + const auto& new_slot = state_->pvwSlot(); + std::string ready_edge_name = firstDstEdgeName(nodes_, new_slot.post_otm_name); + auto ready_edge = nodes_->edges()->findAny(ready_edge_name); + av::Timestamp ready_edge_initial_ts = ready_edge ? ready_edge->lastTS() : NOTS; + postTransitionTask("mixer.cut.ready", cut_ms - wallclock.pts(), + [nodes = nodes_, state = state_, timeline = timeline_, scheduler = scheduler_, + transition_generation, pvw_is_slot_a, scene_name, ready_edge_name, + ready_edge_initial_ts, cut_ms, was_preloaded] { + readyCutTask(nodes, state, timeline, scheduler, transition_generation, + pvw_is_slot_a, scene_name, ready_edge_name, ready_edge_initial_ts, + cut_ms, !was_preloaded); + }); + prep_guard.release(); +} + +} // namespace avp::mixer diff --git a/src/mixer/orchestrator/fade.cpp b/src/mixer/orchestrator/fade.cpp new file mode 100644 index 00000000..1425018e --- /dev/null +++ b/src/mixer/orchestrator/fade.cpp @@ -0,0 +1,177 @@ +// Crossfade through the preheated transition_cuda filter, with the deferred +// routing flip once the last blended frame has been presented. +#include "internal.hpp" + +namespace avp::mixer { + +// --------------------------------------------------------------------------- +// deferredCleanup: flips internal bookkeeping and deletes nodes that can't be +// removed via timeline. The scheduler decides when this runs. +// --------------------------------------------------------------------------- +void MixerOrchestrator::deferredCleanup( + std::shared_ptr nodes, + std::shared_ptr state, + std::shared_ptr timeline, + std::shared_ptr scheduler, + uint64_t transition_generation, + bool new_pgm_is_slot_a, + std::string new_pgm_scene, + int64_t end_pts_ms) { + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Crossfade)) + return; + + // Media time can lag wall time by the configured playout budget. Finish + // only after the selector has produced the first frame at/after the end, + // then release the controls immediately. Do not park the scheduler worker + // while waiting: other commands and shutdown must remain responsive. + const auto presented = edgeLastTsIfExists(nodes, + firstDstEdgeName(nodes, state->source_switcher_name)); + if (!presented.isValid() || presented < av::Timestamp(end_pts_ms, {1, 1000})) { + scheduler->postAfter("mixer.fade.presented", 2, + [nodes, state, timeline, scheduler, transition_generation, + new_pgm_is_slot_a, new_pgm_scene, end_pts_ms] { + deferredCleanup(nodes, state, timeline, scheduler, transition_generation, + new_pgm_is_slot_a, new_pgm_scene, end_pts_ms); + }); + return; + } + + std::lock_guard lock(state->mutex); + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Crossfade)) + return; + try { + MixerOrchestrator orch(nodes, state, timeline, scheduler); + orch.applyPostTransitionRouting(new_pgm_is_slot_a, new_pgm_scene); + orch.finishSnapshot(); + } catch (const std::exception& e) { + logstream << "mixer: deferred cleanup error restoring routing: " << e.what(); + } + state->pgm_is_slot_a = new_pgm_is_slot_a; + state->pgm_scene_name = std::move(new_pgm_scene); + state->pvw_scene_name = ""; + state->transition_mode = MixerState::TransitionMode::Idle; +} + +// --------------------------------------------------------------------------- +// fade: crossfade transition through the permanent preheated CUDA filter. +// All timeline values are computed from the pre-flip state. +// --------------------------------------------------------------------------- +void MixerOrchestrator::fade(const std::string& scene_name, double duration_sec, + int64_t start_pts_ms) { + std::lock_guard lock(state_->mutex); + if (!state_->scenes.count(scene_name)) throw Error("mixer: unknown scene: " + scene_name); + if (!std::isfinite(duration_sec) || duration_sec <= 0) throw Error("mixer: invalid fade duration"); + const auto start = resolveTransitionStartPts(start_pts_ms); + interruptTransition(); + state_->transition_mode = MixerState::TransitionMode::Crossfade; + const auto generation = ++state_->transition_generation; + state_->transition_scene_name = scene_name; + TransitionGuard guard([&] { abortTransition(generation); }); + cutInternal(scene_name, start); + const auto initial = edgeLastTsIfExists(nodes_, firstDstEdgeName(nodes_, state_->pvwSlot().post_otm_name)); + postTransitionTask("mixer.fade.ready", 0, + [orch = *this, scene_name, duration_sec, start, generation, initial]() mutable { + orch.startFadeWhenReady(scene_name, duration_sec, start, generation, initial, wallclock.pts() + 2000); + }); + guard.release(); +} + +void MixerOrchestrator::startFadeWhenReady(std::string scene_name, double duration_sec, + int64_t requested_pts, uint64_t generation, av::Timestamp initial_ts, int64_t deadline_ms) { + std::lock_guard lock(state_->mutex); + if (!transitionIsCurrent(state_, generation, MixerState::TransitionMode::Crossfade)) return; + TransitionGuard guard([&] { abortTransition(generation); }); + const auto ready = edgeLastTsIfExists(nodes_, firstDstEdgeName(nodes_, state_->pvwSlot().post_otm_name)); + auto snapshot = InstanceSharedObjects::get( + nodes_->instanceData(), state_->source_switcher_name + "_snapshot"); + bool output_held; + { + std::lock_guard snapshot_lock(snapshot->mutex); + output_held = snapshot->frames.holding(); + } + if (output_held || !ready.isValid() || (initial_ts.isValid() && ready <= initial_ts)) { + if (wallclock.pts() >= deadline_ms) { + throw Error("mixer.fade: target scene did not produce a fresh frame within 2 seconds"); + } + postTransitionTask("mixer.fade.ready", 2, + [orch = *this, scene_name, duration_sec, requested_pts, generation, initial_ts, deadline_ms]() mutable { + orch.startFadeWhenReady(scene_name, duration_sec, requested_pts, generation, initial_ts, deadline_ms); + }); + guard.release(); + return; + } + startFade(scene_name, duration_sec, std::max(requested_pts, wallclock.pts()), generation); + guard.release(); +} + +void MixerOrchestrator::startFade(const std::string& scene_name, double duration_sec, + int64_t start_ms, uint64_t transition_generation) { + // Capture all needed values from pre-flip state + bool pvw_is_slot_a = !state_->pgm_is_slot_a; + uint32_t pvw_bit = state_->pvwOutputBit(); + const auto& target_slot = pvw_is_slot_a ? state_->slot_a : state_->slot_b; + const auto& old_slot = pvw_is_slot_a ? state_->slot_b : state_->slot_a; + int pvw_sw_idx = state_->pvwSourceSwitcherIndex(); + + auto& target_scene = state_->scenes.at(scene_name); + scheduleSceneControls(target_scene, start_ms); + + // 2. Update the preheated transition_cuda expression while its input + // branches are idle. The filter graph itself remains running. + std::string progress_expr = "clip((t-" + std::to_string(start_ms / 1000.0) + + ")/" + std::to_string(duration_sec) + ",0,1)"; + std::string alpha_expr = pvw_is_slot_a ? "1-" + progress_expr : progress_expr; + + std::string transition_node_name = state_->source_switcher_name.empty() + ? transition_node_name_ + : state_->source_switcher_name + "_transition"; + Parameters alpha_command = { + {"target", "transition_cuda"}, + {"command", "alpha"}, + {"argument", alpha_expr}, + }; + setNodeObject(transition_node_name, "filter_command", alpha_command); + + // 3. Camera routing: applied in loadSceneIntoSlot via rewriteCameraOutputsForSlot + + // 4–5. Timeline: priming post-scene otms (direct+trans) then visible-path switches + int64_t prep_ms = wallclock.pts(); + timeline_->set(state_->slot_a.post_otm_name, "outputs", prep_ms, Parameters(3u)); // 0b11 warmup + timeline_->set(state_->slot_b.post_otm_name, "outputs", prep_ms, Parameters(3u)); + + int64_t end_ms = start_ms + (int64_t)(duration_sec * 1000); + + // At start_ms: switch output to transition + timeline_->set(state_->source_switcher_name, "active", start_ms, + Parameters(MixerState::transSourceSwitcherIndex())); + timeline_->set(state_->slot_a.post_otm_name, "outputs", start_ms, Parameters(2u)); // 0b10 trans only + timeline_->set(state_->slot_b.post_otm_name, "outputs", start_ms, Parameters(2u)); + + // At end_ms: switch output to new PGM direct + timeline_->set(state_->source_switcher_name, "active", end_ms, Parameters(pvw_sw_idx)); + timeline_->set(target_slot.post_otm_name, "outputs", end_ms, Parameters(1u)); // 0b01 direct only + timeline_->set(old_slot.post_otm_name, "outputs", end_ms, Parameters(0u)); // idle + + // Camera cleanup at cleanup_ms: converge to new-PGM-only bitmasks + // pvw_bit == post-flip PGM bit (the PVW slot becomes the new PGM) + int64_t cleanup_ms = end_ms + 100; + for (const auto& [src_name, info] : state_->sources) { + if (info.routed) + continue; + uint32_t new_mask = target_scene.sources.count(src_name) ? pvw_bit : 0u; + timeline_->set(info.otm_node_name, "outputs", cleanup_ms, Parameters(state_->sourceOutputMask(info, new_mask))); + } + publishRoutedRoutesForProgramOnly(pvw_is_slot_a, target_scene, cleanup_ms, false); + timeline_->set(old_slot.compositor_name, "active_inputs", cleanup_ms, Parameters(0u)); + + // 6. Deferred state/routing cleanup. The transition node stays hot. + int64_t flip_delay = end_ms - wallclock.pts(); + postTransitionTask("mixer.fade.cleanup", flip_delay, + [nodes = nodes_, state = state_, timeline = timeline_, scheduler = scheduler_, + transition_generation, pvw_is_slot_a, scene_name, end_ms] { + deferredCleanup(nodes, state, timeline, scheduler, transition_generation, + pvw_is_slot_a, scene_name, end_ms); + }); +} + +} // namespace avp::mixer diff --git a/src/mixer/orchestrator/internal.hpp b/src/mixer/orchestrator/internal.hpp new file mode 100644 index 00000000..5f1d0d08 --- /dev/null +++ b/src/mixer/orchestrator/internal.hpp @@ -0,0 +1,21 @@ +#pragma once +// Shared includes and name imports for the MixerOrchestrator translation units +// (src/mixer/orchestrator/*.cpp). Not part of the public interface. +#include "MixerOrchestrator.hpp" +#include "../graph_ops.hpp" +#include "../primitives/OutputSnapshot.hpp" +#include "../routing.hpp" +#include "../primitives/TransitionGuard.hpp" +#include "../../avutils.hpp" +#include "../../graph_interfaces.hpp" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace avp::mixer { using namespace avp::mixer::graph; } diff --git a/src/mixer/orchestrator/overlay.cpp b/src/mixer/orchestrator/overlay.cpp new file mode 100644 index 00000000..b0729454 --- /dev/null +++ b/src/mixer/orchestrator/overlay.cpp @@ -0,0 +1,113 @@ +// Overlay branch enable/disable with a monotonic-PTS handover on the selector. +#include "internal.hpp" + +namespace avp::mixer { + +namespace { +constexpr int kOverlayDirectInput = 0; +constexpr int kOverlayCompositedInput = 1; +} + +void MixerOrchestrator::setOverlayEnabled(bool enabled, int64_t ready_timeout_ms) { + std::string source_otm_name; + std::string overlay_otm_name; + std::string selector_name; + std::string candidate_edge_name; + std::string selector_output_edge_name; + av::Timestamp visible_ts = NOTS; + av::Timestamp initial_candidate_ts = NOTS; + uint64_t generation = 0; + int64_t timeout_ms = 0; + int64_t poll_ms = 0; + const int candidate_input = enabled ? kOverlayCompositedInput : kOverlayDirectInput; + + { + std::lock_guard lock(state_->mutex); + if (state_->overlay_source_otm_name.empty() || + state_->overlay_otm_name.empty() || + state_->overlay_selector_name.empty()) { + throw Error("mixer.overlay: overlay nodes are not configured"); + } + source_otm_name = state_->overlay_source_otm_name; + overlay_otm_name = state_->overlay_otm_name; + selector_name = state_->overlay_selector_name; + timeout_ms = ready_timeout_ms >= 0 ? ready_timeout_ms : state_->overlay_ready_timeout_ms; + poll_ms = state_->overlay_ready_poll_ms; + generation = ++state_->overlay_generation; + + selector_output_edge_name = firstDstEdgeName(nodes_, selector_name); + candidate_edge_name = edgeNameAt(nodes_, selector_name, "src", candidate_input); + visible_ts = edgeLastTsIfExists(nodes_, selector_output_edge_name); + initial_candidate_ts = edgeLastTsIfExists(nodes_, candidate_edge_name); + + if (candidate_edge_name.empty()) { + throw Error("mixer.overlay: selector " + selector_name + " does not expose input " + + std::to_string(candidate_input)); + } + + if (!setNodeObjectIfCreated(nodes_, selector_name, "drop_non_monotonic", Parameters(true))) { + logstream << "mixer.overlay: queued " << selector_name + << ".drop_non_monotonic for node not created yet"; + } + + if (enabled) { + publishRuntimeObject(selector_name, "active", Parameters(kOverlayDirectInput)); + publishRuntimeObject(source_otm_name, "outputs", Parameters(1u)); + publishRuntimeObject(overlay_otm_name, "outputs", Parameters(3u)); + } else { + // Keep both legs fed until the direct leg has caught up with the last + // visible frame. The selector flips only after the wait below. + publishRuntimeObject(overlay_otm_name, "outputs", Parameters(3u)); + } + + logstream << "mixer.overlay: armed " << (enabled ? "enable" : "disable") + << " selector=" << selector_name + << " candidate_edge=" << candidate_edge_name + << " visible_ts=" << visible_ts + << " initial_candidate_ts=" << initial_candidate_ts + << " timeout_ms=" << timeout_ms; + } + + OverlayReadyResult ready = waitForOverlayBranchReady( + nodes_, state_, generation, candidate_edge_name, initial_candidate_ts, + visible_ts, timeout_ms, poll_ms); + if (ready.cancelled) { + logstream << "mixer.overlay: " << (enabled ? "enable" : "disable") + << " superseded before visible switch"; + return; + } + + { + std::lock_guard lock(state_->mutex); + if (!overlayCommandCurrent(state_, generation)) { + logstream << "mixer.overlay: " << (enabled ? "enable" : "disable") + << " superseded before finalizing"; + return; + } + + if (!ready.ready) { + publishRuntimeObject(selector_name, "active", Parameters(kOverlayDirectInput)); + publishRuntimeObject(overlay_otm_name, "outputs", Parameters(1u)); + publishRuntimeObject(source_otm_name, "outputs", Parameters(0u)); + state_->overlay_enabled = false; + std::ostringstream msg; + msg << "mixer.overlay: " << (enabled ? "overlay" : "direct") + << " branch did not reach monotonic PTS before timeout; edge=" + << candidate_edge_name << " last_ts=" << ready.ready_ts; + throw Error(msg.str()); + } + + publishRuntimeObject(selector_name, "active", Parameters(candidate_input)); + if (!enabled) { + publishRuntimeObject(overlay_otm_name, "outputs", Parameters(1u)); + publishRuntimeObject(source_otm_name, "outputs", Parameters(0u)); + } + state_->overlay_enabled = enabled; + logstream << "mixer.overlay: " << (enabled ? "enabled" : "disabled") + << " waited_ms=" << ready.waited_ms + << " ready_ts=" << ready.ready_ts + << " visible_ts=" << visible_ts; + } +} + +} // namespace avp::mixer diff --git a/src/mixer/orchestrator/scene.cpp b/src/mixer/orchestrator/scene.cpp new file mode 100644 index 00000000..ab4b66a6 --- /dev/null +++ b/src/mixer/orchestrator/scene.cpp @@ -0,0 +1,325 @@ +// Sources, scenes and slot loading: what a scene means in node parameters and +// how the PVW slot is prepared before a transition. +#include "internal.hpp" + +namespace avp::mixer { + +void MixerOrchestrator::defineSource(const std::string& name, const std::string& otm_node, int input_index, + const std::string& cs_node_a, const std::string& cs_node_b) { + std::lock_guard lock(state_->mutex); + MixerState::SourceInfo info; + info.otm_node_name = otm_node; + info.input_index = input_index; + info.cs_node_a = cs_node_a; + info.cs_node_b = cs_node_b; + state_->sources[name] = std::move(info); +} + +void MixerOrchestrator::defineRoutedSource(const std::string& name, const std::string& router_node, + int input_index, + const std::string& route_output_label_a, + const std::string& route_output_label_b, + const std::string& cs_node_a, const std::string& cs_node_b) { + const int output_count = routerOutputCount(nodes_, router_node); + const int route_output_a = routerOutputIndexFromLabel(nodes_, router_node, route_output_label_a); + const int route_output_b = routerOutputIndexFromLabel(nodes_, router_node, route_output_label_b); + + std::lock_guard lock(state_->mutex); + MixerState::SourceInfo info; + info.input_index = input_index; + info.cs_node_a = cs_node_a; + info.cs_node_b = cs_node_b; + info.routed = true; + info.router_node_name = router_node; + info.route_output_label_a = route_output_label_a; + info.route_output_label_b = route_output_label_b; + info.route_output_a = route_output_a; + info.route_output_b = route_output_b; + state_->sources[name] = std::move(info); + + int& stored_output_count = state_->router_output_counts[router_node]; + if (stored_output_count != 0 && stored_output_count != output_count) { + throw Error("mixer.routed_source: router " + router_node + " output count changed from " + + std::to_string(stored_output_count) + " to " + + std::to_string(output_count)); + } + stored_output_count = output_count; + ensureRouteTableSize(*state_, router_node); +} + +void MixerOrchestrator::defineScene(const std::string& name, const SceneDefinition& def) { + std::lock_guard lock(state_->mutex); + if (state_->prewarm_cut_scenes.count(name) && + (!canPrewarmScene(def) || (state_->computeActiveInputsMask(def) & ~state_->prewarm_source_mask))) + state_->prewarm_cut_scenes.erase(name); // Edited source identity takes the ordinary cold path. + state_->scenes[name] = def; +} + +bool MixerOrchestrator::canPrewarmScene(const SceneDefinition& scene) const { + if (!scene.routes.empty() || !scene.controls.empty()) return false; + for (const auto& [name, layout] : scene.sources) { + const auto source = state_->sources.find(name); + if (source == state_->sources.end() || source->second.routed || + !source->second.cs_node_a.empty() || !source->second.cs_node_b.empty() || + !layout.crop_scale_graph.empty()) return false; + } + return !scene.sources.empty(); +} + +void MixerOrchestrator::prewarmCuts(const std::vector& scenes) { + std::lock_guard lock(state_->mutex); + ensureIdle(); + uint32_t mask = 0; + for (const auto& name : scenes) { + const auto scene = state_->scenes.find(name); + if (scene == state_->scenes.end() || !canPrewarmScene(scene->second)) + throw Error("mixer.prewarm: scenes require fixed, filter-free sources without routes or controls: " + name); + mask |= state_->computeActiveInputsMask(scene->second); + } + for (const auto& slot : {state_->slot_a, state_->slot_b}) { + const auto node = nodes_->node(slot.compositor_name); + if (!node->node() || !node->parameters().contains("fps")) + throw Error("mixer.prewarm: requires created clocked compositors"); + } + for (const auto& slot : {state_->slot_a, state_->slot_b}) + setNodeObject(slot.compositor_name, "prewarm_inputs", Parameters(mask)); + state_->prewarm_cut_scenes = {scenes.begin(), scenes.end()}; + state_->prewarm_source_mask = mask; + for (const auto& [name, source] : state_->sources) { + if (source.routed) continue; + uint32_t outputs = state_->scenes.at(state_->pgm_scene_name).sources.count(name) ? state_->pgmOutputBit() : 0u; + if (!state_->pvw_scene_name.empty() && state_->scenes.at(state_->pvw_scene_name).sources.count(name)) + outputs |= state_->pvwOutputBit(); + publishCameraOtmOutputs(source.otm_node_name, state_->sourceOutputMask(source, outputs)); + } +} + +void MixerOrchestrator::applyRoutedSceneRoutesForSlot(bool is_slot_a, const SceneDefinition& scene, + int64_t at_pts_ms, bool immediate) { + auto tables = currentRouterTables(*state_); + setRoutedSlotInTables(*state_, tables, is_slot_a, &scene); + + for (const auto& [router_name, routes] : tables) { + Parameters value = routesToParameters(routes); + if (immediate) { + timeline_->clearKey(router_name, "routes"); + if (!setNodeObjectIfCreated(nodes_, router_name, "routes", value)) { + logstream << "mixer: queued " << router_name + << ".routes for router node not created yet"; + } + state_->router_routes[router_name] = routes; + } + timeline_->set(router_name, "routes", at_pts_ms, value); + } +} + +void MixerOrchestrator::publishRoutedRoutesForProgramOnly(bool pgm_is_slot_a, + const SceneDefinition& scene, + int64_t at_pts_ms, + bool immediate) { + auto tables = currentRouterTables(*state_); + setRoutedSlotInTables(*state_, tables, true, nullptr); + setRoutedSlotInTables(*state_, tables, false, nullptr); + setRoutedSlotInTables(*state_, tables, pgm_is_slot_a, &scene); + + for (const auto& [router_name, routes] : tables) { + Parameters value = routesToParameters(routes); + if (immediate) { + timeline_->clearKey(router_name, "routes"); + if (!setNodeObjectIfCreated(nodes_, router_name, "routes", value)) { + logstream << "mixer: queued " << router_name + << ".routes for router node not created yet"; + } + state_->router_routes[router_name] = routes; + } + timeline_->set(router_name, "routes", at_pts_ms, value); + } +} + +void MixerOrchestrator::initializeRoutedRoutes() { + std::lock_guard lock(state_->mutex); + for (const auto& [router_name, expected_count] : state_->router_output_counts) { + const int actual_count = routerOutputCount(nodes_, router_name); + if (expected_count != actual_count) { + throw Error("mixer.init_routes: router " + router_name + " expected output count " + + std::to_string(expected_count) + " but node dst has " + + std::to_string(actual_count)); + } + ensureRouteTableSize(*state_, router_name); + } + if (state_->pgm_scene_name.empty() || !state_->scenes.count(state_->pgm_scene_name)) + return; + publishRoutedRoutesForProgramOnly( + state_->pgm_is_slot_a, + state_->scenes.at(state_->pgm_scene_name), + wallclock.pts(), + true); +} + +void MixerOrchestrator::applyPostTransitionRouting(bool new_pgm_is_slot_a, + const std::string& new_pgm_scene) { + const auto scene_it = state_->scenes.find(new_pgm_scene); + if (scene_it == state_->scenes.end()) + return; + + const SceneDefinition& scene = scene_it->second; + const uint32_t pgm_bit = new_pgm_is_slot_a ? 1u : 2u; + const uint32_t active = state_->computeActiveInputsMask(scene); + const auto& new_slot = new_pgm_is_slot_a ? state_->slot_a : state_->slot_b; + const auto& old_slot = new_pgm_is_slot_a ? state_->slot_b : state_->slot_a; + + // Source_switcher first: this is the only setting visible at the SDI output. + // Any short window between this and the OTM/compositor flips below would only + // surface if the new direct path were not already producing frames; in both + // callers (ready cut and deferred fade cleanup) it is. + timeline_->clearKey(state_->source_switcher_name, "active"); + setNodeObject(state_->source_switcher_name, "active", + Parameters(new_pgm_is_slot_a ? 0 : 1)); + + // The encoder must not make the receiver wait for the next periodic keyframe: + // a cut changes the whole picture, and a P-frame carrying it can exceed what + // the receiver can recover from. The node coalesces bursts into one keyframe. + if (!state_->keyframe_node_name.empty()) { + try { + setNodeObject(state_->keyframe_node_name, "trigger", Parameters(true)); + } catch (const std::exception& e) { + logstream << "mixer: keyframe trigger failed: " << e.what(); + } + } + + for (const auto& [src_name, info] : state_->sources) { + if (info.routed) + continue; + const bool in_scene = scene.sources.count(src_name) > 0; + const bool active_input = (active & (1u << (unsigned)info.input_index)) != 0; + const uint32_t mask = state_->sourceOutputMask(info, (in_scene && active_input) ? pgm_bit : 0u); + timeline_->clearKey(info.otm_node_name, "outputs"); + setNodeObjectIfCreated(nodes_, info.otm_node_name, "outputs", Parameters(mask)); + } + publishRoutedRoutesForProgramOnly(new_pgm_is_slot_a, scene, wallclock.pts(), true); + + timeline_->clearKey(new_slot.post_otm_name, "outputs"); + timeline_->clearKey(old_slot.post_otm_name, "outputs"); + timeline_->clearKey(new_slot.compositor_name, "active_inputs"); + timeline_->clearKey(old_slot.compositor_name, "active_inputs"); + nodes_->node(new_slot.post_otm_name)->setObject("outputs", Parameters(1u)); + nodes_->node(old_slot.post_otm_name)->setObject("outputs", Parameters(0u)); + nodes_->node(new_slot.compositor_name)->setObject("active_inputs", Parameters(active)); + nodes_->node(old_slot.compositor_name)->setObject("active_inputs", Parameters(0u)); +} + +void MixerOrchestrator::rewriteCameraOutputsForSlot(uint32_t slot_bit, const SceneDefinition& scene) { + uint32_t active = state_->computeActiveInputsMask(scene); + for (const auto& [src_name, info] : state_->sources) { + if (info.routed) + continue; + Parameters current_val; + uint32_t mask = nodes_->node(info.otm_node_name)->getObjectTry("outputs", current_val) + ? current_val.get() + : 0u; + mask &= ~slot_bit; + if (scene.sources.count(src_name) && (active & (1u << (unsigned)info.input_index))) + mask |= slot_bit; + publishCameraOtmOutputs(info.otm_node_name, state_->sourceOutputMask(info, mask)); + } +} + +void MixerOrchestrator::loadSceneIntoSlot(bool is_slot_a, const std::string& scene_name, bool warm_cut) { + auto& scene = state_->scenes.at(scene_name); + const auto& slot = is_slot_a ? state_->slot_a : state_->slot_b; + + // The slot being loaded is the broadcast-inactive PVW slot. Its compositor + // was previously idled with active_inputs=0, so it may still hold frames on + // its input edges. If left there, the next activation starts by rendering + // stale frames and appears to lag behind the scene switch. + flushSlotEdges(is_slot_a); + // Reset on the compositor's worker thread and reject frames from before + // this load, including those still travelling through live upstream edges. + if (warm_cut && state_->prewarm_cut_scenes.count(scene_name) && canPrewarmScene(scene)) + setNodeObject(slot.compositor_name, "warm_reset", Parameters(true)); + else + resetInputIf(nodes_, slot.compositor_name); + + for (const auto& [src_name, layout] : scene.sources) { + auto src_it = state_->sources.find(src_name); + if (src_it == state_->sources.end()) continue; + const auto& info = src_it->second; + const std::string& cs_node = is_slot_a ? info.cs_node_a : info.cs_node_b; + + if (cs_node.empty()) { + if (!layout.crop_scale_graph.empty()) + throw Error("mixer: source " + src_name + " has no filter node for its scene graph"); + continue; + } + + // Only restart the crop/scale node when the graph string actually changed. + // Restarting a filter_video node tears down and rebuilds its FFmpeg filter + // graph, which briefly stops producing frames and allocates a new + // hw_frames_ctx pool. Downstream filter_video nodes now absorb pool + // rotations via a semantic hw_frames_ctx comparison so this no longer + // causes a mid-wipe EXT_NULL gap, but the restart is still a wasted + // stall and a frame-timing hiccup when the graph string is unchanged. + const auto& node_params = nodes_->node(cs_node)->parameters(); + const std::string old_graph = node_params.value("graph", std::string("")); + if (old_graph == layout.crop_scale_graph) { + logstream << "mixer: " << cs_node << " graph unchanged, no restart"; + } else { + logstream << "mixer: " << cs_node << " graph changed (\"" << old_graph << "\" -> \"" + << layout.crop_scale_graph << "\"), restarting"; + setNodeParam(cs_node, "graph", layout.crop_scale_graph); + autoRestartNode(cs_node); + } + } + + setNodeObject(slot.compositor_name, "layers", compositorLayersFromScene(*state_, scene)); + + uint32_t active_mask = state_->computeActiveInputsMask(scene); + // Same pattern as camera otms: cuda_rect_overlay reads "active_inputs" from timeline only. + // clearKey does not touch "layers" or other keys on this compositor channel. + timeline_->clearKey(slot.compositor_name, "active_inputs"); + setNodeObject(slot.compositor_name, "active_inputs", Parameters(active_mask)); + timeline_->set(slot.compositor_name, "active_inputs", wallclock.pts(), Parameters(active_mask)); + + // Drop slot bit for every camera, then enable only sources in scene with active_inputs set. + // Keeps `outputs` consistent with compositor consumption (no frames into unused inputs). + const uint32_t slot_bit = is_slot_a ? 1u : 2u; + rewriteCameraOutputsForSlot(slot_bit, scene); + applyRoutedSceneRoutesForSlot(is_slot_a, scene, wallclock.pts(), true); + + state_->pvw_scene_name = scene_name; +} + +void MixerOrchestrator::scheduleSceneControls(const SceneDefinition& scene, int64_t at_pts_ms) { + for (const auto& control : scene.controls) { + timeline_->set(control.node_name, control.key, at_pts_ms, control.value); + logstream << "mixer scene control: " << control.node_name << "." << control.key + << " at " << at_pts_ms << " -> " << control.value; + } +} + +void MixerOrchestrator::preview(const std::string& scene_name) { + std::lock_guard lock(state_->mutex); + ensureIdle(); + if (!state_->scenes.count(scene_name)) + throw Error("mixer.preview: unknown scene: " + scene_name); + + bool pvw_is_slot_a = !state_->pgm_is_slot_a; + const auto& slot = state_->pvwSlot(); + + if (state_->pvw_scene_name == scene_name) { + logstream << "mixer preview: scene already loaded in PVW: " << scene_name; + } else { + loadSceneIntoSlot(pvw_is_slot_a, scene_name); + resetInputIf(nodes_, slot.norm_ts_name); + } + + int64_t prep_ms = wallclock.pts(); + timeline_->clearKey(slot.post_otm_name, "outputs"); + setNodeObject(slot.post_otm_name, "outputs", Parameters(1u)); + timeline_->set(slot.post_otm_name, "outputs", prep_ms, Parameters(1u)); + logstream << "mixer preview armed: scene=" << scene_name + << " slot=" << (pvw_is_slot_a ? 'A' : 'B') + << " post_otm " << slot.post_otm_name << "->1"; +} + +} // namespace avp::mixer diff --git a/src/mixer/orchestrator/wipe.cpp b/src/mixer/orchestrator/wipe.cpp new file mode 100644 index 00000000..92024e66 --- /dev/null +++ b/src/mixer/orchestrator/wipe.cpp @@ -0,0 +1,373 @@ +// Media wipe: pre-created wipe subgraph, overlay readiness, midpoint scene +// switch hidden behind the opaque wipe, tail drain and teardown. +#include "internal.hpp" + +namespace avp::mixer { + +namespace { +constexpr int64_t kWipeSwitchGraceMs = 500; +} + +// --------------------------------------------------------------------------- +// runWipeMidpointAndCleanup: +// Phase 1 (midpoint): PVW slot prep + timeline source_switcher (hidden under opaque wipe). +// Phase 2 (end): routing cleanup, tear down wipe chain, flip state. +// +// Generation checks prevent an interrupted wipe from changing new routing. +// --------------------------------------------------------------------------- +void MixerOrchestrator::runWipeMidpointAndCleanup( + std::shared_ptr nodes, + std::shared_ptr state, + std::shared_ptr timeline, + std::shared_ptr scheduler, + uint64_t transition_generation, + std::string scene_name, + bool new_pgm_is_slot_a, + int64_t remaining_ms) { + + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return; + + // --- Phase 1: midpoint - do invisible scene switch under the fully-opaque wipe --- + try { + std::lock_guard lock(state->mutex); + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return; + MixerOrchestrator orch(nodes, state, timeline, scheduler); + + // Keep a prewarmed scene intact at the wipe midpoint. + if (state->pvw_scene_name != scene_name) + orch.loadSceneIntoSlot(new_pgm_is_slot_a, scene_name); + + // Same post-scene OTM flip as cutInternal: out_sel will read PVW `sc*_direct`, so that slot's + // `one_to_many` must have outputs=1. If it stays 0 (idle default), frames are popped from + // norm_* with nowhere to go and the wipe path freezes. Stop the old PGM branch to avoid backup. + int64_t Tw = wallclock.pts(); + const auto& new_slot = state->pvwSlot(); + const auto& old_slot = state->pgmSlot(); + timeline->clearKey(old_slot.post_otm_name, "outputs"); + orch.setNodeObject(old_slot.post_otm_name, "outputs", Parameters(0u)); + timeline->set(old_slot.post_otm_name, "outputs", Tw, Parameters(0u)); + timeline->clearKey(new_slot.post_otm_name, "outputs"); + orch.setNodeObject(new_slot.post_otm_name, "outputs", Parameters(1u)); + timeline->set(new_slot.post_otm_name, "outputs", Tw, Parameters(1u)); + + // Switch source_switcher (invisible behind wipe overlay); timeline for consistency with other switches + int sw = state->pvwSourceSwitcherIndex(); + timeline->set(state->source_switcher_name, "active", Tw, Parameters(sw)); + logstream << "mixer wipe midpoint: Tw=" << Tw << " scene=" << scene_name << " out_sel.active=" << sw + << " new_slot post_otm=" << new_slot.post_otm_name << " old_slot post_otm=" + << old_slot.post_otm_name; + resetSlotNormFps(nodes, *state); + } catch (const std::exception& e) { + logstream << "mixer: wipe midpoint error: " << e.what(); + } + + // --- Phase 2: wipe end – tear down and flip --- + // Stage 2a: wait for the wipe source to EOF (or until the planned duration elapses). + bool hit_input_eof = false; + if (remaining_ms > 0) { + int64_t waited_ms = 0; + constexpr int64_t kTailPollMs = 10; + while (waited_ms < remaining_ms) { + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return; + if (!nodeWorkingIfExists(nodes, state->wipe_input_node_name)) { + logstream << "mixer wipe: cleanup pulled to wipe EOF after " << waited_ms + << "ms of remaining tail"; + hit_input_eof = true; + break; + } + int64_t step_ms = std::min(kTailPollMs, remaining_ms - waited_ms); + std::this_thread::sleep_for(std::chrono::milliseconds(step_ms)); + waited_ms += step_ms; + } + } + + // Stage 2b: after source EOF, the tail of the wipe is still propagating through + // wipe_demux → wipe_dec → wipe_fmt → wipe_rt → wipe_rt_fps → wipe_overlay (six + // queues + filter internal buffering). Flipping `wipe_sel` now would cut those + // tail frames. Wait until the last pre-overlay edge (`wipe_tail_edge`) has been + // drained by the overlay, then give the overlay a short grace to emit the final + // blended frames through `wipe_overlay_out` to `wipe_sel`. + if (hit_input_eof && !state->wipe_tail_edge.empty()) { + constexpr int64_t kWipeDrainTimeoutMs = 1000; + constexpr int64_t kWipeDrainPollMs = 10; + // Grace period for the overlay filter to emit any frames already buffered + // in its filter graph after `wipe_tail_edge` drained. 120ms is ~3-4 frames + // at 30fps and ~7-8 frames at 60fps; both are within the typical libavfilter + // internal queue depth. If a future wipe overlay graph buffers more (e.g. + // a multi-stage temporal filter), bump this together with kWipeDrainTimeoutMs. + // Going below ~80ms risks cutting tail blended frames at 30fps. + constexpr int64_t kWipeOverlayTailMs = 120; + int64_t waited = 0; + while (waited < kWipeDrainTimeoutMs) { + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return; + if (edgeOccupiedIfExists(nodes, state->wipe_tail_edge) == 0) + break; + std::this_thread::sleep_for(std::chrono::milliseconds(kWipeDrainPollMs)); + waited += kWipeDrainPollMs; + } + for (int64_t tail = 0; tail < kWipeOverlayTailMs; tail += 5) { + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) return; + std::this_thread::sleep_for(std::chrono::milliseconds(5)); + } + logstream << "mixer wipe: drained " << state->wipe_tail_edge << " in " << waited + << "ms + " << kWipeOverlayTailMs << "ms overlay tail"; + } + + try { + std::lock_guard lock(state->mutex); + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return; + MixerOrchestrator orch(nodes, state, timeline, scheduler); + + int64_t Tw = wallclock.pts(); + if (!state->wipe_otm_name.empty()) { + timeline->clearKey(state->wipe_otm_name, "outputs"); + orch.setNodeObject(state->wipe_otm_name, "outputs", Parameters(1u)); + timeline->set(state->wipe_otm_name, "outputs", Tw, Parameters(1u)); + } + if (!state->wipe_selector_name.empty()) { + timeline->clearKey(state->wipe_selector_name, "active"); + orch.setNodeObject(state->wipe_selector_name, "active", Parameters(0)); + timeline->set(state->wipe_selector_name, "active", Tw, Parameters(0)); + } + logstream << "mixer wipe cleanup: Tw=" << Tw << " otm_final.outputs=1 wipe_sel.active=0" + << " stop_wipe_group_in_ms=" << kWipeSwitchGraceMs; + + uint32_t new_pgm_bit = state->pvwOutputBit(); + auto& scene = state->scenes.at(scene_name); + for (const auto& [src_name, info] : state->sources) { + if (info.routed) + continue; + uint32_t mask = scene.sources.count(src_name) ? new_pgm_bit : 0u; + timeline->set(info.otm_node_name, "outputs", Tw, Parameters(state->sourceOutputMask(info, mask))); + } + orch.publishRoutedRoutesForProgramOnly(new_pgm_is_slot_a, scene, Tw, true); + + const auto& old_slot = state->pgmSlot(); + timeline->set(old_slot.compositor_name, "active_inputs", Tw, Parameters(0u)); + } catch (const std::exception& e) { + logstream << "mixer: wipe cleanup error: " << e.what(); + if (transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + state->transition_mode = MixerState::TransitionMode::Idle; + return; + } + + for (int64_t waited = 0; waited < kWipeSwitchGraceMs; waited += 5) { + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) return; + std::this_thread::sleep_for(std::chrono::milliseconds(5)); + } + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return; + + try { + std::lock_guard lock(state->mutex); + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return; + MixerOrchestrator orch(nodes, state, timeline, scheduler); + + // Stop the pre-created wipe subgraph only after the direct-path switch + // has had time to land on the frame timeline. Tearing it down at the + // same wallclock instant as the switch starves wipe_sel/final_out for a + // few ticks and the encoder-side force_fps visibly repeats the last wipe frame. + if (!state->wipe_group_name.empty()) { + orch.stopGroup(state->wipe_group_name); + // Release any frames still sitting in wipe pipeline edges so they + // don't replay at the start of the next wipe. + orch.flushWipeEdges(); + } + + orch.finishSnapshot(); + state->pgm_is_slot_a = new_pgm_is_slot_a; + state->pgm_scene_name = scene_name; + state->pvw_scene_name = ""; + state->transition_mode = MixerState::TransitionMode::Idle; + } catch (const std::exception& e) { + logstream << "mixer: wipe teardown error: " << e.what(); + if (transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + state->transition_mode = MixerState::TransitionMode::Idle; + } +} + +// --------------------------------------------------------------------------- +// wipe: media wipe transition. Uses the static wipe_otm + wipe_selector +// nodes to route through the overlay without edge rewiring. +// The wipe subgraph (group wipe_group_name) is pre-created but not running +// in steady state; it is started here and stopped at the end of the wipe. +// --------------------------------------------------------------------------- +int64_t MixerOrchestrator::prepareWipe( + std::shared_ptr nodes, + std::shared_ptr state, + std::shared_ptr timeline, + std::shared_ptr scheduler, + uint64_t transition_generation, + std::string scene_name, + std::string wipe_file, + double duration_sec, + bool new_pgm_is_slot_a, + int64_t earliest_visible_pts_ms) { + std::string overlay_edge_name; + av::Timestamp overlay_initial_ts = NOTS; + int64_t prep_ms = wallclock.pts(); + + // Only the serialized transition worker reuses wipe nodes. Finish retiring + // the previous clip outside the control mutex, then recheck cancellation. + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) return -1; + try { + nodes->group(state->wipe_group_name)->stopNodesAndWait(); + std::lock_guard lock(state->mutex); + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return -1; + MixerOrchestrator orch(nodes, state, timeline, scheduler); + + overlay_edge_name = edgeNameAt(nodes, state->wipe_selector_name, "src", 1); + overlay_initial_ts = edgeLastTsIfExists(nodes, overlay_edge_name); + + nodes->node(state->wipe_input_node_name)->stop(true); + orch.setNodeParam(state->wipe_input_node_name, "url", wipe_file); + orch.flushWipeEdges(); + resetInputIf(nodes, state->wipe_base_fps_name); + orch.startGroup(state->wipe_group_name); + + prep_ms = wallclock.pts(); + timeline->clearKey(state->wipe_otm_name, "outputs"); + orch.setNodeObject(state->wipe_otm_name, "outputs", Parameters(3u)); // 0b11 both direct + wipe_in + timeline->set(state->wipe_otm_name, "outputs", prep_ms, Parameters(3u)); + timeline->clearKey(state->wipe_selector_name, "active"); + orch.setNodeObject(state->wipe_selector_name, "active", Parameters(0)); // direct branch while prerolling + timeline->set(state->wipe_selector_name, "active", prep_ms, Parameters(0)); + + int64_t total_ms = (int64_t)(duration_sec * 1000); + int64_t midpoint_ms = total_ms / 2; + logstream << "mixer wipe: scene=" << scene_name << " file=" << wipe_file << " prep_ms=" << prep_ms + << " requested_visible=" << earliest_visible_pts_ms << " total_ms=" << total_ms << " midpoint_ms=" << midpoint_ms + << " new_pgm_slot_" << (new_pgm_is_slot_a ? 'A' : 'B'); + } catch (const std::exception& e) { + logstream << "mixer: wipe prep error: " << e.what(); + std::lock_guard lock(state->mutex); + MixerOrchestrator(nodes, state, timeline, scheduler).abortTransition(transition_generation); + return -1; + } + + WipeReadyResult ready = waitForWipeOverlayReady( + nodes, overlay_edge_name, overlay_initial_ts, earliest_visible_pts_ms, state, transition_generation); + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) return -1; + if (!ready.ready) { + logstream << "mixer: wipe overlay did not become ready within " << kWipeReadyTimeoutMs + << "ms: edge=" << overlay_edge_name << " last_ts=" << ready.ready_ts; + std::lock_guard lock(state->mutex); + MixerOrchestrator(nodes, state, timeline, scheduler).abortTransition(transition_generation); + return -1; + } + + try { + std::lock_guard lock(state->mutex); + if (!transitionIsCurrent(state, transition_generation, MixerState::TransitionMode::Wipe)) + return -1; + MixerOrchestrator orch(nodes, state, timeline, scheduler); + int64_t visible_ms = wallclock.pts(); + orch.setNodeObject(state->wipe_selector_name, "active", Parameters(1)); // wipe_overlay_out + timeline->set(state->wipe_selector_name, "active", visible_ms, Parameters(1)); + logstream << "mixer wipe overlay ready: edge=" << overlay_edge_name + << " waited_ms=" << ready.waited_ms + << " ready_ts=" << ready.ready_ts + << " visible_ms=" << visible_ms + << " requested_visible=" << earliest_visible_pts_ms; + return visible_ms; + } catch (const std::exception& e) { + logstream << "mixer: wipe visible switch error: " << e.what(); + std::lock_guard lock(state->mutex); + MixerOrchestrator(nodes, state, timeline, scheduler).abortTransition(transition_generation); + return -1; + } +} + +void MixerOrchestrator::warmupWipe(const std::string& wipe_file, int64_t timeout_ms) { + std::string overlay_edge_name; + av::Timestamp overlay_initial_ts = NOTS; + uint64_t generation; + const int64_t t0 = wallclock.pts(); + { + std::lock_guard lock(state_->mutex); + if (state_->wipe_otm_name.empty() || state_->wipe_selector_name.empty() || + state_->wipe_group_name.empty() || state_->wipe_input_node_name.empty()) + throw Error("mixer: wipe warm-up requires the wipe subgraph (see mixer.init)"); + if (state_->transition_mode.load() != MixerState::TransitionMode::Idle) + throw Error("mixer: cannot warm up the wipe during a transition"); + generation = state_->transition_generation.load(); + overlay_edge_name = edgeNameAt(nodes_, state_->wipe_selector_name, "src", 1); + overlay_initial_ts = edgeLastTsIfExists(nodes_, overlay_edge_name); + nodes_->group(state_->wipe_group_name)->stopNodesAndWait(); + nodes_->node(state_->wipe_input_node_name)->stop(true); + setNodeParam(state_->wipe_input_node_name, "url", wipe_file); + flushWipeEdges(); + resetInputIf(nodes_, state_->wipe_base_fps_name); + startGroup(state_->wipe_group_name); + // Feed the overlay's program input as a real wipe would; the selector + // stays on the direct branch so nothing of this reaches the output. + timeline_->clearKey(state_->wipe_otm_name, "outputs"); + setNodeObject(state_->wipe_otm_name, "outputs", Parameters(3u)); + } + WipeReadyResult ready = waitForWipeOverlayReady(nodes_, overlay_edge_name, overlay_initial_ts, 0, + state_, generation, timeout_ms); + { + std::lock_guard lock(state_->mutex); + timeline_->clearKey(state_->wipe_otm_name, "outputs"); + setNodeObject(state_->wipe_otm_name, "outputs", Parameters(1u)); + stopGroup(state_->wipe_group_name); + flushWipeEdges(); + } + logstream << "mixer wipe warm-up: file=" << wipe_file << (ready.ready ? " ready" : " NOT ready") + << " after " << (wallclock.pts() - t0) << "ms (overlay waited " << ready.waited_ms << "ms)"; + if (!ready.ready) + throw Error("mixer: wipe warm-up did not produce an overlay frame within " + std::to_string(timeout_ms) + "ms"); +} + +void MixerOrchestrator::wipe(const std::string& scene_name, const std::string& wipe_file, double duration_sec, + int64_t start_pts_ms) { + std::lock_guard lock(state_->mutex); + if (!state_->scenes.count(scene_name)) + throw Error("mixer: unknown scene: " + scene_name); + + if (state_->wipe_otm_name.empty() || state_->wipe_selector_name.empty()) + throw Error("mixer: wipe requires wipe_otm and wipe_selector nodes (see mixer.init)"); + if (state_->wipe_group_name.empty() || state_->wipe_input_node_name.empty()) + throw Error("mixer: wipe requires wipe_group and wipe_input_node (see mixer.init)"); + + if (!std::isfinite(duration_sec) || duration_sec <= 0) throw Error("mixer: invalid wipe duration"); + int64_t start_ms = resolveTransitionStartPts(start_pts_ms); + interruptTransition(); + state_->transition_mode = MixerState::TransitionMode::Wipe; + uint64_t transition_generation = ++state_->transition_generation; + state_->transition_scene_name = scene_name; + TransitionGuard prep_guard([&] { abortTransition(transition_generation); }); + bool pvw_is_slot_a = !state_->pgm_is_slot_a; + cutInternal(scene_name, start_ms); + scheduleSceneControls(state_->scenes.at(scene_name), start_ms); + + int64_t total_ms = (int64_t)(duration_sec * 1000); + int64_t midpoint_ms = total_ms / 2; + int64_t remaining_ms = total_ms - midpoint_ms; + int64_t now_ms = wallclock.pts(); + postTransitionTask("mixer.wipe.prepare", start_ms - now_ms, + [scheduler = scheduler_, nodes = nodes_, state = state_, timeline = timeline_, transition_generation, + scene_name, wipe_file, duration_sec, pvw_is_slot_a, start_ms, midpoint_ms, remaining_ms] { + int64_t visible_ms = prepareWipe(nodes, state, timeline, scheduler, transition_generation, + scene_name, wipe_file, duration_sec, pvw_is_slot_a, + start_ms); + if (visible_ms < 0) + return; + + scheduler->postAfter("mixer.wipe.midpoint", midpoint_ms, + [nodes, state, timeline, scheduler, transition_generation, scene_name, pvw_is_slot_a, remaining_ms] { + runWipeMidpointAndCleanup(nodes, state, timeline, scheduler, transition_generation, + scene_name, pvw_is_slot_a, remaining_ms); + }); + }); + prep_guard.release(); +} + +} // namespace avp::mixer diff --git a/src/mixer/Cadence.hpp b/src/mixer/primitives/Cadence.hpp similarity index 96% rename from src/mixer/Cadence.hpp rename to src/mixer/primitives/Cadence.hpp index 9a9568d2..ae3ae716 100644 --- a/src/mixer/Cadence.hpp +++ b/src/mixer/primitives/Cadence.hpp @@ -1,6 +1,6 @@ #pragma once -#include "FrameRate.hpp" +#include "TickGrid.hpp" #include #include #include @@ -20,14 +20,14 @@ class Cadence { }; private: - FrameRate rate_; + TickGrid rate_; int64_t tolerance_; std::optional next_; int64_t observations_ = 0; std::deque phase_errors_; public: - Cadence(FrameRate rate, int64_t latency_ns) + Cadence(TickGrid rate, int64_t latency_ns) : rate_(rate), tolerance_(std::max(2, rate.nearestIndex(latency_ns) + 1)) {} void advance(int64_t slots) { diff --git a/src/mixer/CutLatency.hpp b/src/mixer/primitives/CutLatency.hpp similarity index 100% rename from src/mixer/CutLatency.hpp rename to src/mixer/primitives/CutLatency.hpp diff --git a/src/mixer/CutLatencyProbe.hpp b/src/mixer/primitives/CutLatencyProbe.hpp similarity index 82% rename from src/mixer/CutLatencyProbe.hpp rename to src/mixer/primitives/CutLatencyProbe.hpp index 1025e20d..d17cbcdb 100644 --- a/src/mixer/CutLatencyProbe.hpp +++ b/src/mixer/primitives/CutLatencyProbe.hpp @@ -1,13 +1,21 @@ #pragma once #include "CutLatency.hpp" -#include "../avutils.hpp" +#include "../../avutils.hpp" #include #include #include namespace avp::mixer { +/// A frame token stored as decimal text in an AVDictionary entry; 0 when absent or malformed. +inline uint64_t parseFrameToken(const char* value) { + uint64_t token = 0; + const char* end = value + std::char_traits::length(value); + const auto parsed = std::from_chars(value, end, token); + return parsed.ec == std::errc{} && parsed.ptr == end ? token : 0; +} + struct CutLatencyProbe { CutLatency timing; const std::string metadata_key; @@ -38,11 +46,7 @@ struct CutLatencyProbe { uint64_t frameToken(const av::VideoFrame& frame) const { if (!frame.raw()) return 0; auto entry = av_dict_get(frame.raw()->metadata, metadata_key.c_str(), nullptr, 0); - if (!entry) return 0; - uint64_t token = 0; - const char* end = entry->value + std::char_traits::length(entry->value); - const auto parsed = std::from_chars(entry->value, end, token); - return parsed.ec == std::errc{} && parsed.ptr == end ? token : 0; + return entry ? parseFrameToken(entry->value) : 0; } }; diff --git a/src/mixer/MixerState.hpp b/src/mixer/primitives/MixerState.hpp similarity index 95% rename from src/mixer/MixerState.hpp rename to src/mixer/primitives/MixerState.hpp index 68cb17ad..b50ce1aa 100644 --- a/src/mixer/MixerState.hpp +++ b/src/mixer/primitives/MixerState.hpp @@ -1,6 +1,6 @@ #pragma once -#include "../instance_shared.hpp" -#include "../util.hpp" +#include "../../instance_shared.hpp" +#include "../../util.hpp" #include #include #include @@ -10,6 +10,8 @@ #include #include "CutLatencyProbe.hpp" +namespace avp::mixer { + struct SourceLayout { std::string crop_scale_graph; // e.g., "crop=1920:1080:0:0,scale_cuda=640:360" /// Layer fields for cuda_rect_overlay (dst_x, dst_y, …) — not including `graph`. @@ -32,12 +34,6 @@ struct SceneDefinition { int width = 1920; int height = 1080; - std::unordered_set sourceNames() const { - std::unordered_set r; - for (const auto& [k, v] : sources) - r.insert(k); - return r; - } }; struct MixerState : public InstanceShared { @@ -141,3 +137,5 @@ struct MixerState : public InstanceShared { return mask; } }; + +} // namespace avp::mixer diff --git a/src/mixer/MonotonicClock.hpp b/src/mixer/primitives/MonotonicClock.hpp similarity index 100% rename from src/mixer/MonotonicClock.hpp rename to src/mixer/primitives/MonotonicClock.hpp diff --git a/src/mixer/OutputSnapshot.hpp b/src/mixer/primitives/OutputSnapshot.hpp similarity index 79% rename from src/mixer/OutputSnapshot.hpp rename to src/mixer/primitives/OutputSnapshot.hpp index 382cd4be..8a070d8e 100644 --- a/src/mixer/OutputSnapshot.hpp +++ b/src/mixer/primitives/OutputSnapshot.hpp @@ -1,8 +1,8 @@ #pragma once #include "Snapshot.hpp" -#include "../instance_shared.hpp" -#include "../avutils.hpp" +#include "../../instance_shared.hpp" +#include "../../avutils.hpp" #include namespace avp::mixer { diff --git a/src/mixer/Snapshot.hpp b/src/mixer/primitives/Snapshot.hpp similarity index 100% rename from src/mixer/Snapshot.hpp rename to src/mixer/primitives/Snapshot.hpp diff --git a/src/mixer/FrameRate.hpp b/src/mixer/primitives/TickGrid.hpp similarity index 69% rename from src/mixer/FrameRate.hpp rename to src/mixer/primitives/TickGrid.hpp index 20cd6656..4d48b0a1 100644 --- a/src/mixer/FrameRate.hpp +++ b/src/mixer/primitives/TickGrid.hpp @@ -1,12 +1,18 @@ #pragma once +#include +extern "C" { +#include +} #include #include namespace avp::mixer { -// Integer frame indices keep fractional rates on one common monotonic grid. -class FrameRate { +// The tick grid of an output rate: integer frame indices on one monotonic +// nanosecond grid, with the index <-> time rounding the playout relies on. +// Not a rational type; the rate itself is an av::Rational. +class TickGrid { int64_t numerator_; int64_t denominator_; @@ -16,8 +22,9 @@ class FrameRate { } public: - FrameRate(int64_t numerator, int64_t denominator) - : numerator_(numerator), denominator_(denominator) { + explicit TickGrid(av::Rational rate) + : numerator_(rate.getNumerator()), denominator_(rate.getDenominator()) { + const int64_t numerator = numerator_, denominator = denominator_; if (numerator <= 0 || denominator <= 0 || denominator > INT64_MAX / 1000000000 || numerator > denominator * 1000000000) @@ -25,8 +32,7 @@ class FrameRate { } int64_t time(int64_t index) const { - return floorDivide(static_cast<__int128>(index) * denominator_ * 1000000000, - numerator_); + return av_rescale_rnd(index, denominator_ * 1000000000, numerator_, AV_ROUND_DOWN); } int64_t nearestIndex(int64_t time_ns) const { diff --git a/src/mixer/primitives/TransitionGuard.hpp b/src/mixer/primitives/TransitionGuard.hpp new file mode 100644 index 00000000..eef3b452 --- /dev/null +++ b/src/mixer/primitives/TransitionGuard.hpp @@ -0,0 +1,34 @@ +#pragma once +// Generation/mode check for transition workers and the abort-on-exception guard +// used while a transition is being prepared under the control mutex. +#include "MixerState.hpp" +#include +#include + +namespace avp::mixer { + +class TransitionGuard { + std::function abort_; + bool active_ = true; + +public: + explicit TransitionGuard(std::function abort) + : abort_(std::move(abort)) {} + + ~TransitionGuard() { + if (active_) abort_(); + } + + void release() { + active_ = false; + } +}; + +inline bool transitionIsCurrent(const std::shared_ptr& state, + uint64_t generation, + MixerState::TransitionMode mode) { + return state->transition_generation.load(std::memory_order_acquire) == generation && + state->transition_mode.load(std::memory_order_acquire) == mode; +} + +} diff --git a/src/mixer/primitives/compositor_color.hpp b/src/mixer/primitives/compositor_color.hpp new file mode 100644 index 00000000..75438d65 --- /dev/null +++ b/src/mixer/primitives/compositor_color.hpp @@ -0,0 +1,17 @@ +#pragma once + +extern "C" { +#include +} + +namespace avp::mixer { + +// Untagged RGB(A) graphics use the compositor's SDR BT.709 interpretation. +// Resolve each missing tag independently; explicit unsupported tags still fail. +// Input frames may be shared, so this does not stamp or modify their metadata. +inline bool isSdrGraphicColor(const AVFrame &frame) { + return (frame.color_trc == AVCOL_TRC_UNSPECIFIED || frame.color_trc == AVCOL_TRC_BT709) && + (frame.color_primaries == AVCOL_PRI_UNSPECIFIED || frame.color_primaries == AVCOL_PRI_BT709); +} + +} diff --git a/src/hwaccel/CompositorGeometry.hpp b/src/mixer/primitives/compositor_geometry.hpp similarity index 91% rename from src/hwaccel/CompositorGeometry.hpp rename to src/mixer/primitives/compositor_geometry.hpp index 11234f14..ea786115 100644 --- a/src/hwaccel/CompositorGeometry.hpp +++ b/src/mixer/primitives/compositor_geometry.hpp @@ -3,7 +3,11 @@ #include #include -namespace avp::compositor { +extern "C" { +#include +} + +namespace avp::mixer { struct Rect { int x = 0, y = 0, w = 0, h = 0; }; struct Placement { Rect source, destination; }; @@ -26,9 +30,9 @@ inline std::optional place(int width, int height, Rect crop, Rect box int w = box.w, h = box.h; if (fit) { if (int64_t(crop.w) * h >= int64_t(crop.h) * w) - h = int(int64_t(crop.h) * w / crop.w); + h = int(av_rescale(crop.h, w, crop.w)); else - w = int(int64_t(crop.w) * h / crop.h); + w = int(av_rescale(crop.w, h, crop.h)); } w = aligned(w, align_x); h = aligned(h, align_y); if (w <= 0 || h <= 0) return {}; @@ -46,7 +50,7 @@ inline std::optional placeInCanvas(int width, int height, Rect crop, auto outer = place(canvas_w, canvas_h, {}, box, true, ax, ay); if (!inner || !outer) return {}; auto map = [](int value, int extent, int canvas, int alignment) { - int result = int(int64_t(value) * extent / canvas); + int result = int(av_rescale(value, extent, canvas)); return result - result % alignment; }; const auto &i = inner->destination, &o = outer->destination; diff --git a/src/mixer/primitives/compositor_layers.hpp b/src/mixer/primitives/compositor_layers.hpp new file mode 100644 index 00000000..f4be46fa --- /dev/null +++ b/src/mixer/primitives/compositor_layers.hpp @@ -0,0 +1,217 @@ +#pragma once +// Layer descriptions for the CUDA compositor: JSON parsing, per-frame metadata +// overrides, and the resolution of layers against the actual source frames into +// an ordered list of draw operations. No CUDA here. +#include "../../util.hpp" +#include "compositor_geometry.hpp" +#include "pixel_layout.hpp" +#include +#include +#include +#include +#include + +namespace avp::mixer { + +struct LayerSpec { + int dst_x = 0; + int dst_y = 0; + int crop_x = 0; + int crop_y = 0; + int dst_w = 0; + int dst_h = 0; + bool fit = false; + int source_canvas_w = 0; + int source_canvas_h = 0; + int crop_w= 0; + int crop_h = 0; + int z = 0; // draw order: lower first, ties by source index + bool blend = false; // honour the source's alpha instead of overwriting + + bool operator==(const LayerSpec &other) const { + return dst_x == other.dst_x && dst_y == other.dst_y && z == other.z && blend == other.blend && + crop_x == other.crop_x && crop_y == other.crop_y && + crop_w == other.crop_w && crop_h == other.crop_h && + dst_w == other.dst_w && dst_h == other.dst_h && fit == other.fit && + source_canvas_w == other.source_canvas_w && source_canvas_h == other.source_canvas_h; + } +}; + +struct DrawOp { + const av::VideoFrame *src = nullptr; + int src_w = 0; + int src_h = 0; + LayerSpec layer; + + bool operator==(const DrawOp &other) const { + return (src != nullptr) == (other.src != nullptr) && + src_w == other.src_w && src_h == other.src_h && + layer == other.layer; + } +}; + +inline void parseLayerFromJson(const Parameters &obj, LayerSpec &out) { + out.dst_x = obj.value("dst_x", 0); + out.dst_y = obj.value("dst_y", 0); + out.dst_w = obj.value("dst_w", 0); + out.dst_h = obj.value("dst_h", 0); + out.z = obj.value("z", 0); + out.blend = obj.value("blend", false); + const std::string fit = obj.value("fit", std::string("stretch")); + if (fit != "stretch" && fit != "contain") + throw Error("cuda_rect_overlay: fit must be stretch or contain"); + out.fit = fit == "contain"; + out.source_canvas_w = out.source_canvas_h = 0; + if (obj.contains("source_canvas")) { + const auto &canvas = obj.at("source_canvas"); + out.source_canvas_w = canvas.at("w").get(); + out.source_canvas_h = canvas.at("h").get(); + if (!out.fit || out.dst_w <= 0 || out.dst_h <= 0 || out.source_canvas_w <= 0 || out.source_canvas_h <= 0) + throw Error("cuda_rect_overlay: source_canvas requires positive dimensions and fit=contain"); + } + if (out.dst_w < 0 || out.dst_h < 0 || (out.dst_w == 0) != (out.dst_h == 0)) + throw Error("cuda_rect_overlay: dst_w and dst_h must both be positive or both omitted"); +#ifndef HAVE_CUDA_RECT_SCALE + if (out.dst_w || out.dst_h) + throw Error("cuda_rect_overlay: destination sizing requires HAVE_NVCC=1"); +#endif + if (obj.contains("crop") && obj["crop"].is_object()) { + const auto &c = obj["crop"]; + out.crop_x = c.value("x", 0); + out.crop_y = c.value("y", 0); + // 0 means "use remaining source width/height from crop_x/crop_y" (resolved per-frame in resolveDrawOps). + out.crop_w = c.value("w", 0); + out.crop_h = c.value("h", 0); + } + // No crop object → crop_x/y/w/h all stay 0 (full source frame from origin). +} + +inline std::vector parseLayersArray(const Parameters &arr) { + std::vector layers; + if (!arr.is_array()) + throw Error("cuda_rect_overlay: layers must be an array"); + for (const auto &item : arr) { + if (!item.is_object()) + throw Error("cuda_rect_overlay: layers entries must be objects"); + LayerSpec s; + parseLayerFromJson(item, s); + layers.push_back(s); + } + return layers; +} + +inline std::vector parseLayersParam(const Parameters ¶ms) { + if (!params.contains("layers") || !params["layers"].is_array()) + throw Error("cuda_rect_overlay: layers array required (one entry per input in src order)"); + return parseLayersArray(params["layers"]); +} + +/// Apply a per-frame metadata override (JSON text) to `layers`: either {"layers": [...]} in src +/// order or {"": {...}} per input. Throws on malformed JSON or entries. +inline void applyLayerMetadata(std::vector &layers, const char *json) { + Parameters md = Parameters::parse(json); + if (md.contains("layers") && md["layers"].is_array()) { + const auto &arr = md["layers"]; + for (size_t i = 0; i < arr.size() && i < layers.size(); ++i) { + if (!arr[i].is_object()) + continue; + parseLayerFromJson(arr[i], layers[i]); + } + } else { + for (size_t i = 0; i < layers.size(); ++i) { + const std::string k = std::to_string(i); + if (md.contains(k) && md[k].is_object()) + parseLayerFromJson(md[k], layers[i]); + } + } +} + +/// Resolve each layer against its source frame: crop/destination clipping, chroma alignment, +/// fit/contain placement, then z-order. Missing sources and empty rectangles yield ops with +/// src == nullptr; a placement rejected by geometry keeps a negative src_w so the log can name it. +inline std::vector resolveDrawOps(const std::vector &sources, + const std::vector &layers, + int canvas_w, int canvas_h, AVPixelFormat canvas_fmt) { + std::vector ops; + ops.reserve(std::min(sources.size(), layers.size())); + for (size_t i = 0; i < sources.size() && i < layers.size(); ++i) { + const av::VideoFrame *srcp = sources[i]; + if (!srcp || !srcp->raw()) { + ops.push_back({}); + continue; + } + LayerSpec L = layers[i]; + if (L.dst_w > 0) { + const Rect crop{L.crop_x, L.crop_y, L.crop_w, L.crop_h}; + const Rect box{L.dst_x, L.dst_y, L.dst_w, L.dst_h}; + const int ax = chromaXAlign(canvas_fmt), ay = chromaYAlign(canvas_fmt); + auto placement = L.source_canvas_w > 0 + ? placeInCanvas(srcp->width(), srcp->height(), crop, box, + L.source_canvas_w, L.source_canvas_h, ax, ay) + : place(srcp->width(), srcp->height(), crop, box, L.fit, ax, ay); + if (!placement || placement->destination.x >= canvas_w || + placement->destination.y >= canvas_h || + int64_t(placement->destination.x) + placement->destination.w <= 0 || + int64_t(placement->destination.y) + placement->destination.h <= 0) { + DrawOp rejected; + rejected.src_w = -srcp->width(); rejected.src_h = srcp->height(); // marks "rejected" in the log + rejected.layer = L; + ops.push_back(rejected); + continue; + } + const auto &p = *placement; + L.crop_x = p.source.x; L.crop_y = p.source.y; + L.crop_w = p.source.w; L.crop_h = p.source.h; + L.dst_x = p.destination.x; L.dst_y = p.destination.y; + L.dst_w = p.destination.w; L.dst_h = p.destination.h; + ops.push_back({srcp, srcp->width(), srcp->height(), L}); + continue; + } + // 0 means "remaining source extent from the crop origin". + if (L.crop_w <= 0) L.crop_w = srcp->width() - L.crop_x; + if (L.crop_h <= 0) L.crop_h = srcp->height() - L.crop_y; + if (!clipRect(L.crop_x, L.crop_y, L.crop_w, L.crop_h, srcp->width(), srcp->height()) || + !clipRect(L.dst_x, L.dst_y, L.crop_w, L.crop_h, canvas_w, canvas_h)) { + ops.push_back({}); + continue; + } + const int ax = chromaXAlign(canvas_fmt); + const int ay = chromaYAlign(canvas_fmt); + L.crop_x = alignCoord(L.crop_x, ax); + L.crop_y = alignCoord(L.crop_y, ay); + L.dst_x = alignCoord(L.dst_x, ax); + L.dst_y = alignCoord(L.dst_y, ay); + if (!clipRect(L.crop_x, L.crop_y, L.crop_w, L.crop_h, srcp->width(), srcp->height()) || + !clipRect(L.dst_x, L.dst_y, L.crop_w, L.crop_h, canvas_w, canvas_h)) { + ops.push_back({}); + continue; + } + ops.push_back({srcp, srcp->width(), srcp->height(), L}); + } + // z decides who draws on top; equal z keeps source order (stable). + std::stable_sort(ops.begin(), ops.end(), + [](const DrawOp &a, const DrawOp &b) { return a.layer.z < b.layer.z; }); + return ops; +} + +/// One-line description of a resolved op list, for change-triggered logging. +inline std::string describeDrawOps(const std::vector &ops) { + std::ostringstream desc; + for (size_t i = 0; i < ops.size(); ++i) { + const DrawOp &op = ops[i]; + const LayerSpec &L = op.layer; + if (!op.src) { + if (op.src_w < 0) + desc << " [" << i << ":REJECTED src " << -op.src_w << "x" << op.src_h << " crop " << L.crop_x + << "," << L.crop_y << " " << L.crop_w << "x" << L.crop_h << " box " << L.dst_x << "," + << L.dst_y << " " << L.dst_w << "x" << L.dst_h << "]"; + continue; + } + desc << " [" << i << ":" << op.src_w << "x" << op.src_h << " crop " << L.crop_x << "," << L.crop_y + << " " << L.crop_w << "x" << L.crop_h << " -> " << L.dst_x << "," << L.dst_y << " " + << L.dst_w << "x" << L.dst_h << " z" << L.z << "]"; + } + return desc.str(); +} + +} diff --git a/src/mixer/primitives/pixel_layout.hpp b/src/mixer/primitives/pixel_layout.hpp new file mode 100644 index 00000000..44754dcc --- /dev/null +++ b/src/mixer/primitives/pixel_layout.hpp @@ -0,0 +1,210 @@ +#pragma once +// Compositor rules on top of libavutil pixel descriptors: storage words and +// padding, chroma alignment, per-plane byte regions, per-plane clear values and +// which source formats a canvas accepts. Plane counts come straight from +// av_pix_fmt_count_planes. No CUDA, no node state. +#include +#include + +extern "C" { +#include +#include +#include +#include +} + +namespace avp::mixer { + +/// Chroma subsampling as luma alignment (2 for 4:2:0 horizontally); 1 for unknown formats. +inline int chromaXAlign(AVPixelFormat fmt) { + int h = 0, v = 0; + return av_pix_fmt_get_chroma_sub_sample(fmt, &h, &v) < 0 ? 1 : 1 << h; +} + +inline int chromaYAlign(AVPixelFormat fmt) { + int h = 0, v = 0; + return av_pix_fmt_get_chroma_sub_sample(fmt, &h, &v) < 0 ? 1 : 1 << v; +} + +inline int alignCoord(int v, int a) { + if (a <= 1) + return v; + return v & ~(a - 1); +} + +inline bool clipRect(int &x, int &y, int &rw, int &rh, int lim_w, int lim_h) { + if (rw <= 0 || rh <= 0 || lim_w <= 0 || lim_h <= 0) + return false; + int x2 = x + rw; + int y2 = y + rh; + x = std::max(0, std::min(x, lim_w)); + y = std::max(0, std::min(y, lim_h)); + x2 = std::max(0, std::min(x2, lim_w)); + y2 = std::max(0, std::min(y2, lim_h)); + rw = x2 - x; + rh = y2 - y; + return rw > 0 && rh > 0; +} + +/// Bytes per stored logical sample on every plane of fmt: 1 for 8-bit, 2 for deeper storage. +inline int sampleBytes(AVPixelFormat fmt) { + const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); + return d && d->comp[0].depth + d->comp[0].shift > 8 ? 2 : 1; +} + +/// Least-significant padding bits below each stored sample (6 for P210/P010, else 0). +inline int storageShift(AVPixelFormat fmt) { + const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); + return d ? d->comp[0].shift : 0; +} + +/// Rectangle in luma/packed pixel units -> byte offset region for a given plane (for memcpy2D). +/// Horizontal extents are av_image_get_linesize of the left and right edges (chroma subsampling +/// and the component step, NV12: 2, P210: 4, come from the descriptor); the left edge must be +/// chroma-aligned. Rows follow log2_chroma_h on the chroma planes of multi-plane formats. +inline void lumaRectToPlaneRegion(AVPixelFormat fmt, int lx, int ly, int lw, int lh, int plane, int &bx, + int &by, int &bw_bytes, int &bh) { + const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); + const int left = av_image_get_linesize(fmt, lx, plane), right = av_image_get_linesize(fmt, lx + lw, plane); + if (!d || left < 0 || right < 0) { + bx = by = bw_bytes = bh = 0; + return; + } + const bool chroma = av_pix_fmt_count_planes(fmt) > 1 && (plane == 1 || plane == 2); + const int sy = chroma ? d->log2_chroma_h : 0; + bx = left; + bw_bytes = right - left; + by = ly >> sy; + bh = AV_CEIL_RSHIFT(ly + lh, sy) - by; +} + +// Returns true when src_fmt can be overlaid onto canvas_fmt by treating the source as fully opaque: +// canvas must have a separate alpha plane, source must not, and all other plane layouts must match. +inline bool isAlphaCompatible(AVPixelFormat src_fmt, AVPixelFormat canvas_fmt) { + const AVPixFmtDescriptor *sd = av_pix_fmt_desc_get(src_fmt); + const AVPixFmtDescriptor *cd = av_pix_fmt_desc_get(canvas_fmt); + if (!sd || !cd) return false; + if (!(cd->flags & AV_PIX_FMT_FLAG_ALPHA)) return false; + if (sd->flags & AV_PIX_FMT_FLAG_ALPHA) return false; + if (sd->nb_components != cd->nb_components - 1) return false; + if (sd->log2_chroma_w != cd->log2_chroma_w) return false; + if (sd->log2_chroma_h != cd->log2_chroma_h) return false; + if (sd->comp[0].depth != cd->comp[0].depth || sd->comp[0].shift != cd->comp[0].shift) return false; + if (av_pix_fmt_count_planes(src_fmt) != av_pix_fmt_count_planes(canvas_fmt) - 1) return false; + return true; +} + +// Packed 8-bit RGB with 3 or 4 bytes per pixel (rgb0, bgr0, rgba, bgra, rgb24, ...): the compositor +// converts such sources onto an NV12 canvas on the GPU, so browser pages and video mix freely. +inline bool isPackedRgb8(AVPixelFormat fmt, int &step, int &r_off, int &g_off, int &b_off) { + const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); + if (!d || !(d->flags & AV_PIX_FMT_FLAG_RGB) || (d->flags & (AV_PIX_FMT_FLAG_PLANAR | AV_PIX_FMT_FLAG_BITSTREAM))) + return false; + if (d->nb_components < 3) return false; + for (int c = 0; c < 3; ++c) + if (d->comp[c].depth != 8 || d->comp[c].plane != 0) return false; + step = d->comp[0].step; + r_off = d->comp[0].offset; + g_off = d->comp[1].offset; + b_off = d->comp[2].offset; + return step == 3 || step == 4; +} + +/// Byte offset of alpha inside a packed 8-bit pixel, or -1 when there is none. +inline int packedAlphaOffset(AVPixelFormat fmt) { + const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); + if (!d || !(d->flags & AV_PIX_FMT_FLAG_ALPHA) || d->nb_components < 4) return -1; + const AVComponentDescriptor &a = d->comp[3]; + return (a.depth == 8 && a.plane == 0) ? a.offset : -1; +} + +// The fused RGB(A) path renders onto semiplanar YUV canvases (NV12, P210): two +// planes, no alpha, interleaved Cb/Cr. Planar-chroma canvases reject packed RGB +// sources instead of growing a third store path here. SDR graphics are embedded +// into the explicitly selected SDR/HLG/PQ canvas color contract. +inline bool isRgbToYuvConvertible(AVPixelFormat src_fmt, AVPixelFormat canvas_fmt) { + const AVPixFmtDescriptor *cd = av_pix_fmt_desc_get(canvas_fmt); + int step, r, g, b; + return cd && !(cd->flags & (AV_PIX_FMT_FLAG_RGB | AV_PIX_FMT_FLAG_ALPHA)) && + cd->nb_components == 3 && av_pix_fmt_count_planes(canvas_fmt) == 2 && + isPackedRgb8(src_fmt, step, r, g, b); +} + +// A lower-depth semiplanar YUV source (e.g. NV12 from NVDEC) drawn onto a deeper +// semiplanar canvas (P210): the fused scaler promotes its codes by 2^(dbits-sbits) +// (NV12->P210 is <<2, 16->64 / 235->940) and resamples the chroma footprint, +// so an 8-bit clip mixes onto a 10-bit program with no separate convert node. +// SDR only, like the RGB path; the source keeps its own subsampling. +inline bool isYuvPromoteConvertible(AVPixelFormat src_fmt, AVPixelFormat canvas_fmt) { + const AVPixFmtDescriptor *sd = av_pix_fmt_desc_get(src_fmt); + const AVPixFmtDescriptor *cd = av_pix_fmt_desc_get(canvas_fmt); + if (!sd || !cd || src_fmt == canvas_fmt) return false; + if ((sd->flags | cd->flags) & (AV_PIX_FMT_FLAG_RGB | AV_PIX_FMT_FLAG_ALPHA)) return false; + if (sd->nb_components != 3 || cd->nb_components != 3) return false; + if (av_pix_fmt_count_planes(src_fmt) != 2 || av_pix_fmt_count_planes(canvas_fmt) != 2) return false; // semiplanar UV + return sd->comp[0].depth <= cd->comp[0].depth; +} + +/// Every source format the canvas accepts: identical, opaque-onto-alpha, packed RGB, or promotable YUV. +inline bool canvasAccepts(AVPixelFormat src_fmt, AVPixelFormat canvas_fmt) { + return src_fmt == canvas_fmt || isAlphaCompatible(src_fmt, canvas_fmt) || + isRgbToYuvConvertible(src_fmt, canvas_fmt) || isYuvPromoteConvertible(src_fmt, canvas_fmt); +} + +// Returns the plane index of the alpha component for planar formats, or -1 if there is none / +// the format is packed (all components on plane 0). +inline int alphaPlaneIndex(AVPixelFormat fmt) { + const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); + if (!d || !(d->flags & AV_PIX_FMT_FLAG_ALPHA)) return -1; + int max_plane = -1; + for (int i = 0; i < d->nb_components; ++i) + max_plane = std::max(max_plane, d->comp[i].plane); + return max_plane > 0 ? max_plane : -1; +} + +inline uint16_t blackLumaValue(AVPixelFormat fmt, const AVFrame *color_src) { + if (color_src && color_src->color_range == AVCOL_RANGE_JPEG) + return 0; + const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); + return (uint16_t)(16 << ((d ? d->comp[0].depth : 8) - 8)); +} + +/// Logical sample value that clears `plane` to opaque black; false when the format has no +/// single-value clear (packed YUV, packed RGBA). +inline bool planeClearValue(AVPixelFormat fmt, const AVFrame *color_src, int plane, uint16_t &value) { + const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); + if (!d) + return false; + const int depth = d->comp[0].depth; + + const int alpha_p = alphaPlaneIndex(fmt); + if (plane == alpha_p) { + value = (uint16_t)((1 << depth) - 1); // 10-bit opaque is 1023, not 255 << 2. + return true; + } + + if (d->flags & AV_PIX_FMT_FLAG_RGB) { + // Packed RGB without alpha can be cleared with all-zero bytes. + // Packed RGBA needs a byte pattern to make opaque black; leave it unsupported here. + if ((d->flags & AV_PIX_FMT_FLAG_ALPHA) && alpha_p < 0) + return false; + value = 0; + return true; + } + + const int planes = av_pix_fmt_count_planes(fmt); + if (planes == 1) { + if (d->nb_components == 1) { + value = blackLumaValue(fmt, color_src); + return true; + } + // Packed YUV (e.g. yuyv422) requires a repeating Y/Cb/Y/Cr pattern. + return false; + } + + // Chroma sits on plane 1 (semiplanar NV12/NV21/P210) or planes 1-2 (planar). + value = (plane == 1 || plane == 2) ? (uint16_t)(1 << (depth - 1)) : blackLumaValue(fmt, color_src); + return true; +} + +} diff --git a/src/mixer/routing.hpp b/src/mixer/routing.hpp new file mode 100644 index 00000000..bef544b1 --- /dev/null +++ b/src/mixer/routing.hpp @@ -0,0 +1,110 @@ +#pragma once +// Pure functions from MixerState/SceneDefinition to node parameters: router +// route tables and the compositor layer array. No node access, unit-testable. +#include "primitives/MixerState.hpp" +#include +#include +#include + +namespace avp::mixer { + +inline Parameters routesToParameters(const std::vector& routes) { + Parameters arr = Parameters::array(); + for (int input_index : routes) + arr.push_back(input_index); + return arr; +} + +inline void ensureRouteTableSize(MixerState& st, const std::string& router_name) { + int count = st.router_output_counts[router_name]; + auto& routes = st.router_routes[router_name]; + if ((int)routes.size() != count) + routes.assign(count, -1); +} + +inline std::unordered_map> currentRouterTables(MixerState& st) { + std::unordered_map> tables; + for (const auto& [router_name, count] : st.router_output_counts) { + ensureRouteTableSize(st, router_name); + tables[router_name] = st.router_routes[router_name]; + if ((int)tables[router_name].size() != count) + tables[router_name].assign(count, -1); + } + return tables; +} + +inline int routeOutputForSlot(const MixerState::SourceInfo& info, bool is_slot_a) { + return is_slot_a ? info.route_output_a : info.route_output_b; +} + +inline std::string routeOutputLabelForSlot(const MixerState::SourceInfo& info, bool is_slot_a) { + return is_slot_a ? info.route_output_label_a : info.route_output_label_b; +} + +/// Rewrite one slot's entries in every router table: clear this slot's outputs, then point the +/// outputs of sources used by `scene` (nullptr: none) at the scene's explicit routes. +inline void setRoutedSlotInTables(MixerState& st, + std::unordered_map>& tables, + bool is_slot_a, + const SceneDefinition* scene) { + for (const auto& [src_name, info] : st.sources) { + if (!info.routed) + continue; + const int output_index = routeOutputForSlot(info, is_slot_a); + if (output_index < 0) + continue; + + auto& routes = tables[info.router_node_name]; + const int count = st.router_output_counts[info.router_node_name]; + if ((int)routes.size() != count) + routes.assign(count, -1); + if (output_index >= (int)routes.size()) + throw Error("mixer: routed source " + src_name + " output index " + + std::to_string(output_index) + " (" + + routeOutputLabelForSlot(info, is_slot_a) + ") exceeds router " + + info.router_node_name + " route table"); + + routes[output_index] = -1; + if (!scene || !scene->sources.count(src_name)) + continue; + + auto route_it = scene->routes.find(src_name); + if (route_it == scene->routes.end()) { + throw Error("mixer: scene " + scene->name + " uses routed source " + + src_name + " without an explicit route"); + } + routes[output_index] = route_it->second; + } +} + +/// One cuda_rect_overlay layer per compositor src index (see mixer.source). Omitted sources use a dummy rect. +inline Parameters compositorLayersFromScene(const MixerState& st, const SceneDefinition& scene) { + static const Parameters kUnusedLayer = Parameters({{"dst_x", 0}, {"dst_y", 0}}); + + int max_idx = -1; + for (const auto& [_, info] : st.sources) + max_idx = std::max(max_idx, info.input_index); + + Parameters arr = Parameters::array(); + for (int i = 0; i <= max_idx; ++i) { + std::string name_at; + for (const auto& [name, info] : st.sources) { + if (info.input_index == i) { + name_at = name; + break; + } + } + if (name_at.empty()) { + arr.push_back(kUnusedLayer); + continue; + } + auto it = scene.sources.find(name_at); + if (it == scene.sources.end()) + arr.push_back(kUnusedLayer); + else + arr.push_back(it->second.layer); + } + return arr; +} + +} diff --git a/src/nodes/clip_cache/clip_cache.cpp b/src/nodes/clip_cache/clip_cache.cpp index fe69b71e..f6228a19 100644 --- a/src/nodes/clip_cache/clip_cache.cpp +++ b/src/nodes/clip_cache/clip_cache.cpp @@ -1,6 +1,6 @@ #include "../node_common.hpp" #include "ClipCache.hpp" -#include "../../mixer/MonotonicClock.hpp" +#include "../../mixer/primitives/MonotonicClock.hpp" extern "C" { #include diff --git a/src/nodes/encoders.cpp b/src/nodes/encoders.cpp index 4099b848..637b4131 100644 --- a/src/nodes/encoders.cpp +++ b/src/nodes/encoders.cpp @@ -3,7 +3,8 @@ #include #include #include "../hwaccel.hpp" -#include "../mixer/CutLatencyProbe.hpp" +#include "../hdr_metadata.hpp" +#include "../mixer/primitives/CutLatencyProbe.hpp" template class Encoder: public NodeSISO, public IEncoder, public ReportsFinishByFlag, public IFlushable, public avp::mixer::CutLatencyObserver { protected: @@ -15,6 +16,7 @@ template class Enc av::Dictionary options_; int enc_flags_ = 0; bool timestamps_passthrough_ = false; + Parameters hdr_metadata_; // optional static HDR side data, serialised verbatim (see hdr_metadata.hpp) av::Timestamp prev_ts_ = NOTS; std::shared_ptr hwaccel_; void emitPacket(const av::Packet& pkt) { @@ -81,6 +83,7 @@ template class Enc } enc_.setTimeBase(getTimeBase()); + if (hdr_metadata_.is_object()) attachHdrMetadata(enc_.raw(), hdr_metadata_); // make copy of options because otherwise enc_.open will remove all consumed ones av::Dictionary options(options_); @@ -199,6 +202,13 @@ template class Enc if (params.count("timestamps_passthrough") > 0) { r->timestamps_passthrough_ = params["timestamps_passthrough"]; } + if (params.count("hdr_metadata") > 0) { + if constexpr (!std::is_same_v) + throw Error("hdr_metadata applies to enc_video only"); + if (!params["hdr_metadata"].is_object()) + throw Error("hdr_metadata must be an object (primaries, white_point, max_luminance, min_luminance, max_cll, max_fall)"); + r->hdr_metadata_ = params["hdr_metadata"]; + } return r; } }; diff --git a/src/nodes/filters.cpp b/src/nodes/filters.cpp index 1a231a20..a148f2d8 100644 --- a/src/nodes/filters.cpp +++ b/src/nodes/filters.cpp @@ -598,8 +598,9 @@ template class FilterNode: p virtual ~FilterNode() { freeFilterGraph(); } - virtual void process() { - // Check for EOF markers on all inputs before normal processing + // Pop EOF markers sitting at the head of any input and close that buffersrc. + // Returns true once every input has reached EOF. + bool consumeEofMarkers() { for (int i = 0; i < (int)this->source_edges_.size(); i++) { if (input_eof_[i]) continue; T* p = this->source_edges_[i]->peek(); @@ -615,7 +616,10 @@ template class FilterNode: p } } } - if (allInputsEof()) { + return allInputsEof(); + } + virtual void process() { + if (consumeEofMarkers()) { drainAndFinish(); return; } @@ -625,6 +629,16 @@ template class FilterNode: p if (source_index >= 0) { std::shared_ptr> edge = this->source_edges_[source_index]; frmin = edge->peek(); + if (frmin && isEofMarker(*frmin)) { + // The marker arrived while findSourceWithData() was waiting, after the + // check above. Treating it as an invalid frame used to finish the node + // immediately and drop every frame still queued on the other inputs + // (a multi-input transition lost its tail whenever one scene ended first). + if (consumeEofMarkers()) { + drainAndFinish(); + } + return; + } if (frmin && (!frmin->isNull()) && frmin->isComplete() && frmin->timeBase().getNumerator() && frmin->timeBase().getDenominator()) { Port &source_port = sources_[source_index]; if (!source_port.checkFrame(*frmin, edge)) { diff --git a/src/nodes/force_fps.cpp b/src/nodes/force_fps.cpp index 8ad7ed5c..5d4076fc 100644 --- a/src/nodes/force_fps.cpp +++ b/src/nodes/force_fps.cpp @@ -31,10 +31,10 @@ template class ForceFPS: public NonBlockingNode>, public const av::Rational fps, const av::Rational timebase, std::string label): NodeSISO(std::move(source), std::move(sink)), fps_(fps), label_(std::move(label)), timebase_(timebase) { if (timebase_.getDenominator()==0 || timebase_.getNumerator()==0) { - timebase_ = av::Rational(fps_.getDenominator(), fps_.getNumerator()); + timebase_ = av_inv_q(fps_.getValue()); frame_delta_ = av::Timestamp(1, timebase_); } else { - frame_delta_ = rescaleTS(av::Timestamp(1, {fps_.getDenominator(), fps_.getNumerator()}), timebase_); + frame_delta_ = rescaleTS(av::Timestamp(1, av_inv_q(fps_.getValue())), timebase_); } logstream << "Set timebase " << timebase_ << ", frame rate " << fps_ << ", frame delta " << frame_delta_; } diff --git a/src/nodes/hwaccel/cuda_rect_draw.cpp b/src/nodes/hwaccel/cuda_rect_draw.cpp new file mode 100644 index 00000000..e7c4fb50 --- /dev/null +++ b/src/nodes/hwaccel/cuda_rect_draw.cpp @@ -0,0 +1,303 @@ +#include "cuda_rect_draw.hpp" +#include "../../mixer/primitives/compositor_color.hpp" +#ifdef HAVE_CUDA_RECT_SCALE +#include "../../../objs/src/nodes/hwaccel/cuda_rect_scale.ptx.h" +#endif + +extern "C" { +#include +} + +#include + +namespace avp::mixer { + +int checkCu(CUresult err, const char *func) { + if (err == CUDA_SUCCESS) + return 0; + const char *err_name = nullptr; + const char *err_string = nullptr; + if (cuGetErrorName && cuGetErrorString) { + cuGetErrorName(err, &err_name); + cuGetErrorString(err, &err_string); + } + logstream << "cuda_rect_overlay: " << func << " failed: " << (err_name ? err_name : "?") << ": " + << (err_string ? err_string : "?"); + return -1; +} + +namespace { + +/// Transfer id the rgb_to_yuv kernels take: 0 = HLG, 1 = PQ, 2 = SDR (BT.709). +int kernelTransfer(AVColorTransferCharacteristic trc) { + return trc == AVCOL_TRC_ARIB_STD_B67 ? 0 : trc == AVCOL_TRC_SMPTE2084 ? 1 : 2; +} + +bool memcpy2d_async(CUstream stream, CUdeviceptr dst, size_t dst_pitch, size_t dst_x_off_bytes, + CUdeviceptr src, size_t src_pitch, size_t src_x_off_bytes, size_t width_bytes, + size_t height) { + CUDA_MEMCPY2D cpy{}; + cpy.srcMemoryType = CU_MEMORYTYPE_DEVICE; + cpy.srcDevice = src + (CUdeviceptr)src_x_off_bytes; + cpy.srcPitch = src_pitch; + cpy.dstMemoryType = CU_MEMORYTYPE_DEVICE; + cpy.dstDevice = dst + (CUdeviceptr)dst_x_off_bytes; + cpy.dstPitch = dst_pitch; + cpy.WidthInBytes = width_bytes; + cpy.Height = height; + return AVP_CHECK_CU(cuMemcpy2DAsync(&cpy, stream)) == 0; +} + +bool blitLayerPlanes(CUstream stream, AVPixelFormat sw_fmt, const AVFrame *src, const AVFrame *dst, + int src_luma_x, int src_luma_y, int lw, int lh, int dst_luma_x, int dst_luma_y) { + const int planes = av_pix_fmt_count_planes(sw_fmt); + for (int p = 0; p < planes && p < AV_NUM_DATA_POINTERS; ++p) { + if (!src->data[p] || !dst->data[p]) + continue; + int sx, sy, sw_bytes, sh; + lumaRectToPlaneRegion(sw_fmt, src_luma_x, src_luma_y, lw, lh, p, sx, sy, sw_bytes, sh); + int dx, dy, dw_bytes, dh; + lumaRectToPlaneRegion(sw_fmt, dst_luma_x, dst_luma_y, lw, lh, p, dx, dy, dw_bytes, dh); + if (sw_bytes <= 0 || sh <= 0 || dw_bytes <= 0 || dh <= 0) + continue; + const size_t src_pitch = (size_t)src->linesize[p]; + const size_t dst_pitch = (size_t)dst->linesize[p]; + CUdeviceptr sbase = (CUdeviceptr)(uintptr_t)src->data[p]; + CUdeviceptr dbase = (CUdeviceptr)(uintptr_t)dst->data[p]; + const size_t src_off = (size_t)sy * src_pitch + (size_t)sx; + const size_t dst_off = (size_t)dy * dst_pitch + (size_t)dx; + if (!memcpy2d_async(stream, dbase, dst_pitch, dst_off, sbase, src_pitch, src_off, (size_t)sw_bytes, + (size_t)sh)) + return false; + } + return true; +} + +// Set a rectangular region of one plane to a constant logical sample value. +// Word-stored formats take the value below the storage shift (64, not 64 << 6) +// and get a 16-bit memset; padding bits stay zero. +void fillPlaneRect(AVPixelFormat fmt, AVFrame *f, int plane, + int lx, int ly, int lw, int lh, uint16_t value) { + if (!f->data[plane] || f->linesize[plane] <= 0) return; + int bx, by, bw, bh; + lumaRectToPlaneRegion(fmt, lx, ly, lw, lh, plane, bx, by, bw, bh); + if (bw <= 0 || bh <= 0) return; + const size_t pitch = (size_t)f->linesize[plane]; + CUdeviceptr base = (CUdeviceptr)(uintptr_t)f->data[plane] + (CUdeviceptr)((size_t)by * pitch + (size_t)bx); + if (sampleBytes(fmt) == 2) + AVP_CHECK_CU(cuMemsetD2D16(base, (unsigned int)pitch, + (unsigned short)(value << storageShift(fmt)), (size_t)bw / 2, (size_t)bh)); + else + AVP_CHECK_CU(cuMemsetD2D8(base, (unsigned int)pitch, (unsigned char)value, (size_t)bw, (size_t)bh)); +} + +bool fillFrameBlack(AVPixelFormat fmt, AVFrame *f, const AVFrame *color_src) { + const int planes = av_pix_fmt_count_planes(fmt); + for (int p = 0; p < planes && p < AV_NUM_DATA_POINTERS; ++p) { + if (!f->data[p]) + continue; + uint16_t value = 0; + if (!planeClearValue(fmt, color_src, p, value)) + return false; + fillPlaneRect(fmt, f, p, 0, 0, f->width, f->height, value); + } + return true; +} + +} // namespace + +AVPixelFormat CudaRectDraw::frameSwFormat(const av::VideoFrame &f) { + if (!f.raw() || !f.raw()->hw_frames_ctx || !f.raw()->hw_frames_ctx->data) + return AV_PIX_FMT_NONE; + AVHWFramesContext *ctx = (AVHWFramesContext *)f.raw()->hw_frames_ctx->data; + return ctx ? ctx->sw_format : AV_PIX_FMT_NONE; +} + +void CudaRectDraw::ensureDevice() { + if (!hwaccel_ || !hwaccel_->deviceContext() || !hwaccel_->deviceContext()->data) + throw Error("cuda_rect_overlay: invalid hwaccel device"); + AVHWDeviceContext *devctx = (AVHWDeviceContext *)hwaccel_->deviceContext()->data; + cuda_dev_ = (AVCUDADeviceContext *)devctx->hwctx; + if (!cuda_dev_ || !cuda_dev_->cuda_ctx) + throw Error("cuda_rect_overlay: CUDA hwctx missing"); + if (AVP_CHECK_CU(cuCtxSetCurrent(cuda_dev_->cuda_ctx))) + throw Error("cuda_rect_overlay: cuCtxSetCurrent failed"); +} + +void CudaRectDraw::ensureKernels() { +#ifdef HAVE_CUDA_RECT_SCALE + if (scale_kernel_) return; + ensureDevice(); + const std::string image(avpl_rect_scale_ptx, avpl_rect_scale_ptx + avpl_rect_scale_ptx_len); + if (AVP_CHECK_CU(cuModuleLoadDataEx(&scale_module_, image.c_str(), 0, nullptr, nullptr)) || + AVP_CHECK_CU(cuModuleGetFunction(&scale_kernel_, scale_module_, "scale_plane")) || + AVP_CHECK_CU(cuModuleGetFunction(&convert_kernel_, scale_module_, "convert_scale_plane")) || + AVP_CHECK_CU(cuModuleGetFunction(&rgb_kernel_, scale_module_, "rgb_to_yuv")) || + AVP_CHECK_CU(cuModuleGetFunction(&rgba_kernel_, scale_module_, "rgba_over_yuv"))) + throw Error("cuda_rect_overlay: cannot load scaling kernel"); +#endif +} + +void CudaRectDraw::unload() { + if (scale_module_) { + cuCtxSetCurrent(cuda_dev_->cuda_ctx); + AVP_CHECK_CU(cuModuleUnload(scale_module_)); + scale_module_ = nullptr; + } +} + +void CudaRectDraw::clearCanvas(AVFrame *canvas, const AVFrame *color_src) { + if (!fillFrameBlack(canvas_.sw_fmt, canvas, color_src)) + throw Error("cuda_rect_overlay: unsupported sw_format for canvas clear"); +} + +/// Draw a packed RGB source onto the semiplanar YUV canvas: one fused scale +/// + color conversion pass, alpha-blended over what is already there when +/// the layer asks for it and the source carries alpha. +void CudaRectDraw::convertRgbLayer(CUstream stream, const AVFrame *src, AVFrame *dst, const LayerSpec &layer, + int step, int r_off, int g_off, int b_off, int a_off) { + ensureKernels(); + const bool blend = layer.blend && a_off >= 0; + if (!rgb_kernel_ || (blend && !rgba_kernel_)) + throw Error("cuda_rect_overlay: RGB conversion kernel unavailable"); + int sx = layer.crop_x, sy = layer.crop_y, sw = layer.crop_w, sh = layer.crop_h; + int dx = layer.dst_x, dy = layer.dst_y; + int dw = layer.dst_w > 0 ? layer.dst_w : layer.crop_w, dh = layer.dst_w > 0 ? layer.dst_h : layer.crop_h; + int cw = canvas_.width, ch = canvas_.height; + const AVPixFmtDescriptor *cd = av_pix_fmt_desc_get(canvas_.sw_fmt); + int dst_sb = sampleBytes(canvas_.sw_fmt), dst_shift = storageShift(canvas_.sw_fmt); + int dst_scale = 1 << (cd->comp[0].depth - 8), sub_x = cd->log2_chroma_w, sub_y = cd->log2_chroma_h; + CUdeviceptr source = (CUdeviceptr)src->data[0], luma = (CUdeviceptr)dst->data[0], + chroma = (CUdeviceptr)dst->data[1]; + int source_pitch = src->linesize[0], luma_pitch = dst->linesize[0], chroma_pitch = dst->linesize[1]; + int transfer = kernelTransfer(canvas_.transfer); + float sdr_white = canvas_.sdr_white, hdr_peak = canvas_.hdr_peak; + void *opaque_args[] = {&source, &source_pitch, &sx, &sy, &sw, &sh, &step, &r_off, &g_off, &b_off, + &luma, &luma_pitch, &chroma, &chroma_pitch, &dx, &dy, &dw, &dh, &cw, &ch, + &dst_sb, &dst_shift, &dst_scale, &sub_x, &sub_y, &transfer, &sdr_white, &hdr_peak}; + void *blend_args[] = {&source, &source_pitch, &sx, &sy, &sw, &sh, &step, &r_off, &g_off, &b_off, &a_off, + &luma, &luma_pitch, &chroma, &chroma_pitch, &dx, &dy, &dw, &dh, &cw, &ch, + &dst_sb, &dst_shift, &dst_scale, &sub_x, &sub_y, &transfer, &sdr_white, &hdr_peak}; + const int bw = 1 << sub_x, bh = 1 << sub_y; + const int blocks_x = (dw + bw - 1) / bw, blocks_y = (dh + bh - 1) / bh; + if (AVP_CHECK_CU(cuLaunchKernel(blend ? rgba_kernel_ : rgb_kernel_, (blocks_x + 31) / 32, (blocks_y + 7) / 8, 1, + 32, 8, 1, 0, stream, blend ? blend_args : opaque_args, nullptr))) + throw Error("cuda_rect_overlay: RGB conversion launch failed"); +} + +void CudaRectDraw::scaleLayer(CUstream stream, const AVFrame *src, AVFrame *dst, const LayerSpec &layer) { + const AVPixelFormat sw_fmt = canvas_.sw_fmt; + const auto *desc = av_pix_fmt_desc_get(sw_fmt); + int sample_bytes = sampleBytes(sw_fmt), shift = storageShift(sw_fmt); + ensureKernels(); + if (!scale_kernel_) throw Error("cuda_rect_overlay: scaling kernel unavailable"); + for (int p = 0; p < av_pix_fmt_count_planes(sw_fmt); ++p) { + if (!src->data[p]) continue; // Opaque input on an alpha canvas. + int lanes = 1; + for (int c = 0; c < desc->nb_components; ++c) + if (desc->comp[c].plane == p) lanes = std::max(lanes, desc->comp[c].step / sample_bytes); + int sx, sy, sw, sh, dx, dy, dw, dh, cx, cy, cw, ch; + lumaRectToPlaneRegion(sw_fmt, layer.crop_x, layer.crop_y, layer.crop_w, layer.crop_h, + p, sx, sy, sw, sh); + lumaRectToPlaneRegion(sw_fmt, layer.dst_x, layer.dst_y, layer.dst_w, layer.dst_h, + p, dx, dy, dw, dh); + lumaRectToPlaneRegion(sw_fmt, 0, 0, canvas_.width, canvas_.height, p, cx, cy, cw, ch); + // Region byte extents -> lane-group coordinates for the kernel. + const int group = lanes * sample_bytes; + sx /= group; sw /= group; dx /= group; dw /= group; cw /= group; + CUdeviceptr source = (CUdeviceptr)src->data[p], destination = (CUdeviceptr)dst->data[p]; + int source_pitch = src->linesize[p], destination_pitch = dst->linesize[p]; + void *args[] = {&source, &source_pitch, &sx, &sy, &sw, &sh, + &destination, &destination_pitch, &dx, &dy, &dw, &dh, &cw, &ch, &lanes, + &sample_bytes, &shift}; + if (AVP_CHECK_CU(cuLaunchKernel(scale_kernel_, (dw + 31) / 32, (dh + 7) / 8, 1, + 32, 8, 1, 0, stream, args, nullptr))) + throw Error("cuda_rect_overlay: scaling launch failed"); + } +} + +// Draw a lower-depth semiplanar source onto the deeper canvas: bilinear +// scale each plane from the source's geometry to the canvas geometry while +// promoting codes by 2^(dst_depth-src_depth). Reads the source's own plane +// layout, writes the canvas layout, so NV12(4:2:0)->P210(4:2:2) works. +void CudaRectDraw::convertLayer(CUstream stream, AVPixelFormat src_fmt, const AVFrame *src, AVFrame *dst, + const LayerSpec &layer) { + const AVPixelFormat sw_fmt = canvas_.sw_fmt; + ensureKernels(); + if (!convert_kernel_) throw Error("cuda_rect_overlay: convert kernel unavailable"); + const AVPixFmtDescriptor *sd = av_pix_fmt_desc_get(src_fmt); + const AVPixFmtDescriptor *dd = av_pix_fmt_desc_get(sw_fmt); + int src_bytes = sampleBytes(src_fmt), src_shift = storageShift(src_fmt); + int dst_bytes = sampleBytes(sw_fmt), dst_shift = storageShift(sw_fmt); + float mul = float(1 << (dd->comp[0].depth - sd->comp[0].depth)); + int dstw = layer.dst_w > 0 ? layer.dst_w : layer.crop_w; + int dsth = layer.dst_w > 0 ? layer.dst_h : layer.crop_h; + for (int p = 0; p < av_pix_fmt_count_planes(sw_fmt); ++p) { + if (!src->data[p]) continue; + int lanes = 1; + for (int c = 0; c < dd->nb_components; ++c) + if (dd->comp[c].plane == p) lanes = std::max(lanes, dd->comp[c].step / dst_bytes); + int sx, sy, sw, sh, dx, dy, dw, dh, cx, cy, cw, ch; + lumaRectToPlaneRegion(src_fmt, layer.crop_x, layer.crop_y, layer.crop_w, layer.crop_h, + p, sx, sy, sw, sh); + lumaRectToPlaneRegion(sw_fmt, layer.dst_x, layer.dst_y, dstw, dsth, p, dx, dy, dw, dh); + lumaRectToPlaneRegion(sw_fmt, 0, 0, canvas_.width, canvas_.height, p, cx, cy, cw, ch); + sx /= lanes * src_bytes; sw /= lanes * src_bytes; + dx /= lanes * dst_bytes; dw /= lanes * dst_bytes; cw /= lanes * dst_bytes; + CUdeviceptr source = (CUdeviceptr)src->data[p], destination = (CUdeviceptr)dst->data[p]; + int source_pitch = src->linesize[p], destination_pitch = dst->linesize[p]; + void *args[] = {&source, &source_pitch, &sx, &sy, &sw, &sh, &src_bytes, &src_shift, + &destination, &destination_pitch, &dx, &dy, &dw, &dh, &cw, &ch, &lanes, + &dst_bytes, &dst_shift, &mul}; + if (AVP_CHECK_CU(cuLaunchKernel(convert_kernel_, (dw + 31) / 32, (dh + 7) / 8, 1, + 32, 8, 1, 0, stream, args, nullptr))) + throw Error("cuda_rect_overlay: convert launch failed"); + } +} + +void CudaRectDraw::drawLayer(CUstream stream, const av::VideoFrame &src, AVFrame *canvas, const LayerSpec &L) { + const AVPixelFormat sw_fmt = canvas_.sw_fmt; + const int canvas_w = canvas_.width, canvas_h = canvas_.height; + const bool sized = L.dst_w > 0; + const bool full_copy = !sized || (L.dst_w == L.crop_w && L.dst_h == L.crop_h && + L.dst_x >= 0 && L.dst_y >= 0 && L.dst_x + L.dst_w <= canvas_w && L.dst_y + L.dst_h <= canvas_h); + const AVPixelFormat src_sw_fmt = frameSwFormat(src); + int rgb_step, r_off, g_off, b_off; + if (canvas_.transfer != AVCOL_TRC_UNSPECIFIED) { + const AVFrame *frame = src.raw(); + const bool rgb = isPackedRgb8(src_sw_fmt, rgb_step, r_off, g_off, b_off); + const bool valid = rgb + ? isSdrGraphicColor(*frame) + : frame->color_trc == canvas->color_trc && + frame->color_primaries == canvas->color_primaries && + frame->colorspace == canvas->colorspace && frame->color_range == AVCOL_RANGE_MPEG; + if (!valid) + throw Error("cuda_rect_overlay: missing or mismatched source color metadata; " + "declare source color and normalize to the canvas before compositing"); + } + if (src_sw_fmt != sw_fmt && isRgbToYuvConvertible(src_sw_fmt, sw_fmt) && + isPackedRgb8(src_sw_fmt, rgb_step, r_off, g_off, b_off)) { + convertRgbLayer(stream, src.raw(), canvas, L, rgb_step, r_off, g_off, b_off, + packedAlphaOffset(src_sw_fmt)); + } else if (src_sw_fmt != sw_fmt && isYuvPromoteConvertible(src_sw_fmt, sw_fmt)) { + convertLayer(stream, src_sw_fmt, src.raw(), canvas, L); + } else if (!full_copy) { + scaleLayer(stream, src.raw(), canvas, L); + } else if (!blitLayerPlanes(stream, sw_fmt, src.raw(), canvas, L.crop_x, L.crop_y, L.crop_w, + L.crop_h, L.dst_x, L.dst_y)) + throw Error("cuda_rect_overlay: GPU blit failed"); + + // When a non-alpha source is drawn onto an alpha canvas, fill the destination + // alpha rect with the depth's opaque maximum so the output alpha is well-defined. + if (src_sw_fmt != AV_PIX_FMT_NONE && src_sw_fmt != sw_fmt) { + const int alpha_p = alphaPlaneIndex(sw_fmt); + int x = L.dst_x, y = L.dst_y; + int w = sized ? L.dst_w : L.crop_w, h = sized ? L.dst_h : L.crop_h; + uint16_t opaque = 255; + if (alpha_p >= 0 && planeClearValue(sw_fmt, nullptr, alpha_p, opaque) && + clipRect(x, y, w, h, canvas_w, canvas_h)) + fillPlaneRect(sw_fmt, canvas, alpha_p, x, y, w, h, opaque); + } +} + +} diff --git a/src/nodes/hwaccel/cuda_rect_draw.hpp b/src/nodes/hwaccel/cuda_rect_draw.hpp new file mode 100644 index 00000000..714dadbd --- /dev/null +++ b/src/nodes/hwaccel/cuda_rect_draw.hpp @@ -0,0 +1,81 @@ +#pragma once +// GPU side of the CUDA compositor: kernel module lifetime, canvas clearing and +// drawing one resolved layer (blit, scale, 8->10-bit promote, packed RGB(A) +// conversion). Owns no scheduling state; the node decides what to draw when. +#include "../../hwaccel.hpp" +#include "../../mixer/primitives/compositor_layers.hpp" +#include + +extern "C" { +#include +#include +} + +#include + +namespace avp::mixer { + +int checkCu(CUresult err, const char *func); +#define AVP_CHECK_CU(x) ::avp::mixer::checkCu((x), #x) + +class CudaRectDraw { +public: + struct Canvas { + int width = 0; + int height = 0; + AVPixelFormat sw_fmt = AV_PIX_FMT_NONE; + // Canvas transfer; unspecified skips color validation and treats graphics as SDR. + AVColorTransferCharacteristic transfer = AVCOL_TRC_UNSPECIFIED; + float sdr_white = 203.f; + float hdr_peak = 1000.f; + }; + + CudaRectDraw(std::shared_ptr hw, Canvas canvas) + : hwaccel_(std::move(hw)), canvas_(canvas) {} + ~CudaRectDraw() { unload(); } + CudaRectDraw(const CudaRectDraw &) = delete; + CudaRectDraw &operator=(const CudaRectDraw &) = delete; + + const Canvas &canvas() const { return canvas_; } + void setColor(AVColorTransferCharacteristic transfer, float sdr_white, float hdr_peak) { + canvas_.transfer = transfer; + canvas_.sdr_white = sdr_white; + canvas_.hdr_peak = hdr_peak; + } + + /// Make the device's CUDA context current on this thread; throws when the hwaccel is unusable. + void ensureDevice(); + /// Load the scaling/convert kernels once (no-op without HAVE_CUDA_RECT_SCALE). + void ensureKernels(); + void unload(); + CUstream stream() const { return (CUstream)cuda_dev_->stream; } + + /// Clear the whole canvas to opaque black in the canvas format. + void clearCanvas(AVFrame *canvas, const AVFrame *color_src); + + /// Validate the source's color tags against the canvas contract (when one is set), then draw + /// the resolved layer: RGB conversion, YUV promote, scale, or plain blit; finally make the + /// destination alpha opaque on alpha canvases fed by non-alpha sources. + void drawLayer(CUstream stream, const av::VideoFrame &src, AVFrame *canvas, const LayerSpec &layer); + + /// sw_format of a hardware frame, AV_PIX_FMT_NONE when it has no frames context. + static AVPixelFormat frameSwFormat(const av::VideoFrame &f); + +private: + std::shared_ptr hwaccel_; + Canvas canvas_; + AVCUDADeviceContext *cuda_dev_ = nullptr; + CUmodule scale_module_ = nullptr; + CUfunction scale_kernel_ = nullptr; + CUfunction convert_kernel_ = nullptr; + CUfunction rgb_kernel_ = nullptr; + CUfunction rgba_kernel_ = nullptr; + + void convertRgbLayer(CUstream stream, const AVFrame *src, AVFrame *dst, const LayerSpec &layer, + int step, int r_off, int g_off, int b_off, int a_off); + void scaleLayer(CUstream stream, const AVFrame *src, AVFrame *dst, const LayerSpec &layer); + void convertLayer(CUstream stream, AVPixelFormat src_fmt, const AVFrame *src, AVFrame *dst, + const LayerSpec &layer); +}; + +} diff --git a/src/nodes/hwaccel/cuda_rect_overlay.cpp b/src/nodes/hwaccel/cuda_rect_overlay.cpp index 4d4bddb6..7662db4c 100644 --- a/src/nodes/hwaccel/cuda_rect_overlay.cpp +++ b/src/nodes/hwaccel/cuda_rect_overlay.cpp @@ -1,19 +1,15 @@ #include "../node_common.hpp" #include "../../hwaccel.hpp" -#include "../../hwaccel/CompositorGeometry.hpp" -#ifdef HAVE_CUDA_RECT_SCALE -#include "../../../objs/src/nodes/hwaccel/cuda_rect_scale.ptx.h" -#endif +#include "../../mixer/primitives/compositor_layers.hpp" #include "../../SharedTimeline.hpp" #include "../../mixer/Playout.hpp" -#include "../../mixer/MonotonicClock.hpp" +#include "../../mixer/primitives/MonotonicClock.hpp" +#include "cuda_rect_draw.hpp" #include -#include extern "C" { #include -#include -#include +#include #include } @@ -25,389 +21,21 @@ extern "C" { #include #include -namespace { - -static int check_cu(CUresult err, const char *func) { - if (err == CUDA_SUCCESS) - return 0; - const char *err_name = nullptr; - const char *err_string = nullptr; - if (cuGetErrorName && cuGetErrorString) { - cuGetErrorName(err, &err_name); - cuGetErrorString(err, &err_string); - } - logstream << "cuda_rect_overlay: " << func << " failed: " << (err_name ? err_name : "?") << ": " - << (err_string ? err_string : "?"); - return -1; -} +using avp::mixer::CudaRectDraw; +using avp::mixer::DrawOp; +using avp::mixer::LayerSpec; -#define CHECK_CU(x) check_cu((x), #x) - -struct LayerSpec { - int dst_x = 0; - int dst_y = 0; - int crop_x = 0; - int crop_y = 0; - int dst_w = 0; - int dst_h = 0; - bool fit = false; - int source_canvas_w = 0; - int source_canvas_h = 0; - int crop_w= 0; - int crop_h = 0; - int z = 0; // draw order: lower first, ties by source index - bool blend = false; // honour the source's alpha instead of overwriting - - bool operator==(const LayerSpec &other) const { - return dst_x == other.dst_x && dst_y == other.dst_y && z == other.z && blend == other.blend && - crop_x == other.crop_x && crop_y == other.crop_y && - crop_w == other.crop_w && crop_h == other.crop_h && - dst_w == other.dst_w && dst_h == other.dst_h && fit == other.fit && - source_canvas_w == other.source_canvas_w && source_canvas_h == other.source_canvas_h; - } -}; - -struct DrawOp { - const av::VideoFrame *src = nullptr; - int src_w = 0; - int src_h = 0; - LayerSpec layer; - - bool operator==(const DrawOp &other) const { - return (src != nullptr) == (other.src != nullptr) && - src_w == other.src_w && src_h == other.src_h && - layer == other.layer; - } -}; +namespace { -static bool frameUsable(const av::VideoFrame &f) { +bool frameUsable(const av::VideoFrame &f) { return !f.isNull() && f.isComplete() && f.raw() && f.pts().isValid(); } - -static void parseLayerFromJson(const Parameters &obj, LayerSpec &out) { - out.dst_x = obj.value("dst_x", 0); - out.dst_y = obj.value("dst_y", 0); - out.dst_w = obj.value("dst_w", 0); - out.dst_h = obj.value("dst_h", 0); - out.z = obj.value("z", 0); - out.blend = obj.value("blend", false); - const std::string fit = obj.value("fit", std::string("stretch")); - if (fit != "stretch" && fit != "contain") - throw Error("cuda_rect_overlay: fit must be stretch or contain"); - out.fit = fit == "contain"; - out.source_canvas_w = out.source_canvas_h = 0; - if (obj.contains("source_canvas")) { - const auto &canvas = obj.at("source_canvas"); - out.source_canvas_w = canvas.at("w").get(); - out.source_canvas_h = canvas.at("h").get(); - if (!out.fit || out.dst_w <= 0 || out.dst_h <= 0 || out.source_canvas_w <= 0 || out.source_canvas_h <= 0) - throw Error("cuda_rect_overlay: source_canvas requires positive dimensions and fit=contain"); - } - if (out.dst_w < 0 || out.dst_h < 0 || (out.dst_w == 0) != (out.dst_h == 0)) - throw Error("cuda_rect_overlay: dst_w and dst_h must both be positive or both omitted"); -#ifndef HAVE_CUDA_RECT_SCALE - if (out.dst_w || out.dst_h) - throw Error("cuda_rect_overlay: destination sizing requires HAVE_NVCC=1"); -#endif - if (obj.contains("crop") && obj["crop"].is_object()) { - const auto &c = obj["crop"]; - out.crop_x = c.value("x", 0); - out.crop_y = c.value("y", 0); - // 0 means "use remaining source width/height from crop_x/crop_y" (resolved per-frame in processComposite). - out.crop_w = c.value("w", 0); - out.crop_h = c.value("h", 0); - } - // No crop object → crop_x/y/w/h all stay 0 (full source frame from origin). -} - -static std::vector parseLayersArray(const Parameters &arr) { - std::vector layers; - if (!arr.is_array()) - throw Error("cuda_rect_overlay: layers must be an array"); - for (const auto &item : arr) { - if (!item.is_object()) - throw Error("cuda_rect_overlay: layers entries must be objects"); - LayerSpec s; - parseLayerFromJson(item, s); - layers.push_back(s); - } - return layers; -} - -static std::vector parseLayersParam(const Parameters ¶ms) { - if (!params.contains("layers") || !params["layers"].is_array()) - throw Error("cuda_rect_overlay: layers array required (one entry per input in src order)"); - return parseLayersArray(params["layers"]); -} - -static int chromaXAlign(AVPixelFormat sw_fmt) { - const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(sw_fmt); - if (!desc || desc->log2_chroma_w < 0) - return 1; - return 1 << desc->log2_chroma_w; -} - -static int chromaYAlign(AVPixelFormat sw_fmt) { - const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(sw_fmt); - if (!desc || desc->log2_chroma_h < 0) - return 1; - return 1 << desc->log2_chroma_h; -} - -static int alignCoord(int v, int a) { - if (a <= 1) - return v; - return v & ~(a - 1); -} - -static bool clipRect(int &x, int &y, int &rw, int &rh, int lim_w, int lim_h) { - if (rw <= 0 || rh <= 0 || lim_w <= 0 || lim_h <= 0) - return false; - int x2 = x + rw; - int y2 = y + rh; - x = std::max(0, std::min(x, lim_w)); - y = std::max(0, std::min(y, lim_h)); - x2 = std::max(0, std::min(x2, lim_w)); - y2 = std::max(0, std::min(y2, lim_h)); - rw = x2 - x; - rh = y2 - y; - return rw > 0 && rh > 0; -} - -static int numPlanes(AVPixelFormat fmt) { - const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); - if (!d) - return 0; - int np = 0; - for (int i = 0; i < d->nb_components; ++i) - np = std::max(np, d->comp[i].plane + 1); - return np; -} - -/// Rectangle in luma/packed pixel units -> byte offset region for a given plane (for memcpy2D). -static void lumaRectToPlaneRegion(AVPixelFormat fmt, int lx, int ly, int lw, int lh, int plane, int &bx, - int &by, int &bw_bytes, int &bh) { - const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); - if (!d) { - bx = by = bw_bytes = bh = 0; - return; - } - if (fmt == AV_PIX_FMT_NV12) { - if (plane == 0) { - bx = lx; - by = ly; - bw_bytes = lw; - bh = lh; - return; - } - const int sx0 = lx >> d->log2_chroma_w; - const int sy0 = ly >> d->log2_chroma_h; - const int sx1 = AV_CEIL_RSHIFT(lx + lw, d->log2_chroma_w); - const int sy1 = AV_CEIL_RSHIFT(ly + lh, d->log2_chroma_h); - bx = sx0 * 2; - by = sy0; - bw_bytes = (sx1 - sx0) * 2; - bh = sy1 - sy0; - return; - } - const int np = numPlanes(fmt); - if (np == 1) { - int step = 1; - for (int i = 0; i < d->nb_components; ++i) - step = std::max(step, d->comp[i].step); - bx = lx * step; - by = ly; - bw_bytes = lw * step; - bh = lh; - return; - } - // Planar YUV (+ alpha): chroma on planes 1–2 follows log2_chroma_*; other planes match luma grid. - const int sx = (plane == 1 || plane == 2) ? d->log2_chroma_w : 0; - const int sy = (plane == 1 || plane == 2) ? d->log2_chroma_h : 0; - const int x0 = lx >> sx; - const int y0 = ly >> sy; - const int x1 = AV_CEIL_RSHIFT(lx + lw, sx); - const int y1 = AV_CEIL_RSHIFT(ly + lh, sy); - int step = 1; - for (int i = 0; i < d->nb_components; ++i) { - if (d->comp[i].plane == plane) { - step = std::max(1, d->comp[i].step); - break; - } - } - bx = x0 * step; - by = y0; - bw_bytes = (x1 - x0) * step; - bh = y1 - y0; -} - -static bool memcpy2d_async(CUstream stream, CUdeviceptr dst, size_t dst_pitch, size_t dst_x_off_bytes, - CUdeviceptr src, size_t src_pitch, size_t src_x_off_bytes, size_t width_bytes, - size_t height) { - CUDA_MEMCPY2D cpy{}; - cpy.srcMemoryType = CU_MEMORYTYPE_DEVICE; - cpy.srcDevice = src + (CUdeviceptr)src_x_off_bytes; - cpy.srcPitch = src_pitch; - cpy.dstMemoryType = CU_MEMORYTYPE_DEVICE; - cpy.dstDevice = dst + (CUdeviceptr)dst_x_off_bytes; - cpy.dstPitch = dst_pitch; - cpy.WidthInBytes = width_bytes; - cpy.Height = height; - return CHECK_CU(cuMemcpy2DAsync(&cpy, stream)) == 0; -} - -static bool blitLayerPlanes(CUstream stream, AVPixelFormat sw_fmt, const AVFrame *src, const AVFrame *dst, - int src_luma_x, int src_luma_y, int lw, int lh, int dst_luma_x, int dst_luma_y) { - const int planes = numPlanes(sw_fmt); - for (int p = 0; p < planes && p < AV_NUM_DATA_POINTERS; ++p) { - if (!src->data[p] || !dst->data[p]) - continue; - int sx, sy, sw_bytes, sh; - lumaRectToPlaneRegion(sw_fmt, src_luma_x, src_luma_y, lw, lh, p, sx, sy, sw_bytes, sh); - int dx, dy, dw_bytes, dh; - lumaRectToPlaneRegion(sw_fmt, dst_luma_x, dst_luma_y, lw, lh, p, dx, dy, dw_bytes, dh); - if (sw_bytes <= 0 || sh <= 0 || dw_bytes <= 0 || dh <= 0) - continue; - const size_t src_pitch = (size_t)src->linesize[p]; - const size_t dst_pitch = (size_t)dst->linesize[p]; - CUdeviceptr sbase = (CUdeviceptr)(uintptr_t)src->data[p]; - CUdeviceptr dbase = (CUdeviceptr)(uintptr_t)dst->data[p]; - const size_t src_off = (size_t)sy * src_pitch + (size_t)sx; - const size_t dst_off = (size_t)dy * dst_pitch + (size_t)dx; - if (!memcpy2d_async(stream, dbase, dst_pitch, dst_off, sbase, src_pitch, src_off, (size_t)sw_bytes, - (size_t)sh)) - return false; - } - return true; -} - -// Returns true when src_fmt can be overlaid onto canvas_fmt by treating the source as fully opaque: -// canvas must have a separate alpha plane, source must not, and all other plane layouts must match. -static bool isAlphaCompatible(AVPixelFormat src_fmt, AVPixelFormat canvas_fmt) { - const AVPixFmtDescriptor *sd = av_pix_fmt_desc_get(src_fmt); - const AVPixFmtDescriptor *cd = av_pix_fmt_desc_get(canvas_fmt); - if (!sd || !cd) return false; - if (!(cd->flags & AV_PIX_FMT_FLAG_ALPHA)) return false; - if (sd->flags & AV_PIX_FMT_FLAG_ALPHA) return false; - if (sd->nb_components != cd->nb_components - 1) return false; - if (sd->log2_chroma_w != cd->log2_chroma_w) return false; - if (sd->log2_chroma_h != cd->log2_chroma_h) return false; - if (numPlanes(src_fmt) != numPlanes(canvas_fmt) - 1) return false; - return true; -} - -// Packed 8-bit RGB with 3 or 4 bytes per pixel (rgb0, bgr0, rgba, bgra, rgb24, ...): the compositor -// converts such sources onto an NV12 canvas on the GPU, so browser pages and video mix freely. -static bool isPackedRgb8(AVPixelFormat fmt, int &step, int &r_off, int &g_off, int &b_off) { - const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); - if (!d || !(d->flags & AV_PIX_FMT_FLAG_RGB) || (d->flags & (AV_PIX_FMT_FLAG_PLANAR | AV_PIX_FMT_FLAG_BITSTREAM))) - return false; - if (d->nb_components < 3) return false; - for (int c = 0; c < 3; ++c) - if (d->comp[c].depth != 8 || d->comp[c].plane != 0) return false; - step = d->comp[0].step; - r_off = d->comp[0].offset; - g_off = d->comp[1].offset; - b_off = d->comp[2].offset; - return step == 3 || step == 4; -} - -/// Byte offset of alpha inside a packed 8-bit pixel, or -1 when there is none. -static int packedAlphaOffset(AVPixelFormat fmt) { - const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); - if (!d || !(d->flags & AV_PIX_FMT_FLAG_ALPHA) || d->nb_components < 4) return -1; - const AVComponentDescriptor &a = d->comp[3]; - return (a.depth == 8 && a.plane == 0) ? a.offset : -1; -} - -static bool isRgbToNv12Convertible(AVPixelFormat src_fmt, AVPixelFormat canvas_fmt) { - int step, r, g, b; - return canvas_fmt == AV_PIX_FMT_NV12 && isPackedRgb8(src_fmt, step, r, g, b); -} - -// Returns the plane index of the alpha component for planar formats, or -1 if there is none / -// the format is packed (all components on plane 0). -static int alphaPlaneIndex(AVPixelFormat fmt) { - const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); - if (!d || !(d->flags & AV_PIX_FMT_FLAG_ALPHA)) return -1; - int max_plane = -1; - for (int i = 0; i < d->nb_components; ++i) - max_plane = std::max(max_plane, d->comp[i].plane); - return max_plane > 0 ? max_plane : -1; -} - -// Set a rectangular region of one plane to a constant byte value. -static void fillPlaneRect(AVPixelFormat fmt, AVFrame *f, int plane, - int lx, int ly, int lw, int lh, uint8_t value) { - if (!f->data[plane] || f->linesize[plane] <= 0) return; - int bx, by, bw, bh; - lumaRectToPlaneRegion(fmt, lx, ly, lw, lh, plane, bx, by, bw, bh); - if (bw <= 0 || bh <= 0) return; - const size_t pitch = (size_t)f->linesize[plane]; - CUdeviceptr base = (CUdeviceptr)(uintptr_t)f->data[plane]; - CHECK_CU(cuMemsetD2D8(base + (CUdeviceptr)((size_t)by * pitch + (size_t)bx), - (unsigned int)pitch, value, (size_t)bw, (size_t)bh)); -} - -static uint8_t blackLumaValue(const AVFrame *color_src) { - return color_src && color_src->color_range == AVCOL_RANGE_JPEG ? 0 : 16; -} - -static bool planeClearValue(AVPixelFormat fmt, const AVFrame *color_src, int plane, uint8_t &value) { - const AVPixFmtDescriptor *d = av_pix_fmt_desc_get(fmt); - if (!d) - return false; - - const int alpha_p = alphaPlaneIndex(fmt); - if (plane == alpha_p) { - value = 255; - return true; - } - - if (d->flags & AV_PIX_FMT_FLAG_RGB) { - // Packed RGB without alpha can be cleared with all-zero bytes. - // Packed RGBA needs a byte pattern to make opaque black; leave it unsupported here. - if ((d->flags & AV_PIX_FMT_FLAG_ALPHA) && alpha_p < 0) - return false; - value = 0; - return true; - } - - if (fmt == AV_PIX_FMT_NV12 || fmt == AV_PIX_FMT_NV21) { - value = plane == 0 ? blackLumaValue(color_src) : 128; - return true; - } - - const int planes = numPlanes(fmt); - if (planes == 1) { - if (d->nb_components == 1) { - value = blackLumaValue(color_src); - return true; - } - // Packed YUV (e.g. yuyv422) requires a repeating Y/Cb/Y/Cr pattern. - return false; - } - - value = (plane == 1 || plane == 2) ? 128 : blackLumaValue(color_src); - return true; -} - -static bool fillFrameBlack(AVPixelFormat fmt, AVFrame *f, const AVFrame *color_src) { - const int planes = numPlanes(fmt); - for (int p = 0; p < planes && p < AV_NUM_DATA_POINTERS; ++p) { - if (!f->data[p]) - continue; - uint8_t value = 0; - if (!planeClearValue(fmt, color_src, p, value)) - return false; - fillPlaneRect(fmt, f, p, 0, 0, f->width, f->height, value); - } - return true; -} - } // namespace +/// Multi-input GPU compositor. Scheduling (clocked playout or timestamp matching), control +/// (`active_inputs`, `layers`, timeline) and output frame plumbing live here; layer geometry is +/// resolved by compositor_layers.hpp and the CUDA work is done by CudaRectDraw. class CudaRectOverlay : public NodeMultiInput, public NodeSingleOutput, public IVideoFormatSource, @@ -415,12 +43,25 @@ class CudaRectOverlay : public NodeMultiInput, public IInputReset, public TimelineReader, public IInputsObjects { - std::shared_ptr hwaccel_; AVBufferRef *out_frames_ref_ = nullptr; int canvas_w_ = 0; int canvas_h_ = 0; AVPixelFormat sw_fmt_ = AV_PIX_FMT_NONE; + // Canvas color contract; unspecified preserves the node's metadata inheritance for non-mixer callers. + AVColorTransferCharacteristic canvas_trc_ = AVCOL_TRC_UNSPECIFIED; + CudaRectDraw draw_; + + bool hasCanvasColor() const { return canvas_trc_ != AVCOL_TRC_UNSPECIFIED; } + + void setCanvasColor(AVFrame *frame) const { + if (!hasCanvasColor()) return; + const bool sdr = canvas_trc_ == AVCOL_TRC_BT709; + frame->color_trc = canvas_trc_; + frame->color_primaries = sdr ? AVCOL_PRI_BT709 : AVCOL_PRI_BT2020; + frame->colorspace = sdr ? AVCOL_SPC_BT709 : AVCOL_SPC_BT2020_NCL; + frame->color_range = AVCOL_RANGE_MPEG; + } std::vector default_layers_; mutable std::mutex layers_mutex_; @@ -428,11 +69,6 @@ class CudaRectOverlay : public NodeMultiInput, int debug_log_every_n_ = 0; uint64_t frame_counter_ = 0; - AVCUDADeviceContext *cuda_dev_ = nullptr; - CUmodule scale_module_ = nullptr; - CUfunction scale_kernel_ = nullptr; - CUfunction rgb_kernel_ = nullptr; - CUfunction rgba_kernel_ = nullptr; std::string last_ops_desc_; bool sent_eof_ = false; @@ -453,7 +89,6 @@ class CudaRectOverlay : public NodeMultiInput, uint32_t applied_active_mask_ = 0; uint32_t applied_prewarm_mask_ = 0; - // Bound how long we will wait for every active layer to produce a fresh // post-activation frame before emitting with whatever we have. <=0 disables // the bound (the historical "wait forever" behavior). The mixer can stall @@ -463,93 +98,10 @@ class CudaRectOverlay : public NodeMultiInput, int64_t warmup_started_pts_ = -1; void freeHwContexts() { - if (scale_module_) { - cuCtxSetCurrent(cuda_dev_->cuda_ctx); - CHECK_CU(cuModuleUnload(scale_module_)); - scale_module_ = nullptr; - } + draw_.unload(); av_buffer_unref(&out_frames_ref_); } - void ensureCudaDevice() { - if (!hwaccel_ || !hwaccel_->deviceContext() || !hwaccel_->deviceContext()->data) - throw Error("cuda_rect_overlay: invalid hwaccel device"); - AVHWDeviceContext *devctx = (AVHWDeviceContext *)hwaccel_->deviceContext()->data; - cuda_dev_ = (AVCUDADeviceContext *)devctx->hwctx; - if (!cuda_dev_ || !cuda_dev_->cuda_ctx) - throw Error("cuda_rect_overlay: CUDA hwctx missing"); - if (CHECK_CU(cuCtxSetCurrent(cuda_dev_->cuda_ctx))) - throw Error("cuda_rect_overlay: cuCtxSetCurrent failed"); - } - - void ensureScaleKernel() { -#ifdef HAVE_CUDA_RECT_SCALE - if (scale_kernel_) return; - ensureCudaDevice(); - const std::string image(avpl_rect_scale_ptx, avpl_rect_scale_ptx + avpl_rect_scale_ptx_len); - if (CHECK_CU(cuModuleLoadDataEx(&scale_module_, image.c_str(), 0, nullptr, nullptr)) || - CHECK_CU(cuModuleGetFunction(&scale_kernel_, scale_module_, "scale_plane")) || - CHECK_CU(cuModuleGetFunction(&rgb_kernel_, scale_module_, "rgb_to_nv12")) || - CHECK_CU(cuModuleGetFunction(&rgba_kernel_, scale_module_, "rgba_over_nv12"))) - throw Error("cuda_rect_overlay: cannot load scaling kernel"); -#endif - } - - /// Draw a packed RGB source onto the NV12 canvas: one fused scale + colour - /// conversion pass, alpha-blended over what is already there when the layer - /// asks for it and the source carries alpha. - void convertRgbLayer(CUstream stream, const AVFrame *src, AVFrame *dst, const LayerSpec &layer, - int step, int r_off, int g_off, int b_off, int a_off = -1) { - ensureScaleKernel(); - const bool blend = layer.blend && a_off >= 0; - if (!rgb_kernel_ || (blend && !rgba_kernel_)) - throw Error("cuda_rect_overlay: RGB conversion kernel unavailable"); - int sx = layer.crop_x, sy = layer.crop_y, sw = layer.crop_w, sh = layer.crop_h; - int dx = layer.dst_x, dy = layer.dst_y; - int dw = layer.dst_w > 0 ? layer.dst_w : layer.crop_w, dh = layer.dst_w > 0 ? layer.dst_h : layer.crop_h; - int cw = canvas_w_, ch = canvas_h_; - CUdeviceptr source = (CUdeviceptr)src->data[0], luma = (CUdeviceptr)dst->data[0], - chroma = (CUdeviceptr)dst->data[1]; - int source_pitch = src->linesize[0], luma_pitch = dst->linesize[0], chroma_pitch = dst->linesize[1]; - void *opaque_args[] = {&source, &source_pitch, &sx, &sy, &sw, &sh, &step, &r_off, &g_off, &b_off, - &luma, &luma_pitch, &chroma, &chroma_pitch, &dx, &dy, &dw, &dh, &cw, &ch}; - void *blend_args[] = {&source, &source_pitch, &sx, &sy, &sw, &sh, &step, &r_off, &g_off, &b_off, &a_off, - &luma, &luma_pitch, &chroma, &chroma_pitch, &dx, &dy, &dw, &dh, &cw, &ch}; - const int blocks_x = (dw + 1) / 2, blocks_y = (dh + 1) / 2; - if (CHECK_CU(cuLaunchKernel(blend ? rgba_kernel_ : rgb_kernel_, (blocks_x + 31) / 32, (blocks_y + 7) / 8, 1, - 32, 8, 1, 0, stream, blend ? blend_args : opaque_args, nullptr))) - throw Error("cuda_rect_overlay: RGB conversion launch failed"); - } - - void scaleLayer(CUstream stream, const AVFrame *src, AVFrame *dst, const LayerSpec &layer) { - const auto *desc = av_pix_fmt_desc_get(sw_fmt_); - for (int c = 0; c < desc->nb_components; ++c) - if (desc->comp[c].depth != 8) - throw Error("cuda_rect_overlay: scaling currently requires 8-bit components"); - ensureScaleKernel(); - if (!scale_kernel_) throw Error("cuda_rect_overlay: scaling kernel unavailable"); - for (int p = 0; p < numPlanes(sw_fmt_); ++p) { - if (!src->data[p]) continue; // Opaque input on an alpha canvas. - int lanes = 1; - for (int c = 0; c < desc->nb_components; ++c) - if (desc->comp[c].plane == p) lanes = std::max(lanes, desc->comp[c].step); - int sx, sy, sw, sh, dx, dy, dw, dh, cx, cy, cw, ch; - lumaRectToPlaneRegion(sw_fmt_, layer.crop_x, layer.crop_y, layer.crop_w, layer.crop_h, - p, sx, sy, sw, sh); - lumaRectToPlaneRegion(sw_fmt_, layer.dst_x, layer.dst_y, layer.dst_w, layer.dst_h, - p, dx, dy, dw, dh); - lumaRectToPlaneRegion(sw_fmt_, 0, 0, canvas_w_, canvas_h_, p, cx, cy, cw, ch); - sx /= lanes; sw /= lanes; dx /= lanes; dw /= lanes; cw /= lanes; - CUdeviceptr source = (CUdeviceptr)src->data[p], destination = (CUdeviceptr)dst->data[p]; - int source_pitch = src->linesize[p], destination_pitch = dst->linesize[p]; - void *args[] = {&source, &source_pitch, &sx, &sy, &sw, &sh, - &destination, &destination_pitch, &dx, &dy, &dw, &dh, &cw, &ch, &lanes}; - if (CHECK_CU(cuLaunchKernel(scale_kernel_, (dw + 31) / 32, (dh + 7) / 8, 1, - 32, 8, 1, 0, stream, args, nullptr))) - throw Error("cuda_rect_overlay: scaling launch failed"); - } - } - std::vector mergeLayersForTick(const av::VideoFrame *metadata_source) { std::vector layers; { @@ -562,21 +114,7 @@ class CudaRectOverlay : public NodeMultiInput, if (!e || !e->value) return layers; try { - Parameters md = Parameters::parse(e->value); - if (md.contains("layers") && md["layers"].is_array()) { - const auto &arr = md["layers"]; - for (size_t i = 0; i < arr.size() && i < layers.size(); ++i) { - if (!arr[i].is_object()) - continue; - parseLayerFromJson(arr[i], layers[i]); - } - } else { - for (size_t i = 0; i < layers.size(); ++i) { - const std::string k = std::to_string(i); - if (md.contains(k) && md[k].is_object()) - parseLayerFromJson(md[k], layers[i]); - } - } + avp::mixer::applyLayerMetadata(layers, e->value); } catch (const std::exception &e) { logstream << "cuda_rect_overlay: ignoring bad per-frame metadata: " << e.what(); } @@ -586,160 +124,37 @@ class CudaRectOverlay : public NodeMultiInput, bool hwSwFormatMatch(const av::VideoFrame &f) const { if (!f.raw() || !f.raw()->hw_frames_ctx || !f.raw()->hw_frames_ctx->data) return false; - AVHWFramesContext *ctx = (AVHWFramesContext *)f.raw()->hw_frames_ctx->data; - if (!ctx) return false; - return ctx->sw_format == sw_fmt_ || isAlphaCompatible(ctx->sw_format, sw_fmt_) || - isRgbToNv12Convertible(ctx->sw_format, sw_fmt_); - } - - static AVPixelFormat frameSwFormat(const av::VideoFrame &f) { - if (!f.raw() || !f.raw()->hw_frames_ctx || !f.raw()->hw_frames_ctx->data) - return AV_PIX_FMT_NONE; - AVHWFramesContext *ctx = (AVHWFramesContext *)f.raw()->hw_frames_ctx->data; - return ctx ? ctx->sw_format : AV_PIX_FMT_NONE; - } - - void clearCanvas(av::VideoFrame &outf, const AVFrame *color_src) { - if (!fillFrameBlack(sw_fmt_, outf.raw(), color_src)) - throw Error("cuda_rect_overlay: unsupported sw_format for canvas clear"); - } - - std::vector resolveDrawOps(const std::vector &sources, - const std::vector &layers) const { - std::vector ops; - ops.reserve(std::min(sources.size(), layers.size())); - for (size_t i = 0; i < sources.size() && i < layers.size(); ++i) { - const av::VideoFrame *srcp = sources[i]; - if (!srcp || !srcp->raw()) { - ops.push_back({}); - continue; - } - LayerSpec L = layers[i]; - if (L.dst_w > 0) { - const avp::compositor::Rect crop{L.crop_x, L.crop_y, L.crop_w, L.crop_h}; - const avp::compositor::Rect box{L.dst_x, L.dst_y, L.dst_w, L.dst_h}; - const int ax = chromaXAlign(sw_fmt_), ay = chromaYAlign(sw_fmt_); - auto placement = L.source_canvas_w > 0 - ? avp::compositor::placeInCanvas(srcp->width(), srcp->height(), crop, box, - L.source_canvas_w, L.source_canvas_h, ax, ay) - : avp::compositor::place(srcp->width(), srcp->height(), crop, box, L.fit, ax, ay); - if (!placement || placement->destination.x >= canvas_w_ || - placement->destination.y >= canvas_h_ || - int64_t(placement->destination.x) + placement->destination.w <= 0 || - int64_t(placement->destination.y) + placement->destination.h <= 0) { - DrawOp rejected; - rejected.src_w = -srcp->width(); rejected.src_h = srcp->height(); // marks "rejected" in the log - rejected.layer = L; - ops.push_back(rejected); - continue; - } - const auto &p = *placement; - L.crop_x = p.source.x; L.crop_y = p.source.y; - L.crop_w = p.source.w; L.crop_h = p.source.h; - L.dst_x = p.destination.x; L.dst_y = p.destination.y; - L.dst_w = p.destination.w; L.dst_h = p.destination.h; - ops.push_back({srcp, srcp->width(), srcp->height(), L}); - continue; - } - // 0 means "remaining source extent from the crop origin". - if (L.crop_w <= 0) L.crop_w = srcp->width() - L.crop_x; - if (L.crop_h <= 0) L.crop_h = srcp->height() - L.crop_y; - if (!clipRect(L.crop_x, L.crop_y, L.crop_w, L.crop_h, srcp->width(), srcp->height()) || - !clipRect(L.dst_x, L.dst_y, L.crop_w, L.crop_h, canvas_w_, canvas_h_)) { - ops.push_back({}); - continue; - } - const int ax = chromaXAlign(sw_fmt_); - const int ay = chromaYAlign(sw_fmt_); - L.crop_x = alignCoord(L.crop_x, ax); - L.crop_y = alignCoord(L.crop_y, ay); - L.dst_x = alignCoord(L.dst_x, ax); - L.dst_y = alignCoord(L.dst_y, ay); - if (!clipRect(L.crop_x, L.crop_y, L.crop_w, L.crop_h, srcp->width(), srcp->height()) || - !clipRect(L.dst_x, L.dst_y, L.crop_w, L.crop_h, canvas_w_, canvas_h_)) { - ops.push_back({}); - continue; - } - ops.push_back({srcp, srcp->width(), srcp->height(), L}); - } - // z decides who draws on top; equal z keeps source order (stable). - std::stable_sort(ops.begin(), ops.end(), - [](const DrawOp &a, const DrawOp &b) { return a.layer.z < b.layer.z; }); - return ops; + return avp::mixer::canvasAccepts(CudaRectDraw::frameSwFormat(f), sw_fmt_); } void processComposite(av::Timestamp pts, const std::vector &sources, const av::VideoFrame *metadata_src) { - ensureCudaDevice(); - CUstream stream = (CUstream)cuda_dev_->stream; + draw_.ensureDevice(); + CUstream stream = draw_.stream(); av::VideoFrame outf; int r = av_hwframe_get_buffer(out_frames_ref_, outf.raw(), 0); if (r < 0) throw Error(std::string("cuda_rect_overlay: av_hwframe_get_buffer failed: ") + av::error2string(r)); - av_buffer_unref(&outf.raw()->hw_frames_ctx); - outf.raw()->hw_frames_ctx = av_buffer_ref(out_frames_ref_); - outf.raw()->format = AV_PIX_FMT_CUDA; - outf.raw()->width = canvas_w_; - outf.raw()->height = canvas_h_; - outf.setComplete(true); + outf.setComplete(true); // av_hwframe_get_buffer set hw_frames_ctx, format and size std::vector layers = mergeLayersForTick(metadata_src); - std::vector ops = resolveDrawOps(sources, layers); + std::vector ops = avp::mixer::resolveDrawOps(sources, layers, canvas_w_, canvas_h_, sw_fmt_); { // Log the resolved layer set whenever it changes (scene switches), not per tick. - std::ostringstream desc; - for (size_t i = 0; i < ops.size(); ++i) { - const DrawOp &op = ops[i]; - const LayerSpec &L = op.layer; - if (!op.src) { - if (op.src_w < 0) - desc << " [" << i << ":REJECTED src " << -op.src_w << "x" << op.src_h << " crop " << L.crop_x - << "," << L.crop_y << " " << L.crop_w << "x" << L.crop_h << " box " << L.dst_x << "," - << L.dst_y << " " << L.dst_w << "x" << L.dst_h << "]"; - continue; - } - desc << " [" << i << ":" << op.src_w << "x" << op.src_h << " crop " << L.crop_x << "," << L.crop_y - << " " << L.crop_w << "x" << L.crop_h << " -> " << L.dst_x << "," << L.dst_y << " " - << L.dst_w << "x" << L.dst_h << " z" << L.z << "]"; - } - if (desc.str() != last_ops_desc_) { - last_ops_desc_ = desc.str(); + std::string desc = avp::mixer::describeDrawOps(ops); + if (desc != last_ops_desc_) { + last_ops_desc_ = std::move(desc); logstream << "cuda_rect_overlay layers:" << last_ops_desc_; } } - clearCanvas(outf, metadata_src ? metadata_src->raw() : nullptr); + setCanvasColor(outf.raw()); + draw_.clearCanvas(outf.raw(), hasCanvasColor() ? outf.raw() : (metadata_src ? metadata_src->raw() : nullptr)); for (const DrawOp &op : ops) { - const av::VideoFrame *srcp = op.src; - if (!srcp) + if (!op.src) continue; - const LayerSpec &L = op.layer; - - const bool sized = L.dst_w > 0; - const bool full_copy = !sized || (L.dst_w == L.crop_w && L.dst_h == L.crop_h && - L.dst_x >= 0 && L.dst_y >= 0 && L.dst_x + L.dst_w <= canvas_w_ && L.dst_y + L.dst_h <= canvas_h_); - const AVPixelFormat src_sw_fmt = frameSwFormat(*srcp); - int rgb_step, r_off, g_off, b_off; - if (src_sw_fmt != sw_fmt_ && sw_fmt_ == AV_PIX_FMT_NV12 && - isPackedRgb8(src_sw_fmt, rgb_step, r_off, g_off, b_off)) { - convertRgbLayer(stream, srcp->raw(), outf.raw(), L, rgb_step, r_off, g_off, b_off, - packedAlphaOffset(src_sw_fmt)); - } else if (!full_copy) { - scaleLayer(stream, srcp->raw(), outf.raw(), L); - } else if (!blitLayerPlanes(stream, sw_fmt_, srcp->raw(), outf.raw(), L.crop_x, L.crop_y, L.crop_w, - L.crop_h, L.dst_x, L.dst_y)) - throw Error("cuda_rect_overlay: GPU blit failed"); - - // When a non-alpha source is drawn onto an alpha canvas, fill the destination - // alpha rect with 255 (fully opaque) so the output alpha is well-defined. - if (src_sw_fmt != AV_PIX_FMT_NONE && src_sw_fmt != sw_fmt_) { - const int alpha_p = alphaPlaneIndex(sw_fmt_); - int x = L.dst_x, y = L.dst_y; - int w = sized ? L.dst_w : L.crop_w, h = sized ? L.dst_h : L.crop_h; - if (alpha_p >= 0 && clipRect(x, y, w, h, canvas_w_, canvas_h_)) - fillPlaneRect(sw_fmt_, outf.raw(), alpha_p, x, y, w, h, 255); - } + draw_.drawLayer(stream, *op.src, outf.raw(), op.layer); } if (metadata_src && metadata_src->raw()) { @@ -747,9 +162,13 @@ class CudaRectOverlay : public NodeMultiInput, if (cpy < 0) throw Error(std::string("cuda_rect_overlay: av_frame_copy_props failed: ") + av::error2string(cpy)); } + setCanvasColor(outf.raw()); + if (hasCanvasColor()) + av_frame_side_data_remove_by_props(&outf.raw()->side_data, &outf.raw()->nb_side_data, + AV_SIDE_DATA_PROP_COLOR_DEPENDENT); outf.setPts(pts); - if (CHECK_CU(cuStreamSynchronize(stream))) + if (AVP_CHECK_CU(cuStreamSynchronize(stream))) throw Error("cuda_rect_overlay: cuStreamSynchronize failed"); if (debug_log_every_n_ > 0 && (frame_counter_ % (uint64_t)debug_log_every_n_) == 0) @@ -765,14 +184,14 @@ class CudaRectOverlay : public NodeMultiInput, CudaRectOverlay(std::unique_ptr> &&sink, std::shared_ptr hw, int cw, int ch, AVPixelFormat sw_fmt, std::vector layers, std::string metadata_key, int dbg_n) : NodeSingleOutput(std::move(sink)), - hwaccel_(std::move(hw)), canvas_w_(cw), canvas_h_(ch), sw_fmt_(sw_fmt), + draw_(hw, CudaRectDraw::Canvas{cw, ch, sw_fmt}), default_layers_(std::move(layers)), metadata_key_(std::move(metadata_key)), debug_log_every_n_(dbg_n) { - out_frames_ref_ = av_hwframe_ctx_alloc(hwaccel_->deviceContext()); + out_frames_ref_ = av_hwframe_ctx_alloc(hw->deviceContext()); if (!out_frames_ref_) throw Error("cuda_rect_overlay: av_hwframe_ctx_alloc failed"); AVHWFramesContext *fc = (AVHWFramesContext *)out_frames_ref_->data; @@ -794,7 +213,7 @@ class CudaRectOverlay : public NodeMultiInput, ~CudaRectOverlay() override { freeHwContexts(); } void init(EdgeManager &edges, const Parameters ¶ms) override { - if (params.value("scale", false)) ensureScaleKernel(); + if (params.value("scale", false)) draw_.ensureKernels(); (void)edges; (void)params; NodeSingleOutput::init(edges, params); @@ -862,7 +281,7 @@ class CudaRectOverlay : public NodeMultiInput, throw Error("cuda_rect_overlay: input must be AV_PIX_FMT_CUDA"); if (!hwSwFormatMatch(*frame)) throw Error("cuda_rect_overlay: input hw sw_format mismatch node sw_format"); - playout_->push(i, *frame, rescaleTS(frame->pts(), {1, 1000000000}).timestamp()); + playout_->push(i, *frame, frame->pts().timestamp({1, 1000000000})); } source_edges_[i]->pop(); } @@ -892,8 +311,7 @@ class CudaRectOverlay : public NodeMultiInput, } // Warm inputs advance their bounded reference queues without allocating // an output surface or issuing any CUDA composition for an idle slot. - if (active) processComposite(av::Timestamp(decision->index, - {frame_rate_.getDenominator(), frame_rate_.getNumerator()}), sources, metadata); + if (active) processComposite(av::Timestamp(decision->index, av_inv_q(frame_rate_.getValue())), sources, metadata); playout_->commit(); if (debug_log_every_n_ > 0 && frame_counter_ % debug_log_every_n_ == 0) { std::ostringstream stats; @@ -1177,14 +595,14 @@ class CudaRectOverlay : public NodeMultiInput, for (auto &edge : source_edges_) edge->producedEvent().signal(); } else if (key == "warm_reset") { if (!playout_) throw Error("cuda_rect_overlay: warm_reset requires clocked playout"); - const auto period = avp::mixer::FrameRate(frame_rate_.getNumerator(), frame_rate_.getDenominator()).time(1); + const auto period = avp::mixer::TickGrid(frame_rate_).time(1); input_valid_from_ns_.store(avp::mixer::monotonicNs() - playout_->latencyNs() - period, std::memory_order_release); preserve_warm_input_.store(true, std::memory_order_release); input_generation_.fetch_add(1, std::memory_order_release); for (auto &edge : source_edges_) edge->producedEvent().signal(); } else if (key == "layers") { - auto new_layers = parseLayersArray(value); + auto new_layers = avp::mixer::parseLayersArray(value); std::lock_guard lock(layers_mutex_); default_layers_ = std::move(new_layers); } @@ -1206,14 +624,13 @@ class CudaRectOverlay : public NodeMultiInput, static std::shared_ptr create(NodeCreationInfo &nci); }; - std::shared_ptr CudaRectOverlay::create(NodeCreationInfo &nci) { EdgeManager &edges = nci.edges; const Parameters ¶ms = nci.params; auto src_names = jsonToStringList(params["src"]); if (src_names.empty()) throw Error("cuda_rect_overlay: at least one input required in src"); - std::vector layers = parseLayersParam(params); + std::vector layers = avp::mixer::parseLayersParam(params); if (layers.size() != src_names.size()) throw Error("cuda_rect_overlay: layers array length must match src count"); @@ -1240,6 +657,20 @@ std::shared_ptr CudaRectOverlay::create(NodeCreationInfo &nci) auto node = std::make_shared( make_unique>(out_edge), std::move(hw), cw, ch, sw_fmt, std::move(layers), mdkey, dbg); + if (params.contains("color")) { + const auto color = params.at("color").get(); + node->canvas_trc_ = color == "sdr" ? AVCOL_TRC_BT709 : color == "hlg" ? AVCOL_TRC_ARIB_STD_B67 : + color == "pq" ? AVCOL_TRC_SMPTE2084 : AVCOL_TRC_UNSPECIFIED; + if (!node->hasCanvasColor()) + throw Error("cuda_rect_overlay: color must be sdr, hlg or pq"); + if (node->canvas_trc_ != AVCOL_TRC_BT709 && av_pix_fmt_desc_get(sw_fmt)->comp[0].depth < 10) + throw Error("cuda_rect_overlay: HDR canvas requires 10-bit storage"); + const float sdr_white = params.value("sdr_white", 203.f); + const float hdr_peak = params.value("hdr_peak", 1000.f); + if (!(sdr_white >= 1.f && sdr_white <= hdr_peak && hdr_peak >= 100.f && hdr_peak <= 10000.f)) + throw Error("cuda_rect_overlay: invalid display white/peak"); + node->draw_.setColor(node->canvas_trc_, sdr_white, hdr_peak); + } node->createSourcesFromParameters(edges, params); out_edge->setProducer(node); node->initTimeline(nci); @@ -1251,8 +682,7 @@ std::shared_ptr CudaRectOverlay::create(NodeCreationInfo &nci) std::optional latency_ms; if (params.contains("latency_ms")) latency_ms = params.at("latency_ms").get(); node->playout_ = std::make_unique>( - src_names.size(), avp::mixer::FrameRate(node->frame_rate_.getNumerator(), - node->frame_rate_.getDenominator()), latency_ms, avp::mixer::TimestampMode::Presentation); + src_names.size(), avp::mixer::TickGrid(node->frame_rate_), latency_ms, avp::mixer::TimestampMode::Presentation); node->input_generation_.store(1); logstream << "cuda_rect_overlay: latency_ms=" << node->playout_->latencyNs() / 1000000.0; } diff --git a/src/nodes/hwaccel/cuda_rect_scale.cu b/src/nodes/hwaccel/cuda_rect_scale.cu index 95fcc4db..8c8ebe7f 100644 --- a/src/nodes/hwaccel/cuda_rect_scale.cu +++ b/src/nodes/hwaccel/cuda_rect_scale.cu @@ -1,12 +1,62 @@ #include +// Logical-sample access shared by every kernel here: a plane holds either +// bytes or little-endian 16-bit words whose meaningful bits sit above `shift` +// (P210/P010 store 10-bit codes as word >> 6; planar 10-bit uses shift 0). +__device__ __forceinline__ float load_sample(const unsigned char *p, int sample_bytes, int shift) { + return sample_bytes == 2 ? float(*(const unsigned short *)p >> shift) : float(*p); +} + +// Round-to-nearest store (+0.5 on a non-negative value); padding bits stay zero. +__device__ __forceinline__ void store_sample(unsigned char *p, int sample_bytes, int shift, float v) { + if (sample_bytes == 2) + *(unsigned short *)p = (unsigned short)((unsigned short)(v + 0.5f) << shift); + else + *p = (unsigned char)(v + 0.5f); +} + // Bilinear interpolation. This lightweight // sampler does not widen its support when downscaling. Each lane is sampled -// separately, including interleaved UV. +// separately, including interleaved UV. Coordinates count lane groups; pitches +// count bytes. extern "C" __global__ void scale_plane( const unsigned char *src, int src_pitch, int sx, int sy, int sw, int sh, unsigned char *dst, int dst_pitch, int dx, int dy, int dw, int dh, - int canvas_w, int canvas_h, int lanes) { + int canvas_w, int canvas_h, int lanes, int sample_bytes, int shift) { + const int ox = blockIdx.x * blockDim.x + threadIdx.x; + const int oy = blockIdx.y * blockDim.y + threadIdx.y; + const int x = dx + ox, y = dy + oy; + if (ox >= dw || oy >= dh || x < 0 || y < 0 || x >= canvas_w || y >= canvas_h) return; + const float fx = (ox + 0.5f) * sw / dw - 0.5f; + const float fy = (oy + 0.5f) * sh / dh - 0.5f; + const int ix = int(floorf(fx)), iy = int(floorf(fy)); + const float tx = fx - ix, ty = fy - iy; + const int x0 = sx + max(0, min(ix, sw - 1)); + const int x1 = sx + max(0, min(ix + 1, sw - 1)); + const int y0 = sy + max(0, min(iy, sh - 1)); + const int y1 = sy + max(0, min(iy + 1, sh - 1)); + for (int c = 0; c < lanes; ++c) { + const float a = load_sample(src + y0 * src_pitch + (x0 * lanes + c) * sample_bytes, sample_bytes, shift); + const float b = load_sample(src + y0 * src_pitch + (x1 * lanes + c) * sample_bytes, sample_bytes, shift); + const float d = load_sample(src + y1 * src_pitch + (x0 * lanes + c) * sample_bytes, sample_bytes, shift); + const float e = load_sample(src + y1 * src_pitch + (x1 * lanes + c) * sample_bytes, sample_bytes, shift); + const float top = a + tx * (b - a), bottom = d + tx * (e - d); + store_sample(dst + y * dst_pitch + (x * lanes + c) * sample_bytes, sample_bytes, shift, + top + ty * (bottom - top)); + } +} + +// Cross-depth plane scaler: read an 8-bit (or narrower) source plane, bilinear +// scale it into the destination region and store at the canvas depth, scaling +// logical codes by `mul` (4 for 8->10-bit SDR promotion, 16->64 / 235->940). +// Separate src/dst sample bytes and shifts let one body cover NV12->P210 luma +// and interleaved chroma; the chroma footprint difference (4:2:0 -> 4:2:2) is +// just the src/dst region sizes, so the same bilinear handles the resample. +extern "C" __global__ void convert_scale_plane( + const unsigned char *src, int src_pitch, int sx, int sy, int sw, int sh, + int src_bytes, int src_shift, + unsigned char *dst, int dst_pitch, int dx, int dy, int dw, int dh, + int canvas_w, int canvas_h, int lanes, int dst_bytes, int dst_shift, float mul) { const int ox = blockIdx.x * blockDim.x + threadIdx.x; const int oy = blockIdx.y * blockDim.y + threadIdx.y; const int x = dx + ox, y = dy + oy; @@ -20,19 +70,23 @@ extern "C" __global__ void scale_plane( const int y0 = sy + max(0, min(iy, sh - 1)); const int y1 = sy + max(0, min(iy + 1, sh - 1)); for (int c = 0; c < lanes; ++c) { - const float a = src[y0 * src_pitch + x0 * lanes + c]; - const float b = src[y0 * src_pitch + x1 * lanes + c]; - const float d = src[y1 * src_pitch + x0 * lanes + c]; - const float e = src[y1 * src_pitch + x1 * lanes + c]; + const float a = load_sample(src + y0 * src_pitch + (x0 * lanes + c) * src_bytes, src_bytes, src_shift); + const float b = load_sample(src + y0 * src_pitch + (x1 * lanes + c) * src_bytes, src_bytes, src_shift); + const float d = load_sample(src + y1 * src_pitch + (x0 * lanes + c) * src_bytes, src_bytes, src_shift); + const float e = load_sample(src + y1 * src_pitch + (x1 * lanes + c) * src_bytes, src_bytes, src_shift); const float top = a + tx * (b - a), bottom = d + tx * (e - d); - dst[y * dst_pitch + x * lanes + c] = (unsigned char)(top + ty * (bottom - top) + 0.5f); + store_sample(dst + y * dst_pitch + (x * lanes + c) * dst_bytes, dst_bytes, dst_shift, + (top + ty * (bottom - top)) * mul); } } -// Packed 8-bit RGB source (any channel order, 3 or 4 bytes per pixel) onto an -// NV12 canvas in one pass: each thread produces one 2x2 luma block and its -// interleaved Cb/Cr pair (BT.709 limited range), sampling RGB bilinearly like -// scale_plane. dst_y/dst_uv are the canvas planes at their own pitches. +// Packed 8-bit RGB source (any channel order, 3 or 4 bytes per pixel) onto a +// semiplanar YUV canvas in one pass: each thread produces one chroma-footprint +// block of luma (2x2 for NV12, 2x1 for P210, 1x1 for 444) and its interleaved +// Cb/Cr pair (BT.709 limited range), sampling RGB bilinearly like scale_plane. +// The 8-bit matrix result is scaled by dst_scale on deeper canvases, so SDR +// graphics promote exactly like SDR video (16 -> 64); this path stays SDR. +// dst_y/dst_uv are the canvas planes at their own pitches. __device__ __forceinline__ void sample_rgb( const unsigned char *src, int src_pitch, int sx, int sy, int sw, int sh, int step, int r_off, int g_off, int b_off, float fx, float fy, float *rgb) { @@ -50,57 +104,113 @@ __device__ __forceinline__ void sample_rgb( } } -extern "C" __global__ void rgb_to_nv12( +// Embed SDR RGB graphics at the same display white as tonemap_cuda. Transfer +// codes match that filter: HLG=0, PQ=1, SDR=2. Alpha stays separate. +__device__ static inline float pq_code(float nits) { + const float p = powf(fmaxf(nits, 0.f) / 10000.f, 0.1593017578125f); + return powf((0.8359375f + 18.8515625f * p) / (1.f + 18.6875f * p), 78.84375f); +} + +__device__ static inline void convert_graphic_rgb(float *rgb, int transfer, float white, float peak) { + if (transfer == 2) return; + float r = powf(fmaxf(rgb[0] / 255.f, 0.f), 2.4f) * white; + float g = powf(fmaxf(rgb[1] / 255.f, 0.f), 2.4f) * white; + float b = powf(fmaxf(rgb[2] / 255.f, 0.f), 2.4f) * white; + rgb[0] = 0.627404f * r + 0.329283f * g + 0.043313f * b; + rgb[1] = 0.069097f * r + 0.919540f * g + 0.011362f * b; + rgb[2] = 0.016391f * r + 0.088013f * g + 0.895595f * b; + if (transfer == 1) { + for (int i = 0; i < 3; ++i) rgb[i] = 255.f * pq_code(rgb[i]); + return; + } + const float luma = 0.2627f * rgb[0] + 0.6780f * rgb[1] + 0.0593f * rgb[2]; + const float gamma = fmaxf(1.f, 1.2f + 0.42f * log10f(peak / 1000.f)); + const float gain = luma > 0.f ? 12.f * powf(luma / peak, 1.f / gamma) / luma : 0.f; + for (int i = 0; i < 3; ++i) { + const float c = fmaxf(rgb[i] * gain, 0.f); + rgb[i] = 255.f * (c <= 1.f ? 0.5f * sqrtf(c) : 0.17883277f * logf(c - 0.28466892f) + 0.55991073f); + } +} + +__device__ static inline float graphic_luma(const float *rgb, int transfer) { + return transfer == 2 ? 16.f + 0.1826f * rgb[0] + 0.6142f * rgb[1] + 0.0620f * rgb[2] + : 16.f + (219.f / 255.f) * (0.2627f * rgb[0] + 0.6780f * rgb[1] + 0.0593f * rgb[2]); +} + +__device__ static inline void graphic_chroma(float r, float g, float b, int transfer, float &cb, float &cr) { + if (transfer == 2) { + cb = 128.f - 0.1006f * r - 0.3386f * g + 0.4392f * b; + cr = 128.f + 0.4392f * r - 0.3989f * g - 0.0403f * b; + } else { + const float y = 0.2627f * r + 0.6780f * g + 0.0593f * b; + cb = 128.f + (224.f / 255.f) * (b - y) / 1.8814f; + cr = 128.f + (224.f / 255.f) * (r - y) / 1.4746f; + } +} + +extern "C" __global__ void rgb_to_yuv( const unsigned char *src, int src_pitch, int sx, int sy, int sw, int sh, int step, int r_off, int g_off, int b_off, unsigned char *dst_y, int y_pitch, unsigned char *dst_uv, int uv_pitch, - int dx, int dy, int dw, int dh, int canvas_w, int canvas_h) { - const int ox = (blockIdx.x * blockDim.x + threadIdx.x) * 2; - const int oy = (blockIdx.y * blockDim.y + threadIdx.y) * 2; + int dx, int dy, int dw, int dh, int canvas_w, int canvas_h, + int dst_sb, int dst_shift, int dst_scale, int sub_x, int sub_y, + int transfer = 2, float white = 203.f, float peak = 1000.f) { + const int bw = 1 << sub_x, bh = 1 << sub_y; + const int ox = (blockIdx.x * blockDim.x + threadIdx.x) * bw; + const int oy = (blockIdx.y * blockDim.y + threadIdx.y) * bh; const int x = dx + ox, y = dy + oy; if (ox >= dw || oy >= dh || x < 0 || y < 0 || x >= canvas_w || y >= canvas_h) return; const float xs = float(sw) / dw, ys = float(sh) / dh; + const float maxv = 256.f * dst_scale - 1.f; float sum[3] = {0.f, 0.f, 0.f}; int n = 0; - for (int j = 0; j < 2; ++j) { - for (int i = 0; i < 2; ++i) { + for (int j = 0; j < bh; ++j) { + for (int i = 0; i < bw; ++i) { if (ox + i >= dw || oy + j >= dh || x + i >= canvas_w || y + j >= canvas_h) continue; float rgb[3]; sample_rgb(src, src_pitch, sx, sy, sw, sh, step, r_off, g_off, b_off, (ox + i + 0.5f) * xs - 0.5f, (oy + j + 0.5f) * ys - 0.5f, rgb); - const float luma = 16.f + 0.1826f * rgb[0] + 0.6142f * rgb[1] + 0.0620f * rgb[2]; - dst_y[(y + j) * y_pitch + x + i] = (unsigned char)min(max(luma + 0.5f, 0.f), 255.f); + convert_graphic_rgb(rgb, transfer, white, peak); + const float luma = graphic_luma(rgb, transfer) * dst_scale; + store_sample(dst_y + (y + j) * y_pitch + (x + i) * dst_sb, dst_sb, dst_shift, + min(max(luma, 0.f), maxv)); sum[0] += rgb[0]; sum[1] += rgb[1]; sum[2] += rgb[2]; ++n; } } const float r = sum[0] / n, g = sum[1] / n, b = sum[2] / n; - const float cb = 128.f - 0.1006f * r - 0.3386f * g + 0.4392f * b; - const float cr = 128.f + 0.4392f * r - 0.3989f * g - 0.0403f * b; - unsigned char *uv = dst_uv + (y / 2) * uv_pitch + (x / 2) * 2; - uv[0] = (unsigned char)min(max(cb + 0.5f, 0.f), 255.f); - uv[1] = (unsigned char)min(max(cr + 0.5f, 0.f), 255.f); + float cb, cr; + graphic_chroma(r, g, b, transfer, cb, cr); + cb *= dst_scale; cr *= dst_scale; + unsigned char *uv = dst_uv + (y >> sub_y) * uv_pitch + (x >> sub_x) * 2 * dst_sb; + store_sample(uv, dst_sb, dst_shift, min(max(cb, 0.f), maxv)); + store_sample(uv + dst_sb, dst_sb, dst_shift, min(max(cr, 0.f), maxv)); } -// Packed 8-bit RGBA over an existing NV12 canvas, BT.709 limited range: the -// source's own alpha decides how much of it survives, so a media wipe or a -// transparent graphic composites in one pass with no separate overlay filter, -// no format round trip and no CPU resize. Same 2x2 block ownership as -// rgb_to_nv12, so chroma is read-modify-written exactly once per block. -extern "C" __global__ void rgba_over_nv12( +// Packed 8-bit RGBA over an existing semiplanar YUV canvas, BT.709 limited +// range: the source's own alpha decides how much of it survives, so a media +// wipe or a transparent graphic composites in one pass with no separate +// overlay filter, no format round trip and no CPU resize. Same chroma-block +// ownership as rgb_to_yuv, so chroma is read-modify-written exactly once per +// block. +extern "C" __global__ void rgba_over_yuv( const unsigned char *src, int src_pitch, int sx, int sy, int sw, int sh, int step, int r_off, int g_off, int b_off, int a_off, unsigned char *dst_y, int y_pitch, unsigned char *dst_uv, int uv_pitch, - int dx, int dy, int dw, int dh, int canvas_w, int canvas_h) { - const int ox = (blockIdx.x * blockDim.x + threadIdx.x) * 2; - const int oy = (blockIdx.y * blockDim.y + threadIdx.y) * 2; + int dx, int dy, int dw, int dh, int canvas_w, int canvas_h, + int dst_sb, int dst_shift, int dst_scale, int sub_x, int sub_y, + int transfer = 2, float white = 203.f, float peak = 1000.f) { + const int bw = 1 << sub_x, bh = 1 << sub_y; + const int ox = (blockIdx.x * blockDim.x + threadIdx.x) * bw; + const int oy = (blockIdx.y * blockDim.y + threadIdx.y) * bh; const int x = dx + ox, y = dy + oy; if (ox >= dw || oy >= dh || x < 0 || y < 0 || x >= canvas_w || y >= canvas_h) return; const float xs = float(sw) / dw, ys = float(sh) / dh; + const float maxv = 256.f * dst_scale - 1.f; float sum[3] = {0.f, 0.f, 0.f}, sum_a = 0.f; int n = 0; - for (int j = 0; j < 2; ++j) { - for (int i = 0; i < 2; ++i) { + for (int j = 0; j < bh; ++j) { + for (int i = 0; i < bw; ++i) { if (ox + i >= dw || oy + j >= dh || x + i >= canvas_w || y + j >= canvas_h) continue; float rgb[3]; const float fx = (ox + i + 0.5f) * xs - 0.5f, fy = (oy + j + 0.5f) * ys - 0.5f; @@ -114,9 +224,11 @@ extern "C" __global__ void rgba_over_nv12( const float a10 = src[y1 * src_pitch + x0 * step + a_off], a11 = src[y1 * src_pitch + x1 * step + a_off]; const float a = ((a00 + tx * (a01 - a00)) + ty * ((a10 + tx * (a11 - a10)) - (a00 + tx * (a01 - a00)))) / 255.f; - const float luma = 16.f + 0.1826f * rgb[0] + 0.6142f * rgb[1] + 0.0620f * rgb[2]; - unsigned char *py = dst_y + (y + j) * y_pitch + x + i; - *py = (unsigned char)min(max(a * luma + (1.f - a) * float(*py) + 0.5f, 0.f), 255.f); + convert_graphic_rgb(rgb, transfer, white, peak); + const float luma = graphic_luma(rgb, transfer) * dst_scale; + unsigned char *py = dst_y + (y + j) * y_pitch + (x + i) * dst_sb; + store_sample(py, dst_sb, dst_shift, + min(max(a * luma + (1.f - a) * load_sample(py, dst_sb, dst_shift), 0.f), maxv)); sum[0] += a * rgb[0]; sum[1] += a * rgb[1]; sum[2] += a * rgb[2]; sum_a += a; ++n; @@ -126,9 +238,12 @@ extern "C" __global__ void rgba_over_nv12( // Premultiplied average keeps a transparent corner of the block from // dragging the blended chroma toward black. const float r = sum[0] / sum_a, g = sum[1] / sum_a, b = sum[2] / sum_a, a = sum_a / n; - const float cb = 128.f - 0.1006f * r - 0.3386f * g + 0.4392f * b; - const float cr = 128.f + 0.4392f * r - 0.3989f * g - 0.0403f * b; - unsigned char *uv = dst_uv + (y / 2) * uv_pitch + (x / 2) * 2; - uv[0] = (unsigned char)min(max(a * cb + (1.f - a) * float(uv[0]) + 0.5f, 0.f), 255.f); - uv[1] = (unsigned char)min(max(a * cr + (1.f - a) * float(uv[1]) + 0.5f, 0.f), 255.f); + float cb, cr; + graphic_chroma(r, g, b, transfer, cb, cr); + cb *= dst_scale; cr *= dst_scale; + unsigned char *uv = dst_uv + (y >> sub_y) * uv_pitch + (x >> sub_x) * 2 * dst_sb; + store_sample(uv, dst_sb, dst_shift, + min(max(a * cb + (1.f - a) * load_sample(uv, dst_sb, dst_shift), 0.f), maxv)); + store_sample(uv + dst_sb, dst_sb, dst_shift, + min(max(a * cr + (1.f - a) * load_sample(uv + dst_sb, dst_sb, dst_shift), 0.f), maxv)); } diff --git a/src/nodes/hwaccel/egl_image_cuda_overlay.cpp b/src/nodes/hwaccel/egl_image_cuda_overlay.cpp index 5f502b86..8627c1d6 100644 --- a/src/nodes/hwaccel/egl_image_cuda_overlay.cpp +++ b/src/nodes/hwaccel/egl_image_cuda_overlay.cpp @@ -3,7 +3,7 @@ #include "../../hwaccel.hpp" #include "../../hwaccel/EglImageFrame.hpp" #include "../../mixer/Playout.hpp" -#include "../../mixer/MonotonicClock.hpp" +#include "../../mixer/primitives/MonotonicClock.hpp" #include "../../../deps/cuda_loader/cuda_drvapi_dynlink_gl.h" extern "C" { @@ -669,8 +669,7 @@ class EglImageCudaOverlay : public NodeMultiInput, if (params.contains("latency_ms")) latency_ms = params.at("latency_ms").get(); node->playout_ = std::make_unique>( - source_names.size(), avp::mixer::FrameRate( - node->frame_rate_.getNumerator(), node->frame_rate_.getDenominator()), latency_ms); + source_names.size(), avp::mixer::TickGrid(node->frame_rate_), latency_ms); logstream << "egl_image_cuda_overlay: latency_ms=" << node->playout_->latencyNs() / 1000000.0; const double ttl_seconds = params.value("cache_ttl", 3.0); diff --git a/src/nodes/hwaccel/v210_to_cuda.cpp b/src/nodes/hwaccel/v210_to_cuda.cpp new file mode 100644 index 00000000..d01e3ae2 --- /dev/null +++ b/src/nodes/hwaccel/v210_to_cuda.cpp @@ -0,0 +1,229 @@ +#include "../node_common.hpp" +#include "../../cuda.hpp" +#include "../../hwaccel.hpp" + +extern "C" { +#include +#include +#include +} + +#include +#include +#include "../../../objs/src/nodes/hwaccel/v210_unpack.ptx.h" + +namespace { + +void checkCuda(CUresult result, const char* operation) { + if (result == CUDA_SUCCESS) return; + const char* description = nullptr; + if (cuGetErrorString) cuGetErrorString(result, &description); + throw Error(std::string("v210_to_cuda: ") + operation + ": " + + (description ? description : std::to_string(result))); +} + +class CurrentContext { +public: + explicit CurrentContext(CUcontext context) { checkCuda(cuCtxPushCurrent(context), "push context"); } + ~CurrentContext() { + CUcontext previous; + CHECK_CU(cuCtxPopCurrent(&previous)); + } + CurrentContext(const CurrentContext&) = delete; + CurrentContext& operator=(const CurrentContext&) = delete; +}; + +struct FramePoolDeleter { + void operator()(AVBufferRef* ref) const { av_buffer_unref(&ref); } +}; + +// A private stream and bounded staging allocation isolate this node's work. +// Destruction also covers partial initialization and failed kernel launches. +struct UploadResources { + CUcontext context = nullptr; + CUstream stream = nullptr; + CUmodule module = nullptr; + CUfunction kernel = nullptr; + CUdeviceptr packed = 0; + void* staging = nullptr; + std::unique_ptr frames; + + ~UploadResources() { + if (!context) return; + if (CHECK_CU(cuCtxPushCurrent(context))) return; + if (stream) CHECK_CU(cuStreamSynchronize(stream)); + frames.reset(); + if (packed) CHECK_CU(cuMemFree(packed)); + if (staging) CHECK_CU(cuMemFreeHost(staging)); + if (module) CHECK_CU(cuModuleUnload(module)); + if (stream) CHECK_CU(cuStreamDestroy(stream)); + CUcontext previous; + CHECK_CU(cuCtxPopCurrent(&previous)); + } +}; + +av::Rational positiveRatio(const std::string& value) { + auto ratio = parseRatio(value); + if (ratio.getNumerator() <= 0 || ratio.getDenominator() <= 0) + throw Error("v210_to_cuda: ratios must be positive"); + return ratio; +} + +int colorOption(const Parameters& params, const char* name, int fallback, + int (*parse)(const char*)) { + if (!params.count(name)) return fallback; + const auto value = params.at(name).get(); + const int result = parse(value.c_str()); + if (result < 0) throw Error(std::string("v210_to_cuda: invalid ") + name + ": " + value); + return result; +} + +} // namespace + +class V210ToCuda : public NodeSISO, public ReportsFinishByFlag, + public IVideoFormatSource, public IFrameRateSource, public ITimeBaseSource { + int width_, height_, stride_; + size_t packed_size_; + AVPixelFormat format_; + av::Rational fps_, timebase_, aspect_; + AVColorRange range_; + AVColorSpace matrix_; + AVColorPrimaries primaries_; + AVColorTransferCharacteristic transfer_; + AVChromaLocation chroma_; + // Keep the FFmpeg device alive until all CUDA resources have been released. + std::shared_ptr device_; + UploadResources gpu_; + + void initialize() { + if (global_cuda.has_errors || !device_ || device_->hardwarePixelFormat() != AV_PIX_FMT_CUDA) + throw Error("v210_to_cuda: requires an initialized CUDA hwaccel device"); + auto* device = reinterpret_cast(device_->deviceContext()->data); + gpu_.context = static_cast(device->hwctx)->cuda_ctx; + CurrentContext context(gpu_.context); + + gpu_.frames.reset(av_hwframe_ctx_alloc(device_->deviceContext())); + if (!gpu_.frames) throw Error("v210_to_cuda: cannot allocate CUDA frame pool"); + auto* frames = reinterpret_cast(gpu_.frames->data); + frames->format = AV_PIX_FMT_CUDA; + frames->sw_format = format_; + frames->width = width_; + frames->height = height_; + const int result = av_hwframe_ctx_init(gpu_.frames.get()); + if (result < 0) + throw Error("v210_to_cuda: CUDA output format requires FFmpeg 8.1 support: " + + av::error2string(result)); + + // CU_STREAM_NON_BLOCKING is absent from the bundled dynlink declarations. + constexpr unsigned kNonBlockingStream = 0x1; + checkCuda(cuStreamCreate(&gpu_.stream, kNonBlockingStream), "create stream"); + checkCuda(cuMemAlloc(&gpu_.packed, packed_size_), "allocate packed buffer"); + checkCuda(cuMemHostAlloc(&gpu_.staging, packed_size_, 0), "allocate upload staging"); + const std::string module(avpl_v210_unpack_ptx, + avpl_v210_unpack_ptx + avpl_v210_unpack_ptx_len); + checkCuda(cuModuleLoadDataEx(&gpu_.module, module.c_str(), 0, nullptr, nullptr), "load kernel"); + checkCuda(cuModuleGetFunction(&gpu_.kernel, gpu_.module, "unpack_v210"), "find kernel"); + } + +public: + V210ToCuda(std::unique_ptr&& source, std::unique_ptr&& sink, + const Parameters& params, std::shared_ptr device) + : NodeSISO(std::move(source), std::move(sink)), + width_(params.at("width").get()), height_(params.at("height").get()), + fps_(positiveRatio(params.at("fps"))), + timebase_(params.count("timebase") ? positiveRatio(params.at("timebase")) + : av::Rational(fps_.getDenominator(), fps_.getNumerator())), + aspect_(positiveRatio(params.value("sample_aspect_ratio", std::string("1/1")))), + range_(static_cast(colorOption(params, "color_range", AVCOL_RANGE_UNSPECIFIED, av_color_range_from_name))), + matrix_(static_cast(colorOption(params, "colorspace", AVCOL_SPC_UNSPECIFIED, av_color_space_from_name))), + primaries_(static_cast(colorOption(params, "color_primaries", AVCOL_PRI_UNSPECIFIED, av_color_primaries_from_name))), + transfer_(static_cast(colorOption(params, "color_trc", AVCOL_TRC_UNSPECIFIED, av_color_transfer_from_name))), + chroma_(static_cast(colorOption(params, "chroma_location", AVCHROMA_LOC_UNSPECIFIED, av_chroma_location_from_name))), + device_(std::move(device)) { + if (width_ <= 0 || height_ <= 0 || width_ % 2 || + av_image_check_size(width_, height_, 0, nullptr) < 0) + throw Error("v210_to_cuda: requires valid dimensions and an even width"); + const int64_t minimum_stride = ((int64_t(width_) * 2 + 2) / 3) * 4; + const int64_t stride = params.value("stride", ((int64_t(width_) + 47) / 48) * 128); + if (stride < minimum_stride || stride % 4 || + stride > std::numeric_limits::max() / height_) + throw Error("v210_to_cuda: stride must be a multiple of four, fit a v210 row and an AVPacket"); + stride_ = static_cast(stride); + packed_size_ = size_t(stride_) * height_; + const auto format = params.value("sw_format", std::string("p210le")); + if (format != "p210le" && format != "yuv422p10le") + throw Error("v210_to_cuda: sw_format must be p210le or yuv422p10le"); + format_ = av_get_pix_fmt(format.c_str()); + if (format_ == AV_PIX_FMT_NONE) throw Error("v210_to_cuda: output pixel format unavailable"); + } + + void process() override { + av::Packet packet = source_->get(); + if (packet.isNull()) return; // Input queue interrupted during shutdown. + if (isEofMarker(packet)) { + onEofConsumed(); + markFinished(); + return; + } + if (packet.size() != packed_size_ || !packet.data()) + throw Error("v210_to_cuda: expected exactly one stride * height packed frame per packet"); + if (!packet.pts().isValid() || packet.timeBase().getNumerator() <= 0 || + packet.timeBase().getDenominator() <= 0) + throw Error("v210_to_cuda: input packet needs valid PTS and a positive time base"); + + av::VideoFrame output; + { + CurrentContext context(gpu_.context); + const int result = av_hwframe_get_buffer(gpu_.frames.get(), output.raw(), 0); + if (result < 0) throw Error("v210_to_cuda: cannot allocate output: " + av::error2string(result)); + // Copy bytes only. Unpacking is entirely on the GPU; staging allows + // arbitrary AVPacket/MXL host buffers without registering their pages. + std::memcpy(gpu_.staging, packet.data(), packed_size_); + auto* frame = output.raw(); + int semiplanar = format_ == AV_PIX_FMT_P210LE; + void* args[] = {&gpu_.packed, &stride_, &width_, &height_, + &frame->data[0], &frame->linesize[0], + &frame->data[1], &frame->linesize[1], + &frame->data[2], &frame->linesize[2], &semiplanar}; + try { + checkCuda(cuMemcpyHtoDAsync(gpu_.packed, gpu_.staging, packed_size_, gpu_.stream), "upload"); + checkCuda(cuLaunchKernel(gpu_.kernel, (width_ / 2 + 31) / 32, (height_ + 7) / 8, 1, + 32, 8, 1, 0, gpu_.stream, args, nullptr), "unpack"); + checkCuda(cuStreamSynchronize(gpu_.stream), "complete frame"); + } catch (...) { + // Do not recycle output/staging while a queued operation uses it. + CHECK_CU(cuStreamSynchronize(gpu_.stream)); + throw; + } + } + output.setTimeBase(timebase_); + output.raw()->time_base = timebase_; + output.raw()->pts = rescaleTS(packet.pts(), timebase_).timestamp(); + output.raw()->pkt_dts = rescaleTS(packet.dts(), timebase_).timestamp(); + output.raw()->duration = av_rescale_q(packet.raw()->duration, packet.timeBase(), timebase_); + output.raw()->sample_aspect_ratio = aspect_; + output.raw()->color_range = range_; + output.raw()->colorspace = matrix_; + output.raw()->color_primaries = primaries_; + output.raw()->color_trc = transfer_; + output.raw()->chroma_location = chroma_; + output.setComplete(true); + sink_->put(output); + } + + int width() override { return width_; } + int height() override { return height_; } + av::PixelFormat pixelFormat() override { return av::PixelFormat(AV_PIX_FMT_CUDA); } + av::PixelFormat realPixelFormat() override { return av::PixelFormat(format_); } + av::Rational frameRate() override { return fps_; } + av::Rational timeBase() override { return timebase_; } + + static std::shared_ptr create(NodeCreationInfo& nci) { + auto device = InstanceSharedObjects::get(nci.instance, nci.params.at("hwaccel")); + auto node = createCommon(nci.edges, nci.params, nci.params, device); + node->initialize(); + return node; + } +}; + +DECLNODE(v210_to_cuda, V210ToCuda) diff --git a/src/nodes/hwaccel/v210_unpack.cu b/src/nodes/hwaccel/v210_unpack.cu new file mode 100644 index 00000000..66d375b4 --- /dev/null +++ b/src/nodes/hwaccel/v210_unpack.cu @@ -0,0 +1,38 @@ +#include +#include + +// One thread owns a 4:2:2 pixel pair. A pair's U/Y/V/Y samples span two +// little-endian v210 words, each containing three samples and two unused bits. +extern "C" __global__ void unpack_v210( + const uint8_t* input, int input_pitch, int width, int height, + uint8_t* output_y, int y_pitch, uint8_t* output_u, int u_pitch, + uint8_t* output_v, int v_pitch, int semiplanar) +{ + const int pair = blockIdx.x * blockDim.x + threadIdx.x; + const int row = blockIdx.y * blockDim.y + threadIdx.y; + if (pair >= width / 2 || row >= height) return; + + const uint32_t* packed = reinterpret_cast( + input + static_cast(row) * input_pitch); + const int sample = pair * 4; + const int word = sample / 3; + const uint64_t bits = ((static_cast(packed[word + 1] & 0x3fffffff) << 30) + | (packed[word] & 0x3fffffff)) >> ((sample % 3) * 10); + const unsigned shift = semiplanar ? 6 : 0; + const uint16_t u = (bits & 1023) << shift; + const uint16_t y0 = ((bits >> 10) & 1023) << shift; + const uint16_t v = ((bits >> 20) & 1023) << shift; + const uint16_t y1 = ((bits >> 30) & 1023) << shift; + + uint16_t* y = reinterpret_cast(output_y + static_cast(row) * y_pitch); + uint16_t* uv = reinterpret_cast(output_u + static_cast(row) * u_pitch); + y[pair * 2] = y0; + y[pair * 2 + 1] = y1; + if (semiplanar) { + uv[pair * 2] = u; + uv[pair * 2 + 1] = v; + } else { + uv[pair] = u; + reinterpret_cast(output_v + static_cast(row) * v_pitch)[pair] = v; + } +} diff --git a/src/nodes/mixer_snapshot.cpp b/src/nodes/mixer_snapshot.cpp index 1681b87a..89928c9f 100644 --- a/src/nodes/mixer_snapshot.cpp +++ b/src/nodes/mixer_snapshot.cpp @@ -1,8 +1,8 @@ #include "node_common.hpp" -#include -#include "../mixer/OutputSnapshot.hpp" -#include "../mixer/FrameRate.hpp" -#include "../mixer/MonotonicClock.hpp" +#include "../mixer/primitives/CutLatencyProbe.hpp" +#include "../mixer/primitives/OutputSnapshot.hpp" +#include "../mixer/primitives/TickGrid.hpp" +#include "../mixer/primitives/MonotonicClock.hpp" // The output instance records committed pictures; the two slot instances can // substitute that same retained frame without a GPU copy or another compositor. @@ -11,17 +11,17 @@ class MixerSnapshot : public NodeSISO, std::shared_ptr state_; int slot_; av::Rational fps_; - avp::mixer::FrameRate rate_; + avp::mixer::TickGrid rate_; int64_t latency_ns_; av::Timestamp last_pts_ = NOTS; - static constexpr const char* marker = "avp.mixer.snapshot"; + static constexpr const char* kMarker = "avp.mixer.snapshot"; public: MixerSnapshot(std::unique_ptr&& source, std::unique_ptr&& sink, std::shared_ptr state, int slot, av::Rational fps, int64_t latency_ns) : NodeSISO(std::move(source), std::move(sink)), state_(std::move(state)), - slot_(slot), fps_(fps), rate_(fps.getNumerator(), fps.getDenominator()), + slot_(slot), fps_(fps), rate_(fps), latency_ns_(latency_ns) { if (slot_ == -1) { std::lock_guard lock(state_->mutex); @@ -35,7 +35,7 @@ class MixerSnapshot : public NodeSISO, } } av::Rational frameRate() override { return fps_; } - av::Rational timeBase() override { return {fps_.getDenominator(), fps_.getNumerator()}; } + av::Rational timeBase() override { return av_inv_q(fps_.getValue()); } void process() override { // A short input wait also wakes a newly requested hold on an idle slot. @@ -45,8 +45,8 @@ class MixerSnapshot : public NodeSISO, bool replace = slot_ == -1 ? frames.holding() : frames.replaces(slot_); bool release = false; if (slot_ == -1 && replace && input && input->isValid() && input->pts().isValid()) { - const auto* tag = av_dict_get(input->raw()->metadata, marker, nullptr, 0); - uint64_t generation = tag ? std::strtoull(tag->value, nullptr, 10) : 0; + const auto* tag = av_dict_get(input->raw()->metadata, kMarker, nullptr, 0); + const uint64_t generation = tag ? avp::mixer::parseFrameToken(tag->value) : 0; release = frames.canRelease(input->pts().timestamp({1, 1000000000}), generation); if (release) replace = false; } @@ -62,7 +62,7 @@ class MixerSnapshot : public NodeSISO, output.setTimeBase(timeBase()); output.setPts(pts); if (slot_ != -1) - av_dict_set(&output.raw()->metadata, marker, std::to_string(frames.generation()).c_str(), 0); + av_dict_set(&output.raw()->metadata, kMarker, std::to_string(frames.generation()).c_str(), 0); } else { if (!input) return; output = *input; diff --git a/src/nodes/repeat_last_frame.cpp b/src/nodes/repeat_last_frame.cpp index 229e811b..9a3935fe 100644 --- a/src/nodes/repeat_last_frame.cpp +++ b/src/nodes/repeat_last_frame.cpp @@ -1,5 +1,5 @@ #include "node_common.hpp" -#include "../mixer/MonotonicClock.hpp" +#include "../mixer/primitives/MonotonicClock.hpp" extern "C" { #include @@ -62,7 +62,7 @@ class RepeatLastFrame: public NodeSISO { auto r = NodeSISO::createCommon(edges, params); if (params.count("fps")) { r->fps_ = parseRatio(params["fps"]); - r->period_ns_ = int64_t(1000000000.0 * r->fps_.getDenominator() / r->fps_.getNumerator()); + r->period_ns_ = av_rescale_q(1, av_inv_q(r->fps_.getValue()), {1, 1000000000}); } return r; } diff --git a/src/nodes/source_switcher.cpp b/src/nodes/source_switcher.cpp index 7d2a408c..1ec76cae 100644 --- a/src/nodes/source_switcher.cpp +++ b/src/nodes/source_switcher.cpp @@ -1,6 +1,6 @@ #include "node_common.hpp" #include "../SharedTimeline.hpp" -#include "../mixer/CutLatencyProbe.hpp" +#include "../mixer/primitives/CutLatencyProbe.hpp" template class MixerSourceSwitcher : public NodeMultiInput, public NodeSingleOutput, @@ -123,10 +123,11 @@ class MixerSourceSwitcher : public NodeMultiInput, public NodeSingleOutput auto activeFor = [&](int input_index, T* data) { if (timeline_reference && input_index == timeline_reference_input_) return reference_active; - av::Timestamp pts = data->pts(); - if (pts.isNoPts()) - pts = data->pts(); - return this->template tlGet("active", pts, active_input_.load(std::memory_order_relaxed)); + const av::Timestamp pts = data->pts(); + const int current = active_input_.load(std::memory_order_relaxed); + // Without a PTS the frame has no place on the timeline; keep the current selection. + if (pts.isNoPts()) return current; + return this->template tlGet("active", pts, current); }; int output_count = 0; diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 00000000..1c9d1ebf --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,31 @@ +"""Shared scaffolding for the standalone C++ unit tests under tests/cpp/.""" +import pathlib +import shutil +import subprocess + +import pytest + +ROOT = pathlib.Path(__file__).resolve().parents[1] + + +@pytest.fixture +def cpp_binary(tmp_path): + """Compile tests/cpp/.cpp into a binary; pkg-config `libs` are linked and their + absence skips the test, `sources` are extra .cpp files relative to the repo root.""" + def build(name, *, libs=(), sources=(), pthread=False): + compiler = shutil.which("g++") or shutil.which("clang++") + if not compiler: + pytest.skip("no C++ compiler available") + flags = [] + if libs: + probe = subprocess.run(["pkg-config", "--cflags", "--libs", *libs], capture_output=True, text=True) + if probe.returncode != 0: + pytest.skip(f"development files not found: {' '.join(libs)}") + flags = probe.stdout.split() + binary = tmp_path / name + subprocess.run([compiler, "-std=c++17", "-O0", "-g", "-Wall", "-Wextra", *(["-pthread"] if pthread else []), + "-I", str(ROOT / "src"), "-I", str(ROOT / "deps/avcpp/src"), "-I", str(ROOT / "deps/include"), + str(ROOT / "tests/cpp" / f"{name}.cpp"), *(str(ROOT / s) for s in sources), + "-o", str(binary), *flags], check=True) + return binary + return build diff --git a/tests/cpp/test_compositor_color.cpp b/tests/cpp/test_compositor_color.cpp new file mode 100644 index 00000000..8e4aa2e9 --- /dev/null +++ b/tests/cpp/test_compositor_color.cpp @@ -0,0 +1,29 @@ +#include "mixer/primitives/compositor_color.hpp" +#include +#include + +int main() { + AVFrame frame{}; + for (auto transfer : {AVCOL_TRC_UNSPECIFIED, AVCOL_TRC_BT709}) { + for (auto primaries : {AVCOL_PRI_UNSPECIFIED, AVCOL_PRI_BT709}) { + frame.color_trc = transfer; + frame.color_primaries = primaries; + assert(avp::mixer::isSdrGraphicColor(frame)); + assert(frame.color_trc == transfer && frame.color_primaries == primaries); + } + } + // A missing companion tag must not turn an explicit HDR/wide-gamut + // declaration into SDR. Fully tagged unsupported combinations also fail. + for (auto transfer : {AVCOL_TRC_ARIB_STD_B67, AVCOL_TRC_SMPTE2084, AVCOL_TRC_GAMMA22}) { + for (auto primaries : {AVCOL_PRI_UNSPECIFIED, AVCOL_PRI_BT709, AVCOL_PRI_BT2020}) { + frame.color_trc = transfer; + frame.color_primaries = primaries; + assert(!avp::mixer::isSdrGraphicColor(frame)); + } + } + for (auto transfer : {AVCOL_TRC_UNSPECIFIED, AVCOL_TRC_BT709}) { + frame.color_trc = transfer; + frame.color_primaries = AVCOL_PRI_BT2020; + assert(!avp::mixer::isSdrGraphicColor(frame)); + } +} diff --git a/tests/cpp/test_compositor_geometry.cpp b/tests/cpp/test_compositor_geometry.cpp index 37772f20..0ecc655f 100644 --- a/tests/cpp/test_compositor_geometry.cpp +++ b/tests/cpp/test_compositor_geometry.cpp @@ -1,7 +1,7 @@ -#include "hwaccel/CompositorGeometry.hpp" +#include "mixer/primitives/compositor_geometry.hpp" #include #include -using namespace avp::compositor; +using namespace avp::mixer; int main() { const Rect tile{540, 240, 540, 240}; auto wide = place(1920, 1080, {}, tile, true, 2, 2); @@ -30,7 +30,8 @@ int main() { assert(direct->destination.w == 426 && direct->destination.h == 240); assert(direct->destination.x == wide->destination.x); auto portrait_canvas = placeInCanvas(720, 1280, {}, {0, 0, 1080, 1920}, 1920, 1080, 2, 2); - assert(portrait_canvas->destination.w == 340 && portrait_canvas->destination.h == 606); + // av_rescale rounds 607.5 to 608 where integer division truncated to 606. + assert(portrait_canvas->destination.w == 342 && portrait_canvas->destination.h == 608); assert(portrait_canvas->destination.x == 368 && portrait_canvas->destination.y == 656); assert(!placeInCanvas(640, 360, {}, tile, 0, 1080, 2, 2)); for (int w = 2; w < 2048; w += 17) { diff --git a/tests/cpp/test_cut_latency.cpp b/tests/cpp/test_cut_latency.cpp index 4ce0e6b9..0ae8b889 100644 --- a/tests/cpp/test_cut_latency.cpp +++ b/tests/cpp/test_cut_latency.cpp @@ -1,4 +1,4 @@ -#include "mixer/CutLatency.hpp" +#include "mixer/primitives/CutLatency.hpp" #include "CommandTiming.hpp" #include #include diff --git a/tests/cpp/test_hdr_metadata.cpp b/tests/cpp/test_hdr_metadata.cpp new file mode 100644 index 00000000..593feeb9 --- /dev/null +++ b/tests/cpp/test_hdr_metadata.cpp @@ -0,0 +1,52 @@ +#include "hdr_metadata.hpp" +#include +#include + +// util.hpp's Error prints a stack trace through this symbol from util.cpp. +void print_stack_trace() {} + +int main() { + const Parameters md = {{"primaries", {{0.708, 0.292}, {0.170, 0.797}, {0.131, 0.046}}}, + {"white_point", {0.3127, 0.3290}}, + {"max_luminance", 1000.0}, {"min_luminance", 0.0001}, + {"max_cll", 1000}, {"max_fall", 400}}; + AVCodecContext *ctx = avcodec_alloc_context3(nullptr); + assert(ctx); + attachHdrMetadata(ctx, md); + assert(ctx->nb_decoded_side_data == 2); + const AVFrameSideData *mastering = av_frame_side_data_get(ctx->decoded_side_data, ctx->nb_decoded_side_data, + AV_FRAME_DATA_MASTERING_DISPLAY_METADATA); + const AVFrameSideData *light = av_frame_side_data_get(ctx->decoded_side_data, ctx->nb_decoded_side_data, + AV_FRAME_DATA_CONTENT_LIGHT_LEVEL); + assert(mastering && light); + size_t mastering_size = 0, light_size = 0; + av_free(av_mastering_display_metadata_alloc_size(&mastering_size)); + av_free(av_content_light_metadata_alloc(&light_size)); + assert(mastering->size == mastering_size && light->size == light_size); + const auto *mdm = reinterpret_cast(mastering->data); + assert(mdm->has_primaries && mdm->has_luminance); + assert(mdm->display_primaries[0][0].num == 35400 && mdm->display_primaries[0][0].den == 50000); // 0.708 + assert(mdm->white_point[0].num == 15635 && mdm->white_point[1].num == 16450); // D65 + assert(mdm->max_luminance.num == 10000000 && mdm->max_luminance.den == 10000); // 1000 nits + assert(mdm->min_luminance.num == 1); // 0.0001 nits + const auto *cll = reinterpret_cast(light->data); + assert(cll->MaxCLL == 1000 && cll->MaxFALL == 400); + avcodec_free_context(&ctx); + + for (const char *missing : {"primaries", "white_point", "max_cll"}) { + Parameters bad = md; + bad.erase(missing); + AVCodecContext *c = avcodec_alloc_context3(nullptr); + bool thrown = false; + try { attachHdrMetadata(c, bad); } catch (const Error &e) { thrown = std::strstr(e.what(), missing) != nullptr; } + assert(thrown); + avcodec_free_context(&c); + } + Parameters shape = md; + shape["primaries"] = {{0.7, 0.3}}; + AVCodecContext *c = avcodec_alloc_context3(nullptr); + bool thrown = false; + try { attachHdrMetadata(c, shape); } catch (const Error &) { thrown = true; } + assert(thrown); + avcodec_free_context(&c); +} diff --git a/tests/cpp/test_mixer_playout.cpp b/tests/cpp/test_mixer_playout.cpp index 497616f5..db0579b7 100644 --- a/tests/cpp/test_mixer_playout.cpp +++ b/tests/cpp/test_mixer_playout.cpp @@ -10,7 +10,7 @@ void burst_keeps_every_frame() { // Two equally paced sources arrive in a burst after a delayed receiver wake. // Neither the early source's future frame nor the late source's first frame // may be discarded just because both are visible at the first deadline. - avp::mixer::Playout mix(2, {60, 1}); + avp::mixer::Playout mix(2, avp::mixer::TickGrid(av::Rational(60, 1))); mix.push(0, 100, 0); mix.push(1, 200, 4000000); mix.push(0, 101, 18000000); @@ -31,7 +31,7 @@ void burst_keeps_every_frame() { } void sixteen_independent_phases() { - avp::mixer::Playout mix(16, {60, 1}); + avp::mixer::Playout mix(16, avp::mixer::TickGrid(av::Rational(60, 1))); int next[16] = {}; int previous[16] = {}; for (int tick = 0; tick < 3600; ++tick) { @@ -58,7 +58,7 @@ void sixteen_independent_phases() { void sparse_missing_paints_do_not_amplify_into_persistent_losses() { using namespace avp::mixer; - const FrameRate rate(60, 1); + const TickGrid rate(av::Rational(60, 1)); for (size_t count : {1u, 4u, 8u, 16u}) { for (double latency_ms : {100.0 / 3, 50.0}) { Playout mix(count, rate, latency_ms); @@ -105,7 +105,7 @@ void sparse_missing_paints_do_not_amplify_into_persistent_losses() { void complete_jitter_plateau_is_not_rate_drift() { using namespace avp::mixer; - const FrameRate rate(60, 1); + const TickGrid rate(av::Rational(60, 1)); Playout mix(1, rate, 50.0); int next = 0; auto offset = [](int frame) -> int64_t { @@ -130,7 +130,7 @@ void complete_jitter_plateau_is_not_rate_drift() { void stale_burst_is_not_retimestamped_as_fresh() { using namespace avp::mixer; - const FrameRate rate(60, 1); + const TickGrid rate(av::Rational(60, 1)); Playout mix(2, rate, 50.0); int next = 0, healthy = 0, previous = -1; for (int tick = 0; tick < 1200; ++tick) { @@ -158,7 +158,7 @@ void stale_burst_is_not_retimestamped_as_fresh() { void rational_source_drift_does_not_amplify() { using namespace avp::mixer; - const FrameRate input(60000, 1001), output(60, 1); + const TickGrid input(av::Rational(60000, 1001)), output(av::Rational(60, 1)); Playout mix(1, output, 50.0); int next = 0, previous = -1; for (int tick = 0; tick < 36000; ++tick) { @@ -181,7 +181,7 @@ void rational_source_drift_does_not_amplify() { } void bounded_queue_counts_overflow() { - avp::mixer::Playout mix(1, {60, 1}); + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1))); for (int id = 0; id < 100; ++id) mix.push(0, id, id * 1000000000LL / 60); CHECK(mix.queued(0) <= 8); CHECK(mix.stats(0).overflow == 92); @@ -189,10 +189,10 @@ void bounded_queue_counts_overflow() { void latency_cannot_exceed_retained_frames() { bool rejected = false; - try { avp::mixer::Playout unsupported(1, {60, 1}, 200.0); } + try { avp::mixer::Playout unsupported(1, avp::mixer::TickGrid(av::Rational(60, 1)), 200.0); } catch (const std::invalid_argument &) { rejected = true; } CHECK(rejected); - avp::mixer::Playout mix(1, {60, 1}, 100.0, + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1)), 100.0, avp::mixer::TimestampMode::Presentation); int next = 0; for (int output = 0; output < 120; ++output) { @@ -209,7 +209,7 @@ void latency_cannot_exceed_retained_frames() { } void missed_deadlines_do_not_catch_up_in_bursts() { - avp::mixer::Playout mix(1, {60, 1}); + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1))); mix.push(0, 0, 0); CHECK(mix.prepare(34000000)); mix.commit(); @@ -226,7 +226,7 @@ void missed_deadlines_do_not_catch_up_in_bursts() { } void latency_and_backpressure() { - avp::mixer::Playout mix(1, {60, 1}, 50.0); + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1)), 50.0); mix.push(0, 42, 0); CHECK(!mix.prepare(49999999)); const auto first = mix.prepare(50000000); @@ -239,7 +239,7 @@ void latency_and_backpressure() { } void source_clock_gap_recovers_bounded_delay() { - avp::mixer::Playout mix(1, {60, 1}); + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1))); mix.push(0, 10, 0); CHECK(mix.prepare(34000000)); mix.commit(); @@ -260,7 +260,7 @@ void source_clock_gap_recovers_bounded_delay() { } void preserves_presentation_timestamps_for_rate_conversion() { - avp::mixer::Playout mix(1, {60, 1}, {}, avp::mixer::TimestampMode::Presentation); + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1)), {}, avp::mixer::TimestampMode::Presentation); mix.push(0, 10, 0); mix.push(0, 11, 33333333); CHECK(*mix.prepare(34000000)->frames[0] == 10); @@ -274,13 +274,13 @@ void preserves_presentation_timestamps_for_rate_conversion() { } void rational_clock_and_reference_lifetime() { - const avp::mixer::FrameRate ntsc(30000, 1001); + const avp::mixer::TickGrid ntsc(av::Rational(30000, 1001)); CHECK(ntsc.time(30000) == 1001000000000LL); CHECK(ntsc.time(1800000) == 60060000000000LL); CHECK(ntsc.atOrBefore(33366666) == 1); CHECK(ntsc.atOrBefore(33366665) == 0); CHECK(ntsc.time(-1) == -33366667); - avp::mixer::Playout> mix(1, {60, 1}); + avp::mixer::Playout> mix(1, avp::mixer::TickGrid(av::Rational(60, 1))); auto frame = std::make_shared(42); std::weak_ptr owner = frame; mix.push(0, frame, 0); @@ -300,17 +300,17 @@ void invalid_parameters_fail_early() { for (double latency : {-1.0, std::numeric_limits::infinity(), std::numeric_limits::quiet_NaN()}) { bool rejected = false; - try { avp::mixer::Playout mix(1, {60, 1}, latency); } + try { avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1)), latency); } catch (const std::invalid_argument &) { rejected = true; } CHECK(rejected); } - avp::mixer::Playout zero(1, {60, 1}, 0.0); + avp::mixer::Playout zero(1, avp::mixer::TickGrid(av::Rational(60, 1)), 0.0); zero.push(0, 7, 0); CHECK(*zero.prepare(0)->frames[0] == 7); } void eof_drains_future_frames_before_finishing() { - avp::mixer::Playout mix(2, {60, 1}); + avp::mixer::Playout mix(2, avp::mixer::TickGrid(av::Rational(60, 1))); mix.push(0, 10, 0); mix.push(0, 11, 16666667); mix.push(1, 20, 0); @@ -327,7 +327,7 @@ void eof_drains_future_frames_before_finishing() { } void inactive_slot_reactivation_does_not_reuse_old_scene() { - avp::mixer::Playout mix(2, {60, 1}); + avp::mixer::Playout mix(2, avp::mixer::TickGrid(av::Rational(60, 1))); mix.push(0, 10, 0); mix.push(1, 20, 0); CHECK(mix.prepare(34000000)); @@ -346,7 +346,7 @@ void inactive_slot_reactivation_does_not_reuse_old_scene() { } void prewarm_waits_for_every_active_slot() { - avp::mixer::Playout mix(2, {60, 1}, {}, avp::mixer::TimestampMode::Presentation); + avp::mixer::Playout mix(2, avp::mixer::TickGrid(av::Rational(60, 1)), {}, avp::mixer::TimestampMode::Presentation); mix.push(0, 10, 0); CHECK(!mix.prepare(34000000, true)); mix.push(1, 20, 16666667); @@ -360,7 +360,7 @@ void prewarm_waits_for_every_active_slot() { } void irregular_first_paints_do_not_shorten_the_playout_delay() { - avp::mixer::Playout mix(1, {60, 1}); + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1))); mix.push(0, 0, 0); int next = 1; for (int tick = 0; tick <= 100; ++tick) { @@ -382,9 +382,9 @@ void prewarmed_slots_share_frame_ids_and_output_ticks() { // Hidden and visible slots start at different times. A late-starting // preview must use the same content grid; it must not restart PTS at zero. using avp::mixer::TimestampMode; - avp::mixer::Playout program(2, {60, 1}, {}, TimestampMode::Presentation); - avp::mixer::Playout preview(2, {60, 1}, {}, TimestampMode::Presentation); - const avp::mixer::FrameRate rate(60, 1); + avp::mixer::Playout program(2, avp::mixer::TickGrid(av::Rational(60, 1)), {}, TimestampMode::Presentation); + avp::mixer::Playout preview(2, avp::mixer::TickGrid(av::Rational(60, 1)), {}, TimestampMode::Presentation); + const avp::mixer::TickGrid rate(av::Rational(60, 1)); for (int tick = 0; tick < 180; ++tick) { // Unequal arrival jitter, while presentation timestamps remain exact. for (int source = 0; source < 2; ++source) { @@ -406,7 +406,7 @@ void prewarmed_slots_share_frame_ids_and_output_ticks() { void scene_reload_discards_prewarm_frames_before_first_visible_frame() { using avp::mixer::TimestampMode; - avp::mixer::Playout mix(2, {60, 1}, {}, TimestampMode::Presentation); + avp::mixer::Playout mix(2, avp::mixer::TickGrid(av::Rational(60, 1)), {}, TimestampMode::Presentation); mix.push(0, 10, 0); mix.push(1, 20, 0); CHECK(mix.prepare(34000000, true)); @@ -429,8 +429,8 @@ void scene_reload_discards_prewarm_frames_before_first_visible_frame() { void stalled_input_does_not_block_healthy_inputs_or_replay_late_burst() { using avp::mixer::TimestampMode; - avp::mixer::Playout mix(2, {60, 1}, {}, TimestampMode::Presentation); - const avp::mixer::FrameRate rate(60, 1); + avp::mixer::Playout mix(2, avp::mixer::TickGrid(av::Rational(60, 1)), {}, TimestampMode::Presentation); + const avp::mixer::TickGrid rate(av::Rational(60, 1)); for (int tick = 0; tick < 15; ++tick) { mix.push(0, tick, rate.time(tick)); if (tick < 5 || tick > 10) mix.push(1, 100 + tick, rate.time(tick)); @@ -450,7 +450,7 @@ void stalled_input_does_not_block_healthy_inputs_or_replay_late_burst() { } void route_reset_rejects_old_frames_still_in_upstream_edges() { - avp::mixer::Playout mix(1, {60, 1}, {}, avp::mixer::TimestampMode::Presentation); + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1)), {}, avp::mixer::TimestampMode::Presentation); mix.push(0, 10, 0); CHECK(mix.prepare(34000000, true)); mix.commit(); @@ -465,8 +465,8 @@ void route_reset_rejects_old_frames_still_in_upstream_edges() { } void presentation_phase_on_half_tick_keeps_every_frame() { - avp::mixer::Playout mix(1, {60, 1}, {}, avp::mixer::TimestampMode::Presentation); - const avp::mixer::FrameRate rate(60, 1); + avp::mixer::Playout mix(1, avp::mixer::TickGrid(av::Rational(60, 1)), {}, avp::mixer::TimestampMode::Presentation); + const avp::mixer::TickGrid rate(av::Rational(60, 1)); // A valid 60 Hz source starts at 25 ms, exactly halfway between output // ticks. Converting 1/60 to integer ns alternates rounding errors every // three frames; those sub-ns errors must not become repeats/skips. @@ -489,7 +489,7 @@ void presentation_phase_on_half_tick_keeps_every_frame() { } void resumed_source_is_live_after_an_eof_marker() { - avp::mixer::Playout mix(2, {60, 1}); + avp::mixer::Playout mix(2, avp::mixer::TickGrid(av::Rational(60, 1))); mix.push(0, 10, 0); mix.push(1, 20, 0); mix.endInput(0); @@ -505,7 +505,7 @@ void resumed_source_is_live_after_an_eof_marker() { void inactive_prewarm_retains_live_frames_without_rendering() { using namespace avp::mixer; - const FrameRate rate(30, 1); + const TickGrid rate(av::Rational(30, 1)); Playout mix(2, rate, {}, TimestampMode::Presentation); mix.setPrewarm(0, true); mix.setPrewarm(1, true); diff --git a/tests/cpp/test_mixer_snapshot.cpp b/tests/cpp/test_mixer_snapshot.cpp index 3a59338b..43148848 100644 --- a/tests/cpp/test_mixer_snapshot.cpp +++ b/tests/cpp/test_mixer_snapshot.cpp @@ -1,4 +1,4 @@ -#include "mixer/Snapshot.hpp" +#include "mixer/primitives/Snapshot.hpp" #include #include diff --git a/tests/cpp/test_pixel_layout.cpp b/tests/cpp/test_pixel_layout.cpp new file mode 100644 index 00000000..97477774 --- /dev/null +++ b/tests/cpp/test_pixel_layout.cpp @@ -0,0 +1,54 @@ +#include "mixer/primitives/pixel_layout.hpp" +#include +using namespace avp::mixer; + +int main() { + assert(av_pix_fmt_count_planes(AV_PIX_FMT_NV12) == 2 && av_pix_fmt_count_planes(AV_PIX_FMT_P210) == 2); + assert(av_pix_fmt_count_planes(AV_PIX_FMT_YUV420P) == 3 && av_pix_fmt_count_planes(AV_PIX_FMT_BGRA) == 1); + assert(sampleBytes(AV_PIX_FMT_NV12) == 1 && sampleBytes(AV_PIX_FMT_P010) == 2); + assert(storageShift(AV_PIX_FMT_P010) == 6 && storageShift(AV_PIX_FMT_NV12) == 0); + assert(chromaXAlign(AV_PIX_FMT_NV12) == 2 && chromaYAlign(AV_PIX_FMT_NV12) == 2); + assert(chromaXAlign(AV_PIX_FMT_P210) == 2 && chromaYAlign(AV_PIX_FMT_P210) == 1); + assert(alignCoord(7, 2) == 6 && alignCoord(7, 1) == 7); + + int x = -4, y = 2, w = 20, h = 30; + assert(clipRect(x, y, w, h, 10, 10) && x == 0 && y == 2 && w == 10 && h == 8); + assert(!clipRect(x, y, w, h, 0, 10)); + + // NV12 chroma plane: half height, same byte width; P210 chroma: full height, 4 bytes per pair. + int bx, by, bw, bh; + lumaRectToPlaneRegion(AV_PIX_FMT_NV12, 4, 6, 8, 10, 1, bx, by, bw, bh); + assert(bx == 4 && by == 3 && bw == 8 && bh == 5); + lumaRectToPlaneRegion(AV_PIX_FMT_P210, 4, 6, 8, 10, 1, bx, by, bw, bh); + assert(bx == 8 && by == 6 && bw == 16 && bh == 10); + lumaRectToPlaneRegion(AV_PIX_FMT_P210, 4, 6, 8, 10, 0, bx, by, bw, bh); + assert(bx == 8 && bw == 16 && bh == 10); + lumaRectToPlaneRegion(AV_PIX_FMT_BGRA, 3, 1, 5, 2, 0, bx, by, bw, bh); + assert(bx == 12 && bw == 20 && bh == 2); + + assert(isYuvPromoteConvertible(AV_PIX_FMT_NV12, AV_PIX_FMT_P210)); + assert(isYuvPromoteConvertible(AV_PIX_FMT_NV12, AV_PIX_FMT_P010)); + assert(!isYuvPromoteConvertible(AV_PIX_FMT_P010, AV_PIX_FMT_NV12)); // never demote + assert(!isYuvPromoteConvertible(AV_PIX_FMT_NV12, AV_PIX_FMT_NV12)); + assert(!isYuvPromoteConvertible(AV_PIX_FMT_YUV420P, AV_PIX_FMT_P010)); // planar chroma + assert(isRgbToYuvConvertible(AV_PIX_FMT_BGRA, AV_PIX_FMT_NV12)); + assert(isRgbToYuvConvertible(AV_PIX_FMT_RGB24, AV_PIX_FMT_P210)); + assert(!isRgbToYuvConvertible(AV_PIX_FMT_BGRA, AV_PIX_FMT_YUV420P)); + assert(isAlphaCompatible(AV_PIX_FMT_YUV420P, AV_PIX_FMT_YUVA420P)); + assert(!isAlphaCompatible(AV_PIX_FMT_NV12, AV_PIX_FMT_YUVA420P)); + assert(packedAlphaOffset(AV_PIX_FMT_BGRA) == 3 && packedAlphaOffset(AV_PIX_FMT_BGR0) == -1); + assert(canvasAccepts(AV_PIX_FMT_NV12, AV_PIX_FMT_NV12) && canvasAccepts(AV_PIX_FMT_BGRA, AV_PIX_FMT_P210)); + assert(!canvasAccepts(AV_PIX_FMT_YUV420P, AV_PIX_FMT_NV12)); + + assert(alphaPlaneIndex(AV_PIX_FMT_YUVA420P) == 3 && alphaPlaneIndex(AV_PIX_FMT_BGRA) == -1); + uint16_t v = 0; + assert(planeClearValue(AV_PIX_FMT_P010, nullptr, 0, v) && v == 64); + assert(planeClearValue(AV_PIX_FMT_P010, nullptr, 1, v) && v == 512); + assert(planeClearValue(AV_PIX_FMT_NV12, nullptr, 1, v) && v == 128); + assert(planeClearValue(AV_PIX_FMT_YUVA420P, nullptr, 3, v) && v == 255); + assert(planeClearValue(AV_PIX_FMT_BGR0, nullptr, 0, v) && v == 0); + assert(!planeClearValue(AV_PIX_FMT_YUYV422, nullptr, 0, v)); + AVFrame full{}; + full.color_range = AVCOL_RANGE_JPEG; + assert(blackLumaValue(AV_PIX_FMT_P010, &full) == 0 && blackLumaValue(AV_PIX_FMT_P010, nullptr) == 64); +} diff --git a/tests/cuda/_harness.py b/tests/cuda/_harness.py new file mode 100644 index 00000000..dd5473e4 --- /dev/null +++ b/tests/cuda/_harness.py @@ -0,0 +1,91 @@ +"""Shared scaffolding for the CUDA smokes: AVPlumber setup, the packed-v210 +ingest chain, plane extraction and the frame drain loop.""" + +import time + +import numpy as np + +EOF = -(1 << 63) + + +def make_avp(hwaccel, capacity=3): + """AVPlumber with a CUDA hwaccel and an error sink; returns (avp, errors).""" + from pyplumber import AVPlumber + avp = AVPlumber() + errors = [] + avp.on_exception = lambda *e: errors.append(tuple(map(str, e))) + avp.edges.planCapacity("*", capacity) + avp.executeCommandsFromString(f'hwaccel.init {{"name":"{hwaccel}","type":"cuda"}}') + return avp, errors + + +def v210_chain(nodes, tag, path, *, width, height, stride, fmt, hwaccel, color, **unpack): + """Input(rawvideo gray, stride x height packets) -> Demux -> V210ToCuda; returns the edge. + The gray demuxer only frames bytes, which also permits nonstandard row strides.""" + from pyplumber.node import Demux, Input, V210ToCuda + nodes += [ + Input({"name": f"in_{tag}", "url": str(path), "format": "rawvideo", "dst": f"pkt_{tag}", + "options": {"pixel_format": "gray", "video_size": f"{stride}x{height}", + "framerate": "60"}}), + Demux({"name": f"demux_{tag}", "src": f"pkt_{tag}", "routing": {"v:0": f"packed_{tag}"}}), + V210ToCuda({"name": f"unpack_{tag}", "src": f"packed_{tag}", "dst": f"gpu_{tag}", + "hwaccel": hwaccel, "width": width, "height": height, "stride": stride, + "fps": "60/1", "timebase": "1/90000", "format": fmt, **color, **unpack}), + ] + return f"gpu_{tag}" + + +def frame_planes(frame, fmt): + """Downloaded frame -> (Y, U, V) logical 10-bit planes.""" + if fmt in ("p010le", "p210le"): + heights = (frame.height, (frame.height + 1) // 2 if fmt == "p010le" else frame.height) + y, uv = [np.frombuffer(d, "> 6, uv[:, 0::2] >> 6, uv[:, 1::2] >> 6 + widths = [frame.width, frame.width // 2, frame.width // 2] if fmt != "yuv444p10le" \ + else [frame.width] * 3 + return tuple(np.frombuffer(d, "10 promotion (x4) and +420<->422 chroma resampling, so every combination has an exact expected value. +Covers same-format copy, same-subsampling depth promote, and cross-subsampling +(420->422) promote-in. Run on the NVIDIA host with the FFmpeg 8.1 avplumber +module + numpy. +""" + +import argparse +from pathlib import Path +import tempfile + +import numpy as np + +from _harness import drain, finish, make_avp, start + +W, H = 256, 128 +# Semiplanar canvas formats under test: (sample_bytes, depth, chroma_h_shift) +FORMATS = { + "nv12": dict(sb=1, depth=8, ch=1), # 4:2:0 8-bit + "p010le": dict(sb=2, depth=10, ch=1), # 4:2:0 10-bit (data in high bits, shift 6) + "p210le": dict(sb=2, depth=10, ch=0), # 4:2:2 10-bit +} +# (source, canvas) pairs: copies, depth promotes, and 420->422 promote-in. +MATRIX = [("nv12", "nv12"), ("p010le", "p010le"), ("p210le", "p210le"), + ("nv12", "p010le"), ("nv12", "p210le"), ("p010le", "p210le")] +# Flat 8-bit source codes; 10-bit sources use these <<2 so promotion is exact. +Y8, U8, V8 = 180, 110, 200 + + +def write_flat(path, fmt): + """One flat-colour raw frame in *fmt*. P010/P210 store the 10-bit code in + the high bits, so the stored word is code<<6.""" + f = FORMATS[fmt] + scale = 4 if f["depth"] == 10 else 1 + hishift = 6 if f["sb"] == 2 else 0 + dt = "> f["ch"], W), dt) # interleaved Cb,Cr at half width + uv[:, 0::2] = (U8 * scale) << hishift + uv[:, 1::2] = (V8 * scale) << hishift + with Path(path).open("wb") as s: + s.write(y.tobytes()); s.write(uv.tobytes()) + + +def run(root, src_fmt, canvas, timeout): + from pyplumber.node import CudaRectOverlay, DecVideo, Demux, FilterVideo, Input + + path = Path(root) / f"{src_fmt}.raw" + if not path.exists(): + write_flat(path, src_fmt) + avp, errors = make_avp("ix_gpu") + nodes = [ + Input({"name": "in", "url": str(path), "format": "rawvideo", "dst": "pkt", + "options": {"pixel_format": src_fmt, "video_size": f"{W}x{H}", "framerate": "60"}}), + Demux({"name": "dx", "src": "pkt", "routing": {"v:0": "raw"}}), + DecVideo({"name": "dec", "src": "raw", "dst": "cpu"}), + FilterVideo({"name": "up", "src": "cpu", "dst": "gpu", "hwaccel": "ix_gpu", "graph": "hwupload"}), + CudaRectOverlay({"name": "comp", "src": ["gpu"], "dst": "scene", "hwaccel": "ix_gpu", + "width": W, "height": H, "sw_format": canvas, "scale": True, + "active_inputs": 1, + "layers": [{"dst_x": 0, "dst_y": 0, "dst_w": W, "dst_h": H}]}), + FilterVideo({"name": "down", "src": "scene", "dst": "out", "hwaccel": "ix_gpu", + "graph": f"hwdownload,format={canvas}"}), + ] + f = FORMATS[canvas] + scale = 4 if f["depth"] == 10 else 1 + ey, eu, ev = Y8 * scale, U8 * scale, V8 * scale + shift, dt = (6, "> shift + uv = np.frombuffer(frame.data[1], dt).reshape(H >> f["ch"], frame.linesize[1] // f["sb"])[:, :W] >> shift + cy, cx = (H >> f["ch"]) // 2, (W // 2) & ~1 # centre: no edge taps on a flat field + assert abs(int(y[H // 2, W // 2]) - ey) <= 1, f"Y {int(y[H//2, W//2])} != {ey}" + assert abs(int(uv[cy, cx]) - eu) <= 1 and abs(int(uv[cy, cx + 1]) - ev) <= 1, "chroma mismatch" + assert not errors, errors + assert state["count"] == 1, "no frame" + print(f"PASS {src_fmt:>7} -> {canvas:<7} (Y {ey} Cb {eu} Cr {ev})", flush=True) + finally: + finish(avp, nodes) + + +def main(): + p = argparse.ArgumentParser(description=__doc__) + p.add_argument("--timeout", type=float, default=30) + args = p.parse_args() + with tempfile.TemporaryDirectory(prefix="avp-ix-") as root: + for src, canvas in MATRIX: + run(root, src, canvas, args.timeout) + print("interop matrix OK") + + +if __name__ == "__main__": + main() diff --git a/tests/cuda/smoke_mixer_10bit.py b/tests/cuda/smoke_mixer_10bit.py new file mode 100644 index 00000000..18c0c1ca --- /dev/null +++ b/tests/cuda/smoke_mixer_10bit.py @@ -0,0 +1,214 @@ +"""Two-source 10-bit mixer smoke: A/B cuda_rect_overlay -> transition_cuda -> +raw verification against an exact CPU reference. + +P210 runs ingest packed v210 (HLG and SDR-promoted families) through +v210_to_cuda; P010 and planar 444 runs upload labeled CPU fixtures. Scene A is a +two-tile grid (scaled path), scene B is source 0 fullscreen (copy path), and +the transition blends them at exact binary-fraction alphas so the float +arithmetic reproduces bit-exactly on the CPU. Run on the NVIDIA host with the +FFmpeg 8.1 avplumber module; the download is solely the verification boundary. +""" + +import argparse +from pathlib import Path +import tempfile + +import numpy as np + +from _harness import drain, finish, frame_planes, make_avp, start, v210_chain +from v210_fixture import COLOR, FAMILIES, frame_stride, write_fixture + +W, H, FRAMES = 384, 216, 6 +# Every run drains to EOF and requires the dual-input transition to flush all +# generated frames: the filter node used to finish on an EOF marker that arrived +# while it was waiting for input, dropping the other scene's queued tail. +# ``--tail-margin N`` generates N extra frames and stops after the scored ones. +TAIL_MARGIN = 0 +TRANSITIONS = (("fade", 0.0), ("fade", 0.25), ("fade", 1.0), ("wipe_left", 0.5)) + + +def planes444(index, source): + row = np.arange(H, dtype=np.int64)[:, None] + x = np.arange(W, dtype=np.int64)[None, :] + k = index + source * 1000 + y = 64 + (x + row * 17 + k * 101) % 877 + u = 64 + (x * 3 + row * 29 + k * 59) % 897 + v = 64 + (x * 7 + row * 13 + k * 83) % 897 + return y, u, v + + +def source_planes(family, index, source): + if family == "420": + y, u, v = planes444(index, source) + return y, u[::2, ::2], v[::2, ::2] + if family == "444": + return planes444(index, source) + return tuple(p.astype(np.int64) for p in FAMILIES[family](W, H, index + source * 1000)) + + +def write_upload_fixture(path, source, family, frames): + with Path(path).open("wb") as stream: + for index in range(frames): + planes = source_planes(family, index, source) + if family == "420": + y, u, v = planes + uv = np.stack((u, v), axis=-1).reshape(H // 2, W) + planes = (y << 6, uv << 6) + for plane in planes: + stream.write(plane.astype("= FRAMES: + continue + reference = blend(scene_a(family, index, n), + source_planes(family, index, 0), mode, coef) + for plane, (actual, expected) in enumerate(zip(frame_planes(frame, fmt), reference)): + np.testing.assert_array_equal(actual, expected, + err_msg=f"frame {index} plane {plane}") + assert not errors, errors + expected = gen_frames if not margin else FRAMES + assert state["count"] == expected, f"delivered {state['count']}/{expected} frames (eof={state['eof']})" + finally: + finish(avp, nodes) + + +def main(): + global FRAMES + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--timeout", type=float, default=60) + parser.add_argument("--families", nargs="+", default=["sdr8", "hlg", "420", "444"]) + parser.add_argument("--tail-margin", type=int, default=TAIL_MARGIN, + help="extra generated frames after the scored ones (default 0: drain to EOF)") + parser.add_argument("--frames", type=int, default=FRAMES) + parser.add_argument("--grid", type=int, default=16, help="sources in the grid run (0 skips it)") + parser.add_argument("--capacity", type=int, default=3, help="edge queue capacity") + parser.add_argument("--no-pairs", action="store_true", help="skip the two-source transition runs") + args = parser.parse_args() + FRAMES = args.frames + with tempfile.TemporaryDirectory(prefix="avp-mix10-") as root: + for family in args.families if not args.no_pairs else (): + fmt = {"420": "p010le", "444": "yuv444p10le"}.get(family, "p210le") + for mode, coef in TRANSITIONS: + run(root, family, fmt, mode, coef, args.timeout, margin=args.tail_margin, capacity=args.capacity) + print(f"PASS {family}/{fmt} {mode} alpha={coef}", flush=True) + # Many simultaneous full-resolution sources drawn as a grid (16 = 4x4). + for family in args.families if args.grid else (): + if family in ("420", "444"): + continue # CPU upload chains add nothing over the v210 grid + run(root, family, "p210le", "fade", 0.25, args.timeout, n=args.grid, + margin=args.tail_margin, capacity=args.capacity) + print(f"PASS {family}/p210le grid of {args.grid}, fade alpha=0.25", flush=True) + + +if __name__ == "__main__": + main() diff --git a/tests/cuda/smoke_output_qualification.py b/tests/cuda/smoke_output_qualification.py new file mode 100644 index 00000000..985f7b7d --- /dev/null +++ b/tests/cuda/smoke_output_qualification.py @@ -0,0 +1,180 @@ +"""Encoded-output qualification: build the real mixer, record a few seconds, probe +what NVENC wrote. + +Run A: SDR NV12 canvas, static SDR bars (v210) -> H.264 reference. +Run B: HLG P210 canvas, the same SDR bars fullscreen + static HLG bars as a PIP -> + HLG HEVC, PQ HEVC (HDR10 static metadata) and a mobius-0.9 SDR H.264 rendition. + +Asserts codec, profile, pixel format and VUI on every leg, the HDR10 mastering +display / MaxCLL SEIs on the PQ leg, and that the SDR round trip through the HLG +canvas (SDR -> HLG -> SDR) reproduces run A's bar colours within a few codes. Needs the CUDA avplumber module and +the custom ffmpeg (hevc/h264 software decoders for probing). +""" + +import argparse +import json +import os +from pathlib import Path +import re +import signal +import subprocess +import sys +import tempfile +import time + +import numpy as np + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +from v210_fixture import pack_v210, write_fixture # noqa: E402 + +W, H, FPS = 1280, 720, 60 +BAR = W // 8 # eight vertical colour bars +PIP = {"x": 760, "y": 40, "w": 480, "h": 270} +PIP_BOTTOM = PIP["y"] + PIP["h"] + 40 # bar centres are sampled below the HLG insert +TOLERANCE = 8 # codes, per channel, at a bar centre + + +def run_mixer(repo, cfg_path, outputs, seconds, timeout): + """Run mixer.py until every output has grown past *seconds* of content, then kill it.""" + env = {**os.environ, "PYTHONPATH": f"{repo}:{os.environ.get('PYTHONPATH', '')}"} + proc = subprocess.Popen([sys.executable, str(repo / "demos/mixer/mixer.py"), "--config", str(cfg_path), + "--output", str(Path(cfg_path).with_suffix(".unused.ts")), "--remote-control-port", "0"], + env=env, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True) + deadline = time.monotonic() + timeout + started = None + try: + while time.monotonic() < deadline: + time.sleep(1) + if proc.poll() is not None: + raise AssertionError("mixer exited early:\n" + proc.stdout.read()[-3000:]) + if all(p.exists() and p.stat().st_size > 0 for p in outputs): + started = started or time.monotonic() + if time.monotonic() - started >= seconds: + return + raise AssertionError(f"mixer did not produce {outputs} within {timeout}s") + finally: + proc.send_signal(signal.SIGKILL) # finite-graph shutdown may hang; the files are complete streams + proc.wait() + + +def probe(ffmpeg, path): + out = subprocess.run([ffmpeg, "-hide_banner", "-i", str(path)], capture_output=True, text=True).stderr + line = next((l for l in out.splitlines() if "Video:" in l), "") + assert line, out + return line + + +def side_data(ffmpeg, path): + out = subprocess.run([ffmpeg, "-hide_banner", "-i", str(path), "-frames:v", "1", "-vf", "showinfo", "-f", "null", "-"], + capture_output=True, text=True).stderr + return "\n".join(l for l in out.splitlines() if "side data" in l.lower() or "MaxCLL" in l) + + +def bar_centres(ffmpeg, path): + """Mean (Y, U, V) of the centre of each colour bar, sampled below the PIP so the + HLG insert never contributes; edges are excluded because every chroma resample + on the HLG path blurs them a little, which is expected and not a colour error.""" + centres = [] + for i in range(8): + crop = f"crop={BAR // 4}:{H - PIP_BOTTOM}:{i * BAR + BAR // 2 - BAR // 8}:{PIP_BOTTOM}" + run = subprocess.run([ffmpeg, "-hide_banner", "-ss", "1", "-i", str(path), "-frames:v", "1", + "-vf", f"{crop},signalstats,metadata=print:file=-", "-f", "null", "-"], + capture_output=True, text=True) + out = run.stdout + run.stderr # metadata=print:file=- writes to stdout + found = [float(re.search(rf"signalstats\.{k}AVG=([0-9.]+)", out).group(1)) for k in "YUV"] + centres.append(found) + return centres + + +def write_bars(path, width, height, frames): + """Flat 75% BT.709 colour bars as limited-range 10-bit v210: flat regions survive + the chroma resampling of the HLG round trip, unlike the pixel-frequency fixtures.""" + rgb = np.array([(1, 1, 1), (1, 1, 0), (0, 1, 1), (0, 1, 0), (1, 0, 1), (1, 0, 0), (0, 0, 1), (0, 0, 0)], + dtype=np.float64) * 0.75 + cols = np.repeat(rgb, width // 8, axis=0)[:width] + r, g, b = (np.tile(cols[:, i], (height, 1)) for i in range(3)) + yp = 0.2126 * r + 0.7152 * g + 0.0722 * b + y = np.rint(64 + 876 * yp).astype(" H.264 BT.709:", line.split("Video: ")[1][:60], flush=True) + + legs = {k: root / f"{k}.ts" for k in ("hlg", "pq", "sdr")} + cfg_b = root / "b.json" + cfg_b.write_text(json.dumps(config(root, {"working_format": "p210le", "color": "hlg"}, [ + {"id": "hlg", "target": str(legs["hlg"]), "codec": "hevc_nvenc", "bitrate_kbps": 8000}, + {"id": "pq", "target": str(legs["pq"]), "codec": "hevc_nvenc", "color": "pq", "max_fall": 400, + "bitrate_kbps": 8000}, + {"id": "sdr", "target": str(legs["sdr"]), "codec": "h264_nvenc", "tonemap": args.sdr_tonemap, + "tonemap_param": args.knee, "bitrate_kbps": 6000}]))) + run_mixer(repo, cfg_b, list(legs.values()), args.seconds, args.timeout) + + line = probe(args.ffmpeg, legs["hlg"]) + assert "hevc (Main 10)" in line and "yuv420p10le(tv, bt2020nc/bt2020/arib-std-b67" in line, line + print("PASS HLG canvas -> HEVC Main 10, BT.2020 HLG VUI", flush=True) + line = probe(args.ffmpeg, legs["pq"]) + assert "hevc (Main 10)" in line and "yuv420p10le(tv, bt2020nc/bt2020/smpte2084" in line, line + sd = side_data(args.ffmpeg, legs["pq"]) + assert "Mastering display" in sd and "MaxCLL=1000" in sd and "MaxFALL=400" in sd, sd + print("PASS PQ rendition -> HEVC Main 10, PQ VUI, HDR10 mastering display + MaxCLL/MaxFALL SEIs", flush=True) + line = probe(args.ffmpeg, legs["sdr"]) + assert "h264" in line and "yuv420p(tv, bt709" in line, line + + if args.keep: + import shutil + Path(args.keep).mkdir(parents=True, exist_ok=True) + for f in (ref, *legs.values()): + shutil.copy(f, Path(args.keep) / f.name) + ref_bars, out_bars = bar_centres(args.ffmpeg, ref), bar_centres(args.ffmpeg, legs["sdr"]) + worst = max(abs(a - b) for r, o in zip(ref_bars, out_bars) for a, b in zip(r, o)) + detail = "; ".join(f"{'/'.join(f'{v:.0f}' for v in r)} -> {'/'.join(f'{v:.0f}' for v in o)}" + for r, o in zip(ref_bars, out_bars)) + assert worst <= TOLERANCE, f"SDR round trip drifted by {worst:.1f} codes: {detail}" + print(f"PASS SDR -> HLG canvas -> SDR round trip ({args.sdr_tonemap}): every bar centre within " + f"{worst:.1f} codes", flush=True) + + +if __name__ == "__main__": + main() diff --git a/tests/cuda/smoke_tonemap_transfers.py b/tests/cuda/smoke_tonemap_transfers.py new file mode 100644 index 00000000..a0c672a8 --- /dev/null +++ b/tests/cuda/smoke_tonemap_transfers.py @@ -0,0 +1,244 @@ +"""Exercise the shipped FFmpeg CUDA filter against display-light references. + +Run on an NVIDIA host: python3 tests/cuda/smoke_tonemap_transfers.py --ffmpeg +CPU uploads/downloads are fixture I/O only; all conversions run on CUDA frames. +No mixer instance or encoder is started or modified by this test. +""" + +import argparse +import itertools +import subprocess + +import numpy as np + +from tonemap_reference import convert_transfer_codes, display_light, encode_display_light, rgb_to_codes + +W, H = 98, 66 # Exercise CUDA pitch padding and partial thread blocks. +TRANSFERS = ("sdr", "hlg", "pq") +TAGS = { + "sdr": ("bt709", "bt709", "bt709"), + "hlg": ("bt2020nc", "bt2020", "arib-std-b67"), + "pq": ("bt2020nc", "bt2020", "smpte2084"), +} + + +def pack(codes, depth): + maximum = (1 << depth) - 1 + y = np.clip(np.rint(codes[..., 0]), 0, maximum) + uv = codes[..., 1:].reshape(H // 2, 2, W // 2, 2, 2).mean(axis=(1, 3)) + uv = np.clip(np.rint(uv), 0, maximum) + dtype = "> 6 + y = a[:W * H].reshape(H, W) + uv = a[W * H:].reshape(H // 2, W // 2, 2).repeat(2, axis=0).repeat(2, axis=1) + return np.concatenate((y[..., None], uv), axis=-1).astype(float) + + +def fixture(transfer, depth): + # Neutral checkpoints, saturated primaries, secondaries and near-black ramps. + palette = np.array([[0, 0, 0], [1, 1, 1], [.18, .18, .18], [.5, .5, .5], + [1, 0, 0], [0, 1, 0], [0, 0, 1], [1, 1, 0], + [0, 1, 1], [1, 0, 1], [.8, .4, .2], [.02, .02, .02]]) + yy, xx = np.indices((H, W)) + rgb = palette[(xx // 2 + yy // 2) % len(palette)] + # Vary luma inside each chroma block to check four-pixel chroma averaging. + rgb = rgb * np.where((xx % 2 + yy % 2) == 0, 1.0, .8)[..., None] + return pack(rgb_to_codes(rgb, transfer, depth), depth) + + +def run_filter(ffmpeg, data, source, input_depth, graph, output_depth, *, fail=None, full_range=False): + space, primaries, trc = TAGS[source] + fmt_in = "p010le" if input_depth == 10 else "nv12" + fmt_out = "p010le" if output_depth == 10 else "nv12" + filters = ( + f"setparams=range={'pc' if full_range else 'tv'}:colorspace={space}" + f":color_primaries={primaries}:color_trc={trc},hwupload_cuda," + f"{graph},hwdownload,format={fmt_out},showinfo" + ) + cmd = [ffmpeg, "-hide_banner", "-nostdin", "-loglevel", "info", "-filter_threads", "1", + "-f", "rawvideo", "-pixel_format", fmt_in, "-video_size", f"{W}x{H}", + "-framerate", "60", "-i", "pipe:0", "-vf", filters, "-frames:v", "3", + "-c:v", "rawvideo", "-threads:v", "1", "-pix_fmt", fmt_out, + "-f", "rawvideo", "pipe:1"] + result = subprocess.run(cmd, input=data * 3, capture_output=True, timeout=90) + log = result.stderr.decode(errors="replace") + if fail: + assert result.returncode and fail in log, (graph, log) + return None, log + assert result.returncode == 0, (graph, log) + size = W * H * 3 // 2 * (2 if output_depth == 10 else 1) + assert len(result.stdout) == 3 * size, (graph, len(result.stdout), log) + frames = [result.stdout[i * size:(i + 1) * size] for i in range(3)] + assert frames[0] == frames[1] == frames[2], "Repeated frames differ" + return frames[0], log + + +def assert_tags(log, transfer): + space, primaries, trc = TAGS[transfer] + for tag in ("color_range:tv", f"color_space:{space}", + f"color_primaries:{primaries}", f"color_trc:{trc}"): + assert tag in log, (tag, log) + + +def check_reference(): + # Published PQ and HLG reference-white checkpoints, independent of the CUDA code. + white = np.full((1, 3), 203.0) + assert np.max(abs(encode_display_light(white, "pq") - .580689)) < 1e-6 + assert np.max(abs(encode_display_light(white, "hlg") - .749877)) < 2e-6 + for transfer in TRANSFERS: + rgb = np.linspace(0, 1, 303).reshape(-1, 3) + restored = encode_display_light(display_light(rgb, transfer), transfer) + assert np.max(abs(restored - rgb)) < 1e-6, transfer + assert np.allclose(display_light(np.ones((1, 3)), "sdr"), 203) + + +def check_directions(ffmpeg): + for source, target, depth in itertools.product(TRANSFERS, TRANSFERS, (8, 10)): + data = fixture(source, depth) + output_depth = depth if source == target else (8 if target == "sdr" else 10) + graph = f"tonemap_cuda=transfer_in={source}:transfer_out={target}:tonemap=none:desat=0" + result, log = run_filter(ffmpeg, data, source, depth, graph, output_depth) + assert_tags(log, target) + if source == target: + assert result == data, f"{source}: identity changed pixels" + else: + expected = pack(convert_transfer_codes(unpack(data, depth), source, target, depth), output_depth) + error = np.max(abs(unpack(result, output_depth) - unpack(expected, output_depth))) + assert error <= 2, (source, target, depth, error) + print(f"PASS {source}->{target} {depth}-bit input, pixels and metadata", flush=True) + + +def check_white_and_roundtrip(ffmpeg): + for target, white, peak in itertools.product(("hlg", "pq"), (100, 203), (1000, 2000)): + data = pack(np.broadcast_to([235, 128, 128], (H, W, 3)), 8) + forward = f"tonemap_cuda=transfer_in=sdr:transfer_out={target}:sdr_white={white}:hdr_peak={peak}" + result, _ = run_filter(ffmpeg, data, "sdr", 8, forward, 10) + encoded = encode_display_light(np.full((1, 3), white), target, white, peak) + expected_y = np.rint(encoded[0, 0] * 876 + 64) + values = unpack(result, 10) + assert np.max(abs(values[..., 0] - expected_y)) <= 1, (target, white, peak) + assert np.max(abs(values[..., 1:] - 512)) <= 1 + # Preserve saturated SDR colors, including after conversion to a P210 canvas + # and back to P010 for delivery. No tone mapping in the return leg. + data = fixture("sdr", 8) + for target in ("hlg", "pq"): + graph = (f"tonemap_cuda=transfer_in=sdr:transfer_out={target}," + "scale_cuda=format=p210le,scale_cuda=format=p010le," + f"tonemap_cuda=transfer_in={target}:transfer_out=sdr:tonemap=none:desat=0") + # Constant vertically, to isolate color errors from the P210 chroma + # resampling. Pixel/chroma averaging is covered by check_directions. + block = unpack(data, 8)[0:1, ::2].repeat(H, 0).repeat(2, 1) + flat = pack(block, 8) + result, _ = run_filter(ffmpeg, flat, "sdr", 8, graph, 8) + error = np.max(abs(unpack(result, 8) - unpack(flat, 8))) + # Near-zero RGB components amplify HDR quantization when decoded through + # SDR gamma. Check the independently quantized roundtrip as well as the + # six-code bound observed at the saturated endpoints of this fixture. + forward = pack(convert_transfer_codes(unpack(flat, 8), "sdr", target, 8), 10) + expected = pack(convert_transfer_codes(unpack(forward, 10), target, "sdr", 10), 8) + assert np.max(abs(unpack(result, 8) - unpack(expected, 8))) <= 2 + assert error <= 6, (target, error) + print(f"PASS SDR->{target}->P210->SDR color roundtrip (max {error} codes)", flush=True) + data = fixture("hlg", 10) + graph = ("tonemap_cuda=transfer_in=hlg:transfer_out=pq," + "tonemap_cuda=transfer_in=pq:transfer_out=hlg") + block = unpack(data, 10)[0:1, ::2].repeat(H, 0).repeat(2, 1) + data = pack(block, 10) + result, _ = run_filter(ffmpeg, data, "hlg", 10, graph, 10) + error = np.max(abs(unpack(result, 10) - unpack(data, 10))) + assert error <= 3, error + print(f"PASS HLG->PQ->HLG display-light roundtrip (max {error} codes)", flush=True) + print("PASS reference-white mapping at 100/203 nits and 1000/2000-nit HLG peaks", flush=True) + + +def check_downmapping(ffmpeg): + for source, op in itertools.product(("hlg", "pq"), + ("linear", "gamma", "clip", "reinhard", "hable", "mobius")): + data = fixture(source, 10) + graph = f"tonemap_cuda=transfer_in={source}:transfer_out=sdr:tonemap={op}:desat=0.5" + result, _ = run_filter(ffmpeg, data, source, 10, graph, 8) + expected = pack(convert_transfer_codes(unpack(data, 10), source, "sdr", 10, + tonemap=op, desat=0.5), 8) + error = np.max(abs(unpack(result, 8) - unpack(expected, 8))) + assert error <= 2, (source, op, error) + print("PASS explicit HLG/PQ->SDR operators and highlight desaturation", flush=True) + + +def check_invalid(ffmpeg): + data = fixture("sdr", 8) + cases = [ + ("transfer_in=sdr", "Set both transfer_in"), + ("transfer_out=hlg", "Set both transfer_in"), + ("transfer_in=sdr:transfer_out=hlg:sdr_white=2000:hdr_peak=1000", "sdr_white must not exceed"), + ] + for options, message in cases: + run_filter(ffmpeg, data, "sdr", 8, "tonemap_cuda=" + options, 10, fail=message) + run_filter(ffmpeg, data, "sdr", 8, "tonemap_cuda=transfer_in=sdr:transfer_out=hlg", 10, + fail="Full-range input is unsupported", full_range=True) + print("PASS invalid/conflicting options and full-range input rejected", flush=True) + + +def check_auto(ffmpeg): + for source, target, depth in itertools.product(TRANSFERS, TRANSFERS, (8, 10)): + data = fixture(source, depth) + outdepth = 8 if target == "sdr" else 10 + fmt = "nv12" if outdepth == 8 else "p010le" + graph = f"tonemap_cuda=transfer_in=auto:transfer_out={target}:format={fmt}:tonemap=none:desat=0" + result, log = run_filter(ffmpeg, data, source, depth, graph, outdepth) + assert_tags(log, target) + if source == target: + expected = pack(unpack(data, depth) * 2.0 ** (outdepth - depth), outdepth) + else: + expected = pack(convert_transfer_codes(unpack(data, depth), source, target, depth), outdepth) + error = np.max(abs(unpack(result, outdepth) - unpack(expected, outdepth))) + assert error <= 2, (source, target, depth, error) + if source == target and depth == outdepth: + assert result == data, "Auto identity changed pixel bytes" + print("PASS auto", source, target, depth, error, flush=True) + # Untagged decodes (most files) are SDR BT.709: identity, output tagged, warning logged. + untagged = "setparams=color_trc=unknown:color_primaries=unknown:colorspace=unknown:range=unknown," + data = fixture("sdr", 8) + result, log = run_filter(ffmpeg, data, "sdr", 8, untagged + "tonemap_cuda=transfer_in=auto:transfer_out=sdr", 8) + assert result == data and "assuming limited-range BT.709 SDR" in log, log + assert_tags(log, "sdr") + print("PASS untagged input assumed SDR", flush=True) + # A transfer with unspecified companions follows the transfer; a tagged companion that + # contradicts the (assumed or declared) transfer still fails. + partial = "setparams=color_primaries=unknown:colorspace=unknown,tonemap_cuda=transfer_in=auto:transfer_out=hlg" + _, log = run_filter(ffmpeg, fixture("hlg", 10), "hlg", 10, partial, 10) + assert_tags(log, "hlg") + graph = "setparams=color_trc=unknown,tonemap_cuda=transfer_in=auto:transfer_out=sdr" + run_filter(ffmpeg, fixture("hlg", 10), "hlg", 10, graph, 8, fail="contradictory") + print("PASS partial tags follow the transfer; BT.2020 without a transfer rejected", flush=True) + graph = "setparams=color_primaries=bt709,tonemap_cuda=transfer_in=auto:transfer_out=sdr" + run_filter(ffmpeg, fixture("hlg", 10), "hlg", 10, graph, 8, fail="contradictory") + run_filter(ffmpeg, fixture("sdr", 8), "sdr", 8, + "tonemap_cuda=transfer_in=auto:transfer_out=sdr", 8, + fail="contradictory", full_range=True) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--ffmpeg", default="ffmpeg") + parser.add_argument("--reference-only", action="store_true") + args = parser.parse_args() + check_reference() + print("PASS independent reference checkpoints", flush=True) + if not args.reference_only: + check_directions(args.ffmpeg) + check_auto(args.ffmpeg) + check_white_and_roundtrip(args.ffmpeg) + check_downmapping(args.ffmpeg) + check_invalid(args.ffmpeg) + + +if __name__ == "__main__": + main() diff --git a/tests/cuda/smoke_v210_to_cuda.py b/tests/cuda/smoke_v210_to_cuda.py new file mode 100644 index 00000000..187f9ded --- /dev/null +++ b/tests/cuda/smoke_v210_to_cuda.py @@ -0,0 +1,113 @@ +"""Raw byte simulator -> GPU v210 unpack -> CUDA scale -> CPU pixel verification. + +Run on an NVIDIA host with the FFmpeg 8.1 avplumber Python module and NumPy. +The download is solely the test's verification boundary. No MXL service or +NVDEC/NVENC is used. Each run also compares against FFmpeg's CPU v210 decoder, +and the HLG fixture family's transfer checkpoints are asserted up front. +""" + +import argparse +from pathlib import Path +import subprocess +import tempfile +import time + +import numpy as np + +from _harness import drain, finish, frame_planes, make_avp, start, v210_chain +from v210_fixture import frame_stride, hlg_planes, sample_planes, write_fixture + +BT709 = {"color_range": "tv", "colorspace": "bt709", "color_primaries": "bt709", + "color_trc": "bt709", "chroma_location": "left"} + + +def check_hlg_fixture(): + """BT.2100 checkpoints E = 0, 1/12, 1 -> Y 64, 502, 940 with neutral chroma.""" + y, u, v = hlg_planes(96, 8, index=3) + for i, code in enumerate((64, 502, 940)): + np.testing.assert_array_equal(y[:2, 12 * i:12 * (i + 1)], code) + np.testing.assert_array_equal(u[:2, 6 * i:6 * (i + 1)], 512) + np.testing.assert_array_equal(v[:2, 6 * i:6 * (i + 1)], 512) + assert y.min() >= 64 and y.max() <= 940 + + +def cpu_reference(ffmpeg, path, width, height, frames): + decoded = subprocess.run([ + ffmpeg, "-v", "error", "-threads", "1", "-f", "v210", + "-video_size", f"{width}x{height}", "-framerate", "60", "-i", str(path), + "-frames:v", str(frames), "-pix_fmt", "yuv422p10le", "-f", "rawvideo", "pipe:1", + ], check=True, capture_output=True, timeout=60).stdout + assert len(decoded) == width * height * 4 * frames + values = np.frombuffer(decoded, dtype=">>(input.data,source_pitch,1,1,2,2, - output.data,target_pitch,origin_x,2,4,4,6,8,lanes); + output.data,target_pitch,origin_x,2,4,4,6,8,lanes,1,0); check(cudaGetLastError()); check(cudaMemcpy(target.data(),output.data,target.size(),cudaMemcpyDeviceToHost)); for (int y=0; y<8; ++y) { @@ -47,10 +47,79 @@ void checkUpscale(int lanes, int origin_x) { } } } +void checkUpscaleWord(int lanes, int origin_x, int shift) { + // The byte ramp times four lands on genuine 10-bit codes; bilinear results + // scale linearly, so the expected table is the byte table times four. + constexpr std::array expected = { + 0,16,48,64, 32,48,80,96, 96,112,144,160, 128,144,176,192 + }; + constexpr int source_pitch_e = 8, target_pitch_e = 16; + std::vector source(source_pitch_e*4, 0xeeee); + for (int y=0; y<2; ++y) + for (int x=0; x<2; ++x) + for (int lane=0; lane target(target_pitch_e*8, 0xa5a5); + DeviceBytes input(source.size()*2), output(target.size()*2); + check(cudaMemcpy(input.data, source.data(), source.size()*2, cudaMemcpyHostToDevice)); + check(cudaMemcpy(output.data, target.data(), target.size()*2, cudaMemcpyHostToDevice)); + scale_plane<<>>(input.data,source_pitch_e*2,1,1,2,2, + output.data,target_pitch_e*2,origin_x,2,4,4,6,8,lanes,2,shift); + check(cudaGetLastError()); + check(cudaMemcpy(target.data(),output.data,target.size()*2,cudaMemcpyDeviceToHost)); + for (int y=0; y<8; ++y) { + for (int e=0; e=0 && ox<4 && oy>=0 && oy<4; + const int logical=lane ? 1023-4*expected[oy*4+ox] : 4*expected[oy*4+ox]; + const int value=written ? logical< 10-bit destination, promotion multiplier 4 (16->64 style). + // Bilinear result then times four, stored at the P210 shift; the byte table + // is the same expected upscale. + constexpr std::array expected = { + 0,16,48,64, 32,48,80,96, 96,112,144,160, 128,144,176,192 + }; + constexpr int source_pitch = 8, target_pitch_e = 16; + std::vector source(source_pitch*4, 0xee); + for (int y=0; y<2; ++y) + for (int x=0; x<2; ++x) + for (int lane=0; lane target(target_pitch_e*8, 0xa5a5); + DeviceBytes input(source.size()), output(target.size()*2); + check(cudaMemcpy(input.data, source.data(), source.size(), cudaMemcpyHostToDevice)); + check(cudaMemcpy(output.data, target.data(), target.size()*2, cudaMemcpyHostToDevice)); + convert_scale_plane<<>>(input.data,source_pitch,1,1,2,2,1,0, + output.data,target_pitch_e*2,2,2,4,4,6,8,lanes,2,dst_shift,4.f); + check(cudaGetLastError()); + check(cudaMemcpy(target.data(),output.data,target.size()*2,cudaMemcpyDeviceToHost)); + for (int y=0; y<8; ++y) { + for (int e=0; e=0 && ox<4 && oy>=0 && oy<4; + const int base=lane ? 255-expected[oy*4+ox] : expected[oy*4+ox]; + const int value=written ? (base*4)<10-bit promotion, crop, pitch and destination clipping\n"; } catch (const std::exception& error) { std::cerr << error.what() << '\n'; return 1; diff --git a/tests/cuda/tonemap_reference.py b/tests/cuda/tonemap_reference.py new file mode 100644 index 00000000..669a3b83 --- /dev/null +++ b/tests/cuda/tonemap_reference.py @@ -0,0 +1,245 @@ +"""NumPy reference (oracle) for tonemap_cuda, ported verbatim from FFmpeg n8.1 +libavfilter/opencl/tonemap.cl + colorspace_common.cl. + +HDR (HLG or PQ, BT.2020) -> SDR (BT.709). Used to validate the CUDA kernel +offline: the CUDA output must match this within rounding. Static peak (no +per-frame detection); operators direct/linear/gamma/clip/reinhard/hable/mobius, +matching the n8.1 kernel exactly. +""" + +import numpy as np + +REFERENCE_WHITE = 100.0 +ST2084_MAX_LUMINANCE = 10000.0 +ST2084_M1 = 0.1593017578125 +ST2084_M2 = 78.84375 +ST2084_C1 = 0.8359375 +ST2084_C2 = 18.8515625 +ST2084_C3 = 18.6875 +HLG_A, HLG_B, HLG_C = 0.17883277, 0.28466892, 0.55991073 +SDR_AVG = 0.25 + +# BT.2020 and BT.709 luma weights. +LUMA_2020 = np.array([0.2627, 0.6780, 0.0593]) +LUMA_709 = np.array([0.2126, 0.7152, 0.0722]) + +# Linear-light BT.2020 -> BT.709 primaries (rgb2rgb), standard NPM product. +RGB2020_TO_709 = np.array([ + [1.660491, -0.587641, -0.072850], + [-0.124550, 1.132900, -0.008349], + [-0.018151, -0.100579, 1.118730], +]) + + +def eotf_st2084(x): + p = np.power(np.maximum(x, 0.0), 1.0 / ST2084_M2) + a = np.maximum(p - ST2084_C1, 0.0) + b = np.maximum(ST2084_C2 - ST2084_C3 * p, 1e-6) + c = np.power(a / b, 1.0 / ST2084_M1) + return np.where(x > 0.0, c * ST2084_MAX_LUMINANCE / REFERENCE_WHITE, 0.0) + + +def inverse_oetf_hlg(x): + a = 4.0 * x * x + b = np.exp((x - HLG_C) / HLG_A) + HLG_B + return np.where(x < 0.5, a, b) + + +def oetf_bt709(c): + c = np.maximum(c, 0.0) + return np.where(c < 0.018, 4.5 * c, 1.099 * np.power(c, 0.45) - 0.099) + + +def ootf_hlg(rgb, peak): + luma = rgb @ LUMA_2020 + gamma = max(1.0, 1.2 + 0.42 * np.log10(peak * REFERENCE_WHITE / 1000.0)) + factor = peak * np.power(np.maximum(luma, 1e-6), gamma - 1.0) / np.power(12.0, gamma) + return rgb * factor[..., None] + + +def hable_f(x): + a, b, c, d, e, f = 0.15, 0.50, 0.10, 0.20, 0.02, 0.30 + return (x * (x * a + b * c) + d * e) / (x * (x * a + b) + d * f) - e / f + + +TONE = { + "direct": lambda s, peak, p: s, + "linear": lambda s, peak, p: s * p / peak, + "clip": lambda s, peak, p: np.clip(s * p, 0.0, 1.0), + "reinhard": lambda s, peak, p: s / (s + p) * (peak + p) / peak, + "hable": lambda s, peak, p: hable_f(s) / hable_f(peak), + "mobius": lambda s, peak, p: _mobius(s, peak, p), + "gamma": lambda s, peak, p: np.where(s > 0.05, (s / peak) ** (1 / p), + s * (0.05 / peak) ** (1 / p) / 0.05), +} + + +def _mobius(s, peak, j): + a = -j * j * (peak - 1.0) / (j * j - 2.0 * j + peak) + b = (j * j - 2.0 * j * peak + peak) / max(peak - 1.0, 1e-6) + out = (b * b + 2.0 * b * j + j * j) / (b - a) * (s + a) / (s + b) + return np.where(s <= j, s, out) + + +DEFAULT_PARAM = {"linear": 1.0, "gamma": 1.8, "clip": 1.0, "reinhard": 0.3, + "hable": 1.0, "mobius": 0.3, "direct": 1.0} + + +def map_one_pixel_rgb(rgb, peak, average, tonemap, param, desat, target_peak=1.0): + sig = np.maximum(np.maximum(rgb[..., 0], np.maximum(rgb[..., 1], rgb[..., 2])), 1e-6) + if target_peak > 1.0: + sig = sig / target_peak + peak = peak / target_peak + sig_old = sig.copy() + slope = min(1.0, SDR_AVG / average) + sig = sig * slope + peak = peak * slope + if desat > 0.0: + luma = rgb @ LUMA_709 + coeff = np.maximum(sig - 0.18, 1e-6) / np.maximum(sig, 1e-6) + coeff = np.power(coeff, 10.0 / desat) + rgb = rgb * (1 - coeff[..., None]) + luma[..., None] * coeff[..., None] + sig = sig * (1 - coeff) + (luma * slope) * coeff + sig = TONE[tonemap](sig, peak, param) + sig = np.minimum(sig, 1.0) + return rgb * (sig / sig_old)[..., None] + + +def tonemap_hdr_to_sdr(rgb_src, transfer, peak, tonemap="hable", param=None, desat=0.5): + """rgb_src: source non-linear RGB in [0,1] (BT.2020). Returns BT.709 SDR + non-linear RGB in [0,1]. peak in REFERENCE_WHITE units (HLG 1000 nits -> 10).""" + if param is None: + param = DEFAULT_PARAM[tonemap] + if transfer == "pq": + lin = eotf_st2084(rgb_src) + elif transfer == "hlg": + lin = inverse_oetf_hlg(rgb_src) + lin = ootf_hlg(lin, peak) + else: + raise ValueError(transfer) + lin = lin @ RGB2020_TO_709.T # lrgb2lrgb: BT.2020 -> BT.709 + lin = map_one_pixel_rgb(lin, peak, SDR_AVG, tonemap, param, desat) + return oetf_bt709(np.clip(lin, 0.0, None)) + + +# --- YCbCr <-> non-linear RGB, matching colorspace_common.cl (limited range) --- + +def yuv2rgb_2020(y10, cb10, cr10): + """BT.2020 limited-range 10-bit YCbCr codes -> non-linear RGB in [0,1].""" + y = (y10 / 1023.0 * 255.0 - 16.0) / 219.0 + u = (cb10 / 1023.0 * 255.0 - 128.0) / 224.0 + v = (cr10 / 1023.0 * 255.0 - 128.0) / 224.0 + r = y + 1.4746 * v + g = y - 0.16455 * u - 0.57135 * v + b = y + 1.8814 * u + return np.stack([r, g, b], axis=-1) + + +def rgb2yuv_709_8bit(rgb): + """BT.709 non-linear RGB [0,1] -> limited-range 8-bit YCbCr codes.""" + r, g, b = rgb[..., 0], rgb[..., 1], rgb[..., 2] + y = 0.2126 * r + 0.7152 * g + 0.0722 * b + u = (b - y) / 1.8556 + v = (r - y) / 1.5748 + y8 = np.rint(219.0 * y + 16.0) + u8 = np.rint(224.0 * u + 128.0) + v8 = np.rint(224.0 * v + 128.0) + return (np.clip(y8, 0, 255).astype(int), np.clip(u8, 0, 255).astype(int), + np.clip(v8, 0, 255).astype(int)) + + +def tonemap_codes(y10, cb10, cr10, transfer, peak, tonemap="hable", param=None, desat=0.5): + """Full pipeline oracle: BT.2020 HLG/PQ 10-bit codes -> BT.709 SDR 8-bit codes.""" + rgb_src = yuv2rgb_2020(np.asarray(y10, float), np.asarray(cb10, float), np.asarray(cr10, float)) + sdr = tonemap_hdr_to_sdr(rgb_src, transfer, peak, tonemap, param, desat) + return rgb2yuv_709_8bit(sdr) + + +# Explicit-transfer reference uses display light in nits (BT.1886/BT.2100). +# Derive the gamut transform from published chromaticities, independently of +# the rounded CUDA coefficient tables. +def _rgb_to_xyz(primaries): + p = np.array(primaries) + basis = np.stack((p[:, 0] / p[:, 1], np.ones(3), + (1 - p.sum(axis=1)) / p[:, 1])) + white = np.array([0.3127 / 0.3290, 1.0, (1 - 0.3127 - 0.3290) / 0.3290]) + return basis * np.linalg.solve(basis, white) + + +_XYZ709 = _rgb_to_xyz([(0.640, 0.330), (0.300, 0.600), (0.150, 0.060)]) +_XYZ2020 = _rgb_to_xyz([(0.708, 0.292), (0.170, 0.797), (0.131, 0.046)]) +_TO2020 = np.linalg.solve(_XYZ2020, _XYZ709) + + +def display_light(rgb, transfer, sdr_white=203.0, hdr_peak=1000.0): + """Encoded RGB -> display-linear nits, ideal black, no surround adjustment.""" + rgb = np.maximum(rgb, 0.0) + if transfer == "sdr": + return rgb ** 2.4 * sdr_white + if transfer == "pq": + return eotf_st2084(rgb) * REFERENCE_WHITE + scene = np.where(rgb <= 0.5, rgb * rgb / 3, + (np.exp((rgb - HLG_C) / HLG_A) + HLG_B) / 12) + gamma = max(1.0, 1.2 + 0.42 * np.log10(hdr_peak / 1000)) + gain = hdr_peak * np.maximum(scene @ LUMA_2020, 1e-12) ** (gamma - 1) + return scene * gain[..., None] + + +def encode_display_light(rgb, transfer, sdr_white=203.0, hdr_peak=1000.0): + """Display-linear nits -> encoded RGB, inverse of display_light.""" + rgb = np.maximum(rgb, 0.0) + if transfer == "sdr": + return (rgb / sdr_white) ** (1 / 2.4) + if transfer == "pq": + p = (rgb / 10000) ** ST2084_M1 + return ((ST2084_C1 + ST2084_C2 * p) / (1 + ST2084_C3 * p)) ** ST2084_M2 + gamma = max(1.0, 1.2 + 0.42 * np.log10(hdr_peak / 1000)) + luma = rgb @ LUMA_2020 + scene = rgb * ((luma / hdr_peak) ** (1 / gamma) / np.maximum(luma, 1e-12))[..., None] + return np.where(scene <= 1 / 12, np.sqrt(3 * scene), + HLG_A * np.log(np.maximum(12 * scene - HLG_B, 1e-12)) + HLG_C) + + +def rgb_to_codes(rgb, transfer, depth): + weights = LUMA_709 if transfer == "sdr" else LUMA_2020 + y = rgb @ weights + u = (rgb[..., 2] - y) / (2 * (1 - weights[2])) + v = (rgb[..., 0] - y) / (2 * (1 - weights[0])) + scale = 1 << (depth - 8) + return np.stack(((219 * y + 16) * scale, (224 * u + 128) * scale, + (224 * v + 128) * scale), axis=-1) + + +def codes_to_rgb(codes, transfer, depth): + weights = LUMA_709 if transfer == "sdr" else LUMA_2020 + codes = codes / (1 << (depth - 8)) + y, u, v = (codes[..., 0] - 16) / 219, (codes[..., 1] - 128) / 224, (codes[..., 2] - 128) / 224 + r = y + 2 * (1 - weights[0]) * v + b = y + 2 * (1 - weights[2]) * u + g = (y - weights[0] * r - weights[2] * b) / weights[1] + return np.stack((r, g, b), axis=-1) + + +def convert_transfer_codes(codes, source, target, depth, *, sdr_white=203.0, + hdr_peak=1000.0, tonemap="direct", desat=0.0): + if source == target: + return codes.copy() + light = display_light(codes_to_rgb(codes, source, depth), source, sdr_white, hdr_peak) + if source == "sdr": + light = light @ _TO2020.T + elif target == "sdr": + light = light @ np.linalg.inv(_TO2020).T + light = map_one_pixel_rgb(light / sdr_white, hdr_peak / sdr_white, SDR_AVG, + tonemap, DEFAULT_PARAM[tonemap], desat) * sdr_white + rgb = np.clip(encode_display_light(light, target, sdr_white, hdr_peak), 0, 1) + return rgb_to_codes(rgb, target, 8 if target == "sdr" else 10) + + +if __name__ == "__main__": + # Reference points to lock the CUDA kernel against (peak=10 -> HLG 1000 nits). + print("transfer y10 cb10 cr10 -> Y8 U8 V8 (hable, peak=10, desat=0.5)") + for tr in ("hlg", "pq"): + for (y, cb, cr) in [(64, 512, 512), (500, 512, 512), (940, 512, 512), + (700, 400, 700), (300, 700, 400), (843, 512, 512)]: + Y, U, V = tonemap_codes(np.array([y]), np.array([cb]), np.array([cr]), tr, 10.0) + print(f"{tr} {y} {cb} {cr} -> {int(Y[0])} {int(U[0])} {int(V[0])}") diff --git a/tests/cuda/v210_fixture.py b/tests/cuda/v210_fixture.py new file mode 100644 index 00000000..92d915af --- /dev/null +++ b/tests/cuda/v210_fixture.py @@ -0,0 +1,120 @@ +"""Generate raw v210 bytes, with no MXL, camera, or compressed-video dependency. + +The moving 10-bit ramps distinguish Y/U/V, frame order, row pitch and sample +alignment. Requires NumPy. Output has no header; consumers need its dimensions, +frame rate and stride supplied separately. +""" + +from pathlib import Path + +import numpy as np + +HLG_A, HLG_B, HLG_C = 0.17883277, 0.28466892, 0.55991073 + + +def hlg_oetf(e): + """BT.2100 HLG OETF on scene-linear values in [0, 1].""" + e = np.asarray(e, dtype=np.float64) + return np.where(e <= 1 / 12, np.sqrt(3 * e), HLG_A * np.log(12 * np.maximum(e, 1 / 12) - HLG_B) + HLG_C) + + +def frame_stride(width, stride=None): + if width <= 0 or width % 2: + raise ValueError("v210 requires a positive even width") + minimum = ((width * 2 + 2) // 3) * 4 + stride = ((width + 47) // 48) * 128 if stride is None else stride + if stride < minimum or stride % 4: + raise ValueError("stride must fit a packed row and be a multiple of four") + return stride + + +def sample_planes(width, height, index=0): + frame_stride(width) + if height <= 0: + raise ValueError("height must be positive") + row = np.arange(height, dtype=np.uint32)[:, None] + x = np.arange(width, dtype=np.uint32)[None, :] + cx = x[:, :width // 2] + y = (x + row * 17 + index * 101) % 1024 + u = (cx * 3 + row * 29 + 341 + index * 59) % 1024 + v = (cx * 7 + row * 13 + 683 + index * 83) % 1024 + return tuple(plane.astype(" 64, + 235 -> 940, neutral 128 -> 512). 8-bit source precision, not HDR.""" + frame_stride(width) + row = np.arange(height, dtype=np.uint32)[:, None] + x = np.arange(width, dtype=np.uint32)[None, :] + cx = x[:, :width // 2] + y = 16 + (x + row * 17 + index * 101) % 220 + u = 16 + (cx * 3 + row * 29 + index * 59) % 225 + v = 16 + (cx * 7 + row * 13 + index * 83) % 225 + return tuple((plane << 2).astype(" 1023) or np.any(plane < 0) for plane in planes): + raise ValueError("samples must be in the 10-bit range") + samples = np.zeros((height, ((width * 2 + 2) // 3) * 3), dtype=np.uint32) + active = samples[:, :width * 2] + active[:, 0::4], active[:, 1::4] = u, y[:, 0::2] + active[:, 2::4], active[:, 3::4] = v, y[:, 1::2] + words = samples.reshape(height, -1, 3) + packed = words[:, :, 0] | (words[:, :, 1] << 10) | (words[:, :, 2] << 20) + output = np.full((height, stride), 0xa5, dtype=np.uint8) + payload = packed.astype(" GCS_PREFIX=gs:/// eka-recorder-smoke.sh IMAGE [LABEL] +# KEEP_BROKER=1 leave the redpanda container running afterwards (default: remove it) +# Run as root or with `sudo KEEP_BROKER=1 bash eka-recorder-smoke.sh ...` (plain sudo drops env). +# +# Downstream check for the eka-recorder / recorder-ai product on a candidate avplumber build: +# the recorder image is built from the recorder repo with avplumber's python module and FFmpeg +# libraries injected (see the recorder repo's audit Dockerfile). Host-specific locations are +# variables so the script can move between hosts. Exit 0 only on PASS. +set -uo pipefail + +IMAGE=${1:?recorder image (built from the recorder repo with this avplumber injected)} +LABEL=${2:-smoke} +ROOT=${ROOT:?harness dir: recorder repo checkout, input excerpt, models, outputs} +REPO_DIR=${REPO_DIR:-$ROOT/repo} # contains ai-recorder/scripts/run_e2e_gcs_test.sh +SOURCE_FILE=${SOURCE_FILE:-$ROOT/simulation-40s.ts} # 40 s stream-copy excerpt, looped by file ingest +MODELS_DIR=${MODELS_DIR:-$ROOT/models} # basketball/{yolo,tracknet_v2,salient-viewport,scoreboard}/*.plan +OUTPUT_BASE_DIR=${OUTPUT_BASE_DIR:-$ROOT/output} +GCS_PREFIX=${GCS_PREFIX:?gs:/// the harness uploads recordings to} +BROKER=${BROKER:-recorder-smoke-kafka} +BROKER_IMAGE=${BROKER_IMAGE:-docker.redpanda.com/redpandadata/redpanda:v24.3.1} +KAFKA_PORT=${KAFKA_PORT:-29092} # the recorder resolves kafka -> 127.0.0.1:29092 +KEEP_BROKER=${KEEP_BROKER:-0} +RUN_SEC=${RUN_SEC:-150} +E2E_WINDOW_SEC=${E2E_WINDOW_SEC:-110} + +log() { echo "[smoke] $*"; } +die() { echo "[smoke] FAIL: $*" >&2; exit 1; } + +# --- preflight ------------------------------------------------------------- +docker image inspect "$IMAGE" >/dev/null 2>&1 || die "image $IMAGE not found" +[ -f "$SOURCE_FILE" ] || die "simulation input missing: $SOURCE_FILE" +[ -f "$REPO_DIR/ai-recorder/scripts/run_e2e_gcs_test.sh" ] || die "harness missing under $REPO_DIR" +for m in yolo tracknet_v2 salient-viewport scoreboard; do + ls "$MODELS_DIR/basketball/$m/"*.plan >/dev/null 2>&1 || die "models missing: $MODELS_DIR/basketball/$m" +done +gcloud storage ls "$GCS_PREFIX/" >/dev/null 2>&1 || die "no GCS access to $GCS_PREFIX" + +# --- broker ---------------------------------------------------------------- +# Redpanda's default schema-registry/pandaproxy/rpc ports (8081/8082/33145) may be taken +# by other containers on a shared host, so they are moved; only the Kafka port matters. +if ! (echo > "/dev/tcp/127.0.0.1/$KAFKA_PORT") 2>/dev/null; then + docker rm -f "$BROKER" >/dev/null 2>&1 || true + docker run -d --name "$BROKER" --network host "$BROKER_IMAGE" redpanda start \ + --overprovisioned --smp 1 --memory 512M --reserve-memory 0M --node-id 0 --check=false \ + --kafka-addr "plaintext://0.0.0.0:$KAFKA_PORT" --advertise-kafka-addr "plaintext://127.0.0.1:$KAFKA_PORT" \ + --schema-registry-addr 0.0.0.0:28081 --pandaproxy-addr 0.0.0.0:28082 --advertise-pandaproxy-addr 127.0.0.1:28082 \ + --rpc-addr 0.0.0.0:33146 --advertise-rpc-addr 127.0.0.1:33146 >/dev/null || die "cannot start $BROKER" + for i in $(seq 1 30); do (echo > "/dev/tcp/127.0.0.1/$KAFKA_PORT") 2>/dev/null && break; sleep 1; done + (echo > "/dev/tcp/127.0.0.1/$KAFKA_PORT") 2>/dev/null || { docker logs --tail 5 "$BROKER"; die "broker not listening on $KAFKA_PORT"; } + log "broker $BROKER up on $KAFKA_PORT" +else + log "port $KAFKA_PORT already accepting connections; reusing" +fi + +# --- harness: file ingest, AI on, camera motion, no Janus --- +export IMAGE REPO_DIR SOURCE_FILE MODELS_DIR OUTPUT_BASE_DIR REBUILD_IMAGE=0 KEEP_LOCAL=1 +export E2E_RECORDING_ID="${LABEL}-smoke-$(date -u +%Y%m%d-%H%M%S)" +export CONTAINER_NAME="recorder-${LABEL}-smoke" MEDIAMTX_CONTAINER="recorder-${LABEL}-unused-mediamtx" +export E2E_INGEST=file E2E_SPORT=basketball E2E_QUALITIES=fhd,hd,hi E2E_OCR=1 E2E_NO_AI=0 +export E2E_CAMERA_MOTION=1 E2E_WEBUI=0 E2E_JANUS=0 E2E_HLS_DEBUG=1 E2E_HLS_DEBUG_UPLOAD=1 +export RECORDER_METADATA_KAFKA_ENABLED=true RECORDER_METADATA_JSONL=1 +export RECORDER_SRC_DIR="$REPO_DIR/ai-recorder/recorder" +export E2E_WINDOW_SEC RUN_SEC MODELS_MOUNT_MODE=rw +log "recording $E2E_RECORDING_ID on $IMAGE (${RUN_SEC}s run, ${E2E_WINDOW_SEC}s window)" +HLOG="$OUTPUT_BASE_DIR/$E2E_RECORDING_ID.harness.log"; mkdir -p "$OUTPUT_BASE_DIR" +bash "$REPO_DIR/ai-recorder/scripts/run_e2e_gcs_test.sh" > "$HLOG" 2>&1 +STATUS=$? + +# --- numbers --------------------------------------------------------------- +RUN="$OUTPUT_BASE_DIR/$E2E_RECORDING_ID" +SEGMENTS=$(find "$RUN/recordings" -name '*.ts' 2>/dev/null | wc -l) +AVPLOG="$GCS_PREFIX/$E2E_RECORDING_ID/logs/avplumber.log" +PANICS=$(gcloud storage cat "$AVPLOG" 2>/dev/null | grep -ci panic) +RESTARTS=$(gcloud storage cat "$AVPLOG" 2>/dev/null | grep -c "initiated group auto-restart") +META=$(gcloud storage cat "$GCS_PREFIX/$E2E_RECORDING_ID/metadata/inference.jsonl" 2>/dev/null | python3 -c ' +import sys, json +pts = [json.loads(l)["frame_pts"] for l in sys.stdin if l.strip()] +print(f"{len(pts)} records spanning {max(pts) - min(pts):.1f} s of frame_pts" if pts else "0 records")') +VERDICT=$(grep -oE "full-flow .* (PASSED|FAILED)" "$HLOG" | tail -1) + +echo "----------------------------------------------------------------" +echo "recording: $E2E_RECORDING_ID" +echo "image: $IMAGE" +echo "harness: exit=$STATUS ${VERDICT:-no verdict line}" +echo "hls segments: $SEGMENTS (all qualities)" +echo "inference: ${META:-unavailable}" +echo "avplumber: panics=${PANICS:-?} input-group restarts=${RESTARTS:-?} (file ingest loops at EOF)" +echo "harness log: $HLOG" +echo "----------------------------------------------------------------" +[ "$KEEP_BROKER" = "1" ] || docker rm -f "$BROKER" >/dev/null 2>&1 +if [ "$STATUS" = "0" ] && [ "${PANICS:-1}" = "0" ]; then echo "RESULT: PASS"; exit 0; else echo "RESULT: FAIL"; exit 1; fi diff --git a/tests/python/smoke_shutdown.py b/tests/python/smoke_shutdown.py new file mode 100644 index 00000000..e536cdaf --- /dev/null +++ b/tests/python/smoke_shutdown.py @@ -0,0 +1,62 @@ +"""Run under an external timeout: shutdown must let Python workers acquire the GIL.""" + +import threading +import time + +from pyplumber import AVPlumber +from pyplumber.node import NodeBase, PythonNode + +started = threading.Event() +stopped = threading.Event() + + +class Worker(PythonNode): + def process(self): + started.set() + time.sleep(0.01) + + def doStop(self): + stopped.set() + + +avp = AVPlumber() +worker = Worker({"name": "worker", "group": "test", "dst": "unused", + "data_type": "VideoFrame", "auto_restart": "panic"}) +avp.addNode(worker) +avp.group("test").startNodes() +assert started.wait(5), "Python worker did not start" +avp.shutdown() +assert stopped.is_set(), "Python doStop did not finish before shutdown returned" +print("Python worker shutdown passed", flush=True) + +# Split makes a final process() call after stop. An empty input must not wait +# again after the stop notification has already been consumed. +avp = AVPlumber() +avp.getEdge("empty", "VideoFrame") +avp.addNode(NodeBase({"type": "split", "name": "idle_split", "group": "test", + "src": "empty", "dst": ["unused"], "data_type": "VideoFrame"})) +avp.group("test").startNodes() +time.sleep(0.1) +avp.shutdown() +print("Idle Split shutdown passed", flush=True) + +# A worker may read again while flushing its tail. Both reads must return once +# stop has been requested; the first must not consume the queue's stop state. +class Reader(PythonNode): + def process(self): + started.set() + self._src.get() + self._src.get() + stopped.set() + +started.clear() +stopped.clear() +avp = AVPlumber() +reader = Reader({"name": "reader", "group": "test", "src": "empty", + "data_type": "VideoFrame"}) +avp.addNode(reader) +avp.group("test").startNodes() +assert started.wait(5), "Reader did not start" +avp.shutdown() +assert stopped.is_set(), "Repeated reads did not stop" +print("Repeated queue reads shutdown passed", flush=True) diff --git a/tests/test_compositor_color.py b/tests/test_compositor_color.py new file mode 100644 index 00000000..a2c24b64 --- /dev/null +++ b/tests/test_compositor_color.py @@ -0,0 +1,6 @@ +"""Color validation for packed RGB(A) compositor inputs; no GPU required.""" +import subprocess + + +def test_compositor_color(cpp_binary): + subprocess.run([str(cpp_binary("test_compositor_color", libs=("libavutil",)))], check=True, timeout=10) diff --git a/tests/test_compositor_geometry.py b/tests/test_compositor_geometry.py index 41f9a42d..ebc3903c 100644 --- a/tests/test_compositor_geometry.py +++ b/tests/test_compositor_geometry.py @@ -1,14 +1,6 @@ """Frame-size changes, crop bounds and chroma alignment without GPU dependencies.""" -import pathlib -import shutil import subprocess -def test_compositor_geometry(tmp_path): - root = pathlib.Path(__file__).resolve().parents[1] - binary = tmp_path / 'compositor_geometry' - compiler = shutil.which('g++') or shutil.which('clang++') - assert compiler, 'a C++ compiler is required' - subprocess.run([compiler, '-std=c++17', '-Wall', '-Wextra', '-I', str(root / 'src'), - str(root / 'tests/cpp/test_compositor_geometry.cpp'), '-o', str(binary)], check=True) - subprocess.run([str(binary)], check=True, timeout=10) +def test_compositor_geometry(cpp_binary): + subprocess.run([str(cpp_binary("test_compositor_geometry", libs=("libavutil",)))], check=True, timeout=10) diff --git a/tests/test_cut_latency.py b/tests/test_cut_latency.py index 641a3f99..f308d168 100644 --- a/tests/test_cut_latency.py +++ b/tests/test_cut_latency.py @@ -1,14 +1,6 @@ """Cut identity and elapsed-clock accounting, without media/GPU dependencies.""" -import pathlib -import shutil import subprocess -def test_cut_latency(tmp_path): - root = pathlib.Path(__file__).resolve().parents[1] - binary = tmp_path / "cut_latency" - compiler = shutil.which("g++") or shutil.which("clang++") - assert compiler, "a C++ compiler is required" - subprocess.run([compiler, "-std=c++17", "-Wall", "-Wextra", "-pthread", "-I", str(root / "src"), - str(root / "tests/cpp/test_cut_latency.cpp"), "-o", str(binary)], check=True) - subprocess.run([str(binary)], check=True, timeout=10) +def test_cut_latency(cpp_binary): + subprocess.run([str(cpp_binary("test_cut_latency", pthread=True))], check=True, timeout=10) diff --git a/tests/test_hdr_metadata.py b/tests/test_hdr_metadata.py new file mode 100644 index 00000000..60f46168 --- /dev/null +++ b/tests/test_hdr_metadata.py @@ -0,0 +1,6 @@ +"""HDR10 static metadata serialised onto an encoder context; needs libavcodec headers (no GPU).""" +import subprocess + + +def test_hdr_metadata(cpp_binary): + subprocess.run([str(cpp_binary("test_hdr_metadata", libs=("libavcodec", "libavutil")))], check=True, timeout=10) diff --git a/tests/test_mixer_color.py b/tests/test_mixer_color.py new file mode 100644 index 00000000..327c12fd --- /dev/null +++ b/tests/test_mixer_color.py @@ -0,0 +1,182 @@ +"""Color contracts and real Python builder topology, without a GPU runtime.""" +import importlib +import itertools +import sys +import types +from pathlib import Path + +import pytest + +from pyplumber.mixer.color import Color, conversion_graph, declared_color, rendition_color +from pyplumber.mixer.config import ConfigError, parse + + +@pytest.mark.parametrize("source,target", list(itertools.product(("sdr", "hlg", "pq"), repeat=2))) +def test_declared_sources_convert_only_when_the_contract_differs(source, target): + fmt = "nv12" if target == "sdr" else "p010le" + graph = conversion_graph(target, fmt, source=source) + assert graph.startswith(Color(source).setparams + ",") + if source == target: + # Identity: stamp the tags, no tone-map pass (it would force 4:2:0). + assert graph == Color(source).setparams + f",scale_cuda=format={fmt}" + assert conversion_graph(target, "p210le", source=source, source_format="p210le") == Color(source).setparams + else: + assert f"transfer_in=auto:transfer_out={target}:format={fmt}" in graph + assert ":tonemap=clip:sdr_white=203:hdr_peak=1000:desat=0" in graph + assert "scale_cuda" not in graph + + +def test_untagged_source_is_never_inferred_from_storage(): + assert declared_color({}) is None + for fmt in ("nv12", "p010le"): + graph = conversion_graph("sdr", "nv12", source_format=fmt) + assert "setparams" not in graph + assert "transfer_in=auto" in graph + with pytest.raises(ValueError, match="incomplete"): + declared_color({"color_trc": "arib-std-b67"}) + with pytest.raises(ValueError, match="contradict"): + declared_color({"color": "hlg", "color_trc": "bt709"}) + + +@pytest.mark.parametrize("tags", [ + {**Color("hlg").tags, "color_primaries": "bt709"}, + {**Color().tags, "color_range": "pc"}, + {**Color().tags, "colorspace": "smpte170m"}, + {**Color().tags, "color_trc": "unknown"}, +]) +def test_unsupported_contracts_fail_instead_of_retagging(tags): + with pytest.raises(ValueError, match="unsupported|contradictory"): + Color.parse(tags) + + +def test_tonemap_converts_storage_in_the_same_pass(): + # 4:2:2 stays inside tonemap_cuda: P210 in, NV12/P210 out, no scale_cuda round trip. + assert conversion_graph("sdr", "nv12", source_format="p210le") == \ + "tonemap_cuda=transfer_in=auto:transfer_out=sdr:format=nv12:tonemap=clip:sdr_white=203:hdr_peak=1000:desat=0" + assert conversion_graph("hlg", "p210le").endswith(":format=p210le:tonemap=clip:sdr_white=203:hdr_peak=1000:desat=0") + assert "scale_cuda" not in conversion_graph("hlg", "p210le", source="sdr", source_format="nv12") + # Planar CUDA storage is re-laid out once before the tone mapper. + assert conversion_graph("sdr", "nv12", source_format="yuv422p10le").startswith("scale_cuda=format=p210le,tonemap_cuda=") + with pytest.raises(ValueError, match="10-bit"): + conversion_graph("hlg", "nv12") + with pytest.raises(ValueError, match="source pixel format"): + conversion_graph("hlg", "p010le", source_format="rgba") + + +@pytest.mark.parametrize("canvas", ("sdr", "hlg", "pq")) +def test_output_codec_defaults_have_correct_color(canvas): + assert rendition_color(canvas, "h264_nvenc") == Color("sdr") + assert rendition_color(canvas, "hevc_nvenc") == Color(canvas) + assert rendition_color(canvas, "hevc_nvenc", "pq") == Color("pq") + with pytest.raises(ValueError, match="H.264"): + rendition_color(canvas, "h264_nvenc", "hlg") + + +def test_raw_inputs_require_a_declaration_and_video_retains_auto(): + doc = {"canvas": {"width": 1920, "height": 1080}, + "sources": [{"id": "x", "kind": "video", "path": ""}], + "scenes": [{"id": "full", "items": [{"source": "x", "dst": {"x": 0, "y": 0, "w": 1920, "h": 1080}}]}]} + assert parse(doc).sources[0].color is None + for kind in ("browser", "v210"): + source = {"id": "x", "kind": kind, "path": "", "url": "https://example.org", "width": 1920, "height": 1080} + with pytest.raises(ConfigError, match="explicit color setting"): + parse({**doc, "sources": [source]}) + assert parse({**doc, "sources": [{**source, "color": "sdr"}]}).sources[0].color == Color() + + +@pytest.fixture +def builder(monkeypatch): + graph = importlib.import_module("pyplumber.mixer.graph") + class Avp: + def __init__(self): + self.nodes = [] + def addNode(self, node): + self.nodes.append(node.parameters) + def executeCommandsFromString(self, command): + pass + avp = Avp() + return graph.MixerGraphBuilder(avp, canvas=(1920, 1080), fps=(60, 1), + working_format="p010le", color="hlg", enable_wipe=False) + + +def test_aliases_share_one_color_conversion_before_all_scene_slots(builder): + for name in ("cam", "cam#2"): + builder.add_source(name, "decoded", "input", default_graph="", color="sdr") + builder.add_scene("full", {"cam": {}, "cam#2": {}}) + builder.set_initial_scene("full") + builder.build() + nodes = {n["name"]: n for n in builder.avp.nodes} + conversions = [n for n in nodes.values() if "tonemap_cuda" in n.get("graph", "")] + assert len(conversions) == 1 + assert conversions[0]["src"] == "decoded" + fanout = nodes["mixer_color_alias_cam"] + assert fanout["src"] == conversions[0]["dst"] + assert len(fanout["dst"]) == 2 + for name, edge in zip(("cam", "cam#2"), fanout["dst"]): + assert nodes[f"mixer_otm_{name}"]["src"] == edge + assert nodes["mixer_comp_a"]["color"] == "hlg" + + +@pytest.mark.parametrize("color", (None, "sdr", "hlg", "pq")) +def test_routed_color_contract_reaches_both_slot_conversions(builder, color): + builder.add_routed_source("cam", "route_a", "route_b", "input", "router", "a", "b", + default_graph="", color=color) + builder.add_scene("full", {"cam": {}}, routes={"cam": 0}) + builder.set_initial_scene("full") + builder.build() + nodes = {n["name"]: n for n in builder.avp.nodes} + expected = conversion_graph("hlg", "p010le", source=color) + for slot in ("a", "b"): + conversion = nodes[f"mixer_color_cam_{slot}"] + assert conversion["src"] == f"route_{slot}" + assert conversion["graph"] == expected + assert nodes[f"mixer_comp_{slot}"]["src"] == [conversion["dst"]] + + +def test_shared_edge_cannot_have_conflicting_contracts(builder): + builder.add_source("a", "decoded", "input", color="sdr") + builder.add_source("b", "decoded", "input", color="pq") + builder.add_scene("full", {"a": {}, "b": {}}) + builder.set_initial_scene("full") + with pytest.raises(ValueError, match="Conflicting color"): + builder.build() + + +def test_rgb_keeps_alpha_and_uses_hdr_compositor(builder): + builder.add_source("page", "rgba", "input", default_graph="", packed_rgb=True, color="sdr") + builder.add_scene("full", {"page": {"blend": True}}) + builder.set_initial_scene("full") + builder.build() + nodes = {n["name"]: n for n in builder.avp.nodes} + assert nodes["mixer_color_page"]["graph"] == Color().setparams + assert nodes["mixer_comp_a"]["color"] == "hlg" + assert nodes["mixer_comp_a"]["layers"][0]["blend"] + + +@pytest.mark.parametrize("wipe_color", (None, "sdr")) +def test_media_wipe_blends_onto_an_hdr_canvas(monkeypatch, wipe_color): + """Auto wipes retain tags for compositor validation; an explicit SDR override + tags the clip before upload. Both retain the RGBA alpha-blend path.""" + graph = importlib.import_module("pyplumber.mixer.graph") + class Avp: + def __init__(self): + self.nodes = [] + def addNode(self, node): + self.nodes.append(node.parameters) + def executeCommandsFromString(self, command): + pass + b = graph.MixerGraphBuilder(Avp(), canvas=(1920, 1080), fps=(60, 1), working_format="p210le", + color="hlg", wipe_color=wipe_color, enable_wipe=True, cache_wipes_mb=512) + b.add_source("cam", "decoded", "input", default_graph="", color="sdr") + b.add_scene("full", {"cam": {}}) + b.set_initial_scene("full") + b.build() + nodes = {n["name"]: n for n in b.avp.nodes} + assert nodes["mixer_wipe_fmt"]["graph"] == (Color().setparams + "," if wipe_color else "") + "format=rgba,hwupload" + overlay = nodes["mixer_wipe_overlay"] + assert overlay["sw_format"] == "p210le" and overlay["color"] == "hlg" + assert overlay["layers"][1]["blend"] is True and overlay["active_inputs"] == 3 + assert nodes["mixer_wipe_cache"]["src"] == "mixer_wipe_rt_out" + with pytest.raises(ValueError, match="require SDR"): + graph.MixerGraphBuilder(Avp(), canvas=(1920, 1080), fps=(60, 1), working_format="p210le", + color="hlg", wipe_color="hlg") diff --git a/tests/test_mixer_playout.py b/tests/test_mixer_playout.py index 01d31b5d..af13e333 100644 --- a/tests/test_mixer_playout.py +++ b/tests/test_mixer_playout.py @@ -8,16 +8,20 @@ @pytest.fixture(scope="module") def playout_binary(tmp_path_factory): - tmp_path = tmp_path_factory.mktemp("mixer-playout") compiler = shutil.which('g++') or shutil.which('clang++') if not compiler: pytest.skip('no C++ compiler available') root = pathlib.Path(__file__).resolve().parents[1] - binary = tmp_path / 'mixer_playout' + # TickGrid takes an av::Rational, whose constructor lives in avcpp's rational.cpp. + probe = subprocess.run(['pkg-config', '--cflags', '--libs', 'libavutil'], capture_output=True, text=True) + if probe.returncode != 0: + pytest.skip('libavutil development files not found') + binary = tmp_path_factory.mktemp("mixer-playout") / 'mixer_playout' subprocess.run([ compiler, '-std=c++17', '-O0', '-g', '-Wall', '-Wextra', - '-I', str(root / 'src'), str(root / 'tests/cpp/test_mixer_playout.cpp'), - '-o', str(binary), + '-I', str(root / 'src'), '-I', str(root / 'deps/avcpp/src'), '-I', str(root / 'deps/include'), + str(root / 'tests/cpp/test_mixer_playout.cpp'), str(root / 'deps/avcpp/src/rational.cpp'), + '-o', str(binary), *probe.stdout.split(), ], check=True, capture_output=True, text=True) return binary diff --git a/tests/test_mixer_snapshot.py b/tests/test_mixer_snapshot.py index 201a1cf5..91c4b149 100644 --- a/tests/test_mixer_snapshot.py +++ b/tests/test_mixer_snapshot.py @@ -1,14 +1,6 @@ """Exact-picture retention and stale-frame rejection, without GPU dependencies.""" -import pathlib -import shutil import subprocess -def test_mixer_snapshot(tmp_path): - root = pathlib.Path(__file__).resolve().parents[1] - binary = tmp_path / 'mixer_snapshot' - compiler = shutil.which('g++') or shutil.which('clang++') - assert compiler, 'a C++ compiler is required' - subprocess.run([compiler, '-std=c++17', '-Wall', '-Wextra', '-I', str(root / 'src'), - str(root / 'tests/cpp/test_mixer_snapshot.cpp'), '-o', str(binary)], check=True) - subprocess.run([str(binary)], check=True, timeout=10) +def test_mixer_snapshot(cpp_binary): + subprocess.run([str(cpp_binary("test_mixer_snapshot"))], check=True, timeout=10) diff --git a/tests/test_pixel_layout.py b/tests/test_pixel_layout.py new file mode 100644 index 00000000..8d3d7565 --- /dev/null +++ b/tests/test_pixel_layout.py @@ -0,0 +1,6 @@ +"""Pixel-format geometry used by the CUDA compositor; needs libavutil headers only (no GPU).""" +import subprocess + + +def test_pixel_layout(cpp_binary): + subprocess.run([str(cpp_binary("test_pixel_layout", libs=("libavutil",)))], check=True, timeout=10)