From 06d3be87d77496a95f405c2dc7ea096263a27d19 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 10:00:47 +0200 Subject: [PATCH 01/11] Build: add isolated FFmpeg 8.1 compatibility patch series --- .dockerignore | 2 + demos/cuda-overlay/Dockerfile | 5 +- demos/cuda-overlay/tests/test_run_matrix.py | 36 +- demos/cuda-overlay/tools/run_matrix.py | 12 +- demos/dmabuf-browser/consumer/Dockerfile.cuda | 4 +- demos/mixer/Dockerfile | 8 +- demos/mixer/docs/guide.md | 2 +- .../0001-swscale-aarch64-argb-yuva420p.patch | 0 ...0002-avfilter-cuda-composition-suite.patch | 0 .../0003-avfilter-npp-cuda13-compat.patch | 0 .../7.1.5}/0004-avcodec-nvdec-intra.patch | 0 .../7.1.5}/0005-avformat-rtp-rfc4175.patch | 0 .../7.1.5}/0006-avdevice-v4l2-compat.patch | 0 .../7.1.5}/0007-avdevice-ndi-v5.patch | 0 .../7.1.5}/README.md | 4 +- deps/ffmpeg/7.1.5/base.env | 3 + deps/ffmpeg/7.1.5/verify.sh | 4 + .../0001-swscale-aarch64-argb-yuva420p.patch | 447 ++ ...0002-avfilter-cuda-composition-suite.patch | 4468 +++++++++++++++++ .../8.1/0003-avfilter-npp-cuda13-compat.patch | 306 ++ .../ffmpeg/8.1/0004-avcodec-nvdec-intra.patch | 27 + .../8.1/0005-avformat-rtp-rfc4175.patch | 168 + .../8.1/0006-avdevice-v4l2-compat.patch | 25 + deps/ffmpeg/8.1/0007-avdevice-ndi-v5.patch | 219 + deps/ffmpeg/8.1/README.md | 61 + deps/ffmpeg/8.1/base.env | 3 + deps/ffmpeg/8.1/verify.sh | 4 + deps/ffmpeg/README.md | 37 + deps/{ffmpeg-patches => ffmpeg}/verify.sh | 24 +- 29 files changed, 5841 insertions(+), 28 deletions(-) rename deps/{ffmpeg-patches => ffmpeg/7.1.5}/0001-swscale-aarch64-argb-yuva420p.patch (100%) rename deps/{ffmpeg-patches => ffmpeg/7.1.5}/0002-avfilter-cuda-composition-suite.patch (100%) rename deps/{ffmpeg-patches => ffmpeg/7.1.5}/0003-avfilter-npp-cuda13-compat.patch (100%) rename deps/{ffmpeg-patches => ffmpeg/7.1.5}/0004-avcodec-nvdec-intra.patch (100%) rename deps/{ffmpeg-patches => ffmpeg/7.1.5}/0005-avformat-rtp-rfc4175.patch (100%) rename deps/{ffmpeg-patches => ffmpeg/7.1.5}/0006-avdevice-v4l2-compat.patch (100%) rename deps/{ffmpeg-patches => ffmpeg/7.1.5}/0007-avdevice-ndi-v5.patch (100%) rename deps/{ffmpeg-patches => ffmpeg/7.1.5}/README.md (94%) create mode 100644 deps/ffmpeg/7.1.5/base.env create mode 100755 deps/ffmpeg/7.1.5/verify.sh create mode 100644 deps/ffmpeg/8.1/0001-swscale-aarch64-argb-yuva420p.patch create mode 100644 deps/ffmpeg/8.1/0002-avfilter-cuda-composition-suite.patch create mode 100644 deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch create mode 100644 deps/ffmpeg/8.1/0004-avcodec-nvdec-intra.patch create mode 100644 deps/ffmpeg/8.1/0005-avformat-rtp-rfc4175.patch create mode 100644 deps/ffmpeg/8.1/0006-avdevice-v4l2-compat.patch create mode 100644 deps/ffmpeg/8.1/0007-avdevice-ndi-v5.patch create mode 100644 deps/ffmpeg/8.1/README.md create mode 100644 deps/ffmpeg/8.1/base.env create mode 100755 deps/ffmpeg/8.1/verify.sh create mode 100644 deps/ffmpeg/README.md rename deps/{ffmpeg-patches => ffmpeg}/verify.sh (70%) diff --git a/.dockerignore b/.dockerignore index 0b219701..a575e575 100644 --- a/.dockerignore +++ b/.dockerignore @@ -8,6 +8,8 @@ tmp/ .vscode/ .rsync-backup-*/ ffmpeg/ +!deps/ffmpeg/ +!deps/ffmpeg/** *.log *.csv *.ts diff --git a/demos/cuda-overlay/Dockerfile b/demos/cuda-overlay/Dockerfile index a9ca27c8..4c882a55 100644 --- a/demos/cuda-overlay/Dockerfile +++ b/demos/cuda-overlay/Dockerfile @@ -35,13 +35,13 @@ RUN git clone --quiet --branch "${NV_CODEC_HEADERS_TAG}" --depth 1 \ && make -C /tmp/nv-codec-headers install PREFIX=/usr/local \ && rm -rf /tmp/nv-codec-headers -COPY deps/ffmpeg-patches /build/deps/ffmpeg-patches +COPY deps/ffmpeg /build/deps/ffmpeg RUN git clone --quiet --branch "${FFMPEG_TAG}" --depth 1 \ https://github.com/FFmpeg/FFmpeg.git /tmp/ffmpeg \ && git -C /tmp/ffmpeg config user.name "cuda-overlay-demo builder" \ && git -C /tmp/ffmpeg config user.email "cuda-overlay-demo@local" \ - && git -C /tmp/ffmpeg am /build/deps/ffmpeg-patches/*.patch + && git -C /tmp/ffmpeg am /build/deps/ffmpeg/${FFMPEG_TAG#n}/*.patch RUN cd /tmp/ffmpeg \ && ./configure \ @@ -117,6 +117,7 @@ COPY demos/cuda-overlay /build/demos/cuda-overlay ENV LD_LIBRARY_PATH=/usr/local/lib:/usr/local/cuda/lib64 ENV AVPLUMBER_REVISION=${AVPLUMBER_REVISION} +ENV FFMPEG_TAG=${FFMPEG_TAG} ENV NVIDIA_DRIVER_CAPABILITIES=compute,utility,video ENV NVIDIA_VISIBLE_DEVICES=all ENV PYPLUMBER_PATH=/build diff --git a/demos/cuda-overlay/tests/test_run_matrix.py b/demos/cuda-overlay/tests/test_run_matrix.py index 51097c4a..f998ba99 100644 --- a/demos/cuda-overlay/tests/test_run_matrix.py +++ b/demos/cuda-overlay/tests/test_run_matrix.py @@ -17,13 +17,47 @@ class RunMatrixTest(unittest.TestCase): def test_patch_identity_reads_the_repository_patch_series(self) -> None: expected_paths = sorted( - (REPOSITORY_DIR / "deps" / "ffmpeg-patches").glob("*.patch") + (REPOSITORY_DIR / "deps" / "ffmpeg" / "7.1.5").glob("*.patch") ) self.assertEqual(REPO_DIR, REPOSITORY_DIR) self.assertEqual(len(expected_paths), 7) self.assertEqual(list(_patch_identity()), [path.name for path in expected_paths]) + def test_patch_identity_selects_ffmpeg81(self) -> None: + expected_paths = sorted( + (REPOSITORY_DIR / "deps" / "ffmpeg" / "8.1").glob("*.patch") + ) + self.assertEqual(len(expected_paths), 7) + self.assertEqual(list(_patch_identity("n8.1")), [path.name for path in expected_paths]) + self.assertNotEqual(_patch_identity(), _patch_identity("n8.1")) + + def test_patch_identity_rejects_unknown_tags(self) -> None: + with self.assertRaises(ValueError): + _patch_identity("n8.0") + + def test_cuda_images_keep_the_default_and_select_matching_patches(self) -> None: + for relative_path in ( + "demos/cuda-overlay/Dockerfile", + "demos/mixer/Dockerfile", + "demos/dmabuf-browser/consumer/Dockerfile.cuda", + ): + with self.subTest(dockerfile=relative_path): + dockerfile = (REPOSITORY_DIR / relative_path).read_text() + self.assertIn("ARG FFMPEG_TAG=n7.1.5", dockerfile) + self.assertIn("COPY deps/ffmpeg /build/deps/ffmpeg", dockerfile) + self.assertIn("/build/deps/ffmpeg/${FFMPEG_TAG#n}/*.patch", dockerfile) + + def test_overlay_image_reports_the_selected_version(self) -> None: + dockerfile = (REPOSITORY_DIR / "demos/cuda-overlay/Dockerfile").read_text() + self.assertIn("ENV FFMPEG_TAG=${FFMPEG_TAG}", dockerfile) + + def test_mixer_checks_runtime_options_without_removed_command_marker(self) -> None: + dockerfile = (REPOSITORY_DIR / "demos/mixer/Dockerfile").read_text() + self.assertIn("-h filter=transition_cuda", dockerfile) + self.assertIn('$1 == "mode" && $3 ~ /T/', dockerfile) + self.assertNotIn('$1 ~ /C/', dockerfile) + if __name__ == "__main__": unittest.main() diff --git a/demos/cuda-overlay/tools/run_matrix.py b/demos/cuda-overlay/tools/run_matrix.py index 30b50f54..aac728d2 100755 --- a/demos/cuda-overlay/tools/run_matrix.py +++ b/demos/cuda-overlay/tools/run_matrix.py @@ -33,9 +33,11 @@ def _command_text(command: list[str]) -> str: return output if output else f"exit {result.returncode}" -def _patch_identity() -> dict[str, str]: +def _patch_identity(ffmpeg_tag: str = "n7.1.5") -> dict[str, str]: + if ffmpeg_tag not in ("n7.1.5", "n8.1"): + raise ValueError(f"unsupported FFmpeg tag: {ffmpeg_tag}") identities: dict[str, str] = {} - for path in sorted((REPO_DIR / "deps" / "ffmpeg-patches").glob("*.patch")): + for path in sorted((REPO_DIR / "deps" / "ffmpeg" / ffmpeg_tag[1:]).glob("*.patch")): identities[path.name] = hashlib.sha256(path.read_bytes()).hexdigest() return identities @@ -78,6 +80,8 @@ def main() -> int: parser.add_argument("--height", type=_positive_dimension, default=HEIGHT) args = parser.parse_args() + ffmpeg_tag = os.environ.get("FFMPEG_TAG", "n7.1.5") + started_at = dt.datetime.now(dt.timezone.utc) run_id = started_at.strftime("run-%Y%m%dT%H%M%SZ") run_dir = args.artifacts / run_id @@ -93,7 +97,7 @@ def main() -> int: "run_id": run_id, "started_at": started_at.isoformat(), "status": "running", - "ffmpeg_tag": "n7.1.5", + "ffmpeg_tag": ffmpeg_tag, "cuda_toolkit_minimum": "11.7", "dimensions": {"width": args.width, "height": args.height}, "requested_overlay_counts": args.counts, @@ -110,7 +114,7 @@ def main() -> int: ] ), "repository_commit": os.environ.get("AVPLUMBER_REVISION", "workspace"), - "patches_sha256": _patch_identity(), + "patches_sha256": _patch_identity(ffmpeg_tag), }, "cases": [], } diff --git a/demos/dmabuf-browser/consumer/Dockerfile.cuda b/demos/dmabuf-browser/consumer/Dockerfile.cuda index 010a443b..32e46b09 100644 --- a/demos/dmabuf-browser/consumer/Dockerfile.cuda +++ b/demos/dmabuf-browser/consumer/Dockerfile.cuda @@ -50,13 +50,13 @@ RUN git clone --quiet --branch "${NV_CODEC_HEADERS_TAG}" --depth 1 \ && make -C /tmp/nv-codec-headers install PREFIX=/usr/local \ && rm -rf /tmp/nv-codec-headers -COPY deps/ffmpeg-patches /build/deps/ffmpeg-patches +COPY deps/ffmpeg /build/deps/ffmpeg RUN git clone --quiet --branch "${FFMPEG_TAG}" --depth 1 \ https://github.com/FFmpeg/FFmpeg.git /tmp/ffmpeg \ && git -C /tmp/ffmpeg config user.name "dmabuf-browser-demo builder" \ && git -C /tmp/ffmpeg config user.email "dmabuf-browser-demo@local" \ - && git -C /tmp/ffmpeg am /build/deps/ffmpeg-patches/*.patch + && git -C /tmp/ffmpeg am /build/deps/ffmpeg/${FFMPEG_TAG#n}/*.patch RUN cd /tmp/ffmpeg \ && ./configure \ diff --git a/demos/mixer/Dockerfile b/demos/mixer/Dockerfile index 01697f3b..78578bbe 100644 --- a/demos/mixer/Dockerfile +++ b/demos/mixer/Dockerfile @@ -36,13 +36,13 @@ RUN git clone --quiet --branch "${NV_CODEC_HEADERS_TAG}" --depth 1 \ && make -C /tmp/nv-codec-headers install PREFIX=/usr/local \ && rm -rf /tmp/nv-codec-headers -COPY deps/ffmpeg-patches /build/deps/ffmpeg-patches +COPY deps/ffmpeg /build/deps/ffmpeg RUN git clone --quiet --branch "${FFMPEG_TAG}" --depth 1 \ https://github.com/FFmpeg/FFmpeg.git /tmp/ffmpeg \ && git -C /tmp/ffmpeg config user.name "mixer-demo builder" \ && git -C /tmp/ffmpeg config user.email "mixer-demo@local" \ - && git -C /tmp/ffmpeg am /build/deps/ffmpeg-patches/*.patch + && git -C /tmp/ffmpeg am /build/deps/ffmpeg/${FFMPEG_TAG#n}/*.patch RUN cd /tmp/ffmpeg \ && ./configure \ @@ -77,8 +77,8 @@ RUN cd /tmp/ffmpeg \ && rm -rf /tmp/ffmpeg RUN /usr/local/bin/ffmpeg -hide_banner -filters | grep -q ' overlay_many_cuda ' \ - && /usr/local/bin/ffmpeg -hide_banner -filters \ - | awk '$2 == "transition_cuda" && $1 ~ /C/ { found = 1 } END { exit !found }' + && /usr/local/bin/ffmpeg -hide_banner -h filter=transition_cuda \ + | awk '$1 == "mode" && $3 ~ /T/ { found = 1 } END { exit !found }' WORKDIR /build diff --git a/demos/mixer/docs/guide.md b/demos/mixer/docs/guide.md index c8bb8cdd..55bdc5e8 100644 --- a/demos/mixer/docs/guide.md +++ b/demos/mixer/docs/guide.md @@ -195,7 +195,7 @@ groups without reordering sources. Follow the [shared NVIDIA setup guide](../../README.md) first. It also provides a local Janus preview if you want WebRTC output. -The demo image builds FFmpeg 7.1 with `deps/ffmpeg-patches`, verifies the +The demo image defaults to FFmpeg 7.1.5 with `deps/ffmpeg/7.1.5`, verifies the patched CUDA overlay and transition filters, and builds the CUDA-enabled AVPlumber Python module against that FFmpeg installation: diff --git a/deps/ffmpeg-patches/0001-swscale-aarch64-argb-yuva420p.patch b/deps/ffmpeg/7.1.5/0001-swscale-aarch64-argb-yuva420p.patch similarity index 100% rename from deps/ffmpeg-patches/0001-swscale-aarch64-argb-yuva420p.patch rename to deps/ffmpeg/7.1.5/0001-swscale-aarch64-argb-yuva420p.patch diff --git a/deps/ffmpeg-patches/0002-avfilter-cuda-composition-suite.patch b/deps/ffmpeg/7.1.5/0002-avfilter-cuda-composition-suite.patch similarity index 100% rename from deps/ffmpeg-patches/0002-avfilter-cuda-composition-suite.patch rename to deps/ffmpeg/7.1.5/0002-avfilter-cuda-composition-suite.patch diff --git a/deps/ffmpeg-patches/0003-avfilter-npp-cuda13-compat.patch b/deps/ffmpeg/7.1.5/0003-avfilter-npp-cuda13-compat.patch similarity index 100% rename from deps/ffmpeg-patches/0003-avfilter-npp-cuda13-compat.patch rename to deps/ffmpeg/7.1.5/0003-avfilter-npp-cuda13-compat.patch diff --git a/deps/ffmpeg-patches/0004-avcodec-nvdec-intra.patch b/deps/ffmpeg/7.1.5/0004-avcodec-nvdec-intra.patch similarity index 100% rename from deps/ffmpeg-patches/0004-avcodec-nvdec-intra.patch rename to deps/ffmpeg/7.1.5/0004-avcodec-nvdec-intra.patch diff --git a/deps/ffmpeg-patches/0005-avformat-rtp-rfc4175.patch b/deps/ffmpeg/7.1.5/0005-avformat-rtp-rfc4175.patch similarity index 100% rename from deps/ffmpeg-patches/0005-avformat-rtp-rfc4175.patch rename to deps/ffmpeg/7.1.5/0005-avformat-rtp-rfc4175.patch diff --git a/deps/ffmpeg-patches/0006-avdevice-v4l2-compat.patch b/deps/ffmpeg/7.1.5/0006-avdevice-v4l2-compat.patch similarity index 100% rename from deps/ffmpeg-patches/0006-avdevice-v4l2-compat.patch rename to deps/ffmpeg/7.1.5/0006-avdevice-v4l2-compat.patch diff --git a/deps/ffmpeg-patches/0007-avdevice-ndi-v5.patch b/deps/ffmpeg/7.1.5/0007-avdevice-ndi-v5.patch similarity index 100% rename from deps/ffmpeg-patches/0007-avdevice-ndi-v5.patch rename to deps/ffmpeg/7.1.5/0007-avdevice-ndi-v5.patch diff --git a/deps/ffmpeg-patches/README.md b/deps/ffmpeg/7.1.5/README.md similarity index 94% rename from deps/ffmpeg-patches/README.md rename to deps/ffmpeg/7.1.5/README.md index b4633f24..b912dcb7 100644 --- a/deps/ffmpeg-patches/README.md +++ b/deps/ffmpeg/7.1.5/README.md @@ -38,7 +38,7 @@ git clone --branch n7.1.5 --depth 1 \ https://github.com/FFmpeg/FFmpeg clean-ffmpeg git -C clean-ffmpeg config user.name "patch application" git -C clean-ffmpeg config user.email "patch-application@local" -git -C clean-ffmpeg am /path/to/avplumber/deps/ffmpeg-patches/*.patch +git -C clean-ffmpeg am /path/to/avplumber/deps/ffmpeg/7.1.5/*.patch ``` ## Verify @@ -48,7 +48,7 @@ commit. It creates and removes an isolated temporary worktree; it does not alter the checkout's active branch: ```bash -deps/ffmpeg-patches/verify.sh /path/to/FFmpeg +deps/ffmpeg/7.1.5/verify.sh /path/to/FFmpeg ``` Verification succeeds only when all seven patches apply and produce the exact diff --git a/deps/ffmpeg/7.1.5/base.env b/deps/ffmpeg/7.1.5/base.env new file mode 100644 index 00000000..ceb83910 --- /dev/null +++ b/deps/ffmpeg/7.1.5/base.env @@ -0,0 +1,3 @@ +base_commit=3a0867c2bfda4a4d4309ca1a8cbdc6175e67f587 +expected_tree=52361f7251069ef74fbb41460e6e1b65d6f9947c +expected_patch_count=7 diff --git a/deps/ffmpeg/7.1.5/verify.sh b/deps/ffmpeg/7.1.5/verify.sh new file mode 100755 index 00000000..1256df3c --- /dev/null +++ b/deps/ffmpeg/7.1.5/verify.sh @@ -0,0 +1,4 @@ +#!/usr/bin/env bash +set -euo pipefail +script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) +exec bash "$script_dir/../verify.sh" 7.1.5 "$@" diff --git a/deps/ffmpeg/8.1/0001-swscale-aarch64-argb-yuva420p.patch b/deps/ffmpeg/8.1/0001-swscale-aarch64-argb-yuva420p.patch new file mode 100644 index 00000000..c6e06fb6 --- /dev/null +++ b/deps/ffmpeg/8.1/0001-swscale-aarch64-argb-yuva420p.patch @@ -0,0 +1,447 @@ +From e65731e98a015ad5b8888367a7929f91fbcba4c9 Mon Sep 17 00:00:00 2001 +From: Jan Pietek +Date: Tue, 15 Sep 2026 09:35:57 +0200 +Subject: [PATCH 1/7] swscale/aarch64: port ARGB to YUVA420P fast paths to 8.1 + +--- + libswscale/aarch64/input.S | 32 ++++ + libswscale/aarch64/swscale.c | 6 + + libswscale/aarch64/swscale_unscaled.c | 125 +++++++++++++ + libswscale/aarch64/swscale_unscaled_neon.S | 200 +++++++++++++++++++++ + 4 files changed, 363 insertions(+) + +diff --git a/libswscale/aarch64/input.S b/libswscale/aarch64/input.S +index c1c0adffc8..a671a2ed60 100644 +--- a/libswscale/aarch64/input.S ++++ b/libswscale/aarch64/input.S +@@ -314,6 +314,38 @@ rgbToUV_neon bgra32, rgba32, element=4 + + rgbToUV_neon abgr32, argb32, element=4, alpha_first=1 + ++// dst = (A << 6) | (A >> 2), as in abgrToA_c. ++function ff_argb32ToA_neon, export=1 ++ cmp w4, #0 ++ b.le 3f ++ cmp w4, #16 ++ b.lt 2f ++1: ++ ld4 {v16.16b, v17.16b, v18.16b, v19.16b}, [x1], #64 ++ uxtl v20.8h, v16.8b ++ uxtl2 v21.8h, v16.16b ++ shl v22.8h, v20.8h, #6 ++ shl v23.8h, v21.8h, #6 ++ ushr v20.8h, v20.8h, #2 ++ ushr v21.8h, v21.8h, #2 ++ orr v22.16b, v22.16b, v20.16b ++ orr v23.16b, v23.16b, v21.16b ++ stp q22, q23, [x0], #32 ++ subs w4, w4, #16 ++ b.gt 1b ++ cbz w4, 3f ++2: ++ ldrb w8, [x1], #4 ++ subs w4, w4, #1 ++ lsl w9, w8, #6 ++ lsr w10, w8, #2 ++ orr w9, w9, w10 ++ strh w9, [x0], #2 ++ cbnz w4, 2b ++3: ++ ret ++endfunc ++ + #if HAVE_DOTPROD + ENABLE_DOTPROD + +diff --git a/libswscale/aarch64/swscale.c b/libswscale/aarch64/swscale.c +index 4f86364f4b..3844db23bd 100644 +--- a/libswscale/aarch64/swscale.c ++++ b/libswscale/aarch64/swscale.c +@@ -291,6 +291,8 @@ NEON_INPUT(bgr24); + NEON_INPUT(bgra32); + NEON_INPUT(rgb24); + NEON_INPUT(rgba32); ++void ff_argb32ToA_neon(uint8_t *dst, const uint8_t *src, const uint8_t *, ++ const uint8_t *, int w, uint32_t *coeffs, void *opq); + NEON_INPUT_DOTPROD(bgra32); + NEON_INPUT_DOTPROD(rgba32); + +@@ -382,6 +384,8 @@ av_cold void ff_sws_init_swscale_aarch64(SwsInternal *c) + c->chrToYV12 = ff_abgr32ToUV_half_neon; + else + c->chrToYV12 = ff_abgr32ToUV_neon; ++ if (c->needAlpha) ++ c->alpToYV12 = ff_argb32ToA_neon; + break; + + case AV_PIX_FMT_ARGB: +@@ -390,6 +394,8 @@ av_cold void ff_sws_init_swscale_aarch64(SwsInternal *c) + c->chrToYV12 = ff_argb32ToUV_half_neon; + else + c->chrToYV12 = ff_argb32ToUV_neon; ++ if (c->needAlpha) ++ c->alpToYV12 = ff_argb32ToA_neon; + break; + case AV_PIX_FMT_BGR24: + c->lumToYV12 = ff_bgr24ToY_neon; +diff --git a/libswscale/aarch64/swscale_unscaled.c b/libswscale/aarch64/swscale_unscaled.c +index ba24775210..75c3991ec0 100644 +--- a/libswscale/aarch64/swscale_unscaled.c ++++ b/libswscale/aarch64/swscale_unscaled.c +@@ -21,6 +21,13 @@ + #include "libswscale/swscale_internal.h" + #include "libavutil/aarch64/cpu.h" + ++void ff_argb_to_yuv420p_neon(const uint8_t *src, uint8_t *ydst, ++ uint8_t *udst, uint8_t *vdst, int width, ++ int height, int lumStride, int chromStride, ++ int srcStride, int32_t *rgb2yuv); ++void ff_argb_to_a8_neon(const uint8_t *src, uint8_t *dst, int width, ++ int height, int srcStride, int dstStride); ++ + #define YUV_TO_RGB_TABLE \ + c->yuv2rgb_v2r_coeff, \ + c->yuv2rgb_u2g_coeff, \ +@@ -205,6 +212,115 @@ static int nv24_to_yuv420p_neon_wrapper(SwsInternal *c, const uint8_t *const src + return srcSliceH; + } + ++static av_always_inline void argb_to_yuva420p_c(const uint8_t *src, uint8_t *ydst, ++ uint8_t *udst, uint8_t *vdst, ++ uint8_t *adst, int width, ++ int height, int lumStride, ++ int chromStride, int srcStride, ++ int alphaStride, int32_t *rgb2yuv) ++{ ++ const int32_t ry = rgb2yuv[RY_IDX], gy = rgb2yuv[GY_IDX], by = rgb2yuv[BY_IDX]; ++ const int32_t ru = rgb2yuv[RU_IDX], gu = rgb2yuv[GU_IDX], bu = rgb2yuv[BU_IDX]; ++ const int32_t rv = rgb2yuv[RV_IDX], gv = rgb2yuv[GV_IDX], bv = rgb2yuv[BV_IDX]; ++ const int chromWidth = width >> 1; ++ const uint8_t *src1 = src; ++ const uint8_t *src2 = src + srcStride; ++ uint8_t *ydst1 = ydst; ++ uint8_t *ydst2 = ydst + lumStride; ++ uint8_t *adst1 = adst; ++ uint8_t *adst2 = adst + alphaStride; ++ ++ for (int y = 0; y < height; y += 2) { ++ if (y + 1 == height) { ++ ydst2 = ydst1; ++ adst2 = adst1; ++ src2 = src1; ++ } ++ ++ for (int i = 0; i < chromWidth; i++) { ++ const unsigned int a11 = src1[8 * i + 0]; ++ const unsigned int r11 = src1[8 * i + 1]; ++ const unsigned int g11 = src1[8 * i + 2]; ++ const unsigned int b11 = src1[8 * i + 3]; ++ const unsigned int a12 = src1[8 * i + 4]; ++ const unsigned int r12 = src1[8 * i + 5]; ++ const unsigned int g12 = src1[8 * i + 6]; ++ const unsigned int b12 = src1[8 * i + 7]; ++ const unsigned int a21 = src2[8 * i + 0]; ++ const unsigned int r21 = src2[8 * i + 1]; ++ const unsigned int g21 = src2[8 * i + 2]; ++ const unsigned int b21 = src2[8 * i + 3]; ++ const unsigned int a22 = src2[8 * i + 4]; ++ const unsigned int r22 = src2[8 * i + 5]; ++ const unsigned int g22 = src2[8 * i + 6]; ++ const unsigned int b22 = src2[8 * i + 7]; ++ ++ const unsigned int Y11 = ((ry * r11 + gy * g11 + by * b11) >> RGB2YUV_SHIFT) + 16; ++ const unsigned int Y12 = ((ry * r12 + gy * g12 + by * b12) >> RGB2YUV_SHIFT) + 16; ++ const unsigned int Y21 = ((ry * r21 + gy * g21 + by * b21) >> RGB2YUV_SHIFT) + 16; ++ const unsigned int Y22 = ((ry * r22 + gy * g22 + by * b22) >> RGB2YUV_SHIFT) + 16; ++ ++ const unsigned int bx = (b11 + b12 + b21 + b22) >> 2; ++ const unsigned int gx = (g11 + g12 + g21 + g22) >> 2; ++ const unsigned int rx = (r11 + r12 + r21 + r22) >> 2; ++ ++ const unsigned int U = ((ru * rx + gu * gx + bu * bx) >> RGB2YUV_SHIFT) + 128; ++ const unsigned int V = ((rv * rx + gv * gx + bv * bx) >> RGB2YUV_SHIFT) + 128; ++ ++ ydst1[2 * i + 0] = Y11; ++ ydst1[2 * i + 1] = Y12; ++ ydst2[2 * i + 0] = Y21; ++ ydst2[2 * i + 1] = Y22; ++ adst1[2 * i + 0] = a11; ++ adst1[2 * i + 1] = a12; ++ adst2[2 * i + 0] = a21; ++ adst2[2 * i + 1] = a22; ++ udst[i] = U; ++ vdst[i] = V; ++ } ++ ++ src1 += srcStride * 2; ++ src2 += srcStride * 2; ++ ydst1 += lumStride * 2; ++ ydst2 += lumStride * 2; ++ adst1 += alphaStride * 2; ++ adst2 += alphaStride * 2; ++ udst += chromStride; ++ vdst += chromStride; ++ } ++} ++ ++static int argb_to_yuva420p_neon_wrapper(SwsInternal *c, const uint8_t *const src[], ++ const int srcStride[], int srcSliceY, int srcSliceH, ++ uint8_t *const dst[], const int dstStride[]) ++{ ++ const int width_aligned = c->opts.src_w & ~15; ++ const uint8_t *src_ptr = src[0]; ++ uint8_t *ydst = dst[0] + srcSliceY * dstStride[0]; ++ uint8_t *udst = dst[1] + (srcSliceY >> 1) * dstStride[1]; ++ uint8_t *vdst = dst[2] + (srcSliceY >> 1) * dstStride[2]; ++ uint8_t *adst = dst[3] + srcSliceY * dstStride[3]; ++ ++ /* Keep strict gating in wrapper too, so odd slice boundaries or non-16 widths ++ * never hit the assembly fast path. */ ++ if ((srcSliceY & 1) || (srcSliceH & 1) || (c->opts.src_w & 15)) { ++ argb_to_yuva420p_c(src_ptr, ydst, udst, vdst, adst, ++ c->opts.src_w, srcSliceH, dstStride[0], dstStride[1], ++ srcStride[0], dstStride[3], c->input_rgb2yuv_table); ++ return srcSliceH; ++ } ++ ++ if (width_aligned > 0) { ++ ff_argb_to_yuv420p_neon(src_ptr, ydst, udst, vdst, width_aligned, ++ srcSliceH, dstStride[0], dstStride[1], ++ srcStride[0], c->input_rgb2yuv_table); ++ ff_argb_to_a8_neon(src_ptr, adst, width_aligned, srcSliceH, ++ srcStride[0], dstStride[3]); ++ } ++ ++ return srcSliceH; ++} ++ + #define DECLARE_FF_NVX_TO_ALL_RGBX_FUNCS(nvx) \ + DECLARE_FF_NVX_TO_RGBX_FUNCS(nvx, argb) \ + DECLARE_FF_NVX_TO_RGBX_FUNCS(nvx, rgba) \ +@@ -259,6 +375,15 @@ static void get_unscaled_swscale_neon(SwsInternal *c) { + (c->opts.src_format == AV_PIX_FMT_NV24 || c->opts.src_format == AV_PIX_FMT_NV42) && + !(c->opts.src_h & 1) && !(c->opts.src_w & 15) && !accurate_rnd) + c->convert_unscaled = nv24_to_yuv420p_neon_wrapper; ++ ++ if (c->opts.src_format == AV_PIX_FMT_ARGB && ++ c->opts.dst_format == AV_PIX_FMT_YUVA420P && ++ !(c->opts.src_h & 1) && ++ !(c->opts.src_w & 15) && ++ !(c->opts.src_w & 1) && !accurate_rnd) { ++ c->convert_unscaled = argb_to_yuva420p_neon_wrapper; ++ c->dst_slice_align = 2; ++ } + } + + void ff_get_unscaled_swscale_aarch64(SwsInternal *c) +diff --git a/libswscale/aarch64/swscale_unscaled_neon.S b/libswscale/aarch64/swscale_unscaled_neon.S +index a074b187e5..0c961ad4af 100644 +--- a/libswscale/aarch64/swscale_unscaled_neon.S ++++ b/libswscale/aarch64/swscale_unscaled_neon.S +@@ -20,6 +20,206 @@ + + #include "libavutil/aarch64/asm.S" + ++#define RGB2YUV_COEFFS 16*4+16*32 ++#define BY v0.h[0] ++#define GY v0.h[1] ++#define RY v0.h[2] ++#define BU v1.h[0] ++#define GU v1.h[1] ++#define RU v1.h[2] ++#define BV v2.h[0] ++#define GV v2.h[1] ++#define RV v2.h[2] ++#define Y_OFFSET v22 ++#define UV_OFFSET v23 ++ ++// convert rgb to 16-bit y, u, or v ++// uses v3 and v4 ++.macro rgbconv16 dst, b, g, r, bc, gc, rc ++ smull v3.4s, \b\().4h, \bc ++ smlal v3.4s, \g\().4h, \gc ++ smlal v3.4s, \r\().4h, \rc ++ smull2 v4.4s, \b\().8h, \bc ++ smlal2 v4.4s, \g\().8h, \gc ++ smlal2 v4.4s, \r\().8h, \rc ++ shrn \dst\().4h, v3.4s, #7 ++ shrn2 \dst\().8h, v4.4s, #7 ++.endm ++ ++// void ff_argb_to_yuv420p_neon(const uint8_t *src, uint8_t *ydst, uint8_t *udst, ++// uint8_t *vdst, int width, int height, int lumStride, ++// int chromStride, int srcStride, int32_t *rgb2yuv); ++function ff_argb_to_yuv420p_neon, export=1 ++// x0 const uint8_t *src ++// x1 uint8_t *ydst ++// x2 uint8_t *udst ++// x3 uint8_t *vdst ++// w4 int width ++// w5 int height ++// w6 int lumStride ++// w7 int chromStride ++ ldrsw x9, [sp] // srcStride ++ ldr x14, [sp, #8] // rgb2yuv ++ ++ // extend width and stride parameters ++ uxtw x4, w4 ++ sxtw x6, w6 ++ sxtw x7, w7 ++ ++ // src1 = x0 ++ // src2 = x10 ++ add x10, x0, x9 // x10 = src + srcStride ++ lsl x9, x9, #1 // srcStride *= 2 ++ lsl x11, x4, #2 // x11 = 4 * width ++ sub x9, x9, x11 // srcPadding = (2 * srcStride) - (4 * width) ++ ++ // ydst1 = x1 ++ // ydst2 = x11 ++ add x11, x1, x6 // x11 = ydst + lumStride ++ lsl x6, x6, #1 // lumStride *= 2 ++ sub x6, x6, x4 // lumPadding = (2 * lumStride) - width ++ ++ sub x7, x7, x4, lsr #1 // chromPadding = chromStride - (width / 2) ++ ++ // load rgb2yuv coefficients into v0, v1, and v2 ++ add x14, x14, #RGB2YUV_COEFFS ++ ld1 {v0.8h-v2.8h}, [x14] ++ ++ // load offset constants ++ movi Y_OFFSET.8h, #0x10, lsl #8 ++ movi UV_OFFSET.8h, #0x80, lsl #8 ++ ++1: ++ mov w14, w4 // w14 = width ++ ++2: ++ // load first line (ARGB) ++ ld4 {v26.16b, v27.16b, v28.16b, v29.16b}, [x0], #64 ++ ++ // widen first line to 16-bit (B,G,R) ++ uxtl v16.8h, v29.8b ++ uxtl v17.8h, v28.8b ++ uxtl v18.8h, v27.8b ++ uxtl2 v19.8h, v29.16b ++ uxtl2 v20.8h, v28.16b ++ uxtl2 v21.8h, v27.16b ++ ++ // calculate Y values for first line ++ rgbconv16 v24, v16, v17, v18, BY, GY, RY ++ rgbconv16 v25, v19, v20, v21, BY, GY, RY ++ ++ // load second line (ARGB) ++ ld4 {v26.16b, v27.16b, v28.16b, v29.16b}, [x10], #64 ++ ++ // pairwise add and save rgb values to calculate average ++ addp v5.8h, v16.8h, v19.8h ++ addp v6.8h, v17.8h, v20.8h ++ addp v7.8h, v18.8h, v21.8h ++ ++ // widen second line to 16-bit (B,G,R) ++ uxtl v16.8h, v29.8b ++ uxtl v17.8h, v28.8b ++ uxtl v18.8h, v27.8b ++ uxtl2 v19.8h, v29.16b ++ uxtl2 v20.8h, v28.16b ++ uxtl2 v21.8h, v27.16b ++ ++ // calculate Y values for second line ++ rgbconv16 v30, v16, v17, v18, BY, GY, RY ++ rgbconv16 v31, v19, v20, v21, BY, GY, RY ++ ++ // pairwise add rgb values to calculate average ++ addp v16.8h, v16.8h, v19.8h ++ addp v17.8h, v17.8h, v20.8h ++ addp v18.8h, v18.8h, v21.8h ++ ++ // calculate average for chroma ++ add v16.8h, v16.8h, v5.8h ++ add v17.8h, v17.8h, v6.8h ++ add v18.8h, v18.8h, v7.8h ++ ushr v16.8h, v16.8h, #2 ++ ushr v17.8h, v17.8h, #2 ++ ushr v18.8h, v18.8h, #2 ++ ++ // calculate U and V values ++ rgbconv16 v28, v16, v17, v18, BU, GU, RU ++ rgbconv16 v29, v16, v17, v18, BV, GV, RV ++ ++ // add offsets and narrow all values ++ addhn v24.8b, v24.8h, Y_OFFSET.8h ++ addhn v25.8b, v25.8h, Y_OFFSET.8h ++ addhn v30.8b, v30.8h, Y_OFFSET.8h ++ addhn v31.8b, v31.8h, Y_OFFSET.8h ++ addhn v28.8b, v28.8h, UV_OFFSET.8h ++ addhn v29.8b, v29.8h, UV_OFFSET.8h ++ ++ subs w14, w14, #16 ++ ++ // store output ++ st1 {v24.8b, v25.8b}, [x1], #16 ++ st1 {v30.8b, v31.8b}, [x11], #16 ++ st1 {v28.8b}, [x2], #8 ++ st1 {v29.8b}, [x3], #8 ++ ++ b.gt 2b ++ ++ subs w5, w5, #2 ++ ++ // row += 2 ++ add x0, x0, x9 ++ add x10, x10, x9 ++ add x1, x1, x6 ++ add x11, x11, x6 ++ add x2, x2, x7 ++ add x3, x3, x7 ++ b.gt 1b ++ ++ ret ++endfunc ++ ++// void ff_argb_to_a8_neon(const uint8_t *src, uint8_t *dst, int width, int height, ++// int srcStride, int dstStride); ++function ff_argb_to_a8_neon, export=1 ++// x0 const uint8_t *src ++// x1 uint8_t *dst ++// w2 int width ++// w3 int height ++// w4 int srcStride ++// w5 int dstStride ++ cmp w2, #0 ++ b.le 4f ++ cmp w3, #0 ++ b.le 4f ++ ++ sub w4, w4, w2, lsl #2 // srcPadding = srcStride - width * 4 ++ sub w5, w5, w2 // dstPadding = dstStride - width ++ ++1: ++ mov w6, w2 ++2: ++ cmp w6, #16 ++ b.lt 3f ++ ld4 {v0.16b, v1.16b, v2.16b, v3.16b}, [x0], #64 ++ st1 {v0.16b}, [x1], #16 ++ sub w6, w6, #16 ++ b 2b ++3: ++ cbz w6, 5f ++6: ++ ldrb w7, [x0], #4 ++ subs w6, w6, #1 ++ strb w7, [x1], #1 ++ b.gt 6b ++5: ++ subs w3, w3, #1 ++ b.eq 4f ++ add x0, x0, w4, sxtw ++ add x1, x1, w5, sxtw ++ b 1b ++4: ++ ret ++endfunc ++ + function ff_nv24_to_yuv420p_chroma_neon, export=1 + // x0 uint8_t *dst1 + // x1 int dstStride1 diff --git a/deps/ffmpeg/8.1/0002-avfilter-cuda-composition-suite.patch b/deps/ffmpeg/8.1/0002-avfilter-cuda-composition-suite.patch new file mode 100644 index 00000000..f1e51adf --- /dev/null +++ b/deps/ffmpeg/8.1/0002-avfilter-cuda-composition-suite.patch @@ -0,0 +1,4468 @@ +From d9643102a8b37addf0905f9d18b199d6fb2c0321 Mon Sep 17 00:00:00 2001 +From: Jan Pietek +Date: Tue, 15 Sep 2026 09:37:19 +0200 +Subject: [PATCH 2/7] avfilter: port CUDA composition suite to FFmpeg 8.1 + +Retain the custom pad implementation and existing processing behavior. Adapt FFFilter registration, preserve upstream scale_cuda format support, and use the upstream CUDA architecture fallback. + +Co-authored-by: Teodor Wozniak +--- + configure | 7 + + doc/filters.texi | 47 ++ + libavfilter/Makefile | 8 + + libavfilter/allfilters.c | 4 + + libavfilter/vf_convert_cuda.c | 511 +++++++++++++++ + libavfilter/vf_convert_cuda.cu | 273 ++++++++ + libavfilter/vf_crop_cuda.c | 561 +++++++++++++++++ + libavfilter/vf_overlay_cuda.c | 1 + + libavfilter/vf_overlay_many_cuda.c | 559 +++++++++++++++++ + libavfilter/vf_overlay_many_cuda.cu | 252 ++++++++ + libavfilter/vf_overlay_many_cuda.h | 68 ++ + libavfilter/vf_pad_cuda.c | 922 +++++++++++++--------------- + libavfilter/vf_pad_cuda.cu | 84 +-- + libavfilter/vf_scale_cuda.c | 5 + + libavfilter/vf_scale_cuda.cu | 104 +++- + libavfilter/vf_transition_cuda.c | 621 +++++++++++++++++++ + libavfilter/vf_transition_cuda.cu | 56 ++ + libavutil/hwcontext_cuda.c | 1 + + 18 files changed, 3504 insertions(+), 580 deletions(-) + create mode 100644 libavfilter/vf_convert_cuda.c + create mode 100644 libavfilter/vf_convert_cuda.cu + create mode 100644 libavfilter/vf_crop_cuda.c + create mode 100644 libavfilter/vf_overlay_many_cuda.c + create mode 100644 libavfilter/vf_overlay_many_cuda.cu + create mode 100644 libavfilter/vf_overlay_many_cuda.h + create mode 100644 libavfilter/vf_transition_cuda.c + create mode 100644 libavfilter/vf_transition_cuda.cu + +diff --git a/configure b/configure +index 1759694274..584e1df313 100755 +--- a/configure ++++ b/configure +@@ -3515,6 +3515,8 @@ chromakey_cuda_filter_deps="ffnvcodec" + chromakey_cuda_filter_deps_any="cuda_nvcc cuda_llvm" + colorspace_cuda_filter_deps="ffnvcodec" + colorspace_cuda_filter_deps_any="cuda_nvcc cuda_llvm" ++crop_cuda_filter_deps="ffnvcodec" ++crop_cuda_filter_deps_any="cuda_nvcc cuda_llvm" + hwupload_cuda_filter_deps="ffnvcodec" + scale_npp_filter_deps="ffnvcodec libnpp" + scale2ref_npp_filter_deps="ffnvcodec libnpp" +@@ -3525,6 +3527,11 @@ thumbnail_cuda_filter_deps_any="cuda_nvcc cuda_llvm" + transpose_npp_filter_deps="ffnvcodec libnpp" + overlay_cuda_filter_deps="ffnvcodec" + overlay_cuda_filter_deps_any="cuda_nvcc cuda_llvm" ++overlay_many_cuda_filter_deps="ffnvcodec cuda_nvcc" ++convert_cuda_filter_deps="ffnvcodec" ++convert_cuda_filter_deps_any="cuda_nvcc cuda_llvm" ++transition_cuda_filter_deps="ffnvcodec" ++transition_cuda_filter_deps_any="cuda_nvcc cuda_llvm" + pad_cuda_filter_deps="ffnvcodec" + pad_cuda_filter_deps_any="cuda_nvcc cuda_llvm" + sharpen_npp_filter_deps="ffnvcodec libnpp" +diff --git a/doc/filters.texi b/doc/filters.texi +index 5d222c6b96..5ae2fc5080 100644 +--- a/doc/filters.texi ++++ b/doc/filters.texi +@@ -19412,6 +19412,53 @@ testsrc=s=100x100, split=4 [in0][in1][in2][in3]; + + @end itemize + ++@anchor{overlay_many_cuda} ++@section overlay_many_cuda ++ ++Overlay several CUDA video streams on top of a main CUDA video stream. ++ ++The first input is the main video. Every subsequent input is a full-frame ++overlay, applied in input order. The filter supports between 2 and 16 total ++inputs. Unavailable overlay frames are skipped independently; the main frame is ++passed through unchanged only when no overlay frame is available. ++ ++The supported software-format combinations inside the CUDA frames are: ++ ++@itemize ++@item ++@code{yuv420p} main with @code{yuva420p} overlays; ++@item ++@code{yuv420p} main with @code{yuva444p} overlays; ++@item ++@code{yuv444p} main with @code{yuva444p} overlays. ++@end itemize ++ ++All overlays must use the same software format and must be at least as large as ++the main input. Composition starts at the top-left corner. Processing remains ++on the CUDA device and uses the stream associated with the main input. ++ ++This filter requires CUDA Toolkit 11.7 or newer and a GPU with compute ++capability 7.0 or newer. ++ ++It accepts the following options: ++ ++@table @option ++@item inputs ++Set the total number of inputs, including the main input. The accepted range is ++2 to 16. The default is 2. ++ ++@item eof_action ++See @ref{framesync}. ++ ++@item shortest ++See @ref{framesync}. ++ ++@item repeatlast ++See @ref{framesync}. ++@end table ++ ++This filter also supports the @ref{framesync} options. ++ + @section owdenoise + + Apply Overcomplete Wavelet denoiser. +diff --git a/libavfilter/Makefile b/libavfilter/Makefile +index a530cfae29..2fae9853a6 100644 +--- a/libavfilter/Makefile ++++ b/libavfilter/Makefile +@@ -253,6 +253,8 @@ OBJS-$(CONFIG_COLORSPACE_CUDA_FILTER) += vf_colorspace_cuda.o \ + vf_colorspace_cuda.ptx.o \ + cuda/load_helper.o + OBJS-$(CONFIG_COLORTEMPERATURE_FILTER) += vf_colortemperature.o ++OBJS-$(CONFIG_CONVERT_CUDA_FILTER) += vf_convert_cuda.o vf_convert_cuda.ptx.o \ ++ cuda/load_helper.o + OBJS-$(CONFIG_CONVOLUTION_FILTER) += vf_convolution.o + OBJS-$(CONFIG_CONVOLUTION_OPENCL_FILTER) += vf_convolution_opencl.o opencl.o \ + opencl/convolution.o +@@ -262,6 +264,7 @@ OBJS-$(CONFIG_COREIMAGE_FILTER) += vf_coreimage.o + OBJS-$(CONFIG_CORR_FILTER) += vf_corr.o framesync.o + OBJS-$(CONFIG_COVER_RECT_FILTER) += vf_cover_rect.o lavfutils.o + OBJS-$(CONFIG_CROP_FILTER) += vf_crop.o ++OBJS-$(CONFIG_CROP_CUDA_FILTER) += vf_crop_cuda.o + OBJS-$(CONFIG_CROPDETECT_FILTER) += vf_cropdetect.o edge_common.o + OBJS-$(CONFIG_CUE_FILTER) += f_cue.o + OBJS-$(CONFIG_CURVES_FILTER) += vf_curves.o +@@ -422,6 +425,9 @@ OBJS-$(CONFIG_OSCILLOSCOPE_FILTER) += vf_datascope.o + OBJS-$(CONFIG_OVERLAY_FILTER) += vf_overlay.o framesync.o + OBJS-$(CONFIG_OVERLAY_CUDA_FILTER) += vf_overlay_cuda.o framesync.o vf_overlay_cuda.ptx.o \ + cuda/load_helper.o ++OBJS-$(CONFIG_OVERLAY_MANY_CUDA_FILTER) += vf_overlay_many_cuda.o framesync.o \ ++ vf_overlay_many_cuda.ptx.o cuda/load_helper.o ++ + OBJS-$(CONFIG_OVERLAY_OPENCL_FILTER) += vf_overlay_opencl.o opencl.o \ + opencl/overlay.o framesync.o + OBJS-$(CONFIG_OVERLAY_QSV_FILTER) += vf_overlay_qsv.o framesync.o +@@ -545,6 +551,8 @@ OBJS-$(CONFIG_TONEMAP_OPENCL_FILTER) += vf_tonemap_opencl.o opencl.o \ + opencl/tonemap.o opencl/colorspace_common.o + OBJS-$(CONFIG_TONEMAP_VAAPI_FILTER) += vf_tonemap_vaapi.o vaapi_vpp.o + OBJS-$(CONFIG_TPAD_FILTER) += vf_tpad.o ++OBJS-$(CONFIG_TRANSITION_CUDA_FILTER) += vf_transition_cuda.o framesync.o vf_transition_cuda.ptx.o \ ++ cuda/load_helper.o + OBJS-$(CONFIG_TRANSPOSE_FILTER) += vf_transpose.o + OBJS-$(CONFIG_TRANSPOSE_NPP_FILTER) += vf_transpose_npp.o + OBJS-$(CONFIG_TRANSPOSE_OPENCL_FILTER) += vf_transpose_opencl.o opencl.o opencl/transpose.o +diff --git a/libavfilter/allfilters.c b/libavfilter/allfilters.c +index e26859e159..01c75b2971 100644 +--- a/libavfilter/allfilters.c ++++ b/libavfilter/allfilters.c +@@ -230,6 +230,7 @@ extern const FFFilter ff_vf_colormatrix; + extern const FFFilter ff_vf_colorspace; + extern const FFFilter ff_vf_colorspace_cuda; + extern const FFFilter ff_vf_colortemperature; ++extern const FFFilter ff_vf_convert_cuda; + extern const FFFilter ff_vf_convolution; + extern const FFFilter ff_vf_convolution_opencl; + extern const FFFilter ff_vf_convolve; +@@ -238,6 +239,7 @@ extern const FFFilter ff_vf_coreimage; + extern const FFFilter ff_vf_corr; + extern const FFFilter ff_vf_cover_rect; + extern const FFFilter ff_vf_crop; ++extern const FFFilter ff_vf_crop_cuda; + extern const FFFilter ff_vf_cropdetect; + extern const FFFilter ff_vf_cue; + extern const FFFilter ff_vf_curves; +@@ -399,6 +401,7 @@ extern const FFFilter ff_vf_overlay_qsv; + extern const FFFilter ff_vf_overlay_vaapi; + extern const FFFilter ff_vf_overlay_vulkan; + extern const FFFilter ff_vf_overlay_cuda; ++extern const FFFilter ff_vf_overlay_many_cuda; + extern const FFFilter ff_vf_owdenoise; + extern const FFFilter ff_vf_pad; + extern const FFFilter ff_vf_pad_cuda; +@@ -512,6 +515,7 @@ extern const FFFilter ff_vf_tonemap; + extern const FFFilter ff_vf_tonemap_opencl; + extern const FFFilter ff_vf_tonemap_vaapi; + extern const FFFilter ff_vf_tpad; ++extern const FFFilter ff_vf_transition_cuda; + extern const FFFilter ff_vf_transpose; + extern const FFFilter ff_vf_transpose_npp; + extern const FFFilter ff_vf_transpose_opencl; +diff --git a/libavfilter/vf_convert_cuda.c b/libavfilter/vf_convert_cuda.c +new file mode 100644 +index 0000000000..5f49d2126e +--- /dev/null ++++ b/libavfilter/vf_convert_cuda.c +@@ -0,0 +1,511 @@ ++#include "libavutil/hwcontext.h" ++#include "libavutil/hwcontext_cuda_internal.h" ++#include "libavutil/cuda_check.h" ++#include "libavutil/opt.h" ++#include "libavutil/pixdesc.h" ++#include "libavutil/eval.h" ++#include "libavutil/colorspace.h" ++ ++#include "avfilter.h" ++#include "formats.h" ++#include "avfilter_internal.h" ++#include "video.h" ++ ++#include "cuda/load_helper.h" ++ ++static enum AVPixelFormat supported_in_formats[] = { ++ AV_PIX_FMT_YUVA444P12LE, AV_PIX_FMT_YUVA444P, ++ AV_PIX_FMT_ARGB, AV_PIX_FMT_RGBA, AV_PIX_FMT_ABGR, AV_PIX_FMT_BGRA, ++ AV_PIX_FMT_NV12, AV_PIX_FMT_YUV420P, ++}; ++ ++static enum AVPixelFormat supported_out_formats[] = { ++ AV_PIX_FMT_YUVA420P, AV_PIX_FMT_YUVA444P, ++ AV_PIX_FMT_NV12, AV_PIX_FMT_YUV420P, ++}; ++ ++#define BLOCK_X 32 ++#define BLOCK_Y 16 ++ ++#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) ) ++#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, cu, x) ++#define ARRAY_COUNT(x) (sizeof(x)/sizeof(*(x))) ++ ++typedef struct ConvertCUDAContext { ++ const AVClass *class; ++ AVCUDADeviceContext *hwctx; ++ AVBufferRef *hw_device_ctx; ++ ++ AVBufferRef *frames_ctx; ++ AVFrame *frame; ++ AVFrame *tmp_frame; ++ ++ CUmodule cu_module; ++ // cu_functions is an array where we load handles to CUDA shader functions. ++ // For some in/out format combinations we do the whole convert with 1 shader. ++ // For others we need to use 2 separate shaders (mostly for 444p... -> 420p). ++ // This is organized in this way to minimize the final number of shaders needed by this filter. ++ CUfunction cu_functions[2]; ++ CUstream cu_stream; ++ ++ char *format_str; ++ ++ enum AVPixelFormat format_in; ++ enum AVPixelFormat format_out; ++} ConvertCUDAContext; ++ ++ ++ ++static int format_input_is_supported(enum AVPixelFormat fmt) ++{ ++ for (int i = 0; i < FF_ARRAY_ELEMS(supported_in_formats); i++) ++ if (supported_in_formats[i] == fmt) return 1; ++ return 0; ++} ++ ++static int format_output_is_supported(enum AVPixelFormat fmt) ++{ ++ for (int i = 0; i < FF_ARRAY_ELEMS(supported_out_formats); i++) ++ if (supported_out_formats[i] == fmt) return 1; ++ return 0; ++} ++ ++static av_cold int init(AVFilterContext *avctx) ++{ ++ ConvertCUDAContext *ctx = avctx->priv; ++ ++ ctx->frame = av_frame_alloc(); ++ if (!ctx->frame) ++ return AVERROR(ENOMEM); ++ ++ ctx->tmp_frame = av_frame_alloc(); ++ if (!ctx->tmp_frame) ++ return AVERROR(ENOMEM); ++ ++ return 0; ++} ++ ++static av_cold void uninit(AVFilterContext *avctx) ++{ ++ ConvertCUDAContext *ctx = avctx->priv; ++ ++ if (ctx->hwctx && ctx->cu_module) { ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext dummy; ++ ++ CHECK_CU(cu->cuCtxPushCurrent(ctx->hwctx->cuda_ctx)); ++ CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); ++ ctx->cu_module = NULL; ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ } ++ ++ av_frame_free(&ctx->frame); ++ av_buffer_unref(&ctx->frames_ctx); ++ av_frame_free(&ctx->tmp_frame); ++} ++ ++static av_cold int output_config_props(AVFilterLink *outlink) ++{ ++ extern const unsigned char ff_vf_convert_cuda_ptx_data[]; ++ extern const unsigned int ff_vf_convert_cuda_ptx_len; ++ ++ FilterLink *outl = ff_filter_link(outlink); ++ AVFilterContext *avctx = outlink->src; ++ AVFilterLink *inlink = outlink->src->inputs[0]; ++ FilterLink *inl = ff_filter_link(inlink); ++ ConvertCUDAContext *ctx = avctx->priv; ++ AVHWFramesContext *frames_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; ++ int ret; ++ ++ if (!frames_ctx) { ++ av_log(ctx, AV_LOG_ERROR, "No hw context provided on main input\n"); ++ return AVERROR(EINVAL); ++ } ++ ++ ctx->format_in = frames_ctx->sw_format; ++ if (!format_input_is_supported(ctx->format_in)) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported input format: %s\n", ++ av_get_pix_fmt_name(ctx->format_in)); ++ return AVERROR(ENOSYS); ++ } ++ ++ if (!ctx->format_str || !strlen(ctx->format_str)) { ++ av_log(ctx, AV_LOG_ERROR, "Specify output pixel format argument (example: format=yuva420p, format=yuva444p)\n"); ++ return AVERROR(ENOSYS); ++ } ++ ++ ctx->format_out = av_get_pix_fmt(ctx->format_str); ++ if (!format_output_is_supported(ctx->format_out)) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported output format: %s (%s)\n", ++ av_get_pix_fmt_name(ctx->format_out), ctx->format_str); ++ return AVERROR(ENOSYS); ++ } ++ ++ ++ // initialize ++ ctx->hw_device_ctx = av_buffer_ref(frames_ctx->device_ref); ++ if (!ctx->hw_device_ctx) ++ return AVERROR(ENOMEM); ++ ctx->hwctx = ((AVHWDeviceContext*)ctx->hw_device_ctx->data)->hwctx; ++ ctx->cu_stream = ctx->hwctx->stream; ++ ++ // load cuda functions ++ { ++ CUcontext cuda_ctx = ctx->hwctx->cuda_ctx; ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext dummy; ++ const char *fmt_str_in[2] = {0}; ++ const char *fmt_str_out = 0; ++ static_assert(ARRAY_COUNT(fmt_str_in) == ARRAY_COUNT(ctx->cu_functions)); ++ ++ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); ++ if (ret < 0) { ++ return ret; ++ } ++ ++ ret = ff_cuda_load_module(ctx, ctx->hwctx, &ctx->cu_module, ff_vf_convert_cuda_ptx_data, ff_vf_convert_cuda_ptx_len); ++ if (ret < 0) { ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return ret; ++ } ++ ++ switch (ctx->format_in) { ++ case AV_PIX_FMT_ARGB: fmt_str_in[0] = "ARGB"; break; ++ case AV_PIX_FMT_RGBA: fmt_str_in[0] = "RGBA"; break; ++ case AV_PIX_FMT_BGRA: fmt_str_in[0] = "BGRA"; break; ++ case AV_PIX_FMT_ABGR: fmt_str_in[0] = "ABGR"; break; ++ ++ case AV_PIX_FMT_YUVA444P12LE: ++ fmt_str_in[0] = "YUVA444P12LE_ya"; ++ fmt_str_in[1] = "YUVA444P12LE_uv"; ++ break; ++ ++ case AV_PIX_FMT_YUVA444P: ++ fmt_str_in[0] = "YUVA444P_ya"; ++ fmt_str_in[1] = "YUVA444P_uv"; ++ break; ++ ++ case AV_PIX_FMT_NV12: ++ fmt_str_in[0] = "NV12_y"; ++ fmt_str_in[1] = "NV12_uv"; ++ break; ++ ++ case AV_PIX_FMT_YUV420P: ++ fmt_str_in[0] = "YUV420P_y"; ++ fmt_str_in[1] = "YUV420P_uv"; ++ break; ++ } ++ ++ switch (ctx->format_out) { ++ case AV_PIX_FMT_YUVA420P: fmt_str_out = "YUVA420P"; break; ++ case AV_PIX_FMT_YUVA444P: fmt_str_out = "YUVA444P"; break; ++ case AV_PIX_FMT_NV12: fmt_str_out = "NV12"; break; ++ case AV_PIX_FMT_YUV420P: fmt_str_out = "YUV420P"; break; ++ } ++ ++ for (int i = 0; i < ARRAY_COUNT(fmt_str_in); i++) { ++ if (fmt_str_in[i] && fmt_str_out) { ++ char func_name[256]; ++ snprintf(func_name, sizeof(func_name), "Convert_%s_to_%s", fmt_str_in[i], fmt_str_out); ++ ret = CHECK_CU(cu->cuModuleGetFunction(&ctx->cu_functions[i], ctx->cu_module, func_name)); ++ ++ if (ret < 0) { ++ av_log(ctx, AV_LOG_FATAL, "CUDA function %s wasn't implemented\n", func_name); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return ret; ++ } ++ } ++ } ++ ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ } ++ ++ outlink->w = inlink->w; ++ outlink->h = inlink->h; ++ ++ // prepare output buffer ++ { ++ AVHWFramesContext *out_ctx; ++ AVBufferRef *out_ref = av_hwframe_ctx_alloc(ctx->hw_device_ctx); ++ if (!out_ref) ++ return AVERROR(ENOMEM); ++ ++ out_ctx = (AVHWFramesContext*)out_ref->data; ++ out_ctx->format = AV_PIX_FMT_CUDA; ++ out_ctx->sw_format = ctx->format_out; ++ out_ctx->width = FFALIGN(inlink->w, 32); ++ out_ctx->height = FFALIGN(inlink->h, 32); ++ ++ ret = av_hwframe_ctx_init(out_ref); ++ if (ret < 0) ++ goto output_buffer_fail; ++ ++ av_frame_unref(ctx->frame); ++ ret = av_hwframe_get_buffer(out_ref, ctx->frame, 0); ++ if (ret < 0) ++ goto output_buffer_fail; ++ ++ ctx->frame->width = inlink->w; ++ ctx->frame->height = inlink->h; ++ ++ ctx->frames_ctx = out_ref; ++ ++ if (ret < 0) { ++output_buffer_fail: ++ av_buffer_unref(&out_ref); ++ return ret; ++ } ++ ++ outl->hw_frames_ctx = av_buffer_ref(ctx->frames_ctx); ++ if (!outl->hw_frames_ctx) ++ return AVERROR(ENOMEM); ++ } ++ ++ return 0; ++} ++ ++static int fill_buffers(AVFilterContext *avctx, AVFrame *out, AVFrame *in) ++{ ++ ConvertCUDAContext *ctx = avctx->priv; ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext cuda_ctx = ctx->hwctx->cuda_ctx; ++ CUcontext dummy; ++ int ret; ++ ++ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); ++ if (ret < 0) return ret; ++ ++ ++#define LaunchKernelBlock(Func, Width, Height, BlockX, BlockY) CHECK_CU(cu->cuLaunchKernel(\ ++ Func, DIV_UP((Width), (BlockX)), DIV_UP((Height), (BlockY)), 1,\ ++ (BlockX), (BlockY), 1, 0, ctx->cu_stream, kernel_args, NULL)) ++ ++#define LaunchKernel(Func, Width, Height) LaunchKernelBlock(Func, Width, Height, BLOCK_X, BLOCK_Y) ++ ++ switch (ctx->format_in) ++ { ++ case AV_PIX_FMT_ARGB: ++ case AV_PIX_FMT_RGBA: ++ case AV_PIX_FMT_ABGR: ++ case AV_PIX_FMT_BGRA: { ++ void *kernel_args[] = { ++ &in->data[0], &in->linesize[0], ++ &out->data[0], &out->linesize[0], ++ &out->data[1], &out->linesize[1], ++ &out->data[2], &out->linesize[2], ++ &out->data[3], &out->linesize[3], ++ }; ++ ++ unsigned int out_width = out->width; ++ unsigned int out_height = out->height; ++ unsigned int block_width = BLOCK_X; ++ unsigned int block_height = BLOCK_Y; ++ if (ctx->format_out == AV_PIX_FMT_YUVA420P) { ++ out_width /= 2; ++ out_height /= 2; ++ block_width /= 2; ++ block_height /= 2; ++ } ++ ++ ret = LaunchKernelBlock(ctx->cu_functions[0], ++ out_width, out_height, ++ block_width, block_height); ++ if (ret < 0) return ret; ++ } break; ++ ++ case AV_PIX_FMT_YUVA444P: ++ case AV_PIX_FMT_YUVA444P12LE: { ++ unsigned int out_uv_width = out->width; ++ unsigned int out_uv_height = out->height; ++ if (ctx->format_out == AV_PIX_FMT_YUVA420P) { ++ out_uv_width /= 2; ++ out_uv_height /= 2; ++ } ++ ++ { // y ++ void *kernel_args[] = { ++ &out->data[0], &out->linesize[0], ++ &in->data[0], &in->linesize[0] ++ }; ++ ret = LaunchKernel(ctx->cu_functions[0], out->width, out->height); ++ if (ret < 0) return ret; ++ } ++ { // u ++ void *kernel_args[] = { ++ &out->data[1], &out->linesize[1], ++ &in->data[1], &in->linesize[1] ++ }; ++ ret = LaunchKernel(ctx->cu_functions[1], out_uv_width, out_uv_height); ++ if (ret < 0) return ret; ++ } ++ { // v ++ void *kernel_args[] = { ++ &out->data[2], &out->linesize[2], ++ &in->data[2], &in->linesize[2] ++ }; ++ ret = LaunchKernel(ctx->cu_functions[1], out_uv_width, out_uv_height); ++ if (ret < 0) return ret; ++ } ++ { // alpha ++ void *kernel_args[] = { ++ &out->data[3], &out->linesize[3], ++ &in->data[3], &in->linesize[3] ++ }; ++ ret = LaunchKernel(ctx->cu_functions[0], out->width, out->height); ++ if (ret < 0) return ret; ++ } ++ } break; ++ ++ case AV_PIX_FMT_NV12: { ++ // Y plane: copy verbatim ++ { ++ void *kernel_args[] = { ++ &out->data[0], &out->linesize[0], ++ &in->data[0], &in->linesize[0], ++ }; ++ ret = LaunchKernel(ctx->cu_functions[0], out->width, out->height); ++ if (ret < 0) return ret; ++ } ++ // UV plane: deinterleave NV12 UV -> separate YUV420P U and V ++ { ++ void *kernel_args[] = { ++ &out->data[1], &out->linesize[1], ++ &out->data[2], &out->linesize[2], ++ &in->data[1], &in->linesize[1], ++ }; ++ ret = LaunchKernel(ctx->cu_functions[1], out->width / 2, out->height / 2); ++ if (ret < 0) return ret; ++ } ++ } break; ++ ++ case AV_PIX_FMT_YUV420P: { ++ // Y plane: copy verbatim ++ { ++ void *kernel_args[] = { ++ &out->data[0], &out->linesize[0], ++ &in->data[0], &in->linesize[0], ++ }; ++ ret = LaunchKernel(ctx->cu_functions[0], out->width, out->height); ++ if (ret < 0) return ret; ++ } ++ // UV plane: interleave YUV420P U and V -> NV12 UV ++ { ++ void *kernel_args[] = { ++ &out->data[1], &out->linesize[1], ++ &in->data[1], &in->linesize[1], ++ &in->data[2], &in->linesize[2], ++ }; ++ ret = LaunchKernel(ctx->cu_functions[1], out->width / 2, out->height / 2); ++ if (ret < 0) return ret; ++ } ++ } break; ++ ++ default: { ++ av_log(ctx, AV_LOG_ERROR, "Unexpected lack of support for pixel format %s\n", ++ av_get_pix_fmt_name(ctx->format_in)); ++ av_frame_free(&out); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return AVERROR_BUG; ++ } break; ++ } ++ ++ ret = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ if (ret < 0) return ret; ++ ++ return 0; ++} ++ ++static int filter_frame(AVFilterLink *link, AVFrame *in) ++{ ++ AVFilterContext *avctx = link->dst; ++ ConvertCUDAContext *ctx = avctx->priv; ++ AVFilterLink *outlink = avctx->outputs[0]; ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ ++ AVFrame *out = NULL; ++ CUcontext dummy; ++ int ret = 0; ++ ++ out = av_frame_alloc(); ++ if (!out) { ++ ret = AVERROR(ENOMEM); ++ goto fail; ++ } ++ ++ ret = CHECK_CU(cu->cuCtxPushCurrent(ctx->hwctx->cuda_ctx)); ++ if (ret < 0) ++ goto fail; ++ ++ // fill and prepare the output frame ++ { ++ ret = fill_buffers(avctx, ctx->frame, in); ++ if (ret < 0) ++ return ret; ++ ++ ret = av_hwframe_get_buffer(ctx->frame->hw_frames_ctx, ctx->tmp_frame, 0); ++ if (ret < 0) ++ return ret; ++ ++ av_frame_move_ref(out, ctx->frame); ++ av_frame_move_ref(ctx->frame, ctx->tmp_frame); ++ ++ ctx->frame->width = outlink->w; ++ ctx->frame->height = outlink->h; ++ ++ ret = av_frame_copy_props(out, in); ++ if (ret < 0) ++ return ret; ++ } ++ ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ if (ret < 0) ++ goto fail; ++ ++ av_frame_free(&in); ++ return ff_filter_frame(outlink, out); ++fail: ++ av_frame_free(&in); ++ av_frame_free(&out); ++ return ret; ++} ++ ++ ++ ++#define OFFSET(x) offsetof(ConvertCUDAContext, x) ++#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM|AV_OPT_FLAG_VIDEO_PARAM) ++ ++static const AVOption convert_cuda_options[] = { ++ { "format", "Output pixel format", OFFSET(format_str), AV_OPT_TYPE_STRING, { .str = "" }, 0, 0, FLAGS }, ++ { NULL } ++}; ++ ++AVFILTER_DEFINE_CLASS(convert_cuda); ++ ++static const AVFilterPad inputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .filter_frame = filter_frame, ++ }, ++}; ++ ++static const AVFilterPad outputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .config_props = output_config_props, ++ }, ++}; ++ ++const FFFilter ff_vf_convert_cuda = { ++ .p.name = "convert_cuda", ++ .p.description = NULL_IF_CONFIG_SMALL("CUDA accelerated pixel format conversion"), ++ .priv_size = sizeof(ConvertCUDAContext), ++ .p.priv_class = &convert_cuda_class, ++ .init = init, ++ .uninit = uninit, ++ FILTER_INPUTS(inputs), ++ FILTER_OUTPUTS(outputs), ++ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), ++ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, ++}; +diff --git a/libavfilter/vf_convert_cuda.cu b/libavfilter/vf_convert_cuda.cu +new file mode 100644 +index 0000000000..beabc0cd91 +--- /dev/null ++++ b/libavfilter/vf_convert_cuda.cu +@@ -0,0 +1,273 @@ ++typedef unsigned char uint8_t; ++typedef unsigned short uint16_t; ++ ++// @todo: 3 shaders here are indentical to Convert_YUVA444P12LE_ya_to_YUVA420P ++// Boilerplate/organization code in vf_convert_cuda.c ++// could be refactored a bit so there is less copypasta. ++ ++ ++// ++// Color constants for converting from RGB to YUV ++// in this file are following BT709 color profile. ++// They are derived from libavutil/colorspace.h ++// ++ ++// offsets into color components ++// <0, 1, 2, 3> for argb ++// <3, 2, 1, 0> for bgra ++template ++__device__ static inline void convert_templated_argb_to_yuva420p( ++ uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, ++ uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, ++ uint8_t *out_a, int out_a_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ int x2 = x*2; ++ int y2 = y*2; ++ ++ uint8_t *pixel0 = &in_argb[(x2 )*4 + (y2 )*in_linesize]; ++ uint8_t *pixel1 = &in_argb[(x2 + 1)*4 + (y2 )*in_linesize]; ++ uint8_t *pixel2 = &in_argb[(x2 )*4 + (y2 + 1)*in_linesize]; ++ uint8_t *pixel3 = &in_argb[(x2 + 1)*4 + (y2 + 1)*in_linesize]; ++ ++ uint8_t r0 = pixel0[ri]; uint8_t g0 = pixel0[gi]; uint8_t b0 = pixel0[bi]; ++ uint8_t r1 = pixel1[ri]; uint8_t g1 = pixel1[gi]; uint8_t b1 = pixel1[bi]; ++ uint8_t r2 = pixel2[ri]; uint8_t g2 = pixel2[gi]; uint8_t b2 = pixel2[bi]; ++ uint8_t r3 = pixel3[ri]; uint8_t g3 = pixel3[gi]; uint8_t b3 = pixel3[bi]; ++ ++ // calculate Y for 4 pixels ++ out_y[(x2 ) + (y2 )*out_y_linesize] = lroundf(0.18258588f*r0 + 0.6142305882f*g0 + 0.062007059f*b0 + 16.f); ++ out_y[(x2 + 1) + (y2 )*out_y_linesize] = lroundf(0.18258588f*r1 + 0.6142305882f*g1 + 0.062007059f*b1 + 16.f); ++ out_y[(x2 ) + (y2 + 1)*out_y_linesize] = lroundf(0.18258588f*r2 + 0.6142305882f*g2 + 0.062007059f*b2 + 16.f); ++ out_y[(x2 + 1) + (y2 + 1)*out_y_linesize] = lroundf(0.18258588f*r3 + 0.6142305882f*g3 + 0.062007059f*b3 + 16.f); ++ ++ // copy alpha as is for 4 pixels ++ out_a[(x2 ) + (y2 )*out_a_linesize] = pixel0[ai]; ++ out_a[(x2 + 1) + (y2 )*out_a_linesize] = pixel1[ai]; ++ out_a[(x2 ) + (y2 + 1)*out_a_linesize] = pixel2[ai]; ++ out_a[(x2 + 1) + (y2 + 1)*out_a_linesize] = pixel3[ai]; ++ ++ // average out 4 rgb pixels ++ uint8_t avg_r = (uint8_t)((r0 + r1 + r2 + r3)/4); ++ uint8_t avg_g = (uint8_t)((g0 + g1 + g2 + g3)/4); ++ uint8_t avg_b = (uint8_t)((b0 + b1 + b2 + b3)/4); ++ ++ // write out U and V ++ out_u[x + (y * out_u_linesize)] = lroundf((-0.100641882f*avg_r - 0.3385738039f*avg_g + 0.439215686f*avg_b) + 128.f); ++ out_v[x + (y * out_v_linesize)] = lroundf(( 0.439215686f*avg_r - 0.3989396078f*avg_g - 0.040276078f*avg_b) + 128.f); ++} ++ ++ ++// offsets into color components ++// <0, 1, 2, 3> for argb ++// <3, 2, 1, 0> for bgra ++template ++__device__ static inline void convert_templated_argb_to_yuva444p( ++ uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, ++ uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, ++ uint8_t *out_a, int out_a_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ ++ uint8_t *pixel0 = &in_argb[x*4 + y*in_linesize]; ++ uint8_t r0 = pixel0[ri]; uint8_t g0 = pixel0[gi]; uint8_t b0 = pixel0[bi]; ++ ++ // write out Y and A ++ out_y[x + y*out_y_linesize] = lroundf(0.18258588f*r0 + 0.6142305882f*g0 + 0.062007059f*b0 + 16.f); ++ out_a[x + y*out_a_linesize] = pixel0[ai]; ++ ++ // write out U and V ++ out_u[x + y*out_u_linesize] = lroundf((-0.100641882f*r0 - 0.3385738039f*g0 + 0.439215686f*b0) + 128.f); ++ out_v[x + y*out_v_linesize] = lroundf(( 0.439215686f*r0 - 0.3989396078f*g0 - 0.040276078f*b0) + 128.f); ++} ++ ++ ++ ++extern "C" { ++ ++// ARGB (+variants) -> YUVA420P ++__global__ void Convert_ARGB_to_YUVA420P(uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) ++{ ++ convert_templated_argb_to_yuva420p<0,1,2,3>(in_argb, in_linesize, ++ out_y, out_y_linesize, out_u, out_u_linesize, ++ out_v, out_v_linesize, out_a, out_a_linesize); ++} ++ ++__global__ void Convert_RGBA_to_YUVA420P(uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) ++{ ++ convert_templated_argb_to_yuva420p<3,0,1,2>(in_argb, in_linesize, ++ out_y, out_y_linesize, out_u, out_u_linesize, ++ out_v, out_v_linesize, out_a, out_a_linesize); ++} ++ ++__global__ void Convert_BGRA_to_YUVA420P(uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) ++{ ++ convert_templated_argb_to_yuva420p<3,2,1,0>(in_argb, in_linesize, ++ out_y, out_y_linesize, out_u, out_u_linesize, ++ out_v, out_v_linesize, out_a, out_a_linesize); ++} ++ ++__global__ void Convert_ABGR_to_YUVA420P(uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) ++{ ++ convert_templated_argb_to_yuva420p<0,3,2,1>(in_argb, in_linesize, ++ out_y, out_y_linesize, out_u, out_u_linesize, ++ out_v, out_v_linesize, out_a, out_a_linesize); ++} ++ ++ ++ ++ ++// ARGB (+variants) -> YUVA444P ++__global__ void Convert_ARGB_to_YUVA444P(uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) ++{ ++ convert_templated_argb_to_yuva444p<0,1,2,3>(in_argb, in_linesize, ++ out_y, out_y_linesize, out_u, out_u_linesize, ++ out_v, out_v_linesize, out_a, out_a_linesize); ++} ++ ++__global__ void Convert_RGBA_to_YUVA444P(uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) ++{ ++ convert_templated_argb_to_yuva444p<3,0,1,2>(in_argb, in_linesize, ++ out_y, out_y_linesize, out_u, out_u_linesize, ++ out_v, out_v_linesize, out_a, out_a_linesize); ++} ++ ++__global__ void Convert_BGRA_to_YUVA444P(uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) ++{ ++ convert_templated_argb_to_yuva444p<3,2,1,0>(in_argb, in_linesize, ++ out_y, out_y_linesize, out_u, out_u_linesize, ++ out_v, out_v_linesize, out_a, out_a_linesize); ++} ++ ++__global__ void Convert_ABGR_to_YUVA444P(uint8_t *in_argb, int in_linesize, ++ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) ++{ ++ convert_templated_argb_to_yuva444p<0,3,2,1>(in_argb, in_linesize, ++ out_y, out_y_linesize, out_u, out_u_linesize, ++ out_v, out_v_linesize, out_a, out_a_linesize); ++} ++ ++ ++ ++ ++// YUVA444P12LE -> YUVA420P ++__global__ void Convert_YUVA444P12LE_ya_to_YUVA420P( ++ uint8_t *main, int main_linesize, ++ uint16_t *overlay, int overlay_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ main[x + y*main_linesize] = overlay[x + y*(overlay_linesize/2)] >> 4; ++} ++ ++__global__ void Convert_YUVA444P12LE_uv_to_YUVA420P( ++ uint8_t *main, int main_linesize, ++ uint16_t *overlay, int overlay_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ main[x + y*main_linesize] = overlay[x*2 + (y*2)*(overlay_linesize/2)] >> 4; ++} ++ ++// YUVA444P12LE -> YUVA444P ++__global__ void Convert_YUVA444P12LE_ya_to_YUVA444P( ++ uint8_t *main, int main_linesize, ++ uint16_t *overlay, int overlay_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ main[x + y*main_linesize] = overlay[x + y*(overlay_linesize/2)] >> 4; ++} ++ ++__global__ void Convert_YUVA444P12LE_uv_to_YUVA444P( ++ uint8_t *main, int main_linesize, ++ uint16_t *overlay, int overlay_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ main[x + y*main_linesize] = overlay[x + y*(overlay_linesize/2)] >> 4; ++} ++ ++// NV12 to YUV420P ++__global__ void Convert_NV12_y_to_YUV420P( ++ uint8_t *out_y, int out_y_linesize, ++ uint8_t *in_y, int in_y_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ out_y[x + y * out_y_linesize] = in_y[x + y * in_y_linesize]; ++} ++ ++__global__ void Convert_NV12_uv_to_YUV420P( ++ uint8_t *out_u, int out_u_linesize, ++ uint8_t *out_v, int out_v_linesize, ++ uint8_t *in_uv, int in_uv_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ out_u[x + y * out_u_linesize] = in_uv[x * 2 + y * in_uv_linesize]; ++ out_v[x + y * out_v_linesize] = in_uv[x * 2 + 1 + y * in_uv_linesize]; ++} ++ ++// YUV420P to NV12 ++__global__ void Convert_YUV420P_y_to_NV12( ++ uint8_t *out_y, int out_y_linesize, ++ uint8_t *in_y, int in_y_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ out_y[x + y * out_y_linesize] = in_y[x + y * in_y_linesize]; ++} ++ ++__global__ void Convert_YUV420P_uv_to_NV12( ++ uint8_t *out_uv, int out_uv_linesize, ++ uint8_t *in_u, int in_u_linesize, ++ uint8_t *in_v, int in_v_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ out_uv[x * 2 + y * out_uv_linesize] = in_u[x + y * in_u_linesize]; ++ out_uv[x * 2 + 1 + y * out_uv_linesize] = in_v[x + y * in_v_linesize]; ++} ++ ++// YUVA444P -> YUVA420P ++__global__ void Convert_YUVA444P_ya_to_YUVA420P( ++ uint8_t *main, int main_linesize, ++ uint8_t *overlay, int overlay_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ main[x + y*main_linesize] = overlay[x + y*(overlay_linesize)]; ++} ++ ++__global__ void Convert_YUVA444P_uv_to_YUVA420P( ++ uint8_t *main, int main_linesize, ++ uint8_t *overlay, int overlay_linesize) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ main[x + y*main_linesize] = overlay[x*2 + (y*2)*(overlay_linesize)]; ++} ++ ++} +diff --git a/libavfilter/vf_crop_cuda.c b/libavfilter/vf_crop_cuda.c +new file mode 100644 +index 0000000000..a8586793d7 +--- /dev/null ++++ b/libavfilter/vf_crop_cuda.c +@@ -0,0 +1,561 @@ ++/* ++ * Copyright (c) 2019 - 2022 ++ * ++ * This file is part of FFmpeg. ++ * ++ * FFmpeg is free software; you can redistribute it and/or ++ * modify it under the terms of the GNU Lesser General Public ++ * License as published by the Free Software Foundation; either ++ * version 2.1 of the License, or (at your option) any later version. ++ * ++ * FFmpeg is distributed in the hope that it will be useful, ++ * but WITHOUT ANY WARRANTY; without even the implied warranty of ++ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ * Lesser General Public License for more details. ++ * ++ * You should have received a copy of the GNU Lesser General Public ++ * License along with FFmpeg; if not, write to the Free Software ++ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA ++ */ ++ ++#include ++ ++#include "avfilter.h" ++#include "filters.h" ++#include "video.h" ++#include "libavutil/avstring.h" ++#include "libavutil/cuda_check.h" ++#include "libavutil/eval.h" ++#include "libavutil/hwcontext.h" ++#include "libavutil/hwcontext_cuda_internal.h" ++#include "libavutil/imgutils.h" ++#include "libavutil/internal.h" ++#include "libavutil/libm.h" ++#include "libavutil/mathematics.h" ++#include "libavutil/opt.h" ++ ++static const char *const var_names[] = { ++ "in_w", "iw", ++ "in_h", "ih", ++ "out_w", "ow", ++ "out_h", "oh", ++ "a", ++ "sar", ++ "dar", ++ "hsub", ++ "vsub", ++ "x", ++ "y", ++ "n", ++#if FF_API_FRAME_PKT ++ "pos", ++#endif ++ "t", ++ NULL ++}; ++ ++enum var_name { ++ VAR_IN_W, VAR_IW, ++ VAR_IN_H, VAR_IH, ++ VAR_OUT_W, VAR_OW, ++ VAR_OUT_H, VAR_OH, ++ VAR_A, ++ VAR_SAR, ++ VAR_DAR, ++ VAR_HSUB, ++ VAR_VSUB, ++ VAR_X, ++ VAR_Y, ++ VAR_N, ++#if FF_API_FRAME_PKT ++ VAR_POS, ++#endif ++ VAR_T, ++ VAR_VARS_NB ++}; ++ ++typedef struct CropCUDAContext { ++ const AVClass *class; ++ AVCUDADeviceContext *hwctx; ++ AVBufferRef *hw_device_ctx; ++ ++ AVBufferRef *frames_ctx; ++ AVFrame *frame; ++ AVFrame *tmp_frame; ++ ++ enum AVPixelFormat sw_format; ++ ++ int x; ++ int y; ++ int w; ++ int h; ++ ++ AVRational out_sar; ++ int keep_aspect; ++ int exact; ++ ++ int max_step[4]; ++ int hsub, vsub; ++ char *x_expr, *y_expr, *w_expr, *h_expr; ++ AVExpr *x_pexpr, *y_pexpr; ++ double var_values[VAR_VARS_NB]; ++} CropCUDAContext; ++ ++#define CHECK_CU(x) FF_CUDA_CHECK_DL(s, cu, x) ++ ++static av_cold int init(AVFilterContext *ctx) ++{ ++ CropCUDAContext *s = ctx->priv; ++ ++ s->frame = av_frame_alloc(); ++ if (!s->frame) ++ return AVERROR(ENOMEM); ++ ++ s->tmp_frame = av_frame_alloc(); ++ if (!s->tmp_frame) ++ return AVERROR(ENOMEM); ++ ++ return 0; ++} ++ ++static av_cold void uninit(AVFilterContext *ctx) ++{ ++ CropCUDAContext *s = ctx->priv; ++ ++ av_expr_free(s->x_pexpr); ++ s->x_pexpr = NULL; ++ av_expr_free(s->y_pexpr); ++ s->y_pexpr = NULL; ++ ++ av_frame_free(&s->frame); ++ av_buffer_unref(&s->hw_device_ctx); ++ av_buffer_unref(&s->frames_ctx); ++ av_frame_free(&s->tmp_frame); ++} ++ ++static inline int normalize_double(int *n, double d) ++{ ++ int ret = 0; ++ ++ if (isnan(d)) { ++ ret = AVERROR(EINVAL); ++ } else if (d > INT_MAX || d < INT_MIN) { ++ *n = d > INT_MAX ? INT_MAX : INT_MIN; ++ ret = AVERROR(EINVAL); ++ } else { ++ *n = lrint(d); ++ } ++ ++ return ret; ++} ++ ++static int config_input(AVFilterLink *link) ++{ ++ FilterLink *l = ff_filter_link(link); ++ AVFilterContext *ctx = link->dst; ++ CropCUDAContext *s = ctx->priv; ++ AVHWFramesContext *frames_ctx; ++ const AVPixFmtDescriptor *pix_desc; ++ int ret; ++ const char *expr; ++ double res; ++ ++ if (!l->hw_frames_ctx) { ++ av_log(ctx, AV_LOG_ERROR, "No hw context provided on main input\n"); ++ return AVERROR(EINVAL); ++ } ++ frames_ctx = (AVHWFramesContext *)l->hw_frames_ctx->data; ++ ++ s->hw_device_ctx = av_buffer_ref(frames_ctx->device_ref); ++ if (!s->hw_device_ctx) ++ return AVERROR(ENOMEM); ++ s->hwctx = ((AVHWDeviceContext *)s->hw_device_ctx->data)->hwctx; ++ ++ s->sw_format = frames_ctx->sw_format; ++ pix_desc = av_pix_fmt_desc_get(s->sw_format); ++ ++ s->var_values[VAR_IN_W] = s->var_values[VAR_IW] = ctx->inputs[0]->w; ++ s->var_values[VAR_IN_H] = s->var_values[VAR_IH] = ctx->inputs[0]->h; ++ s->var_values[VAR_A] = (float)link->w / link->h; ++ s->var_values[VAR_SAR] = link->sample_aspect_ratio.num ? av_q2d(link->sample_aspect_ratio) : 1; ++ s->var_values[VAR_DAR] = s->var_values[VAR_A] * s->var_values[VAR_SAR]; ++ s->var_values[VAR_HSUB] = 1 << pix_desc->log2_chroma_w; ++ s->var_values[VAR_VSUB] = 1 << pix_desc->log2_chroma_h; ++ s->var_values[VAR_X] = NAN; ++ s->var_values[VAR_Y] = NAN; ++ s->var_values[VAR_OUT_W] = s->var_values[VAR_OW] = NAN; ++ s->var_values[VAR_OUT_H] = s->var_values[VAR_OH] = NAN; ++ s->var_values[VAR_N] = 0; ++ s->var_values[VAR_T] = NAN; ++#if FF_API_FRAME_PKT ++ s->var_values[VAR_POS] = NAN; ++#endif ++ ++ av_image_fill_max_pixsteps(s->max_step, NULL, pix_desc); ++ ++ s->hsub = pix_desc->log2_chroma_w; ++ s->vsub = pix_desc->log2_chroma_h; ++ ++ av_expr_parse_and_eval(&res, (expr = s->w_expr), ++ var_names, s->var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx); ++ s->var_values[VAR_OUT_W] = s->var_values[VAR_OW] = res; ++ if ((ret = av_expr_parse_and_eval(&res, (expr = s->h_expr), ++ var_names, s->var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) ++ goto fail_expr; ++ s->var_values[VAR_OUT_H] = s->var_values[VAR_OH] = res; ++ if ((ret = av_expr_parse_and_eval(&res, (expr = s->w_expr), ++ var_names, s->var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) ++ goto fail_expr; ++ ++ s->var_values[VAR_OUT_W] = s->var_values[VAR_OW] = res; ++ if (normalize_double(&s->w, s->var_values[VAR_OUT_W]) < 0 || ++ normalize_double(&s->h, s->var_values[VAR_OUT_H]) < 0) { ++ av_log(ctx, AV_LOG_ERROR, ++ "Too big value or invalid expression for out_w/ow or out_h/oh. " ++ "Maybe the expression for out_w:'%s' or for out_h:'%s' is self-referencing.\n", ++ s->w_expr, s->h_expr); ++ return AVERROR(EINVAL); ++ } ++ ++ if (!s->exact) { ++ s->w &= ~((1 << s->hsub) - 1); ++ s->h &= ~((1 << s->vsub) - 1); ++ } ++ ++ av_expr_free(s->x_pexpr); ++ av_expr_free(s->y_pexpr); ++ s->x_pexpr = s->y_pexpr = NULL; ++ if ((ret = av_expr_parse(&s->x_pexpr, s->x_expr, var_names, ++ NULL, NULL, NULL, NULL, 0, ctx)) < 0 || ++ (ret = av_expr_parse(&s->y_pexpr, s->y_expr, var_names, ++ NULL, NULL, NULL, NULL, 0, ctx)) < 0) ++ return AVERROR(EINVAL); ++ ++ if (s->keep_aspect) { ++ AVRational dar = av_mul_q(link->sample_aspect_ratio, ++ (AVRational){ link->w, link->h }); ++ av_reduce(&s->out_sar.num, &s->out_sar.den, ++ (int64_t)dar.num * s->h, (int64_t)dar.den * s->w, INT_MAX); ++ } else { ++ s->out_sar = link->sample_aspect_ratio; ++ } ++ ++ av_log(ctx, AV_LOG_VERBOSE, "w:%d h:%d sar:%d/%d -> w:%d h:%d sar:%d/%d\n", ++ link->w, link->h, link->sample_aspect_ratio.num, link->sample_aspect_ratio.den, ++ s->w, s->h, s->out_sar.num, s->out_sar.den); ++ ++ if (s->w <= 0 || s->h <= 0 || s->w > link->w || s->h > link->h) { ++ av_log(ctx, AV_LOG_ERROR, ++ "Invalid too big or non positive size for width '%d' or height '%d'\n", ++ s->w, s->h); ++ return AVERROR(EINVAL); ++ } ++ ++ s->x = (link->w - s->w) / 2; ++ s->y = (link->h - s->h) / 2; ++ if (!s->exact) { ++ s->x &= ~((1 << s->hsub) - 1); ++ s->y &= ~((1 << s->vsub) - 1); ++ } ++ ++ { ++ FilterLink *outl = ff_filter_link(ctx->outputs[0]); ++ AVHWFramesContext *out_ctx; ++ AVBufferRef *out_ref = av_hwframe_ctx_alloc(s->hw_device_ctx); ++ if (!out_ref) ++ return AVERROR(ENOMEM); ++ ++ out_ctx = (AVHWFramesContext *)out_ref->data; ++ out_ctx->format = AV_PIX_FMT_CUDA; ++ out_ctx->sw_format = s->sw_format; ++ out_ctx->width = FFALIGN(s->w, 32); ++ out_ctx->height = FFALIGN(s->h, 32); ++ ++ ret = av_hwframe_ctx_init(out_ref); ++ if (ret < 0) ++ goto output_buffer_fail; ++ ++ av_frame_unref(s->frame); ++ ret = av_hwframe_get_buffer(out_ref, s->frame, 0); ++ if (ret < 0) ++ goto output_buffer_fail; ++ ++ s->frame->width = s->w; ++ s->frame->height = s->h; ++ ++ av_buffer_unref(&s->frames_ctx); ++ s->frames_ctx = out_ref; ++ ++ if (ret < 0) { ++output_buffer_fail: ++ av_buffer_unref(&out_ref); ++ return ret; ++ } ++ ++ av_buffer_unref(&outl->hw_frames_ctx); ++ outl->hw_frames_ctx = av_buffer_ref(s->frames_ctx); ++ if (!outl->hw_frames_ctx) ++ return AVERROR(ENOMEM); ++ } ++ ++ return 0; ++ ++fail_expr: ++ av_log(ctx, AV_LOG_ERROR, "Error when evaluating the expression '%s'\n", expr); ++ return ret; ++} ++ ++static int config_output(AVFilterLink *link) ++{ ++ CropCUDAContext *s = link->src->priv; ++ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(link->format); ++ ++ if (desc->flags & AV_PIX_FMT_FLAG_HWACCEL && link->format != AV_PIX_FMT_CUDA) { ++ /* Hardware frames adjust the cropping regions rather than changing the frame size. */ ++ } else { ++ link->w = s->w; ++ link->h = s->h; ++ } ++ link->sample_aspect_ratio = s->out_sar; ++ ++ return 0; ++} ++ ++static int filter_frame(AVFilterLink *link, AVFrame *frame_in) ++{ ++ FilterLink *l = ff_filter_link(link); ++ AVFilterContext *ctx = link->dst; ++ CropCUDAContext *s = ctx->priv; ++ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(link->format); ++ AVFrame *frame_end; ++ int ret; ++ ++ frame_end = av_frame_alloc(); ++ if (!frame_end) ++ return AVERROR(ENOMEM); ++ ++ s->var_values[VAR_N] = l->frame_count_out; ++ s->var_values[VAR_T] = frame_in->pts == AV_NOPTS_VALUE ? ++ NAN : frame_in->pts * av_q2d(link->time_base); ++#if FF_API_FRAME_PKT ++FF_DISABLE_DEPRECATION_WARNINGS ++ s->var_values[VAR_POS] = frame_in->pkt_pos == -1 ? NAN : frame_in->pkt_pos; ++FF_ENABLE_DEPRECATION_WARNINGS ++#endif ++ s->var_values[VAR_X] = av_expr_eval(s->x_pexpr, s->var_values, NULL); ++ s->var_values[VAR_Y] = av_expr_eval(s->y_pexpr, s->var_values, NULL); ++ s->var_values[VAR_X] = av_expr_eval(s->x_pexpr, s->var_values, NULL); ++ ++ normalize_double(&s->x, s->var_values[VAR_X]); ++ normalize_double(&s->y, s->var_values[VAR_Y]); ++ ++ if (s->x < 0) ++ s->x = 0; ++ if (s->y < 0) ++ s->y = 0; ++ if ((unsigned)s->x + (unsigned)s->w > link->w) ++ s->x = link->w - s->w; ++ if ((unsigned)s->y + (unsigned)s->h > link->h) ++ s->y = link->h - s->h; ++ if (!s->exact) { ++ s->x &= ~((1 << s->hsub) - 1); ++ s->y &= ~((1 << s->vsub) - 1); ++ } ++ ++ av_log(ctx, AV_LOG_TRACE, "n:%d t:%f x:%d y:%d x+w:%d y+h:%d\n", ++ (int)s->var_values[VAR_N], s->var_values[VAR_T], ++ s->x, s->y, s->x + s->w, s->y + s->h); ++ ++ { ++ CUcontext cuda_ctx = s->hwctx->cuda_ctx; ++ CudaFunctions *cu = s->hwctx->internal->cuda_dl; ++ CUstream cu_stream = s->hwctx->stream; ++ CUcontext dummy; ++ AVFrame *frame_out = s->frame; ++ int src_y_in_bytes = s->y * frame_in->linesize[0]; ++ int src_x_in_bytes = s->x * s->max_step[0]; ++ int err; ++ CUDA_MEMCPY2D cpy = { ++ .dstMemoryType = CU_MEMORYTYPE_DEVICE, ++ .dstPitch = frame_out->linesize[0], ++ .dstDevice = (CUdeviceptr)frame_out->data[0], ++ .srcMemoryType = CU_MEMORYTYPE_DEVICE, ++ .srcPitch = frame_in->linesize[0], ++ .srcDevice = (CUdeviceptr)(frame_in->data[0] + src_y_in_bytes + src_x_in_bytes), ++ .WidthInBytes = s->w * s->max_step[0], ++ .Height = s->h, ++ }; ++ ++ err = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); ++ if (err < 0) { ++ av_frame_free(&frame_end); ++ return err; ++ } ++ ++ err = CHECK_CU(cu->cuMemcpy2DAsync(&cpy, cu_stream)); ++ if (err < 0) { ++ av_log(ctx, AV_LOG_ERROR, "cuMemcpy2D error on plane [0]\n"); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ av_frame_free(&frame_end); ++ return err; ++ } ++ ++ if (!(desc->flags & AV_PIX_FMT_FLAG_PAL)) { ++ for (int i = 1; i < 3; i++) { ++ if (frame_in->data[i]) { ++ src_y_in_bytes = (s->y >> s->vsub) * frame_in->linesize[i]; ++ src_x_in_bytes = (s->x * s->max_step[i]) >> s->hsub; ++ ++ cpy.dstPitch = frame_out->linesize[i]; ++ cpy.dstDevice = (CUdeviceptr)frame_out->data[i]; ++ cpy.srcPitch = frame_in->linesize[i]; ++ cpy.srcDevice = (CUdeviceptr)(frame_in->data[i] + src_y_in_bytes + src_x_in_bytes); ++ cpy.WidthInBytes = (s->w * s->max_step[i]) >> s->hsub; ++ cpy.Height = s->h >> s->vsub; ++ ++ err = CHECK_CU(cu->cuMemcpy2DAsync(&cpy, cu_stream)); ++ if (err < 0) { ++ av_log(ctx, AV_LOG_ERROR, "cuMemcpy2D error on plane [%d]\n", i); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ av_frame_free(&frame_end); ++ return err; ++ } ++ } ++ } ++ } ++ ++ if (frame_in->data[3]) { ++ src_y_in_bytes = s->y * frame_in->linesize[3]; ++ src_x_in_bytes = s->x * s->max_step[3]; ++ ++ cpy.dstPitch = frame_out->linesize[3]; ++ cpy.dstDevice = (CUdeviceptr)frame_out->data[3]; ++ cpy.srcPitch = frame_in->linesize[3]; ++ cpy.srcDevice = (CUdeviceptr)(frame_in->data[3] + src_y_in_bytes + src_x_in_bytes); ++ cpy.WidthInBytes = s->w * s->max_step[3]; ++ cpy.Height = s->h; ++ ++ err = CHECK_CU(cu->cuMemcpy2DAsync(&cpy, cu_stream)); ++ if (err < 0) { ++ av_log(ctx, AV_LOG_ERROR, "cuMemcpy2D error on plane [3]\n"); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ av_frame_free(&frame_end); ++ return err; ++ } ++ } ++ ++ err = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ if (err < 0) { ++ av_frame_free(&frame_end); ++ return err; ++ } ++ } ++ ++ ret = av_hwframe_get_buffer(s->frame->hw_frames_ctx, s->tmp_frame, 0); ++ if (ret < 0) { ++ av_frame_free(&frame_end); ++ return ret; ++ } ++ ++ av_frame_move_ref(frame_end, s->frame); ++ av_frame_move_ref(s->frame, s->tmp_frame); ++ ++ s->frame->width = s->w; ++ s->frame->height = s->h; ++ ++ ret = av_frame_copy_props(frame_end, frame_in); ++ if (ret < 0) { ++ av_frame_free(&frame_in); ++ av_frame_free(&frame_end); ++ return ret; ++ } ++ ++ av_frame_free(&frame_in); ++ return ff_filter_frame(ctx->outputs[0], frame_end); ++} ++ ++static int process_command(AVFilterContext *ctx, const char *cmd, const char *args, ++ char *res, int res_len, int flags) ++{ ++ CropCUDAContext *s = ctx->priv; ++ int ret; ++ ++ if (!strcmp(cmd, "out_w") || !strcmp(cmd, "w") || ++ !strcmp(cmd, "out_h") || !strcmp(cmd, "h") || ++ !strcmp(cmd, "x") || !strcmp(cmd, "y")) { ++ int old_x = s->x; ++ int old_y = s->y; ++ int old_w = s->w; ++ int old_h = s->h; ++ AVFilterLink *outlink = ctx->outputs[0]; ++ AVFilterLink *inlink = ctx->inputs[0]; ++ ++ av_opt_set(s, cmd, args, 0); ++ ++ if ((ret = config_input(inlink)) < 0) { ++ s->x = old_x; ++ s->y = old_y; ++ s->w = old_w; ++ s->h = old_h; ++ return ret; ++ } ++ ++ ret = config_output(outlink); ++ } else { ++ ret = AVERROR(ENOSYS); ++ } ++ ++ return ret; ++} ++ ++#define OFFSET(x) offsetof(CropCUDAContext, x) ++#define FLAGS AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM ++#define TFLAGS AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM | AV_OPT_FLAG_RUNTIME_PARAM ++ ++static const AVOption crop_cuda_options[] = { ++ { "out_w", "set the width crop area expression", OFFSET(w_expr), AV_OPT_TYPE_STRING, { .str = "iw" }, 0, 0, TFLAGS }, ++ { "w", "set the width crop area expression", OFFSET(w_expr), AV_OPT_TYPE_STRING, { .str = "iw" }, 0, 0, TFLAGS }, ++ { "out_h", "set the height crop area expression", OFFSET(h_expr), AV_OPT_TYPE_STRING, { .str = "ih" }, 0, 0, TFLAGS }, ++ { "h", "set the height crop area expression", OFFSET(h_expr), AV_OPT_TYPE_STRING, { .str = "ih" }, 0, 0, TFLAGS }, ++ { "x", "set the x crop area expression", OFFSET(x_expr), AV_OPT_TYPE_STRING, { .str = "(in_w-out_w)/2" }, 0, 0, TFLAGS }, ++ { "y", "set the y crop area expression", OFFSET(y_expr), AV_OPT_TYPE_STRING, { .str = "(in_h-out_h)/2" }, 0, 0, TFLAGS }, ++ { "keep_aspect", "keep aspect ratio", OFFSET(keep_aspect), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS }, ++ { "exact", "do exact cropping", OFFSET(exact), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS }, ++ { NULL } ++}; ++ ++AVFILTER_DEFINE_CLASS(crop_cuda); ++ ++static const AVFilterPad crop_cuda_inputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .filter_frame = filter_frame, ++ .config_props = config_input, ++ }, ++}; ++ ++static const AVFilterPad crop_cuda_outputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .config_props = config_output, ++ }, ++}; ++ ++const FFFilter ff_vf_crop_cuda = { ++ .p.name = "crop_cuda", ++ .p.description = NULL_IF_CONFIG_SMALL("Crop the input CUDA video."), ++ .priv_size = sizeof(CropCUDAContext), ++ .p.priv_class = &crop_cuda_class, ++ .init = init, ++ .uninit = uninit, ++ FILTER_INPUTS(crop_cuda_inputs), ++ FILTER_OUTPUTS(crop_cuda_outputs), ++ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), ++ .process_command = process_command, ++ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, ++}; +diff --git a/libavfilter/vf_overlay_cuda.c b/libavfilter/vf_overlay_cuda.c +index 3d39458f3f..dcf768cab5 100644 +--- a/libavfilter/vf_overlay_cuda.c ++++ b/libavfilter/vf_overlay_cuda.c +@@ -477,6 +477,7 @@ static int overlay_cuda_config_output(AVFilterLink *outlink) + ctx->hwctx = ((AVHWDeviceContext*)ctx->hw_device_ctx->data)->hwctx; + + cuda_ctx = ctx->hwctx->cuda_ctx; ++ ctx->cu_ctx = cuda_ctx; + ctx->fs.time_base = inlink->time_base; + + ctx->cu_stream = ctx->hwctx->stream; +diff --git a/libavfilter/vf_overlay_many_cuda.c b/libavfilter/vf_overlay_many_cuda.c +new file mode 100644 +index 0000000000..20e5e9c9e6 +--- /dev/null ++++ b/libavfilter/vf_overlay_many_cuda.c +@@ -0,0 +1,559 @@ ++/** ++ * @file ++ * Overlay multiple videos on top of each other using CUDA hardware acceleration ++ */ ++ ++#include ++#include ++ ++#include "libavutil/avstring.h" ++#include "libavutil/cuda_check.h" ++#include "libavutil/hwcontext.h" ++#include "libavutil/hwcontext_cuda_internal.h" ++#include "libavutil/log.h" ++#include "libavutil/opt.h" ++#include "libavutil/pixdesc.h" ++ ++#include "avfilter.h" ++#include "avfilter_internal.h" ++#include "filters.h" ++#include "framesync.h" ++#include "video.h" ++ ++#include "cuda/load_helper.h" ++#include "vf_overlay_many_cuda.h" ++ ++#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, ctx->hwctx->internal->cuda_dl, x) ++#define DIV_UP(a, b) (((a) + (b) - 1) / (b)) ++ ++#define BLOCK_X 32 ++#define BLOCK_Y 16 ++ ++#define MAX_OVERLAY_COUNT OVERLAY_MANY_CUDA_MAX_OVERLAYS ++#define MAX_INPUT_COUNT (MAX_OVERLAY_COUNT + 1) ++ ++static const enum AVPixelFormat supported_main_formats[] = { ++ AV_PIX_FMT_YUV420P, ++ AV_PIX_FMT_YUV444P, ++ AV_PIX_FMT_NONE, ++}; ++ ++static const enum AVPixelFormat supported_overlay_formats[] = { ++ AV_PIX_FMT_YUVA420P, ++ AV_PIX_FMT_YUVA444P, ++ AV_PIX_FMT_NONE, ++}; ++ ++typedef struct OverlayManyCUDAContext { ++ const AVClass *class; ++ ++ AVBufferRef *hw_device_ctx; ++ AVBufferRef *frames_ctx; ++ AVCUDADeviceContext *hwctx; ++ ++ CUcontext cu_ctx; ++ CUmodule cu_module; ++ CUfunction cu_func; ++ CUstream cu_stream; ++ ++ FFFrameSync fs; ++ ++ int nb_inputs; ++ enum AVPixelFormat format_main; ++ enum AVPixelFormat format_overlay; ++} OverlayManyCUDAContext; ++ ++static int format_is_supported(const enum AVPixelFormat formats[], ++ enum AVPixelFormat format) ++{ ++ for (int i = 0; formats[i] != AV_PIX_FMT_NONE; i++) { ++ if (formats[i] == format) ++ return 1; ++ } ++ return 0; ++} ++ ++static int formats_match(enum AVPixelFormat format_main, ++ enum AVPixelFormat format_overlay) ++{ ++ switch (format_main) { ++ case AV_PIX_FMT_YUV420P: ++ return format_overlay == AV_PIX_FMT_YUVA420P || ++ format_overlay == AV_PIX_FMT_YUVA444P; ++ case AV_PIX_FMT_YUV444P: ++ return format_overlay == AV_PIX_FMT_YUVA444P; ++ default: ++ return 0; ++ } ++} ++ ++static int pack_frame(OverlayManyCUDAFrame *descriptor, ++ const AVFrame *frame, int plane_count) ++{ ++ for (int plane = 0; plane < plane_count; plane++) { ++ if (!frame->data[plane] || frame->linesize[plane] <= 0) ++ return AVERROR(EINVAL); ++ ++ descriptor->plane[plane].ptr = (uint64_t)(uintptr_t)frame->data[plane]; ++ descriptor->plane[plane].pitch = frame->linesize[plane]; ++ } ++ ++ return 0; ++} ++ ++static int pack_launch_params(OverlayManyCUDAParams *params, ++ const AVFrame *output, ++ const AVFrame *main, ++ AVFrame *const overlays[], ++ unsigned int overlay_count) ++{ ++ int ret; ++ ++ if (main->width <= 0 || main->height <= 0 || ++ overlay_count > MAX_OVERLAY_COUNT) ++ return AVERROR(EINVAL); ++ ++ memset(params, 0, sizeof(*params)); ++ params->width = main->width; ++ params->height = main->height; ++ params->overlay_count = overlay_count; ++ ++ ret = pack_frame(¶ms->dst, output, 3); ++ if (ret < 0) ++ return ret; ++ ret = pack_frame(¶ms->main, main, 3); ++ if (ret < 0) ++ return ret; ++ ++ for (unsigned int i = 0; i < overlay_count; i++) { ++ if (overlays[i]->width < main->width || ++ overlays[i]->height < main->height) ++ return AVERROR(EINVAL); ++ ++ ret = pack_frame(¶ms->overlay[i], overlays[i], 4); ++ if (ret < 0) ++ return ret; ++ } ++ ++ for (int dst_plane = 0; dst_plane < 3; dst_plane++) { ++ for (int src_plane = 0; src_plane < 3; src_plane++) { ++ if (output->data[dst_plane] == main->data[src_plane]) ++ return AVERROR(EINVAL); ++ } ++ for (unsigned int i = 0; i < overlay_count; i++) { ++ for (int src_plane = 0; src_plane < 4; src_plane++) { ++ if (output->data[dst_plane] == overlays[i]->data[src_plane]) ++ return AVERROR(EINVAL); ++ } ++ } ++ } ++ ++ return 0; ++} ++ ++static int overlay_many_cuda_blend(FFFrameSync *fs) ++{ ++ AVFilterContext *avctx = fs->parent; ++ OverlayManyCUDAContext *ctx = avctx->priv; ++ AVFilterLink *outlink = avctx->outputs[0]; ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext dummy; ++ AVFrame *main = NULL; ++ AVFrame *output = NULL; ++ AVFrame *overlays[MAX_OVERLAY_COUNT]; ++ OverlayManyCUDAParams params; ++ unsigned int overlay_count = 0; ++ int context_pushed = 0; ++ int ret; ++ ++ ret = ff_framesync_get_frame(fs, 0, &main, 1); ++ if (ret < 0) ++ return ret; ++ if (!main) ++ return AVERROR_BUG; ++ ++ for (int i = 1; i < ctx->nb_inputs; i++) { ++ AVFrame *overlay = NULL; ++ ++ ret = ff_framesync_get_frame(fs, i, &overlay, 0); ++ if (ret < 0) ++ goto fail; ++ if (overlay) ++ overlays[overlay_count++] = overlay; ++ } ++ ++ main->pts = av_rescale_q(fs->pts, fs->time_base, outlink->time_base); ++ if (avctx->is_disabled) ++ overlay_count = 0; ++ ++ output = ff_get_video_buffer(outlink, outlink->w, outlink->h); ++ if (!output) { ++ ret = AVERROR(ENOMEM); ++ goto fail; ++ } ++ ++ ret = av_frame_copy_props(output, main); ++ if (ret < 0) ++ goto fail; ++ ++ ret = pack_launch_params(¶ms, output, main, overlays, overlay_count); ++ if (ret < 0) { ++ av_log(ctx, AV_LOG_ERROR, "Invalid CUDA frame layout for overlay\n"); ++ goto fail; ++ } ++ ++ ret = CHECK_CU(cu->cuCtxPushCurrent(ctx->cu_ctx)); ++ if (ret < 0) ++ goto fail; ++ context_pushed = 1; ++ ++ { ++ void *kernel_args[] = { ¶ms }; ++ ++ ret = CHECK_CU(cu->cuLaunchKernel( ++ ctx->cu_func, ++ DIV_UP(params.width, BLOCK_X), ++ DIV_UP(params.height, BLOCK_Y), ++ 1, BLOCK_X, BLOCK_Y, 1, ++ 0, ctx->cu_stream, kernel_args, NULL)); ++ } ++ ++ { ++ int pop_ret = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ ++ context_pushed = 0; ++ if (ret >= 0) ++ ret = pop_ret; ++ } ++ ++ if (ret < 0) ++ goto fail; ++ ++ av_frame_free(&main); ++ return ff_filter_frame(outlink, output); ++ ++fail: ++ if (context_pushed) { ++ int pop_ret = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ ++ if (ret >= 0) ++ ret = pop_ret; ++ } ++ av_frame_free(&main); ++ av_frame_free(&output); ++ return ret; ++} ++ ++static av_cold int overlay_many_cuda_init(AVFilterContext *avctx) ++{ ++ OverlayManyCUDAContext *ctx = avctx->priv; ++ ++ ctx->fs.on_event = overlay_many_cuda_blend; ++ ++ for (int i = 0; i < ctx->nb_inputs; i++) { ++ AVFilterPad pad = { 0 }; ++ int ret; ++ ++ pad.type = AVMEDIA_TYPE_VIDEO; ++ pad.name = av_asprintf("input%d", i); ++ if (!pad.name) ++ return AVERROR(ENOMEM); ++ ++ ret = ff_append_inpad_free_name(avctx, &pad); ++ if (ret < 0) ++ return ret; ++ } ++ ++ return 0; ++} ++ ++static av_cold void overlay_many_cuda_uninit(AVFilterContext *avctx) ++{ ++ OverlayManyCUDAContext *ctx = avctx->priv; ++ ++ ff_framesync_uninit(&ctx->fs); ++ ++ if (ctx->hwctx && ctx->cu_module) { ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext dummy; ++ int ret; ++ ++ ret = CHECK_CU(cu->cuCtxPushCurrent(ctx->cu_ctx)); ++ if (ret >= 0) { ++ CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ } ++ } ++ ++ av_buffer_unref(&ctx->frames_ctx); ++ av_buffer_unref(&ctx->hw_device_ctx); ++ ctx->hwctx = NULL; ++} ++ ++static int overlay_many_cuda_activate(AVFilterContext *avctx) ++{ ++ OverlayManyCUDAContext *ctx = avctx->priv; ++ ++ return ff_framesync_activate(&ctx->fs); ++} ++ ++static int overlay_many_cuda_config_output(AVFilterLink *outlink) ++{ ++ extern const unsigned char ff_vf_overlay_many_cuda_ptx_data[]; ++ extern const unsigned int ff_vf_overlay_many_cuda_ptx_len; ++ ++ FilterLink *outl = ff_filter_link(outlink); ++ AVFilterContext *avctx = outlink->src; ++ OverlayManyCUDAContext *ctx = avctx->priv; ++ AVFilterLink *inlink_main = avctx->inputs[0]; ++ FilterLink *inl_main = ff_filter_link(inlink_main); ++ AVHWFramesContext *frames_ctx_main; ++ CudaFunctions *cu; ++ CUcontext dummy; ++ const char *function_name; ++ int err; ++ ++ if (avctx->nb_inputs != ctx->nb_inputs) { ++ av_log(ctx, AV_LOG_ERROR, ++ "The inputs option (%d) does not match the input pad count (%d)\n", ++ ctx->nb_inputs, avctx->nb_inputs); ++ return AVERROR_BUG; ++ } ++ ++ if (!inl_main->hw_frames_ctx) { ++ av_log(ctx, AV_LOG_ERROR, "No hardware frame context on main input\n"); ++ return AVERROR(EINVAL); ++ } ++ frames_ctx_main = (AVHWFramesContext *)inl_main->hw_frames_ctx->data; ++ if (!frames_ctx_main->device_ref) { ++ av_log(ctx, AV_LOG_ERROR, "No CUDA device context on main input\n"); ++ return AVERROR(EINVAL); ++ } ++ ++ ctx->format_main = frames_ctx_main->sw_format; ++ if (!format_is_supported(supported_main_formats, ctx->format_main)) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported main input format: %s\n", ++ av_get_pix_fmt_name(ctx->format_main)); ++ return AVERROR(ENOSYS); ++ } ++ ++ for (int i = 1; i < ctx->nb_inputs; i++) { ++ AVFilterLink *inlink_overlay = avctx->inputs[i]; ++ FilterLink *inl_overlay = ff_filter_link(inlink_overlay); ++ AVHWFramesContext *frames_ctx_overlay; ++ ++ if (!inl_overlay->hw_frames_ctx) { ++ av_log(ctx, AV_LOG_ERROR, ++ "No hardware frame context on overlay input %d\n", i); ++ return AVERROR(EINVAL); ++ } ++ frames_ctx_overlay = ++ (AVHWFramesContext *)inl_overlay->hw_frames_ctx->data; ++ if (!frames_ctx_overlay->device_ref) { ++ av_log(ctx, AV_LOG_ERROR, ++ "No CUDA device context on overlay input %d\n", i); ++ return AVERROR(EINVAL); ++ } ++ ++ if (!format_is_supported(supported_overlay_formats, ++ frames_ctx_overlay->sw_format)) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported overlay input %d format: %s\n", ++ i, av_get_pix_fmt_name(frames_ctx_overlay->sw_format)); ++ return AVERROR(ENOSYS); ++ } ++ ++ if (i == 1) ++ ctx->format_overlay = frames_ctx_overlay->sw_format; ++ else if (ctx->format_overlay != frames_ctx_overlay->sw_format) { ++ av_log(ctx, AV_LOG_ERROR, ++ "All overlay inputs must use the same software format\n"); ++ return AVERROR(EINVAL); ++ } ++ ++ if (frames_ctx_overlay->device_ref->data != ++ frames_ctx_main->device_ref->data) { ++ av_log(ctx, AV_LOG_ERROR, ++ "Overlay input %d uses a different CUDA device context\n", i); ++ return AVERROR(EINVAL); ++ } ++ ++ if (inlink_overlay->w < inlink_main->w || ++ inlink_overlay->h < inlink_main->h) { ++ av_log(ctx, AV_LOG_ERROR, ++ "Overlay input %d (%dx%d) is smaller than the main input " ++ "(%dx%d)\n", ++ i, inlink_overlay->w, inlink_overlay->h, ++ inlink_main->w, inlink_main->h); ++ return AVERROR(ENOSYS); ++ } ++ } ++ ++ if (!formats_match(ctx->format_main, ctx->format_overlay)) { ++ av_log(ctx, AV_LOG_ERROR, "Cannot overlay %s on %s\n", ++ av_get_pix_fmt_name(ctx->format_overlay), ++ av_get_pix_fmt_name(ctx->format_main)); ++ return AVERROR(EINVAL); ++ } ++ ++ outlink->w = inlink_main->w; ++ outlink->h = inlink_main->h; ++ outlink->time_base = inlink_main->time_base; ++ outlink->sample_aspect_ratio = inlink_main->sample_aspect_ratio; ++ ff_filter_link(outlink)->frame_rate = ++ ff_filter_link(inlink_main)->frame_rate; ++ ++ ctx->hw_device_ctx = av_buffer_ref(frames_ctx_main->device_ref); ++ if (!ctx->hw_device_ctx) ++ return AVERROR(ENOMEM); ++ ctx->hwctx = ((AVHWDeviceContext *)ctx->hw_device_ctx->data)->hwctx; ++ ctx->cu_ctx = ctx->hwctx->cuda_ctx; ++ ctx->cu_stream = ctx->hwctx->stream; ++ cu = ctx->hwctx->internal->cuda_dl; ++ ++ { ++ int compute_major; ++ ++ err = CHECK_CU(cu->cuDeviceGetAttribute( ++ &compute_major, ++ 75 /* CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR */, ++ ctx->hwctx->internal->cuda_device)); ++ if (err < 0) ++ return err; ++ if (compute_major < 7) { ++ av_log(ctx, AV_LOG_ERROR, ++ "overlay_many_cuda requires CUDA compute capability 7.0 " ++ "or newer\n"); ++ return AVERROR(ENOSYS); ++ } ++ } ++ ++ { ++ AVBufferRef *out_ref = av_hwframe_ctx_alloc(ctx->hw_device_ctx); ++ AVHWFramesContext *out_ctx; ++ ++ if (!out_ref) ++ return AVERROR(ENOMEM); ++ ++ out_ctx = (AVHWFramesContext *)out_ref->data; ++ out_ctx->format = AV_PIX_FMT_CUDA; ++ out_ctx->sw_format = ctx->format_main; ++ out_ctx->width = outlink->w; ++ out_ctx->height = outlink->h; ++ ++ err = av_hwframe_ctx_init(out_ref); ++ if (err < 0) { ++ av_buffer_unref(&out_ref); ++ return err; ++ } ++ ++ ctx->frames_ctx = out_ref; ++ outl->hw_frames_ctx = av_buffer_ref(ctx->frames_ctx); ++ if (!outl->hw_frames_ctx) ++ return AVERROR(ENOMEM); ++ } ++ ++ if (ctx->format_main == AV_PIX_FMT_YUV420P && ++ ctx->format_overlay == AV_PIX_FMT_YUVA420P) ++ function_name = "OverlayMany420"; ++ else if (ctx->format_main == AV_PIX_FMT_YUV444P) ++ function_name = "OverlayMany444"; ++ else ++ function_name = "OverlayMany420From444"; ++ ++ err = CHECK_CU(cu->cuCtxPushCurrent(ctx->cu_ctx)); ++ if (err < 0) ++ return err; ++ ++ err = ff_cuda_load_module(ctx, ctx->hwctx, &ctx->cu_module, ++ ff_vf_overlay_many_cuda_ptx_data, ++ ff_vf_overlay_many_cuda_ptx_len); ++ if (err >= 0) { ++ err = CHECK_CU(cu->cuModuleGetFunction( ++ &ctx->cu_func, ctx->cu_module, function_name)); ++ if (err < 0) ++ av_log(ctx, AV_LOG_FATAL, "CUDA function %s is unavailable\n", ++ function_name); ++ } ++ ++ { ++ int pop_ret = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ ++ if (err >= 0) ++ err = pop_ret; ++ } ++ if (err < 0) ++ return err; ++ ++ { ++ FFFrameSync *fs = &ctx->fs; ++ ++ err = ff_framesync_init(fs, avctx, ctx->nb_inputs); ++ if (err < 0) ++ return err; ++ ++ fs->in[0].time_base = avctx->inputs[0]->time_base; ++ fs->in[0].sync = ctx->nb_inputs; ++ fs->in[0].before = EXT_STOP; ++ fs->in[0].after = EXT_INFINITY; ++ ++ for (int i = 1; i < ctx->nb_inputs; i++) { ++ fs->in[i].time_base = avctx->inputs[i]->time_base; ++ fs->in[i].sync = ctx->nb_inputs - i; ++ fs->in[i].before = EXT_NULL; ++ fs->in[i].after = EXT_INFINITY; ++ } ++ } ++ ++ return ff_framesync_configure(&ctx->fs); ++} ++ ++#define OFFSET(x) offsetof(OverlayManyCUDAContext, x) ++#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM) ++ ++static const AVOption overlay_many_cuda_options[] = { ++ { "inputs", "set number of inputs", OFFSET(nb_inputs), AV_OPT_TYPE_INT, ++ { .i64 = 2 }, 2, MAX_INPUT_COUNT, FLAGS }, ++ { "eof_action", "action on secondary input EOF", ++ OFFSET(fs.opt_eof_action), AV_OPT_TYPE_INT, ++ { .i64 = EOF_ACTION_REPEAT }, EOF_ACTION_REPEAT, EOF_ACTION_PASS, ++ FLAGS, "eof_action" }, ++ { "repeat", "repeat the previous frame", 0, AV_OPT_TYPE_CONST, ++ { .i64 = EOF_ACTION_REPEAT }, .flags = FLAGS, ++ .unit = "eof_action" }, ++ { "endall", "end both streams", 0, AV_OPT_TYPE_CONST, ++ { .i64 = EOF_ACTION_ENDALL }, .flags = FLAGS, ++ .unit = "eof_action" }, ++ { "pass", "pass through the main input", 0, AV_OPT_TYPE_CONST, ++ { .i64 = EOF_ACTION_PASS }, .flags = FLAGS, ++ .unit = "eof_action" }, ++ { "shortest", "force termination when the shortest input terminates", ++ OFFSET(fs.opt_shortest), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS }, ++ { "repeatlast", "repeat the last overlay frame", ++ OFFSET(fs.opt_repeatlast), AV_OPT_TYPE_BOOL, { .i64 = 1 }, 0, 1, FLAGS }, ++ { NULL } ++}; ++ ++FRAMESYNC_DEFINE_CLASS(overlay_many_cuda, OverlayManyCUDAContext, fs); ++ ++static const AVFilterPad overlay_many_cuda_outputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .config_props = overlay_many_cuda_config_output, ++ }, ++}; ++ ++const FFFilter ff_vf_overlay_many_cuda = { ++ .p.name = "overlay_many_cuda", ++ .p.description = NULL_IF_CONFIG_SMALL( ++ "Overlay multiple videos on top of each other using CUDA"), ++ .priv_size = sizeof(OverlayManyCUDAContext), ++ .p.priv_class = &overlay_many_cuda_class, ++ .init = overlay_many_cuda_init, ++ .uninit = overlay_many_cuda_uninit, ++ .activate = overlay_many_cuda_activate, ++ FILTER_OUTPUTS(overlay_many_cuda_outputs), ++ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), ++ .preinit = overlay_many_cuda_framesync_preinit, ++ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, ++}; +diff --git a/libavfilter/vf_overlay_many_cuda.cu b/libavfilter/vf_overlay_many_cuda.cu +new file mode 100644 +index 0000000000..46d45df61c +--- /dev/null ++++ b/libavfilter/vf_overlay_many_cuda.cu +@@ -0,0 +1,252 @@ ++/* ++ * This file is part of FFmpeg. ++ * ++ * FFmpeg is free software; you can redistribute it and/or ++ * modify it under the terms of the GNU Lesser General Public ++ * License as published by the Free Software Foundation; either ++ * version 2.1 of the License, or (at your option) any later version. ++ * ++ * FFmpeg is distributed in the hope that it will be useful, ++ * but WITHOUT ANY WARRANTY; without even the implied warranty of ++ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ * Lesser General Public License for more details. ++ * ++ * You should have received a copy of the GNU Lesser General Public ++ * License along with FFmpeg; if not, write to the Free Software ++ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA ++ */ ++ ++#include "vf_overlay_many_cuda.h" ++ ++#if !defined(__CUDACC_VER_MAJOR__) || \ ++ (__CUDACC_VER_MAJOR__ < 11) || \ ++ (__CUDACC_VER_MAJOR__ == 11 && __CUDACC_VER_MINOR__ < 7) ++#error "overlay_many_cuda requires CUDA Toolkit 11.7 or newer" ++#endif ++ ++#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ < 700 ++#error "overlay_many_cuda requires compute capability 7.0 or newer" ++#endif ++ ++static __device__ __forceinline__ const unsigned char * ++plane_src(const OverlayManyCUDAPlane &plane) ++{ ++ return (const unsigned char *)(uintptr_t)plane.ptr; ++} ++ ++static __device__ __forceinline__ unsigned char * ++plane_dst(const OverlayManyCUDAPlane &plane) ++{ ++ return (unsigned char *)(uintptr_t)plane.ptr; ++} ++ ++static __device__ __forceinline__ unsigned int ++load_sample(const OverlayManyCUDAPlane &plane, unsigned int x, unsigned int y) ++{ ++ return plane_src(plane)[x + (size_t)y * plane.pitch]; ++} ++ ++static __device__ __forceinline__ void ++store_sample(const OverlayManyCUDAPlane &plane, unsigned int x, unsigned int y, ++ unsigned int value) ++{ ++ plane_dst(plane)[x + (size_t)y * plane.pitch] = (unsigned char)value; ++} ++ ++static __device__ __forceinline__ unsigned int ++blend_sample(unsigned int background, unsigned int foreground, ++ unsigned int alpha) ++{ ++ unsigned int sum = alpha * foreground + (255U - alpha) * background; ++ unsigned int rounded = sum + 128U; ++ ++ /* Exact rounded division by 255 without an integer division. */ ++ return (rounded + (rounded >> 8)) >> 8; ++} ++ ++static __device__ __forceinline__ unsigned int ++alpha_420(const OverlayManyCUDAFrame &overlay, unsigned int x, unsigned int y, ++ unsigned int width, unsigned int height) ++{ ++ const OverlayManyCUDAPlane &alpha = overlay.plane[3]; ++ unsigned int sum = load_sample(alpha, x, y); ++ unsigned int count = 1; ++ ++ if (x + 1 < width) { ++ sum += load_sample(alpha, x + 1, y); ++ count++; ++ } ++ if (y + 1 < height) { ++ sum += load_sample(alpha, x, y + 1); ++ count++; ++ if (x + 1 < width) { ++ sum += load_sample(alpha, x + 1, y + 1); ++ count++; ++ } ++ } ++ ++ return (sum + count / 2) / count; ++} ++ ++extern "C" { ++ ++__global__ void OverlayMany420( ++ const __grid_constant__ OverlayManyCUDAParams params) ++{ ++ unsigned int x = blockIdx.x * blockDim.x + threadIdx.x; ++ unsigned int y = blockIdx.y * blockDim.y + threadIdx.y; ++ ++ if (x >= params.width || y >= params.height) ++ return; ++ ++ unsigned int luma = load_sample(params.main.plane[0], x, y); ++ ++ for (unsigned int i = 0; i < params.overlay_count; i++) { ++ const OverlayManyCUDAFrame &overlay = params.overlay[i]; ++ unsigned int alpha = load_sample(overlay.plane[3], x, y); ++ ++ luma = blend_sample(luma, load_sample(overlay.plane[0], x, y), alpha); ++ } ++ store_sample(params.dst.plane[0], x, y, luma); ++ ++ if ((x & 1) || (y & 1)) ++ return; ++ ++ unsigned int cx = x >> 1; ++ unsigned int cy = y >> 1; ++ unsigned int chroma_u = load_sample(params.main.plane[1], cx, cy); ++ unsigned int chroma_v = load_sample(params.main.plane[2], cx, cy); ++ ++ for (unsigned int i = 0; i < params.overlay_count; i++) { ++ const OverlayManyCUDAFrame &overlay = params.overlay[i]; ++ unsigned int alpha = alpha_420(overlay, x, y, ++ params.width, params.height); ++ ++ chroma_u = blend_sample(chroma_u, ++ load_sample(overlay.plane[1], cx, cy), alpha); ++ chroma_v = blend_sample(chroma_v, ++ load_sample(overlay.plane[2], cx, cy), alpha); ++ } ++ ++ store_sample(params.dst.plane[1], cx, cy, chroma_u); ++ store_sample(params.dst.plane[2], cx, cy, chroma_v); ++} ++ ++__global__ void OverlayMany444( ++ const __grid_constant__ OverlayManyCUDAParams params) ++{ ++ unsigned int x = blockIdx.x * blockDim.x + threadIdx.x; ++ unsigned int y = blockIdx.y * blockDim.y + threadIdx.y; ++ ++ if (x >= params.width || y >= params.height) ++ return; ++ ++ unsigned int luma = load_sample(params.main.plane[0], x, y); ++ unsigned int chroma_u = load_sample(params.main.plane[1], x, y); ++ unsigned int chroma_v = load_sample(params.main.plane[2], x, y); ++ ++ for (unsigned int i = 0; i < params.overlay_count; i++) { ++ const OverlayManyCUDAFrame &overlay = params.overlay[i]; ++ unsigned int alpha = load_sample(overlay.plane[3], x, y); ++ ++ luma = blend_sample(luma, load_sample(overlay.plane[0], x, y), alpha); ++ chroma_u = blend_sample(chroma_u, ++ load_sample(overlay.plane[1], x, y), alpha); ++ chroma_v = blend_sample(chroma_v, ++ load_sample(overlay.plane[2], x, y), alpha); ++ } ++ ++ store_sample(params.dst.plane[0], x, y, luma); ++ store_sample(params.dst.plane[1], x, y, chroma_u); ++ store_sample(params.dst.plane[2], x, y, chroma_v); ++} ++ ++__global__ void OverlayMany420From444( ++ const __grid_constant__ OverlayManyCUDAParams params) ++{ ++ unsigned int x = blockIdx.x * blockDim.x + threadIdx.x; ++ unsigned int y = blockIdx.y * blockDim.y + threadIdx.y; ++ ++ if (x >= params.width || y >= params.height) ++ return; ++ ++ unsigned int luma = load_sample(params.main.plane[0], x, y); ++ ++ for (unsigned int i = 0; i < params.overlay_count; i++) { ++ const OverlayManyCUDAFrame &overlay = params.overlay[i]; ++ unsigned int alpha = load_sample(overlay.plane[3], x, y); ++ ++ luma = blend_sample(luma, load_sample(overlay.plane[0], x, y), alpha); ++ } ++ store_sample(params.dst.plane[0], x, y, luma); ++ ++ if ((x & 1) || (y & 1)) ++ return; ++ ++ unsigned int cx = x >> 1; ++ unsigned int cy = y >> 1; ++ unsigned int base_u = load_sample(params.main.plane[1], cx, cy); ++ unsigned int base_v = load_sample(params.main.plane[2], cx, cy); ++ ++ /* ++ * Compose all four 4:4:4 chroma positions independently, then downsample ++ * the final values. Downsampling each layer first changes alpha semantics. ++ */ ++ unsigned int u00 = base_u; ++ unsigned int u01 = base_u; ++ unsigned int u10 = base_u; ++ unsigned int u11 = base_u; ++ unsigned int v00 = base_v; ++ unsigned int v01 = base_v; ++ unsigned int v10 = base_v; ++ unsigned int v11 = base_v; ++ bool has_x1 = x + 1 < params.width; ++ bool has_y1 = y + 1 < params.height; ++ ++ for (unsigned int i = 0; i < params.overlay_count; i++) { ++ const OverlayManyCUDAFrame &overlay = params.overlay[i]; ++ unsigned int alpha = load_sample(overlay.plane[3], x, y); ++ ++ u00 = blend_sample(u00, load_sample(overlay.plane[1], x, y), alpha); ++ v00 = blend_sample(v00, load_sample(overlay.plane[2], x, y), alpha); ++ ++ if (has_x1) { ++ alpha = load_sample(overlay.plane[3], x + 1, y); ++ u01 = blend_sample(u01, ++ load_sample(overlay.plane[1], x + 1, y), alpha); ++ v01 = blend_sample(v01, ++ load_sample(overlay.plane[2], x + 1, y), alpha); ++ } ++ ++ if (has_y1) { ++ alpha = load_sample(overlay.plane[3], x, y + 1); ++ u10 = blend_sample(u10, ++ load_sample(overlay.plane[1], x, y + 1), alpha); ++ v10 = blend_sample(v10, ++ load_sample(overlay.plane[2], x, y + 1), alpha); ++ ++ if (has_x1) { ++ alpha = load_sample(overlay.plane[3], x + 1, y + 1); ++ u11 = blend_sample( ++ u11, load_sample(overlay.plane[1], x + 1, y + 1), alpha); ++ v11 = blend_sample( ++ v11, load_sample(overlay.plane[2], x + 1, y + 1), alpha); ++ } ++ } ++ } ++ ++ unsigned int sample_count = (1U + has_x1) * (1U + has_y1); ++ unsigned int sum_u = u00 + (has_x1 ? u01 : 0) + ++ (has_y1 ? u10 : 0) + ++ (has_x1 && has_y1 ? u11 : 0); ++ unsigned int sum_v = v00 + (has_x1 ? v01 : 0) + ++ (has_y1 ? v10 : 0) + ++ (has_x1 && has_y1 ? v11 : 0); ++ ++ store_sample(params.dst.plane[1], cx, cy, ++ (sum_u + sample_count / 2) / sample_count); ++ store_sample(params.dst.plane[2], cx, cy, ++ (sum_v + sample_count / 2) / sample_count); ++} ++ ++} /* extern "C" */ +diff --git a/libavfilter/vf_overlay_many_cuda.h b/libavfilter/vf_overlay_many_cuda.h +new file mode 100644 +index 0000000000..4f42b883cd +--- /dev/null ++++ b/libavfilter/vf_overlay_many_cuda.h +@@ -0,0 +1,68 @@ ++/* ++ * CUDA overlay-many launch descriptor ++ * ++ * This file is part of FFmpeg. ++ * ++ * FFmpeg is free software; you can redistribute it and/or ++ * modify it under the terms of the GNU Lesser General Public ++ * License as published by the Free Software Foundation; either ++ * version 2.1 of the License, or (at your option) any later version. ++ * ++ * FFmpeg is distributed in the hope that it will be useful, ++ * but WITHOUT ANY WARRANTY; without even the implied warranty of ++ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ * Lesser General Public License for more details. ++ * ++ * You should have received a copy of the GNU Lesser General Public ++ * License along with FFmpeg; if not, write to the Free Software ++ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA ++ */ ++ ++#ifndef AVFILTER_VF_OVERLAY_MANY_CUDA_H ++#define AVFILTER_VF_OVERLAY_MANY_CUDA_H ++ ++#include ++#include ++ ++#define OVERLAY_MANY_CUDA_MAX_OVERLAYS 15 ++ ++typedef struct OverlayManyCUDAPlane { ++ uint64_t ptr; ++ uint32_t pitch; ++ uint32_t reserved; ++} OverlayManyCUDAPlane; ++ ++typedef struct OverlayManyCUDAFrame { ++ OverlayManyCUDAPlane plane[4]; ++} OverlayManyCUDAFrame; ++ ++typedef struct OverlayManyCUDAParams { ++ OverlayManyCUDAFrame dst; ++ OverlayManyCUDAFrame main; ++ uint32_t width; ++ uint32_t height; ++ uint32_t overlay_count; ++ uint32_t reserved; ++ OverlayManyCUDAFrame overlay[OVERLAY_MANY_CUDA_MAX_OVERLAYS]; ++} OverlayManyCUDAParams; ++ ++#ifdef __CUDACC__ ++#define OVERLAY_MANY_CUDA_STATIC_ASSERT static_assert ++#else ++#define OVERLAY_MANY_CUDA_STATIC_ASSERT _Static_assert ++#endif ++ ++OVERLAY_MANY_CUDA_STATIC_ASSERT(sizeof(OverlayManyCUDAPlane) == 16, ++ "unexpected CUDA plane descriptor layout"); ++OVERLAY_MANY_CUDA_STATIC_ASSERT(sizeof(uintptr_t) <= sizeof(uint64_t), ++ "CUDA device pointers do not fit the descriptor"); ++OVERLAY_MANY_CUDA_STATIC_ASSERT(sizeof(OverlayManyCUDAFrame) == 64, ++ "unexpected CUDA frame descriptor layout"); ++OVERLAY_MANY_CUDA_STATIC_ASSERT(offsetof(OverlayManyCUDAParams, overlay) == 144, ++ "unexpected CUDA overlay array offset"); ++OVERLAY_MANY_CUDA_STATIC_ASSERT(sizeof(OverlayManyCUDAParams) == 1104, ++ "unexpected CUDA launch parameter size"); ++ ++#undef OVERLAY_MANY_CUDA_STATIC_ASSERT ++ ++#endif /* AVFILTER_VF_OVERLAY_MANY_CUDA_H */ +diff --git a/libavfilter/vf_pad_cuda.c b/libavfilter/vf_pad_cuda.c +index 425d6d2a94..7003c1a8df 100644 +--- a/libavfilter/vf_pad_cuda.c ++++ b/libavfilter/vf_pad_cuda.c +@@ -1,636 +1,538 @@ +-/* +- * This file is part of FFmpeg. +- * +- * FFmpeg is free software; you can redistribute it and/or +- * modify it under the terms of the GNU Lesser General Public +- * License as published by the Free Software Foundation; either +- * version 2.1 of the License, or (at your option) any later version. +- * +- * FFmpeg is distributed in the hope that it will be useful, +- * but WITHOUT ANY WARRANTY; without even the implied warranty of +- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +- * Lesser General Public License for more details. +- * +- * You should have received a copy of the GNU Lesser General Public +- * License along with FFmpeg; if not, write to the Free Software +- * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +- */ +- +-/** +- * @file +- * CUDA video padding filter +- */ +- +-#include +- +-#include "filters.h" +-#include "libavutil/avstring.h" +-#include "libavutil/common.h" +-#include "libavutil/cuda_check.h" +-#include "libavutil/eval.h" + #include "libavutil/hwcontext.h" + #include "libavutil/hwcontext_cuda_internal.h" +-#include "libavutil/imgutils.h" +-#include "libavutil/internal.h" ++#include "libavutil/cuda_check.h" + #include "libavutil/opt.h" + #include "libavutil/pixdesc.h" ++#include "libavutil/eval.h" + #include "libavutil/colorspace.h" + ++#include "avfilter.h" ++#include "formats.h" ++#include "avfilter_internal.h" ++#include "video.h" ++ + #include "cuda/load_helper.h" + +-#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, device_hwctx->internal->cuda_dl, x) +-#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) ) + #define BLOCK_X 32 + #define BLOCK_Y 16 + +-static const enum AVPixelFormat supported_formats[] = { +- AV_PIX_FMT_YUV420P, +- AV_PIX_FMT_YUV444P, +- AV_PIX_FMT_YUVA420P, +- AV_PIX_FMT_YUVA444P, +- AV_PIX_FMT_NV12, ++enum var_name { ++ VAR_IN_W, VAR_IW, ++ VAR_IN_H, VAR_IH, ++ VAR_OUT_W, VAR_OW, ++ VAR_OUT_H, VAR_OH, ++ VAR_X, ++ VAR_Y, ++ VAR_A, ++ VAR_SAR, ++ VAR_DAR, ++ VARS_NB + }; + +-typedef struct CUDAPadContext { +- const AVClass *class; +- +- AVBufferRef *frames_ctx; +- +- int w, h; ///< output dimensions, a value of 0 will result in the input size +- int x, y; ///< offsets of the input area with respect to the padded area +- int in_w, in_h; ///< width and height for the padded input video +- +- char *w_expr; ///< width expression +- char *h_expr; ///< height expression +- char *x_expr; ///< x offset expression +- char *y_expr; ///< y offset expression +- +- uint8_t rgba_color[4]; ///< color for the padding area +- uint8_t parsed_color[4]; +- AVRational aspect; +- +- int eval_mode; +- +- int last_out_w, last_out_h; ///< used to evaluate the prior output width and height with the incoming frame +- +- AVCUDADeviceContext *hwctx; +- CUmodule cu_module; +- CUfunction cu_func_uchar; +- CUfunction cu_func_uchar2; +-} CUDAPadContext; +- + static const char *const var_names[] = { +- "in_w", "iw", +- "in_h", "ih", +- "out_w", "ow", +- "out_h", "oh", ++ "in_w", "iw", ++ "in_h", "ih", ++ "out_w", "ow", ++ "out_h", "oh", + "x", + "y", + "a", + "sar", + "dar", +- "hsub", +- "vsub", + NULL + }; + +-enum { +- VAR_IN_W, +- VAR_IW, +- VAR_IN_H, +- VAR_IH, +- VAR_OUT_W, +- VAR_OW, +- VAR_OUT_H, +- VAR_OH, +- VAR_X, +- VAR_Y, +- VAR_A, +- VAR_SAR, +- VAR_DAR, +- VAR_HSUB, +- VAR_VSUB, +- VARS_NB +-}; +- +-enum EvalMode { +- EVAL_MODE_INIT, +- EVAL_MODE_FRAME, +- EVAL_MODE_NB ++static enum AVPixelFormat supported_formats[] = { ++ AV_PIX_FMT_NV12, ++ AV_PIX_FMT_YUV420P, AV_PIX_FMT_YUVA420P, ++ AV_PIX_FMT_YUV444P, AV_PIX_FMT_YUVA444P, + }; + +-static int eval_expr(AVFilterContext *ctx) +-{ +- CUDAPadContext *s = ctx->priv; +- AVFilterLink *inlink = ctx->inputs[0]; +- const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(inlink->format); +- +- double var_values[VARS_NB], res; +- char *expr; +- int ret; +- +- var_values[VAR_IN_W] = var_values[VAR_IW] = s->in_w; +- var_values[VAR_IN_H] = var_values[VAR_IH] = s->in_h; +- var_values[VAR_OUT_W] = var_values[VAR_OW] = NAN; +- var_values[VAR_OUT_H] = var_values[VAR_OH] = NAN; +- var_values[VAR_A] = (double)s->in_w / s->in_h; +- var_values[VAR_SAR] = inlink->sample_aspect_ratio.num ? +- (double)inlink->sample_aspect_ratio.num / +- inlink->sample_aspect_ratio.den : 1; +- var_values[VAR_DAR] = var_values[VAR_A] * var_values[VAR_SAR]; +- var_values[VAR_HSUB] = 1 << desc->log2_chroma_w; +- var_values[VAR_VSUB] = 1 << desc->log2_chroma_h; +- +- expr = s->w_expr; +- ret = av_expr_parse_and_eval(&res, expr, var_names, var_values, NULL, NULL, NULL, NULL, NULL, 0, ctx); +- if (ret < 0) +- goto fail; +- +- s->w = res; +- if (s->w < 0) { +- av_log(ctx, AV_LOG_ERROR, "Width expression is negative.\n"); +- ret = AVERROR(EINVAL); +- goto fail; +- } +- +- var_values[VAR_OUT_W] = var_values[VAR_OW] = s->w; +- +- expr = s->h_expr; +- ret = av_expr_parse_and_eval(&res, expr, var_names, var_values, NULL, NULL, NULL, NULL, NULL, 0, ctx); +- if (ret < 0) +- goto fail; +- +- s->h = res; +- if (s->h < 0) { +- av_log(ctx, AV_LOG_ERROR, "Height expression is negative.\n"); +- ret = AVERROR(EINVAL); +- goto fail; +- } +- var_values[VAR_OUT_H] = var_values[VAR_OH] = s->h; +- +- if (!s->h) +- s->h = s->in_h; +- +- var_values[VAR_OUT_H] = var_values[VAR_OH] = s->h; +- +- +- expr = s->w_expr; +- ret = av_expr_parse_and_eval(&res, expr, var_names, var_values, NULL, NULL, NULL, NULL, NULL, 0, ctx); +- if (ret < 0) +- goto fail; +- +- s->w = res; +- if (s->w < 0) { +- av_log(ctx, AV_LOG_ERROR, "Width expression is negative.\n"); +- ret = AVERROR(EINVAL); +- goto fail; +- } +- if (!s->w) +- s->w = s->in_w; +- +- var_values[VAR_OUT_W] = var_values[VAR_OW] = s->w; +- +- +- expr = s->x_expr; +- ret = av_expr_parse_and_eval(&res, expr, var_names, var_values, NULL, NULL, NULL, NULL, NULL, 0, ctx); +- if (ret < 0) +- goto fail; +- +- s->x = res; ++#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) ) ++#define BLOCKX 32 ++#define BLOCKY 16 + ++#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, cu, x) + +- expr = s->y_expr; +- ret = av_expr_parse_and_eval(&res, expr, var_names, var_values, NULL, NULL, NULL, NULL, NULL, 0, ctx); +- if (ret < 0) +- goto fail; +- +- s->y = res; ++typedef struct PadCUDAContext { ++ const AVClass *class; ++ AVCUDADeviceContext *hwctx; ++ AVBufferRef *hw_device_ctx; ++ ++ AVBufferRef *frames_ctx; ++ AVFrame *frame; ++ AVFrame *tmp_frame; + +- if (s->x < 0 || s->x + s->in_w > s->w) { +- s->x = (s->w - s->in_w) / 2; +- av_log(ctx, AV_LOG_VERBOSE, "centering X offset.\n"); +- } ++ CUmodule cu_module; ++ CUfunction cu_func; ++ CUstream cu_stream; + +- if (s->y < 0 || s->y + s->in_h > s->h) { +- s->y = (s->h - s->in_h) / 2; +- av_log(ctx, AV_LOG_VERBOSE, "centering Y offset.\n"); +- } ++ enum AVPixelFormat sw_format; + +- s->w = av_clip(s->w, 1, INT_MAX); +- s->h = av_clip(s->h, 1, INT_MAX); ++ char *w_expr, *h_expr, *x_expr, *y_expr; ++ AVRational aspect; ++ int w, h, x, y; ++ uint8_t pad_rgba[4]; ++ uint8_t pad_color[4]; ++} PadCUDAContext; + +- if (s->w < s->in_w || s->h < s->in_h) { +- av_log(ctx, AV_LOG_ERROR, "Padded size < input size.\n"); +- return AVERROR(EINVAL); +- } + +- av_log(ctx, AV_LOG_DEBUG, +- "w:%d h:%d -> w:%d h:%d x:%d y:%d color:0x%02X%02X%02X%02X\n", +- inlink->w, inlink->h, s->w, s->h, s->x, s->y, s->rgba_color[0], +- s->rgba_color[1], s->rgba_color[2], s->rgba_color[3]); + ++static int format_is_supported(enum AVPixelFormat fmt) ++{ ++ for (int i = 0; i < FF_ARRAY_ELEMS(supported_formats); i++) ++ if (supported_formats[i] == fmt) ++ return 1; + return 0; +- +-fail: +- av_log(ctx, AV_LOG_ERROR, "Error evaluating '%s'\n", expr); +- return ret; + } + +-static int cuda_pad_alloc_out_frames_ctx(AVFilterContext *ctx, AVBufferRef **out_frames_ctx, const int width, const int height) ++static av_cold int pad_cuda_init(AVFilterContext *avctx) + { +- AVFilterLink *inlink = ctx->inputs[0]; +- FilterLink *inl = ff_filter_link(inlink); +- AVHWFramesContext *in_frames_ctx = (AVHWFramesContext *)inl->hw_frames_ctx->data; +- int ret; ++ PadCUDAContext *ctx = avctx->priv; + +- *out_frames_ctx = av_hwframe_ctx_alloc(in_frames_ctx->device_ref); +- if (!*out_frames_ctx) { ++ ctx->frame = av_frame_alloc(); ++ if (!ctx->frame) + return AVERROR(ENOMEM); +- } + +- AVHWFramesContext *out_fc = (AVHWFramesContext *)(*out_frames_ctx)->data; +- out_fc->format = AV_PIX_FMT_CUDA; +- out_fc->sw_format = in_frames_ctx->sw_format; +- +- out_fc->width = FFALIGN(width, 32); +- out_fc->height = FFALIGN(height, 32); +- +- ret = av_hwframe_ctx_init(*out_frames_ctx); +- if (ret < 0) { +- av_log(ctx, AV_LOG_ERROR, "Failed to init output ctx\n"); +- av_buffer_unref(out_frames_ctx); +- return ret; +- } +- +- return 0; +-} +- +-static av_cold int cuda_pad_init(AVFilterContext *ctx) +-{ +- CUDAPadContext *s = ctx->priv; +- +- s->last_out_w = -1; +- s->last_out_h = -1; ++ ctx->tmp_frame = av_frame_alloc(); ++ if (!ctx->tmp_frame) ++ return AVERROR(ENOMEM); + + return 0; + } + +-static av_cold void cuda_pad_uninit(AVFilterContext *ctx) ++static av_cold void pad_cuda_uninit(AVFilterContext *avctx) + { +- CUDAPadContext *s = ctx->priv; +- CUcontext dummy; ++ PadCUDAContext *ctx = avctx->priv; + +- av_buffer_unref(&s->frames_ctx); ++ if (ctx->hwctx && ctx->cu_module) { ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext dummy; + +- if (s->hwctx && s->cu_module) { +- CudaFunctions *cu = s->hwctx->internal->cuda_dl; +- AVCUDADeviceContext *device_hwctx = s->hwctx; +- CHECK_CU(cu->cuCtxPushCurrent(s->hwctx->cuda_ctx)); +- CHECK_CU(cu->cuModuleUnload(s->cu_module)); ++ CHECK_CU(cu->cuCtxPushCurrent(ctx->hwctx->cuda_ctx)); ++ CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); + CHECK_CU(cu->cuCtxPopCurrent(&dummy)); + } + +- s->cu_module = NULL; +- s->hwctx = NULL; ++ av_frame_free(&ctx->frame); ++ av_buffer_unref(&ctx->hw_device_ctx); ++ av_buffer_unref(&ctx->frames_ctx); ++ av_frame_free(&ctx->tmp_frame); ++ ctx->hwctx = NULL; + } + +-static av_cold int cuda_pad_load_functions(AVFilterContext *ctx) ++static av_cold int pad_cuda_output_config_props(AVFilterLink *outlink) + { +- CUDAPadContext *s = ctx->priv; +- CudaFunctions *cu = s->hwctx->internal->cuda_dl; +- CUcontext dummy_cu_ctx; +- int ret; +- +- AVCUDADeviceContext *device_hwctx = s->hwctx; +- + extern const unsigned char ff_vf_pad_cuda_ptx_data[]; + extern const unsigned int ff_vf_pad_cuda_ptx_len; + +- ret = CHECK_CU(cu->cuCtxPushCurrent(device_hwctx->cuda_ctx)); +- if (ret < 0) +- return ret; +- +- ret = ff_cuda_load_module(ctx, device_hwctx, &s->cu_module, +- ff_vf_pad_cuda_ptx_data, ff_vf_pad_cuda_ptx_len); +- if (ret < 0) { +- av_log(ctx, AV_LOG_ERROR, "Failed to load CUDA module\n"); +- goto end; +- } +- +- ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_uchar, s->cu_module, "pad_uchar")); +- if (ret < 0) { +- av_log(ctx, AV_LOG_ERROR, "Failed to load pad_planar_cuda\n"); +- goto end; +- } +- +- ret = CHECK_CU(cu->cuModuleGetFunction(&s->cu_func_uchar2, s->cu_module, "pad_uchar2")); +- if (ret < 0) +- av_log(ctx, AV_LOG_ERROR, "Failed to load pad_uv_cuda\n"); +- +-end: +- CHECK_CU(cu->cuCtxPopCurrent(&dummy_cu_ctx)); +- +- return ret; +-} +- +-static int cuda_pad_config_props(AVFilterLink *outlink) +-{ +- AVFilterContext *ctx = outlink->src; +- CUDAPadContext *s = ctx->priv; +- +- AVFilterLink *inlink = ctx->inputs[0]; ++ FilterLink *outl = ff_filter_link(outlink); ++ AVFilterContext *avctx = outlink->src; ++ AVFilterLink *inlink = outlink->src->inputs[0]; + FilterLink *inl = ff_filter_link(inlink); +- +- FilterLink *ol = ff_filter_link(outlink); +- +- AVHWFramesContext *in_frames_ctx; +- int format_supported = 0; +- int ret; +- +- s->in_w = inlink->w; +- s->in_h = inlink->h; +- ret = eval_expr(ctx); +- if (ret < 0) +- return ret; +- +- if (!inl->hw_frames_ctx) { +- av_log(ctx, AV_LOG_ERROR, "No hw context provided on input\n"); ++ PadCUDAContext *ctx = avctx->priv; ++ AVHWFramesContext *frames_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; ++ ++ AVRational adjusted_aspect = ctx->aspect; ++ double var_values[VARS_NB], res; ++ int err, ret; ++ ++ if (!frames_ctx) { ++ av_log(ctx, AV_LOG_ERROR, "No hw context provided on main input\n"); + return AVERROR(EINVAL); + } + +- in_frames_ctx = (AVHWFramesContext *)inl->hw_frames_ctx->data; +- s->hwctx = in_frames_ctx->device_ctx->hwctx; +- +- for (int i = 0; i < FF_ARRAY_ELEMS(supported_formats); i++) { +- if (in_frames_ctx->sw_format == supported_formats[i]) { +- format_supported = 1; +- break; +- } ++ ctx->sw_format = frames_ctx->sw_format; ++ if (!format_is_supported(ctx->sw_format)) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported input format: %s\n", ++ av_get_pix_fmt_name(ctx->sw_format)); ++ return AVERROR(ENOSYS); + } +- if (!format_supported) { +- av_log(ctx, AV_LOG_ERROR, "Unsupported input format.\n"); +- return AVERROR(EINVAL); +- } +- +- s->parsed_color[0] = RGB_TO_Y_BT709(s->rgba_color[0], s->rgba_color[1], s->rgba_color[2]); +- s->parsed_color[1] = RGB_TO_U_BT709(s->rgba_color[0], s->rgba_color[1], s->rgba_color[2], 0); +- s->parsed_color[2] = RGB_TO_V_BT709(s->rgba_color[0], s->rgba_color[1], s->rgba_color[2], 0); +- s->parsed_color[3] = s->rgba_color[3]; + +- ret = cuda_pad_alloc_out_frames_ctx(ctx, &s->frames_ctx, s->w, s->h); +- if (ret < 0) +- return ret; +- +- ol->hw_frames_ctx = av_buffer_ref(s->frames_ctx); +- if (!ol->hw_frames_ctx) ++ // initialize ++ ctx->hw_device_ctx = av_buffer_ref(frames_ctx->device_ref); ++ if (!ctx->hw_device_ctx) + return AVERROR(ENOMEM); ++ ctx->hwctx = ((AVHWDeviceContext*)frames_ctx->device_ref->data)->hwctx; ++ ctx->cu_stream = ctx->hwctx->stream; ++ ++ // load functions ++ { ++ CUcontext cuda_ctx = ctx->hwctx->cuda_ctx; ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext dummy; ++ ++ err = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); ++ if (err < 0) { ++ return err; ++ } + +- outlink->w = s->w; +- outlink->h = s->h; +- outlink->time_base = inlink->time_base; +- outlink->format = AV_PIX_FMT_CUDA; +- +- s->last_out_w = s->w; +- s->last_out_h = s->h; +- +- ret = cuda_pad_load_functions(ctx); +- if (ret < 0) +- return ret; +- +- return 0; +-} +- +-static int cuda_pad_pad(AVFilterContext *ctx, AVFrame *out, const AVFrame *in) +-{ +- CUDAPadContext *s = ctx->priv; +- FilterLink *inl = ff_filter_link(ctx->inputs[0]); +- +- AVHWFramesContext *in_frames_ctx = (AVHWFramesContext *)inl->hw_frames_ctx->data; +- const AVPixFmtDescriptor *pixdesc = av_pix_fmt_desc_get(in_frames_ctx->sw_format); +- +- CudaFunctions *cu = s->hwctx->internal->cuda_dl; +- AVCUDADeviceContext *device_hwctx = s->hwctx; +- int ret; +- ++ err = ff_cuda_load_module(ctx, ctx->hwctx, &ctx->cu_module, ff_vf_pad_cuda_ptx_data, ff_vf_pad_cuda_ptx_len); ++ if (err < 0) { ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return err; ++ } + +- const int nb_planes = av_pix_fmt_count_planes(in_frames_ctx->sw_format); +- for (int plane = 0; plane < nb_planes; plane++) { +- const AVComponentDescriptor *cur_comp = &pixdesc->comp[0]; +- for (int comp = 1; comp < pixdesc->nb_components && cur_comp->plane != plane; comp++) +- cur_comp = &pixdesc->comp[comp]; ++ err = CHECK_CU(cu->cuModuleGetFunction(&ctx->cu_func, ctx->cu_module, "Pad_Cuda")); ++ if (err < 0) { ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return err; ++ } + +- int hsub = (plane == 1 || plane == 2) ? pixdesc->log2_chroma_w : 0; +- int vsub = (plane == 1 || plane == 2) ? pixdesc->log2_chroma_h : 0; ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ } + +- int src_w = AV_CEIL_RSHIFT(s->in_w, hsub); +- int src_h = AV_CEIL_RSHIFT(s->in_h, vsub); ++ // process filter parameters ++ { ++ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(ctx->sw_format); ++ ++ var_values[VAR_IN_W] = var_values[VAR_IW] = inlink->w; ++ var_values[VAR_IN_H] = var_values[VAR_IH] = inlink->h; ++ var_values[VAR_OUT_W] = var_values[VAR_OW] = NAN; ++ var_values[VAR_OUT_H] = var_values[VAR_OH] = NAN; ++ var_values[VAR_A] = (double) inlink->w / inlink->h; ++ var_values[VAR_SAR] = inlink->sample_aspect_ratio.num ? ++ (double) inlink->sample_aspect_ratio.num / inlink->sample_aspect_ratio.den : 1; ++ var_values[VAR_DAR] = var_values[VAR_A] * var_values[VAR_SAR]; ++ ++ av_expr_parse_and_eval(&res, ctx->w_expr, ++ var_names, var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx); ++ ctx->w = var_values[VAR_OUT_W] = var_values[VAR_OW] = res; ++ if ((ret = av_expr_parse_and_eval(&res, ctx->h_expr, ++ var_names, var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) ++ return ret; ++ ctx->h = var_values[VAR_OUT_H] = var_values[VAR_OH] = res; ++ if (!ctx->h) ++ var_values[VAR_OUT_H] = var_values[VAR_OH] = ctx->h = inlink->h; ++ ++ /* evaluate the width again, as it may depend on the evaluated output height */ ++ if ((ret = av_expr_parse_and_eval(&res, ctx->w_expr, ++ var_names, var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) ++ return ret; ++ ctx->w = var_values[VAR_OUT_W] = var_values[VAR_OW] = res; ++ if (!ctx->w) ++ var_values[VAR_OUT_W] = var_values[VAR_OW] = ctx->w = inlink->w; ++ ++ if (adjusted_aspect.num && adjusted_aspect.den) { ++ adjusted_aspect = av_div_q(adjusted_aspect, inlink->sample_aspect_ratio); ++ if (ctx->h < av_rescale(ctx->w, adjusted_aspect.den, adjusted_aspect.num)) { ++ ctx->h = var_values[VAR_OUT_H] = var_values[VAR_OH] = av_rescale(ctx->w, adjusted_aspect.den, adjusted_aspect.num); ++ } else { ++ ctx->w = var_values[VAR_OUT_W] = var_values[VAR_OW] = av_rescale(ctx->h, adjusted_aspect.num, adjusted_aspect.den); ++ } ++ } + +- int dst_w = AV_CEIL_RSHIFT(s->w, hsub); +- int dst_h = AV_CEIL_RSHIFT(s->h, vsub); ++ /* evaluate x and y */ ++ av_expr_parse_and_eval(&res, ctx->x_expr, ++ var_names, var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx); ++ ctx->x = var_values[VAR_X] = res; ++ if ((ret = av_expr_parse_and_eval(&res, ctx->y_expr, ++ var_names, var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) ++ return ret; ++ ctx->y = var_values[VAR_Y] = res; ++ /* evaluate x again, as it may depend on the evaluated y value */ ++ if ((ret = av_expr_parse_and_eval(&res, ctx->x_expr, ++ var_names, var_values, ++ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) ++ return ret; ++ ctx->x = var_values[VAR_X] = res; + +- int y_plane_offset = AV_CEIL_RSHIFT(s->y, vsub); +- int x_plane_offset = AV_CEIL_RSHIFT(s->x, hsub); ++ if (ctx->x < 0 || ctx->x + inlink->w > ctx->w) ++ ctx->x = var_values[VAR_X] = (ctx->w - inlink->w) / 2; ++ if (ctx->y < 0 || ctx->y + inlink->h > ctx->h) ++ ctx->y = var_values[VAR_Y] = (ctx->h - inlink->h) / 2; + +- if (x_plane_offset + src_w > dst_w || y_plane_offset + src_h > dst_h) { +- av_log(ctx, AV_LOG_ERROR, +- "ROI out of bounds in plane %d: offset=(%d,%d) in=(%dx%d) " +- "out=(%dx%d)\n", +- plane, x_plane_offset, y_plane_offset, src_w, src_h, dst_w, dst_h); ++ /* sanity check params */ ++ if (ctx->w < inlink->w || ctx->h < inlink->h) { ++ av_log(ctx, AV_LOG_ERROR, "Padded dimensions cannot be smaller than input dimensions.\n"); + return AVERROR(EINVAL); + } + +- int dst_linesize = out->linesize[plane] / cur_comp->step; +- int src_linesize = in->linesize[plane] / cur_comp->step; ++ // Align x, y offsets between planes ++ ctx->x &= ~((1 << desc->log2_chroma_w) - 1); ++ ctx->y &= ~((1 << desc->log2_chroma_h) - 1); ++ ++ ctx->pad_color[0] = RGB_TO_Y_BT709(ctx->pad_rgba[0], ctx->pad_rgba[1], ctx->pad_rgba[2]); ++ ctx->pad_color[1] = RGB_TO_U_BT709(ctx->pad_rgba[0], ctx->pad_rgba[1], ctx->pad_rgba[2], 0); ++ ctx->pad_color[2] = RGB_TO_V_BT709(ctx->pad_rgba[0], ctx->pad_rgba[1], ctx->pad_rgba[2], 0); ++ ctx->pad_color[3] = ctx->pad_rgba[3]; ++ } ++ + +- CUdeviceptr d_dst = (CUdeviceptr)out->data[plane]; +- CUdeviceptr d_src = (CUdeviceptr)in->data[plane]; ++ outlink->w = ctx->w; ++ outlink->h = ctx->h; + +- CUfunction cuda_func; + +- if (cur_comp->step == 1 && cur_comp->depth == 8) +- cuda_func = s->cu_func_uchar; +- else if(cur_comp->step == 2 && cur_comp->depth == 8) +- cuda_func = s->cu_func_uchar2; +- else +- return AVERROR_BUG; ++ // prepare output buffer ++ { ++ AVHWFramesContext *out_ctx; ++ AVBufferRef *out_ref = av_hwframe_ctx_alloc(ctx->hw_device_ctx); ++ if (!out_ref) ++ return AVERROR(ENOMEM); + +- void *kernel_args[] = { +- &d_dst, &dst_linesize, &dst_w, &dst_h, +- &d_src, &src_linesize, &src_w, &src_h, +- &x_plane_offset, &y_plane_offset, &s->parsed_color[plane] +- }; ++ out_ctx = (AVHWFramesContext*)out_ref->data; ++ out_ctx->format = AV_PIX_FMT_CUDA; ++ out_ctx->sw_format = ctx->sw_format; ++ out_ctx->width = FFALIGN(ctx->w, 32); ++ out_ctx->height = FFALIGN(ctx->h, 32); ++ ++ ret = av_hwframe_ctx_init(out_ref); ++ if (ret < 0) ++ goto output_buffer_fail; + +- unsigned int grid_x = DIV_UP(dst_w, BLOCK_X); +- unsigned int grid_y = DIV_UP(dst_h, BLOCK_Y); ++ av_frame_unref(ctx->frame); ++ ret = av_hwframe_get_buffer(out_ref, ctx->frame, 0); ++ if (ret < 0) ++ goto output_buffer_fail; ++ ++ ctx->frame->width = ctx->w; ++ ctx->frame->height = ctx->h; + +- ret = CHECK_CU(cu->cuLaunchKernel(cuda_func, grid_x, grid_y, 1, +- BLOCK_X, BLOCK_Y, 1, +- 0, s->hwctx->stream, kernel_args, NULL)); ++ ctx->frames_ctx = out_ref; + + if (ret < 0) { +- av_log(ctx, AV_LOG_ERROR, "Failed to launch kernel for plane %d\n", plane); ++output_buffer_fail: ++ av_buffer_unref(&out_ref); + return ret; + } ++ ++ outl->hw_frames_ctx = av_buffer_ref(inl->hw_frames_ctx); ++ if (!outl->hw_frames_ctx) { ++ return AVERROR(ENOMEM); ++ } + } + + return 0; + } + +-static int cuda_pad_filter_frame(AVFilterLink *inlink, AVFrame *in) +-{ +- AVFilterContext *ctx = inlink->dst; +- CUDAPadContext *s = ctx->priv; +- AVFilterLink *outlink = ctx->outputs[0]; +- +- FilterLink *outl = ff_filter_link(outlink); +- +- AVHWFramesContext *out_frames_ctx = (AVHWFramesContext *)outl->hw_frames_ctx->data; +- AVCUDADeviceContext *device_hwctx = out_frames_ctx->device_ctx->hwctx; ++static int pad_cuda_call_kernel( ++ PadCUDAContext *ctx, ++ uint8_t *main_data, int main_linesize, ++ int main_width, int main_height, ++ uint8_t *overlay_data, int overlay_linesize, ++ int overlay_width, int overlay_height, ++ int offset_x, int offset_y, ++ uint8_t fill_color1, uint8_t fill_color2) { ++ ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ ++ void* kernel_args[] = { ++ &main_data, &main_linesize, ++ &overlay_data, &overlay_linesize, ++ &overlay_width, &overlay_height, ++ &offset_x, &offset_y, ++ &fill_color1, &fill_color2 ++ }; ++ ++ return CHECK_CU(cu->cuLaunchKernel( ++ ctx->cu_func, ++ DIV_UP(main_width, BLOCK_X), DIV_UP(main_height, BLOCK_Y), 1, ++ BLOCK_X, BLOCK_Y, 1, ++ 0, ctx->cu_stream, kernel_args, NULL)); ++} + ++static int pad_cuda_fill_buffers(AVFilterContext *avctx, AVFrame *out, AVFrame *in) ++{ ++ PadCUDAContext *ctx = avctx->priv; ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext cuda_ctx = ctx->hwctx->cuda_ctx; ++ CUcontext dummy; + int ret; ++ uint8_t color_y = ctx->pad_color[0]; ++ uint8_t color_u = ctx->pad_color[1]; ++ uint8_t color_v = ctx->pad_color[2]; ++ uint8_t color_a = ctx->pad_color[3]; + +- if (s->eval_mode == EVAL_MODE_FRAME) { +- s->in_w = in->width; +- s->in_h = in->height; +- s->aspect = in->sample_aspect_ratio; ++ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); ++ if (ret < 0) ++ return ret; + +- ret = eval_expr(ctx); +- if (ret < 0) { +- av_frame_free(&in); +- return ret; +- } ++ // overlay first plane ++ pad_cuda_call_kernel(ctx, ++ out->data[0], out->linesize[0], ++ out->width, out->height, ++ in->data[0], in->linesize[0], ++ in->width, in->height, ++ ctx->x, ctx->y, ++ color_y, color_y); ++ ++ // overlay color offset planes depending on the pixel format ++ switch(ctx->sw_format) { ++ case AV_PIX_FMT_NV12: ++ pad_cuda_call_kernel(ctx, ++ out->data[1], out->linesize[1], ++ out->width, out->height / 2, ++ in->data[1], in->linesize[1], ++ in->width, in->height / 2, ++ ctx->x, ctx->y / 2, ++ color_u, color_v); ++ break; ++ ++ case AV_PIX_FMT_YUV420P: ++ case AV_PIX_FMT_YUVA420P: ++ pad_cuda_call_kernel(ctx, ++ out->data[1], out->linesize[1], ++ out->width / 2, out->height / 2, ++ in->data[1], in->linesize[1], ++ in->width / 2, in->height / 2, ++ ctx->x / 2, ctx->y / 2, ++ color_u, color_u); ++ pad_cuda_call_kernel(ctx, ++ out->data[2], out->linesize[2], ++ out->width / 2, out->height / 2, ++ in->data[2], in->linesize[2], ++ in->width / 2, in->height / 2, ++ ctx->x / 2, ctx->y / 2, ++ color_v, color_v); ++ break; ++ ++ case AV_PIX_FMT_YUV444P: ++ case AV_PIX_FMT_YUVA444P: ++ pad_cuda_call_kernel(ctx, ++ out->data[1], out->linesize[1], ++ out->width, out->height, ++ in->data[1], in->linesize[1], ++ in->width, in->height, ++ ctx->x, ctx->y, ++ color_u, color_u); ++ pad_cuda_call_kernel(ctx, ++ out->data[2], out->linesize[2], ++ out->width, out->height, ++ in->data[2], in->linesize[2], ++ in->width, in->height, ++ ctx->x, ctx->y, ++ color_v, color_v); ++ break; ++ ++ default: ++ av_log(ctx, AV_LOG_ERROR, "Passed unsupported overlay pixel format\n"); ++ av_frame_free(&out); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return AVERROR_BUG; + } + +- +- if (s->x == 0 && s->y == 0 && +- s->w == in->width && s->h == in->height) { +- av_log(ctx, AV_LOG_DEBUG, "No border. Passing the frame unmodified.\n"); +- s->last_out_w = s->w; +- s->last_out_h = s->h; +- return ff_filter_frame(outlink, in); ++ if (in->data[3]) { ++ pad_cuda_call_kernel(ctx, ++ out->data[3], out->linesize[3], ++ out->width, out->height, ++ in->data[3], in->linesize[3], ++ in->width, in->height, ++ ctx->x, ctx->y, ++ color_a, color_a); + } ++ ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return 0; ++} + ++static int pad_cuda_filter_frame(AVFilterLink *link, AVFrame *in) ++{ ++ AVFilterContext *avctx = link->dst; ++ PadCUDAContext *ctx = avctx->priv; ++ AVFilterLink *outlink = avctx->outputs[0]; ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; + +- if (s->w != s->last_out_w || s->h != s->last_out_h) { ++ AVFrame *out = NULL; ++ CUcontext dummy; ++ int ret = 0; + +- av_buffer_unref(&s->frames_ctx); ++ out = av_frame_alloc(); ++ if (!out) { ++ ret = AVERROR(ENOMEM); ++ goto fail; ++ } + +- ret = cuda_pad_alloc_out_frames_ctx(ctx, &s->frames_ctx, s->w, s->h); ++ ret = CHECK_CU(cu->cuCtxPushCurrent(ctx->hwctx->cuda_ctx)); ++ if (ret < 0) ++ goto fail; ++ ++ // fill and prepare the output frame ++ { ++ ret = pad_cuda_fill_buffers(avctx, ctx->frame, in); + if (ret < 0) + return ret; + +- av_buffer_unref(&outl->hw_frames_ctx); +- outl->hw_frames_ctx = av_buffer_ref(s->frames_ctx); +- if (!outl->hw_frames_ctx) { +- av_frame_free(&in); +- av_log(ctx, AV_LOG_ERROR, "Failed to allocate output frame context.\n"); +- return AVERROR(ENOMEM); +- } +- outlink->w = s->w; +- outlink->h = s->h; +- +- s->last_out_w = s->w; +- s->last_out_h = s->h; +- } +- +- AVFrame *out = av_frame_alloc(); +- if (!out) { +- av_frame_free(&in); +- av_log(ctx, AV_LOG_ERROR, "Failed to allocate output AVFrame.\n"); +- return AVERROR(ENOMEM); +- } +- ret = av_hwframe_get_buffer(outl->hw_frames_ctx, out, 0); +- if (ret < 0) { +- av_log(ctx, AV_LOG_ERROR, "Unable to get output buffer: %s\n", +- av_err2str(ret)); +- av_frame_free(&out); +- av_frame_free(&in); +- return ret; +- } +- +- CUcontext dummy; +- ret = CHECK_CU(device_hwctx->internal->cuda_dl->cuCtxPushCurrent( +- device_hwctx->cuda_ctx)); +- if (ret < 0) { +- av_frame_free(&out); +- av_frame_free(&in); +- return ret; +- } ++ ret = av_hwframe_get_buffer(ctx->frame->hw_frames_ctx, ctx->tmp_frame, 0); ++ if (ret < 0) ++ return ret; + +- ret = cuda_pad_pad(ctx, out, in); ++ av_frame_move_ref(out, ctx->frame); ++ av_frame_move_ref(ctx->frame, ctx->tmp_frame); + +- CHECK_CU(device_hwctx->internal->cuda_dl->cuCtxPopCurrent(&dummy)); ++ ctx->frame->width = outlink->w; ++ ctx->frame->height = outlink->h; + +- if (ret < 0) { +- av_frame_free(&out); +- av_frame_free(&in); +- return ret; ++ ret = av_frame_copy_props(out, in); ++ if (ret < 0) ++ return ret; + } + +- av_frame_copy_props(out, in); +- out->width = s->w; +- out->height = s->h; +- ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ if (ret < 0) ++ goto fail; + + av_reduce(&out->sample_aspect_ratio.num, &out->sample_aspect_ratio.den, +- (int64_t)in->sample_aspect_ratio.num * out->height * in->width, +- (int64_t)in->sample_aspect_ratio.den * out->width * in->height, ++ (int64_t)in->sample_aspect_ratio.num * outlink->h * link->w, ++ (int64_t)in->sample_aspect_ratio.den * outlink->w * link->h, + INT_MAX); + + av_frame_free(&in); + return ff_filter_frame(outlink, out); ++fail: ++ av_frame_free(&in); ++ av_frame_free(&out); ++ return ret; + } + +-#define OFFSET(x) offsetof(CUDAPadContext, x) +-#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM) +- +-static const AVOption cuda_pad_options[] = { +- { "width", "set the pad area width expression", OFFSET(w_expr), AV_OPT_TYPE_STRING, {.str = "iw"}, 0, 0, FLAGS }, +- { "w", "set the pad area width expression", OFFSET(w_expr), AV_OPT_TYPE_STRING, {.str = "iw"}, 0, 0, FLAGS }, +- { "height", "set the pad area height expression", OFFSET(h_expr), AV_OPT_TYPE_STRING, {.str = "ih"}, 0, 0, FLAGS }, +- { "h", "set the pad area height expression", OFFSET(h_expr), AV_OPT_TYPE_STRING, {.str = "ih"}, 0, 0, FLAGS }, +- { "x", "set the x offset expression for the input image position", OFFSET(x_expr), AV_OPT_TYPE_STRING, {.str = "0"}, 0, 0, FLAGS }, +- { "y", "set the y offset expression for the input image position", OFFSET(y_expr), AV_OPT_TYPE_STRING, {.str = "0"}, 0, 0, FLAGS }, +- { "color", "set the color of the padded area border", OFFSET(rgba_color), AV_OPT_TYPE_COLOR, {.str = "black"}, .flags = FLAGS }, +- { "eval", "specify when to evaluate expressions", OFFSET(eval_mode), AV_OPT_TYPE_INT, {.i64 = EVAL_MODE_INIT}, 0, EVAL_MODE_NB-1, FLAGS, .unit = "eval" }, +- { "init", "eval expressions once during initialization", 0, AV_OPT_TYPE_CONST, {.i64=EVAL_MODE_INIT}, .flags = FLAGS, .unit = "eval" }, +- { "frame", "eval expressions during initialization and per-frame", 0, AV_OPT_TYPE_CONST, {.i64=EVAL_MODE_FRAME}, .flags = FLAGS, .unit = "eval" }, +- { "aspect", "pad to fit an aspect instead of a resolution", OFFSET(aspect), AV_OPT_TYPE_RATIONAL, {.dbl = 0}, 0, DBL_MAX, FLAGS }, ++ ++ ++#define OFFSET(x) offsetof(PadCUDAContext, x) ++#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM|AV_OPT_FLAG_VIDEO_PARAM) ++ ++static const AVOption pad_cuda_options[] = { ++ { "width", "set the pad area width", OFFSET(w_expr), AV_OPT_TYPE_STRING, {.str = "iw"}, 0, 0, FLAGS }, ++ { "w", "set the pad area width", OFFSET(w_expr), AV_OPT_TYPE_STRING, {.str = "iw"}, 0, 0, FLAGS }, ++ { "height", "set the pad area height", OFFSET(h_expr), AV_OPT_TYPE_STRING, {.str = "ih"}, 0, 0, FLAGS }, ++ { "h", "set the pad area height", OFFSET(h_expr), AV_OPT_TYPE_STRING, {.str = "ih"}, 0, 0, FLAGS }, ++ { "x", "set the x offset for the input image position", OFFSET(x_expr), AV_OPT_TYPE_STRING, {.str = "0"}, 0, INT16_MAX, FLAGS }, ++ { "y", "set the y offset for the input image position", OFFSET(y_expr), AV_OPT_TYPE_STRING, {.str = "0"}, 0, INT16_MAX, FLAGS }, ++ { "color", "set the color of the padded area border", OFFSET(pad_rgba), AV_OPT_TYPE_COLOR, { .str = "black" }, 0, 0, FLAGS }, ++ { "aspect", "pad to fit an aspect instead of a resolution", OFFSET(aspect), AV_OPT_TYPE_RATIONAL, {.dbl = 0}, 0, INT16_MAX, FLAGS }, + { NULL } + }; + +-static const AVClass cuda_pad_class = { +- .class_name = "pad_cuda", +- .item_name = av_default_item_name, +- .option = cuda_pad_options, +- .version = LIBAVUTIL_VERSION_INT, +-}; ++AVFILTER_DEFINE_CLASS(pad_cuda); + +-static const AVFilterPad cuda_pad_inputs[] = {{ +- .name = "default", +- .type = AVMEDIA_TYPE_VIDEO, +- .filter_frame = cuda_pad_filter_frame +-}}; ++static const AVFilterPad pad_cuda_inputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .filter_frame = pad_cuda_filter_frame, ++ }, ++}; + +-static const AVFilterPad cuda_pad_outputs[] = {{ ++static const AVFilterPad pad_cuda_outputs[] = { ++ { + .name = "default", + .type = AVMEDIA_TYPE_VIDEO, +- .config_props = cuda_pad_config_props, +-}}; ++ .config_props = pad_cuda_output_config_props, ++ }, ++}; + + const FFFilter ff_vf_pad_cuda = { +- .p.name = "pad_cuda", +- .p.description = NULL_IF_CONFIG_SMALL("CUDA-based GPU padding filter"), +- .init = cuda_pad_init, +- .uninit = cuda_pad_uninit, +- +- .p.priv_class = &cuda_pad_class, +- +- FILTER_INPUTS(cuda_pad_inputs), +- FILTER_OUTPUTS(cuda_pad_outputs), +- ++ .p.name = "pad_cuda", ++ .p.description = NULL_IF_CONFIG_SMALL("Pad CUDA accelerated video using solid color"), ++ .priv_size = sizeof(PadCUDAContext), ++ .p.priv_class = &pad_cuda_class, ++ .init = pad_cuda_init, ++ .uninit = pad_cuda_uninit, ++ FILTER_INPUTS(pad_cuda_inputs), ++ FILTER_OUTPUTS(pad_cuda_outputs), + FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), +- +- .priv_size = sizeof(CUDAPadContext), + .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, + }; +diff --git a/libavfilter/vf_pad_cuda.cu b/libavfilter/vf_pad_cuda.cu +index f1323d122f..62f7988de5 100644 +--- a/libavfilter/vf_pad_cuda.cu ++++ b/libavfilter/vf_pad_cuda.cu +@@ -1,61 +1,37 @@ +-/* +- * This file is part of FFmpeg. +- * +- * FFmpeg is free software; you can redistribute it and/or +- * modify it under the terms of the GNU Lesser General Public +- * License as published by the Free Software Foundation; either +- * version 2.1 of the License, or (at your option) any later version. +- * +- * FFmpeg is distributed in the hope that it will be useful, +- * but WITHOUT ANY WARRANTY; without even the implied warranty of +- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +- * Lesser General Public License for more details. +- * +- * You should have received a copy of the GNU Lesser General Public +- * License along with FFmpeg; if not, write to the Free Software +- * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +- */ ++extern "C" { + +-template +-__device__ void pad_impl(T* dst, int dst_pitch, int dst_w, int dst_h, +- const T* src, int src_pitch, int src_w, int src_h, +- int roi_x, int roi_y, T fill_val) ++__global__ void Pad_Cuda( ++ unsigned char* main, int main_linesize, ++ unsigned char* overlay, int overlay_linesize, ++ int overlay_w, int overlay_h, ++ int offset_x, int offset_y, ++ unsigned char fill_color1, unsigned char fill_color2) + { +- const int x = blockIdx.x * blockDim.x + threadIdx.x; +- const int y = blockIdx.y * blockDim.y + threadIdx.y; +- +- if (x >= dst_w || y >= dst_h) { +- return; ++ // ++ // We pass two colors to handle NV12 plane with interleaved UV. ++ // So fill_color1 would correspond to U and fill_color2 to V. ++ // For other formats both fill_colors should be the same. ++ // ++ unsigned char color[2] = {fill_color1, fill_color2}; ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ ++ unsigned char *m = &main[x + y*main_linesize]; ++ ++ if (x < offset_x || ++ y < offset_y || ++ x >= overlay_w + offset_x || ++ y >= overlay_h + offset_y) ++ { ++ // Picking colors[0] for even value of x; and colors[1] for odd value of x; ++ *m = color[x & 1]; + } +- +- if (x >= roi_x && x < (roi_x + src_w) && y >= roi_y && y < (roi_y + src_h)) { +- const int src_x = x - roi_x; +- const int src_y = y - roi_y; +- dst[y * dst_pitch + x] = src[src_y * src_pitch + src_x]; +- } else { +- dst[y * dst_pitch + x] = fill_val; ++ else ++ { ++ int overlay_x = x - offset_x; ++ int overlay_y = y - offset_y; ++ *m = overlay[overlay_x + overlay_y*overlay_linesize]; + } + } + +- +-extern "C" { +- +-__global__ void pad_uchar(unsigned char* dst, int dst_pitch, int dst_w, int dst_h, +- const unsigned char* src, int src_pitch, int src_w, int src_h, +- int roi_x, int roi_y, unsigned char fill_val) +-{ +- pad_impl(dst, dst_pitch, dst_w, dst_h, +- src, src_pitch, src_w, src_h, +- roi_x, roi_y, fill_val); +-} +- +-__global__ void pad_uchar2(uchar2* dst, int dst_pitch, int dst_w, int dst_h, +- const uchar2* src, int src_pitch, int src_w, int src_h, +- int roi_x, int roi_y, uchar2 fill_val) +-{ +- pad_impl(dst, dst_pitch, dst_w, dst_h, +- src, src_pitch, src_w, src_h, +- roi_x, roi_y, fill_val); +-} +- + } +diff --git a/libavfilter/vf_scale_cuda.c b/libavfilter/vf_scale_cuda.c +index 5fd757161b..56c4d6350d 100644 +--- a/libavfilter/vf_scale_cuda.c ++++ b/libavfilter/vf_scale_cuda.c +@@ -487,6 +487,11 @@ static int scalecuda_resize(AVFilterContext *ctx, + + for (i = 0; i < s->in_planes; i++) { + CUDA_TEXTURE_DESC tex_desc = { ++ .addressMode = { ++ CU_TR_ADDRESS_MODE_CLAMP, ++ CU_TR_ADDRESS_MODE_CLAMP, ++ CU_TR_ADDRESS_MODE_CLAMP, ++ }, + .filterMode = s->interp_use_linear ? + CU_TR_FILTER_MODE_LINEAR : + CU_TR_FILTER_MODE_POINT, +diff --git a/libavfilter/vf_scale_cuda.cu b/libavfilter/vf_scale_cuda.cu +index d674c0885a..eec20db1a1 100644 +--- a/libavfilter/vf_scale_cuda.cu ++++ b/libavfilter/vf_scale_cuda.cu +@@ -1043,24 +1043,29 @@ struct Convert_rgb0_rgba + + typedef float4 (*coeffs_function_t)(float, float); + +-__device__ static inline float4 lanczos_coeffs(float x, float param) ++__device__ static inline float lanczos_sinc(float x) + { + const float pi = 3.141592654f; + ++ if (fabsf(x) < 1.0e-5f) ++ return 1.0f; ++ ++ x *= pi; ++ return __sinf(x) / x; ++} ++ ++__device__ static inline float lanczos_kernel(float x) ++{ ++ return lanczos_sinc(x) * lanczos_sinc(x / 2.0f); ++} ++ ++__device__ static inline float4 lanczos_coeffs(float x, float param) ++{ + float4 res = make_float4( +- pi * (x + 1), +- pi * x, +- pi * (x - 1), +- pi * (x - 2)); +- +- res.x = res.x == 0.0f ? 1.0f : +- __sinf(res.x) * __sinf(res.x / 2.0f) / (res.x * res.x / 2.0f); +- res.y = res.y == 0.0f ? 1.0f : +- __sinf(res.y) * __sinf(res.y / 2.0f) / (res.y * res.y / 2.0f); +- res.z = res.z == 0.0f ? 1.0f : +- __sinf(res.z) * __sinf(res.z / 2.0f) / (res.z * res.z / 2.0f); +- res.w = res.w == 0.0f ? 1.0f : +- __sinf(res.w) * __sinf(res.w / 2.0f) / (res.w * res.w / 2.0f); ++ lanczos_kernel(x + 1), ++ lanczos_kernel(x), ++ lanczos_kernel(x - 1), ++ lanczos_kernel(x - 2)); + + return res / (res.x + res.y + res.z + res.w); + } +@@ -1089,6 +1094,71 @@ __device__ static inline V apply_coeffs(float4 coeffs, V c0, V c1, V c2, V c3) + return res; + } + ++__device__ static inline float clamp_texture_coord(float v, int size) ++{ ++ return fminf(fmaxf(v, 0.0f), (float)(size - 1)); ++} ++ ++__device__ static inline uchar clip_float_to_uchar(float v) ++{ ++ return (uchar)fminf(fmaxf(v, 0.0f), 255.0f); ++} ++ ++__device__ static inline ushort clip_float_to_ushort(float v) ++{ ++ return (ushort)fminf(fmaxf(v, 0.0f), 65535.0f); ++} ++ ++template ++__device__ static inline T from_scaled_floatN(const V &v) ++{ ++ return from_floatN(v); ++} ++ ++template<> ++__device__ inline uchar from_scaled_floatN(const float &v) ++{ ++ return clip_float_to_uchar(v); ++} ++ ++template<> ++__device__ inline uchar2 from_scaled_floatN(const float2 &v) ++{ ++ return make_uchar2(clip_float_to_uchar(v.x), ++ clip_float_to_uchar(v.y)); ++} ++ ++template<> ++__device__ inline uchar4 from_scaled_floatN(const float4 &v) ++{ ++ return make_uchar4(clip_float_to_uchar(v.x), ++ clip_float_to_uchar(v.y), ++ clip_float_to_uchar(v.z), ++ clip_float_to_uchar(v.w)); ++} ++ ++template<> ++__device__ inline ushort from_scaled_floatN(const float &v) ++{ ++ return clip_float_to_ushort(v); ++} ++ ++template<> ++__device__ inline ushort2 from_scaled_floatN(const float2 &v) ++{ ++ return make_ushort2(clip_float_to_ushort(v.x), ++ clip_float_to_ushort(v.y)); ++} ++ ++template<> ++__device__ inline ushort4 from_scaled_floatN(const float4 &v) ++{ ++ return make_ushort4(clip_float_to_ushort(v.x), ++ clip_float_to_ushort(v.y), ++ clip_float_to_ushort(v.z), ++ clip_float_to_ushort(v.w)); ++} ++ + template + __device__ static inline T Subsample_Nearest(cudaTextureObject_t tex, + int xo, int yo, +@@ -1159,9 +1229,11 @@ __device__ static inline T Subsample_Bicubic(cudaTextureObject_t tex, + float4 coeffsX = coeffs_function(fx, param); + float4 coeffsY = coeffs_function(fy, param); + +-#define PIX(x, y) tex2D(tex, (x), (y)) ++#define PIX(x, y) tex2D(tex, \ ++ clamp_texture_coord((x), src_width), \ ++ clamp_texture_coord((y), src_height)) + +- return from_floatN( ++ return from_scaled_floatN( + apply_coeffs(coeffsY, + apply_coeffs(coeffsX, PIX(px - 1, py - 1), PIX(px, py - 1), PIX(px + 1, py - 1), PIX(px + 2, py - 1)), + apply_coeffs(coeffsX, PIX(px - 1, py ), PIX(px, py ), PIX(px + 1, py ), PIX(px + 2, py )), +diff --git a/libavfilter/vf_transition_cuda.c b/libavfilter/vf_transition_cuda.c +new file mode 100644 +index 0000000000..0c055a71e0 +--- /dev/null ++++ b/libavfilter/vf_transition_cuda.c +@@ -0,0 +1,621 @@ ++/* ++ * Copyright (c) 2020 Yaroslav Pogrebnyak ++ * ++ * This file is part of FFmpeg. ++ * ++ * FFmpeg is free software; you can redistribute it and/or ++ * modify it under the terms of the GNU Lesser General Public ++ * License as published by the Free Software Foundation; either ++ * version 2.1 of the License, or (at your option) any later version. ++ * ++ * FFmpeg is distributed in the hope that it will be useful, ++ * but WITHOUT ANY WARRANTY; without even the implied warranty of ++ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ * Lesser General Public License for more details. ++ * ++ * You should have received a copy of the GNU Lesser General Public ++ * License along with FFmpeg; if not, write to the Free Software ++ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA ++ */ ++ ++/** ++ * @file ++ * Overlay one video on top of another or transition between them using cuda hardware acceleration ++ */ ++ ++#include "libavutil/log.h" ++#include "libavutil/mem.h" ++#include "libavutil/opt.h" ++#include "libavutil/pixdesc.h" ++#include "libavutil/hwcontext.h" ++#include "libavutil/hwcontext_cuda_internal.h" ++#include "libavutil/cuda_check.h" ++#include "libavutil/eval.h" ++ ++#include "avfilter.h" ++#include "framesync.h" ++#include "avfilter_internal.h" ++ ++#include "cuda/load_helper.h" ++ ++#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, ctx->hwctx->internal->cuda_dl, x) ++#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) ) ++ ++#define BLOCK_X 32 ++#define BLOCK_Y 16 ++ ++#define MAIN 0 ++#define OVERLAY 1 ++ ++static const enum AVPixelFormat supported_main_formats[] = { ++ AV_PIX_FMT_NV12, ++ AV_PIX_FMT_YUV420P, ++ AV_PIX_FMT_NONE, ++}; ++ ++static const enum AVPixelFormat supported_overlay_formats[] = { ++ AV_PIX_FMT_NV12, ++ AV_PIX_FMT_YUV420P, ++ AV_PIX_FMT_YUVA420P, ++ AV_PIX_FMT_NONE, ++}; ++ ++enum var_name { ++ VAR_MAIN_W, VAR_MW, ++ VAR_MAIN_H, VAR_MH, ++ VAR_OVERLAY_W, VAR_OW, ++ VAR_OVERLAY_H, VAR_OH, ++ VAR_ALPHA, ++ VAR_N, ++ VAR_POS, ++ VAR_T, ++ VAR_VARS_NB ++}; ++ ++enum EvalMode { ++ EVAL_MODE_INIT, ++ EVAL_MODE_FRAME, ++ EVAL_MODE_NB ++}; ++ ++enum TransitionMode { ++ TRANSITION_MODE_FADE, ++ TRANSITION_MODE_WIPE_LEFT, ++ TRANSITION_MODE_WIPE_RIGHT, ++ TRANSITION_MODE_WIPE_DOWN, ++ TRANSITION_MODE_WIPE_UP, ++ TRANSITION_MODE_NB ++}; ++ ++static const char *const var_names[] = { ++ "main_w", "W", ///< width of the main video ++ "main_h", "H", ///< height of the main video ++ "overlay_w", "w", ///< width of the overlay video ++ "overlay_h", "h", ///< height of the overlay video ++ "alpha", ++ "n", ///< number of frame ++ "pos", ///< position in the file ++ "t", ///< timestamp expressed in seconds ++ NULL ++}; ++ ++/** ++ * TransitionCUDAContext ++ */ ++typedef struct TransitionCUDAContext { ++ const AVClass *class; ++ ++ enum AVPixelFormat in_format_overlay; ++ enum AVPixelFormat in_format_main; ++ ++ AVBufferRef *hw_device_ctx; ++ AVCUDADeviceContext *hwctx; ++ ++ CUcontext cu_ctx; ++ CUmodule cu_module; ++ CUfunction cu_func; ++ CUstream cu_stream; ++ ++ FFFrameSync fs; ++ ++ int eval_mode; ++ int transition_mode; ++ float alpha_coef; ++ ++ double var_values[VAR_VARS_NB]; ++ char *alpha_expr; ++ ++ AVExpr *alpha_pexpr; ++} TransitionCUDAContext; ++ ++/** ++ * Helper to find out if provided format is supported by filter ++ */ ++static int format_is_supported(const enum AVPixelFormat formats[], enum AVPixelFormat fmt) ++{ ++ for (int i = 0; formats[i] != AV_PIX_FMT_NONE; i++) ++ if (formats[i] == fmt) ++ return 1; ++ return 0; ++} ++ ++static void eval_expr(AVFilterContext *ctx) ++{ ++ TransitionCUDAContext *s = ctx->priv; ++ ++ s->var_values[VAR_ALPHA] = av_expr_eval(s->alpha_pexpr, s->var_values, NULL); ++ s->alpha_coef = (float)s->var_values[VAR_ALPHA]; ++ ++ if (isnan(s->alpha_coef) || s->alpha_coef > 1.f) ++ s->alpha_coef = 1.f; ++ ++ if (s->alpha_coef < 0.f) ++ s->alpha_coef = 0.f; ++} ++ ++static int set_expr(AVExpr **pexpr, const char *expr, const char *option, void *log_ctx) ++{ ++ int ret; ++ AVExpr *old = NULL; ++ ++ if (*pexpr) ++ old = *pexpr; ++ ret = av_expr_parse(pexpr, expr, var_names, ++ NULL, NULL, NULL, NULL, 0, log_ctx); ++ if (ret < 0) { ++ av_log(log_ctx, AV_LOG_ERROR, ++ "Error when evaluating the expression '%s' for %s\n", ++ expr, option); ++ *pexpr = old; ++ return ret; ++ } ++ ++ av_expr_free(old); ++ return 0; ++} ++ ++/** ++ * Helper checks if we can process main and overlay pixel formats ++ */ ++static int formats_match(const enum AVPixelFormat format_main, const enum AVPixelFormat format_overlay) { ++ switch(format_main) { ++ case AV_PIX_FMT_NV12: ++ return format_overlay == AV_PIX_FMT_NV12; ++ case AV_PIX_FMT_YUV420P: ++ return format_overlay == AV_PIX_FMT_YUV420P || ++ format_overlay == AV_PIX_FMT_YUVA420P; ++ default: ++ return 0; ++ } ++} ++ ++/** ++ * Call transition kernell for a plane ++ */ ++static int transition_cuda_call_kernel( ++ TransitionCUDAContext *ctx, ++ uint8_t* main_data, int main_linesize, ++ int main_width, int main_height, ++ uint8_t* overlay_data, int overlay_linesize, ++ int overlay_width, int overlay_height, ++ uint8_t* alpha_data, int alpha_linesize, ++ int alpha_adj_x, int alpha_adj_y, ++ float alpha_coef, ++ int transition_mode) { ++ ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ ++ void* kernel_args[] = { ++ &main_data, &main_linesize, ++ &overlay_data, &overlay_linesize, ++ &overlay_width, &overlay_height, ++ &alpha_data, &alpha_linesize, ++ &alpha_adj_x, &alpha_adj_y, ++ &alpha_coef, ++ &transition_mode ++ }; ++ ++ return CHECK_CU(cu->cuLaunchKernel( ++ ctx->cu_func, ++ DIV_UP(main_width, BLOCK_X), DIV_UP(main_height, BLOCK_Y), 1, ++ BLOCK_X, BLOCK_Y, 1, ++ 0, ctx->cu_stream, kernel_args, NULL)); ++} ++ ++/** ++ * Perform blend overlay picture over main picture ++ */ ++static int transition_cuda_blend(FFFrameSync *fs) ++{ ++ int ret; ++ ++ AVFilterContext *avctx = fs->parent; ++ TransitionCUDAContext *ctx = avctx->priv; ++ AVFilterLink *outlink = avctx->outputs[0]; ++ AVFilterLink *inlink = avctx->inputs[0]; ++ FilterLink *inl = ff_filter_link(inlink); ++ ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CUcontext dummy, cuda_ctx = ctx->hwctx->cuda_ctx; ++ ++ AVFrame *input_main, *input_overlay; ++ ++ ctx->cu_ctx = cuda_ctx; ++ ++ // read main and overlay frames from inputs ++ ret = ff_framesync_dualinput_get(fs, &input_main, &input_overlay); ++ if (ret < 0) ++ return ret; ++ ++ if (!input_main) ++ return AVERROR_BUG; ++ ++ if (!input_overlay) ++ return ff_filter_frame(outlink, input_main); ++ ++ ret = av_frame_make_writable(input_main); ++ if (ret < 0) { ++ av_frame_free(&input_main); ++ return ret; ++ } ++ ++ // push cuda context ++ ++ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); ++ if (ret < 0) { ++ av_frame_free(&input_main); ++ return ret; ++ } ++ ++ if (ctx->eval_mode == EVAL_MODE_FRAME) { ++ ctx->var_values[VAR_N] = inl->frame_count_out; ++ ctx->var_values[VAR_T] = input_main->pts == AV_NOPTS_VALUE ? ++ NAN : input_main->pts * av_q2d(inlink->time_base); ++ // FFmpeg 8 no longer exposes the packet byte position on AVFrame. ++ ctx->var_values[VAR_POS] = NAN; ++ ctx->var_values[VAR_OVERLAY_W] = ctx->var_values[VAR_OW] = input_overlay->width; ++ ctx->var_values[VAR_OVERLAY_H] = ctx->var_values[VAR_OH] = input_overlay->height; ++ ctx->var_values[VAR_MAIN_W ] = ctx->var_values[VAR_MW] = input_main->width; ++ ctx->var_values[VAR_MAIN_H ] = ctx->var_values[VAR_MH] = input_main->height; ++ ++ eval_expr(avctx); ++ ++ av_log(avctx, AV_LOG_DEBUG, "n:%f t:%f pos:%f alpha: %f\n", ++ ctx->var_values[VAR_N], ctx->var_values[VAR_T], ctx->var_values[VAR_POS], ++ ctx->alpha_coef); ++ } ++ ++ // overlay first plane ++ ++ transition_cuda_call_kernel(ctx, ++ input_main->data[0], input_main->linesize[0], ++ input_main->width, input_main->height, ++ input_overlay->data[0], input_overlay->linesize[0], ++ input_overlay->width, input_overlay->height, ++ input_overlay->data[3], input_overlay->linesize[3], 1, 1, ++ ctx->alpha_coef, ctx->transition_mode); ++ ++ // overlay rest planes depending on pixel format ++ ++ switch(ctx->in_format_overlay) { ++ case AV_PIX_FMT_NV12: ++ transition_cuda_call_kernel(ctx, ++ input_main->data[1], input_main->linesize[1], ++ input_main->width, input_main->height / 2, ++ input_overlay->data[1], input_overlay->linesize[1], ++ input_overlay->width, input_overlay->height / 2, ++ 0, 0, 0, 0, ++ ctx->alpha_coef, ctx->transition_mode); ++ break; ++ case AV_PIX_FMT_YUV420P: ++ case AV_PIX_FMT_YUVA420P: ++ transition_cuda_call_kernel(ctx, ++ input_main->data[1], input_main->linesize[1], ++ input_main->width / 2, input_main->height / 2, ++ input_overlay->data[1], input_overlay->linesize[1], ++ input_overlay->width / 2, input_overlay->height / 2, ++ input_overlay->data[3], input_overlay->linesize[3], 2, 2, ++ ctx->alpha_coef, ctx->transition_mode); ++ ++ transition_cuda_call_kernel(ctx, ++ input_main->data[2], input_main->linesize[2], ++ input_main->width / 2, input_main->height / 2, ++ input_overlay->data[2], input_overlay->linesize[2], ++ input_overlay->width / 2, input_overlay->height / 2, ++ input_overlay->data[3], input_overlay->linesize[3], 2, 2, ++ ctx->alpha_coef, ctx->transition_mode); ++ break; ++ default: ++ av_log(ctx, AV_LOG_ERROR, "Passed unsupported overlay pixel format\n"); ++ av_frame_free(&input_main); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return AVERROR_BUG; ++ } ++ ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ ++ return ff_filter_frame(outlink, input_main); ++} ++ ++static int config_input_overlay(AVFilterLink *inlink) ++{ ++ AVFilterContext *ctx = inlink->dst; ++ TransitionCUDAContext *s = inlink->dst->priv; ++ int ret; ++ ++ ++ /* Finish the configuration by evaluating the expressions ++ now when both inputs are configured. */ ++ s->var_values[VAR_MAIN_W ] = s->var_values[VAR_MW] = ctx->inputs[MAIN ]->w; ++ s->var_values[VAR_MAIN_H ] = s->var_values[VAR_MH] = ctx->inputs[MAIN ]->h; ++ s->var_values[VAR_OVERLAY_W] = s->var_values[VAR_OW] = ctx->inputs[OVERLAY]->w; ++ s->var_values[VAR_OVERLAY_H] = s->var_values[VAR_OH] = ctx->inputs[OVERLAY]->h; ++ s->var_values[VAR_ALPHA] = NAN; ++ s->var_values[VAR_N] = 0; ++ s->var_values[VAR_T] = NAN; ++ s->var_values[VAR_POS] = NAN; ++ ++ if ((ret = set_expr(&s->alpha_pexpr, s->alpha_expr, "alpha", ctx)) < 0) ++ return ret; ++ ++ if (s->eval_mode == EVAL_MODE_INIT) { ++ eval_expr(ctx); ++ av_log(ctx, AV_LOG_VERBOSE, "alpha: %f\n", s->alpha_coef); ++ } ++ ++ return 0; ++} ++ ++/** ++ * Initialize transition_cuda ++ */ ++static av_cold int transition_cuda_init(AVFilterContext *avctx) ++{ ++ TransitionCUDAContext* ctx = avctx->priv; ++ ctx->fs.on_event = &transition_cuda_blend; ++ return 0; ++} ++ ++/** ++ * Uninitialize transition_cuda ++ */ ++static av_cold void transition_cuda_uninit(AVFilterContext *avctx) ++{ ++ TransitionCUDAContext* ctx = avctx->priv; ++ ++ ff_framesync_uninit(&ctx->fs); ++ ++ if (ctx->hwctx && ctx->cu_module) { ++ CUcontext dummy; ++ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; ++ CHECK_CU(cu->cuCtxPushCurrent(ctx->cu_ctx)); ++ CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ } ++ ++ av_expr_free(ctx->alpha_pexpr); ctx->alpha_pexpr = NULL; ++ av_buffer_unref(&ctx->hw_device_ctx); ++ ctx->hwctx = NULL; ++} ++ ++/** ++ * Activate transition_cuda ++ */ ++static int transition_cuda_activate(AVFilterContext *avctx) ++{ ++ TransitionCUDAContext *ctx = avctx->priv; ++ return ff_framesync_activate(&ctx->fs); ++} ++ ++/** ++ * Configure output ++ */ ++static int transition_cuda_config_output(AVFilterLink *outlink) ++{ ++ extern const unsigned char ff_vf_transition_cuda_ptx_data[]; ++ extern const unsigned int ff_vf_transition_cuda_ptx_len; ++ ++ int err; ++ FilterLink *outl = ff_filter_link(outlink); ++ AVFilterContext *avctx = outlink->src; ++ TransitionCUDAContext *ctx = avctx->priv; ++ ++ AVFilterLink *inlink = avctx->inputs[0]; ++ FilterLink *inl = ff_filter_link(inlink); ++ AVHWFramesContext *frames_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; ++ ++ AVFilterLink *inlink_overlay = avctx->inputs[1]; ++ FilterLink *inl_overlay = ff_filter_link(inlink_overlay); ++ AVHWFramesContext *frames_ctx_overlay = (AVHWFramesContext*)inl_overlay->hw_frames_ctx->data; ++ ++ CUcontext dummy, cuda_ctx; ++ CudaFunctions *cu; ++ ++ // check main input formats ++ ++ if (!frames_ctx) { ++ av_log(ctx, AV_LOG_ERROR, "No hw context provided on main input\n"); ++ return AVERROR(EINVAL); ++ } ++ ++ ctx->in_format_main = frames_ctx->sw_format; ++ if (!format_is_supported(supported_main_formats, ctx->in_format_main)) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported main input format: %s\n", ++ av_get_pix_fmt_name(ctx->in_format_main)); ++ return AVERROR(ENOSYS); ++ } ++ ++ // check transition input formats ++ ++ if (!frames_ctx_overlay) { ++ av_log(ctx, AV_LOG_ERROR, "No hw context provided on transition input\n"); ++ return AVERROR(EINVAL); ++ } ++ ++ ctx->in_format_overlay = frames_ctx_overlay->sw_format; ++ if (!format_is_supported(supported_overlay_formats, ctx->in_format_overlay)) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported overlay input format: %s\n", ++ av_get_pix_fmt_name(ctx->in_format_overlay)); ++ return AVERROR(ENOSYS); ++ } ++ ++ // check we can overlay pictures with those pixel formats ++ ++ if (!formats_match(ctx->in_format_main, ctx->in_format_overlay)) { ++ av_log(ctx, AV_LOG_ERROR, "Can't overlay %s on %s \n", ++ av_get_pix_fmt_name(ctx->in_format_overlay), av_get_pix_fmt_name(ctx->in_format_main)); ++ return AVERROR(EINVAL); ++ } ++ ++ // initialize ++ ++ ctx->hw_device_ctx = av_buffer_ref(frames_ctx->device_ref); ++ if (!ctx->hw_device_ctx) ++ return AVERROR(ENOMEM); ++ ctx->hwctx = ((AVHWDeviceContext*)ctx->hw_device_ctx->data)->hwctx; ++ ++ cuda_ctx = ctx->hwctx->cuda_ctx; ++ ctx->fs.time_base = inlink->time_base; ++ ++ ctx->cu_stream = ctx->hwctx->stream; ++ ++ outl->hw_frames_ctx = av_buffer_ref(inl->hw_frames_ctx); ++ if (!outl->hw_frames_ctx) ++ return AVERROR(ENOMEM); ++ ++ // load functions ++ ++ cu = ctx->hwctx->internal->cuda_dl; ++ ++ err = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); ++ if (err < 0) { ++ return err; ++ } ++ ++ err = ff_cuda_load_module(ctx, ctx->hwctx, &ctx->cu_module, ff_vf_transition_cuda_ptx_data, ff_vf_transition_cuda_ptx_len); ++ if (err < 0) { ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return err; ++ } ++ ++ err = CHECK_CU(cu->cuModuleGetFunction(&ctx->cu_func, ctx->cu_module, "Transition_Cuda")); ++ if (err < 0) { ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ return err; ++ } ++ ++ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); ++ ++ // init dual input ++ ++ err = ff_framesync_init_dualinput(&ctx->fs, avctx); ++ if (err < 0) { ++ return err; ++ } ++ ++ return ff_framesync_configure(&ctx->fs); ++} ++ ++static int transition_cuda_process_command(AVFilterContext *avctx, ++ const char *cmd, ++ const char *args, ++ char *res, ++ int res_len, ++ int flags) ++{ ++ TransitionCUDAContext *ctx = avctx->priv; ++ AVExpr *new_expr = NULL; ++ char *new_alpha_expr; ++ int ret; ++ ++ (void)res; ++ (void)res_len; ++ (void)flags; ++ ++ if (!strcmp(cmd, "mode")) ++ return av_opt_set(ctx, "mode", args, 0); ++ ++ if (strcmp(cmd, "alpha")) ++ return AVERROR(ENOSYS); ++ ++ ret = set_expr(&new_expr, args, "alpha", avctx); ++ if (ret < 0) ++ return ret; ++ ++ new_alpha_expr = av_strdup(args); ++ if (!new_alpha_expr) { ++ av_expr_free(new_expr); ++ return AVERROR(ENOMEM); ++ } ++ ++ av_expr_free(ctx->alpha_pexpr); ++ ctx->alpha_pexpr = new_expr; ++ av_freep(&ctx->alpha_expr); ++ ctx->alpha_expr = new_alpha_expr; ++ return 0; ++} ++ ++ ++#define OFFSET(x) offsetof(TransitionCUDAContext, x) ++#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM) ++#define RUNTIME_FLAGS (FLAGS | AV_OPT_FLAG_RUNTIME_PARAM) ++ ++static const AVOption transition_cuda_options[] = { ++ { "alpha", "set the alpha expression of overlay in range [0.0-1.0] (default is 1.0)", OFFSET(alpha_expr), AV_OPT_TYPE_STRING, { .str = "1.0" }, 0, 0, RUNTIME_FLAGS }, ++ { "mode", "set the CUDA transition mode", OFFSET(transition_mode), AV_OPT_TYPE_INT, { .i64 = TRANSITION_MODE_FADE }, 0, TRANSITION_MODE_NB - 1, RUNTIME_FLAGS, "mode" }, ++ { "fade", "crossfade", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_FADE }, .flags = RUNTIME_FLAGS, .unit = "mode" }, ++ { "wipe_left", "wipe from left", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_WIPE_LEFT }, .flags = RUNTIME_FLAGS, .unit = "mode" }, ++ { "wipe_right", "wipe from right", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_WIPE_RIGHT }, .flags = RUNTIME_FLAGS, .unit = "mode" }, ++ { "wipe_down", "wipe from top to bottom", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_WIPE_DOWN }, .flags = RUNTIME_FLAGS, .unit = "mode" }, ++ { "wipe_up", "wipe from bottom to top", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_WIPE_UP }, .flags = RUNTIME_FLAGS, .unit = "mode" }, ++ { "eof_action", "Action to take when encountering EOF from secondary input ", ++ OFFSET(fs.opt_eof_action), AV_OPT_TYPE_INT, { .i64 = EOF_ACTION_REPEAT }, ++ EOF_ACTION_REPEAT, EOF_ACTION_PASS, .flags = FLAGS, "eof_action" }, ++ { "repeat", "Repeat the previous frame.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_REPEAT }, .flags = FLAGS, "eof_action" }, ++ { "endall", "End both streams.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_ENDALL }, .flags = FLAGS, "eof_action" }, ++ { "pass", "Pass through the main input.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_PASS }, .flags = FLAGS, "eof_action" }, ++ { "eval", "specify when to evaluate expressions", OFFSET(eval_mode), AV_OPT_TYPE_INT, { .i64 = EVAL_MODE_FRAME }, 0, EVAL_MODE_NB - 1, FLAGS, "eval" }, ++ { "init", "eval expressions once during initialization", 0, AV_OPT_TYPE_CONST, { .i64=EVAL_MODE_INIT }, .flags = FLAGS, .unit = "eval" }, ++ { "frame", "eval expressions per-frame", 0, AV_OPT_TYPE_CONST, { .i64=EVAL_MODE_FRAME }, .flags = FLAGS, .unit = "eval" }, ++ { "shortest", "force termination when the shortest input terminates", OFFSET(fs.opt_shortest), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS }, ++ { "repeatlast", "repeat overlay of the last overlay frame", OFFSET(fs.opt_repeatlast), AV_OPT_TYPE_BOOL, {.i64=1}, 0, 1, FLAGS }, ++ { NULL }, ++}; ++ ++FRAMESYNC_DEFINE_CLASS(transition_cuda, TransitionCUDAContext, fs); ++ ++static const AVFilterPad transition_cuda_inputs[] = { ++ { ++ .name = "main", ++ .type = AVMEDIA_TYPE_VIDEO, ++ }, ++ { ++ .name = "overlay", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .config_props = config_input_overlay, ++ }, ++}; ++ ++static const AVFilterPad transition_cuda_outputs[] = { ++ { ++ .name = "default", ++ .type = AVMEDIA_TYPE_VIDEO, ++ .config_props = &transition_cuda_config_output, ++ }, ++}; ++ ++const FFFilter ff_vf_transition_cuda = { ++ .p.name = "transition_cuda", ++ .p.description = NULL_IF_CONFIG_SMALL("Transition between videos using CUDA"), ++ .priv_size = sizeof(TransitionCUDAContext), ++ .p.priv_class = &transition_cuda_class, ++ .init = &transition_cuda_init, ++ .uninit = &transition_cuda_uninit, ++ .activate = &transition_cuda_activate, ++ .process_command = &transition_cuda_process_command, ++ FILTER_INPUTS(transition_cuda_inputs), ++ FILTER_OUTPUTS(transition_cuda_outputs), ++ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), ++ .preinit = transition_cuda_framesync_preinit, ++ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, ++}; +diff --git a/libavfilter/vf_transition_cuda.cu b/libavfilter/vf_transition_cuda.cu +new file mode 100644 +index 0000000000..c688e66cca +--- /dev/null ++++ b/libavfilter/vf_transition_cuda.cu +@@ -0,0 +1,56 @@ ++/* ++ * Copyright (c) 2020 Yaroslav Pogrebnyak ++ * ++ * This file is part of FFmpeg. ++ * ++ * FFmpeg is free software; you can redistribute it and/or ++ * modify it under the terms of the GNU Lesser General Public ++ * License as published by the Free Software Foundation; either ++ * version 2.1 of the License, or (at your option) any later version. ++ * ++ * FFmpeg is distributed in the hope that it will be useful, ++ * but WITHOUT ANY WARRANTY; without even the implied warranty of ++ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ * Lesser General Public License for more details. ++ * ++ * You should have received a copy of the GNU Lesser General Public ++ * License along with FFmpeg; if not, write to the Free Software ++ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA ++ */ ++ ++extern "C" { ++ ++__global__ void Transition_Cuda( ++ unsigned char* main, int main_linesize, ++ unsigned char* overlay, int overlay_linesize, ++ int overlay_w, int overlay_h, ++ unsigned char* overlay_alpha, int alpha_linesize, ++ int alpha_adj_x, int alpha_adj_y, ++ float alpha_coef, int transition_mode) ++{ ++ int x = blockIdx.x * blockDim.x + threadIdx.x; ++ int y = blockIdx.y * blockDim.y + threadIdx.y; ++ ++ if (x >= overlay_w || ++ y >= overlay_h) { ++ return; ++ } ++ ++ float alpha = alpha_coef; ++ if (transition_mode == 1) { ++ alpha = (x + 0.5f) / overlay_w <= alpha_coef; ++ } else if (transition_mode == 2) { ++ alpha = (x + 0.5f) / overlay_w >= 1.f - alpha_coef; ++ } else if (transition_mode == 3) { ++ alpha = (y + 0.5f) / overlay_h <= alpha_coef; ++ } else if (transition_mode == 4) { ++ alpha = (y + 0.5f) / overlay_h >= 1.f - alpha_coef; ++ } ++ if (alpha_linesize) { ++ alpha *= overlay_alpha[alpha_adj_x * x + alpha_adj_y * y * alpha_linesize] / 255.f; ++ } ++ ++ main[x + y*main_linesize] = alpha * overlay[x + y*overlay_linesize] + (1.f - alpha) * main[x + y*main_linesize]; ++} ++ ++} +diff --git a/libavutil/hwcontext_cuda.c b/libavutil/hwcontext_cuda.c +index b0b65b2446..372c4722b5 100644 +--- a/libavutil/hwcontext_cuda.c ++++ b/libavutil/hwcontext_cuda.c +@@ -46,6 +46,7 @@ static const enum AVPixelFormat supported_formats[] = { + AV_PIX_FMT_YUV420P, + AV_PIX_FMT_YUVA420P, + AV_PIX_FMT_YUV444P, ++ AV_PIX_FMT_YUVA444P, + AV_PIX_FMT_P010, + AV_PIX_FMT_P016, + AV_PIX_FMT_P210, diff --git a/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch b/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch new file mode 100644 index 00000000..1e231e62 --- /dev/null +++ b/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch @@ -0,0 +1,306 @@ +From 4040eeb26441ebae9f07a0b175b7d7f77688def7 Mon Sep 17 00:00:00 2001 +From: Jan Pietek +Date: Wed, 12 Aug 2026 15:54:13 +0000 +Subject: [PATCH 3/7] avfilter/npp: support CUDA 13 stream context APIs + +Add an NPP compatibility layer and use explicit CUDA stream contexts in scale, sharpen, and transpose filters. + +Original-commit: 30b851f13d97bad9b0b28e7310122d914ac6fc1d +--- + libavfilter/cuda/npp_compat.h | 65 ++++++++++++++++++++++++++++++++++ + libavfilter/vf_scale_npp.c | 56 ++++++++++++++++++++--------- + libavfilter/vf_sharpen_npp.c | 12 +++++-- + libavfilter/vf_transpose_npp.c | 35 ++++++++++++------ + 4 files changed, 140 insertions(+), 28 deletions(-) + create mode 100644 libavfilter/cuda/npp_compat.h + +diff --git a/libavfilter/cuda/npp_compat.h b/libavfilter/cuda/npp_compat.h +new file mode 100644 +index 0000000000..bf5d73e6cd +--- /dev/null ++++ b/libavfilter/cuda/npp_compat.h +@@ -0,0 +1,65 @@ ++/* ++ * CUDA NPP compatibility helpers. ++ * ++ * This file is part of FFmpeg. ++ */ ++ ++#ifndef AVFILTER_CUDA_NPP_COMPAT_H ++#define AVFILTER_CUDA_NPP_COMPAT_H ++ ++#include ++ ++#include ++ ++#include "libavutil/cuda_check.h" ++#include "libavutil/hwcontext_cuda_internal.h" ++ ++#ifndef CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_BLOCK ++#define CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_BLOCK ((CUdevice_attribute)1) ++#endif ++#ifndef CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK ++#define CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK ((CUdevice_attribute)8) ++#endif ++#ifndef CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_MULTIPROCESSOR ++#define CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_MULTIPROCESSOR ((CUdevice_attribute)39) ++#endif ++ ++static inline int ff_npp_get_stream_context(void *logctx, AVCUDADeviceContext *device_hwctx, ++ NppStreamContext *npp_ctx) ++{ ++ CudaFunctions *cu = device_hwctx->internal->cuda_dl; ++ CUdevice dev = device_hwctx->internal->cuda_device; ++ int value; ++ int ret; ++ ++ memset(npp_ctx, 0, sizeof(*npp_ctx)); ++ npp_ctx->hStream = (cudaStream_t)device_hwctx->stream; ++ npp_ctx->nCudaDeviceId = dev; ++ ++#define GET_NPP_DEVICE_ATTR(dst, attr) do { \ ++ ret = FF_CUDA_CHECK_DL(logctx, cu, \ ++ cu->cuDeviceGetAttribute(&(dst), (attr), dev)); \ ++ if (ret < 0) \ ++ return ret; \ ++ } while (0) ++ ++ GET_NPP_DEVICE_ATTR(npp_ctx->nMultiProcessorCount, ++ CU_DEVICE_ATTRIBUTE_MULTIPROCESSOR_COUNT); ++ GET_NPP_DEVICE_ATTR(npp_ctx->nMaxThreadsPerMultiProcessor, ++ CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_MULTIPROCESSOR); ++ GET_NPP_DEVICE_ATTR(npp_ctx->nMaxThreadsPerBlock, ++ CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_BLOCK); ++ GET_NPP_DEVICE_ATTR(value, ++ CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK); ++ npp_ctx->nSharedMemPerBlock = value; ++ GET_NPP_DEVICE_ATTR(npp_ctx->nCudaDevAttrComputeCapabilityMajor, ++ CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR); ++ GET_NPP_DEVICE_ATTR(npp_ctx->nCudaDevAttrComputeCapabilityMinor, ++ CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR); ++ ++#undef GET_NPP_DEVICE_ATTR ++ ++ return 0; ++} ++ ++#endif /* AVFILTER_CUDA_NPP_COMPAT_H */ +diff --git a/libavfilter/vf_scale_npp.c b/libavfilter/vf_scale_npp.c +index 8e9113521c..de179ed580 100644 +--- a/libavfilter/vf_scale_npp.c ++++ b/libavfilter/vf_scale_npp.c +@@ -36,6 +36,7 @@ + #include "libavutil/pixdesc.h" + + #include "avfilter.h" ++#include "cuda/npp_compat.h" + #include "filters.h" + #include "formats.h" + #include "scale_eval.h" +@@ -696,14 +697,22 @@ static int nppscale_deinterleave(AVFilterContext *ctx, NPPScaleStageContext *sta + AVFrame *out, AVFrame *in) + { + AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; ++ AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; ++ NppStreamContext npp_ctx; + NppStatus err; ++ int ret; ++ ++ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); ++ if (ret < 0) ++ return ret; + + switch (in_frames_ctx->sw_format) { + case AV_PIX_FMT_NV12: +- err = nppiYCbCr420_8u_P2P3R(in->data[0], in->linesize[0], +- in->data[1], in->linesize[1], +- out->data, out->linesize, +- (NppiSize){ in->width, in->height }); ++ err = nppiYCbCr420_8u_P2P3R_Ctx(in->data[0], in->linesize[0], ++ in->data[1], in->linesize[1], ++ out->data, out->linesize, ++ (NppiSize){ in->width, in->height }, ++ npp_ctx); + break; + default: + return AVERROR_BUG; +@@ -719,9 +728,16 @@ static int nppscale_deinterleave(AVFilterContext *ctx, NPPScaleStageContext *sta + static int nppscale_resize(AVFilterContext *ctx, NPPScaleStageContext *stage, + AVFrame *out, AVFrame *in) + { ++ AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; ++ AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; + NPPScaleContext *s = ctx->priv; ++ NppStreamContext npp_ctx; + NppStatus err; +- int i; ++ int i, ret; ++ ++ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); ++ if (ret < 0) ++ return ret; + + for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { + int iw = stage->planes_in[i].width; +@@ -729,12 +745,12 @@ static int nppscale_resize(AVFilterContext *ctx, NPPScaleStageContext *stage, + int ow = stage->planes_out[i].width; + int oh = stage->planes_out[i].height; + +- err = nppiResizeSqrPixel_8u_C1R(in->data[i], (NppiSize){ iw, ih }, +- in->linesize[i], (NppiRect){ 0, 0, iw, ih }, +- out->data[i], out->linesize[i], +- (NppiRect){ 0, 0, ow, oh }, +- (double)ow / iw, (double)oh / ih, +- 0.0, 0.0, s->interp_algo); ++ err = nppiResizeSqrPixel_8u_C1R_Ctx(in->data[i], (NppiSize){ iw, ih }, ++ in->linesize[i], (NppiRect){ 0, 0, iw, ih }, ++ out->data[i], out->linesize[i], ++ (NppiRect){ 0, 0, ow, oh }, ++ (double)ow / iw, (double)oh / ih, ++ 0.0, 0.0, s->interp_algo, npp_ctx); + if (err != NPP_SUCCESS) { + av_log(ctx, AV_LOG_ERROR, "NPP resize error: %d\n", err); + return AVERROR_UNKNOWN; +@@ -748,15 +764,23 @@ static int nppscale_interleave(AVFilterContext *ctx, NPPScaleStageContext *stage + AVFrame *out, AVFrame *in) + { + AVHWFramesContext *out_frames_ctx = (AVHWFramesContext*)out->hw_frames_ctx->data; ++ AVCUDADeviceContext *device_hwctx = out_frames_ctx->device_ctx->hwctx; ++ NppStreamContext npp_ctx; + NppStatus err; ++ int ret; ++ ++ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); ++ if (ret < 0) ++ return ret; + + switch (out_frames_ctx->sw_format) { + case AV_PIX_FMT_NV12: +- err = nppiYCbCr420_8u_P3P2R((const uint8_t**)in->data, +- in->linesize, +- out->data[0], out->linesize[0], +- out->data[1], out->linesize[1], +- (NppiSize){ in->width, in->height }); ++ err = nppiYCbCr420_8u_P3P2R_Ctx((const uint8_t**)in->data, ++ in->linesize, ++ out->data[0], out->linesize[0], ++ out->data[1], out->linesize[1], ++ (NppiSize){ in->width, in->height }, ++ npp_ctx); + break; + default: + return AVERROR_BUG; +diff --git a/libavfilter/vf_sharpen_npp.c b/libavfilter/vf_sharpen_npp.c +index 3ec74f8c0c..e3f378292f 100644 +--- a/libavfilter/vf_sharpen_npp.c ++++ b/libavfilter/vf_sharpen_npp.c +@@ -24,6 +24,7 @@ + #include + #include + ++#include "cuda/npp_compat.h" + #include "filters.h" + #include "libavutil/pixdesc.h" + #include "libavutil/cuda_check.h" +@@ -159,17 +160,24 @@ static int nppsharpen_sharpen(AVFilterContext* ctx, AVFrame* out, AVFrame* in) + { + FilterLink *inl = ff_filter_link(ctx->inputs[0]); + AVHWFramesContext* in_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; ++ AVCUDADeviceContext *device_hwctx = in_ctx->device_ctx->hwctx; + NPPSharpenContext* s = ctx->priv; + + const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(in_ctx->sw_format); ++ NppStreamContext npp_ctx; ++ int ret; ++ ++ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); ++ if (ret < 0) ++ return ret; + + for (int i = 0; i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { + int ow = AV_CEIL_RSHIFT(in->width, (i == 1 || i == 2) ? desc->log2_chroma_w : 0); + int oh = AV_CEIL_RSHIFT(in->height, (i == 1 || i == 2) ? desc->log2_chroma_h : 0); + +- NppStatus err = nppiFilterSharpenBorder_8u_C1R( ++ NppStatus err = nppiFilterSharpenBorder_8u_C1R_Ctx( + in->data[i], in->linesize[i], (NppiSize){ow, oh}, (NppiPoint){0, 0}, +- out->data[i], out->linesize[i], (NppiSize){ow, oh}, s->border_type); ++ out->data[i], out->linesize[i], (NppiSize){ow, oh}, s->border_type, npp_ctx); + if (err != NPP_SUCCESS) { + av_log(ctx, AV_LOG_ERROR, "NPP sharpen error: %d\n", err); + return AVERROR_EXTERNAL; +diff --git a/libavfilter/vf_transpose_npp.c b/libavfilter/vf_transpose_npp.c +index 2315b1043a..ad43356523 100644 +--- a/libavfilter/vf_transpose_npp.c ++++ b/libavfilter/vf_transpose_npp.c +@@ -29,6 +29,7 @@ + #include "libavutil/pixdesc.h" + + #include "avfilter.h" ++#include "cuda/npp_compat.h" + #include "filters.h" + #include "formats.h" + #include "video.h" +@@ -294,9 +295,16 @@ static int npptranspose_config_props(AVFilterLink *outlink) + static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *stage, + AVFrame *out, AVFrame *in) + { ++ AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; ++ AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; + NPPTransposeContext *s = ctx->priv; ++ NppStreamContext npp_ctx; + NppStatus err; +- int i; ++ int i, ret; ++ ++ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); ++ if (ret < 0) ++ return ret; + + for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { + int iw = stage->planes_in[i].width; +@@ -311,11 +319,11 @@ static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *s + int shiftw = (s->dir == NPP_TRANSPOSE_CLOCK || s->dir == NPP_TRANSPOSE_CLOCK_FLIP) ? ow - 1 : 0; + int shifth = (s->dir == NPP_TRANSPOSE_CCLOCK || s->dir == NPP_TRANSPOSE_CLOCK_FLIP) ? oh - 1 : 0; + +- err = nppiRotate_8u_C1R(in->data[i], (NppiSize){ iw, ih }, +- in->linesize[i], (NppiRect){ 0, 0, iw, ih }, +- out->data[i], out->linesize[i], +- (NppiRect){ 0, 0, ow, oh }, +- angle, shiftw, shifth, NPPI_INTER_NN); ++ err = nppiRotate_8u_C1R_Ctx(in->data[i], (NppiSize){ iw, ih }, ++ in->linesize[i], (NppiRect){ 0, 0, iw, ih }, ++ out->data[i], out->linesize[i], ++ (NppiRect){ 0, 0, ow, oh }, ++ angle, shiftw, shifth, NPPI_INTER_NN, npp_ctx); + if (err != NPP_SUCCESS) { + av_log(ctx, AV_LOG_ERROR, "NPP rotate error: %d\n", err); + return AVERROR_UNKNOWN; +@@ -328,16 +336,23 @@ static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *s + static int npptranspose_transpose(AVFilterContext *ctx, NPPTransposeStageContext *stage, + AVFrame *out, AVFrame *in) + { ++ AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; ++ AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; ++ NppStreamContext npp_ctx; + NppStatus err; +- int i; ++ int i, ret; ++ ++ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); ++ if (ret < 0) ++ return ret; + + for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { + int iw = stage->planes_in[i].width; + int ih = stage->planes_in[i].height; + +- err = nppiTranspose_8u_C1R(in->data[i], in->linesize[i], +- out->data[i], out->linesize[i], +- (NppiSize){ iw, ih }); ++ err = nppiTranspose_8u_C1R_Ctx(in->data[i], in->linesize[i], ++ out->data[i], out->linesize[i], ++ (NppiSize){ iw, ih }, npp_ctx); + if (err != NPP_SUCCESS) { + av_log(ctx, AV_LOG_ERROR, "NPP transpose error: %d\n", err); + return AVERROR_UNKNOWN; diff --git a/deps/ffmpeg/8.1/0004-avcodec-nvdec-intra.patch b/deps/ffmpeg/8.1/0004-avcodec-nvdec-intra.patch new file mode 100644 index 00000000..dcdeb64b --- /dev/null +++ b/deps/ffmpeg/8.1/0004-avcodec-nvdec-intra.patch @@ -0,0 +1,27 @@ +From 5bcee64dd15453ff5c0bb88ce7b9d189a82c93a3 Mon Sep 17 00:00:00 2001 +From: Jan Pietek +Date: Tue, 15 Sep 2026 09:37:43 +0200 +Subject: [PATCH 4/7] avcodec/nvdec: preserve intra-only stream handling on 8.1 + +--- + libavcodec/cuviddec.c | 7 +++++++ + 1 file changed, 7 insertions(+) + +diff --git a/libavcodec/cuviddec.c b/libavcodec/cuviddec.c +index be183fce35..8a1b7e3042 100644 +--- a/libavcodec/cuviddec.c ++++ b/libavcodec/cuviddec.c +@@ -398,6 +398,13 @@ static int CUDAAPI cuvid_handle_video_sequence(void *opaque, CUVIDEOFORMAT* form + cuinfo.bitDepthMinus8 = format->bit_depth_luma_minus8; + cuinfo.DeinterlaceMode = ctx->deint_mode_current; + ++ if(avctx->gop_size == 100000) { ++ av_log(avctx, AV_LOG_INFO, "init decoder intra only\n"); ++ cuinfo.ulIntraDecodeOnly = 1; ++ } ++ ++ av_log(avctx, AV_LOG_INFO, "init decoder %d %d\n", ctx->nb_surfaces, avctx->refs); ++ + ctx->internal_error = CHECK_CU(ctx->cvdl->cuvidCreateDecoder(&ctx->cudecoder, &cuinfo)); + if (ctx->internal_error < 0) + return 0; diff --git a/deps/ffmpeg/8.1/0005-avformat-rtp-rfc4175.patch b/deps/ffmpeg/8.1/0005-avformat-rtp-rfc4175.patch new file mode 100644 index 00000000..f9f24c8d --- /dev/null +++ b/deps/ffmpeg/8.1/0005-avformat-rtp-rfc4175.patch @@ -0,0 +1,168 @@ +From bab76e3c3f8cf7b7a26bae454cd7736cbb8c30a7 Mon Sep 17 00:00:00 2001 +From: Jan Pietek +Date: Wed, 12 Aug 2026 15:54:13 +0000 +Subject: [PATCH 5/7] avformat/rtp: extend RFC 4175 frame handling + +Support the legacy 4:2:0 payload layout and skip incomplete RFC 4175 frames without emitting corrupted output. + +Original-commits: e4b4dbdccaa82b007bf3f320d7fd761f1cbd9a14 364d9773b677feedc39ed4a3e30a15e95878b811 +--- + libavformat/rtpdec_rfc4175.c | 88 ++++++++++++++++++++++++++++++++---- + 1 file changed, 80 insertions(+), 8 deletions(-) + +diff --git a/libavformat/rtpdec_rfc4175.c b/libavformat/rtpdec_rfc4175.c +index b49fc55d2d..2f0dff0bf0 100644 +--- a/libavformat/rtpdec_rfc4175.c ++++ b/libavformat/rtpdec_rfc4175.c +@@ -43,6 +43,9 @@ struct PayloadContext { + unsigned int frame_size; + unsigned int pgroup; /* size of the pixel group in bytes */ + unsigned int xinc; ++ int is_yuv420; ++ int next_offset; ++ int next_line; + + uint32_t timestamp; + }; +@@ -53,6 +56,7 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) + int tag; + const AVPixFmtDescriptor *desc; + ++ data->is_yuv420 = 0; + if (!strncmp(data->sampling, "YCbCr-4:2:2", 11)) { + tag = MKTAG('U', 'Y', 'V', 'Y'); + data->xinc = 2; +@@ -71,6 +75,7 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) + } else if (!strncmp(data->sampling, "YCbCr-4:2:0", 11)) { + tag = MKTAG('I', '4', '2', '0'); + data->xinc = 4; ++ data->is_yuv420 = 1; + + if (data->depth == 8) { + data->pgroup = 6; +@@ -108,6 +113,8 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) + stream->codecpar->codec_tag = tag; + stream->codecpar->bits_per_coded_sample = av_get_bits_per_pixel(desc); + data->frame_size = data->width * data->height * data->pgroup / data->xinc; ++ data->next_offset = 0; ++ data->next_line = 0; + + if (data->interlaced) + stream->codecpar->field_order = AV_FIELD_TT; +@@ -225,6 +232,8 @@ static int rfc4175_finalize_packet(PayloadContext *data, AVPacket *pkt, + av_freep(&data->frame); + } + data->frame = NULL; ++ data->next_offset = 0; ++ data->next_line = 0; + } + + data->field = 0; +@@ -232,6 +241,14 @@ static int rfc4175_finalize_packet(PayloadContext *data, AVPacket *pkt, + return ret; + } + ++static void rfc4175_discard_packet(PayloadContext *data) ++{ ++ av_freep(&data->frame); ++ data->frame = NULL; ++ data->next_offset = 0; ++ data->next_line = 0; ++} ++ + static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, + AVStream *st, AVPacket *pkt, uint32_t *timestamp, + const uint8_t * buf, int len, +@@ -255,8 +272,12 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, + * previous frame (or pair of fields) anyway by filling the AVPacket. + */ + av_log(ctx, AV_LOG_ERROR, "Missed previous RTP Marker\n"); +- missed_last_packet = 1; +- rfc4175_finalize_packet(data, pkt, st->index); ++ if (ctx->flags & AVFMT_FLAG_DISCARD_CORRUPT) { ++ rfc4175_discard_packet(data); ++ } else { ++ missed_last_packet = 1; ++ rfc4175_finalize_packet(data, pkt, st->index); ++ } + } + + if (!data->frame) +@@ -270,6 +291,11 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, + } + } + ++ if (!data->frame) { ++ /* buffer already freed by discard, or duplicate packet */ ++ return AVERROR(EAGAIN); ++ } ++ + /* + * looks for the 'Continuation bit' in scan lines' headers + * to find where data start +@@ -310,13 +336,59 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, + if (line >= data->height) + return AVERROR_INVALIDDATA; + +- /* prevent ill-formed packets to write after buffer's end */ +- copy_offset = (line * data->width + offset) * data->pgroup / data->xinc; +- if (copy_offset + length > data->frame_size || !data->frame) +- return AVERROR_INVALIDDATA; ++ if ((ctx->flags & AVFMT_FLAG_DISCARD_CORRUPT) && ++ (data->next_line != line || data->next_offset != offset)) { ++ av_log(ctx, AV_LOG_ERROR, ++ "packet loss: line %d -> %d, offset %d -> %d\n", ++ data->next_line, line, data->next_offset, offset); ++ rfc4175_discard_packet(data); ++ return AVERROR(EAGAIN); ++ } ++ if (ctx->flags & AVFMT_FLAG_DISCARD_CORRUPT) { ++ data->next_offset = offset + length * data->xinc / ++ (data->pgroup * (1 + !!data->is_yuv420)); ++ if (data->next_offset >= data->width) { ++ data->next_offset = 0; ++ data->next_line = line + 1 + !!data->is_yuv420; ++ } ++ } + +- dest = data->frame + copy_offset; +- memcpy(dest, payload, length); ++ if (data->is_yuv420) { ++ uint8_t *yd1p, *yd2p, *up, *vp, *udp, *vdp; ++ const uint8_t *p; ++ int uvoff, i; ++ ++ /* libavformat stores YUV420 planar frames, depacketized payload is interleaved */ ++ if (line % 2 != 0 || !data->frame || (offset % 2) != 0 || (length % 6) != 0) ++ return AVERROR_INVALIDDATA; ++ ++ yd1p = data->frame + line * data->width + offset; ++ yd2p = yd1p + data->width; ++ up = data->frame + data->width * data->height; ++ vp = up + data->width / 2 * data->height / 2; ++ uvoff = line / 2 * data->width / 2 + offset / 2; ++ udp = up + uvoff; ++ vdp = vp + uvoff; ++ p = payload; ++ ++ for (i = 0; i < length; i += 6) { ++ *yd1p++ = p[0]; ++ *yd1p++ = p[1]; ++ *yd2p++ = p[2]; ++ *yd2p++ = p[3]; ++ *udp++ = p[4]; ++ *vdp++ = p[5]; ++ p += 6; ++ } ++ } else { ++ /* prevent ill-formed packets to write after buffer's end */ ++ copy_offset = (line * data->width + offset) * data->pgroup / data->xinc; ++ if (copy_offset + length > data->frame_size || !data->frame) ++ return AVERROR_INVALIDDATA; ++ ++ dest = data->frame + copy_offset; ++ memcpy(dest, payload, length); ++ } + + payload += length; + payload_len -= length; diff --git a/deps/ffmpeg/8.1/0006-avdevice-v4l2-compat.patch b/deps/ffmpeg/8.1/0006-avdevice-v4l2-compat.patch new file mode 100644 index 00000000..f8b073f8 --- /dev/null +++ b/deps/ffmpeg/8.1/0006-avdevice-v4l2-compat.patch @@ -0,0 +1,25 @@ +From 74c71ce7e961dd4d99bbcbd14fc084e7a4d7a0c0 Mon Sep 17 00:00:00 2001 +From: Jan Pietek +Date: Wed, 12 Aug 2026 15:54:13 +0000 +Subject: [PATCH 6/7] avdevice/v4l2: retain source timestamps + +Preserve the legacy V4L2 timestamp behavior used by avplumber inputs. + +Original-commit: 961ec93cbc352178ae5572c8d7f03f508edc3147 +--- + libavdevice/v4l2.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/libavdevice/v4l2.c b/libavdevice/v4l2.c +index c38ecbb378..11203f8fec 100644 +--- a/libavdevice/v4l2.c ++++ b/libavdevice/v4l2.c +@@ -584,7 +584,7 @@ static int mmap_read_frame(AVFormatContext *ctx, AVPacket *pkt) + if (ctx->video_codec_id == AV_CODEC_ID_CPIA) + s->frame_size = bytesused; + +- if (s->frame_size > 0 && bytesused != s->frame_size) { ++ if (s->frame_size > 0 && bytesused < s->frame_size) { + av_log(ctx, AV_LOG_WARNING, + "Dequeued v4l2 buffer contains %d bytes, but %d were expected. Flags: 0x%08X.\n", + bytesused, s->frame_size, buf.flags); diff --git a/deps/ffmpeg/8.1/0007-avdevice-ndi-v5.patch b/deps/ffmpeg/8.1/0007-avdevice-ndi-v5.patch new file mode 100644 index 00000000..9daeb08e --- /dev/null +++ b/deps/ffmpeg/8.1/0007-avdevice-ndi-v5.patch @@ -0,0 +1,219 @@ +From 8b939f7ee621f63ac2ecc6175e0c9644eca7861d Mon Sep 17 00:00:00 2001 +From: Jan Pietek +Date: Tue, 15 Sep 2026 09:38:18 +0200 +Subject: [PATCH 7/7] avdevice/ndi: rebase optional NDI v5 registration on 8.1 + +--- + configure | 7 +++++ + doc/indevs.texi | 65 ++++++++++++++++++++++++++++++++++++++++ + doc/outdevs.texi | 45 ++++++++++++++++++++++++++++ + libavdevice/Makefile | 4 +++ + libavdevice/alldevices.c | 2 ++ + 5 files changed, 123 insertions(+) + +diff --git a/configure b/configure +index 584e1df313..f086594121 100755 +--- a/configure ++++ b/configure +@@ -323,6 +323,7 @@ External library support: + --enable-lv2 enable LV2 audio filtering [no] + --disable-lzma disable lzma [autodetect] + --enable-decklink enable Blackmagic DeckLink I/O support [no] ++ --enable-libndi_newtek enable Newteck NDI I/O support [no] + --enable-mbedtls enable mbedTLS, needed for https support + if openssl, gnutls or libtls is not used [no] + --enable-mediacodec enable Android MediaCodec support [no] +@@ -2003,6 +2004,7 @@ EXTERNAL_LIBRARY_GPL_LIST=" + + EXTERNAL_LIBRARY_NONFREE_LIST=" + decklink ++ libndi_newtek + libfdk_aac + libmpeghdec + " +@@ -3983,6 +3985,10 @@ decklink_indev_suggest="libzvbi" + decklink_outdev_deps="decklink threads" + decklink_outdev_suggest="libklvanc" + decklink_outdev_extralibs="-lstdc++" ++libndi_newtek_indev_deps="libndi_newtek" ++libndi_newtek_indev_extralibs="-lndi" ++libndi_newtek_outdev_deps="libndi_newtek" ++libndi_newtek_outdev_extralibs="-lndi" + dshow_indev_deps="IBaseFilter" + dshow_indev_extralibs="-lpsapi -lole32 -lstrmiids -luuid -loleaut32 -lshlwapi" + fbdev_indev_deps="linux_fb_h" +@@ -7233,6 +7239,7 @@ enabled chromaprint && { check_pkg_config chromaprint libchromaprint "chro + require chromaprint chromaprint.h chromaprint_get_version -lchromaprint; } + enabled decklink && { require_headers DeckLinkAPI.h && + { test_cpp_condition DeckLinkAPIVersion.h "BLACKMAGIC_DECKLINK_API_VERSION >= 0x0a0b0000" || die "ERROR: Decklink API version must be >= 10.11"; } } ++enabled libndi_newtek && require libndi_newtek Processing.NDI.Lib.h NDIlib_initialize -lndi + enabled frei0r && require_headers "frei0r.h" + enabled gmp && require gmp gmp.h mpz_export -lgmp + enabled gnutls && require_pkg_config gnutls gnutls gnutls/gnutls.h gnutls_global_init +diff --git a/doc/indevs.texi b/doc/indevs.texi +index 8822e070fe..5ef5d4c208 100644 +--- a/doc/indevs.texi ++++ b/doc/indevs.texi +@@ -1121,6 +1121,71 @@ Set the video size given as a string such as @code{640x480} or @code{hd720}. + Default is @code{qvga}. + @end table + ++@section libndi_newtek ++ ++The libndi_newtek input device provides capture capabilities for using NDI (Network ++Device Interface, standard created by NewTek). ++ ++Input filename is a NDI source name that could be found by sending -find_sources 1 ++to command line - it has no specific syntax but human-readable formatted. ++ ++To enable this input device, you need the NDI SDK and you ++need to configure with the appropriate @code{--extra-cflags} ++and @code{--extra-ldflags}. ++ ++@subsection Options ++ ++@table @option ++ ++@item find_sources ++If set to @option{true}, print a list of found/available NDI sources and exit. ++Defaults to @option{false}. ++ ++@item wait_sources ++Override time to wait until the number of online sources have changed. ++Defaults to @option{0.5}. ++ ++@item allow_video_fields ++When this flag is @option{false}, all video that you receive will be progressive. ++Defaults to @option{true}. ++ ++@item extra_ips ++If is set to list of comma separated ip addresses, scan for sources not only ++using mDNS but also use unicast ip addresses specified by this list. ++ ++@end table ++ ++@subsection Examples ++ ++@itemize ++ ++@item ++List input devices: ++@example ++ffmpeg -f libndi_newtek -find_sources 1 -i dummy ++@end example ++ ++@item ++List local and remote input devices: ++@example ++ffmpeg -f libndi_newtek -extra_ips "192.168.10.10" -find_sources 1 -i dummy ++@end example ++ ++@item ++Restream to NDI: ++@example ++ffmpeg -f libndi_newtek -i "DEV-5.INTERNAL.M1STEREO.TV (NDI_SOURCE_NAME_1)" -f libndi_newtek -y NDI_SOURCE_NAME_2 ++@end example ++ ++@item ++Restream remote NDI to local NDI: ++@example ++ffmpeg -f libndi_newtek -extra_ips "192.168.10.10" -i "DEV-5.REMOTE.M1STEREO.TV (NDI_SOURCE_NAME_1)" -f libndi_newtek -y NDI_SOURCE_NAME_2 ++@end example ++ ++ ++@end itemize ++ + @section openal + + The OpenAL input device provides audio capture on all systems with a +diff --git a/doc/outdevs.texi b/doc/outdevs.texi +index 86c78f31b7..62d342f2c2 100644 +--- a/doc/outdevs.texi ++++ b/doc/outdevs.texi +@@ -301,6 +301,51 @@ ffmpeg -re -i INPUT -c:v rawvideo -pix_fmt bgra -f fbdev /dev/fb0 + + See also @url{http://linux-fbdev.sourceforge.net/}, and fbset(1). + ++@section libndi_newtek ++ ++The libndi_newtek output device provides playback capabilities for using NDI (Network ++Device Interface, standard created by NewTek). ++ ++Output filename is a NDI name. ++ ++To enable this output device, you need the NDI SDK and you ++need to configure with the appropriate @code{--extra-cflags} ++and @code{--extra-ldflags}. ++ ++NDI uses uyvy422 pixel format natively, but also supports bgra, bgr0, rgba and ++rgb0. ++ ++@subsection Options ++ ++@table @option ++ ++@item reference_level ++The audio reference level in dB. This specifies how many dB above the ++reference level (+4dBU) is the full range of 16 bit audio. ++Defaults to @option{0}. ++ ++@item clock_video ++These specify whether video "clock" themselves. ++Defaults to @option{false}. ++ ++@item clock_audio ++These specify whether audio "clock" themselves. ++Defaults to @option{false}. ++ ++@end table ++ ++@subsection Examples ++ ++@itemize ++ ++@item ++Play video clip: ++@example ++ffmpeg -i "udp://@@239.1.1.1:10480?fifo_size=1000000&overrun_nonfatal=1" -vf "scale=720:576,fps=fps=25,setdar=dar=16/9,format=pix_fmts=uyvy422" -f libndi_newtek NEW_NDI1 ++@end example ++ ++@end itemize ++ + @section oss + + OSS (Open Sound System) output device. +diff --git a/libavdevice/Makefile b/libavdevice/Makefile +index a226368d16..272c80deed 100644 +--- a/libavdevice/Makefile ++++ b/libavdevice/Makefile +@@ -21,6 +21,8 @@ OBJS-$(CONFIG_AVFOUNDATION_INDEV) += avfoundation.o + OBJS-$(CONFIG_CACA_OUTDEV) += caca.o + OBJS-$(CONFIG_DECKLINK_OUTDEV) += decklink_enc.o decklink_enc_c.o decklink_common.o + OBJS-$(CONFIG_DECKLINK_INDEV) += decklink_dec.o decklink_dec_c.o decklink_common.o ++OBJS-$(CONFIG_LIBNDI_NEWTEK_OUTDEV) += libndi_newtek_enc.o ++OBJS-$(CONFIG_LIBNDI_NEWTEK_INDEV) += libndi_newtek_dec.o + OBJS-$(CONFIG_DSHOW_INDEV) += dshow_crossbar.o dshow.o dshow_enummediatypes.o \ + dshow_enumpins.o dshow_filter.o \ + dshow_pin.o dshow_common.o +@@ -62,6 +64,8 @@ SHLIBOBJS-$(HAVE_GNU_WINDRES) += avdeviceres.o + SKIPHEADERS += decklink_common.h + SKIPHEADERS-$(CONFIG_DECKLINK) += decklink_enc.h decklink_dec.h \ + decklink_common_c.h ++SKIPHEADERS-$(CONFIG_LIBNDI_NEWTEK_INDEV) += libndi_newtek_common.h ++SKIPHEADERS-$(CONFIG_LIBNDI_NEWTEK_OUTDEV) += libndi_newtek_common.h + SKIPHEADERS-$(CONFIG_DSHOW_INDEV) += dshow_capture.h + SKIPHEADERS-$(CONFIG_FBDEV_INDEV) += fbdev_common.h + SKIPHEADERS-$(CONFIG_FBDEV_OUTDEV) += fbdev_common.h +diff --git a/libavdevice/alldevices.c b/libavdevice/alldevices.c +index 573595f416..3b516e6d3e 100644 +--- a/libavdevice/alldevices.c ++++ b/libavdevice/alldevices.c +@@ -35,6 +35,8 @@ extern const FFInputFormat ff_avfoundation_demuxer; + extern const FFOutputFormat ff_caca_muxer; + extern const FFInputFormat ff_decklink_demuxer; + extern const FFOutputFormat ff_decklink_muxer; ++extern const FFInputFormat ff_libndi_newtek_demuxer; ++extern const FFOutputFormat ff_libndi_newtek_muxer; + extern const FFInputFormat ff_dshow_demuxer; + extern const FFInputFormat ff_fbdev_demuxer; + extern const FFOutputFormat ff_fbdev_muxer; diff --git a/deps/ffmpeg/8.1/README.md b/deps/ffmpeg/8.1/README.md new file mode 100644 index 00000000..7f0f9bfc --- /dev/null +++ b/deps/ffmpeg/8.1/README.md @@ -0,0 +1,61 @@ +# FFmpeg 8.1 compatibility series + +Base: upstream `n8.1`, commit `9047fa1b084f76b1b4d065af2d743df1b40dfb56`. +The exact patched tree is recorded in `base.env`. + +This is the 8.1 adaptation of the seven features in `../7.1.5`, not a new +pixel-format or mixer pipeline design. + +## Adaptations + +1. **AArch64 ARGB conversion:** use `SwsInternal`, `opts` fields and the updated + unscaled callback signature. Retain the existing conversion algorithms. +2. **CUDA composition:** use `FFFilter` registrations for `convert_cuda`, + `crop_cuda`, `overlay_many_cuda`, `pad_cuda` and `transition_cuda`. + The custom `pad_cuda` implementation intentionally replaces the upstream + implementation in this series; switching padding semantics is out of scope. + Keep upstream 8.1 `scale_cuda`, including its expanded pixel formats, and + carry the existing scaling-edge and overlay-context fixes. Retain YUVA444P + CUDA frame support. Use upstream's existing compute-75 compiler fallback. +3. **NPP CUDA 13 compatibility:** carry the stream-context helper and filter + changes; the demo build keeps NPP disabled. +4. **NVDEC intra-only handling:** move the existing initialization hunk to its + corresponding 8.1 location. +5. **RFC 4175:** carry the existing frame handling patch. +6. **V4L2:** carry the existing source-timestamp patch. +7. **NDI v5 registration:** rebase configuration and documentation. As in the + 7.1.5 series, this patch only registers the optional integration; it does not + supply the NDI device implementation files or SDK. NDI remains disabled in + the demo build and is not covered by its compile check. + +FFmpeg 8 removed `AVFrame.pkt_pos`. The legacy `transition_cuda` expression +variable `pos` therefore evaluates to `NAN` (unavailable). `crop_cuda` already +guards that legacy variable by FFmpeg API version. Time/frame-based expressions +and the demo transition configuration do not use packet byte positions. + +FFmpeg 8 also removed the `C` command-support marker from `-filters` output. +The mixer Dockerfile checks the transition's runtime-capable `mode` option in +filter help instead; this check works with both series. + +## Apply and verify + +```bash +git clone --branch n8.1 --depth 1 https://github.com/FFmpeg/FFmpeg +git -C am /deps/ffmpeg/8.1/*.patch +deps/ffmpeg/8.1/verify.sh +``` + +Compilation/linking and filter registration are the acceptance criteria for this +port. They do not establish runtime correctness, performance, or support for +uncompiled optional NPP, NDI or AArch64 paths. Do not deploy over an existing +demo until separate runtime validation is completed. + +Validated on 2026-09-15 in an isolated x86-64 CUDA development container: + +- Both ordered patch series reproduce their pinned trees. +- FFmpeg 8.1 builds; all seven composition/scaling filters are registered. +- Pinned avcpp `31de3f4f937ed3bb30d083275e5e76192dfc9cb3` builds unchanged. +- avplumber binary and Python module build with CUDA/NVCC/DRM/GL enabled, + FRUC/neural/TensorRT disabled. EGL/CUDA binary linking uses the toolkit's + driver stub; the module import check also uses that link-only stub. +- No GPU media graph or running demo was changed by this compile check. diff --git a/deps/ffmpeg/8.1/base.env b/deps/ffmpeg/8.1/base.env new file mode 100644 index 00000000..78bf8fca --- /dev/null +++ b/deps/ffmpeg/8.1/base.env @@ -0,0 +1,3 @@ +base_commit=9047fa1b084f76b1b4d065af2d743df1b40dfb56 +expected_tree=cd89f4187aeadcb94062868063926c5d34db15cf +expected_patch_count=7 diff --git a/deps/ffmpeg/8.1/verify.sh b/deps/ffmpeg/8.1/verify.sh new file mode 100755 index 00000000..ab01ece5 --- /dev/null +++ b/deps/ffmpeg/8.1/verify.sh @@ -0,0 +1,4 @@ +#!/usr/bin/env bash +set -euo pipefail +script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) +exec bash "$script_dir/../verify.sh" 8.1 "$@" diff --git a/deps/ffmpeg/README.md b/deps/ffmpeg/README.md new file mode 100644 index 00000000..18e7c612 --- /dev/null +++ b/deps/ffmpeg/README.md @@ -0,0 +1,37 @@ +# FFmpeg patch series + +Each version directory contains a complete series for one exact upstream tag. +Do not apply the 7.1.5 series and then the 8.1 series to the same checkout. + +| Directory | Upstream tag | Purpose | +| --- | --- | --- | +| `7.1.5/` | `n7.1.5` | Existing default; patch contents preserved unchanged. | +| `8.1/` | `n8.1` | Compile-compatibility port; not a validated demo upgrade. | + +The mixer, CUDA-overlay and DMA-BUF CUDA consumer Dockerfiles select the series +using their existing `FFMPEG_TAG` argument. Their default remains `n7.1.5`. +An isolated 8.1 mixer build can be requested from the repository root with: + +```bash +docker build --build-arg FFMPEG_TAG=n8.1 \ + -f demos/mixer/Dockerfile -t avplumber-mixer:ffmpeg8.1 . +``` + +Build on an NVIDIA development host, with the required submodules populated. +This command creates a separate image; it does not replace a running container. +avcpp and avplumber must be rebuilt against the selected FFmpeg libraries. +Keep the current avcpp revision unless a verified compatibility issue requires +a change. The mixer graph remains 8-bit NV12; this port does not add MXL or +10-bit composition. + +Each version's `base.env` pins the upstream commit, patched tree and patch count. +Verify either series without changing the source checkout's branch: + +```bash +deps/ffmpeg/7.1.5/verify.sh +deps/ffmpeg/8.1/verify.sh +``` + +The supplied checkout must contain the corresponding upstream commit. The +shared verifier uses an isolated worktree and checks the entire ordered series, +not just individual patches. diff --git a/deps/ffmpeg-patches/verify.sh b/deps/ffmpeg/verify.sh similarity index 70% rename from deps/ffmpeg-patches/verify.sh rename to deps/ffmpeg/verify.sh index f0913bb3..7d63c0b1 100755 --- a/deps/ffmpeg-patches/verify.sh +++ b/deps/ffmpeg/verify.sh @@ -1,24 +1,25 @@ #!/usr/bin/env bash set -euo pipefail -base_commit=3a0867c2bfda4a4d4309ca1a8cbdc6175e67f587 -expected_tree=52361f7251069ef74fbb41460e6e1b65d6f9947c -expected_patch_count=7 - -if [[ $# -ne 1 ]]; then - echo "usage: $0 /path/to/FFmpeg" >&2 +if [[ $# -ne 2 ]]; then + echo "usage: $0 <7.1.5|8.1> /path/to/FFmpeg" >&2 exit 2 fi -source_repo=$(git -C "$1" rev-parse --show-toplevel) +script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) +case "$1" in + 7.1.5|8.1) series_dir="$script_dir/$1" ;; + *) echo "unsupported FFmpeg series: $1" >&2; exit 2 ;; +esac +source "$series_dir/base.env" +source_repo=$(git -C "$2" rev-parse --show-toplevel) if ! git -C "$source_repo" cat-file -e "${base_commit}^{commit}"; then echo "FFmpeg checkout does not contain base commit ${base_commit}" >&2 exit 2 fi -script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) shopt -s nullglob -patches=("$script_dir"/*.patch) +patches=("$series_dir"/*.patch) if [[ ${#patches[@]} -ne $expected_patch_count ]]; then echo "expected ${expected_patch_count} patches, found ${#patches[@]}" >&2 exit 1 @@ -38,9 +39,8 @@ trap cleanup EXIT git -C "$source_repo" worktree add --detach "$audit_worktree" "$base_commit" \ >/dev/null -git -C "$audit_worktree" config user.name "avplumber patch verifier" -git -C "$audit_worktree" config user.email "patch-verifier@local" -git -C "$audit_worktree" am --whitespace=nowarn "${patches[@]}" >/dev/null +git -C "$audit_worktree" -c user.name="avplumber patch verifier" \ + -c user.email="patch-verifier@local" am --whitespace=nowarn "${patches[@]}" >/dev/null actual_tree=$(git -C "$audit_worktree" rev-parse 'HEAD^{tree}') if [[ "$actual_tree" != "$expected_tree" ]]; then From bcffb3343e8c1dce6d6baad56176c8f2f01ec4ae Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 10:16:42 +0200 Subject: [PATCH 02/11] Build: probe patched NPP stream-context API on FFmpeg 8.1 --- .../8.1/0003-avfilter-npp-cuda13-compat.patch | 20 +++++++++++++++++-- deps/ffmpeg/8.1/README.md | 3 ++- deps/ffmpeg/8.1/base.env | 2 +- 3 files changed, 21 insertions(+), 4 deletions(-) diff --git a/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch b/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch index 1e231e62..53559f75 100644 --- a/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch +++ b/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch @@ -1,4 +1,4 @@ -From 4040eeb26441ebae9f07a0b175b7d7f77688def7 Mon Sep 17 00:00:00 2001 +From 9492ef1cc9a0473ee4e99549cff3eb08166c9675 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Wed, 12 Aug 2026 15:54:13 +0000 Subject: [PATCH 3/7] avfilter/npp: support CUDA 13 stream context APIs @@ -7,13 +7,29 @@ Add an NPP compatibility layer and use explicit CUDA stream contexts in scale, s Original-commit: 30b851f13d97bad9b0b28e7310122d914ac6fc1d --- + configure | 4 +-- libavfilter/cuda/npp_compat.h | 65 ++++++++++++++++++++++++++++++++++ libavfilter/vf_scale_npp.c | 56 ++++++++++++++++++++--------- libavfilter/vf_sharpen_npp.c | 12 +++++-- libavfilter/vf_transpose_npp.c | 35 ++++++++++++------ - 4 files changed, 140 insertions(+), 28 deletions(-) + 5 files changed, 142 insertions(+), 30 deletions(-) create mode 100644 libavfilter/cuda/npp_compat.h +diff --git a/configure b/configure +index 584e1df313..932be316d4 100755 +--- a/configure ++++ b/configure +@@ -7338,8 +7338,8 @@ enabled libnpp && { test_cpp_condition "$(cd "$source_path"; pwd)/lib + { check_lib libnpp npp.h nppGetLibVersion -lnppig -lnppicc -lnppc -lnppidei -lnppif || + check_lib libnpp npp.h nppGetLibVersion -lnppi -lnppif -lnppc -lnppidei || + die "ERROR: libnpp not found"; } && +- { check_func_headers "nppi.h" nppiYCbCr420_8u_P2P3R $libnpp_extralibs || +- die "ERROR: libnpp support is deprecated, version 13.0 and up are not supported"; } ++ { check_func_headers "nppi.h" nppiYCbCr420_8u_P2P3R_Ctx $libnpp_extralibs || ++ die "ERROR: libnpp stream context APIs not found"; } + enabled libopencore_amrnb && { check_pkg_config libopencore_amrnb opencore-amrnb opencore-amrnb/interf_dec.h Decoder_Interface_init || + require libopencore_amrnb opencore-amrnb/interf_dec.h Decoder_Interface_init -lopencore-amrnb; } + enabled libopencore_amrwb && { check_pkg_config libopencore_amrwb opencore-amrwb opencore-amrwb/dec_if.h D_IF_init || diff --git a/libavfilter/cuda/npp_compat.h b/libavfilter/cuda/npp_compat.h new file mode 100644 index 0000000000..bf5d73e6cd diff --git a/deps/ffmpeg/8.1/README.md b/deps/ffmpeg/8.1/README.md index 7f0f9bfc..92a998ce 100644 --- a/deps/ffmpeg/8.1/README.md +++ b/deps/ffmpeg/8.1/README.md @@ -18,7 +18,8 @@ pixel-format or mixer pipeline design. carry the existing scaling-edge and overlay-context fixes. Retain YUVA444P CUDA frame support. Use upstream's existing compute-75 compiler fallback. 3. **NPP CUDA 13 compatibility:** carry the stream-context helper and filter - changes; the demo build keeps NPP disabled. + changes; probe the stream-context API in configure, since NPP 13 removes + the legacy API checked by upstream. The mixer demo build keeps NPP disabled. 4. **NVDEC intra-only handling:** move the existing initialization hunk to its corresponding 8.1 location. 5. **RFC 4175:** carry the existing frame handling patch. diff --git a/deps/ffmpeg/8.1/base.env b/deps/ffmpeg/8.1/base.env index 78bf8fca..0ccaa766 100644 --- a/deps/ffmpeg/8.1/base.env +++ b/deps/ffmpeg/8.1/base.env @@ -1,3 +1,3 @@ base_commit=9047fa1b084f76b1b4d065af2d743df1b40dfb56 -expected_tree=cd89f4187aeadcb94062868063926c5d34db15cf +expected_tree=3f4f8ba98c493fe9c0e0bfae5a61cd8b83fc30ca expected_patch_count=7 From 818ccdb76395711c9b514edb9dffebd839b66d9d Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 10:40:26 +0200 Subject: [PATCH 03/11] Deps: backport avcpp custom-IO and CMake fixes without API migration --- deps/avcpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/deps/avcpp b/deps/avcpp index 31de3f4f..dac00a23 160000 --- a/deps/avcpp +++ b/deps/avcpp @@ -1 +1 @@ -Subproject commit 31de3f4f937ed3bb30d083275e5e76192dfc9cb3 +Subproject commit dac00a2390000d6b16628cf0d5d0a4a3b0b5bbbb From 9edf53f38fc600472046d42c13d86cd611bdc424 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 10:57:02 +0200 Subject: [PATCH 04/11] Initialize CUDA buffer source parameters before FFmpeg filters --- deps/ffmpeg/8.1/README.md | 8 ++++++++ src/nodes/filters.cpp | 24 ++++++++++++++++-------- 2 files changed, 24 insertions(+), 8 deletions(-) diff --git a/deps/ffmpeg/8.1/README.md b/deps/ffmpeg/8.1/README.md index 92a998ce..304b807b 100644 --- a/deps/ffmpeg/8.1/README.md +++ b/deps/ffmpeg/8.1/README.md @@ -38,6 +38,10 @@ FFmpeg 8 also removed the `C` command-support marker from `-filters` output. The mixer Dockerfile checks the transition's runtime-capable `mode` option in filter help instead; this check works with both series. +The AVP filter node sets the buffer source's `hw_frames_ctx` before initializing +the filter. FFmpeg 8.1 validates CUDA input formats during initialization; +setting the context after `avfilter_graph_create_filter` is too late. + ## Apply and verify ```bash @@ -60,3 +64,7 @@ Validated on 2026-09-15 in an isolated x86-64 CUDA development container: FRUC/neural/TensorRT disabled. EGL/CUDA binary linking uses the toolkit's driver stub; the module import check also uses that link-only stub. - No GPU media graph or running demo was changed by this compile check. + +The current avcpp pin additionally backports custom-IO allocation/cleanup fixes +and CMake link-list handling. These retain the existing wrapper API; they are +maintenance fixes, not requirements for FFmpeg 8.1 compilation or a v3 migration. diff --git a/src/nodes/filters.cpp b/src/nodes/filters.cpp index d3a20611..41bd59f8 100644 --- a/src/nodes/filters.cpp +++ b/src/nodes/filters.cpp @@ -191,16 +191,13 @@ template class FilterNode: p logstream << "Unable to init source filter " << name << ": in_args_ not initialized"; } - // create buffersrc filter const AVFilter* buffersrc = avfilter_get_by_name(ms_.source_filter_name); - int ret = avfilter_graph_create_filter(&ctx_, buffersrc, name.c_str(), in_args_.c_str(), nullptr, filter_graph); - if (ret < 0) { - throw Error("Couldn't create buffer source"); + if (!buffersrc) { + throw Error("Couldn't find buffer source filter"); } - - ret = avfilter_link(ctx_, 0, dst->filter_ctx, dst->pad_idx); - if (ret != 0) { - throw Error("Couldn't link " + name); + ctx_ = avfilter_graph_alloc_filter(filter_graph, buffersrc, name.c_str()); + if (!ctx_) { + throw Error("Couldn't allocate buffer source"); } // Prefer copying hw_frames_ctx from the first frame (if captured), @@ -238,6 +235,17 @@ template class FilterNode: p av_buffer_unref(¶ms->hw_frames_ctx); av_freep(¶ms); } + + // FFmpeg 8.1 validates hardware inputs during initialization, so the + // buffersrc must already have its hw_frames_ctx at this point. + int ret = avfilter_init_str(ctx_, in_args_.c_str()); + if (ret < 0) { + throw Error("Couldn't initialize buffer source: " + av::error2string(ret)); + } + ret = avfilter_link(ctx_, 0, dst->filter_ctx, dst->pad_idx); + if (ret != 0) { + throw Error("Couldn't link " + name); + } } void initSinkFilter(const int index, AVFilterGraph *filter_graph, AVFilterInOut *src) { std::string name = "out" + std::to_string(index); From daf078653b76fbb6dc242c844587ce51c9380878 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 12:24:31 +0200 Subject: [PATCH 05/11] Initialize hardware filter devices before graph parsing completes --- deps/ffmpeg/8.1/README.md | 7 ++++++ src/nodes/filters.cpp | 46 +++++++++++++++++++++++++-------------- 2 files changed, 37 insertions(+), 16 deletions(-) diff --git a/deps/ffmpeg/8.1/README.md b/deps/ffmpeg/8.1/README.md index 304b807b..aa8ba60e 100644 --- a/deps/ffmpeg/8.1/README.md +++ b/deps/ffmpeg/8.1/README.md @@ -41,6 +41,13 @@ filter help instead; this check works with both series. The AVP filter node sets the buffer source's `hw_frames_ctx` before initializing the filter. FFmpeg 8.1 validates CUDA input formats during initialization; setting the context after `avfilter_graph_create_filter` is too late. +Filters that request a hardware device also receive it before initialization. +The AVP node uses FFmpeg's segmented graph parser to attach `hw_device_ctx` +between filter allocation and initialization; this is needed by `hwupload` +when preloading alpha wipes. These APIs are also available in FFmpeg 7.1.5, +so the same AVP filter source supports both versions without a version fork. +The binaries and Python modules must still be built separately for each +FFmpeg ABI. The 7.1.5 patch series and default Docker build version are unchanged. ## Apply and verify diff --git a/src/nodes/filters.cpp b/src/nodes/filters.cpp index 41bd59f8..1a231a20 100644 --- a/src/nodes/filters.cpp +++ b/src/nodes/filters.cpp @@ -360,6 +360,35 @@ template class FilterNode: p sinks_.resize(this->sink_edges_.size()); input_eof_.resize(this->source_edges_.size(), false); } + int parseFilterGraph(AVFilterInOut **inputs, AVFilterInOut **outputs) { + #ifdef AVFILTER_FLAG_HWDEVICE + if (hwaccel_) { + AVFilterGraphSegment *raw_segment = nullptr; + int ret = avfilter_graph_segment_parse(filter_graph_, graph_desc_.c_str(), + 0, &raw_segment); + auto free_segment = [](AVFilterGraphSegment *segment) { + avfilter_graph_segment_free(&segment); + }; + std::unique_ptr + segment(raw_segment, free_segment); + if (ret < 0) return ret; + ret = avfilter_graph_segment_create_filters(segment.get(), 0); + if (ret < 0) return ret; + + // FFmpeg 8.1's hwupload requires the device during init, not just + // format negotiation. Attach it before segment_apply initializes. + for (unsigned j = 0; j < filter_graph_->nb_filters; j++) { + AVFilterContext *fctx = filter_graph_->filters[j]; + if (fctx->filter->flags & AVFILTER_FLAG_HWDEVICE) { + fctx->hw_device_ctx = hwaccel_->refDeviceContext(); + if (!fctx->hw_device_ctx) return AVERROR(ENOMEM); + } + } + return avfilter_graph_segment_apply(segment.get(), 0, inputs, outputs); + } + #endif + return avfilter_graph_parse2(filter_graph_, graph_desc_.c_str(), inputs, outputs); + } bool maybeInitFilterGraph(bool frame_waiting = false) { if (filter_graph_ != nullptr) { freeFilterGraph(); @@ -392,27 +421,12 @@ template class FilterNode: p AVFilterInOut* outputs = nullptr; int ret; - ret = avfilter_graph_parse2(filter_graph_, graph_desc_.c_str(), &inputs, &outputs); + ret = parseFilterGraph(&inputs, &outputs); if (ret < 0) { log_init("error_parse", elapsed_ms()); throw Error("Couldn't parse filter graph"); } - #ifdef AVFILTER_FLAG_HWDEVICE - // Provide the hardware device to any filter in the graph that requests it - // (e.g. hwupload_cuda, hwupload, scale_cuda). This lets those filters - // allocate output HW frames even when the buffersrc is a software format. - // ffmpeg 6.1+ is required for this to work. - if (hwaccel_) { - for (unsigned j = 0; j < filter_graph_->nb_filters; j++) { - AVFilterContext *fctx = filter_graph_->filters[j]; - if (fctx->filter->flags & AVFILTER_FLAG_HWDEVICE) { - fctx->hw_device_ctx = hwaccel_->refDeviceContext(); - } - } - } - #endif - auto forEachInOut = [](AVFilterInOut *inout, std::function cb) { for (; inout != nullptr; inout = inout->next) { cb(inout); From 05215f4eda60cc4c27b57d91dce039e1c444f433 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 12:48:37 +0200 Subject: [PATCH 06/11] Document dual-version filter and GPU mixer runtime validation --- deps/ffmpeg/8.1/README.md | 15 +++++++++++++-- deps/ffmpeg/README.md | 2 +- 2 files changed, 14 insertions(+), 3 deletions(-) diff --git a/deps/ffmpeg/8.1/README.md b/deps/ffmpeg/8.1/README.md index aa8ba60e..cf6fd4d4 100644 --- a/deps/ffmpeg/8.1/README.md +++ b/deps/ffmpeg/8.1/README.md @@ -68,10 +68,21 @@ Validated on 2026-09-15 in an isolated x86-64 CUDA development container: - FFmpeg 8.1 builds; all seven composition/scaling filters are registered. - Pinned avcpp `31de3f4f937ed3bb30d083275e5e76192dfc9cb3` builds unchanged. - avplumber binary and Python module build with CUDA/NVCC/DRM/GL enabled, - FRUC/neural/TensorRT disabled. EGL/CUDA binary linking uses the toolkit's - driver stub; the module import check also uses that link-only stub. + FRUC/neural/TensorRT disabled. - No GPU media graph or running demo was changed by this compile check. +Subsequent GPU runtime checks covered the 8-bit mixer at 30 and 60 fps, 41 prewarmed +scenes, video and DMA-BUF browser inputs, all five wipe-cache loads, cut/fade/wipe +commands, and NVENC output received by a WebRTC browser. The same filter source +also compiled against FFmpeg 7.1.5 and passed CUDA scaling and alpha-upload tests. + +When reusing a build tree with different GL feature flags, rebuild +`deps/cuda_loader/cuda_drvapi_dynlink.o` with the new flags. A loader compiled +without GL lacks the EGL function-pointer variables. Linking `libcuda` directly +to satisfy those missing symbols is incorrect: it supplies functions where AVP +expects variables and crashes during DMA-BUF import. The validated mixer module +uses the GL-enabled dynamic loader, without direct `libcuda` linkage. + The current avcpp pin additionally backports custom-IO allocation/cleanup fixes and CMake link-list handling. These retain the existing wrapper API; they are maintenance fixes, not requirements for FFmpeg 8.1 compilation or a v3 migration. diff --git a/deps/ffmpeg/README.md b/deps/ffmpeg/README.md index 18e7c612..33477759 100644 --- a/deps/ffmpeg/README.md +++ b/deps/ffmpeg/README.md @@ -6,7 +6,7 @@ Do not apply the 7.1.5 series and then the 8.1 series to the same checkout. | Directory | Upstream tag | Purpose | | --- | --- | --- | | `7.1.5/` | `n7.1.5` | Existing default; patch contents preserved unchanged. | -| `8.1/` | `n8.1` | Compile-compatibility port; not a validated demo upgrade. | +| `8.1/` | `n8.1` | Compatibility port with 8-bit CUDA mixer runtime checks; see its README for coverage. | The mixer, CUDA-overlay and DMA-BUF CUDA consumer Dockerfiles select the series using their existing `FFMPEG_TAG` argument. Their default remains `n7.1.5`. From bc5b2ac98359014575beb7a3f9e4d6ab6b3e9b5c Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 18:31:05 +0200 Subject: [PATCH 07/11] Complete FFmpeg 8.1 CUDA crop and live EOF compatibility --- deps/ffmpeg/8.1/README.md | 71 ++++++++++++++++ doc/NODES.md | 7 ++ src/nodes/assume_metadata.cpp | 34 ++++++-- src/nodes/crop_metadata_cuda.cpp | 19 ++++- tests/cuda/smoke_crop_filter_chain.py | 111 ++++++++++++++++++++++++++ tests/smoke_format_eof.py | 77 ++++++++++++++++++ 6 files changed, 309 insertions(+), 10 deletions(-) create mode 100644 tests/cuda/smoke_crop_filter_chain.py create mode 100644 tests/smoke_format_eof.py diff --git a/deps/ffmpeg/8.1/README.md b/deps/ffmpeg/8.1/README.md index cf6fd4d4..4aaef1e0 100644 --- a/deps/ffmpeg/8.1/README.md +++ b/deps/ffmpeg/8.1/README.md @@ -41,6 +41,9 @@ filter help instead; this check works with both series. The AVP filter node sets the buffer source's `hw_frames_ctx` before initializing the filter. FFmpeg 8.1 validates CUDA input formats during initialization; setting the context after `avfilter_graph_create_filter` is too late. +The metadata-driven CUDA crop node builds its own graph and follows the same +allocate -> attach frames context -> initialize sequence. Updating only the +generic filter node does not cover crop, portrait, or two-box output paths. Filters that request a hardware device also receive it before initialization. The AVP node uses FFmpeg's segmented graph parser to attach `hw_device_ctx` between filter allocation and initialization; this is needed by `hwupload` @@ -49,6 +52,34 @@ so the same AVP filter source supports both versions without a version fork. The binaries and Python modules must still be built separately for each FFmpeg ABI. The 7.1.5 patch series and default Docker build version are unchanged. +## Hardware acceleration gains and limits + +Compared with upstream n7.1.5, the n8.1 CUDA/NVIDIA path makes these features +available to applications that select the corresponding formats and codecs: + +| Capability | Gain | Requirement / current coverage | +| --- | --- | --- | +| H.264 10-bit NVDEC/NVENC | Hardware High10 decode and encode | Blackwell GPU, SDK 13 headers and compatible driver; not tested on Blackwell. | +| H.264 / HEVC 4:2:2 NVDEC/NVENC | Hardware paths for higher chroma resolution, including 10-bit 4:2:2 | Blackwell GPU, SDK 13 headers and compatible driver; not provided by a T4 or L4 upgrade to FFmpeg alone. | +| CUDA scaling formats | Adds planar 4:2:2, NV16, P210/P216 and planar 10-bit 4:2:0/4:2:2/4:4:4 to upstream `scale_cuda` | CUDA format/scaling support is distinct from hardware codec support. P010 10-bit 4:2:0 already existed in 7.1.5. | +| Existing HEVC Main10 | Remains available on supporting GPUs | Not a new 8.1 capability; this PR does not qualify an end-to-end Main10 graph. | +| Existing custom CUDA composition and CUDA 13 NPP | Keeps the seven-patch suite buildable and usable with the new FFmpeg API | Tested 8-bit paths; custom padding/overlays/inference do not become 10-bit or 4:2:2 automatically. | + +The recorder and mixer remain configured for 8-bit NV12/4:2:0. A 10-bit or 4:2:2 +end-to-end product pipeline still needs compatible decode, filter, composition, +inference and encode stages, plus matching frame metadata. This update does not +add HDR tone mapping or qualify HDR metadata preservation. T4 testing cannot +establish Blackwell codec support or throughput gains. + +SDK-dependent NVENC options are compiled conditionally. The mixer demo still +pins `NV_CODEC_HEADERS_TAG=n12.1.14.0`; a Blackwell build must select SDK 13-era +headers and a matching driver as well as `FFMPEG_TAG=n8.1`. + +Sources: [NVIDIA SDK 13 release notes](https://docs.nvidia.com/video-technologies/video-codec-sdk/13.0/read-me/index.html), +[FFmpeg n8.1 H.264 NVENC profiles](https://github.com/FFmpeg/FFmpeg/blob/n8.1/libavcodec/nvenc_h264.c), +[FFmpeg n8.1 CUDA scaler](https://github.com/FFmpeg/FFmpeg/blob/n8.1/libavfilter/vf_scale_cuda.c), +[FFmpeg n7.1.5 CUDA scaler](https://github.com/FFmpeg/FFmpeg/blob/n7.1.5/libavfilter/vf_scale_cuda.c). + ## Apply and verify ```bash @@ -86,3 +117,43 @@ uses the GL-enabled dynamic loader, without direct `libcuda` linkage. The current avcpp pin additionally backports custom-IO allocation/cleanup fixes and CMake link-list handling. These retain the existing wrapper API; they are maintenance fixes, not requirements for FFmpeg 8.1 compilation or a v3 migration. + +## Full reframer and composition checks + +The 2026-09-15 T4 check with the reframer's eight-patch FFmpeg 8.1 runtime, +CUDA 13/NPP, TensorRT and legacy float TrackNet covered native 1080p25 input, +H=20 camera-pan planning, Player 360p, salient detection, frame classification, +15 Hz scoreboard OCR, DMA-BUF browser overlays, portrait/square crops and eight +HLS video renditions. All 2,502 measured frames reached every pre-NVENC branch; +the latency collector reported no incomplete frames or dropped packets. All +eight finalized renditions were 25 fps and 100.2 seconds long. + +This validates functionality, not steady low-latency performance: processing +had catch-up bursts, with post-NVDEC-to-pre-NVENC latency of 1.095 s median, +3.635 s p95 and 4.359 s maximum. GPU utilization was 54% median and peak device +memory was 2,558 MiB. Native 60 fps remains unqualified. + +The independent `demos/cuda-overlay` pixel-reference matrix passed all 45 cases +on FFmpeg 8.1: 1-15 overlays in 420/420, 420/444 and 444/444 combinations, +including a 641-pixel-wide canvas. Every compared YUV sample matched. + +`tests/cuda/smoke_crop_filter_chain.py` exercises the AVP crop node together +with CUDA padding, scaling, format conversion and NVENC. +Use `--scaler scale_npp` to cover NPP and `--band-blur` when the reframer's +optional `band_blur_cuda` patch is installed. CPU decoding is only the final +encoded-output assertion, not a transfer inside the CUDA processing chain. + +## Live recorder EOF regression + +A live SRT disconnect can finish the input group and propagate EOF into the +permanent pre-sentinel format declaration. `ignore_eof=true` on +`fake_video_format` / `fake_audio_metadata` keeps those nodes accepting frames +across reconnection. This is opt-in; default finite-graph EOF still propagates. + +The FFmpeg 8.1 T4 recorder check survived two SRT disconnects with one recorder +generation. Its 1,600 consecutive 25 fps metadata records matched Kafka and GCS +JSONL; all primary HLS outputs contained 64 finalized one-second segments. +Native audio/video reconnect tests preserve their decoded frame timestamp +sequences, while default finite-EOF tests still finish. The unpatched image +fails the live-EOF regression. Full-recorder finite-VOD completion remains +separate work; the recorder still applies its live restart policy to file input. diff --git a/doc/NODES.md b/doc/NODES.md index d90e1e9f..da9e5791 100644 --- a/doc/NODES.md +++ b/doc/NODES.md @@ -284,6 +284,13 @@ when real metadata aren't available yet. 1 input, 1 output: `av::VideoFrame` or `av::AudioSamples` +Shared parameter (also supported by `fake_video_format` / `fake_audio_metadata`): + +- `ignore_eof` (bool, default `false`) - discard upstream EOF markers and keep + processing subsequent frames. Enable only at a live reconnect boundary before + a sentinel. The default forwards EOF and finishes; explicit shutdown and errors + are unaffected. + Parameters for video: - `width` (int) - default 1920 - `height` (int) - default 1080 diff --git a/src/nodes/assume_metadata.cpp b/src/nodes/assume_metadata.cpp index be3bf67f..b9e55e2b 100644 --- a/src/nodes/assume_metadata.cpp +++ b/src/nodes/assume_metadata.cpp @@ -1,6 +1,24 @@ #include "node_common.hpp" -class AssumeAudioFormat: public TransparentNode, public IAudioMetadataSource, public ITimeBaseSource { +// Live format declarations must remain available while upstream reconnects. +// Keep ordinary EOF propagation as the default for finite graphs. +template class FormatDeclaration: public TransparentNode { +public: + using TransparentNode::TransparentNode; + bool ignore_eof = false; + + bool consumeEofIfPresent() override { + if (!ignore_eof) return TransparentNode::consumeEofIfPresent(); + T* frame = this->source_->peek(0); + if (frame && isEofMarker(*frame)) { + this->source_->pop(); + logstream << "Ignoring upstream EOF at live format boundary"; + } + return false; + } +}; + +class AssumeAudioFormat: public FormatDeclaration, public IAudioMetadataSource, public ITimeBaseSource { private: int sample_rate_; av::SampleFormat sample_format_; @@ -19,7 +37,7 @@ class AssumeAudioFormat: public TransparentNode, public IAudio return {1, sample_rate_}; } AssumeAudioFormat(std::unique_ptr> &&source, std::unique_ptr> &&sink, const int sample_rate, const av::SampleFormat sample_format, const uint64_t channel_layout): - TransparentNode(std::move(source), std::move(sink)), + FormatDeclaration(std::move(source), std::move(sink)), sample_rate_(sample_rate), sample_format_(sample_format), channel_layout_(channel_layout) { } static std::shared_ptr create(NodeCreationInfo &nci) { @@ -31,13 +49,15 @@ class AssumeAudioFormat: public TransparentNode, public IAudio if (params.count("sample_rate")) sr = params["sample_rate"]; if (params.count("sample_format")) fmt = av::SampleFormat(params["sample_format"].get()); if (params.count("channel_layout")) chl = stringToChannelLayout(params["channel_layout"].get().c_str()); - return NodeSISO::template createCommon(edges, params, sr, fmt, chl); + auto node = NodeSISO::template createCommon(edges, params, sr, fmt, chl); + node->ignore_eof = params.value("ignore_eof", false); + return node; } virtual ~AssumeAudioFormat() { } }; -class AssumeVideoFormat: public TransparentNode, public IVideoFormatSource { +class AssumeVideoFormat: public FormatDeclaration, public IVideoFormatSource { private: int width_, height_; av::PixelFormat pix_fmt_; @@ -57,7 +77,7 @@ class AssumeVideoFormat: public TransparentNode, public IVideoFo } AssumeVideoFormat(std::unique_ptr> &&source, std::unique_ptr> &&sink, const int width, const int height, const av::PixelFormat pixfmt, const av::PixelFormat real_pixfmt): - TransparentNode(std::move(source), std::move(sink)), + FormatDeclaration(std::move(source), std::move(sink)), width_(width), height_(height), pix_fmt_(pixfmt), real_pix_fmt_(real_pixfmt) { } static std::shared_ptr create(NodeCreationInfo &nci) { @@ -71,7 +91,9 @@ class AssumeVideoFormat: public TransparentNode, public IVideoFo if (params.count("height")) height = params["height"]; if (params.count("pixel_format")) pixel_format = av::PixelFormat(params["pixel_format"].get()); if (params.count("real_pixel_format")) real_pixel_format = av::PixelFormat(params["real_pixel_format"].get()); - return NodeSISO::template createCommon(edges, params, width, height, pixel_format, real_pixel_format); + auto node = NodeSISO::template createCommon(edges, params, width, height, pixel_format, real_pixel_format); + node->ignore_eof = params.value("ignore_eof", false); + return node; } }; diff --git a/src/nodes/crop_metadata_cuda.cpp b/src/nodes/crop_metadata_cuda.cpp index f52ccfd9..fb49f566 100644 --- a/src/nodes/crop_metadata_cuda.cpp +++ b/src/nodes/crop_metadata_cuda.cpp @@ -152,9 +152,9 @@ class MetadataDrivenCudaCrop: public NodeSISO, } const std::string in_args = buildSourceArgsString(); - int ret = avfilter_graph_create_filter(&buffersrc_ctx_, buffersrc, "in", in_args.c_str(), nullptr, filter_graph_); - if (ret < 0) { - throw Error("crop_metadata_cuda: couldn't create buffer source"); + buffersrc_ctx_ = avfilter_graph_alloc_filter(filter_graph_, buffersrc, "in"); + if (!buffersrc_ctx_) { + throw Error("crop_metadata_cuda: couldn't allocate buffer source"); } AVBufferSrcParameters *src_params = av_buffersrc_parameters_alloc(); @@ -162,13 +162,24 @@ class MetadataDrivenCudaCrop: public NodeSISO, throw Error("crop_metadata_cuda: av_buffersrc_parameters_alloc failed"); } src_params->hw_frames_ctx = av_buffer_ref(initial_hw_frames_ctx_); - ret = av_buffersrc_parameters_set(buffersrc_ctx_, src_params); + if (!src_params->hw_frames_ctx) { + av_freep(&src_params); + throw Error("crop_metadata_cuda: couldn't reference buffer source CUDA frames"); + } + int ret = av_buffersrc_parameters_set(buffersrc_ctx_, src_params); av_buffer_unref(&src_params->hw_frames_ctx); av_freep(&src_params); if (ret < 0) { throw Error("crop_metadata_cuda: av_buffersrc_parameters_set failed"); } + // FFmpeg 8.1 validates CUDA sources during init, so the frames context + // must be attached before initializing, not just before graph config. + ret = avfilter_init_str(buffersrc_ctx_, in_args.c_str()); + if (ret < 0) { + throw Error("crop_metadata_cuda: couldn't initialize buffer source: " + av::error2string(ret)); + } + std::stringstream crop_args; crop_args << "w=" << dst_width_ << ":h=" << dst_height_ << ":x=" << last_crop_x_ << ":y=" << last_crop_y_; diff --git a/tests/cuda/smoke_crop_filter_chain.py b/tests/cuda/smoke_crop_filter_chain.py new file mode 100644 index 00000000..1a0454f8 --- /dev/null +++ b/tests/cuda/smoke_crop_filter_chain.py @@ -0,0 +1,111 @@ +"""Bounded NVDEC -> metadata crop -> pad/scale/format conversion -> NVENC smoke. + +Run on an isolated NVIDIA host with the patched FFmpeg libraries and a video +fixture at least 640x360. Frames stay on CUDA throughout the processing graph; +the independent software decode below only validates the encoded output. +""" +import argparse +import json +from pathlib import Path +import subprocess +import time + +from pyplumber import AVPlumber +from pyplumber.node import ( + AssumeVideoFormat, CropMetadataCuda, DecVideo, Demux, EncVideo, + FilterVideo, InputRec, +) + + +def run(args): + args.output.parent.mkdir(parents=True, exist_ok=True) + fixture = args.output.with_suffix(".input.mkv") + # Stream-copy a bounded encoded fixture; no pixel conversion is involved. + subprocess.run([args.ffmpeg, "-v", "error", "-y", "-i", args.input, + "-map", "0:v:0", "-c", "copy", "-frames:v", str(args.frames + 16), + str(fixture)], check=True, timeout=30) + avp = AVPlumber() + errors = [] + avp.on_exception = lambda *error: errors.append(tuple(map(str, error))) + avp.executeCommandsFromString('hwaccel.init {"name":"filter_chain_gpu","type":"cuda"}') + avp.edges.planCapacity("*", 4) + graph = ( + f"pad_cuda=672:384:16:12:color=black,{args.scaler}=640:360," + "convert_cuda=format=yuv420p,convert_cuda=format=nv12" + ) + if args.band_blur: + graph += ",band_blur_cuda" + nodes = [ + InputRec({"name": "input", "url": str(fixture), "loop": False, "dst": "packets"}), + Demux({"name": "demux", "src": "packets", "routing": {"v:0": "video_packets"}}), + DecVideo({"name": "decode", "src": "video_packets", "dst": "decoded", + "hwaccel": "filter_chain_gpu", "pixel_format": "cuda"}), + # Empty metadata selects the crop node's documented center fallback. + # The FFmpeg metadata filter edits properties, not CUDA pixel data. + FilterVideo({"name": "metadata", "src": "decoded", "dst": "marked", + "graph": "metadata=mode=add:key=reframer_bbox:value='{}'"}), + CropMetadataCuda({"name": "crop", "src": "marked", "dst": "cropped", + "dst_width": 640, "dst_height": 360, + "offset_log_path": str(args.output.with_suffix(".crop.log"))}), + FilterVideo({"name": "chain", "src": "cropped", "dst": "filtered", "graph": graph, + "hwaccel": "filter_chain_gpu", "dst_width": 640, "dst_height": 360, + "dst_pixel_format": "cuda", "dst_frame_rate": "25/1", + "defer_preliminary_init": True}), + AssumeVideoFormat({"name": "format", "src": "filtered", "dst": "nv12", + "width": 640, "height": 360, "pixel_format": "cuda", + "real_pixel_format": "nv12"}), + EncVideo({"name": "encode", "src": "nv12", "dst": "encoded", + "hwaccel": "filter_chain_gpu", "codec": "h264_nvenc", + "options": {"preset": "p1", "tune": "ull", "bf": 0, + "rc-lookahead": 0, "delay": 0, "g": 25}}), + ] + for node in nodes: + node.parameters.update({"group": "test", "auto_restart": "off"}) + avp.addNode(node) + del node + encoded = avp.getEdge("encoded", "Packet") + packets = [] + try: + avp.group("test").startNodes() + deadline = time.monotonic() + args.timeout + finished = False + while time.monotonic() < deadline and not errors: + packet = encoded.tryGet(100) + if packet is None: + continue + # AVPlumber's packet EOF marker has a one-byte payload and NOPTS. + if packet.pts.timestamp == -(1 << 63): + finished = True + break + if packet.size > 0: + if len(packets) < args.frames: + packets.append(packet.data) + assert not errors, errors + assert finished, "encoded stream did not reach EOF" + assert len(packets) == args.frames, f"only {len(packets)}/{args.frames} encoded frames" + finally: + # The crop node is EOF-driven, so use a finite fixture and drain the + # stream instead of forcing a stop midway through its filter graph. + nodes.clear() + avp.shutdown() + args.output.write_bytes(b"".join(packets)) + decoded = subprocess.run([ + args.ffmpeg, "-v", "error", "-threads", "1", "-f", "h264", "-i", str(args.output), + "-vf", "scale=2:2", "-pix_fmt", "gray", "-fps_mode", "passthrough", + "-f", "rawvideo", "pipe:1", + ], check=True, capture_output=True, timeout=30).stdout + assert len(decoded) == args.frames * 4, "independent decode frame count differs" + print(json.dumps({"status": "passed", "frames": args.frames, "scaler": args.scaler, + "graph": graph, "output": str(args.output)}), flush=True) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--input", required=True) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--frames", type=int, default=60) + parser.add_argument("--timeout", type=float, default=30) + parser.add_argument("--scaler", choices=("scale_cuda", "scale_npp"), default="scale_cuda") + parser.add_argument("--band-blur", action="store_true", help="also check the reframer's optional band_blur_cuda patch") + parser.add_argument("--ffmpeg", default="/usr/local/bin/ffmpeg") + run(parser.parse_args()) diff --git a/tests/smoke_format_eof.py b/tests/smoke_format_eof.py new file mode 100644 index 00000000..ce9c3e43 --- /dev/null +++ b/tests/smoke_format_eof.py @@ -0,0 +1,77 @@ +"""Native regression: decode a short A/V fixture, reach EOF, then reconnect. + +Run in the built avplumber environment: + python3 tests/smoke_format_eof.py +""" + +import sys +import time + +from pyplumber import AVPlumber +from pyplumber import node as n + + +def run(path, media, live): + avp = AVPlumber() + errors = [] + avp.on_exception = lambda *error: errors.append(error) + video = media == "video" + data_type = "VideoFrame" if video else "AudioSamples" + decoder = n.DecVideo if video else n.DecAudio + declaration = n.FakeVideoFormat if video else n.FakeAudioMetadata + avp.edges.planCapacity("*", 4096) + for cls, params in ( + (n.InputRec, {"name": "input", "url": path, "dst": "packets"}), + (n.Demux, {"name": "demux", "src": "packets", "routing": { + "v:0" if video else "a:0": "selected"}}), + (decoder, {"name": "decode", "src": "selected", "dst": "decoded"}), + ): + avp.addNode(cls({"group": "input", **params})) + params = {"name": "format", "group": "output", "src": "decoded", "dst": "result"} + if live: + params["ignore_eof"] = True + avp.addNode(declaration(params)) + source = avp.getEdge("decoded", data_type) + result = avp.getEdge("result", data_type) + passes = [] + try: + avp.group("output").startNodes() + for attempt in range(2 if live else 1): + previous_input = source.enqueued_total + if attempt: + avp.group("input").restartNodes() + else: + avp.group("input").startNodes() + deadline = time.monotonic() + 15 + while True: + assert not errors, errors + assert time.monotonic() < deadline, "decode/EOF did not complete" + if (source.enqueued_total > previous_input + and not avp.node("decode").isWorking + and source.occupied == 0): + if live or not avp.node("format").isWorking: + break + time.sleep(0.01) + rows = [] + eof = 0 + while result.occupied: + frame = result.wait_dequeue() + if frame.pts.timestamp == -(1 << 63): + eof += 1 + else: + rows.append((frame.pts.timestamp, frame.pts.timebase.num, + frame.pts.timebase.den)) + assert rows, "no decoded media reached the output" + assert eof == (0 if live else 1), (live, eof) + assert avp.node("format").isWorking == live + passes.append(rows) + if live: + assert passes[0] == passes[1], "reconnect changed or lost decoded frames" + print(f"PASS {media} {'live reconnect' if live else 'default finite EOF'}: " + f"{len(passes[0])} frames per pass", flush=True) + finally: + avp.shutdown() + + +if __name__ == "__main__": + run(sys.argv[1], sys.argv[2], sys.argv[3] == "live") From 11a9c778e5a87be84967fd0236a09246d2400eaa Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 18:37:49 +0200 Subject: [PATCH 08/11] Make crop smoke startup and EOF teardown deterministic --- tests/cuda/smoke_crop_filter_chain.py | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/tests/cuda/smoke_crop_filter_chain.py b/tests/cuda/smoke_crop_filter_chain.py index 1a0454f8..9876b2d5 100644 --- a/tests/cuda/smoke_crop_filter_chain.py +++ b/tests/cuda/smoke_crop_filter_chain.py @@ -13,7 +13,7 @@ from pyplumber import AVPlumber from pyplumber.node import ( AssumeVideoFormat, CropMetadataCuda, DecVideo, Demux, EncVideo, - FilterVideo, InputRec, + FilterVideo, ForceFPS, InputRec, ) @@ -51,7 +51,10 @@ def run(args): "hwaccel": "filter_chain_gpu", "dst_width": 640, "dst_height": 360, "dst_pixel_format": "cuda", "dst_frame_rate": "25/1", "defer_preliminary_init": True}), - AssumeVideoFormat({"name": "format", "src": "filtered", "dst": "nv12", + # Match the recorder's static timebase boundary before NVENC. The + # deferred CUDA filter has no output timebase until its first frame. + ForceFPS({"name": "fps", "src": "filtered", "dst": "paced", "fps": "25/1"}), + AssumeVideoFormat({"name": "format", "src": "paced", "dst": "nv12", "width": 640, "height": 360, "pixel_format": "cuda", "real_pixel_format": "nv12"}), EncVideo({"name": "encode", "src": "nv12", "dst": "encoded", @@ -83,6 +86,14 @@ def run(args): assert not errors, errors assert finished, "encoded stream did not reach EOF" assert len(packets) == args.frames, f"only {len(packets)}/{args.frames} encoded frames" + # The encoder can enqueue EOF before its worker (or an upstream + # decoder) has unwound. Await those EOF-driven workers before shutdown; + # they deliberately have no interface for an abrupt mid-frame stop. + eof_workers = [node.parameters["name"] for node in nodes + if node.parameters["name"] != "fps"] + while any(avp.node(name).isWorking for name in eof_workers): + assert time.monotonic() < deadline, "EOF workers did not finish" + time.sleep(0.01) finally: # The crop node is EOF-driven, so use a finite fixture and drain the # stream instead of forcing a stop midway through its filter graph. From 23bd4b552bd3674ad2993232a42f006367d6cd0d Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Tue, 15 Sep 2026 20:40:37 +0200 Subject: [PATCH 09/11] Mixer: rate-limit forced keyframes with a configurable 150 ms default Coalesce pending requests across cuts, RTCP feedback, and periodic forcing without delaying video frames. Add the mixer CLI option, native node tests, and a 16-input NVENC rapid-cut/fade regression. Validated 133 mixer/timing tests and 18 native limiter tests, plus isolated 60 fps NVENC A/B runs. --- avpmixer/janus.py | 6 ++ demos/mixer/README.md | 6 ++ demos/mixer/mixer.py | 14 ++- demos/mixer/tests/test_graph.py | 40 +++++++- doc/NODES.md | 18 +++- src/nodes/force_keyframe.cpp | 55 +++++++++-- tests/cuda/smoke_keyframe_limit.py | 148 ++++++++++++++++++++++++++++ tests/test_force_keyframe_native.py | 144 +++++++++++++++++++++++++++ 8 files changed, 417 insertions(+), 14 deletions(-) create mode 100644 tests/cuda/smoke_keyframe_limit.py create mode 100644 tests/test_force_keyframe_native.py diff --git a/avpmixer/janus.py b/avpmixer/janus.py index 4395c1f2..d755ba0d 100644 --- a/avpmixer/janus.py +++ b/avpmixer/janus.py @@ -5,6 +5,7 @@ from dataclasses import dataclass RTP_PACKET_SIZE = 1_200 +DEFAULT_KEYFRAME_MIN_INTERVAL_MS = 150 @dataclass(frozen=True) @@ -16,6 +17,7 @@ class JanusVideoConfig: bitrate_kbps: int = 4_500 rtcp_bind: str = "0.0.0.0" rtcp_port: int = 0 + keyframe_min_interval_ms: int = DEFAULT_KEYFRAME_MIN_INTERVAL_MS def __post_init__(self) -> None: if not self.host: @@ -30,6 +32,9 @@ def __post_init__(self) -> None: raise ValueError("Janus bitrate must be positive") if not 0 <= self.rtcp_port <= 65535: raise ValueError("RTCP port must be between 0 and 65535") + if (type(self.keyframe_min_interval_ms) is not int + or not 0 <= self.keyframe_min_interval_ms <= 2_147_483_647): + raise ValueError("keyframe_min_interval_ms must be a non-negative integer") @property def rtcp_port_remote(self) -> int: @@ -58,6 +63,7 @@ def build_janus_output(avp, api, src_edge: str, janus: JanusVideoConfig, *, fps: avp.addNode(api.ForceKeyFrame({ "name": JANUS_KEYFRAME_NODE, "src": "janus_fps", "dst": "janus_keyframed", "interval_sec": "1/1", "auto_restart": "panic", "group": group, + "min_interval_ms": janus.keyframe_min_interval_ms, })) avp.addNode(api.AssumeVideoFormat({ "name": "janus_format", "src": "janus_keyframed", "dst": "janus_video", diff --git a/demos/mixer/README.md b/demos/mixer/README.md index 503c7ad9..21eba91b 100644 --- a/demos/mixer/README.md +++ b/demos/mixer/README.md @@ -149,6 +149,12 @@ python3 demos/mixer/tests/frame_codes.py media --sources 16 --width 1920 --heigh ## Under the hood +Janus output limits forced keyframes to one per 150 ms by default (9 frames at +60 fps). Override with `--keyframe-min-interval-ms 200`; `0` disables the limit. +The option also applies to Janus renditions loaded with `--config`. Cuts and +ordinary frames are not delayed: pending cut/RTCP requests coalesce until the +next eligible frame. Periodic keyframes share the same limit. + Grouped mixer graph: inputs, two compositor slots, transitions, media wipe and output. Click for the full ungrouped graph. Two compositor slots draw every scene; a transition filter blends them and the diff --git a/demos/mixer/mixer.py b/demos/mixer/mixer.py index be059d75..b566a9e6 100644 --- a/demos/mixer/mixer.py +++ b/demos/mixer/mixer.py @@ -18,7 +18,8 @@ from avpmixer.dmabuf_inputs import (dmabuf_cuda_input_nodes, is_dmabuf_url, open_browser_windows, open_windows, refresh_windows, wait_for_sockets, window_id) from avpmixer.inputs import build_input -from avpmixer.janus import JANUS_KEYFRAME_NODE, JanusVideoConfig, build_janus_output +from avpmixer.janus import (DEFAULT_KEYFRAME_MIN_INTERVAL_MS, JANUS_KEYFRAME_NODE, + JanusVideoConfig, build_janus_output) try: from .layouts import ( @@ -69,6 +70,7 @@ class GraphOptions: janus_video_pt: int = JANUS_DEFAULT_VIDEO_PT janus_video_ssrc: int = JANUS_DEFAULT_VIDEO_SSRC janus_video_bitrate_kbps: int = JANUS_DEFAULT_VIDEO_BITRATE_KBPS + keyframe_min_interval_ms: int = DEFAULT_KEYFRAME_MIN_INTERVAL_MS janus_rtcp_bind: str = "0.0.0.0" janus_rtcp_port: int = 0 preheat_timeout_sec: float = 60.0 @@ -107,6 +109,9 @@ def validate(self) -> None: raise ValueError("janus_rtcp_port must be between 0 and 65535") if self.preheat_timeout_sec <= 0: raise ValueError("preheat_timeout_sec must be positive") + if (type(self.keyframe_min_interval_ms) is not int + or not 0 <= self.keyframe_min_interval_ms <= 2_147_483_647): + raise ValueError("keyframe_min_interval_ms must be a non-negative integer") if any(v <= 0 for v in self.dmabuf_size): raise ValueError("--dmabuf-size must be WxH with positive numbers") ids = self.dmabuf_inputs @@ -502,6 +507,7 @@ def _build_outputs(avp, api, options: GraphOptions, mixer_edge: str, *, host=options.janus_host, video_port=options.janus_video_port, payload_type=options.janus_video_pt, ssrc=options.janus_video_ssrc, bitrate_kbps=options.janus_video_bitrate_kbps, + keyframe_min_interval_ms=options.keyframe_min_interval_ms, rtcp_bind=options.janus_rtcp_bind, rtcp_port=options.janus_rtcp_port, ), fps=options.fps, fps_den=FPS_DEN, width=width, height=height, @@ -609,6 +615,7 @@ def _build_renditions(avp, api, options: GraphOptions, cfg, mixer_edge: str): video_port=rendition.port or options.janus_video_port, payload_type=options.janus_video_pt, ssrc=options.janus_video_ssrc, bitrate_kbps=rendition.bitrate_kbps, + keyframe_min_interval_ms=options.keyframe_min_interval_ms, rtcp_bind=options.janus_rtcp_bind, rtcp_port=options.janus_rtcp_port, ), fps=rendition.fps, fps_den=FPS_DEN, width=rendition.width, height=rendition.height, @@ -755,6 +762,10 @@ def parse_args(argv: list[str] | None = None) -> GraphOptions: default=JANUS_DEFAULT_VIDEO_BITRATE_KBPS, ) parser.add_argument("--janus-rtcp-bind", default="0.0.0.0") + parser.add_argument("--keyframe-min-interval-ms", type=int, + default=DEFAULT_KEYFRAME_MIN_INTERVAL_MS, + help="Minimum forced-keyframe spacing for Janus output in media time " + "(default: 150 ms; 0 disables rate limiting)") parser.add_argument("--janus-rtcp-port", type=int, default=0) parser.add_argument("--preheat-timeout", type=float, default=60.0) parser.add_argument("--wipe-file", help="Alpha wipe clip to warm the media-wipe chain up with at start " @@ -787,6 +798,7 @@ def parse_args(argv: list[str] | None = None) -> GraphOptions: janus_video_pt=args.janus_video_pt, janus_video_ssrc=args.janus_video_ssrc, janus_video_bitrate_kbps=args.janus_video_bitrate_kbps, + keyframe_min_interval_ms=args.keyframe_min_interval_ms, janus_rtcp_bind=args.janus_rtcp_bind, janus_rtcp_port=args.janus_rtcp_port, preheat_timeout_sec=args.preheat_timeout, diff --git a/demos/mixer/tests/test_graph.py b/demos/mixer/tests/test_graph.py index 9d38aff9..3c424788 100644 --- a/demos/mixer/tests/test_graph.py +++ b/demos/mixer/tests/test_graph.py @@ -568,14 +568,52 @@ def test_cli_requires_inputs_or_config(): def test_transitions_trigger_a_keyframe_only_when_streaming(): FakeMixer.instances.clear() - build_application(GraphOptions(inputs=("a.mp4",), janus_output=True), api=fake_api()) + application = build_application(GraphOptions(inputs=("a.mp4",), janus_output=True), api=fake_api()) assert FakeMixer.instances[-1].parameters["keyframe_node"] == "janus_force_keyframe" + node = next(n for n in application.avp.nodes if n.parameters.get("name") == "janus_force_keyframe") + assert node.parameters["min_interval_ms"] == 150 + assert node.parameters["interval_sec"] == "1/1" FakeMixer.instances.clear() build_application(GraphOptions(inputs=("a.mp4",), output="p.mp4"), api=fake_api()) assert FakeMixer.instances[-1].parameters["keyframe_node"] is None +@pytest.mark.parametrize("minimum", [0, 100, 150, 200, 500]) +def test_janus_keyframe_limit_is_configurable(minimum): + from avpmixer.janus import JanusVideoConfig, build_janus_output + avp = FakeAvp() + build_janus_output(avp, fake_api(), "program", JanusVideoConfig(keyframe_min_interval_ms=minimum), + fps=60, width=1920, height=1080) + node = next(n for n in avp.nodes if n.parameters.get("name") == "janus_force_keyframe") + assert node.parameters["min_interval_ms"] == minimum + + +@pytest.mark.parametrize("minimum", [-1, 0.2, True, "200", 2**31]) +def test_janus_rejects_invalid_keyframe_limit(minimum): + from avpmixer.janus import JanusVideoConfig + with pytest.raises(ValueError, match="keyframe_min_interval_ms"): + JanusVideoConfig(keyframe_min_interval_ms=minimum) + + +@pytest.mark.parametrize("configured", [False, True]) +def test_mixer_keyframe_option_reaches_each_output_path(tmp_path, configured): + if configured: + path = tmp_path / "mixer.json" + path.write_text(json.dumps({ + **CONFIG, "sources": CONFIG["sources"][:1], "scenes": CONFIG["scenes"][:1], + "initial_scene": "full", "wipes": [], + "renditions": [{"id": "program", "target": "janus"}], + })) + options = GraphOptions(config=str(path), janus_output=True, keyframe_min_interval_ms=200) + else: + options = parse_args(["--input", "a.mp4", "--janus-output", "--keyframe-min-interval-ms", "200"]) + application = build_application(options, api=fake_api()) + node = next(n for n in application.avp.nodes if n.parameters.get("name") == "janus_force_keyframe") + assert node.parameters["min_interval_ms"] == 200 + assert parse_args(["--input", "a.mp4", "--janus-output"]).keyframe_min_interval_ms == 150 + + def test_wipe_dir_scans_a_library_and_explicit_entries_win(tmp_path): from avpmixer import config as mc for name in ("b_swoosh.mov", "a_dip.webm", "notes.txt", "c_star.mp4"): diff --git a/doc/NODES.md b/doc/NODES.md index da9e5791..6bfc408e 100644 --- a/doc/NODES.md +++ b/doc/NODES.md @@ -432,11 +432,24 @@ encoder option in FFmpeg, works with non-integer FPS. - `interval_sec` (int / float / string of rational) - optional, keyframe interval, in seconds. If omitted, periodic forcing is disabled and the node acts as a pass-through until triggered externally. +- `min_interval_ms` (non-negative int, default `0`) - minimum media-PTS + spacing between forced keyframes, shared by periodic and external requests. + `0` preserves unrestricted forcing. Requires valid PTS when enabled; a + backwards jump before the last forced PTS resets the limiter. The Janus + mixer output defaults to `150` (9 frames at 60 fps), configurable with + `--keyframe-min-interval-ms` or `JanusVideoConfig.keyframe_min_interval_ms`. + +While rate-limited, ordinary frames pass through without waiting. Requests +coalesce and remain pending until the first eligible frame, including requests +from RTCP PLI/FIR; these may wait up to the minimum interval plus frame rounding. +Periodic forcing cannot bypass the limit. This limits the node's forced frames, +not keyframes independently inserted by an encoder, and is not a wall-clock RTP +packet pacer. Runtime control (via `node.object.set` / `node.object.get`): - `node.object.set trigger true` — request one keyframe on the next - processed frame (one-shot, edge-triggered). Also accepted as key `force` + eligible frame (one-shot, edge-triggered). Also accepted as key `force` or `request`. - `node.object.get status` — returns a JSON object: ```json @@ -446,7 +459,8 @@ Runtime control (via `node.object.set` / `node.object.get`): "pending": false, "triggered_frames": 2, "periodic_frames": 60, - "interval_enabled": true + "interval_enabled": true, + "min_interval_ms": 150 } ``` - `node.object.get pending` — bool, `true` if a trigger has been diff --git a/src/nodes/force_keyframe.cpp b/src/nodes/force_keyframe.cpp index 97089ca5..f8f725e8 100644 --- a/src/nodes/force_keyframe.cpp +++ b/src/nodes/force_keyframe.cpp @@ -1,5 +1,6 @@ #include "node_common.hpp" #include +#include class ForceKeyFrame: public NodeSISO, public IInputsObjects, @@ -8,6 +9,8 @@ class ForceKeyFrame: public NodeSISO, av::Rational interval_sec_; bool interval_enabled_ = false; int64_t last_result_ = -(1L<<62); + int min_interval_ms_ = 0; + av::Timestamp last_forced_pts_; std::atomic requested_generation_{0}; std::atomic forced_generation_{0}; std::atomic triggered_frames_{0}; @@ -45,11 +48,11 @@ class ForceKeyFrame: public NodeSISO, return true; } - // Coalescing semantics: any number of triggers received between two frames - // results in *one* forced keyframe. forced_generation_ is set to the latest - // requested value, so callers polling getObject("pending") cannot distinguish + // Coalescing semantics: any number of triggers received before the next + // eligible frame results in *one* forced keyframe. forced_generation_ is set + // to the latest requested value, so callers polling getObject("pending") cannot distinguish // their individual trigger from triggers emitted by other clients in the - // same inter-frame window. This is intentional — emitting one keyframe per + // same rate-limit window. This is intentional — emitting one keyframe per // received trigger would let a misbehaving controller spike the bitrate // arbitrarily. bool shouldForceTriggered() { @@ -73,13 +76,32 @@ class ForceKeyFrame: public NodeSISO, return; } av::VideoFrame frm = *ptr; - const bool force_triggered = shouldForceTriggered(); - const bool force_periodic = shouldForcePeriodic(frm); + const auto pts = frm.pts(); + if (min_interval_ms_ > 0) { + if (!pts.isValid() || pts.timebase().getNumerator() <= 0 || + pts.timebase().getDenominator() <= 0) { + throw Error("force_keyframe: min_interval_ms requires valid frame PTS"); + } + // A seek/reconnect starts a new media timeline. Do not wait for the + // old timeline to catch up before allowing a recovery keyframe. + if (last_forced_pts_.isValid() && pts < last_forced_pts_) { + last_forced_pts_ = av::Timestamp(); + last_result_ = -(1L<<62); + } + } + const bool eligible = min_interval_ms_ == 0 || !last_forced_pts_.isValid() || + pts - last_forced_pts_ >= av::Timestamp(min_interval_ms_, av::Rational(1, 1000)); + // Do not acknowledge triggers or advance the periodic bucket while + // throttled: both remain due until one eligible frame satisfies them. + const bool force_triggered = eligible && shouldForceTriggered(); + const bool force_periodic = eligible && shouldForcePeriodic(frm); if (force_triggered || force_periodic) { frm.setPictureType(AV_PICTURE_TYPE_I); frm.setKeyFrame(true); + last_forced_pts_ = pts; } else { frm.setPictureType(AV_PICTURE_TYPE_NONE); + frm.setKeyFrame(false); } this->sink_->put(frm); this->source_->pop(); @@ -113,6 +135,7 @@ class ForceKeyFrame: public NodeSISO, status["triggered_frames"] = triggered_frames_.load(std::memory_order_relaxed); status["periodic_frames"] = periodic_frames_.load(std::memory_order_relaxed); status["interval_enabled"] = interval_enabled_; + status["min_interval_ms"] = min_interval_ms_; return status; } if (key == "pending") { @@ -128,10 +151,12 @@ class ForceKeyFrame: public NodeSISO, std::unique_ptr> &&source, std::unique_ptr> &&sink, bool interval_enabled, - const av::Rational interval_sec + const av::Rational interval_sec, + int min_interval_ms ): NodeSISO(std::move(source), std::move(sink)), interval_sec_(interval_sec), - interval_enabled_(interval_enabled) { + interval_enabled_(interval_enabled), + min_interval_ms_(min_interval_ms) { } static std::shared_ptr create(NodeCreationInfo &nci) { @@ -139,6 +164,15 @@ class ForceKeyFrame: public NodeSISO, const Parameters ¶ms = nci.params; bool interval_enabled = false; av::Rational interval_sec(0, 1); + int min_interval_ms = 0; + if (params.contains("min_interval_ms")) { + const auto& value = params["min_interval_ms"]; + if (!value.is_number_integer() || value.get() < 0 || + value.get() > std::numeric_limits::max()) { + throw Error("min_interval_ms must be a non-negative integer in milliseconds"); + } + min_interval_ms = value.get(); + } if (params.count("interval_sec") > 0) { interval_sec = parseInterval(params["interval_sec"]); if (interval_sec.getNumerator() <= 0 || interval_sec.getDenominator() <= 0) { @@ -150,9 +184,10 @@ class ForceKeyFrame: public NodeSISO, edges, params, interval_enabled, - interval_sec + interval_sec, + min_interval_ms ); } }; -DECLNODE(force_keyframe, ForceKeyFrame); +DECLNODE(force_keyframe, ForceKeyFrame) diff --git a/tests/cuda/smoke_keyframe_limit.py b/tests/cuda/smoke_keyframe_limit.py new file mode 100644 index 00000000..a6abfe7d --- /dev/null +++ b/tests/cuda/smoke_keyframe_limit.py @@ -0,0 +1,148 @@ +"""Isolated 16-input, 60-fps mixer/NVENC keyframe-spam regression. + +Uses the demo's actual Janus encoder graph but terminates at an encoded-packet +reader, never a live Janus mountpoint. CPU decoding is only the post-encode +pixel oracle. No CPU/GPU round trip is added to the mixer processing graph. +Run separate processes with --minimum-ms 0 and 150 for the A/B comparison. +""" + +import argparse +import json +from pathlib import Path +import statistics +import subprocess +import sys +import threading +import time + +sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "demos" / "mixer")) +from mixer import GraphOptions, build_application + + +def run(args): + stop = threading.Event() + app = build_application(GraphOptions( + inputs=tuple(args.inputs[i % 2] for i in range(args.source_count)), + fps=60, loop_inputs=True, janus_output=True, + janus_video_bitrate_kbps=2700, keyframe_min_interval_ms=args.minimum_ms, + remote_control_port=args.port, wipe_cache_mb=0, + prewarm_cut_scenes=("*",), + )) + # The owned test ends at the encoder; leave RTP/BSF lifecycle to its own + # tests, and never publish this graph to the demo's streaming mountpoint. + app.avp.executeCommandsFromString("node.delete janus_rtp_output\n" + "node.delete janus_mux\nnode.delete janus_repeat_headers") + app.rtcp_feedback_listener = None + errors, frames, cuts = [], [], [] + app.avp.on_exception = lambda *error: errors.append(error) + + def capture(packet): + if packet.size <= 0 or packet.pts.timestamp == -(1 << 63): + return + tb = packet.pts.timebase + frames.append({"at": time.monotonic(), "pts": round(packet.pts.timestamp * tb.num / tb.den * 60), + "key": bool(packet.flags & 1), "data": packet.data}) + + # A queue reader avoids Python callbacks from a native encoder thread + # during shutdown (shutdown holds the GIL while joining native workers). + encoded = app.avp.getEdge("janus_encoded", "Packet") + + def read_encoded(): + while not stop.is_set(): + packet = encoded.tryGet(100) + if packet is not None: + capture(packet) + + packet_reader = threading.Thread(target=read_encoded, daemon=True) + packet_reader.start() + + def wait_for(predicate, label, timeout=10): + deadline = time.monotonic() + timeout + while not predicate(): + assert not errors, errors + assert time.monotonic() < deadline, label + time.sleep(.005) + + try: + app.start() + app.avp.registerWithWebUI(args.webui, "keyframe-limit-smoke", "") + wait_for(lambda: len(frames) >= 60, "encoder startup") + state = app.avp.node("janus_force_keyframe").getObject("status") + assert state["min_interval_ms"] == args.minimum_ms, state + begin = time.monotonic() + for index in range(args.cuts): + target = 1 - index % 2 + cuts.append({"at": time.monotonic(), "target": target}) + app.mixer.cut(f"fullscreen_{target}") + time.sleep(.05) + # Let the final cut/keyframe request finish after spam stops. + time.sleep(.4) + assert not app.avp.node("janus_force_keyframe").getObject("pending") + cut_end = time.monotonic() + # Exercise interrupted crossfades, then a fully completed fade. + for index in range(12): + app.mixer.fade(f"fullscreen_{index % 2}", duration_sec=.25) + time.sleep(.075) + app.mixer.fade("fullscreen_0", duration_sec=.25) + time.sleep(.6) + end = time.monotonic() + state = app.avp.node("janus_force_keyframe").getObject("status") + assert not state["pending"], state + assert not errors, errors + captured = tuple(frames) + measured = [f for f in captured if begin <= f["at"] <= end] + pts = [f["pts"] for f in measured] + assert all(b - a == 1 for a, b in zip(pts, pts[1:])), "output frame gap or duplicate PTS" + keys = [f["pts"] for f in captured if f["key"]] + key_spacing = [b - a for a, b in zip(keys, keys[1:])] + assert len(key_spacing) >= 3, "insufficient keyframes" + assert all(g * 1000 >= args.minimum_ms * 60 for g in key_spacing), key_spacing + gaps = [(b["at"] - a["at"]) * 1000 for a, b in zip(measured, measured[1:])] + assert max(gaps) < 150, ("encoded output stalled", max(gaps)) + + # Red/blue solid input fixtures make each cut's actual first picture + # distinguishable. This also proves that cuts need not wait for an IDR. + decoded = subprocess.run([ + args.ffmpeg, "-v", "error", "-threads", "1", "-f", "h264", "-i", "pipe:0", + "-vf", "crop=2:2:540:960", "-pix_fmt", "gray", "-fps_mode", "passthrough", + "-f", "rawvideo", "pipe:1", + ], input=b"".join(f["data"] for f in captured), capture_output=True, check=True, timeout=30).stdout + assert len(decoded) == len(captured) * 4, (len(decoded), len(captured)) + for i, frame in enumerate(captured): + y = decoded[i * 4] + frame["source"] = 0 if 60 <= y <= 95 else 1 if 15 <= y <= 45 else None + latencies, non_key_cuts = [], 0 + for i, cut in enumerate(cuts): + deadline = cuts[i + 1]["at"] if i + 1 < len(cuts) else cut_end + matches = [f for f in captured if cut["at"] <= f["at"] < deadline + and f["source"] == cut["target"]] + assert matches, ("cut picture did not arrive before the next cut", i, cut) + latencies.append((matches[0]["at"] - cut["at"]) * 1000) + non_key_cuts += not matches[0]["key"] + if args.minimum_ms: + assert non_key_cuts > len(cuts) // 2, "cuts appear to wait for keyframes" + print("KEYFRAME_LIMIT_RESULT " + json.dumps({ + "minimum_ms": args.minimum_ms, "inputs": args.source_count, + "cuts": len(cuts), "cut_pictures_without_keyframe": non_key_cuts, + "cut_latency_ms": {"median": statistics.median(latencies), "max": max(latencies)}, + "frames": len(measured), "fps": (len(measured) - 1) / (measured[-1]["at"] - measured[0]["at"]), + "encoded_gap_ms": {"p99": sorted(gaps)[int(.99 * (len(gaps) - 1))], "max": max(gaps)}, + "keyframes": len(keys), "min_keyframe_spacing_frames": min(key_spacing), + "status": state, + }), flush=True) + finally: + app.stop() + stop.set() + packet_reader.join(timeout=2) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--inputs", nargs=2, required=True, metavar=("RED_H264", "BLUE_H264")) + parser.add_argument("--port", type=int, required=True, help="unused control port for this isolated graph") + parser.add_argument("--webui", required=True, help="existing WebUI backend") + parser.add_argument("--source-count", type=int, choices=(2, 16), default=16) + parser.add_argument("--minimum-ms", type=int, choices=(0, 100, 150, 200), default=150) + parser.add_argument("--cuts", type=int, default=80) + parser.add_argument("--ffmpeg", default="/usr/local/bin/ffmpeg") + run(parser.parse_args()) diff --git a/tests/test_force_keyframe_native.py b/tests/test_force_keyframe_native.py new file mode 100644 index 00000000..41f3a7c1 --- /dev/null +++ b/tests/test_force_keyframe_native.py @@ -0,0 +1,144 @@ +"""Native node tests; run with the built pyplumber extension on the test host.""" + +from contextlib import contextmanager +from fractions import Fraction +import subprocess + +import pytest + +pytest.importorskip("_avplumber") + +from pyplumber import AVPlumber +from pyplumber.node import DecVideo, Demux, ForceKeyFrame, Input + + +@pytest.fixture(scope="module") +def frames(tmp_path_factory): + def generate(fps="60/1", count=125): + path = tmp_path_factory.mktemp("keyframe-frames") / "frames.nut" + subprocess.run([ + "ffmpeg", "-v", "error", "-n", "-f", "lavfi", "-i", + f"testsrc2=size=64x64:rate={fps}", "-frames:v", str(count), + "-c:v", "rawvideo", "-f", "nut", str(path), + ], check=True, capture_output=True, timeout=15) + avp = AVPlumber() + errors = [] + avp.on_exception = lambda *error: errors.append(error) + for cls, params in ( + (Input, {"url": str(path), "dst": "packets"}), + (Demux, {"src": "packets", "routing": {"v:0": "video"}}), + (DecVideo, {"src": "video", "dst": "frames"}), + ): + avp.addNode(cls({"group": "fixture", **params})) + edge = avp.getEdge("frames", "VideoFrame") + result = [] + try: + avp.group("fixture").startNodes() + for _ in range(count): + frame = edge.get(5000) + assert not errors, errors + result.append(frame) + return result + finally: + avp.shutdown() + return generate + + +@contextmanager +def limiter(**params): + avp = AVPlumber() + errors = [] + avp.on_exception = lambda *error: errors.append(error) + try: + avp.addNode(ForceKeyFrame({ + "name": "limit", "src": "input", "dst": "output", **params, + }), early_create=True) + source = avp.getEdge("input", "VideoFrame") + output = avp.getEdge("output", "VideoFrame") + avp.node("limit").start() + + def step(frame, requests=0): + for _ in range(requests): + avp.executeCommandsFromString("node.object.set limit trigger true") + source.enqueue(frame) + result = output.get(5000) + assert not errors, errors + assert result.pts.timestamp == frame.pts.timestamp + assert result.data == frame.data + return result.keyFrame + + yield step, lambda: avp.node("limit").getObject("status") + finally: + avp.shutdown() + + +@pytest.mark.parametrize("fps,minimum,spacing", [ + ("60/1", 100, 6), ("60/1", 150, 9), ("60/1", 200, 12), + ("60000/1001", 150, 9), ("60000/1001", 200, 12), ("24000/1001", 200, 5), +]) +def test_spam_is_coalesced_without_losing_frames(frames, fps, minimum, spacing): + footage = frames(fps=fps) + with limiter(min_interval_ms=minimum, interval_sec="1/1") as (step, status): + keys = [i for i, frame in enumerate(footage) if step(frame, requests=5)] + assert keys == list(range(0, len(footage), spacing)) + state = status() + assert state["requested_generation"] == len(footage) * 5 + assert state["triggered_frames"] == len(keys) + assert state["min_interval_ms"] == minimum + assert all(Fraction(b - a, 1) / Fraction(fps) >= Fraction(minimum, 1000) + for a, b in zip(keys, keys[1:])) + + +def test_last_request_survives_the_cooldown_and_is_not_repeated(frames): + with limiter(min_interval_ms=200) as (step, status): + keys = [] + for i, frame in enumerate(frames(count=30)): + if step(frame, requests=1 if i in (0, 1) else 0): + keys.append(i) + if 1 <= i < 12: + assert status()["pending"] + assert status()["forced_generation"] == 1 + if i >= 12: + assert not status()["pending"] + assert keys == [0, 12] + assert status()["forced_generation"] == 2 + + +def test_periodic_keyframe_cannot_bypass_a_recent_trigger(frames): + with limiter(min_interval_ms=200, interval_sec="1/1") as (step, status): + keys = [i for i, frame in enumerate(frames()) if step(frame, requests=int(i == 59))] + assert keys == [0, 59, 71, 120] + assert status()["periodic_frames"] == 3 + assert status()["triggered_frames"] == 1 + + +@pytest.mark.parametrize("params", [{}, {"min_interval_ms": 0}]) +def test_default_preserves_unlimited_triggering(frames, params): + with limiter(**params) as (step, status): + assert all(step(frame, requests=1) for frame in frames(count=15)) + assert status()["triggered_frames"] == 15 + assert not status()["pending"] + + +def test_periodic_forcing_can_be_disabled(frames): + with limiter(min_interval_ms=200) as (step, status): + assert not any(step(frame) for frame in frames()) + assert status()["periodic_frames"] == 0 + + +def test_backwards_pts_restarts_the_limit_and_duplicate_pts_do_not(frames): + footage = frames(count=30) + with limiter(min_interval_ms=200) as (step, _): + assert step(footage[24], requests=1) + assert not step(footage[24], requests=1) + assert step(footage[0]) # Pending request survives the timeline reset. + assert not step(footage[0], requests=1) + assert not step(footage[11]) + assert step(footage[12]) + + +@pytest.mark.parametrize("value", [-1, -0.1, 0.2, True, "200", 2**31]) +def test_invalid_interval_is_rejected(value): + with pytest.raises(RuntimeError, match="min_interval_ms"): + with limiter(min_interval_ms=value): + pytest.fail("invalid interval was accepted") From 5f7d9569bd3123a28482291557d3598ad13b67b9 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Wed, 16 Sep 2026 20:02:29 +0200 Subject: [PATCH 10/11] deps/ffmpeg: one patch series for FFmpeg 8.0 and 8.1; drop 7.1.5 Replace the per-version directories with a single deps/ffmpeg/8/ series that applies to both upstream n8.0 and n8.1. The composition-suite sources are new files (no base context), and the remaining patches are regenerated with minimal context; the only real 8.0/8.1 difference is the libnpp configure check 8.1 added that fails on CUDA 13, which the shared apply.sh rewrites to the stream-context probe (a no-op on 8.0). verify.sh checks either base against a pinned tree in 8/bases.env. Remove the 7.1.5 series and the triplicated per-version verify/README/ base.env; the three demo Dockerfiles call apply.sh and default to n8.1. Net: -5.1k lines of duplicated patch text, one README instead of three. Co-Authored-By: Claude Fable 5.1 --- demos/cuda-overlay/Dockerfile | 4 +- demos/dmabuf-browser/consumer/Dockerfile.cuda | 4 +- demos/mixer/Dockerfile | 4 +- .../0001-swscale-aarch64-argb-yuva420p.patch | 451 -- ...0002-avfilter-cuda-composition-suite.patch | 3928 ----------------- .../7.1.5/0004-avcodec-nvdec-intra.patch | 30 - .../7.1.5/0006-avdevice-v4l2-compat.patch | 25 - deps/ffmpeg/7.1.5/0007-avdevice-ndi-v5.patch | 222 - deps/ffmpeg/7.1.5/README.md | 56 - deps/ffmpeg/7.1.5/base.env | 3 - deps/ffmpeg/7.1.5/verify.sh | 4 - .../8.1/0003-avfilter-npp-cuda13-compat.patch | 322 -- .../8.1/0005-avformat-rtp-rfc4175.patch | 168 - deps/ffmpeg/8.1/README.md | 159 - deps/ffmpeg/8.1/base.env | 3 - deps/ffmpeg/8.1/verify.sh | 4 - .../0001-swscale-aarch64-argb-yuva420p.patch | 0 ...0002-avfilter-cuda-composition-suite.patch | 0 .../0003-avfilter-npp-cuda13-compat.patch | 100 +- .../{8.1 => 8}/0004-avcodec-nvdec-intra.patch | 0 .../0005-avformat-rtp-rfc4175.patch | 68 +- .../0006-avdevice-v4l2-compat.patch | 0 .../{8.1 => 8}/0007-avdevice-ndi-v5.patch | 75 +- deps/ffmpeg/8/bases.env | 5 + deps/ffmpeg/README.md | 88 +- deps/ffmpeg/apply.sh | 19 + deps/ffmpeg/verify.sh | 67 +- 27 files changed, 169 insertions(+), 5640 deletions(-) delete mode 100644 deps/ffmpeg/7.1.5/0001-swscale-aarch64-argb-yuva420p.patch delete mode 100644 deps/ffmpeg/7.1.5/0002-avfilter-cuda-composition-suite.patch delete mode 100644 deps/ffmpeg/7.1.5/0004-avcodec-nvdec-intra.patch delete mode 100644 deps/ffmpeg/7.1.5/0006-avdevice-v4l2-compat.patch delete mode 100644 deps/ffmpeg/7.1.5/0007-avdevice-ndi-v5.patch delete mode 100644 deps/ffmpeg/7.1.5/README.md delete mode 100644 deps/ffmpeg/7.1.5/base.env delete mode 100755 deps/ffmpeg/7.1.5/verify.sh delete mode 100644 deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch delete mode 100644 deps/ffmpeg/8.1/0005-avformat-rtp-rfc4175.patch delete mode 100644 deps/ffmpeg/8.1/README.md delete mode 100644 deps/ffmpeg/8.1/base.env delete mode 100755 deps/ffmpeg/8.1/verify.sh rename deps/ffmpeg/{8.1 => 8}/0001-swscale-aarch64-argb-yuva420p.patch (100%) rename deps/ffmpeg/{8.1 => 8}/0002-avfilter-cuda-composition-suite.patch (100%) rename deps/ffmpeg/{7.1.5 => 8}/0003-avfilter-npp-cuda13-compat.patch (72%) rename deps/ffmpeg/{8.1 => 8}/0004-avcodec-nvdec-intra.patch (100%) rename deps/ffmpeg/{7.1.5 => 8}/0005-avformat-rtp-rfc4175.patch (65%) rename deps/ffmpeg/{8.1 => 8}/0006-avdevice-v4l2-compat.patch (100%) rename deps/ffmpeg/{8.1 => 8}/0007-avdevice-ndi-v5.patch (65%) create mode 100644 deps/ffmpeg/8/bases.env create mode 100755 deps/ffmpeg/apply.sh diff --git a/demos/cuda-overlay/Dockerfile b/demos/cuda-overlay/Dockerfile index 4c882a55..8c3aa0ea 100644 --- a/demos/cuda-overlay/Dockerfile +++ b/demos/cuda-overlay/Dockerfile @@ -2,7 +2,7 @@ FROM nvidia/cuda:11.7.1-devel-ubuntu22.04 ARG DEBIAN_FRONTEND=noninteractive ARG AVPLUMBER_REVISION=workspace -ARG FFMPEG_TAG=n7.1.5 +ARG FFMPEG_TAG=n8.1 ARG NV_CODEC_HEADERS_TAG=n12.1.14.0 RUN apt-get update \ @@ -41,7 +41,7 @@ RUN git clone --quiet --branch "${FFMPEG_TAG}" --depth 1 \ https://github.com/FFmpeg/FFmpeg.git /tmp/ffmpeg \ && git -C /tmp/ffmpeg config user.name "cuda-overlay-demo builder" \ && git -C /tmp/ffmpeg config user.email "cuda-overlay-demo@local" \ - && git -C /tmp/ffmpeg am /build/deps/ffmpeg/${FFMPEG_TAG#n}/*.patch + && bash /build/deps/ffmpeg/apply.sh /tmp/ffmpeg RUN cd /tmp/ffmpeg \ && ./configure \ diff --git a/demos/dmabuf-browser/consumer/Dockerfile.cuda b/demos/dmabuf-browser/consumer/Dockerfile.cuda index 32e46b09..0d0036ae 100644 --- a/demos/dmabuf-browser/consumer/Dockerfile.cuda +++ b/demos/dmabuf-browser/consumer/Dockerfile.cuda @@ -16,7 +16,7 @@ ARG UBUNTU_VERSION=22.04 FROM nvidia/cuda:${CUDA_IMAGE_VERSION}-devel-ubuntu${UBUNTU_VERSION} AS builder ARG DEBIAN_FRONTEND=noninteractive -ARG FFMPEG_TAG=n7.1.5 +ARG FFMPEG_TAG=n8.1 ARG NV_CODEC_HEADERS_TAG=n12.1.14.0 ARG CUDA_NVCC_FLAGS="-gencode arch=compute_70,code=compute_70 -O2" @@ -56,7 +56,7 @@ RUN git clone --quiet --branch "${FFMPEG_TAG}" --depth 1 \ https://github.com/FFmpeg/FFmpeg.git /tmp/ffmpeg \ && git -C /tmp/ffmpeg config user.name "dmabuf-browser-demo builder" \ && git -C /tmp/ffmpeg config user.email "dmabuf-browser-demo@local" \ - && git -C /tmp/ffmpeg am /build/deps/ffmpeg/${FFMPEG_TAG#n}/*.patch + && bash /build/deps/ffmpeg/apply.sh /tmp/ffmpeg RUN cd /tmp/ffmpeg \ && ./configure \ diff --git a/demos/mixer/Dockerfile b/demos/mixer/Dockerfile index 78578bbe..69bf8341 100644 --- a/demos/mixer/Dockerfile +++ b/demos/mixer/Dockerfile @@ -2,7 +2,7 @@ FROM nvidia/cuda:11.7.1-devel-ubuntu22.04 ARG DEBIAN_FRONTEND=noninteractive ARG AVPLUMBER_REVISION=workspace -ARG FFMPEG_TAG=n7.1.5 +ARG FFMPEG_TAG=n8.1 ARG NV_CODEC_HEADERS_TAG=n12.1.14.0 RUN apt-get update \ @@ -42,7 +42,7 @@ RUN git clone --quiet --branch "${FFMPEG_TAG}" --depth 1 \ https://github.com/FFmpeg/FFmpeg.git /tmp/ffmpeg \ && git -C /tmp/ffmpeg config user.name "mixer-demo builder" \ && git -C /tmp/ffmpeg config user.email "mixer-demo@local" \ - && git -C /tmp/ffmpeg am /build/deps/ffmpeg/${FFMPEG_TAG#n}/*.patch + && bash /build/deps/ffmpeg/apply.sh /tmp/ffmpeg RUN cd /tmp/ffmpeg \ && ./configure \ diff --git a/deps/ffmpeg/7.1.5/0001-swscale-aarch64-argb-yuva420p.patch b/deps/ffmpeg/7.1.5/0001-swscale-aarch64-argb-yuva420p.patch deleted file mode 100644 index d5a9aa69..00000000 --- a/deps/ffmpeg/7.1.5/0001-swscale-aarch64-argb-yuva420p.patch +++ /dev/null @@ -1,451 +0,0 @@ -From 127f4c991739e9af8cdc5ac1315511c5e1bd85ef Mon Sep 17 00:00:00 2001 -From: Teodor Wozniak -Date: Wed, 12 Aug 2026 15:53:23 +0000 -Subject: [PATCH 1/7] swscale/aarch64: add fast ARGB to YUVA420P conversion - -Add an unscaled NEON conversion path for ARGB input to YUVA420P output. - -Original-commit: 3708bfab707a6eeb151829fd5c92807470c0f6d8 ---- - libswscale/aarch64/input.S | 36 ++++ - libswscale/aarch64/swscale.c | 6 + - libswscale/aarch64/swscale_unscaled.c | 125 +++++++++++++ - libswscale/aarch64/swscale_unscaled_neon.S | 200 +++++++++++++++++++++ - 4 files changed, 367 insertions(+) - -diff --git a/libswscale/aarch64/input.S b/libswscale/aarch64/input.S -index 5cb1871..bab2700 100644 ---- a/libswscale/aarch64/input.S -+++ b/libswscale/aarch64/input.S -@@ -313,3 +313,39 @@ rgbToUV_neon bgr24, rgb24, element=3 - rgbToUV_neon bgra32, rgba32, element=4 - - rgbToUV_neon abgr32, argb32, element=4, alpha_first=1 -+ -+// void ff_argb32ToA_neon(uint8_t *dst, const uint8_t *src, const uint8_t *unused1, -+// const uint8_t *unused2, int width, uint32_t *unused, void *opq) -+// dst stores one int16_t per pixel with the same scaling as abgrToA_c: -+// dst = (A << 6) | (A >> 2) -+function ff_argb32ToA_neon, export=1 -+ cmp w4, #0 -+ b.le 3f -+ -+ cmp w4, #16 -+ b.lt 2f -+1: -+ ld4 {v16.16b, v17.16b, v18.16b, v19.16b}, [x1], #64 -+ uxtl v20.8h, v16.8b -+ uxtl2 v21.8h, v16.16b -+ shl v22.8h, v20.8h, #6 -+ shl v23.8h, v21.8h, #6 -+ ushr v20.8h, v20.8h, #2 -+ ushr v21.8h, v21.8h, #2 -+ orr v22.16b, v22.16b, v20.16b -+ orr v23.16b, v23.16b, v21.16b -+ stp q22, q23, [x0], #32 -+ subs w4, w4, #16 -+ b.gt 1b -+ cbz w4, 3f -+2: -+ ldrb w8, [x1], #4 -+ subs w4, w4, #1 -+ lsl w9, w8, #6 -+ lsr w10, w8, #2 -+ orr w9, w9, w10 -+ strh w9, [x0], #2 -+ cbnz w4, 2b -+3: -+ ret -+endfunc -diff --git a/libswscale/aarch64/swscale.c b/libswscale/aarch64/swscale.c -index eb90728..a2abfe5 100644 ---- a/libswscale/aarch64/swscale.c -+++ b/libswscale/aarch64/swscale.c -@@ -217,6 +217,8 @@ NEON_INPUT(bgr24); - NEON_INPUT(bgra32); - NEON_INPUT(rgb24); - NEON_INPUT(rgba32); -+void ff_argb32ToA_neon(uint8_t *dst, const uint8_t *src, const uint8_t *, -+ const uint8_t *, int w, uint32_t *coeffs, void *opq); - - void ff_lumRangeFromJpeg_neon(int16_t *dst, int width); - void ff_chrRangeFromJpeg_neon(int16_t *dstU, int16_t *dstV, int width); -@@ -256,6 +258,8 @@ av_cold void ff_sws_init_swscale_aarch64(SwsContext *c) - c->chrToYV12 = ff_abgr32ToUV_half_neon; - else - c->chrToYV12 = ff_abgr32ToUV_neon; -+ if (c->needAlpha) -+ c->alpToYV12 = ff_argb32ToA_neon; - break; - - case AV_PIX_FMT_ARGB: -@@ -264,6 +268,8 @@ av_cold void ff_sws_init_swscale_aarch64(SwsContext *c) - c->chrToYV12 = ff_argb32ToUV_half_neon; - else - c->chrToYV12 = ff_argb32ToUV_neon; -+ if (c->needAlpha) -+ c->alpToYV12 = ff_argb32ToA_neon; - break; - case AV_PIX_FMT_BGR24: - c->lumToYV12 = ff_bgr24ToY_neon; -diff --git a/libswscale/aarch64/swscale_unscaled.c b/libswscale/aarch64/swscale_unscaled.c -index 9dfccc0..9626dd3 100644 ---- a/libswscale/aarch64/swscale_unscaled.c -+++ b/libswscale/aarch64/swscale_unscaled.c -@@ -21,6 +21,13 @@ - #include "libswscale/swscale_internal.h" - #include "libavutil/aarch64/cpu.h" - -+void ff_argb_to_yuv420p_neon(const uint8_t *src, uint8_t *ydst, -+ uint8_t *udst, uint8_t *vdst, int width, -+ int height, int lumStride, int chromStride, -+ int srcStride, int32_t *rgb2yuv); -+void ff_argb_to_a8_neon(const uint8_t *src, uint8_t *dst, int width, -+ int height, int srcStride, int dstStride); -+ - #define YUV_TO_RGB_TABLE \ - c->yuv2rgb_v2r_coeff, \ - c->yuv2rgb_u2g_coeff, \ -@@ -164,6 +171,115 @@ static int nv24_to_yuv420p_neon_wrapper(SwsContext *c, const uint8_t *src[], - return srcSliceH; - } - -+static av_always_inline void argb_to_yuva420p_c(const uint8_t *src, uint8_t *ydst, -+ uint8_t *udst, uint8_t *vdst, -+ uint8_t *adst, int width, -+ int height, int lumStride, -+ int chromStride, int srcStride, -+ int alphaStride, int32_t *rgb2yuv) -+{ -+ const int32_t ry = rgb2yuv[RY_IDX], gy = rgb2yuv[GY_IDX], by = rgb2yuv[BY_IDX]; -+ const int32_t ru = rgb2yuv[RU_IDX], gu = rgb2yuv[GU_IDX], bu = rgb2yuv[BU_IDX]; -+ const int32_t rv = rgb2yuv[RV_IDX], gv = rgb2yuv[GV_IDX], bv = rgb2yuv[BV_IDX]; -+ const int chromWidth = width >> 1; -+ const uint8_t *src1 = src; -+ const uint8_t *src2 = src + srcStride; -+ uint8_t *ydst1 = ydst; -+ uint8_t *ydst2 = ydst + lumStride; -+ uint8_t *adst1 = adst; -+ uint8_t *adst2 = adst + alphaStride; -+ -+ for (int y = 0; y < height; y += 2) { -+ if (y + 1 == height) { -+ ydst2 = ydst1; -+ adst2 = adst1; -+ src2 = src1; -+ } -+ -+ for (int i = 0; i < chromWidth; i++) { -+ const unsigned int a11 = src1[8 * i + 0]; -+ const unsigned int r11 = src1[8 * i + 1]; -+ const unsigned int g11 = src1[8 * i + 2]; -+ const unsigned int b11 = src1[8 * i + 3]; -+ const unsigned int a12 = src1[8 * i + 4]; -+ const unsigned int r12 = src1[8 * i + 5]; -+ const unsigned int g12 = src1[8 * i + 6]; -+ const unsigned int b12 = src1[8 * i + 7]; -+ const unsigned int a21 = src2[8 * i + 0]; -+ const unsigned int r21 = src2[8 * i + 1]; -+ const unsigned int g21 = src2[8 * i + 2]; -+ const unsigned int b21 = src2[8 * i + 3]; -+ const unsigned int a22 = src2[8 * i + 4]; -+ const unsigned int r22 = src2[8 * i + 5]; -+ const unsigned int g22 = src2[8 * i + 6]; -+ const unsigned int b22 = src2[8 * i + 7]; -+ -+ const unsigned int Y11 = ((ry * r11 + gy * g11 + by * b11) >> RGB2YUV_SHIFT) + 16; -+ const unsigned int Y12 = ((ry * r12 + gy * g12 + by * b12) >> RGB2YUV_SHIFT) + 16; -+ const unsigned int Y21 = ((ry * r21 + gy * g21 + by * b21) >> RGB2YUV_SHIFT) + 16; -+ const unsigned int Y22 = ((ry * r22 + gy * g22 + by * b22) >> RGB2YUV_SHIFT) + 16; -+ -+ const unsigned int bx = (b11 + b12 + b21 + b22) >> 2; -+ const unsigned int gx = (g11 + g12 + g21 + g22) >> 2; -+ const unsigned int rx = (r11 + r12 + r21 + r22) >> 2; -+ -+ const unsigned int U = ((ru * rx + gu * gx + bu * bx) >> RGB2YUV_SHIFT) + 128; -+ const unsigned int V = ((rv * rx + gv * gx + bv * bx) >> RGB2YUV_SHIFT) + 128; -+ -+ ydst1[2 * i + 0] = Y11; -+ ydst1[2 * i + 1] = Y12; -+ ydst2[2 * i + 0] = Y21; -+ ydst2[2 * i + 1] = Y22; -+ adst1[2 * i + 0] = a11; -+ adst1[2 * i + 1] = a12; -+ adst2[2 * i + 0] = a21; -+ adst2[2 * i + 1] = a22; -+ udst[i] = U; -+ vdst[i] = V; -+ } -+ -+ src1 += srcStride * 2; -+ src2 += srcStride * 2; -+ ydst1 += lumStride * 2; -+ ydst2 += lumStride * 2; -+ adst1 += alphaStride * 2; -+ adst2 += alphaStride * 2; -+ udst += chromStride; -+ vdst += chromStride; -+ } -+} -+ -+static int argb_to_yuva420p_neon_wrapper(SwsContext *c, const uint8_t *src[], -+ int srcStride[], int srcSliceY, int srcSliceH, -+ uint8_t *dst[], int dstStride[]) -+{ -+ const int width_aligned = c->srcW & ~15; -+ const uint8_t *src_ptr = src[0]; -+ uint8_t *ydst = dst[0] + srcSliceY * dstStride[0]; -+ uint8_t *udst = dst[1] + (srcSliceY >> 1) * dstStride[1]; -+ uint8_t *vdst = dst[2] + (srcSliceY >> 1) * dstStride[2]; -+ uint8_t *adst = dst[3] + srcSliceY * dstStride[3]; -+ -+ /* Keep strict gating in wrapper too, so odd slice boundaries or non-16 widths -+ * never hit the assembly fast path. */ -+ if ((srcSliceY & 1) || (srcSliceH & 1) || (c->srcW & 15)) { -+ argb_to_yuva420p_c(src_ptr, ydst, udst, vdst, adst, -+ c->srcW, srcSliceH, dstStride[0], dstStride[1], -+ srcStride[0], dstStride[3], c->input_rgb2yuv_table); -+ return srcSliceH; -+ } -+ -+ if (width_aligned > 0) { -+ ff_argb_to_yuv420p_neon(src_ptr, ydst, udst, vdst, width_aligned, -+ srcSliceH, dstStride[0], dstStride[1], -+ srcStride[0], c->input_rgb2yuv_table); -+ ff_argb_to_a8_neon(src_ptr, adst, width_aligned, srcSliceH, -+ srcStride[0], dstStride[3]); -+ } -+ -+ return srcSliceH; -+} -+ - #define DECLARE_FF_NVX_TO_ALL_RGBX_FUNCS(nvx) \ - DECLARE_FF_NVX_TO_RGBX_FUNCS(nvx, argb) \ - DECLARE_FF_NVX_TO_RGBX_FUNCS(nvx, rgba) \ -@@ -207,6 +323,15 @@ static void get_unscaled_swscale_neon(SwsContext *c) { - (c->srcFormat == AV_PIX_FMT_NV24 || c->srcFormat == AV_PIX_FMT_NV42) && - !(c->srcH & 1) && !(c->srcW & 15) && !accurate_rnd) - c->convert_unscaled = nv24_to_yuv420p_neon_wrapper; -+ -+ if (c->srcFormat == AV_PIX_FMT_ARGB && -+ c->dstFormat == AV_PIX_FMT_YUVA420P && -+ !(c->srcH & 1) && -+ !(c->srcW & 15) && -+ !(c->srcW & 1) && !accurate_rnd) { -+ c->convert_unscaled = argb_to_yuva420p_neon_wrapper; -+ c->dst_slice_align = 2; -+ } - } - - void ff_get_unscaled_swscale_aarch64(SwsContext *c) -diff --git a/libswscale/aarch64/swscale_unscaled_neon.S b/libswscale/aarch64/swscale_unscaled_neon.S -index a074b18..0c961ad 100644 ---- a/libswscale/aarch64/swscale_unscaled_neon.S -+++ b/libswscale/aarch64/swscale_unscaled_neon.S -@@ -20,6 +20,206 @@ - - #include "libavutil/aarch64/asm.S" - -+#define RGB2YUV_COEFFS 16*4+16*32 -+#define BY v0.h[0] -+#define GY v0.h[1] -+#define RY v0.h[2] -+#define BU v1.h[0] -+#define GU v1.h[1] -+#define RU v1.h[2] -+#define BV v2.h[0] -+#define GV v2.h[1] -+#define RV v2.h[2] -+#define Y_OFFSET v22 -+#define UV_OFFSET v23 -+ -+// convert rgb to 16-bit y, u, or v -+// uses v3 and v4 -+.macro rgbconv16 dst, b, g, r, bc, gc, rc -+ smull v3.4s, \b\().4h, \bc -+ smlal v3.4s, \g\().4h, \gc -+ smlal v3.4s, \r\().4h, \rc -+ smull2 v4.4s, \b\().8h, \bc -+ smlal2 v4.4s, \g\().8h, \gc -+ smlal2 v4.4s, \r\().8h, \rc -+ shrn \dst\().4h, v3.4s, #7 -+ shrn2 \dst\().8h, v4.4s, #7 -+.endm -+ -+// void ff_argb_to_yuv420p_neon(const uint8_t *src, uint8_t *ydst, uint8_t *udst, -+// uint8_t *vdst, int width, int height, int lumStride, -+// int chromStride, int srcStride, int32_t *rgb2yuv); -+function ff_argb_to_yuv420p_neon, export=1 -+// x0 const uint8_t *src -+// x1 uint8_t *ydst -+// x2 uint8_t *udst -+// x3 uint8_t *vdst -+// w4 int width -+// w5 int height -+// w6 int lumStride -+// w7 int chromStride -+ ldrsw x9, [sp] // srcStride -+ ldr x14, [sp, #8] // rgb2yuv -+ -+ // extend width and stride parameters -+ uxtw x4, w4 -+ sxtw x6, w6 -+ sxtw x7, w7 -+ -+ // src1 = x0 -+ // src2 = x10 -+ add x10, x0, x9 // x10 = src + srcStride -+ lsl x9, x9, #1 // srcStride *= 2 -+ lsl x11, x4, #2 // x11 = 4 * width -+ sub x9, x9, x11 // srcPadding = (2 * srcStride) - (4 * width) -+ -+ // ydst1 = x1 -+ // ydst2 = x11 -+ add x11, x1, x6 // x11 = ydst + lumStride -+ lsl x6, x6, #1 // lumStride *= 2 -+ sub x6, x6, x4 // lumPadding = (2 * lumStride) - width -+ -+ sub x7, x7, x4, lsr #1 // chromPadding = chromStride - (width / 2) -+ -+ // load rgb2yuv coefficients into v0, v1, and v2 -+ add x14, x14, #RGB2YUV_COEFFS -+ ld1 {v0.8h-v2.8h}, [x14] -+ -+ // load offset constants -+ movi Y_OFFSET.8h, #0x10, lsl #8 -+ movi UV_OFFSET.8h, #0x80, lsl #8 -+ -+1: -+ mov w14, w4 // w14 = width -+ -+2: -+ // load first line (ARGB) -+ ld4 {v26.16b, v27.16b, v28.16b, v29.16b}, [x0], #64 -+ -+ // widen first line to 16-bit (B,G,R) -+ uxtl v16.8h, v29.8b -+ uxtl v17.8h, v28.8b -+ uxtl v18.8h, v27.8b -+ uxtl2 v19.8h, v29.16b -+ uxtl2 v20.8h, v28.16b -+ uxtl2 v21.8h, v27.16b -+ -+ // calculate Y values for first line -+ rgbconv16 v24, v16, v17, v18, BY, GY, RY -+ rgbconv16 v25, v19, v20, v21, BY, GY, RY -+ -+ // load second line (ARGB) -+ ld4 {v26.16b, v27.16b, v28.16b, v29.16b}, [x10], #64 -+ -+ // pairwise add and save rgb values to calculate average -+ addp v5.8h, v16.8h, v19.8h -+ addp v6.8h, v17.8h, v20.8h -+ addp v7.8h, v18.8h, v21.8h -+ -+ // widen second line to 16-bit (B,G,R) -+ uxtl v16.8h, v29.8b -+ uxtl v17.8h, v28.8b -+ uxtl v18.8h, v27.8b -+ uxtl2 v19.8h, v29.16b -+ uxtl2 v20.8h, v28.16b -+ uxtl2 v21.8h, v27.16b -+ -+ // calculate Y values for second line -+ rgbconv16 v30, v16, v17, v18, BY, GY, RY -+ rgbconv16 v31, v19, v20, v21, BY, GY, RY -+ -+ // pairwise add rgb values to calculate average -+ addp v16.8h, v16.8h, v19.8h -+ addp v17.8h, v17.8h, v20.8h -+ addp v18.8h, v18.8h, v21.8h -+ -+ // calculate average for chroma -+ add v16.8h, v16.8h, v5.8h -+ add v17.8h, v17.8h, v6.8h -+ add v18.8h, v18.8h, v7.8h -+ ushr v16.8h, v16.8h, #2 -+ ushr v17.8h, v17.8h, #2 -+ ushr v18.8h, v18.8h, #2 -+ -+ // calculate U and V values -+ rgbconv16 v28, v16, v17, v18, BU, GU, RU -+ rgbconv16 v29, v16, v17, v18, BV, GV, RV -+ -+ // add offsets and narrow all values -+ addhn v24.8b, v24.8h, Y_OFFSET.8h -+ addhn v25.8b, v25.8h, Y_OFFSET.8h -+ addhn v30.8b, v30.8h, Y_OFFSET.8h -+ addhn v31.8b, v31.8h, Y_OFFSET.8h -+ addhn v28.8b, v28.8h, UV_OFFSET.8h -+ addhn v29.8b, v29.8h, UV_OFFSET.8h -+ -+ subs w14, w14, #16 -+ -+ // store output -+ st1 {v24.8b, v25.8b}, [x1], #16 -+ st1 {v30.8b, v31.8b}, [x11], #16 -+ st1 {v28.8b}, [x2], #8 -+ st1 {v29.8b}, [x3], #8 -+ -+ b.gt 2b -+ -+ subs w5, w5, #2 -+ -+ // row += 2 -+ add x0, x0, x9 -+ add x10, x10, x9 -+ add x1, x1, x6 -+ add x11, x11, x6 -+ add x2, x2, x7 -+ add x3, x3, x7 -+ b.gt 1b -+ -+ ret -+endfunc -+ -+// void ff_argb_to_a8_neon(const uint8_t *src, uint8_t *dst, int width, int height, -+// int srcStride, int dstStride); -+function ff_argb_to_a8_neon, export=1 -+// x0 const uint8_t *src -+// x1 uint8_t *dst -+// w2 int width -+// w3 int height -+// w4 int srcStride -+// w5 int dstStride -+ cmp w2, #0 -+ b.le 4f -+ cmp w3, #0 -+ b.le 4f -+ -+ sub w4, w4, w2, lsl #2 // srcPadding = srcStride - width * 4 -+ sub w5, w5, w2 // dstPadding = dstStride - width -+ -+1: -+ mov w6, w2 -+2: -+ cmp w6, #16 -+ b.lt 3f -+ ld4 {v0.16b, v1.16b, v2.16b, v3.16b}, [x0], #64 -+ st1 {v0.16b}, [x1], #16 -+ sub w6, w6, #16 -+ b 2b -+3: -+ cbz w6, 5f -+6: -+ ldrb w7, [x0], #4 -+ subs w6, w6, #1 -+ strb w7, [x1], #1 -+ b.gt 6b -+5: -+ subs w3, w3, #1 -+ b.eq 4f -+ add x0, x0, w4, sxtw -+ add x1, x1, w5, sxtw -+ b 1b -+4: -+ ret -+endfunc -+ - function ff_nv24_to_yuv420p_chroma_neon, export=1 - // x0 uint8_t *dst1 - // x1 int dstStride1 diff --git a/deps/ffmpeg/7.1.5/0002-avfilter-cuda-composition-suite.patch b/deps/ffmpeg/7.1.5/0002-avfilter-cuda-composition-suite.patch deleted file mode 100644 index ef437c10..00000000 --- a/deps/ffmpeg/7.1.5/0002-avfilter-cuda-composition-suite.patch +++ /dev/null @@ -1,3928 +0,0 @@ -From 7bc571c2fdfcc90943f7d4c98ed99822fe181c71 Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Wed, 12 Aug 2026 15:53:55 +0000 -Subject: [PATCH 2/7] avfilter: add CUDA composition filter suite - -Add and register the CUDA pad, convert, crop, overlay-many, and transition filters used by avplumber composition graphs. Include runtime overlay descriptors, procedural transitions, scale edge fixes, overlay context propagation, and CUDA architecture compatibility. - -This squashes the incomplete registration, missing-file, compatibility, and follow-up patches into the complete feature they implement. - -Original-commits: 3f0a41055cfc542662ce7ad68b131dbda2ed0bc4 1a54abb0cad9fd5992166ae23649bda9e9b14753 aa3aff11d4a032bda6ba2dffd6418969e66df953 534042a4c5b297fe7baca2adf9c231acb171ee12 de361ec1e6d9f8169f4ed8ebf0e9cbc469c57027 8a109d1d41400391ea3c3ac9ee6faeb925c402be 58a00475094535b234aa2a5b360fd8460b72fb86 ae1c26b083fbeb885b9850e1c54ec97ee9873d2a 2105ce91bf8f27d423d2ed55f49be0b4cf3557b4 0a384ecc48ba46ff8e5f15d960dd9bb652ea7b52 c3474bdec62f807fd4283669d9d5405f46da1390 5d0ac856ccf9738f6e84853172ea2656082f176c 1bbd5e268f460893dd900c2d7ba541f44446a350 b3c7f6584796fdd2d0dc439ba8cf8dc4746fb27a - -Co-authored-by: Teodor Wozniak ---- - configure | 5 +- - doc/filters.texi | 47 +++ - libavfilter/Makefile | 9 + - libavfilter/allfilters.c | 5 + - libavfilter/vf_convert_cuda.c | 511 +++++++++++++++++++++++ - libavfilter/vf_convert_cuda.cu | 273 ++++++++++++ - libavfilter/vf_crop_cuda.c | 561 +++++++++++++++++++++++++ - libavfilter/vf_overlay_cuda.c | 1 + - libavfilter/vf_overlay_many_cuda.c | 559 +++++++++++++++++++++++++ - libavfilter/vf_overlay_many_cuda.cu | 252 +++++++++++ - libavfilter/vf_overlay_many_cuda.h | 68 +++ - libavfilter/vf_pad_cuda.c | 538 ++++++++++++++++++++++++ - libavfilter/vf_pad_cuda.cu | 37 ++ - libavfilter/vf_scale_cuda.c | 5 + - libavfilter/vf_scale_cuda.cu | 104 ++++- - libavfilter/vf_transition_cuda.c | 621 ++++++++++++++++++++++++++++ - libavfilter/vf_transition_cuda.cu | 56 +++ - libavutil/hwcontext_cuda.c | 1 + - 18 files changed, 3636 insertions(+), 17 deletions(-) - create mode 100644 libavfilter/vf_convert_cuda.c - create mode 100644 libavfilter/vf_convert_cuda.cu - create mode 100644 libavfilter/vf_crop_cuda.c - create mode 100644 libavfilter/vf_overlay_many_cuda.c - create mode 100644 libavfilter/vf_overlay_many_cuda.cu - create mode 100644 libavfilter/vf_overlay_many_cuda.h - create mode 100644 libavfilter/vf_pad_cuda.c - create mode 100644 libavfilter/vf_pad_cuda.cu - create mode 100644 libavfilter/vf_transition_cuda.c - create mode 100644 libavfilter/vf_transition_cuda.cu - -diff --git a/configure b/configure -index 1652b0d..2cd7884 100755 ---- a/configure -+++ b/configure -@@ -3309,6 +3309,8 @@ chromakey_cuda_filter_deps="ffnvcodec" - chromakey_cuda_filter_deps_any="cuda_nvcc cuda_llvm" - colorspace_cuda_filter_deps="ffnvcodec" - colorspace_cuda_filter_deps_any="cuda_nvcc cuda_llvm" -+crop_cuda_filter_deps="ffnvcodec" -+crop_cuda_filter_deps_any="cuda_nvcc cuda_llvm" - hwupload_cuda_filter_deps="ffnvcodec" - scale_npp_filter_deps="ffnvcodec libnpp" - scale2ref_npp_filter_deps="ffnvcodec libnpp" -@@ -3319,6 +3321,7 @@ thumbnail_cuda_filter_deps_any="cuda_nvcc cuda_llvm" - transpose_npp_filter_deps="ffnvcodec libnpp" - overlay_cuda_filter_deps="ffnvcodec" - overlay_cuda_filter_deps_any="cuda_nvcc cuda_llvm" -+overlay_many_cuda_filter_deps="ffnvcodec cuda_nvcc" - sharpen_npp_filter_deps="ffnvcodec libnpp" - - ddagrab_filter_deps="d3d11va IDXGIOutput1 DXGI_OUTDUPL_FRAME_INFO" -@@ -4705,7 +4708,7 @@ set_default nvcc - - if enabled cuda_nvcc; then - if $nvcc $nvccflags_default 2>&1 | grep -qi unsupported; then -- nvccflags_default="-gencode arch=compute_60,code=sm_60 -O2" -+ nvccflags_default="-gencode arch=compute_75,code=sm_75 -O2" - fi - fi - -diff --git a/doc/filters.texi b/doc/filters.texi -index 428986a..a4ae86c 100644 ---- a/doc/filters.texi -+++ b/doc/filters.texi -@@ -19050,6 +19050,53 @@ See @ref{framesync}. - - This filter also supports the @ref{framesync} options. - -+@anchor{overlay_many_cuda} -+@section overlay_many_cuda -+ -+Overlay several CUDA video streams on top of a main CUDA video stream. -+ -+The first input is the main video. Every subsequent input is a full-frame -+overlay, applied in input order. The filter supports between 2 and 16 total -+inputs. Unavailable overlay frames are skipped independently; the main frame is -+passed through unchanged only when no overlay frame is available. -+ -+The supported software-format combinations inside the CUDA frames are: -+ -+@itemize -+@item -+@code{yuv420p} main with @code{yuva420p} overlays; -+@item -+@code{yuv420p} main with @code{yuva444p} overlays; -+@item -+@code{yuv444p} main with @code{yuva444p} overlays. -+@end itemize -+ -+All overlays must use the same software format and must be at least as large as -+the main input. Composition starts at the top-left corner. Processing remains -+on the CUDA device and uses the stream associated with the main input. -+ -+This filter requires CUDA Toolkit 11.7 or newer and a GPU with compute -+capability 7.0 or newer. -+ -+It accepts the following options: -+ -+@table @option -+@item inputs -+Set the total number of inputs, including the main input. The accepted range is -+2 to 16. The default is 2. -+ -+@item eof_action -+See @ref{framesync}. -+ -+@item shortest -+See @ref{framesync}. -+ -+@item repeatlast -+See @ref{framesync}. -+@end table -+ -+This filter also supports the @ref{framesync} options. -+ - @section owdenoise - - Apply Overcomplete Wavelet denoiser. -diff --git a/libavfilter/Makefile b/libavfilter/Makefile -index 91487af..c80bece 100644 ---- a/libavfilter/Makefile -+++ b/libavfilter/Makefile -@@ -247,6 +247,8 @@ OBJS-$(CONFIG_COLORSPACE_CUDA_FILTER) += vf_colorspace_cuda.o \ - vf_colorspace_cuda.ptx.o \ - cuda/load_helper.o - OBJS-$(CONFIG_COLORTEMPERATURE_FILTER) += vf_colortemperature.o -+OBJS-$(CONFIG_CONVERT_CUDA_FILTER) += vf_convert_cuda.o vf_convert_cuda.ptx.o \ -+ cuda/load_helper.o - OBJS-$(CONFIG_CONVOLUTION_FILTER) += vf_convolution.o - OBJS-$(CONFIG_CONVOLUTION_OPENCL_FILTER) += vf_convolution_opencl.o opencl.o \ - opencl/convolution.o -@@ -256,6 +258,7 @@ OBJS-$(CONFIG_COREIMAGE_FILTER) += vf_coreimage.o - OBJS-$(CONFIG_CORR_FILTER) += vf_corr.o framesync.o - OBJS-$(CONFIG_COVER_RECT_FILTER) += vf_cover_rect.o lavfutils.o - OBJS-$(CONFIG_CROP_FILTER) += vf_crop.o -+OBJS-$(CONFIG_CROP_CUDA_FILTER) += vf_crop_cuda.o - OBJS-$(CONFIG_CROPDETECT_FILTER) += vf_cropdetect.o edge_common.o - OBJS-$(CONFIG_CUE_FILTER) += f_cue.o - OBJS-$(CONFIG_CURVES_FILTER) += vf_curves.o -@@ -410,6 +413,9 @@ OBJS-$(CONFIG_OSCILLOSCOPE_FILTER) += vf_datascope.o - OBJS-$(CONFIG_OVERLAY_FILTER) += vf_overlay.o framesync.o - OBJS-$(CONFIG_OVERLAY_CUDA_FILTER) += vf_overlay_cuda.o framesync.o vf_overlay_cuda.ptx.o \ - cuda/load_helper.o -+OBJS-$(CONFIG_OVERLAY_MANY_CUDA_FILTER) += vf_overlay_many_cuda.o framesync.o \ -+ vf_overlay_many_cuda.ptx.o cuda/load_helper.o -+ - OBJS-$(CONFIG_OVERLAY_OPENCL_FILTER) += vf_overlay_opencl.o opencl.o \ - opencl/overlay.o framesync.o - OBJS-$(CONFIG_OVERLAY_QSV_FILTER) += vf_overlay_qsv.o framesync.o -@@ -417,6 +423,7 @@ OBJS-$(CONFIG_OVERLAY_VAAPI_FILTER) += vf_overlay_vaapi.o framesync.o v - OBJS-$(CONFIG_OVERLAY_VULKAN_FILTER) += vf_overlay_vulkan.o vulkan.o vulkan_filter.o - OBJS-$(CONFIG_OWDENOISE_FILTER) += vf_owdenoise.o - OBJS-$(CONFIG_PAD_FILTER) += vf_pad.o -+OBJS-$(CONFIG_PAD_CUDA_FILTER) += vf_pad_cuda.o vf_pad_cuda.ptx.o cuda/load_helper.o - OBJS-$(CONFIG_PAD_OPENCL_FILTER) += vf_pad_opencl.o opencl.o opencl/pad.o - OBJS-$(CONFIG_PALETTEGEN_FILTER) += vf_palettegen.o palette.o - OBJS-$(CONFIG_PALETTEUSE_FILTER) += vf_paletteuse.o framesync.o palette.o -@@ -528,6 +535,8 @@ OBJS-$(CONFIG_TONEMAP_OPENCL_FILTER) += vf_tonemap_opencl.o opencl.o \ - opencl/tonemap.o opencl/colorspace_common.o - OBJS-$(CONFIG_TONEMAP_VAAPI_FILTER) += vf_tonemap_vaapi.o vaapi_vpp.o - OBJS-$(CONFIG_TPAD_FILTER) += vf_tpad.o -+OBJS-$(CONFIG_TRANSITION_CUDA_FILTER) += vf_transition_cuda.o framesync.o vf_transition_cuda.ptx.o \ -+ cuda/load_helper.o - OBJS-$(CONFIG_TRANSPOSE_FILTER) += vf_transpose.o - OBJS-$(CONFIG_TRANSPOSE_NPP_FILTER) += vf_transpose_npp.o - OBJS-$(CONFIG_TRANSPOSE_OPENCL_FILTER) += vf_transpose_opencl.o opencl.o opencl/transpose.o -diff --git a/libavfilter/allfilters.c b/libavfilter/allfilters.c -index 9819f0f..640b3f8 100644 ---- a/libavfilter/allfilters.c -+++ b/libavfilter/allfilters.c -@@ -226,6 +226,7 @@ extern const AVFilter ff_vf_colormatrix; - extern const AVFilter ff_vf_colorspace; - extern const AVFilter ff_vf_colorspace_cuda; - extern const AVFilter ff_vf_colortemperature; -+extern const AVFilter ff_vf_convert_cuda; - extern const AVFilter ff_vf_convolution; - extern const AVFilter ff_vf_convolution_opencl; - extern const AVFilter ff_vf_convolve; -@@ -234,6 +235,7 @@ extern const AVFilter ff_vf_coreimage; - extern const AVFilter ff_vf_corr; - extern const AVFilter ff_vf_cover_rect; - extern const AVFilter ff_vf_crop; -+extern const AVFilter ff_vf_crop_cuda; - extern const AVFilter ff_vf_cropdetect; - extern const AVFilter ff_vf_cue; - extern const AVFilter ff_vf_curves; -@@ -390,8 +392,10 @@ extern const AVFilter ff_vf_overlay_qsv; - extern const AVFilter ff_vf_overlay_vaapi; - extern const AVFilter ff_vf_overlay_vulkan; - extern const AVFilter ff_vf_overlay_cuda; -+extern const AVFilter ff_vf_overlay_many_cuda; - extern const AVFilter ff_vf_owdenoise; - extern const AVFilter ff_vf_pad; -+extern const AVFilter ff_vf_pad_cuda; - extern const AVFilter ff_vf_pad_opencl; - extern const AVFilter ff_vf_palettegen; - extern const AVFilter ff_vf_paletteuse; -@@ -497,6 +501,7 @@ extern const AVFilter ff_vf_tonemap; - extern const AVFilter ff_vf_tonemap_opencl; - extern const AVFilter ff_vf_tonemap_vaapi; - extern const AVFilter ff_vf_tpad; -+extern const AVFilter ff_vf_transition_cuda; - extern const AVFilter ff_vf_transpose; - extern const AVFilter ff_vf_transpose_npp; - extern const AVFilter ff_vf_transpose_opencl; -diff --git a/libavfilter/vf_convert_cuda.c b/libavfilter/vf_convert_cuda.c -new file mode 100644 -index 0000000..12f9299 ---- /dev/null -+++ b/libavfilter/vf_convert_cuda.c -@@ -0,0 +1,511 @@ -+#include "libavutil/hwcontext.h" -+#include "libavutil/hwcontext_cuda_internal.h" -+#include "libavutil/cuda_check.h" -+#include "libavutil/opt.h" -+#include "libavutil/pixdesc.h" -+#include "libavutil/eval.h" -+#include "libavutil/colorspace.h" -+ -+#include "avfilter.h" -+#include "formats.h" -+#include "avfilter_internal.h" -+#include "video.h" -+ -+#include "cuda/load_helper.h" -+ -+static enum AVPixelFormat supported_in_formats[] = { -+ AV_PIX_FMT_YUVA444P12LE, AV_PIX_FMT_YUVA444P, -+ AV_PIX_FMT_ARGB, AV_PIX_FMT_RGBA, AV_PIX_FMT_ABGR, AV_PIX_FMT_BGRA, -+ AV_PIX_FMT_NV12, AV_PIX_FMT_YUV420P, -+}; -+ -+static enum AVPixelFormat supported_out_formats[] = { -+ AV_PIX_FMT_YUVA420P, AV_PIX_FMT_YUVA444P, -+ AV_PIX_FMT_NV12, AV_PIX_FMT_YUV420P, -+}; -+ -+#define BLOCK_X 32 -+#define BLOCK_Y 16 -+ -+#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) ) -+#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, cu, x) -+#define ARRAY_COUNT(x) (sizeof(x)/sizeof(*(x))) -+ -+typedef struct ConvertCUDAContext { -+ const AVClass *class; -+ AVCUDADeviceContext *hwctx; -+ AVBufferRef *hw_device_ctx; -+ -+ AVBufferRef *frames_ctx; -+ AVFrame *frame; -+ AVFrame *tmp_frame; -+ -+ CUmodule cu_module; -+ // cu_functions is an array where we load handles to CUDA shader functions. -+ // For some in/out format combinations we do the whole convert with 1 shader. -+ // For others we need to use 2 separate shaders (mostly for 444p... -> 420p). -+ // This is organized in this way to minimize the final number of shaders needed by this filter. -+ CUfunction cu_functions[2]; -+ CUstream cu_stream; -+ -+ char *format_str; -+ -+ enum AVPixelFormat format_in; -+ enum AVPixelFormat format_out; -+} ConvertCUDAContext; -+ -+ -+ -+static int format_input_is_supported(enum AVPixelFormat fmt) -+{ -+ for (int i = 0; i < FF_ARRAY_ELEMS(supported_in_formats); i++) -+ if (supported_in_formats[i] == fmt) return 1; -+ return 0; -+} -+ -+static int format_output_is_supported(enum AVPixelFormat fmt) -+{ -+ for (int i = 0; i < FF_ARRAY_ELEMS(supported_out_formats); i++) -+ if (supported_out_formats[i] == fmt) return 1; -+ return 0; -+} -+ -+static av_cold int init(AVFilterContext *avctx) -+{ -+ ConvertCUDAContext *ctx = avctx->priv; -+ -+ ctx->frame = av_frame_alloc(); -+ if (!ctx->frame) -+ return AVERROR(ENOMEM); -+ -+ ctx->tmp_frame = av_frame_alloc(); -+ if (!ctx->tmp_frame) -+ return AVERROR(ENOMEM); -+ -+ return 0; -+} -+ -+static av_cold void uninit(AVFilterContext *avctx) -+{ -+ ConvertCUDAContext *ctx = avctx->priv; -+ -+ if (ctx->hwctx && ctx->cu_module) { -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext dummy; -+ -+ CHECK_CU(cu->cuCtxPushCurrent(ctx->hwctx->cuda_ctx)); -+ CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); -+ ctx->cu_module = NULL; -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ } -+ -+ av_frame_free(&ctx->frame); -+ av_buffer_unref(&ctx->frames_ctx); -+ av_frame_free(&ctx->tmp_frame); -+} -+ -+static av_cold int output_config_props(AVFilterLink *outlink) -+{ -+ extern const unsigned char ff_vf_convert_cuda_ptx_data[]; -+ extern const unsigned int ff_vf_convert_cuda_ptx_len; -+ -+ FilterLink *outl = ff_filter_link(outlink); -+ AVFilterContext *avctx = outlink->src; -+ AVFilterLink *inlink = outlink->src->inputs[0]; -+ FilterLink *inl = ff_filter_link(inlink); -+ ConvertCUDAContext *ctx = avctx->priv; -+ AVHWFramesContext *frames_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; -+ int ret; -+ -+ if (!frames_ctx) { -+ av_log(ctx, AV_LOG_ERROR, "No hw context provided on main input\n"); -+ return AVERROR(EINVAL); -+ } -+ -+ ctx->format_in = frames_ctx->sw_format; -+ if (!format_input_is_supported(ctx->format_in)) { -+ av_log(ctx, AV_LOG_ERROR, "Unsupported input format: %s\n", -+ av_get_pix_fmt_name(ctx->format_in)); -+ return AVERROR(ENOSYS); -+ } -+ -+ if (!ctx->format_str || !strlen(ctx->format_str)) { -+ av_log(ctx, AV_LOG_ERROR, "Specify output pixel format argument (example: format=yuva420p, format=yuva444p)\n"); -+ return AVERROR(ENOSYS); -+ } -+ -+ ctx->format_out = av_get_pix_fmt(ctx->format_str); -+ if (!format_output_is_supported(ctx->format_out)) { -+ av_log(ctx, AV_LOG_ERROR, "Unsupported output format: %s (%s)\n", -+ av_get_pix_fmt_name(ctx->format_out), ctx->format_str); -+ return AVERROR(ENOSYS); -+ } -+ -+ -+ // initialize -+ ctx->hw_device_ctx = av_buffer_ref(frames_ctx->device_ref); -+ if (!ctx->hw_device_ctx) -+ return AVERROR(ENOMEM); -+ ctx->hwctx = ((AVHWDeviceContext*)ctx->hw_device_ctx->data)->hwctx; -+ ctx->cu_stream = ctx->hwctx->stream; -+ -+ // load cuda functions -+ { -+ CUcontext cuda_ctx = ctx->hwctx->cuda_ctx; -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext dummy; -+ const char *fmt_str_in[2] = {0}; -+ const char *fmt_str_out = 0; -+ static_assert(ARRAY_COUNT(fmt_str_in) == ARRAY_COUNT(ctx->cu_functions)); -+ -+ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); -+ if (ret < 0) { -+ return ret; -+ } -+ -+ ret = ff_cuda_load_module(ctx, ctx->hwctx, &ctx->cu_module, ff_vf_convert_cuda_ptx_data, ff_vf_convert_cuda_ptx_len); -+ if (ret < 0) { -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return ret; -+ } -+ -+ switch (ctx->format_in) { -+ case AV_PIX_FMT_ARGB: fmt_str_in[0] = "ARGB"; break; -+ case AV_PIX_FMT_RGBA: fmt_str_in[0] = "RGBA"; break; -+ case AV_PIX_FMT_BGRA: fmt_str_in[0] = "BGRA"; break; -+ case AV_PIX_FMT_ABGR: fmt_str_in[0] = "ABGR"; break; -+ -+ case AV_PIX_FMT_YUVA444P12LE: -+ fmt_str_in[0] = "YUVA444P12LE_ya"; -+ fmt_str_in[1] = "YUVA444P12LE_uv"; -+ break; -+ -+ case AV_PIX_FMT_YUVA444P: -+ fmt_str_in[0] = "YUVA444P_ya"; -+ fmt_str_in[1] = "YUVA444P_uv"; -+ break; -+ -+ case AV_PIX_FMT_NV12: -+ fmt_str_in[0] = "NV12_y"; -+ fmt_str_in[1] = "NV12_uv"; -+ break; -+ -+ case AV_PIX_FMT_YUV420P: -+ fmt_str_in[0] = "YUV420P_y"; -+ fmt_str_in[1] = "YUV420P_uv"; -+ break; -+ } -+ -+ switch (ctx->format_out) { -+ case AV_PIX_FMT_YUVA420P: fmt_str_out = "YUVA420P"; break; -+ case AV_PIX_FMT_YUVA444P: fmt_str_out = "YUVA444P"; break; -+ case AV_PIX_FMT_NV12: fmt_str_out = "NV12"; break; -+ case AV_PIX_FMT_YUV420P: fmt_str_out = "YUV420P"; break; -+ } -+ -+ for (int i = 0; i < ARRAY_COUNT(fmt_str_in); i++) { -+ if (fmt_str_in[i] && fmt_str_out) { -+ char func_name[256]; -+ snprintf(func_name, sizeof(func_name), "Convert_%s_to_%s", fmt_str_in[i], fmt_str_out); -+ ret = CHECK_CU(cu->cuModuleGetFunction(&ctx->cu_functions[i], ctx->cu_module, func_name)); -+ -+ if (ret < 0) { -+ av_log(ctx, AV_LOG_FATAL, "CUDA function %s wasn't implemented\n", func_name); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return ret; -+ } -+ } -+ } -+ -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ } -+ -+ outlink->w = inlink->w; -+ outlink->h = inlink->h; -+ -+ // prepare output buffer -+ { -+ AVHWFramesContext *out_ctx; -+ AVBufferRef *out_ref = av_hwframe_ctx_alloc(ctx->hw_device_ctx); -+ if (!out_ref) -+ return AVERROR(ENOMEM); -+ -+ out_ctx = (AVHWFramesContext*)out_ref->data; -+ out_ctx->format = AV_PIX_FMT_CUDA; -+ out_ctx->sw_format = ctx->format_out; -+ out_ctx->width = FFALIGN(inlink->w, 32); -+ out_ctx->height = FFALIGN(inlink->h, 32); -+ -+ ret = av_hwframe_ctx_init(out_ref); -+ if (ret < 0) -+ goto output_buffer_fail; -+ -+ av_frame_unref(ctx->frame); -+ ret = av_hwframe_get_buffer(out_ref, ctx->frame, 0); -+ if (ret < 0) -+ goto output_buffer_fail; -+ -+ ctx->frame->width = inlink->w; -+ ctx->frame->height = inlink->h; -+ -+ ctx->frames_ctx = out_ref; -+ -+ if (ret < 0) { -+output_buffer_fail: -+ av_buffer_unref(&out_ref); -+ return ret; -+ } -+ -+ outl->hw_frames_ctx = av_buffer_ref(ctx->frames_ctx); -+ if (!outl->hw_frames_ctx) -+ return AVERROR(ENOMEM); -+ } -+ -+ return 0; -+} -+ -+static int fill_buffers(AVFilterContext *avctx, AVFrame *out, AVFrame *in) -+{ -+ ConvertCUDAContext *ctx = avctx->priv; -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext cuda_ctx = ctx->hwctx->cuda_ctx; -+ CUcontext dummy; -+ int ret; -+ -+ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); -+ if (ret < 0) return ret; -+ -+ -+#define LaunchKernelBlock(Func, Width, Height, BlockX, BlockY) CHECK_CU(cu->cuLaunchKernel(\ -+ Func, DIV_UP((Width), (BlockX)), DIV_UP((Height), (BlockY)), 1,\ -+ (BlockX), (BlockY), 1, 0, ctx->cu_stream, kernel_args, NULL)) -+ -+#define LaunchKernel(Func, Width, Height) LaunchKernelBlock(Func, Width, Height, BLOCK_X, BLOCK_Y) -+ -+ switch (ctx->format_in) -+ { -+ case AV_PIX_FMT_ARGB: -+ case AV_PIX_FMT_RGBA: -+ case AV_PIX_FMT_ABGR: -+ case AV_PIX_FMT_BGRA: { -+ void *kernel_args[] = { -+ &in->data[0], &in->linesize[0], -+ &out->data[0], &out->linesize[0], -+ &out->data[1], &out->linesize[1], -+ &out->data[2], &out->linesize[2], -+ &out->data[3], &out->linesize[3], -+ }; -+ -+ unsigned int out_width = out->width; -+ unsigned int out_height = out->height; -+ unsigned int block_width = BLOCK_X; -+ unsigned int block_height = BLOCK_Y; -+ if (ctx->format_out == AV_PIX_FMT_YUVA420P) { -+ out_width /= 2; -+ out_height /= 2; -+ block_width /= 2; -+ block_height /= 2; -+ } -+ -+ ret = LaunchKernelBlock(ctx->cu_functions[0], -+ out_width, out_height, -+ block_width, block_height); -+ if (ret < 0) return ret; -+ } break; -+ -+ case AV_PIX_FMT_YUVA444P: -+ case AV_PIX_FMT_YUVA444P12LE: { -+ unsigned int out_uv_width = out->width; -+ unsigned int out_uv_height = out->height; -+ if (ctx->format_out == AV_PIX_FMT_YUVA420P) { -+ out_uv_width /= 2; -+ out_uv_height /= 2; -+ } -+ -+ { // y -+ void *kernel_args[] = { -+ &out->data[0], &out->linesize[0], -+ &in->data[0], &in->linesize[0] -+ }; -+ ret = LaunchKernel(ctx->cu_functions[0], out->width, out->height); -+ if (ret < 0) return ret; -+ } -+ { // u -+ void *kernel_args[] = { -+ &out->data[1], &out->linesize[1], -+ &in->data[1], &in->linesize[1] -+ }; -+ ret = LaunchKernel(ctx->cu_functions[1], out_uv_width, out_uv_height); -+ if (ret < 0) return ret; -+ } -+ { // v -+ void *kernel_args[] = { -+ &out->data[2], &out->linesize[2], -+ &in->data[2], &in->linesize[2] -+ }; -+ ret = LaunchKernel(ctx->cu_functions[1], out_uv_width, out_uv_height); -+ if (ret < 0) return ret; -+ } -+ { // alpha -+ void *kernel_args[] = { -+ &out->data[3], &out->linesize[3], -+ &in->data[3], &in->linesize[3] -+ }; -+ ret = LaunchKernel(ctx->cu_functions[0], out->width, out->height); -+ if (ret < 0) return ret; -+ } -+ } break; -+ -+ case AV_PIX_FMT_NV12: { -+ // Y plane: copy verbatim -+ { -+ void *kernel_args[] = { -+ &out->data[0], &out->linesize[0], -+ &in->data[0], &in->linesize[0], -+ }; -+ ret = LaunchKernel(ctx->cu_functions[0], out->width, out->height); -+ if (ret < 0) return ret; -+ } -+ // UV plane: deinterleave NV12 UV -> separate YUV420P U and V -+ { -+ void *kernel_args[] = { -+ &out->data[1], &out->linesize[1], -+ &out->data[2], &out->linesize[2], -+ &in->data[1], &in->linesize[1], -+ }; -+ ret = LaunchKernel(ctx->cu_functions[1], out->width / 2, out->height / 2); -+ if (ret < 0) return ret; -+ } -+ } break; -+ -+ case AV_PIX_FMT_YUV420P: { -+ // Y plane: copy verbatim -+ { -+ void *kernel_args[] = { -+ &out->data[0], &out->linesize[0], -+ &in->data[0], &in->linesize[0], -+ }; -+ ret = LaunchKernel(ctx->cu_functions[0], out->width, out->height); -+ if (ret < 0) return ret; -+ } -+ // UV plane: interleave YUV420P U and V -> NV12 UV -+ { -+ void *kernel_args[] = { -+ &out->data[1], &out->linesize[1], -+ &in->data[1], &in->linesize[1], -+ &in->data[2], &in->linesize[2], -+ }; -+ ret = LaunchKernel(ctx->cu_functions[1], out->width / 2, out->height / 2); -+ if (ret < 0) return ret; -+ } -+ } break; -+ -+ default: { -+ av_log(ctx, AV_LOG_ERROR, "Unexpected lack of support for pixel format %s\n", -+ av_get_pix_fmt_name(ctx->format_in)); -+ av_frame_free(&out); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return AVERROR_BUG; -+ } break; -+ } -+ -+ ret = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ if (ret < 0) return ret; -+ -+ return 0; -+} -+ -+static int filter_frame(AVFilterLink *link, AVFrame *in) -+{ -+ AVFilterContext *avctx = link->dst; -+ ConvertCUDAContext *ctx = avctx->priv; -+ AVFilterLink *outlink = avctx->outputs[0]; -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ -+ AVFrame *out = NULL; -+ CUcontext dummy; -+ int ret = 0; -+ -+ out = av_frame_alloc(); -+ if (!out) { -+ ret = AVERROR(ENOMEM); -+ goto fail; -+ } -+ -+ ret = CHECK_CU(cu->cuCtxPushCurrent(ctx->hwctx->cuda_ctx)); -+ if (ret < 0) -+ goto fail; -+ -+ // fill and prepare the output frame -+ { -+ ret = fill_buffers(avctx, ctx->frame, in); -+ if (ret < 0) -+ return ret; -+ -+ ret = av_hwframe_get_buffer(ctx->frame->hw_frames_ctx, ctx->tmp_frame, 0); -+ if (ret < 0) -+ return ret; -+ -+ av_frame_move_ref(out, ctx->frame); -+ av_frame_move_ref(ctx->frame, ctx->tmp_frame); -+ -+ ctx->frame->width = outlink->w; -+ ctx->frame->height = outlink->h; -+ -+ ret = av_frame_copy_props(out, in); -+ if (ret < 0) -+ return ret; -+ } -+ -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ if (ret < 0) -+ goto fail; -+ -+ av_frame_free(&in); -+ return ff_filter_frame(outlink, out); -+fail: -+ av_frame_free(&in); -+ av_frame_free(&out); -+ return ret; -+} -+ -+ -+ -+#define OFFSET(x) offsetof(ConvertCUDAContext, x) -+#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM|AV_OPT_FLAG_VIDEO_PARAM) -+ -+static const AVOption convert_cuda_options[] = { -+ { "format", "Output pixel format", OFFSET(format_str), AV_OPT_TYPE_STRING, { .str = "" }, 0, 0, FLAGS }, -+ { NULL } -+}; -+ -+AVFILTER_DEFINE_CLASS(convert_cuda); -+ -+static const AVFilterPad inputs[] = { -+ { -+ .name = "default", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .filter_frame = filter_frame, -+ }, -+}; -+ -+static const AVFilterPad outputs[] = { -+ { -+ .name = "default", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .config_props = output_config_props, -+ }, -+}; -+ -+const AVFilter ff_vf_convert_cuda = { -+ .name = "convert_cuda", -+ .description = NULL_IF_CONFIG_SMALL("CUDA accelerated pixel format conversion"), -+ .priv_size = sizeof(ConvertCUDAContext), -+ .priv_class = &convert_cuda_class, -+ .init = init, -+ .uninit = uninit, -+ FILTER_INPUTS(inputs), -+ FILTER_OUTPUTS(outputs), -+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), -+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, -+}; -diff --git a/libavfilter/vf_convert_cuda.cu b/libavfilter/vf_convert_cuda.cu -new file mode 100644 -index 0000000..beabc0c ---- /dev/null -+++ b/libavfilter/vf_convert_cuda.cu -@@ -0,0 +1,273 @@ -+typedef unsigned char uint8_t; -+typedef unsigned short uint16_t; -+ -+// @todo: 3 shaders here are indentical to Convert_YUVA444P12LE_ya_to_YUVA420P -+// Boilerplate/organization code in vf_convert_cuda.c -+// could be refactored a bit so there is less copypasta. -+ -+ -+// -+// Color constants for converting from RGB to YUV -+// in this file are following BT709 color profile. -+// They are derived from libavutil/colorspace.h -+// -+ -+// offsets into color components -+// <0, 1, 2, 3> for argb -+// <3, 2, 1, 0> for bgra -+template -+__device__ static inline void convert_templated_argb_to_yuva420p( -+ uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, -+ uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, -+ uint8_t *out_a, int out_a_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ int x2 = x*2; -+ int y2 = y*2; -+ -+ uint8_t *pixel0 = &in_argb[(x2 )*4 + (y2 )*in_linesize]; -+ uint8_t *pixel1 = &in_argb[(x2 + 1)*4 + (y2 )*in_linesize]; -+ uint8_t *pixel2 = &in_argb[(x2 )*4 + (y2 + 1)*in_linesize]; -+ uint8_t *pixel3 = &in_argb[(x2 + 1)*4 + (y2 + 1)*in_linesize]; -+ -+ uint8_t r0 = pixel0[ri]; uint8_t g0 = pixel0[gi]; uint8_t b0 = pixel0[bi]; -+ uint8_t r1 = pixel1[ri]; uint8_t g1 = pixel1[gi]; uint8_t b1 = pixel1[bi]; -+ uint8_t r2 = pixel2[ri]; uint8_t g2 = pixel2[gi]; uint8_t b2 = pixel2[bi]; -+ uint8_t r3 = pixel3[ri]; uint8_t g3 = pixel3[gi]; uint8_t b3 = pixel3[bi]; -+ -+ // calculate Y for 4 pixels -+ out_y[(x2 ) + (y2 )*out_y_linesize] = lroundf(0.18258588f*r0 + 0.6142305882f*g0 + 0.062007059f*b0 + 16.f); -+ out_y[(x2 + 1) + (y2 )*out_y_linesize] = lroundf(0.18258588f*r1 + 0.6142305882f*g1 + 0.062007059f*b1 + 16.f); -+ out_y[(x2 ) + (y2 + 1)*out_y_linesize] = lroundf(0.18258588f*r2 + 0.6142305882f*g2 + 0.062007059f*b2 + 16.f); -+ out_y[(x2 + 1) + (y2 + 1)*out_y_linesize] = lroundf(0.18258588f*r3 + 0.6142305882f*g3 + 0.062007059f*b3 + 16.f); -+ -+ // copy alpha as is for 4 pixels -+ out_a[(x2 ) + (y2 )*out_a_linesize] = pixel0[ai]; -+ out_a[(x2 + 1) + (y2 )*out_a_linesize] = pixel1[ai]; -+ out_a[(x2 ) + (y2 + 1)*out_a_linesize] = pixel2[ai]; -+ out_a[(x2 + 1) + (y2 + 1)*out_a_linesize] = pixel3[ai]; -+ -+ // average out 4 rgb pixels -+ uint8_t avg_r = (uint8_t)((r0 + r1 + r2 + r3)/4); -+ uint8_t avg_g = (uint8_t)((g0 + g1 + g2 + g3)/4); -+ uint8_t avg_b = (uint8_t)((b0 + b1 + b2 + b3)/4); -+ -+ // write out U and V -+ out_u[x + (y * out_u_linesize)] = lroundf((-0.100641882f*avg_r - 0.3385738039f*avg_g + 0.439215686f*avg_b) + 128.f); -+ out_v[x + (y * out_v_linesize)] = lroundf(( 0.439215686f*avg_r - 0.3989396078f*avg_g - 0.040276078f*avg_b) + 128.f); -+} -+ -+ -+// offsets into color components -+// <0, 1, 2, 3> for argb -+// <3, 2, 1, 0> for bgra -+template -+__device__ static inline void convert_templated_argb_to_yuva444p( -+ uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, -+ uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, -+ uint8_t *out_a, int out_a_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ -+ uint8_t *pixel0 = &in_argb[x*4 + y*in_linesize]; -+ uint8_t r0 = pixel0[ri]; uint8_t g0 = pixel0[gi]; uint8_t b0 = pixel0[bi]; -+ -+ // write out Y and A -+ out_y[x + y*out_y_linesize] = lroundf(0.18258588f*r0 + 0.6142305882f*g0 + 0.062007059f*b0 + 16.f); -+ out_a[x + y*out_a_linesize] = pixel0[ai]; -+ -+ // write out U and V -+ out_u[x + y*out_u_linesize] = lroundf((-0.100641882f*r0 - 0.3385738039f*g0 + 0.439215686f*b0) + 128.f); -+ out_v[x + y*out_v_linesize] = lroundf(( 0.439215686f*r0 - 0.3989396078f*g0 - 0.040276078f*b0) + 128.f); -+} -+ -+ -+ -+extern "C" { -+ -+// ARGB (+variants) -> YUVA420P -+__global__ void Convert_ARGB_to_YUVA420P(uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) -+{ -+ convert_templated_argb_to_yuva420p<0,1,2,3>(in_argb, in_linesize, -+ out_y, out_y_linesize, out_u, out_u_linesize, -+ out_v, out_v_linesize, out_a, out_a_linesize); -+} -+ -+__global__ void Convert_RGBA_to_YUVA420P(uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) -+{ -+ convert_templated_argb_to_yuva420p<3,0,1,2>(in_argb, in_linesize, -+ out_y, out_y_linesize, out_u, out_u_linesize, -+ out_v, out_v_linesize, out_a, out_a_linesize); -+} -+ -+__global__ void Convert_BGRA_to_YUVA420P(uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) -+{ -+ convert_templated_argb_to_yuva420p<3,2,1,0>(in_argb, in_linesize, -+ out_y, out_y_linesize, out_u, out_u_linesize, -+ out_v, out_v_linesize, out_a, out_a_linesize); -+} -+ -+__global__ void Convert_ABGR_to_YUVA420P(uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) -+{ -+ convert_templated_argb_to_yuva420p<0,3,2,1>(in_argb, in_linesize, -+ out_y, out_y_linesize, out_u, out_u_linesize, -+ out_v, out_v_linesize, out_a, out_a_linesize); -+} -+ -+ -+ -+ -+// ARGB (+variants) -> YUVA444P -+__global__ void Convert_ARGB_to_YUVA444P(uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) -+{ -+ convert_templated_argb_to_yuva444p<0,1,2,3>(in_argb, in_linesize, -+ out_y, out_y_linesize, out_u, out_u_linesize, -+ out_v, out_v_linesize, out_a, out_a_linesize); -+} -+ -+__global__ void Convert_RGBA_to_YUVA444P(uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) -+{ -+ convert_templated_argb_to_yuva444p<3,0,1,2>(in_argb, in_linesize, -+ out_y, out_y_linesize, out_u, out_u_linesize, -+ out_v, out_v_linesize, out_a, out_a_linesize); -+} -+ -+__global__ void Convert_BGRA_to_YUVA444P(uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) -+{ -+ convert_templated_argb_to_yuva444p<3,2,1,0>(in_argb, in_linesize, -+ out_y, out_y_linesize, out_u, out_u_linesize, -+ out_v, out_v_linesize, out_a, out_a_linesize); -+} -+ -+__global__ void Convert_ABGR_to_YUVA444P(uint8_t *in_argb, int in_linesize, -+ uint8_t *out_y, int out_y_linesize, uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, uint8_t *out_a, int out_a_linesize) -+{ -+ convert_templated_argb_to_yuva444p<0,3,2,1>(in_argb, in_linesize, -+ out_y, out_y_linesize, out_u, out_u_linesize, -+ out_v, out_v_linesize, out_a, out_a_linesize); -+} -+ -+ -+ -+ -+// YUVA444P12LE -> YUVA420P -+__global__ void Convert_YUVA444P12LE_ya_to_YUVA420P( -+ uint8_t *main, int main_linesize, -+ uint16_t *overlay, int overlay_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ main[x + y*main_linesize] = overlay[x + y*(overlay_linesize/2)] >> 4; -+} -+ -+__global__ void Convert_YUVA444P12LE_uv_to_YUVA420P( -+ uint8_t *main, int main_linesize, -+ uint16_t *overlay, int overlay_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ main[x + y*main_linesize] = overlay[x*2 + (y*2)*(overlay_linesize/2)] >> 4; -+} -+ -+// YUVA444P12LE -> YUVA444P -+__global__ void Convert_YUVA444P12LE_ya_to_YUVA444P( -+ uint8_t *main, int main_linesize, -+ uint16_t *overlay, int overlay_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ main[x + y*main_linesize] = overlay[x + y*(overlay_linesize/2)] >> 4; -+} -+ -+__global__ void Convert_YUVA444P12LE_uv_to_YUVA444P( -+ uint8_t *main, int main_linesize, -+ uint16_t *overlay, int overlay_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ main[x + y*main_linesize] = overlay[x + y*(overlay_linesize/2)] >> 4; -+} -+ -+// NV12 to YUV420P -+__global__ void Convert_NV12_y_to_YUV420P( -+ uint8_t *out_y, int out_y_linesize, -+ uint8_t *in_y, int in_y_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ out_y[x + y * out_y_linesize] = in_y[x + y * in_y_linesize]; -+} -+ -+__global__ void Convert_NV12_uv_to_YUV420P( -+ uint8_t *out_u, int out_u_linesize, -+ uint8_t *out_v, int out_v_linesize, -+ uint8_t *in_uv, int in_uv_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ out_u[x + y * out_u_linesize] = in_uv[x * 2 + y * in_uv_linesize]; -+ out_v[x + y * out_v_linesize] = in_uv[x * 2 + 1 + y * in_uv_linesize]; -+} -+ -+// YUV420P to NV12 -+__global__ void Convert_YUV420P_y_to_NV12( -+ uint8_t *out_y, int out_y_linesize, -+ uint8_t *in_y, int in_y_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ out_y[x + y * out_y_linesize] = in_y[x + y * in_y_linesize]; -+} -+ -+__global__ void Convert_YUV420P_uv_to_NV12( -+ uint8_t *out_uv, int out_uv_linesize, -+ uint8_t *in_u, int in_u_linesize, -+ uint8_t *in_v, int in_v_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ out_uv[x * 2 + y * out_uv_linesize] = in_u[x + y * in_u_linesize]; -+ out_uv[x * 2 + 1 + y * out_uv_linesize] = in_v[x + y * in_v_linesize]; -+} -+ -+// YUVA444P -> YUVA420P -+__global__ void Convert_YUVA444P_ya_to_YUVA420P( -+ uint8_t *main, int main_linesize, -+ uint8_t *overlay, int overlay_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ main[x + y*main_linesize] = overlay[x + y*(overlay_linesize)]; -+} -+ -+__global__ void Convert_YUVA444P_uv_to_YUVA420P( -+ uint8_t *main, int main_linesize, -+ uint8_t *overlay, int overlay_linesize) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ main[x + y*main_linesize] = overlay[x*2 + (y*2)*(overlay_linesize)]; -+} -+ -+} -diff --git a/libavfilter/vf_crop_cuda.c b/libavfilter/vf_crop_cuda.c -new file mode 100644 -index 0000000..d08dc38 ---- /dev/null -+++ b/libavfilter/vf_crop_cuda.c -@@ -0,0 +1,561 @@ -+/* -+ * Copyright (c) 2019 - 2022 -+ * -+ * This file is part of FFmpeg. -+ * -+ * FFmpeg is free software; you can redistribute it and/or -+ * modify it under the terms of the GNU Lesser General Public -+ * License as published by the Free Software Foundation; either -+ * version 2.1 of the License, or (at your option) any later version. -+ * -+ * FFmpeg is distributed in the hope that it will be useful, -+ * but WITHOUT ANY WARRANTY; without even the implied warranty of -+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU -+ * Lesser General Public License for more details. -+ * -+ * You should have received a copy of the GNU Lesser General Public -+ * License along with FFmpeg; if not, write to the Free Software -+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA -+ */ -+ -+#include -+ -+#include "avfilter.h" -+#include "filters.h" -+#include "video.h" -+#include "libavutil/avstring.h" -+#include "libavutil/cuda_check.h" -+#include "libavutil/eval.h" -+#include "libavutil/hwcontext.h" -+#include "libavutil/hwcontext_cuda_internal.h" -+#include "libavutil/imgutils.h" -+#include "libavutil/internal.h" -+#include "libavutil/libm.h" -+#include "libavutil/mathematics.h" -+#include "libavutil/opt.h" -+ -+static const char *const var_names[] = { -+ "in_w", "iw", -+ "in_h", "ih", -+ "out_w", "ow", -+ "out_h", "oh", -+ "a", -+ "sar", -+ "dar", -+ "hsub", -+ "vsub", -+ "x", -+ "y", -+ "n", -+#if FF_API_FRAME_PKT -+ "pos", -+#endif -+ "t", -+ NULL -+}; -+ -+enum var_name { -+ VAR_IN_W, VAR_IW, -+ VAR_IN_H, VAR_IH, -+ VAR_OUT_W, VAR_OW, -+ VAR_OUT_H, VAR_OH, -+ VAR_A, -+ VAR_SAR, -+ VAR_DAR, -+ VAR_HSUB, -+ VAR_VSUB, -+ VAR_X, -+ VAR_Y, -+ VAR_N, -+#if FF_API_FRAME_PKT -+ VAR_POS, -+#endif -+ VAR_T, -+ VAR_VARS_NB -+}; -+ -+typedef struct CropCUDAContext { -+ const AVClass *class; -+ AVCUDADeviceContext *hwctx; -+ AVBufferRef *hw_device_ctx; -+ -+ AVBufferRef *frames_ctx; -+ AVFrame *frame; -+ AVFrame *tmp_frame; -+ -+ enum AVPixelFormat sw_format; -+ -+ int x; -+ int y; -+ int w; -+ int h; -+ -+ AVRational out_sar; -+ int keep_aspect; -+ int exact; -+ -+ int max_step[4]; -+ int hsub, vsub; -+ char *x_expr, *y_expr, *w_expr, *h_expr; -+ AVExpr *x_pexpr, *y_pexpr; -+ double var_values[VAR_VARS_NB]; -+} CropCUDAContext; -+ -+#define CHECK_CU(x) FF_CUDA_CHECK_DL(s, cu, x) -+ -+static av_cold int init(AVFilterContext *ctx) -+{ -+ CropCUDAContext *s = ctx->priv; -+ -+ s->frame = av_frame_alloc(); -+ if (!s->frame) -+ return AVERROR(ENOMEM); -+ -+ s->tmp_frame = av_frame_alloc(); -+ if (!s->tmp_frame) -+ return AVERROR(ENOMEM); -+ -+ return 0; -+} -+ -+static av_cold void uninit(AVFilterContext *ctx) -+{ -+ CropCUDAContext *s = ctx->priv; -+ -+ av_expr_free(s->x_pexpr); -+ s->x_pexpr = NULL; -+ av_expr_free(s->y_pexpr); -+ s->y_pexpr = NULL; -+ -+ av_frame_free(&s->frame); -+ av_buffer_unref(&s->hw_device_ctx); -+ av_buffer_unref(&s->frames_ctx); -+ av_frame_free(&s->tmp_frame); -+} -+ -+static inline int normalize_double(int *n, double d) -+{ -+ int ret = 0; -+ -+ if (isnan(d)) { -+ ret = AVERROR(EINVAL); -+ } else if (d > INT_MAX || d < INT_MIN) { -+ *n = d > INT_MAX ? INT_MAX : INT_MIN; -+ ret = AVERROR(EINVAL); -+ } else { -+ *n = lrint(d); -+ } -+ -+ return ret; -+} -+ -+static int config_input(AVFilterLink *link) -+{ -+ FilterLink *l = ff_filter_link(link); -+ AVFilterContext *ctx = link->dst; -+ CropCUDAContext *s = ctx->priv; -+ AVHWFramesContext *frames_ctx; -+ const AVPixFmtDescriptor *pix_desc; -+ int ret; -+ const char *expr; -+ double res; -+ -+ if (!l->hw_frames_ctx) { -+ av_log(ctx, AV_LOG_ERROR, "No hw context provided on main input\n"); -+ return AVERROR(EINVAL); -+ } -+ frames_ctx = (AVHWFramesContext *)l->hw_frames_ctx->data; -+ -+ s->hw_device_ctx = av_buffer_ref(frames_ctx->device_ref); -+ if (!s->hw_device_ctx) -+ return AVERROR(ENOMEM); -+ s->hwctx = ((AVHWDeviceContext *)s->hw_device_ctx->data)->hwctx; -+ -+ s->sw_format = frames_ctx->sw_format; -+ pix_desc = av_pix_fmt_desc_get(s->sw_format); -+ -+ s->var_values[VAR_IN_W] = s->var_values[VAR_IW] = ctx->inputs[0]->w; -+ s->var_values[VAR_IN_H] = s->var_values[VAR_IH] = ctx->inputs[0]->h; -+ s->var_values[VAR_A] = (float)link->w / link->h; -+ s->var_values[VAR_SAR] = link->sample_aspect_ratio.num ? av_q2d(link->sample_aspect_ratio) : 1; -+ s->var_values[VAR_DAR] = s->var_values[VAR_A] * s->var_values[VAR_SAR]; -+ s->var_values[VAR_HSUB] = 1 << pix_desc->log2_chroma_w; -+ s->var_values[VAR_VSUB] = 1 << pix_desc->log2_chroma_h; -+ s->var_values[VAR_X] = NAN; -+ s->var_values[VAR_Y] = NAN; -+ s->var_values[VAR_OUT_W] = s->var_values[VAR_OW] = NAN; -+ s->var_values[VAR_OUT_H] = s->var_values[VAR_OH] = NAN; -+ s->var_values[VAR_N] = 0; -+ s->var_values[VAR_T] = NAN; -+#if FF_API_FRAME_PKT -+ s->var_values[VAR_POS] = NAN; -+#endif -+ -+ av_image_fill_max_pixsteps(s->max_step, NULL, pix_desc); -+ -+ s->hsub = pix_desc->log2_chroma_w; -+ s->vsub = pix_desc->log2_chroma_h; -+ -+ av_expr_parse_and_eval(&res, (expr = s->w_expr), -+ var_names, s->var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx); -+ s->var_values[VAR_OUT_W] = s->var_values[VAR_OW] = res; -+ if ((ret = av_expr_parse_and_eval(&res, (expr = s->h_expr), -+ var_names, s->var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) -+ goto fail_expr; -+ s->var_values[VAR_OUT_H] = s->var_values[VAR_OH] = res; -+ if ((ret = av_expr_parse_and_eval(&res, (expr = s->w_expr), -+ var_names, s->var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) -+ goto fail_expr; -+ -+ s->var_values[VAR_OUT_W] = s->var_values[VAR_OW] = res; -+ if (normalize_double(&s->w, s->var_values[VAR_OUT_W]) < 0 || -+ normalize_double(&s->h, s->var_values[VAR_OUT_H]) < 0) { -+ av_log(ctx, AV_LOG_ERROR, -+ "Too big value or invalid expression for out_w/ow or out_h/oh. " -+ "Maybe the expression for out_w:'%s' or for out_h:'%s' is self-referencing.\n", -+ s->w_expr, s->h_expr); -+ return AVERROR(EINVAL); -+ } -+ -+ if (!s->exact) { -+ s->w &= ~((1 << s->hsub) - 1); -+ s->h &= ~((1 << s->vsub) - 1); -+ } -+ -+ av_expr_free(s->x_pexpr); -+ av_expr_free(s->y_pexpr); -+ s->x_pexpr = s->y_pexpr = NULL; -+ if ((ret = av_expr_parse(&s->x_pexpr, s->x_expr, var_names, -+ NULL, NULL, NULL, NULL, 0, ctx)) < 0 || -+ (ret = av_expr_parse(&s->y_pexpr, s->y_expr, var_names, -+ NULL, NULL, NULL, NULL, 0, ctx)) < 0) -+ return AVERROR(EINVAL); -+ -+ if (s->keep_aspect) { -+ AVRational dar = av_mul_q(link->sample_aspect_ratio, -+ (AVRational){ link->w, link->h }); -+ av_reduce(&s->out_sar.num, &s->out_sar.den, -+ (int64_t)dar.num * s->h, (int64_t)dar.den * s->w, INT_MAX); -+ } else { -+ s->out_sar = link->sample_aspect_ratio; -+ } -+ -+ av_log(ctx, AV_LOG_VERBOSE, "w:%d h:%d sar:%d/%d -> w:%d h:%d sar:%d/%d\n", -+ link->w, link->h, link->sample_aspect_ratio.num, link->sample_aspect_ratio.den, -+ s->w, s->h, s->out_sar.num, s->out_sar.den); -+ -+ if (s->w <= 0 || s->h <= 0 || s->w > link->w || s->h > link->h) { -+ av_log(ctx, AV_LOG_ERROR, -+ "Invalid too big or non positive size for width '%d' or height '%d'\n", -+ s->w, s->h); -+ return AVERROR(EINVAL); -+ } -+ -+ s->x = (link->w - s->w) / 2; -+ s->y = (link->h - s->h) / 2; -+ if (!s->exact) { -+ s->x &= ~((1 << s->hsub) - 1); -+ s->y &= ~((1 << s->vsub) - 1); -+ } -+ -+ { -+ FilterLink *outl = ff_filter_link(ctx->outputs[0]); -+ AVHWFramesContext *out_ctx; -+ AVBufferRef *out_ref = av_hwframe_ctx_alloc(s->hw_device_ctx); -+ if (!out_ref) -+ return AVERROR(ENOMEM); -+ -+ out_ctx = (AVHWFramesContext *)out_ref->data; -+ out_ctx->format = AV_PIX_FMT_CUDA; -+ out_ctx->sw_format = s->sw_format; -+ out_ctx->width = FFALIGN(s->w, 32); -+ out_ctx->height = FFALIGN(s->h, 32); -+ -+ ret = av_hwframe_ctx_init(out_ref); -+ if (ret < 0) -+ goto output_buffer_fail; -+ -+ av_frame_unref(s->frame); -+ ret = av_hwframe_get_buffer(out_ref, s->frame, 0); -+ if (ret < 0) -+ goto output_buffer_fail; -+ -+ s->frame->width = s->w; -+ s->frame->height = s->h; -+ -+ av_buffer_unref(&s->frames_ctx); -+ s->frames_ctx = out_ref; -+ -+ if (ret < 0) { -+output_buffer_fail: -+ av_buffer_unref(&out_ref); -+ return ret; -+ } -+ -+ av_buffer_unref(&outl->hw_frames_ctx); -+ outl->hw_frames_ctx = av_buffer_ref(s->frames_ctx); -+ if (!outl->hw_frames_ctx) -+ return AVERROR(ENOMEM); -+ } -+ -+ return 0; -+ -+fail_expr: -+ av_log(ctx, AV_LOG_ERROR, "Error when evaluating the expression '%s'\n", expr); -+ return ret; -+} -+ -+static int config_output(AVFilterLink *link) -+{ -+ CropCUDAContext *s = link->src->priv; -+ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(link->format); -+ -+ if (desc->flags & AV_PIX_FMT_FLAG_HWACCEL && link->format != AV_PIX_FMT_CUDA) { -+ /* Hardware frames adjust the cropping regions rather than changing the frame size. */ -+ } else { -+ link->w = s->w; -+ link->h = s->h; -+ } -+ link->sample_aspect_ratio = s->out_sar; -+ -+ return 0; -+} -+ -+static int filter_frame(AVFilterLink *link, AVFrame *frame_in) -+{ -+ FilterLink *l = ff_filter_link(link); -+ AVFilterContext *ctx = link->dst; -+ CropCUDAContext *s = ctx->priv; -+ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(link->format); -+ AVFrame *frame_end; -+ int ret; -+ -+ frame_end = av_frame_alloc(); -+ if (!frame_end) -+ return AVERROR(ENOMEM); -+ -+ s->var_values[VAR_N] = l->frame_count_out; -+ s->var_values[VAR_T] = frame_in->pts == AV_NOPTS_VALUE ? -+ NAN : frame_in->pts * av_q2d(link->time_base); -+#if FF_API_FRAME_PKT -+FF_DISABLE_DEPRECATION_WARNINGS -+ s->var_values[VAR_POS] = frame_in->pkt_pos == -1 ? NAN : frame_in->pkt_pos; -+FF_ENABLE_DEPRECATION_WARNINGS -+#endif -+ s->var_values[VAR_X] = av_expr_eval(s->x_pexpr, s->var_values, NULL); -+ s->var_values[VAR_Y] = av_expr_eval(s->y_pexpr, s->var_values, NULL); -+ s->var_values[VAR_X] = av_expr_eval(s->x_pexpr, s->var_values, NULL); -+ -+ normalize_double(&s->x, s->var_values[VAR_X]); -+ normalize_double(&s->y, s->var_values[VAR_Y]); -+ -+ if (s->x < 0) -+ s->x = 0; -+ if (s->y < 0) -+ s->y = 0; -+ if ((unsigned)s->x + (unsigned)s->w > link->w) -+ s->x = link->w - s->w; -+ if ((unsigned)s->y + (unsigned)s->h > link->h) -+ s->y = link->h - s->h; -+ if (!s->exact) { -+ s->x &= ~((1 << s->hsub) - 1); -+ s->y &= ~((1 << s->vsub) - 1); -+ } -+ -+ av_log(ctx, AV_LOG_TRACE, "n:%d t:%f x:%d y:%d x+w:%d y+h:%d\n", -+ (int)s->var_values[VAR_N], s->var_values[VAR_T], -+ s->x, s->y, s->x + s->w, s->y + s->h); -+ -+ { -+ CUcontext cuda_ctx = s->hwctx->cuda_ctx; -+ CudaFunctions *cu = s->hwctx->internal->cuda_dl; -+ CUstream cu_stream = s->hwctx->stream; -+ CUcontext dummy; -+ AVFrame *frame_out = s->frame; -+ int src_y_in_bytes = s->y * frame_in->linesize[0]; -+ int src_x_in_bytes = s->x * s->max_step[0]; -+ int err; -+ CUDA_MEMCPY2D cpy = { -+ .dstMemoryType = CU_MEMORYTYPE_DEVICE, -+ .dstPitch = frame_out->linesize[0], -+ .dstDevice = (CUdeviceptr)frame_out->data[0], -+ .srcMemoryType = CU_MEMORYTYPE_DEVICE, -+ .srcPitch = frame_in->linesize[0], -+ .srcDevice = (CUdeviceptr)(frame_in->data[0] + src_y_in_bytes + src_x_in_bytes), -+ .WidthInBytes = s->w * s->max_step[0], -+ .Height = s->h, -+ }; -+ -+ err = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); -+ if (err < 0) { -+ av_frame_free(&frame_end); -+ return err; -+ } -+ -+ err = CHECK_CU(cu->cuMemcpy2DAsync(&cpy, cu_stream)); -+ if (err < 0) { -+ av_log(ctx, AV_LOG_ERROR, "cuMemcpy2D error on plane [0]\n"); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ av_frame_free(&frame_end); -+ return err; -+ } -+ -+ if (!(desc->flags & AV_PIX_FMT_FLAG_PAL)) { -+ for (int i = 1; i < 3; i++) { -+ if (frame_in->data[i]) { -+ src_y_in_bytes = (s->y >> s->vsub) * frame_in->linesize[i]; -+ src_x_in_bytes = (s->x * s->max_step[i]) >> s->hsub; -+ -+ cpy.dstPitch = frame_out->linesize[i]; -+ cpy.dstDevice = (CUdeviceptr)frame_out->data[i]; -+ cpy.srcPitch = frame_in->linesize[i]; -+ cpy.srcDevice = (CUdeviceptr)(frame_in->data[i] + src_y_in_bytes + src_x_in_bytes); -+ cpy.WidthInBytes = (s->w * s->max_step[i]) >> s->hsub; -+ cpy.Height = s->h >> s->vsub; -+ -+ err = CHECK_CU(cu->cuMemcpy2DAsync(&cpy, cu_stream)); -+ if (err < 0) { -+ av_log(ctx, AV_LOG_ERROR, "cuMemcpy2D error on plane [%d]\n", i); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ av_frame_free(&frame_end); -+ return err; -+ } -+ } -+ } -+ } -+ -+ if (frame_in->data[3]) { -+ src_y_in_bytes = s->y * frame_in->linesize[3]; -+ src_x_in_bytes = s->x * s->max_step[3]; -+ -+ cpy.dstPitch = frame_out->linesize[3]; -+ cpy.dstDevice = (CUdeviceptr)frame_out->data[3]; -+ cpy.srcPitch = frame_in->linesize[3]; -+ cpy.srcDevice = (CUdeviceptr)(frame_in->data[3] + src_y_in_bytes + src_x_in_bytes); -+ cpy.WidthInBytes = s->w * s->max_step[3]; -+ cpy.Height = s->h; -+ -+ err = CHECK_CU(cu->cuMemcpy2DAsync(&cpy, cu_stream)); -+ if (err < 0) { -+ av_log(ctx, AV_LOG_ERROR, "cuMemcpy2D error on plane [3]\n"); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ av_frame_free(&frame_end); -+ return err; -+ } -+ } -+ -+ err = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ if (err < 0) { -+ av_frame_free(&frame_end); -+ return err; -+ } -+ } -+ -+ ret = av_hwframe_get_buffer(s->frame->hw_frames_ctx, s->tmp_frame, 0); -+ if (ret < 0) { -+ av_frame_free(&frame_end); -+ return ret; -+ } -+ -+ av_frame_move_ref(frame_end, s->frame); -+ av_frame_move_ref(s->frame, s->tmp_frame); -+ -+ s->frame->width = s->w; -+ s->frame->height = s->h; -+ -+ ret = av_frame_copy_props(frame_end, frame_in); -+ if (ret < 0) { -+ av_frame_free(&frame_in); -+ av_frame_free(&frame_end); -+ return ret; -+ } -+ -+ av_frame_free(&frame_in); -+ return ff_filter_frame(ctx->outputs[0], frame_end); -+} -+ -+static int process_command(AVFilterContext *ctx, const char *cmd, const char *args, -+ char *res, int res_len, int flags) -+{ -+ CropCUDAContext *s = ctx->priv; -+ int ret; -+ -+ if (!strcmp(cmd, "out_w") || !strcmp(cmd, "w") || -+ !strcmp(cmd, "out_h") || !strcmp(cmd, "h") || -+ !strcmp(cmd, "x") || !strcmp(cmd, "y")) { -+ int old_x = s->x; -+ int old_y = s->y; -+ int old_w = s->w; -+ int old_h = s->h; -+ AVFilterLink *outlink = ctx->outputs[0]; -+ AVFilterLink *inlink = ctx->inputs[0]; -+ -+ av_opt_set(s, cmd, args, 0); -+ -+ if ((ret = config_input(inlink)) < 0) { -+ s->x = old_x; -+ s->y = old_y; -+ s->w = old_w; -+ s->h = old_h; -+ return ret; -+ } -+ -+ ret = config_output(outlink); -+ } else { -+ ret = AVERROR(ENOSYS); -+ } -+ -+ return ret; -+} -+ -+#define OFFSET(x) offsetof(CropCUDAContext, x) -+#define FLAGS AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM -+#define TFLAGS AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM | AV_OPT_FLAG_RUNTIME_PARAM -+ -+static const AVOption crop_cuda_options[] = { -+ { "out_w", "set the width crop area expression", OFFSET(w_expr), AV_OPT_TYPE_STRING, { .str = "iw" }, 0, 0, TFLAGS }, -+ { "w", "set the width crop area expression", OFFSET(w_expr), AV_OPT_TYPE_STRING, { .str = "iw" }, 0, 0, TFLAGS }, -+ { "out_h", "set the height crop area expression", OFFSET(h_expr), AV_OPT_TYPE_STRING, { .str = "ih" }, 0, 0, TFLAGS }, -+ { "h", "set the height crop area expression", OFFSET(h_expr), AV_OPT_TYPE_STRING, { .str = "ih" }, 0, 0, TFLAGS }, -+ { "x", "set the x crop area expression", OFFSET(x_expr), AV_OPT_TYPE_STRING, { .str = "(in_w-out_w)/2" }, 0, 0, TFLAGS }, -+ { "y", "set the y crop area expression", OFFSET(y_expr), AV_OPT_TYPE_STRING, { .str = "(in_h-out_h)/2" }, 0, 0, TFLAGS }, -+ { "keep_aspect", "keep aspect ratio", OFFSET(keep_aspect), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS }, -+ { "exact", "do exact cropping", OFFSET(exact), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS }, -+ { NULL } -+}; -+ -+AVFILTER_DEFINE_CLASS(crop_cuda); -+ -+static const AVFilterPad crop_cuda_inputs[] = { -+ { -+ .name = "default", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .filter_frame = filter_frame, -+ .config_props = config_input, -+ }, -+}; -+ -+static const AVFilterPad crop_cuda_outputs[] = { -+ { -+ .name = "default", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .config_props = config_output, -+ }, -+}; -+ -+const AVFilter ff_vf_crop_cuda = { -+ .name = "crop_cuda", -+ .description = NULL_IF_CONFIG_SMALL("Crop the input CUDA video."), -+ .priv_size = sizeof(CropCUDAContext), -+ .priv_class = &crop_cuda_class, -+ .init = init, -+ .uninit = uninit, -+ FILTER_INPUTS(crop_cuda_inputs), -+ FILTER_OUTPUTS(crop_cuda_outputs), -+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), -+ .process_command = process_command, -+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, -+}; -diff --git a/libavfilter/vf_overlay_cuda.c b/libavfilter/vf_overlay_cuda.c -index a35f6ed..ac5a789 100644 ---- a/libavfilter/vf_overlay_cuda.c -+++ b/libavfilter/vf_overlay_cuda.c -@@ -495,6 +495,7 @@ static int overlay_cuda_config_output(AVFilterLink *outlink) - ctx->hwctx = ((AVHWDeviceContext*)ctx->hw_device_ctx->data)->hwctx; - - cuda_ctx = ctx->hwctx->cuda_ctx; -+ ctx->cu_ctx = cuda_ctx; - ctx->fs.time_base = inlink->time_base; - - ctx->cu_stream = ctx->hwctx->stream; -diff --git a/libavfilter/vf_overlay_many_cuda.c b/libavfilter/vf_overlay_many_cuda.c -new file mode 100644 -index 0000000..da5b915 ---- /dev/null -+++ b/libavfilter/vf_overlay_many_cuda.c -@@ -0,0 +1,559 @@ -+/** -+ * @file -+ * Overlay multiple videos on top of each other using CUDA hardware acceleration -+ */ -+ -+#include -+#include -+ -+#include "libavutil/avstring.h" -+#include "libavutil/cuda_check.h" -+#include "libavutil/hwcontext.h" -+#include "libavutil/hwcontext_cuda_internal.h" -+#include "libavutil/log.h" -+#include "libavutil/opt.h" -+#include "libavutil/pixdesc.h" -+ -+#include "avfilter.h" -+#include "avfilter_internal.h" -+#include "filters.h" -+#include "framesync.h" -+#include "video.h" -+ -+#include "cuda/load_helper.h" -+#include "vf_overlay_many_cuda.h" -+ -+#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, ctx->hwctx->internal->cuda_dl, x) -+#define DIV_UP(a, b) (((a) + (b) - 1) / (b)) -+ -+#define BLOCK_X 32 -+#define BLOCK_Y 16 -+ -+#define MAX_OVERLAY_COUNT OVERLAY_MANY_CUDA_MAX_OVERLAYS -+#define MAX_INPUT_COUNT (MAX_OVERLAY_COUNT + 1) -+ -+static const enum AVPixelFormat supported_main_formats[] = { -+ AV_PIX_FMT_YUV420P, -+ AV_PIX_FMT_YUV444P, -+ AV_PIX_FMT_NONE, -+}; -+ -+static const enum AVPixelFormat supported_overlay_formats[] = { -+ AV_PIX_FMT_YUVA420P, -+ AV_PIX_FMT_YUVA444P, -+ AV_PIX_FMT_NONE, -+}; -+ -+typedef struct OverlayManyCUDAContext { -+ const AVClass *class; -+ -+ AVBufferRef *hw_device_ctx; -+ AVBufferRef *frames_ctx; -+ AVCUDADeviceContext *hwctx; -+ -+ CUcontext cu_ctx; -+ CUmodule cu_module; -+ CUfunction cu_func; -+ CUstream cu_stream; -+ -+ FFFrameSync fs; -+ -+ int nb_inputs; -+ enum AVPixelFormat format_main; -+ enum AVPixelFormat format_overlay; -+} OverlayManyCUDAContext; -+ -+static int format_is_supported(const enum AVPixelFormat formats[], -+ enum AVPixelFormat format) -+{ -+ for (int i = 0; formats[i] != AV_PIX_FMT_NONE; i++) { -+ if (formats[i] == format) -+ return 1; -+ } -+ return 0; -+} -+ -+static int formats_match(enum AVPixelFormat format_main, -+ enum AVPixelFormat format_overlay) -+{ -+ switch (format_main) { -+ case AV_PIX_FMT_YUV420P: -+ return format_overlay == AV_PIX_FMT_YUVA420P || -+ format_overlay == AV_PIX_FMT_YUVA444P; -+ case AV_PIX_FMT_YUV444P: -+ return format_overlay == AV_PIX_FMT_YUVA444P; -+ default: -+ return 0; -+ } -+} -+ -+static int pack_frame(OverlayManyCUDAFrame *descriptor, -+ const AVFrame *frame, int plane_count) -+{ -+ for (int plane = 0; plane < plane_count; plane++) { -+ if (!frame->data[plane] || frame->linesize[plane] <= 0) -+ return AVERROR(EINVAL); -+ -+ descriptor->plane[plane].ptr = (uint64_t)(uintptr_t)frame->data[plane]; -+ descriptor->plane[plane].pitch = frame->linesize[plane]; -+ } -+ -+ return 0; -+} -+ -+static int pack_launch_params(OverlayManyCUDAParams *params, -+ const AVFrame *output, -+ const AVFrame *main, -+ AVFrame *const overlays[], -+ unsigned int overlay_count) -+{ -+ int ret; -+ -+ if (main->width <= 0 || main->height <= 0 || -+ overlay_count > MAX_OVERLAY_COUNT) -+ return AVERROR(EINVAL); -+ -+ memset(params, 0, sizeof(*params)); -+ params->width = main->width; -+ params->height = main->height; -+ params->overlay_count = overlay_count; -+ -+ ret = pack_frame(¶ms->dst, output, 3); -+ if (ret < 0) -+ return ret; -+ ret = pack_frame(¶ms->main, main, 3); -+ if (ret < 0) -+ return ret; -+ -+ for (unsigned int i = 0; i < overlay_count; i++) { -+ if (overlays[i]->width < main->width || -+ overlays[i]->height < main->height) -+ return AVERROR(EINVAL); -+ -+ ret = pack_frame(¶ms->overlay[i], overlays[i], 4); -+ if (ret < 0) -+ return ret; -+ } -+ -+ for (int dst_plane = 0; dst_plane < 3; dst_plane++) { -+ for (int src_plane = 0; src_plane < 3; src_plane++) { -+ if (output->data[dst_plane] == main->data[src_plane]) -+ return AVERROR(EINVAL); -+ } -+ for (unsigned int i = 0; i < overlay_count; i++) { -+ for (int src_plane = 0; src_plane < 4; src_plane++) { -+ if (output->data[dst_plane] == overlays[i]->data[src_plane]) -+ return AVERROR(EINVAL); -+ } -+ } -+ } -+ -+ return 0; -+} -+ -+static int overlay_many_cuda_blend(FFFrameSync *fs) -+{ -+ AVFilterContext *avctx = fs->parent; -+ OverlayManyCUDAContext *ctx = avctx->priv; -+ AVFilterLink *outlink = avctx->outputs[0]; -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext dummy; -+ AVFrame *main = NULL; -+ AVFrame *output = NULL; -+ AVFrame *overlays[MAX_OVERLAY_COUNT]; -+ OverlayManyCUDAParams params; -+ unsigned int overlay_count = 0; -+ int context_pushed = 0; -+ int ret; -+ -+ ret = ff_framesync_get_frame(fs, 0, &main, 1); -+ if (ret < 0) -+ return ret; -+ if (!main) -+ return AVERROR_BUG; -+ -+ for (int i = 1; i < ctx->nb_inputs; i++) { -+ AVFrame *overlay = NULL; -+ -+ ret = ff_framesync_get_frame(fs, i, &overlay, 0); -+ if (ret < 0) -+ goto fail; -+ if (overlay) -+ overlays[overlay_count++] = overlay; -+ } -+ -+ main->pts = av_rescale_q(fs->pts, fs->time_base, outlink->time_base); -+ if (avctx->is_disabled) -+ overlay_count = 0; -+ -+ output = ff_get_video_buffer(outlink, outlink->w, outlink->h); -+ if (!output) { -+ ret = AVERROR(ENOMEM); -+ goto fail; -+ } -+ -+ ret = av_frame_copy_props(output, main); -+ if (ret < 0) -+ goto fail; -+ -+ ret = pack_launch_params(¶ms, output, main, overlays, overlay_count); -+ if (ret < 0) { -+ av_log(ctx, AV_LOG_ERROR, "Invalid CUDA frame layout for overlay\n"); -+ goto fail; -+ } -+ -+ ret = CHECK_CU(cu->cuCtxPushCurrent(ctx->cu_ctx)); -+ if (ret < 0) -+ goto fail; -+ context_pushed = 1; -+ -+ { -+ void *kernel_args[] = { ¶ms }; -+ -+ ret = CHECK_CU(cu->cuLaunchKernel( -+ ctx->cu_func, -+ DIV_UP(params.width, BLOCK_X), -+ DIV_UP(params.height, BLOCK_Y), -+ 1, BLOCK_X, BLOCK_Y, 1, -+ 0, ctx->cu_stream, kernel_args, NULL)); -+ } -+ -+ { -+ int pop_ret = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ -+ context_pushed = 0; -+ if (ret >= 0) -+ ret = pop_ret; -+ } -+ -+ if (ret < 0) -+ goto fail; -+ -+ av_frame_free(&main); -+ return ff_filter_frame(outlink, output); -+ -+fail: -+ if (context_pushed) { -+ int pop_ret = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ -+ if (ret >= 0) -+ ret = pop_ret; -+ } -+ av_frame_free(&main); -+ av_frame_free(&output); -+ return ret; -+} -+ -+static av_cold int overlay_many_cuda_init(AVFilterContext *avctx) -+{ -+ OverlayManyCUDAContext *ctx = avctx->priv; -+ -+ ctx->fs.on_event = overlay_many_cuda_blend; -+ -+ for (int i = 0; i < ctx->nb_inputs; i++) { -+ AVFilterPad pad = { 0 }; -+ int ret; -+ -+ pad.type = AVMEDIA_TYPE_VIDEO; -+ pad.name = av_asprintf("input%d", i); -+ if (!pad.name) -+ return AVERROR(ENOMEM); -+ -+ ret = ff_append_inpad_free_name(avctx, &pad); -+ if (ret < 0) -+ return ret; -+ } -+ -+ return 0; -+} -+ -+static av_cold void overlay_many_cuda_uninit(AVFilterContext *avctx) -+{ -+ OverlayManyCUDAContext *ctx = avctx->priv; -+ -+ ff_framesync_uninit(&ctx->fs); -+ -+ if (ctx->hwctx && ctx->cu_module) { -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext dummy; -+ int ret; -+ -+ ret = CHECK_CU(cu->cuCtxPushCurrent(ctx->cu_ctx)); -+ if (ret >= 0) { -+ CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ } -+ } -+ -+ av_buffer_unref(&ctx->frames_ctx); -+ av_buffer_unref(&ctx->hw_device_ctx); -+ ctx->hwctx = NULL; -+} -+ -+static int overlay_many_cuda_activate(AVFilterContext *avctx) -+{ -+ OverlayManyCUDAContext *ctx = avctx->priv; -+ -+ return ff_framesync_activate(&ctx->fs); -+} -+ -+static int overlay_many_cuda_config_output(AVFilterLink *outlink) -+{ -+ extern const unsigned char ff_vf_overlay_many_cuda_ptx_data[]; -+ extern const unsigned int ff_vf_overlay_many_cuda_ptx_len; -+ -+ FilterLink *outl = ff_filter_link(outlink); -+ AVFilterContext *avctx = outlink->src; -+ OverlayManyCUDAContext *ctx = avctx->priv; -+ AVFilterLink *inlink_main = avctx->inputs[0]; -+ FilterLink *inl_main = ff_filter_link(inlink_main); -+ AVHWFramesContext *frames_ctx_main; -+ CudaFunctions *cu; -+ CUcontext dummy; -+ const char *function_name; -+ int err; -+ -+ if (avctx->nb_inputs != ctx->nb_inputs) { -+ av_log(ctx, AV_LOG_ERROR, -+ "The inputs option (%d) does not match the input pad count (%d)\n", -+ ctx->nb_inputs, avctx->nb_inputs); -+ return AVERROR_BUG; -+ } -+ -+ if (!inl_main->hw_frames_ctx) { -+ av_log(ctx, AV_LOG_ERROR, "No hardware frame context on main input\n"); -+ return AVERROR(EINVAL); -+ } -+ frames_ctx_main = (AVHWFramesContext *)inl_main->hw_frames_ctx->data; -+ if (!frames_ctx_main->device_ref) { -+ av_log(ctx, AV_LOG_ERROR, "No CUDA device context on main input\n"); -+ return AVERROR(EINVAL); -+ } -+ -+ ctx->format_main = frames_ctx_main->sw_format; -+ if (!format_is_supported(supported_main_formats, ctx->format_main)) { -+ av_log(ctx, AV_LOG_ERROR, "Unsupported main input format: %s\n", -+ av_get_pix_fmt_name(ctx->format_main)); -+ return AVERROR(ENOSYS); -+ } -+ -+ for (int i = 1; i < ctx->nb_inputs; i++) { -+ AVFilterLink *inlink_overlay = avctx->inputs[i]; -+ FilterLink *inl_overlay = ff_filter_link(inlink_overlay); -+ AVHWFramesContext *frames_ctx_overlay; -+ -+ if (!inl_overlay->hw_frames_ctx) { -+ av_log(ctx, AV_LOG_ERROR, -+ "No hardware frame context on overlay input %d\n", i); -+ return AVERROR(EINVAL); -+ } -+ frames_ctx_overlay = -+ (AVHWFramesContext *)inl_overlay->hw_frames_ctx->data; -+ if (!frames_ctx_overlay->device_ref) { -+ av_log(ctx, AV_LOG_ERROR, -+ "No CUDA device context on overlay input %d\n", i); -+ return AVERROR(EINVAL); -+ } -+ -+ if (!format_is_supported(supported_overlay_formats, -+ frames_ctx_overlay->sw_format)) { -+ av_log(ctx, AV_LOG_ERROR, "Unsupported overlay input %d format: %s\n", -+ i, av_get_pix_fmt_name(frames_ctx_overlay->sw_format)); -+ return AVERROR(ENOSYS); -+ } -+ -+ if (i == 1) -+ ctx->format_overlay = frames_ctx_overlay->sw_format; -+ else if (ctx->format_overlay != frames_ctx_overlay->sw_format) { -+ av_log(ctx, AV_LOG_ERROR, -+ "All overlay inputs must use the same software format\n"); -+ return AVERROR(EINVAL); -+ } -+ -+ if (frames_ctx_overlay->device_ref->data != -+ frames_ctx_main->device_ref->data) { -+ av_log(ctx, AV_LOG_ERROR, -+ "Overlay input %d uses a different CUDA device context\n", i); -+ return AVERROR(EINVAL); -+ } -+ -+ if (inlink_overlay->w < inlink_main->w || -+ inlink_overlay->h < inlink_main->h) { -+ av_log(ctx, AV_LOG_ERROR, -+ "Overlay input %d (%dx%d) is smaller than the main input " -+ "(%dx%d)\n", -+ i, inlink_overlay->w, inlink_overlay->h, -+ inlink_main->w, inlink_main->h); -+ return AVERROR(ENOSYS); -+ } -+ } -+ -+ if (!formats_match(ctx->format_main, ctx->format_overlay)) { -+ av_log(ctx, AV_LOG_ERROR, "Cannot overlay %s on %s\n", -+ av_get_pix_fmt_name(ctx->format_overlay), -+ av_get_pix_fmt_name(ctx->format_main)); -+ return AVERROR(EINVAL); -+ } -+ -+ outlink->w = inlink_main->w; -+ outlink->h = inlink_main->h; -+ outlink->time_base = inlink_main->time_base; -+ outlink->sample_aspect_ratio = inlink_main->sample_aspect_ratio; -+ ff_filter_link(outlink)->frame_rate = -+ ff_filter_link(inlink_main)->frame_rate; -+ -+ ctx->hw_device_ctx = av_buffer_ref(frames_ctx_main->device_ref); -+ if (!ctx->hw_device_ctx) -+ return AVERROR(ENOMEM); -+ ctx->hwctx = ((AVHWDeviceContext *)ctx->hw_device_ctx->data)->hwctx; -+ ctx->cu_ctx = ctx->hwctx->cuda_ctx; -+ ctx->cu_stream = ctx->hwctx->stream; -+ cu = ctx->hwctx->internal->cuda_dl; -+ -+ { -+ int compute_major; -+ -+ err = CHECK_CU(cu->cuDeviceGetAttribute( -+ &compute_major, -+ 75 /* CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR */, -+ ctx->hwctx->internal->cuda_device)); -+ if (err < 0) -+ return err; -+ if (compute_major < 7) { -+ av_log(ctx, AV_LOG_ERROR, -+ "overlay_many_cuda requires CUDA compute capability 7.0 " -+ "or newer\n"); -+ return AVERROR(ENOSYS); -+ } -+ } -+ -+ { -+ AVBufferRef *out_ref = av_hwframe_ctx_alloc(ctx->hw_device_ctx); -+ AVHWFramesContext *out_ctx; -+ -+ if (!out_ref) -+ return AVERROR(ENOMEM); -+ -+ out_ctx = (AVHWFramesContext *)out_ref->data; -+ out_ctx->format = AV_PIX_FMT_CUDA; -+ out_ctx->sw_format = ctx->format_main; -+ out_ctx->width = outlink->w; -+ out_ctx->height = outlink->h; -+ -+ err = av_hwframe_ctx_init(out_ref); -+ if (err < 0) { -+ av_buffer_unref(&out_ref); -+ return err; -+ } -+ -+ ctx->frames_ctx = out_ref; -+ outl->hw_frames_ctx = av_buffer_ref(ctx->frames_ctx); -+ if (!outl->hw_frames_ctx) -+ return AVERROR(ENOMEM); -+ } -+ -+ if (ctx->format_main == AV_PIX_FMT_YUV420P && -+ ctx->format_overlay == AV_PIX_FMT_YUVA420P) -+ function_name = "OverlayMany420"; -+ else if (ctx->format_main == AV_PIX_FMT_YUV444P) -+ function_name = "OverlayMany444"; -+ else -+ function_name = "OverlayMany420From444"; -+ -+ err = CHECK_CU(cu->cuCtxPushCurrent(ctx->cu_ctx)); -+ if (err < 0) -+ return err; -+ -+ err = ff_cuda_load_module(ctx, ctx->hwctx, &ctx->cu_module, -+ ff_vf_overlay_many_cuda_ptx_data, -+ ff_vf_overlay_many_cuda_ptx_len); -+ if (err >= 0) { -+ err = CHECK_CU(cu->cuModuleGetFunction( -+ &ctx->cu_func, ctx->cu_module, function_name)); -+ if (err < 0) -+ av_log(ctx, AV_LOG_FATAL, "CUDA function %s is unavailable\n", -+ function_name); -+ } -+ -+ { -+ int pop_ret = CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ -+ if (err >= 0) -+ err = pop_ret; -+ } -+ if (err < 0) -+ return err; -+ -+ { -+ FFFrameSync *fs = &ctx->fs; -+ -+ err = ff_framesync_init(fs, avctx, ctx->nb_inputs); -+ if (err < 0) -+ return err; -+ -+ fs->in[0].time_base = avctx->inputs[0]->time_base; -+ fs->in[0].sync = ctx->nb_inputs; -+ fs->in[0].before = EXT_STOP; -+ fs->in[0].after = EXT_INFINITY; -+ -+ for (int i = 1; i < ctx->nb_inputs; i++) { -+ fs->in[i].time_base = avctx->inputs[i]->time_base; -+ fs->in[i].sync = ctx->nb_inputs - i; -+ fs->in[i].before = EXT_NULL; -+ fs->in[i].after = EXT_INFINITY; -+ } -+ } -+ -+ return ff_framesync_configure(&ctx->fs); -+} -+ -+#define OFFSET(x) offsetof(OverlayManyCUDAContext, x) -+#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM) -+ -+static const AVOption overlay_many_cuda_options[] = { -+ { "inputs", "set number of inputs", OFFSET(nb_inputs), AV_OPT_TYPE_INT, -+ { .i64 = 2 }, 2, MAX_INPUT_COUNT, FLAGS }, -+ { "eof_action", "action on secondary input EOF", -+ OFFSET(fs.opt_eof_action), AV_OPT_TYPE_INT, -+ { .i64 = EOF_ACTION_REPEAT }, EOF_ACTION_REPEAT, EOF_ACTION_PASS, -+ FLAGS, "eof_action" }, -+ { "repeat", "repeat the previous frame", 0, AV_OPT_TYPE_CONST, -+ { .i64 = EOF_ACTION_REPEAT }, .flags = FLAGS, -+ .unit = "eof_action" }, -+ { "endall", "end both streams", 0, AV_OPT_TYPE_CONST, -+ { .i64 = EOF_ACTION_ENDALL }, .flags = FLAGS, -+ .unit = "eof_action" }, -+ { "pass", "pass through the main input", 0, AV_OPT_TYPE_CONST, -+ { .i64 = EOF_ACTION_PASS }, .flags = FLAGS, -+ .unit = "eof_action" }, -+ { "shortest", "force termination when the shortest input terminates", -+ OFFSET(fs.opt_shortest), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS }, -+ { "repeatlast", "repeat the last overlay frame", -+ OFFSET(fs.opt_repeatlast), AV_OPT_TYPE_BOOL, { .i64 = 1 }, 0, 1, FLAGS }, -+ { NULL } -+}; -+ -+FRAMESYNC_DEFINE_CLASS(overlay_many_cuda, OverlayManyCUDAContext, fs); -+ -+static const AVFilterPad overlay_many_cuda_outputs[] = { -+ { -+ .name = "default", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .config_props = overlay_many_cuda_config_output, -+ }, -+}; -+ -+const AVFilter ff_vf_overlay_many_cuda = { -+ .name = "overlay_many_cuda", -+ .description = NULL_IF_CONFIG_SMALL( -+ "Overlay multiple videos on top of each other using CUDA"), -+ .priv_size = sizeof(OverlayManyCUDAContext), -+ .priv_class = &overlay_many_cuda_class, -+ .init = overlay_many_cuda_init, -+ .uninit = overlay_many_cuda_uninit, -+ .activate = overlay_many_cuda_activate, -+ FILTER_OUTPUTS(overlay_many_cuda_outputs), -+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), -+ .preinit = overlay_many_cuda_framesync_preinit, -+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, -+}; -diff --git a/libavfilter/vf_overlay_many_cuda.cu b/libavfilter/vf_overlay_many_cuda.cu -new file mode 100644 -index 0000000..46d45df ---- /dev/null -+++ b/libavfilter/vf_overlay_many_cuda.cu -@@ -0,0 +1,252 @@ -+/* -+ * This file is part of FFmpeg. -+ * -+ * FFmpeg is free software; you can redistribute it and/or -+ * modify it under the terms of the GNU Lesser General Public -+ * License as published by the Free Software Foundation; either -+ * version 2.1 of the License, or (at your option) any later version. -+ * -+ * FFmpeg is distributed in the hope that it will be useful, -+ * but WITHOUT ANY WARRANTY; without even the implied warranty of -+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU -+ * Lesser General Public License for more details. -+ * -+ * You should have received a copy of the GNU Lesser General Public -+ * License along with FFmpeg; if not, write to the Free Software -+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA -+ */ -+ -+#include "vf_overlay_many_cuda.h" -+ -+#if !defined(__CUDACC_VER_MAJOR__) || \ -+ (__CUDACC_VER_MAJOR__ < 11) || \ -+ (__CUDACC_VER_MAJOR__ == 11 && __CUDACC_VER_MINOR__ < 7) -+#error "overlay_many_cuda requires CUDA Toolkit 11.7 or newer" -+#endif -+ -+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ < 700 -+#error "overlay_many_cuda requires compute capability 7.0 or newer" -+#endif -+ -+static __device__ __forceinline__ const unsigned char * -+plane_src(const OverlayManyCUDAPlane &plane) -+{ -+ return (const unsigned char *)(uintptr_t)plane.ptr; -+} -+ -+static __device__ __forceinline__ unsigned char * -+plane_dst(const OverlayManyCUDAPlane &plane) -+{ -+ return (unsigned char *)(uintptr_t)plane.ptr; -+} -+ -+static __device__ __forceinline__ unsigned int -+load_sample(const OverlayManyCUDAPlane &plane, unsigned int x, unsigned int y) -+{ -+ return plane_src(plane)[x + (size_t)y * plane.pitch]; -+} -+ -+static __device__ __forceinline__ void -+store_sample(const OverlayManyCUDAPlane &plane, unsigned int x, unsigned int y, -+ unsigned int value) -+{ -+ plane_dst(plane)[x + (size_t)y * plane.pitch] = (unsigned char)value; -+} -+ -+static __device__ __forceinline__ unsigned int -+blend_sample(unsigned int background, unsigned int foreground, -+ unsigned int alpha) -+{ -+ unsigned int sum = alpha * foreground + (255U - alpha) * background; -+ unsigned int rounded = sum + 128U; -+ -+ /* Exact rounded division by 255 without an integer division. */ -+ return (rounded + (rounded >> 8)) >> 8; -+} -+ -+static __device__ __forceinline__ unsigned int -+alpha_420(const OverlayManyCUDAFrame &overlay, unsigned int x, unsigned int y, -+ unsigned int width, unsigned int height) -+{ -+ const OverlayManyCUDAPlane &alpha = overlay.plane[3]; -+ unsigned int sum = load_sample(alpha, x, y); -+ unsigned int count = 1; -+ -+ if (x + 1 < width) { -+ sum += load_sample(alpha, x + 1, y); -+ count++; -+ } -+ if (y + 1 < height) { -+ sum += load_sample(alpha, x, y + 1); -+ count++; -+ if (x + 1 < width) { -+ sum += load_sample(alpha, x + 1, y + 1); -+ count++; -+ } -+ } -+ -+ return (sum + count / 2) / count; -+} -+ -+extern "C" { -+ -+__global__ void OverlayMany420( -+ const __grid_constant__ OverlayManyCUDAParams params) -+{ -+ unsigned int x = blockIdx.x * blockDim.x + threadIdx.x; -+ unsigned int y = blockIdx.y * blockDim.y + threadIdx.y; -+ -+ if (x >= params.width || y >= params.height) -+ return; -+ -+ unsigned int luma = load_sample(params.main.plane[0], x, y); -+ -+ for (unsigned int i = 0; i < params.overlay_count; i++) { -+ const OverlayManyCUDAFrame &overlay = params.overlay[i]; -+ unsigned int alpha = load_sample(overlay.plane[3], x, y); -+ -+ luma = blend_sample(luma, load_sample(overlay.plane[0], x, y), alpha); -+ } -+ store_sample(params.dst.plane[0], x, y, luma); -+ -+ if ((x & 1) || (y & 1)) -+ return; -+ -+ unsigned int cx = x >> 1; -+ unsigned int cy = y >> 1; -+ unsigned int chroma_u = load_sample(params.main.plane[1], cx, cy); -+ unsigned int chroma_v = load_sample(params.main.plane[2], cx, cy); -+ -+ for (unsigned int i = 0; i < params.overlay_count; i++) { -+ const OverlayManyCUDAFrame &overlay = params.overlay[i]; -+ unsigned int alpha = alpha_420(overlay, x, y, -+ params.width, params.height); -+ -+ chroma_u = blend_sample(chroma_u, -+ load_sample(overlay.plane[1], cx, cy), alpha); -+ chroma_v = blend_sample(chroma_v, -+ load_sample(overlay.plane[2], cx, cy), alpha); -+ } -+ -+ store_sample(params.dst.plane[1], cx, cy, chroma_u); -+ store_sample(params.dst.plane[2], cx, cy, chroma_v); -+} -+ -+__global__ void OverlayMany444( -+ const __grid_constant__ OverlayManyCUDAParams params) -+{ -+ unsigned int x = blockIdx.x * blockDim.x + threadIdx.x; -+ unsigned int y = blockIdx.y * blockDim.y + threadIdx.y; -+ -+ if (x >= params.width || y >= params.height) -+ return; -+ -+ unsigned int luma = load_sample(params.main.plane[0], x, y); -+ unsigned int chroma_u = load_sample(params.main.plane[1], x, y); -+ unsigned int chroma_v = load_sample(params.main.plane[2], x, y); -+ -+ for (unsigned int i = 0; i < params.overlay_count; i++) { -+ const OverlayManyCUDAFrame &overlay = params.overlay[i]; -+ unsigned int alpha = load_sample(overlay.plane[3], x, y); -+ -+ luma = blend_sample(luma, load_sample(overlay.plane[0], x, y), alpha); -+ chroma_u = blend_sample(chroma_u, -+ load_sample(overlay.plane[1], x, y), alpha); -+ chroma_v = blend_sample(chroma_v, -+ load_sample(overlay.plane[2], x, y), alpha); -+ } -+ -+ store_sample(params.dst.plane[0], x, y, luma); -+ store_sample(params.dst.plane[1], x, y, chroma_u); -+ store_sample(params.dst.plane[2], x, y, chroma_v); -+} -+ -+__global__ void OverlayMany420From444( -+ const __grid_constant__ OverlayManyCUDAParams params) -+{ -+ unsigned int x = blockIdx.x * blockDim.x + threadIdx.x; -+ unsigned int y = blockIdx.y * blockDim.y + threadIdx.y; -+ -+ if (x >= params.width || y >= params.height) -+ return; -+ -+ unsigned int luma = load_sample(params.main.plane[0], x, y); -+ -+ for (unsigned int i = 0; i < params.overlay_count; i++) { -+ const OverlayManyCUDAFrame &overlay = params.overlay[i]; -+ unsigned int alpha = load_sample(overlay.plane[3], x, y); -+ -+ luma = blend_sample(luma, load_sample(overlay.plane[0], x, y), alpha); -+ } -+ store_sample(params.dst.plane[0], x, y, luma); -+ -+ if ((x & 1) || (y & 1)) -+ return; -+ -+ unsigned int cx = x >> 1; -+ unsigned int cy = y >> 1; -+ unsigned int base_u = load_sample(params.main.plane[1], cx, cy); -+ unsigned int base_v = load_sample(params.main.plane[2], cx, cy); -+ -+ /* -+ * Compose all four 4:4:4 chroma positions independently, then downsample -+ * the final values. Downsampling each layer first changes alpha semantics. -+ */ -+ unsigned int u00 = base_u; -+ unsigned int u01 = base_u; -+ unsigned int u10 = base_u; -+ unsigned int u11 = base_u; -+ unsigned int v00 = base_v; -+ unsigned int v01 = base_v; -+ unsigned int v10 = base_v; -+ unsigned int v11 = base_v; -+ bool has_x1 = x + 1 < params.width; -+ bool has_y1 = y + 1 < params.height; -+ -+ for (unsigned int i = 0; i < params.overlay_count; i++) { -+ const OverlayManyCUDAFrame &overlay = params.overlay[i]; -+ unsigned int alpha = load_sample(overlay.plane[3], x, y); -+ -+ u00 = blend_sample(u00, load_sample(overlay.plane[1], x, y), alpha); -+ v00 = blend_sample(v00, load_sample(overlay.plane[2], x, y), alpha); -+ -+ if (has_x1) { -+ alpha = load_sample(overlay.plane[3], x + 1, y); -+ u01 = blend_sample(u01, -+ load_sample(overlay.plane[1], x + 1, y), alpha); -+ v01 = blend_sample(v01, -+ load_sample(overlay.plane[2], x + 1, y), alpha); -+ } -+ -+ if (has_y1) { -+ alpha = load_sample(overlay.plane[3], x, y + 1); -+ u10 = blend_sample(u10, -+ load_sample(overlay.plane[1], x, y + 1), alpha); -+ v10 = blend_sample(v10, -+ load_sample(overlay.plane[2], x, y + 1), alpha); -+ -+ if (has_x1) { -+ alpha = load_sample(overlay.plane[3], x + 1, y + 1); -+ u11 = blend_sample( -+ u11, load_sample(overlay.plane[1], x + 1, y + 1), alpha); -+ v11 = blend_sample( -+ v11, load_sample(overlay.plane[2], x + 1, y + 1), alpha); -+ } -+ } -+ } -+ -+ unsigned int sample_count = (1U + has_x1) * (1U + has_y1); -+ unsigned int sum_u = u00 + (has_x1 ? u01 : 0) + -+ (has_y1 ? u10 : 0) + -+ (has_x1 && has_y1 ? u11 : 0); -+ unsigned int sum_v = v00 + (has_x1 ? v01 : 0) + -+ (has_y1 ? v10 : 0) + -+ (has_x1 && has_y1 ? v11 : 0); -+ -+ store_sample(params.dst.plane[1], cx, cy, -+ (sum_u + sample_count / 2) / sample_count); -+ store_sample(params.dst.plane[2], cx, cy, -+ (sum_v + sample_count / 2) / sample_count); -+} -+ -+} /* extern "C" */ -diff --git a/libavfilter/vf_overlay_many_cuda.h b/libavfilter/vf_overlay_many_cuda.h -new file mode 100644 -index 0000000..4f42b88 ---- /dev/null -+++ b/libavfilter/vf_overlay_many_cuda.h -@@ -0,0 +1,68 @@ -+/* -+ * CUDA overlay-many launch descriptor -+ * -+ * This file is part of FFmpeg. -+ * -+ * FFmpeg is free software; you can redistribute it and/or -+ * modify it under the terms of the GNU Lesser General Public -+ * License as published by the Free Software Foundation; either -+ * version 2.1 of the License, or (at your option) any later version. -+ * -+ * FFmpeg is distributed in the hope that it will be useful, -+ * but WITHOUT ANY WARRANTY; without even the implied warranty of -+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU -+ * Lesser General Public License for more details. -+ * -+ * You should have received a copy of the GNU Lesser General Public -+ * License along with FFmpeg; if not, write to the Free Software -+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA -+ */ -+ -+#ifndef AVFILTER_VF_OVERLAY_MANY_CUDA_H -+#define AVFILTER_VF_OVERLAY_MANY_CUDA_H -+ -+#include -+#include -+ -+#define OVERLAY_MANY_CUDA_MAX_OVERLAYS 15 -+ -+typedef struct OverlayManyCUDAPlane { -+ uint64_t ptr; -+ uint32_t pitch; -+ uint32_t reserved; -+} OverlayManyCUDAPlane; -+ -+typedef struct OverlayManyCUDAFrame { -+ OverlayManyCUDAPlane plane[4]; -+} OverlayManyCUDAFrame; -+ -+typedef struct OverlayManyCUDAParams { -+ OverlayManyCUDAFrame dst; -+ OverlayManyCUDAFrame main; -+ uint32_t width; -+ uint32_t height; -+ uint32_t overlay_count; -+ uint32_t reserved; -+ OverlayManyCUDAFrame overlay[OVERLAY_MANY_CUDA_MAX_OVERLAYS]; -+} OverlayManyCUDAParams; -+ -+#ifdef __CUDACC__ -+#define OVERLAY_MANY_CUDA_STATIC_ASSERT static_assert -+#else -+#define OVERLAY_MANY_CUDA_STATIC_ASSERT _Static_assert -+#endif -+ -+OVERLAY_MANY_CUDA_STATIC_ASSERT(sizeof(OverlayManyCUDAPlane) == 16, -+ "unexpected CUDA plane descriptor layout"); -+OVERLAY_MANY_CUDA_STATIC_ASSERT(sizeof(uintptr_t) <= sizeof(uint64_t), -+ "CUDA device pointers do not fit the descriptor"); -+OVERLAY_MANY_CUDA_STATIC_ASSERT(sizeof(OverlayManyCUDAFrame) == 64, -+ "unexpected CUDA frame descriptor layout"); -+OVERLAY_MANY_CUDA_STATIC_ASSERT(offsetof(OverlayManyCUDAParams, overlay) == 144, -+ "unexpected CUDA overlay array offset"); -+OVERLAY_MANY_CUDA_STATIC_ASSERT(sizeof(OverlayManyCUDAParams) == 1104, -+ "unexpected CUDA launch parameter size"); -+ -+#undef OVERLAY_MANY_CUDA_STATIC_ASSERT -+ -+#endif /* AVFILTER_VF_OVERLAY_MANY_CUDA_H */ -diff --git a/libavfilter/vf_pad_cuda.c b/libavfilter/vf_pad_cuda.c -new file mode 100644 -index 0000000..7daa227 ---- /dev/null -+++ b/libavfilter/vf_pad_cuda.c -@@ -0,0 +1,538 @@ -+#include "libavutil/hwcontext.h" -+#include "libavutil/hwcontext_cuda_internal.h" -+#include "libavutil/cuda_check.h" -+#include "libavutil/opt.h" -+#include "libavutil/pixdesc.h" -+#include "libavutil/eval.h" -+#include "libavutil/colorspace.h" -+ -+#include "avfilter.h" -+#include "formats.h" -+#include "avfilter_internal.h" -+#include "video.h" -+ -+#include "cuda/load_helper.h" -+ -+#define BLOCK_X 32 -+#define BLOCK_Y 16 -+ -+enum var_name { -+ VAR_IN_W, VAR_IW, -+ VAR_IN_H, VAR_IH, -+ VAR_OUT_W, VAR_OW, -+ VAR_OUT_H, VAR_OH, -+ VAR_X, -+ VAR_Y, -+ VAR_A, -+ VAR_SAR, -+ VAR_DAR, -+ VARS_NB -+}; -+ -+static const char *const var_names[] = { -+ "in_w", "iw", -+ "in_h", "ih", -+ "out_w", "ow", -+ "out_h", "oh", -+ "x", -+ "y", -+ "a", -+ "sar", -+ "dar", -+ NULL -+}; -+ -+static enum AVPixelFormat supported_formats[] = { -+ AV_PIX_FMT_NV12, -+ AV_PIX_FMT_YUV420P, AV_PIX_FMT_YUVA420P, -+ AV_PIX_FMT_YUV444P, AV_PIX_FMT_YUVA444P, -+}; -+ -+#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) ) -+#define BLOCKX 32 -+#define BLOCKY 16 -+ -+#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, cu, x) -+ -+typedef struct PadCUDAContext { -+ const AVClass *class; -+ AVCUDADeviceContext *hwctx; -+ AVBufferRef *hw_device_ctx; -+ -+ AVBufferRef *frames_ctx; -+ AVFrame *frame; -+ AVFrame *tmp_frame; -+ -+ CUmodule cu_module; -+ CUfunction cu_func; -+ CUstream cu_stream; -+ -+ enum AVPixelFormat sw_format; -+ -+ char *w_expr, *h_expr, *x_expr, *y_expr; -+ AVRational aspect; -+ int w, h, x, y; -+ uint8_t pad_rgba[4]; -+ uint8_t pad_color[4]; -+} PadCUDAContext; -+ -+ -+ -+static int format_is_supported(enum AVPixelFormat fmt) -+{ -+ for (int i = 0; i < FF_ARRAY_ELEMS(supported_formats); i++) -+ if (supported_formats[i] == fmt) -+ return 1; -+ return 0; -+} -+ -+static av_cold int pad_cuda_init(AVFilterContext *avctx) -+{ -+ PadCUDAContext *ctx = avctx->priv; -+ -+ ctx->frame = av_frame_alloc(); -+ if (!ctx->frame) -+ return AVERROR(ENOMEM); -+ -+ ctx->tmp_frame = av_frame_alloc(); -+ if (!ctx->tmp_frame) -+ return AVERROR(ENOMEM); -+ -+ return 0; -+} -+ -+static av_cold void pad_cuda_uninit(AVFilterContext *avctx) -+{ -+ PadCUDAContext *ctx = avctx->priv; -+ -+ if (ctx->hwctx && ctx->cu_module) { -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext dummy; -+ -+ CHECK_CU(cu->cuCtxPushCurrent(ctx->hwctx->cuda_ctx)); -+ CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ } -+ -+ av_frame_free(&ctx->frame); -+ av_buffer_unref(&ctx->hw_device_ctx); -+ av_buffer_unref(&ctx->frames_ctx); -+ av_frame_free(&ctx->tmp_frame); -+ ctx->hwctx = NULL; -+} -+ -+static av_cold int pad_cuda_output_config_props(AVFilterLink *outlink) -+{ -+ extern const unsigned char ff_vf_pad_cuda_ptx_data[]; -+ extern const unsigned int ff_vf_pad_cuda_ptx_len; -+ -+ FilterLink *outl = ff_filter_link(outlink); -+ AVFilterContext *avctx = outlink->src; -+ AVFilterLink *inlink = outlink->src->inputs[0]; -+ FilterLink *inl = ff_filter_link(inlink); -+ PadCUDAContext *ctx = avctx->priv; -+ AVHWFramesContext *frames_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; -+ -+ AVRational adjusted_aspect = ctx->aspect; -+ double var_values[VARS_NB], res; -+ int err, ret; -+ -+ if (!frames_ctx) { -+ av_log(ctx, AV_LOG_ERROR, "No hw context provided on main input\n"); -+ return AVERROR(EINVAL); -+ } -+ -+ ctx->sw_format = frames_ctx->sw_format; -+ if (!format_is_supported(ctx->sw_format)) { -+ av_log(ctx, AV_LOG_ERROR, "Unsupported input format: %s\n", -+ av_get_pix_fmt_name(ctx->sw_format)); -+ return AVERROR(ENOSYS); -+ } -+ -+ // initialize -+ ctx->hw_device_ctx = av_buffer_ref(frames_ctx->device_ref); -+ if (!ctx->hw_device_ctx) -+ return AVERROR(ENOMEM); -+ ctx->hwctx = ((AVHWDeviceContext*)frames_ctx->device_ref->data)->hwctx; -+ ctx->cu_stream = ctx->hwctx->stream; -+ -+ // load functions -+ { -+ CUcontext cuda_ctx = ctx->hwctx->cuda_ctx; -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext dummy; -+ -+ err = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); -+ if (err < 0) { -+ return err; -+ } -+ -+ err = ff_cuda_load_module(ctx, ctx->hwctx, &ctx->cu_module, ff_vf_pad_cuda_ptx_data, ff_vf_pad_cuda_ptx_len); -+ if (err < 0) { -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return err; -+ } -+ -+ err = CHECK_CU(cu->cuModuleGetFunction(&ctx->cu_func, ctx->cu_module, "Pad_Cuda")); -+ if (err < 0) { -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return err; -+ } -+ -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ } -+ -+ // process filter parameters -+ { -+ const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(ctx->sw_format); -+ -+ var_values[VAR_IN_W] = var_values[VAR_IW] = inlink->w; -+ var_values[VAR_IN_H] = var_values[VAR_IH] = inlink->h; -+ var_values[VAR_OUT_W] = var_values[VAR_OW] = NAN; -+ var_values[VAR_OUT_H] = var_values[VAR_OH] = NAN; -+ var_values[VAR_A] = (double) inlink->w / inlink->h; -+ var_values[VAR_SAR] = inlink->sample_aspect_ratio.num ? -+ (double) inlink->sample_aspect_ratio.num / inlink->sample_aspect_ratio.den : 1; -+ var_values[VAR_DAR] = var_values[VAR_A] * var_values[VAR_SAR]; -+ -+ av_expr_parse_and_eval(&res, ctx->w_expr, -+ var_names, var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx); -+ ctx->w = var_values[VAR_OUT_W] = var_values[VAR_OW] = res; -+ if ((ret = av_expr_parse_and_eval(&res, ctx->h_expr, -+ var_names, var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) -+ return ret; -+ ctx->h = var_values[VAR_OUT_H] = var_values[VAR_OH] = res; -+ if (!ctx->h) -+ var_values[VAR_OUT_H] = var_values[VAR_OH] = ctx->h = inlink->h; -+ -+ /* evaluate the width again, as it may depend on the evaluated output height */ -+ if ((ret = av_expr_parse_and_eval(&res, ctx->w_expr, -+ var_names, var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) -+ return ret; -+ ctx->w = var_values[VAR_OUT_W] = var_values[VAR_OW] = res; -+ if (!ctx->w) -+ var_values[VAR_OUT_W] = var_values[VAR_OW] = ctx->w = inlink->w; -+ -+ if (adjusted_aspect.num && adjusted_aspect.den) { -+ adjusted_aspect = av_div_q(adjusted_aspect, inlink->sample_aspect_ratio); -+ if (ctx->h < av_rescale(ctx->w, adjusted_aspect.den, adjusted_aspect.num)) { -+ ctx->h = var_values[VAR_OUT_H] = var_values[VAR_OH] = av_rescale(ctx->w, adjusted_aspect.den, adjusted_aspect.num); -+ } else { -+ ctx->w = var_values[VAR_OUT_W] = var_values[VAR_OW] = av_rescale(ctx->h, adjusted_aspect.num, adjusted_aspect.den); -+ } -+ } -+ -+ /* evaluate x and y */ -+ av_expr_parse_and_eval(&res, ctx->x_expr, -+ var_names, var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx); -+ ctx->x = var_values[VAR_X] = res; -+ if ((ret = av_expr_parse_and_eval(&res, ctx->y_expr, -+ var_names, var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) -+ return ret; -+ ctx->y = var_values[VAR_Y] = res; -+ /* evaluate x again, as it may depend on the evaluated y value */ -+ if ((ret = av_expr_parse_and_eval(&res, ctx->x_expr, -+ var_names, var_values, -+ NULL, NULL, NULL, NULL, NULL, 0, ctx)) < 0) -+ return ret; -+ ctx->x = var_values[VAR_X] = res; -+ -+ if (ctx->x < 0 || ctx->x + inlink->w > ctx->w) -+ ctx->x = var_values[VAR_X] = (ctx->w - inlink->w) / 2; -+ if (ctx->y < 0 || ctx->y + inlink->h > ctx->h) -+ ctx->y = var_values[VAR_Y] = (ctx->h - inlink->h) / 2; -+ -+ /* sanity check params */ -+ if (ctx->w < inlink->w || ctx->h < inlink->h) { -+ av_log(ctx, AV_LOG_ERROR, "Padded dimensions cannot be smaller than input dimensions.\n"); -+ return AVERROR(EINVAL); -+ } -+ -+ // Align x, y offsets between planes -+ ctx->x &= ~((1 << desc->log2_chroma_w) - 1); -+ ctx->y &= ~((1 << desc->log2_chroma_h) - 1); -+ -+ ctx->pad_color[0] = RGB_TO_Y_BT709(ctx->pad_rgba[0], ctx->pad_rgba[1], ctx->pad_rgba[2]); -+ ctx->pad_color[1] = RGB_TO_U_BT709(ctx->pad_rgba[0], ctx->pad_rgba[1], ctx->pad_rgba[2], 0); -+ ctx->pad_color[2] = RGB_TO_V_BT709(ctx->pad_rgba[0], ctx->pad_rgba[1], ctx->pad_rgba[2], 0); -+ ctx->pad_color[3] = ctx->pad_rgba[3]; -+ } -+ -+ -+ outlink->w = ctx->w; -+ outlink->h = ctx->h; -+ -+ -+ // prepare output buffer -+ { -+ AVHWFramesContext *out_ctx; -+ AVBufferRef *out_ref = av_hwframe_ctx_alloc(ctx->hw_device_ctx); -+ if (!out_ref) -+ return AVERROR(ENOMEM); -+ -+ out_ctx = (AVHWFramesContext*)out_ref->data; -+ out_ctx->format = AV_PIX_FMT_CUDA; -+ out_ctx->sw_format = ctx->sw_format; -+ out_ctx->width = FFALIGN(ctx->w, 32); -+ out_ctx->height = FFALIGN(ctx->h, 32); -+ -+ ret = av_hwframe_ctx_init(out_ref); -+ if (ret < 0) -+ goto output_buffer_fail; -+ -+ av_frame_unref(ctx->frame); -+ ret = av_hwframe_get_buffer(out_ref, ctx->frame, 0); -+ if (ret < 0) -+ goto output_buffer_fail; -+ -+ ctx->frame->width = ctx->w; -+ ctx->frame->height = ctx->h; -+ -+ ctx->frames_ctx = out_ref; -+ -+ if (ret < 0) { -+output_buffer_fail: -+ av_buffer_unref(&out_ref); -+ return ret; -+ } -+ -+ outl->hw_frames_ctx = av_buffer_ref(inl->hw_frames_ctx); -+ if (!outl->hw_frames_ctx) { -+ return AVERROR(ENOMEM); -+ } -+ } -+ -+ return 0; -+} -+ -+static int pad_cuda_call_kernel( -+ PadCUDAContext *ctx, -+ uint8_t *main_data, int main_linesize, -+ int main_width, int main_height, -+ uint8_t *overlay_data, int overlay_linesize, -+ int overlay_width, int overlay_height, -+ int offset_x, int offset_y, -+ uint8_t fill_color1, uint8_t fill_color2) { -+ -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ -+ void* kernel_args[] = { -+ &main_data, &main_linesize, -+ &overlay_data, &overlay_linesize, -+ &overlay_width, &overlay_height, -+ &offset_x, &offset_y, -+ &fill_color1, &fill_color2 -+ }; -+ -+ return CHECK_CU(cu->cuLaunchKernel( -+ ctx->cu_func, -+ DIV_UP(main_width, BLOCK_X), DIV_UP(main_height, BLOCK_Y), 1, -+ BLOCK_X, BLOCK_Y, 1, -+ 0, ctx->cu_stream, kernel_args, NULL)); -+} -+ -+static int pad_cuda_fill_buffers(AVFilterContext *avctx, AVFrame *out, AVFrame *in) -+{ -+ PadCUDAContext *ctx = avctx->priv; -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext cuda_ctx = ctx->hwctx->cuda_ctx; -+ CUcontext dummy; -+ int ret; -+ uint8_t color_y = ctx->pad_color[0]; -+ uint8_t color_u = ctx->pad_color[1]; -+ uint8_t color_v = ctx->pad_color[2]; -+ uint8_t color_a = ctx->pad_color[3]; -+ -+ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); -+ if (ret < 0) -+ return ret; -+ -+ // overlay first plane -+ pad_cuda_call_kernel(ctx, -+ out->data[0], out->linesize[0], -+ out->width, out->height, -+ in->data[0], in->linesize[0], -+ in->width, in->height, -+ ctx->x, ctx->y, -+ color_y, color_y); -+ -+ // overlay color offset planes depending on the pixel format -+ switch(ctx->sw_format) { -+ case AV_PIX_FMT_NV12: -+ pad_cuda_call_kernel(ctx, -+ out->data[1], out->linesize[1], -+ out->width, out->height / 2, -+ in->data[1], in->linesize[1], -+ in->width, in->height / 2, -+ ctx->x, ctx->y / 2, -+ color_u, color_v); -+ break; -+ -+ case AV_PIX_FMT_YUV420P: -+ case AV_PIX_FMT_YUVA420P: -+ pad_cuda_call_kernel(ctx, -+ out->data[1], out->linesize[1], -+ out->width / 2, out->height / 2, -+ in->data[1], in->linesize[1], -+ in->width / 2, in->height / 2, -+ ctx->x / 2, ctx->y / 2, -+ color_u, color_u); -+ pad_cuda_call_kernel(ctx, -+ out->data[2], out->linesize[2], -+ out->width / 2, out->height / 2, -+ in->data[2], in->linesize[2], -+ in->width / 2, in->height / 2, -+ ctx->x / 2, ctx->y / 2, -+ color_v, color_v); -+ break; -+ -+ case AV_PIX_FMT_YUV444P: -+ case AV_PIX_FMT_YUVA444P: -+ pad_cuda_call_kernel(ctx, -+ out->data[1], out->linesize[1], -+ out->width, out->height, -+ in->data[1], in->linesize[1], -+ in->width, in->height, -+ ctx->x, ctx->y, -+ color_u, color_u); -+ pad_cuda_call_kernel(ctx, -+ out->data[2], out->linesize[2], -+ out->width, out->height, -+ in->data[2], in->linesize[2], -+ in->width, in->height, -+ ctx->x, ctx->y, -+ color_v, color_v); -+ break; -+ -+ default: -+ av_log(ctx, AV_LOG_ERROR, "Passed unsupported overlay pixel format\n"); -+ av_frame_free(&out); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return AVERROR_BUG; -+ } -+ -+ if (in->data[3]) { -+ pad_cuda_call_kernel(ctx, -+ out->data[3], out->linesize[3], -+ out->width, out->height, -+ in->data[3], in->linesize[3], -+ in->width, in->height, -+ ctx->x, ctx->y, -+ color_a, color_a); -+ } -+ -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return 0; -+} -+ -+static int pad_cuda_filter_frame(AVFilterLink *link, AVFrame *in) -+{ -+ AVFilterContext *avctx = link->dst; -+ PadCUDAContext *ctx = avctx->priv; -+ AVFilterLink *outlink = avctx->outputs[0]; -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ -+ AVFrame *out = NULL; -+ CUcontext dummy; -+ int ret = 0; -+ -+ out = av_frame_alloc(); -+ if (!out) { -+ ret = AVERROR(ENOMEM); -+ goto fail; -+ } -+ -+ ret = CHECK_CU(cu->cuCtxPushCurrent(ctx->hwctx->cuda_ctx)); -+ if (ret < 0) -+ goto fail; -+ -+ // fill and prepare the output frame -+ { -+ ret = pad_cuda_fill_buffers(avctx, ctx->frame, in); -+ if (ret < 0) -+ return ret; -+ -+ ret = av_hwframe_get_buffer(ctx->frame->hw_frames_ctx, ctx->tmp_frame, 0); -+ if (ret < 0) -+ return ret; -+ -+ av_frame_move_ref(out, ctx->frame); -+ av_frame_move_ref(ctx->frame, ctx->tmp_frame); -+ -+ ctx->frame->width = outlink->w; -+ ctx->frame->height = outlink->h; -+ -+ ret = av_frame_copy_props(out, in); -+ if (ret < 0) -+ return ret; -+ } -+ -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ if (ret < 0) -+ goto fail; -+ -+ av_reduce(&out->sample_aspect_ratio.num, &out->sample_aspect_ratio.den, -+ (int64_t)in->sample_aspect_ratio.num * outlink->h * link->w, -+ (int64_t)in->sample_aspect_ratio.den * outlink->w * link->h, -+ INT_MAX); -+ -+ av_frame_free(&in); -+ return ff_filter_frame(outlink, out); -+fail: -+ av_frame_free(&in); -+ av_frame_free(&out); -+ return ret; -+} -+ -+ -+ -+#define OFFSET(x) offsetof(PadCUDAContext, x) -+#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM|AV_OPT_FLAG_VIDEO_PARAM) -+ -+static const AVOption pad_cuda_options[] = { -+ { "width", "set the pad area width", OFFSET(w_expr), AV_OPT_TYPE_STRING, {.str = "iw"}, 0, 0, FLAGS }, -+ { "w", "set the pad area width", OFFSET(w_expr), AV_OPT_TYPE_STRING, {.str = "iw"}, 0, 0, FLAGS }, -+ { "height", "set the pad area height", OFFSET(h_expr), AV_OPT_TYPE_STRING, {.str = "ih"}, 0, 0, FLAGS }, -+ { "h", "set the pad area height", OFFSET(h_expr), AV_OPT_TYPE_STRING, {.str = "ih"}, 0, 0, FLAGS }, -+ { "x", "set the x offset for the input image position", OFFSET(x_expr), AV_OPT_TYPE_STRING, {.str = "0"}, 0, INT16_MAX, FLAGS }, -+ { "y", "set the y offset for the input image position", OFFSET(y_expr), AV_OPT_TYPE_STRING, {.str = "0"}, 0, INT16_MAX, FLAGS }, -+ { "color", "set the color of the padded area border", OFFSET(pad_rgba), AV_OPT_TYPE_COLOR, { .str = "black" }, 0, 0, FLAGS }, -+ { "aspect", "pad to fit an aspect instead of a resolution", OFFSET(aspect), AV_OPT_TYPE_RATIONAL, {.dbl = 0}, 0, INT16_MAX, FLAGS }, -+ { NULL } -+}; -+ -+AVFILTER_DEFINE_CLASS(pad_cuda); -+ -+static const AVFilterPad pad_cuda_inputs[] = { -+ { -+ .name = "default", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .filter_frame = pad_cuda_filter_frame, -+ }, -+}; -+ -+static const AVFilterPad pad_cuda_outputs[] = { -+ { -+ .name = "default", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .config_props = pad_cuda_output_config_props, -+ }, -+}; -+ -+const AVFilter ff_vf_pad_cuda = { -+ .name = "pad_cuda", -+ .description = NULL_IF_CONFIG_SMALL("Pad CUDA accelerated video using solid color"), -+ .priv_size = sizeof(PadCUDAContext), -+ .priv_class = &pad_cuda_class, -+ .init = pad_cuda_init, -+ .uninit = pad_cuda_uninit, -+ FILTER_INPUTS(pad_cuda_inputs), -+ FILTER_OUTPUTS(pad_cuda_outputs), -+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), -+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, -+}; -diff --git a/libavfilter/vf_pad_cuda.cu b/libavfilter/vf_pad_cuda.cu -new file mode 100644 -index 0000000..62f7988 ---- /dev/null -+++ b/libavfilter/vf_pad_cuda.cu -@@ -0,0 +1,37 @@ -+extern "C" { -+ -+__global__ void Pad_Cuda( -+ unsigned char* main, int main_linesize, -+ unsigned char* overlay, int overlay_linesize, -+ int overlay_w, int overlay_h, -+ int offset_x, int offset_y, -+ unsigned char fill_color1, unsigned char fill_color2) -+{ -+ // -+ // We pass two colors to handle NV12 plane with interleaved UV. -+ // So fill_color1 would correspond to U and fill_color2 to V. -+ // For other formats both fill_colors should be the same. -+ // -+ unsigned char color[2] = {fill_color1, fill_color2}; -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ -+ unsigned char *m = &main[x + y*main_linesize]; -+ -+ if (x < offset_x || -+ y < offset_y || -+ x >= overlay_w + offset_x || -+ y >= overlay_h + offset_y) -+ { -+ // Picking colors[0] for even value of x; and colors[1] for odd value of x; -+ *m = color[x & 1]; -+ } -+ else -+ { -+ int overlay_x = x - offset_x; -+ int overlay_y = y - offset_y; -+ *m = overlay[overlay_x + overlay_y*overlay_linesize]; -+ } -+} -+ -+} -diff --git a/libavfilter/vf_scale_cuda.c b/libavfilter/vf_scale_cuda.c -index 54a3409..2db84dd 100644 ---- a/libavfilter/vf_scale_cuda.c -+++ b/libavfilter/vf_scale_cuda.c -@@ -446,6 +446,11 @@ static int scalecuda_resize(AVFilterContext *ctx, - - for (i = 0; i < s->in_planes; i++) { - CUDA_TEXTURE_DESC tex_desc = { -+ .addressMode = { -+ CU_TR_ADDRESS_MODE_CLAMP, -+ CU_TR_ADDRESS_MODE_CLAMP, -+ CU_TR_ADDRESS_MODE_CLAMP, -+ }, - .filterMode = s->interp_use_linear ? - CU_TR_FILTER_MODE_LINEAR : - CU_TR_FILTER_MODE_POINT, -diff --git a/libavfilter/vf_scale_cuda.cu b/libavfilter/vf_scale_cuda.cu -index de06ba9..414649e 100644 ---- a/libavfilter/vf_scale_cuda.cu -+++ b/libavfilter/vf_scale_cuda.cu -@@ -1013,24 +1013,29 @@ struct Convert_rgb0_rgba - - typedef float4 (*coeffs_function_t)(float, float); - --__device__ static inline float4 lanczos_coeffs(float x, float param) -+__device__ static inline float lanczos_sinc(float x) - { - const float pi = 3.141592654f; - -+ if (fabsf(x) < 1.0e-5f) -+ return 1.0f; -+ -+ x *= pi; -+ return __sinf(x) / x; -+} -+ -+__device__ static inline float lanczos_kernel(float x) -+{ -+ return lanczos_sinc(x) * lanczos_sinc(x / 2.0f); -+} -+ -+__device__ static inline float4 lanczos_coeffs(float x, float param) -+{ - float4 res = make_float4( -- pi * (x + 1), -- pi * x, -- pi * (x - 1), -- pi * (x - 2)); -- -- res.x = res.x == 0.0f ? 1.0f : -- __sinf(res.x) * __sinf(res.x / 2.0f) / (res.x * res.x / 2.0f); -- res.y = res.y == 0.0f ? 1.0f : -- __sinf(res.y) * __sinf(res.y / 2.0f) / (res.y * res.y / 2.0f); -- res.z = res.z == 0.0f ? 1.0f : -- __sinf(res.z) * __sinf(res.z / 2.0f) / (res.z * res.z / 2.0f); -- res.w = res.w == 0.0f ? 1.0f : -- __sinf(res.w) * __sinf(res.w / 2.0f) / (res.w * res.w / 2.0f); -+ lanczos_kernel(x + 1), -+ lanczos_kernel(x), -+ lanczos_kernel(x - 1), -+ lanczos_kernel(x - 2)); - - return res / (res.x + res.y + res.z + res.w); - } -@@ -1059,6 +1064,71 @@ __device__ static inline V apply_coeffs(float4 coeffs, V c0, V c1, V c2, V c3) - return res; - } - -+__device__ static inline float clamp_texture_coord(float v, int size) -+{ -+ return fminf(fmaxf(v, 0.0f), (float)(size - 1)); -+} -+ -+__device__ static inline uchar clip_float_to_uchar(float v) -+{ -+ return (uchar)fminf(fmaxf(v, 0.0f), 255.0f); -+} -+ -+__device__ static inline ushort clip_float_to_ushort(float v) -+{ -+ return (ushort)fminf(fmaxf(v, 0.0f), 65535.0f); -+} -+ -+template -+__device__ static inline T from_scaled_floatN(const V &v) -+{ -+ return from_floatN(v); -+} -+ -+template<> -+__device__ inline uchar from_scaled_floatN(const float &v) -+{ -+ return clip_float_to_uchar(v); -+} -+ -+template<> -+__device__ inline uchar2 from_scaled_floatN(const float2 &v) -+{ -+ return make_uchar2(clip_float_to_uchar(v.x), -+ clip_float_to_uchar(v.y)); -+} -+ -+template<> -+__device__ inline uchar4 from_scaled_floatN(const float4 &v) -+{ -+ return make_uchar4(clip_float_to_uchar(v.x), -+ clip_float_to_uchar(v.y), -+ clip_float_to_uchar(v.z), -+ clip_float_to_uchar(v.w)); -+} -+ -+template<> -+__device__ inline ushort from_scaled_floatN(const float &v) -+{ -+ return clip_float_to_ushort(v); -+} -+ -+template<> -+__device__ inline ushort2 from_scaled_floatN(const float2 &v) -+{ -+ return make_ushort2(clip_float_to_ushort(v.x), -+ clip_float_to_ushort(v.y)); -+} -+ -+template<> -+__device__ inline ushort4 from_scaled_floatN(const float4 &v) -+{ -+ return make_ushort4(clip_float_to_ushort(v.x), -+ clip_float_to_ushort(v.y), -+ clip_float_to_ushort(v.z), -+ clip_float_to_ushort(v.w)); -+} -+ - template - __device__ static inline T Subsample_Nearest(cudaTextureObject_t tex, - int xo, int yo, -@@ -1126,9 +1196,11 @@ __device__ static inline T Subsample_Bicubic(cudaTextureObject_t tex, - float4 coeffsX = coeffs_function(fx, param); - float4 coeffsY = coeffs_function(fy, param); - --#define PIX(x, y) tex2D(tex, (x), (y)) -+#define PIX(x, y) tex2D(tex, \ -+ clamp_texture_coord((x), src_width), \ -+ clamp_texture_coord((y), src_height)) - -- return from_floatN( -+ return from_scaled_floatN( - apply_coeffs(coeffsY, - apply_coeffs(coeffsX, PIX(px - 1, py - 1), PIX(px, py - 1), PIX(px + 1, py - 1), PIX(px + 2, py - 1)), - apply_coeffs(coeffsX, PIX(px - 1, py ), PIX(px, py ), PIX(px + 1, py ), PIX(px + 2, py )), -diff --git a/libavfilter/vf_transition_cuda.c b/libavfilter/vf_transition_cuda.c -new file mode 100644 -index 0000000..44a198f ---- /dev/null -+++ b/libavfilter/vf_transition_cuda.c -@@ -0,0 +1,621 @@ -+/* -+ * Copyright (c) 2020 Yaroslav Pogrebnyak -+ * -+ * This file is part of FFmpeg. -+ * -+ * FFmpeg is free software; you can redistribute it and/or -+ * modify it under the terms of the GNU Lesser General Public -+ * License as published by the Free Software Foundation; either -+ * version 2.1 of the License, or (at your option) any later version. -+ * -+ * FFmpeg is distributed in the hope that it will be useful, -+ * but WITHOUT ANY WARRANTY; without even the implied warranty of -+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU -+ * Lesser General Public License for more details. -+ * -+ * You should have received a copy of the GNU Lesser General Public -+ * License along with FFmpeg; if not, write to the Free Software -+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA -+ */ -+ -+/** -+ * @file -+ * Overlay one video on top of another or transition between them using cuda hardware acceleration -+ */ -+ -+#include "libavutil/log.h" -+#include "libavutil/mem.h" -+#include "libavutil/opt.h" -+#include "libavutil/pixdesc.h" -+#include "libavutil/hwcontext.h" -+#include "libavutil/hwcontext_cuda_internal.h" -+#include "libavutil/cuda_check.h" -+#include "libavutil/eval.h" -+ -+#include "avfilter.h" -+#include "framesync.h" -+#include "avfilter_internal.h" -+ -+#include "cuda/load_helper.h" -+ -+#define CHECK_CU(x) FF_CUDA_CHECK_DL(ctx, ctx->hwctx->internal->cuda_dl, x) -+#define DIV_UP(a, b) ( ((a) + (b) - 1) / (b) ) -+ -+#define BLOCK_X 32 -+#define BLOCK_Y 16 -+ -+#define MAIN 0 -+#define OVERLAY 1 -+ -+static const enum AVPixelFormat supported_main_formats[] = { -+ AV_PIX_FMT_NV12, -+ AV_PIX_FMT_YUV420P, -+ AV_PIX_FMT_NONE, -+}; -+ -+static const enum AVPixelFormat supported_overlay_formats[] = { -+ AV_PIX_FMT_NV12, -+ AV_PIX_FMT_YUV420P, -+ AV_PIX_FMT_YUVA420P, -+ AV_PIX_FMT_NONE, -+}; -+ -+enum var_name { -+ VAR_MAIN_W, VAR_MW, -+ VAR_MAIN_H, VAR_MH, -+ VAR_OVERLAY_W, VAR_OW, -+ VAR_OVERLAY_H, VAR_OH, -+ VAR_ALPHA, -+ VAR_N, -+ VAR_POS, -+ VAR_T, -+ VAR_VARS_NB -+}; -+ -+enum EvalMode { -+ EVAL_MODE_INIT, -+ EVAL_MODE_FRAME, -+ EVAL_MODE_NB -+}; -+ -+enum TransitionMode { -+ TRANSITION_MODE_FADE, -+ TRANSITION_MODE_WIPE_LEFT, -+ TRANSITION_MODE_WIPE_RIGHT, -+ TRANSITION_MODE_WIPE_DOWN, -+ TRANSITION_MODE_WIPE_UP, -+ TRANSITION_MODE_NB -+}; -+ -+static const char *const var_names[] = { -+ "main_w", "W", ///< width of the main video -+ "main_h", "H", ///< height of the main video -+ "overlay_w", "w", ///< width of the overlay video -+ "overlay_h", "h", ///< height of the overlay video -+ "alpha", -+ "n", ///< number of frame -+ "pos", ///< position in the file -+ "t", ///< timestamp expressed in seconds -+ NULL -+}; -+ -+/** -+ * TransitionCUDAContext -+ */ -+typedef struct TransitionCUDAContext { -+ const AVClass *class; -+ -+ enum AVPixelFormat in_format_overlay; -+ enum AVPixelFormat in_format_main; -+ -+ AVBufferRef *hw_device_ctx; -+ AVCUDADeviceContext *hwctx; -+ -+ CUcontext cu_ctx; -+ CUmodule cu_module; -+ CUfunction cu_func; -+ CUstream cu_stream; -+ -+ FFFrameSync fs; -+ -+ int eval_mode; -+ int transition_mode; -+ float alpha_coef; -+ -+ double var_values[VAR_VARS_NB]; -+ char *alpha_expr; -+ -+ AVExpr *alpha_pexpr; -+} TransitionCUDAContext; -+ -+/** -+ * Helper to find out if provided format is supported by filter -+ */ -+static int format_is_supported(const enum AVPixelFormat formats[], enum AVPixelFormat fmt) -+{ -+ for (int i = 0; formats[i] != AV_PIX_FMT_NONE; i++) -+ if (formats[i] == fmt) -+ return 1; -+ return 0; -+} -+ -+static void eval_expr(AVFilterContext *ctx) -+{ -+ TransitionCUDAContext *s = ctx->priv; -+ -+ s->var_values[VAR_ALPHA] = av_expr_eval(s->alpha_pexpr, s->var_values, NULL); -+ s->alpha_coef = (float)s->var_values[VAR_ALPHA]; -+ -+ if (isnan(s->alpha_coef) || s->alpha_coef > 1.f) -+ s->alpha_coef = 1.f; -+ -+ if (s->alpha_coef < 0.f) -+ s->alpha_coef = 0.f; -+} -+ -+static int set_expr(AVExpr **pexpr, const char *expr, const char *option, void *log_ctx) -+{ -+ int ret; -+ AVExpr *old = NULL; -+ -+ if (*pexpr) -+ old = *pexpr; -+ ret = av_expr_parse(pexpr, expr, var_names, -+ NULL, NULL, NULL, NULL, 0, log_ctx); -+ if (ret < 0) { -+ av_log(log_ctx, AV_LOG_ERROR, -+ "Error when evaluating the expression '%s' for %s\n", -+ expr, option); -+ *pexpr = old; -+ return ret; -+ } -+ -+ av_expr_free(old); -+ return 0; -+} -+ -+/** -+ * Helper checks if we can process main and overlay pixel formats -+ */ -+static int formats_match(const enum AVPixelFormat format_main, const enum AVPixelFormat format_overlay) { -+ switch(format_main) { -+ case AV_PIX_FMT_NV12: -+ return format_overlay == AV_PIX_FMT_NV12; -+ case AV_PIX_FMT_YUV420P: -+ return format_overlay == AV_PIX_FMT_YUV420P || -+ format_overlay == AV_PIX_FMT_YUVA420P; -+ default: -+ return 0; -+ } -+} -+ -+/** -+ * Call transition kernell for a plane -+ */ -+static int transition_cuda_call_kernel( -+ TransitionCUDAContext *ctx, -+ uint8_t* main_data, int main_linesize, -+ int main_width, int main_height, -+ uint8_t* overlay_data, int overlay_linesize, -+ int overlay_width, int overlay_height, -+ uint8_t* alpha_data, int alpha_linesize, -+ int alpha_adj_x, int alpha_adj_y, -+ float alpha_coef, -+ int transition_mode) { -+ -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ -+ void* kernel_args[] = { -+ &main_data, &main_linesize, -+ &overlay_data, &overlay_linesize, -+ &overlay_width, &overlay_height, -+ &alpha_data, &alpha_linesize, -+ &alpha_adj_x, &alpha_adj_y, -+ &alpha_coef, -+ &transition_mode -+ }; -+ -+ return CHECK_CU(cu->cuLaunchKernel( -+ ctx->cu_func, -+ DIV_UP(main_width, BLOCK_X), DIV_UP(main_height, BLOCK_Y), 1, -+ BLOCK_X, BLOCK_Y, 1, -+ 0, ctx->cu_stream, kernel_args, NULL)); -+} -+ -+/** -+ * Perform blend overlay picture over main picture -+ */ -+static int transition_cuda_blend(FFFrameSync *fs) -+{ -+ int ret; -+ -+ AVFilterContext *avctx = fs->parent; -+ TransitionCUDAContext *ctx = avctx->priv; -+ AVFilterLink *outlink = avctx->outputs[0]; -+ AVFilterLink *inlink = avctx->inputs[0]; -+ FilterLink *inl = ff_filter_link(inlink); -+ -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CUcontext dummy, cuda_ctx = ctx->hwctx->cuda_ctx; -+ -+ AVFrame *input_main, *input_overlay; -+ -+ ctx->cu_ctx = cuda_ctx; -+ -+ // read main and overlay frames from inputs -+ ret = ff_framesync_dualinput_get(fs, &input_main, &input_overlay); -+ if (ret < 0) -+ return ret; -+ -+ if (!input_main) -+ return AVERROR_BUG; -+ -+ if (!input_overlay) -+ return ff_filter_frame(outlink, input_main); -+ -+ ret = av_frame_make_writable(input_main); -+ if (ret < 0) { -+ av_frame_free(&input_main); -+ return ret; -+ } -+ -+ // push cuda context -+ -+ ret = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); -+ if (ret < 0) { -+ av_frame_free(&input_main); -+ return ret; -+ } -+ -+ if (ctx->eval_mode == EVAL_MODE_FRAME) { -+ int pos = input_main->pkt_pos; -+ ctx->var_values[VAR_N] = inl->frame_count_out; -+ ctx->var_values[VAR_T] = input_main->pts == AV_NOPTS_VALUE ? -+ NAN : input_main->pts * av_q2d(inlink->time_base); -+ ctx->var_values[VAR_POS] = pos == -1 ? NAN : pos; -+ ctx->var_values[VAR_OVERLAY_W] = ctx->var_values[VAR_OW] = input_overlay->width; -+ ctx->var_values[VAR_OVERLAY_H] = ctx->var_values[VAR_OH] = input_overlay->height; -+ ctx->var_values[VAR_MAIN_W ] = ctx->var_values[VAR_MW] = input_main->width; -+ ctx->var_values[VAR_MAIN_H ] = ctx->var_values[VAR_MH] = input_main->height; -+ -+ eval_expr(avctx); -+ -+ av_log(avctx, AV_LOG_DEBUG, "n:%f t:%f pos:%f alpha: %f\n", -+ ctx->var_values[VAR_N], ctx->var_values[VAR_T], ctx->var_values[VAR_POS], -+ ctx->alpha_coef); -+ } -+ -+ // overlay first plane -+ -+ transition_cuda_call_kernel(ctx, -+ input_main->data[0], input_main->linesize[0], -+ input_main->width, input_main->height, -+ input_overlay->data[0], input_overlay->linesize[0], -+ input_overlay->width, input_overlay->height, -+ input_overlay->data[3], input_overlay->linesize[3], 1, 1, -+ ctx->alpha_coef, ctx->transition_mode); -+ -+ // overlay rest planes depending on pixel format -+ -+ switch(ctx->in_format_overlay) { -+ case AV_PIX_FMT_NV12: -+ transition_cuda_call_kernel(ctx, -+ input_main->data[1], input_main->linesize[1], -+ input_main->width, input_main->height / 2, -+ input_overlay->data[1], input_overlay->linesize[1], -+ input_overlay->width, input_overlay->height / 2, -+ 0, 0, 0, 0, -+ ctx->alpha_coef, ctx->transition_mode); -+ break; -+ case AV_PIX_FMT_YUV420P: -+ case AV_PIX_FMT_YUVA420P: -+ transition_cuda_call_kernel(ctx, -+ input_main->data[1], input_main->linesize[1], -+ input_main->width / 2, input_main->height / 2, -+ input_overlay->data[1], input_overlay->linesize[1], -+ input_overlay->width / 2, input_overlay->height / 2, -+ input_overlay->data[3], input_overlay->linesize[3], 2, 2, -+ ctx->alpha_coef, ctx->transition_mode); -+ -+ transition_cuda_call_kernel(ctx, -+ input_main->data[2], input_main->linesize[2], -+ input_main->width / 2, input_main->height / 2, -+ input_overlay->data[2], input_overlay->linesize[2], -+ input_overlay->width / 2, input_overlay->height / 2, -+ input_overlay->data[3], input_overlay->linesize[3], 2, 2, -+ ctx->alpha_coef, ctx->transition_mode); -+ break; -+ default: -+ av_log(ctx, AV_LOG_ERROR, "Passed unsupported overlay pixel format\n"); -+ av_frame_free(&input_main); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return AVERROR_BUG; -+ } -+ -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ -+ return ff_filter_frame(outlink, input_main); -+} -+ -+static int config_input_overlay(AVFilterLink *inlink) -+{ -+ AVFilterContext *ctx = inlink->dst; -+ TransitionCUDAContext *s = inlink->dst->priv; -+ int ret; -+ -+ -+ /* Finish the configuration by evaluating the expressions -+ now when both inputs are configured. */ -+ s->var_values[VAR_MAIN_W ] = s->var_values[VAR_MW] = ctx->inputs[MAIN ]->w; -+ s->var_values[VAR_MAIN_H ] = s->var_values[VAR_MH] = ctx->inputs[MAIN ]->h; -+ s->var_values[VAR_OVERLAY_W] = s->var_values[VAR_OW] = ctx->inputs[OVERLAY]->w; -+ s->var_values[VAR_OVERLAY_H] = s->var_values[VAR_OH] = ctx->inputs[OVERLAY]->h; -+ s->var_values[VAR_ALPHA] = NAN; -+ s->var_values[VAR_N] = 0; -+ s->var_values[VAR_T] = NAN; -+ s->var_values[VAR_POS] = NAN; -+ -+ if ((ret = set_expr(&s->alpha_pexpr, s->alpha_expr, "alpha", ctx)) < 0) -+ return ret; -+ -+ if (s->eval_mode == EVAL_MODE_INIT) { -+ eval_expr(ctx); -+ av_log(ctx, AV_LOG_VERBOSE, "alpha: %f\n", s->alpha_coef); -+ } -+ -+ return 0; -+} -+ -+/** -+ * Initialize transition_cuda -+ */ -+static av_cold int transition_cuda_init(AVFilterContext *avctx) -+{ -+ TransitionCUDAContext* ctx = avctx->priv; -+ ctx->fs.on_event = &transition_cuda_blend; -+ return 0; -+} -+ -+/** -+ * Uninitialize transition_cuda -+ */ -+static av_cold void transition_cuda_uninit(AVFilterContext *avctx) -+{ -+ TransitionCUDAContext* ctx = avctx->priv; -+ -+ ff_framesync_uninit(&ctx->fs); -+ -+ if (ctx->hwctx && ctx->cu_module) { -+ CUcontext dummy; -+ CudaFunctions *cu = ctx->hwctx->internal->cuda_dl; -+ CHECK_CU(cu->cuCtxPushCurrent(ctx->cu_ctx)); -+ CHECK_CU(cu->cuModuleUnload(ctx->cu_module)); -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ } -+ -+ av_expr_free(ctx->alpha_pexpr); ctx->alpha_pexpr = NULL; -+ av_buffer_unref(&ctx->hw_device_ctx); -+ ctx->hwctx = NULL; -+} -+ -+/** -+ * Activate transition_cuda -+ */ -+static int transition_cuda_activate(AVFilterContext *avctx) -+{ -+ TransitionCUDAContext *ctx = avctx->priv; -+ return ff_framesync_activate(&ctx->fs); -+} -+ -+/** -+ * Configure output -+ */ -+static int transition_cuda_config_output(AVFilterLink *outlink) -+{ -+ extern const unsigned char ff_vf_transition_cuda_ptx_data[]; -+ extern const unsigned int ff_vf_transition_cuda_ptx_len; -+ -+ int err; -+ FilterLink *outl = ff_filter_link(outlink); -+ AVFilterContext *avctx = outlink->src; -+ TransitionCUDAContext *ctx = avctx->priv; -+ -+ AVFilterLink *inlink = avctx->inputs[0]; -+ FilterLink *inl = ff_filter_link(inlink); -+ AVHWFramesContext *frames_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; -+ -+ AVFilterLink *inlink_overlay = avctx->inputs[1]; -+ FilterLink *inl_overlay = ff_filter_link(inlink_overlay); -+ AVHWFramesContext *frames_ctx_overlay = (AVHWFramesContext*)inl_overlay->hw_frames_ctx->data; -+ -+ CUcontext dummy, cuda_ctx; -+ CudaFunctions *cu; -+ -+ // check main input formats -+ -+ if (!frames_ctx) { -+ av_log(ctx, AV_LOG_ERROR, "No hw context provided on main input\n"); -+ return AVERROR(EINVAL); -+ } -+ -+ ctx->in_format_main = frames_ctx->sw_format; -+ if (!format_is_supported(supported_main_formats, ctx->in_format_main)) { -+ av_log(ctx, AV_LOG_ERROR, "Unsupported main input format: %s\n", -+ av_get_pix_fmt_name(ctx->in_format_main)); -+ return AVERROR(ENOSYS); -+ } -+ -+ // check transition input formats -+ -+ if (!frames_ctx_overlay) { -+ av_log(ctx, AV_LOG_ERROR, "No hw context provided on transition input\n"); -+ return AVERROR(EINVAL); -+ } -+ -+ ctx->in_format_overlay = frames_ctx_overlay->sw_format; -+ if (!format_is_supported(supported_overlay_formats, ctx->in_format_overlay)) { -+ av_log(ctx, AV_LOG_ERROR, "Unsupported overlay input format: %s\n", -+ av_get_pix_fmt_name(ctx->in_format_overlay)); -+ return AVERROR(ENOSYS); -+ } -+ -+ // check we can overlay pictures with those pixel formats -+ -+ if (!formats_match(ctx->in_format_main, ctx->in_format_overlay)) { -+ av_log(ctx, AV_LOG_ERROR, "Can't overlay %s on %s \n", -+ av_get_pix_fmt_name(ctx->in_format_overlay), av_get_pix_fmt_name(ctx->in_format_main)); -+ return AVERROR(EINVAL); -+ } -+ -+ // initialize -+ -+ ctx->hw_device_ctx = av_buffer_ref(frames_ctx->device_ref); -+ if (!ctx->hw_device_ctx) -+ return AVERROR(ENOMEM); -+ ctx->hwctx = ((AVHWDeviceContext*)ctx->hw_device_ctx->data)->hwctx; -+ -+ cuda_ctx = ctx->hwctx->cuda_ctx; -+ ctx->fs.time_base = inlink->time_base; -+ -+ ctx->cu_stream = ctx->hwctx->stream; -+ -+ outl->hw_frames_ctx = av_buffer_ref(inl->hw_frames_ctx); -+ if (!outl->hw_frames_ctx) -+ return AVERROR(ENOMEM); -+ -+ // load functions -+ -+ cu = ctx->hwctx->internal->cuda_dl; -+ -+ err = CHECK_CU(cu->cuCtxPushCurrent(cuda_ctx)); -+ if (err < 0) { -+ return err; -+ } -+ -+ err = ff_cuda_load_module(ctx, ctx->hwctx, &ctx->cu_module, ff_vf_transition_cuda_ptx_data, ff_vf_transition_cuda_ptx_len); -+ if (err < 0) { -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return err; -+ } -+ -+ err = CHECK_CU(cu->cuModuleGetFunction(&ctx->cu_func, ctx->cu_module, "Transition_Cuda")); -+ if (err < 0) { -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ return err; -+ } -+ -+ CHECK_CU(cu->cuCtxPopCurrent(&dummy)); -+ -+ // init dual input -+ -+ err = ff_framesync_init_dualinput(&ctx->fs, avctx); -+ if (err < 0) { -+ return err; -+ } -+ -+ return ff_framesync_configure(&ctx->fs); -+} -+ -+static int transition_cuda_process_command(AVFilterContext *avctx, -+ const char *cmd, -+ const char *args, -+ char *res, -+ int res_len, -+ int flags) -+{ -+ TransitionCUDAContext *ctx = avctx->priv; -+ AVExpr *new_expr = NULL; -+ char *new_alpha_expr; -+ int ret; -+ -+ (void)res; -+ (void)res_len; -+ (void)flags; -+ -+ if (!strcmp(cmd, "mode")) -+ return av_opt_set(ctx, "mode", args, 0); -+ -+ if (strcmp(cmd, "alpha")) -+ return AVERROR(ENOSYS); -+ -+ ret = set_expr(&new_expr, args, "alpha", avctx); -+ if (ret < 0) -+ return ret; -+ -+ new_alpha_expr = av_strdup(args); -+ if (!new_alpha_expr) { -+ av_expr_free(new_expr); -+ return AVERROR(ENOMEM); -+ } -+ -+ av_expr_free(ctx->alpha_pexpr); -+ ctx->alpha_pexpr = new_expr; -+ av_freep(&ctx->alpha_expr); -+ ctx->alpha_expr = new_alpha_expr; -+ return 0; -+} -+ -+ -+#define OFFSET(x) offsetof(TransitionCUDAContext, x) -+#define FLAGS (AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM) -+#define RUNTIME_FLAGS (FLAGS | AV_OPT_FLAG_RUNTIME_PARAM) -+ -+static const AVOption transition_cuda_options[] = { -+ { "alpha", "set the alpha expression of overlay in range [0.0-1.0] (default is 1.0)", OFFSET(alpha_expr), AV_OPT_TYPE_STRING, { .str = "1.0" }, 0, 0, RUNTIME_FLAGS }, -+ { "mode", "set the CUDA transition mode", OFFSET(transition_mode), AV_OPT_TYPE_INT, { .i64 = TRANSITION_MODE_FADE }, 0, TRANSITION_MODE_NB - 1, RUNTIME_FLAGS, "mode" }, -+ { "fade", "crossfade", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_FADE }, .flags = RUNTIME_FLAGS, .unit = "mode" }, -+ { "wipe_left", "wipe from left", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_WIPE_LEFT }, .flags = RUNTIME_FLAGS, .unit = "mode" }, -+ { "wipe_right", "wipe from right", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_WIPE_RIGHT }, .flags = RUNTIME_FLAGS, .unit = "mode" }, -+ { "wipe_down", "wipe from top to bottom", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_WIPE_DOWN }, .flags = RUNTIME_FLAGS, .unit = "mode" }, -+ { "wipe_up", "wipe from bottom to top", 0, AV_OPT_TYPE_CONST, { .i64 = TRANSITION_MODE_WIPE_UP }, .flags = RUNTIME_FLAGS, .unit = "mode" }, -+ { "eof_action", "Action to take when encountering EOF from secondary input ", -+ OFFSET(fs.opt_eof_action), AV_OPT_TYPE_INT, { .i64 = EOF_ACTION_REPEAT }, -+ EOF_ACTION_REPEAT, EOF_ACTION_PASS, .flags = FLAGS, "eof_action" }, -+ { "repeat", "Repeat the previous frame.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_REPEAT }, .flags = FLAGS, "eof_action" }, -+ { "endall", "End both streams.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_ENDALL }, .flags = FLAGS, "eof_action" }, -+ { "pass", "Pass through the main input.", 0, AV_OPT_TYPE_CONST, { .i64 = EOF_ACTION_PASS }, .flags = FLAGS, "eof_action" }, -+ { "eval", "specify when to evaluate expressions", OFFSET(eval_mode), AV_OPT_TYPE_INT, { .i64 = EVAL_MODE_FRAME }, 0, EVAL_MODE_NB - 1, FLAGS, "eval" }, -+ { "init", "eval expressions once during initialization", 0, AV_OPT_TYPE_CONST, { .i64=EVAL_MODE_INIT }, .flags = FLAGS, .unit = "eval" }, -+ { "frame", "eval expressions per-frame", 0, AV_OPT_TYPE_CONST, { .i64=EVAL_MODE_FRAME }, .flags = FLAGS, .unit = "eval" }, -+ { "shortest", "force termination when the shortest input terminates", OFFSET(fs.opt_shortest), AV_OPT_TYPE_BOOL, { .i64 = 0 }, 0, 1, FLAGS }, -+ { "repeatlast", "repeat overlay of the last overlay frame", OFFSET(fs.opt_repeatlast), AV_OPT_TYPE_BOOL, {.i64=1}, 0, 1, FLAGS }, -+ { NULL }, -+}; -+ -+FRAMESYNC_DEFINE_CLASS(transition_cuda, TransitionCUDAContext, fs); -+ -+static const AVFilterPad transition_cuda_inputs[] = { -+ { -+ .name = "main", -+ .type = AVMEDIA_TYPE_VIDEO, -+ }, -+ { -+ .name = "overlay", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .config_props = config_input_overlay, -+ }, -+}; -+ -+static const AVFilterPad transition_cuda_outputs[] = { -+ { -+ .name = "default", -+ .type = AVMEDIA_TYPE_VIDEO, -+ .config_props = &transition_cuda_config_output, -+ }, -+}; -+ -+const AVFilter ff_vf_transition_cuda = { -+ .name = "transition_cuda", -+ .description = NULL_IF_CONFIG_SMALL("Transition between videos using CUDA"), -+ .priv_size = sizeof(TransitionCUDAContext), -+ .priv_class = &transition_cuda_class, -+ .init = &transition_cuda_init, -+ .uninit = &transition_cuda_uninit, -+ .activate = &transition_cuda_activate, -+ .process_command = &transition_cuda_process_command, -+ FILTER_INPUTS(transition_cuda_inputs), -+ FILTER_OUTPUTS(transition_cuda_outputs), -+ FILTER_SINGLE_PIXFMT(AV_PIX_FMT_CUDA), -+ .preinit = transition_cuda_framesync_preinit, -+ .flags_internal = FF_FILTER_FLAG_HWFRAME_AWARE, -+}; -diff --git a/libavfilter/vf_transition_cuda.cu b/libavfilter/vf_transition_cuda.cu -new file mode 100644 -index 0000000..c688e66 ---- /dev/null -+++ b/libavfilter/vf_transition_cuda.cu -@@ -0,0 +1,56 @@ -+/* -+ * Copyright (c) 2020 Yaroslav Pogrebnyak -+ * -+ * This file is part of FFmpeg. -+ * -+ * FFmpeg is free software; you can redistribute it and/or -+ * modify it under the terms of the GNU Lesser General Public -+ * License as published by the Free Software Foundation; either -+ * version 2.1 of the License, or (at your option) any later version. -+ * -+ * FFmpeg is distributed in the hope that it will be useful, -+ * but WITHOUT ANY WARRANTY; without even the implied warranty of -+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU -+ * Lesser General Public License for more details. -+ * -+ * You should have received a copy of the GNU Lesser General Public -+ * License along with FFmpeg; if not, write to the Free Software -+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA -+ */ -+ -+extern "C" { -+ -+__global__ void Transition_Cuda( -+ unsigned char* main, int main_linesize, -+ unsigned char* overlay, int overlay_linesize, -+ int overlay_w, int overlay_h, -+ unsigned char* overlay_alpha, int alpha_linesize, -+ int alpha_adj_x, int alpha_adj_y, -+ float alpha_coef, int transition_mode) -+{ -+ int x = blockIdx.x * blockDim.x + threadIdx.x; -+ int y = blockIdx.y * blockDim.y + threadIdx.y; -+ -+ if (x >= overlay_w || -+ y >= overlay_h) { -+ return; -+ } -+ -+ float alpha = alpha_coef; -+ if (transition_mode == 1) { -+ alpha = (x + 0.5f) / overlay_w <= alpha_coef; -+ } else if (transition_mode == 2) { -+ alpha = (x + 0.5f) / overlay_w >= 1.f - alpha_coef; -+ } else if (transition_mode == 3) { -+ alpha = (y + 0.5f) / overlay_h <= alpha_coef; -+ } else if (transition_mode == 4) { -+ alpha = (y + 0.5f) / overlay_h >= 1.f - alpha_coef; -+ } -+ if (alpha_linesize) { -+ alpha *= overlay_alpha[alpha_adj_x * x + alpha_adj_y * y * alpha_linesize] / 255.f; -+ } -+ -+ main[x + y*main_linesize] = alpha * overlay[x + y*overlay_linesize] + (1.f - alpha) * main[x + y*main_linesize]; -+} -+ -+} -diff --git a/libavutil/hwcontext_cuda.c b/libavutil/hwcontext_cuda.c -index 3de3847..f8ff9a4 100644 ---- a/libavutil/hwcontext_cuda.c -+++ b/libavutil/hwcontext_cuda.c -@@ -44,6 +44,7 @@ static const enum AVPixelFormat supported_formats[] = { - AV_PIX_FMT_NV12, - AV_PIX_FMT_YUV420P, - AV_PIX_FMT_YUVA420P, -+ AV_PIX_FMT_YUVA444P, - AV_PIX_FMT_YUV444P, - AV_PIX_FMT_P010, - AV_PIX_FMT_P016, diff --git a/deps/ffmpeg/7.1.5/0004-avcodec-nvdec-intra.patch b/deps/ffmpeg/7.1.5/0004-avcodec-nvdec-intra.patch deleted file mode 100644 index e70a242a..00000000 --- a/deps/ffmpeg/7.1.5/0004-avcodec-nvdec-intra.patch +++ /dev/null @@ -1,30 +0,0 @@ -From 100f0aa5a67eb5500a1c38576e4003db59255f4b Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Wed, 12 Aug 2026 15:54:13 +0000 -Subject: [PATCH 4/7] avcodec/nvdec: handle intra-only streams - -Preserve the legacy CUVID handling required for intra-only decoded streams. - -Original-commit: 4bb0479af64eab3994fecbbda2911609db892f22 ---- - libavcodec/cuviddec.c | 7 +++++++ - 1 file changed, 7 insertions(+) - -diff --git a/libavcodec/cuviddec.c b/libavcodec/cuviddec.c -index 3fae9c1..7ad59f1 100644 ---- a/libavcodec/cuviddec.c -+++ b/libavcodec/cuviddec.c -@@ -342,6 +342,13 @@ static int CUDAAPI cuvid_handle_video_sequence(void *opaque, CUVIDEOFORMAT* form - cuinfo.bitDepthMinus8 = format->bit_depth_luma_minus8; - cuinfo.DeinterlaceMode = ctx->deint_mode_current; - -+ if(avctx->gop_size == 100000) { -+ av_log(avctx, AV_LOG_INFO, "init decoder intra only\n"); -+ cuinfo.ulIntraDecodeOnly = 1; -+ } -+ -+ av_log(avctx, AV_LOG_INFO, "init decoder %d %d\n", ctx->nb_surfaces, avctx->refs); -+ - if (ctx->deint_mode_current != cudaVideoDeinterlaceMode_Weave && !ctx->drop_second_field) - avctx->framerate = av_mul_q(avctx->framerate, (AVRational){2, 1}); - diff --git a/deps/ffmpeg/7.1.5/0006-avdevice-v4l2-compat.patch b/deps/ffmpeg/7.1.5/0006-avdevice-v4l2-compat.patch deleted file mode 100644 index e59e29ae..00000000 --- a/deps/ffmpeg/7.1.5/0006-avdevice-v4l2-compat.patch +++ /dev/null @@ -1,25 +0,0 @@ -From c89a55032f642c8c6afd7df0c52febe16a3c23d0 Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Wed, 12 Aug 2026 15:54:13 +0000 -Subject: [PATCH 6/7] avdevice/v4l2: retain source timestamps - -Preserve the legacy V4L2 timestamp behavior used by avplumber inputs. - -Original-commit: 961ec93cbc352178ae5572c8d7f03f508edc3147 ---- - libavdevice/v4l2.c | 2 +- - 1 file changed, 1 insertion(+), 1 deletion(-) - -diff --git a/libavdevice/v4l2.c b/libavdevice/v4l2.c -index 0ae6872..644752f 100644 ---- a/libavdevice/v4l2.c -+++ b/libavdevice/v4l2.c -@@ -584,7 +584,7 @@ static int mmap_read_frame(AVFormatContext *ctx, AVPacket *pkt) - if (ctx->video_codec_id == AV_CODEC_ID_CPIA) - s->frame_size = bytesused; - -- if (s->frame_size > 0 && bytesused != s->frame_size) { -+ if (s->frame_size > 0 && bytesused < s->frame_size) { - av_log(ctx, AV_LOG_WARNING, - "Dequeued v4l2 buffer contains %d bytes, but %d were expected. Flags: 0x%08X.\n", - bytesused, s->frame_size, buf.flags); diff --git a/deps/ffmpeg/7.1.5/0007-avdevice-ndi-v5.patch b/deps/ffmpeg/7.1.5/0007-avdevice-ndi-v5.patch deleted file mode 100644 index 6ad66c6c..00000000 --- a/deps/ffmpeg/7.1.5/0007-avdevice-ndi-v5.patch +++ /dev/null @@ -1,222 +0,0 @@ -From 2d74ed07e99ba7a0f3a09ccf2d20cddbb9f39b71 Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Wed, 12 Aug 2026 15:54:13 +0000 -Subject: [PATCH 7/7] avdevice/ndi: restore NDI v5 integration - -Register the existing NDI v5 input and output devices, add configure detection, and restore their user documentation. - -Original-commit: 6f4d2c58f2f6dd88e0612621ede61d51bc12980c ---- - configure | 7 +++++ - doc/indevs.texi | 65 ++++++++++++++++++++++++++++++++++++++++ - doc/outdevs.texi | 45 ++++++++++++++++++++++++++++ - libavdevice/Makefile | 4 +++ - libavdevice/alldevices.c | 2 ++ - 5 files changed, 123 insertions(+) - -diff --git a/configure b/configure -index 2cd7884..5ae4594 100755 ---- a/configure -+++ b/configure -@@ -316,6 +316,7 @@ External library support: - --enable-lv2 enable LV2 audio filtering [no] - --disable-lzma disable lzma [autodetect] - --enable-decklink enable Blackmagic DeckLink I/O support [no] -+ --enable-libndi_newtek enable Newteck NDI I/O support [no] - --enable-mbedtls enable mbedTLS, needed for https support - if openssl, gnutls or libtls is not used [no] - --enable-mediacodec enable Android MediaCodec support [no] -@@ -1878,6 +1879,7 @@ EXTERNAL_LIBRARY_GPL_LIST=" - - EXTERNAL_LIBRARY_NONFREE_LIST=" - decklink -+ libndi_newtek - libfdk_aac - libtls - " -@@ -3726,6 +3728,10 @@ decklink_indev_suggest="libzvbi" - decklink_outdev_deps="decklink threads" - decklink_outdev_suggest="libklvanc" - decklink_outdev_extralibs="-lstdc++" -+libndi_newtek_indev_deps="libndi_newtek" -+libndi_newtek_indev_extralibs="-lndi" -+libndi_newtek_outdev_deps="libndi_newtek" -+libndi_newtek_outdev_extralibs="-lndi" - dshow_indev_deps="IBaseFilter" - dshow_indev_extralibs="-lpsapi -lole32 -lstrmiids -luuid -loleaut32 -lshlwapi" - fbdev_indev_deps="linux_fb_h" -@@ -6883,6 +6889,7 @@ enabled chromaprint && { check_pkg_config chromaprint libchromaprint "chro - require chromaprint chromaprint.h chromaprint_get_version -lchromaprint; } - enabled decklink && { require_headers DeckLinkAPI.h && - { test_cpp_condition DeckLinkAPIVersion.h "BLACKMAGIC_DECKLINK_API_VERSION >= 0x0a0b0000" || die "ERROR: Decklink API version must be >= 10.11"; } } -+enabled libndi_newtek && require libndi_newtek Processing.NDI.Lib.h NDIlib_initialize -lndi - enabled frei0r && require_headers "frei0r.h" - enabled gmp && require gmp gmp.h mpz_export -lgmp - enabled gnutls && require_pkg_config gnutls gnutls gnutls/gnutls.h gnutls_global_init -diff --git a/doc/indevs.texi b/doc/indevs.texi -index cdf44a6..30679b0 100644 ---- a/doc/indevs.texi -+++ b/doc/indevs.texi -@@ -1156,6 +1156,71 @@ Set the video size given as a string such as @code{640x480} or @code{hd720}. - Default is @code{qvga}. - @end table - -+@section libndi_newtek -+ -+The libndi_newtek input device provides capture capabilities for using NDI (Network -+Device Interface, standard created by NewTek). -+ -+Input filename is a NDI source name that could be found by sending -find_sources 1 -+to command line - it has no specific syntax but human-readable formatted. -+ -+To enable this input device, you need the NDI SDK and you -+need to configure with the appropriate @code{--extra-cflags} -+and @code{--extra-ldflags}. -+ -+@subsection Options -+ -+@table @option -+ -+@item find_sources -+If set to @option{true}, print a list of found/available NDI sources and exit. -+Defaults to @option{false}. -+ -+@item wait_sources -+Override time to wait until the number of online sources have changed. -+Defaults to @option{0.5}. -+ -+@item allow_video_fields -+When this flag is @option{false}, all video that you receive will be progressive. -+Defaults to @option{true}. -+ -+@item extra_ips -+If is set to list of comma separated ip addresses, scan for sources not only -+using mDNS but also use unicast ip addresses specified by this list. -+ -+@end table -+ -+@subsection Examples -+ -+@itemize -+ -+@item -+List input devices: -+@example -+ffmpeg -f libndi_newtek -find_sources 1 -i dummy -+@end example -+ -+@item -+List local and remote input devices: -+@example -+ffmpeg -f libndi_newtek -extra_ips "192.168.10.10" -find_sources 1 -i dummy -+@end example -+ -+@item -+Restream to NDI: -+@example -+ffmpeg -f libndi_newtek -i "DEV-5.INTERNAL.M1STEREO.TV (NDI_SOURCE_NAME_1)" -f libndi_newtek -y NDI_SOURCE_NAME_2 -+@end example -+ -+@item -+Restream remote NDI to local NDI: -+@example -+ffmpeg -f libndi_newtek -extra_ips "192.168.10.10" -i "DEV-5.REMOTE.M1STEREO.TV (NDI_SOURCE_NAME_1)" -f libndi_newtek -y NDI_SOURCE_NAME_2 -+@end example -+ -+ -+@end itemize -+ - @section openal - - The OpenAL input device provides audio capture on all systems with a -diff --git a/doc/outdevs.texi b/doc/outdevs.texi -index 9ee8575..71f467d 100644 ---- a/doc/outdevs.texi -+++ b/doc/outdevs.texi -@@ -301,6 +301,51 @@ ffmpeg -re -i INPUT -c:v rawvideo -pix_fmt bgra -f fbdev /dev/fb0 - - See also @url{http://linux-fbdev.sourceforge.net/}, and fbset(1). - -+@section libndi_newtek -+ -+The libndi_newtek output device provides playback capabilities for using NDI (Network -+Device Interface, standard created by NewTek). -+ -+Output filename is a NDI name. -+ -+To enable this output device, you need the NDI SDK and you -+need to configure with the appropriate @code{--extra-cflags} -+and @code{--extra-ldflags}. -+ -+NDI uses uyvy422 pixel format natively, but also supports bgra, bgr0, rgba and -+rgb0. -+ -+@subsection Options -+ -+@table @option -+ -+@item reference_level -+The audio reference level in dB. This specifies how many dB above the -+reference level (+4dBU) is the full range of 16 bit audio. -+Defaults to @option{0}. -+ -+@item clock_video -+These specify whether video "clock" themselves. -+Defaults to @option{false}. -+ -+@item clock_audio -+These specify whether audio "clock" themselves. -+Defaults to @option{false}. -+ -+@end table -+ -+@subsection Examples -+ -+@itemize -+ -+@item -+Play video clip: -+@example -+ffmpeg -i "udp://@@239.1.1.1:10480?fifo_size=1000000&overrun_nonfatal=1" -vf "scale=720:576,fps=fps=25,setdar=dar=16/9,format=pix_fmts=uyvy422" -f libndi_newtek NEW_NDI1 -+@end example -+ -+@end itemize -+ - @section opengl - OpenGL output device. Deprecated and will be removed. - -diff --git a/libavdevice/Makefile b/libavdevice/Makefile -index c304492..290ae8b 100644 ---- a/libavdevice/Makefile -+++ b/libavdevice/Makefile -@@ -22,6 +22,8 @@ OBJS-$(CONFIG_BKTR_INDEV) += bktr.o - OBJS-$(CONFIG_CACA_OUTDEV) += caca.o - OBJS-$(CONFIG_DECKLINK_OUTDEV) += decklink_enc.o decklink_enc_c.o decklink_common.o - OBJS-$(CONFIG_DECKLINK_INDEV) += decklink_dec.o decklink_dec_c.o decklink_common.o -+OBJS-$(CONFIG_LIBNDI_NEWTEK_OUTDEV) += libndi_newtek_enc.o -+OBJS-$(CONFIG_LIBNDI_NEWTEK_INDEV) += libndi_newtek_dec.o - OBJS-$(CONFIG_DSHOW_INDEV) += dshow_crossbar.o dshow.o dshow_enummediatypes.o \ - dshow_enumpins.o dshow_filter.o \ - dshow_pin.o dshow_common.o -@@ -65,6 +67,8 @@ SHLIBOBJS-$(HAVE_GNU_WINDRES) += avdeviceres.o - SKIPHEADERS += decklink_common.h - SKIPHEADERS-$(CONFIG_DECKLINK) += decklink_enc.h decklink_dec.h \ - decklink_common_c.h -+SKIPHEADERS-$(CONFIG_LIBNDI_NEWTEK_INDEV) += libndi_newtek_common.h -+SKIPHEADERS-$(CONFIG_LIBNDI_NEWTEK_OUTDEV) += libndi_newtek_common.h - SKIPHEADERS-$(CONFIG_DSHOW_INDEV) += dshow_capture.h - SKIPHEADERS-$(CONFIG_FBDEV_INDEV) += fbdev_common.h - SKIPHEADERS-$(CONFIG_FBDEV_OUTDEV) += fbdev_common.h -diff --git a/libavdevice/alldevices.c b/libavdevice/alldevices.c -index 9b9a914..ac942ec 100644 ---- a/libavdevice/alldevices.c -+++ b/libavdevice/alldevices.c -@@ -36,6 +36,8 @@ extern const FFInputFormat ff_bktr_demuxer; - extern const FFOutputFormat ff_caca_muxer; - extern const FFInputFormat ff_decklink_demuxer; - extern const FFOutputFormat ff_decklink_muxer; -+extern const FFInputFormat ff_libndi_newtek_demuxer; -+extern const FFOutputFormat ff_libndi_newtek_muxer; - extern const FFInputFormat ff_dshow_demuxer; - extern const FFInputFormat ff_fbdev_demuxer; - extern const FFOutputFormat ff_fbdev_muxer; diff --git a/deps/ffmpeg/7.1.5/README.md b/deps/ffmpeg/7.1.5/README.md deleted file mode 100644 index b912dcb7..00000000 --- a/deps/ffmpeg/7.1.5/README.md +++ /dev/null @@ -1,56 +0,0 @@ -# FFmpeg Patch Stack - -This directory contains the public FFmpeg patch stack used by avplumber's CUDA -composition and media-input workflows. - -## Base - -- Upstream repository: `https://github.com/FFmpeg/FFmpeg` -- Upstream tag: `n7.1.5` -- Upstream commit: `3a0867c2bfda4a4d4309ca1a8cbdc6175e67f587` -- Expected patched tree: `52361f7251069ef74fbb41460e6e1b65d6f9947c` - -## Series - -The seven patches are ordered by filename and grouped by feature rather than by -the chronology of incomplete ports and follow-up fixes: - -1. `0001-swscale-aarch64-argb-yuva420p.patch` — AArch64 fast color conversion. -2. `0002-avfilter-cuda-composition-suite.patch` — CUDA pad, convert, crop, - overlay, overlay-many, scale edge handling, transitions, and procedural - wipes. -3. `0003-avfilter-npp-cuda13-compat.patch` — CUDA 13 NPP compatibility. -4. `0004-avcodec-nvdec-intra.patch` — NVDEC intra-only stream handling. -5. `0005-avformat-rtp-rfc4175.patch` — RFC 4175 4:2:0 and incomplete-frame - handling. -6. `0006-avdevice-v4l2-compat.patch` — V4L2 timestamp compatibility. -7. `0007-avdevice-ndi-v5.patch` — NDI v5 device registration and documentation. - -Each patch message lists the original exported FFmpeg commits it replaces. - -The old FFmpeg `af_whisper` port is intentionally absent. Speech-to-text belongs -in an AVPlumber node and is not part of this FFmpeg variant. - -## Apply - -```bash -git clone --branch n7.1.5 --depth 1 \ - https://github.com/FFmpeg/FFmpeg clean-ffmpeg -git -C clean-ffmpeg config user.name "patch application" -git -C clean-ffmpeg config user.email "patch-application@local" -git -C clean-ffmpeg am /path/to/avplumber/deps/ffmpeg/7.1.5/*.patch -``` - -## Verify - -Run the verifier with any FFmpeg Git checkout that contains the documented base -commit. It creates and removes an isolated temporary worktree; it does not alter -the checkout's active branch: - -```bash -deps/ffmpeg/7.1.5/verify.sh /path/to/FFmpeg -``` - -Verification succeeds only when all seven patches apply and produce the exact -expected Git tree. Runtime validation is provided by `demos/cuda-overlay` and -`demos/mixer`, whose Dockerfiles build this series against FFmpeg `n7.1.5`. diff --git a/deps/ffmpeg/7.1.5/base.env b/deps/ffmpeg/7.1.5/base.env deleted file mode 100644 index ceb83910..00000000 --- a/deps/ffmpeg/7.1.5/base.env +++ /dev/null @@ -1,3 +0,0 @@ -base_commit=3a0867c2bfda4a4d4309ca1a8cbdc6175e67f587 -expected_tree=52361f7251069ef74fbb41460e6e1b65d6f9947c -expected_patch_count=7 diff --git a/deps/ffmpeg/7.1.5/verify.sh b/deps/ffmpeg/7.1.5/verify.sh deleted file mode 100755 index 1256df3c..00000000 --- a/deps/ffmpeg/7.1.5/verify.sh +++ /dev/null @@ -1,4 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail -script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) -exec bash "$script_dir/../verify.sh" 7.1.5 "$@" diff --git a/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch b/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch deleted file mode 100644 index 53559f75..00000000 --- a/deps/ffmpeg/8.1/0003-avfilter-npp-cuda13-compat.patch +++ /dev/null @@ -1,322 +0,0 @@ -From 9492ef1cc9a0473ee4e99549cff3eb08166c9675 Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Wed, 12 Aug 2026 15:54:13 +0000 -Subject: [PATCH 3/7] avfilter/npp: support CUDA 13 stream context APIs - -Add an NPP compatibility layer and use explicit CUDA stream contexts in scale, sharpen, and transpose filters. - -Original-commit: 30b851f13d97bad9b0b28e7310122d914ac6fc1d ---- - configure | 4 +-- - libavfilter/cuda/npp_compat.h | 65 ++++++++++++++++++++++++++++++++++ - libavfilter/vf_scale_npp.c | 56 ++++++++++++++++++++--------- - libavfilter/vf_sharpen_npp.c | 12 +++++-- - libavfilter/vf_transpose_npp.c | 35 ++++++++++++------ - 5 files changed, 142 insertions(+), 30 deletions(-) - create mode 100644 libavfilter/cuda/npp_compat.h - -diff --git a/configure b/configure -index 584e1df313..932be316d4 100755 ---- a/configure -+++ b/configure -@@ -7338,8 +7338,8 @@ enabled libnpp && { test_cpp_condition "$(cd "$source_path"; pwd)/lib - { check_lib libnpp npp.h nppGetLibVersion -lnppig -lnppicc -lnppc -lnppidei -lnppif || - check_lib libnpp npp.h nppGetLibVersion -lnppi -lnppif -lnppc -lnppidei || - die "ERROR: libnpp not found"; } && -- { check_func_headers "nppi.h" nppiYCbCr420_8u_P2P3R $libnpp_extralibs || -- die "ERROR: libnpp support is deprecated, version 13.0 and up are not supported"; } -+ { check_func_headers "nppi.h" nppiYCbCr420_8u_P2P3R_Ctx $libnpp_extralibs || -+ die "ERROR: libnpp stream context APIs not found"; } - enabled libopencore_amrnb && { check_pkg_config libopencore_amrnb opencore-amrnb opencore-amrnb/interf_dec.h Decoder_Interface_init || - require libopencore_amrnb opencore-amrnb/interf_dec.h Decoder_Interface_init -lopencore-amrnb; } - enabled libopencore_amrwb && { check_pkg_config libopencore_amrwb opencore-amrwb opencore-amrwb/dec_if.h D_IF_init || -diff --git a/libavfilter/cuda/npp_compat.h b/libavfilter/cuda/npp_compat.h -new file mode 100644 -index 0000000000..bf5d73e6cd ---- /dev/null -+++ b/libavfilter/cuda/npp_compat.h -@@ -0,0 +1,65 @@ -+/* -+ * CUDA NPP compatibility helpers. -+ * -+ * This file is part of FFmpeg. -+ */ -+ -+#ifndef AVFILTER_CUDA_NPP_COMPAT_H -+#define AVFILTER_CUDA_NPP_COMPAT_H -+ -+#include -+ -+#include -+ -+#include "libavutil/cuda_check.h" -+#include "libavutil/hwcontext_cuda_internal.h" -+ -+#ifndef CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_BLOCK -+#define CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_BLOCK ((CUdevice_attribute)1) -+#endif -+#ifndef CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK -+#define CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK ((CUdevice_attribute)8) -+#endif -+#ifndef CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_MULTIPROCESSOR -+#define CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_MULTIPROCESSOR ((CUdevice_attribute)39) -+#endif -+ -+static inline int ff_npp_get_stream_context(void *logctx, AVCUDADeviceContext *device_hwctx, -+ NppStreamContext *npp_ctx) -+{ -+ CudaFunctions *cu = device_hwctx->internal->cuda_dl; -+ CUdevice dev = device_hwctx->internal->cuda_device; -+ int value; -+ int ret; -+ -+ memset(npp_ctx, 0, sizeof(*npp_ctx)); -+ npp_ctx->hStream = (cudaStream_t)device_hwctx->stream; -+ npp_ctx->nCudaDeviceId = dev; -+ -+#define GET_NPP_DEVICE_ATTR(dst, attr) do { \ -+ ret = FF_CUDA_CHECK_DL(logctx, cu, \ -+ cu->cuDeviceGetAttribute(&(dst), (attr), dev)); \ -+ if (ret < 0) \ -+ return ret; \ -+ } while (0) -+ -+ GET_NPP_DEVICE_ATTR(npp_ctx->nMultiProcessorCount, -+ CU_DEVICE_ATTRIBUTE_MULTIPROCESSOR_COUNT); -+ GET_NPP_DEVICE_ATTR(npp_ctx->nMaxThreadsPerMultiProcessor, -+ CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_MULTIPROCESSOR); -+ GET_NPP_DEVICE_ATTR(npp_ctx->nMaxThreadsPerBlock, -+ CU_DEVICE_ATTRIBUTE_MAX_THREADS_PER_BLOCK); -+ GET_NPP_DEVICE_ATTR(value, -+ CU_DEVICE_ATTRIBUTE_MAX_SHARED_MEMORY_PER_BLOCK); -+ npp_ctx->nSharedMemPerBlock = value; -+ GET_NPP_DEVICE_ATTR(npp_ctx->nCudaDevAttrComputeCapabilityMajor, -+ CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR); -+ GET_NPP_DEVICE_ATTR(npp_ctx->nCudaDevAttrComputeCapabilityMinor, -+ CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR); -+ -+#undef GET_NPP_DEVICE_ATTR -+ -+ return 0; -+} -+ -+#endif /* AVFILTER_CUDA_NPP_COMPAT_H */ -diff --git a/libavfilter/vf_scale_npp.c b/libavfilter/vf_scale_npp.c -index 8e9113521c..de179ed580 100644 ---- a/libavfilter/vf_scale_npp.c -+++ b/libavfilter/vf_scale_npp.c -@@ -36,6 +36,7 @@ - #include "libavutil/pixdesc.h" - - #include "avfilter.h" -+#include "cuda/npp_compat.h" - #include "filters.h" - #include "formats.h" - #include "scale_eval.h" -@@ -696,14 +697,22 @@ static int nppscale_deinterleave(AVFilterContext *ctx, NPPScaleStageContext *sta - AVFrame *out, AVFrame *in) - { - AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; -+ AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; -+ NppStreamContext npp_ctx; - NppStatus err; -+ int ret; -+ -+ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); -+ if (ret < 0) -+ return ret; - - switch (in_frames_ctx->sw_format) { - case AV_PIX_FMT_NV12: -- err = nppiYCbCr420_8u_P2P3R(in->data[0], in->linesize[0], -- in->data[1], in->linesize[1], -- out->data, out->linesize, -- (NppiSize){ in->width, in->height }); -+ err = nppiYCbCr420_8u_P2P3R_Ctx(in->data[0], in->linesize[0], -+ in->data[1], in->linesize[1], -+ out->data, out->linesize, -+ (NppiSize){ in->width, in->height }, -+ npp_ctx); - break; - default: - return AVERROR_BUG; -@@ -719,9 +728,16 @@ static int nppscale_deinterleave(AVFilterContext *ctx, NPPScaleStageContext *sta - static int nppscale_resize(AVFilterContext *ctx, NPPScaleStageContext *stage, - AVFrame *out, AVFrame *in) - { -+ AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; -+ AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; - NPPScaleContext *s = ctx->priv; -+ NppStreamContext npp_ctx; - NppStatus err; -- int i; -+ int i, ret; -+ -+ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); -+ if (ret < 0) -+ return ret; - - for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { - int iw = stage->planes_in[i].width; -@@ -729,12 +745,12 @@ static int nppscale_resize(AVFilterContext *ctx, NPPScaleStageContext *stage, - int ow = stage->planes_out[i].width; - int oh = stage->planes_out[i].height; - -- err = nppiResizeSqrPixel_8u_C1R(in->data[i], (NppiSize){ iw, ih }, -- in->linesize[i], (NppiRect){ 0, 0, iw, ih }, -- out->data[i], out->linesize[i], -- (NppiRect){ 0, 0, ow, oh }, -- (double)ow / iw, (double)oh / ih, -- 0.0, 0.0, s->interp_algo); -+ err = nppiResizeSqrPixel_8u_C1R_Ctx(in->data[i], (NppiSize){ iw, ih }, -+ in->linesize[i], (NppiRect){ 0, 0, iw, ih }, -+ out->data[i], out->linesize[i], -+ (NppiRect){ 0, 0, ow, oh }, -+ (double)ow / iw, (double)oh / ih, -+ 0.0, 0.0, s->interp_algo, npp_ctx); - if (err != NPP_SUCCESS) { - av_log(ctx, AV_LOG_ERROR, "NPP resize error: %d\n", err); - return AVERROR_UNKNOWN; -@@ -748,15 +764,23 @@ static int nppscale_interleave(AVFilterContext *ctx, NPPScaleStageContext *stage - AVFrame *out, AVFrame *in) - { - AVHWFramesContext *out_frames_ctx = (AVHWFramesContext*)out->hw_frames_ctx->data; -+ AVCUDADeviceContext *device_hwctx = out_frames_ctx->device_ctx->hwctx; -+ NppStreamContext npp_ctx; - NppStatus err; -+ int ret; -+ -+ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); -+ if (ret < 0) -+ return ret; - - switch (out_frames_ctx->sw_format) { - case AV_PIX_FMT_NV12: -- err = nppiYCbCr420_8u_P3P2R((const uint8_t**)in->data, -- in->linesize, -- out->data[0], out->linesize[0], -- out->data[1], out->linesize[1], -- (NppiSize){ in->width, in->height }); -+ err = nppiYCbCr420_8u_P3P2R_Ctx((const uint8_t**)in->data, -+ in->linesize, -+ out->data[0], out->linesize[0], -+ out->data[1], out->linesize[1], -+ (NppiSize){ in->width, in->height }, -+ npp_ctx); - break; - default: - return AVERROR_BUG; -diff --git a/libavfilter/vf_sharpen_npp.c b/libavfilter/vf_sharpen_npp.c -index 3ec74f8c0c..e3f378292f 100644 ---- a/libavfilter/vf_sharpen_npp.c -+++ b/libavfilter/vf_sharpen_npp.c -@@ -24,6 +24,7 @@ - #include - #include - -+#include "cuda/npp_compat.h" - #include "filters.h" - #include "libavutil/pixdesc.h" - #include "libavutil/cuda_check.h" -@@ -159,17 +160,24 @@ static int nppsharpen_sharpen(AVFilterContext* ctx, AVFrame* out, AVFrame* in) - { - FilterLink *inl = ff_filter_link(ctx->inputs[0]); - AVHWFramesContext* in_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; -+ AVCUDADeviceContext *device_hwctx = in_ctx->device_ctx->hwctx; - NPPSharpenContext* s = ctx->priv; - - const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(in_ctx->sw_format); -+ NppStreamContext npp_ctx; -+ int ret; -+ -+ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); -+ if (ret < 0) -+ return ret; - - for (int i = 0; i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { - int ow = AV_CEIL_RSHIFT(in->width, (i == 1 || i == 2) ? desc->log2_chroma_w : 0); - int oh = AV_CEIL_RSHIFT(in->height, (i == 1 || i == 2) ? desc->log2_chroma_h : 0); - -- NppStatus err = nppiFilterSharpenBorder_8u_C1R( -+ NppStatus err = nppiFilterSharpenBorder_8u_C1R_Ctx( - in->data[i], in->linesize[i], (NppiSize){ow, oh}, (NppiPoint){0, 0}, -- out->data[i], out->linesize[i], (NppiSize){ow, oh}, s->border_type); -+ out->data[i], out->linesize[i], (NppiSize){ow, oh}, s->border_type, npp_ctx); - if (err != NPP_SUCCESS) { - av_log(ctx, AV_LOG_ERROR, "NPP sharpen error: %d\n", err); - return AVERROR_EXTERNAL; -diff --git a/libavfilter/vf_transpose_npp.c b/libavfilter/vf_transpose_npp.c -index 2315b1043a..ad43356523 100644 ---- a/libavfilter/vf_transpose_npp.c -+++ b/libavfilter/vf_transpose_npp.c -@@ -29,6 +29,7 @@ - #include "libavutil/pixdesc.h" - - #include "avfilter.h" -+#include "cuda/npp_compat.h" - #include "filters.h" - #include "formats.h" - #include "video.h" -@@ -294,9 +295,16 @@ static int npptranspose_config_props(AVFilterLink *outlink) - static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *stage, - AVFrame *out, AVFrame *in) - { -+ AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; -+ AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; - NPPTransposeContext *s = ctx->priv; -+ NppStreamContext npp_ctx; - NppStatus err; -- int i; -+ int i, ret; -+ -+ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); -+ if (ret < 0) -+ return ret; - - for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { - int iw = stage->planes_in[i].width; -@@ -311,11 +319,11 @@ static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *s - int shiftw = (s->dir == NPP_TRANSPOSE_CLOCK || s->dir == NPP_TRANSPOSE_CLOCK_FLIP) ? ow - 1 : 0; - int shifth = (s->dir == NPP_TRANSPOSE_CCLOCK || s->dir == NPP_TRANSPOSE_CLOCK_FLIP) ? oh - 1 : 0; - -- err = nppiRotate_8u_C1R(in->data[i], (NppiSize){ iw, ih }, -- in->linesize[i], (NppiRect){ 0, 0, iw, ih }, -- out->data[i], out->linesize[i], -- (NppiRect){ 0, 0, ow, oh }, -- angle, shiftw, shifth, NPPI_INTER_NN); -+ err = nppiRotate_8u_C1R_Ctx(in->data[i], (NppiSize){ iw, ih }, -+ in->linesize[i], (NppiRect){ 0, 0, iw, ih }, -+ out->data[i], out->linesize[i], -+ (NppiRect){ 0, 0, ow, oh }, -+ angle, shiftw, shifth, NPPI_INTER_NN, npp_ctx); - if (err != NPP_SUCCESS) { - av_log(ctx, AV_LOG_ERROR, "NPP rotate error: %d\n", err); - return AVERROR_UNKNOWN; -@@ -328,16 +336,23 @@ static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *s - static int npptranspose_transpose(AVFilterContext *ctx, NPPTransposeStageContext *stage, - AVFrame *out, AVFrame *in) - { -+ AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; -+ AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; -+ NppStreamContext npp_ctx; - NppStatus err; -- int i; -+ int i, ret; -+ -+ ret = ff_npp_get_stream_context(ctx, device_hwctx, &npp_ctx); -+ if (ret < 0) -+ return ret; - - for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { - int iw = stage->planes_in[i].width; - int ih = stage->planes_in[i].height; - -- err = nppiTranspose_8u_C1R(in->data[i], in->linesize[i], -- out->data[i], out->linesize[i], -- (NppiSize){ iw, ih }); -+ err = nppiTranspose_8u_C1R_Ctx(in->data[i], in->linesize[i], -+ out->data[i], out->linesize[i], -+ (NppiSize){ iw, ih }, npp_ctx); - if (err != NPP_SUCCESS) { - av_log(ctx, AV_LOG_ERROR, "NPP transpose error: %d\n", err); - return AVERROR_UNKNOWN; diff --git a/deps/ffmpeg/8.1/0005-avformat-rtp-rfc4175.patch b/deps/ffmpeg/8.1/0005-avformat-rtp-rfc4175.patch deleted file mode 100644 index f9f24c8d..00000000 --- a/deps/ffmpeg/8.1/0005-avformat-rtp-rfc4175.patch +++ /dev/null @@ -1,168 +0,0 @@ -From bab76e3c3f8cf7b7a26bae454cd7736cbb8c30a7 Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Wed, 12 Aug 2026 15:54:13 +0000 -Subject: [PATCH 5/7] avformat/rtp: extend RFC 4175 frame handling - -Support the legacy 4:2:0 payload layout and skip incomplete RFC 4175 frames without emitting corrupted output. - -Original-commits: e4b4dbdccaa82b007bf3f320d7fd761f1cbd9a14 364d9773b677feedc39ed4a3e30a15e95878b811 ---- - libavformat/rtpdec_rfc4175.c | 88 ++++++++++++++++++++++++++++++++---- - 1 file changed, 80 insertions(+), 8 deletions(-) - -diff --git a/libavformat/rtpdec_rfc4175.c b/libavformat/rtpdec_rfc4175.c -index b49fc55d2d..2f0dff0bf0 100644 ---- a/libavformat/rtpdec_rfc4175.c -+++ b/libavformat/rtpdec_rfc4175.c -@@ -43,6 +43,9 @@ struct PayloadContext { - unsigned int frame_size; - unsigned int pgroup; /* size of the pixel group in bytes */ - unsigned int xinc; -+ int is_yuv420; -+ int next_offset; -+ int next_line; - - uint32_t timestamp; - }; -@@ -53,6 +56,7 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) - int tag; - const AVPixFmtDescriptor *desc; - -+ data->is_yuv420 = 0; - if (!strncmp(data->sampling, "YCbCr-4:2:2", 11)) { - tag = MKTAG('U', 'Y', 'V', 'Y'); - data->xinc = 2; -@@ -71,6 +75,7 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) - } else if (!strncmp(data->sampling, "YCbCr-4:2:0", 11)) { - tag = MKTAG('I', '4', '2', '0'); - data->xinc = 4; -+ data->is_yuv420 = 1; - - if (data->depth == 8) { - data->pgroup = 6; -@@ -108,6 +113,8 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) - stream->codecpar->codec_tag = tag; - stream->codecpar->bits_per_coded_sample = av_get_bits_per_pixel(desc); - data->frame_size = data->width * data->height * data->pgroup / data->xinc; -+ data->next_offset = 0; -+ data->next_line = 0; - - if (data->interlaced) - stream->codecpar->field_order = AV_FIELD_TT; -@@ -225,6 +232,8 @@ static int rfc4175_finalize_packet(PayloadContext *data, AVPacket *pkt, - av_freep(&data->frame); - } - data->frame = NULL; -+ data->next_offset = 0; -+ data->next_line = 0; - } - - data->field = 0; -@@ -232,6 +241,14 @@ static int rfc4175_finalize_packet(PayloadContext *data, AVPacket *pkt, - return ret; - } - -+static void rfc4175_discard_packet(PayloadContext *data) -+{ -+ av_freep(&data->frame); -+ data->frame = NULL; -+ data->next_offset = 0; -+ data->next_line = 0; -+} -+ - static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - AVStream *st, AVPacket *pkt, uint32_t *timestamp, - const uint8_t * buf, int len, -@@ -255,8 +272,12 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - * previous frame (or pair of fields) anyway by filling the AVPacket. - */ - av_log(ctx, AV_LOG_ERROR, "Missed previous RTP Marker\n"); -- missed_last_packet = 1; -- rfc4175_finalize_packet(data, pkt, st->index); -+ if (ctx->flags & AVFMT_FLAG_DISCARD_CORRUPT) { -+ rfc4175_discard_packet(data); -+ } else { -+ missed_last_packet = 1; -+ rfc4175_finalize_packet(data, pkt, st->index); -+ } - } - - if (!data->frame) -@@ -270,6 +291,11 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - } - } - -+ if (!data->frame) { -+ /* buffer already freed by discard, or duplicate packet */ -+ return AVERROR(EAGAIN); -+ } -+ - /* - * looks for the 'Continuation bit' in scan lines' headers - * to find where data start -@@ -310,13 +336,59 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - if (line >= data->height) - return AVERROR_INVALIDDATA; - -- /* prevent ill-formed packets to write after buffer's end */ -- copy_offset = (line * data->width + offset) * data->pgroup / data->xinc; -- if (copy_offset + length > data->frame_size || !data->frame) -- return AVERROR_INVALIDDATA; -+ if ((ctx->flags & AVFMT_FLAG_DISCARD_CORRUPT) && -+ (data->next_line != line || data->next_offset != offset)) { -+ av_log(ctx, AV_LOG_ERROR, -+ "packet loss: line %d -> %d, offset %d -> %d\n", -+ data->next_line, line, data->next_offset, offset); -+ rfc4175_discard_packet(data); -+ return AVERROR(EAGAIN); -+ } -+ if (ctx->flags & AVFMT_FLAG_DISCARD_CORRUPT) { -+ data->next_offset = offset + length * data->xinc / -+ (data->pgroup * (1 + !!data->is_yuv420)); -+ if (data->next_offset >= data->width) { -+ data->next_offset = 0; -+ data->next_line = line + 1 + !!data->is_yuv420; -+ } -+ } - -- dest = data->frame + copy_offset; -- memcpy(dest, payload, length); -+ if (data->is_yuv420) { -+ uint8_t *yd1p, *yd2p, *up, *vp, *udp, *vdp; -+ const uint8_t *p; -+ int uvoff, i; -+ -+ /* libavformat stores YUV420 planar frames, depacketized payload is interleaved */ -+ if (line % 2 != 0 || !data->frame || (offset % 2) != 0 || (length % 6) != 0) -+ return AVERROR_INVALIDDATA; -+ -+ yd1p = data->frame + line * data->width + offset; -+ yd2p = yd1p + data->width; -+ up = data->frame + data->width * data->height; -+ vp = up + data->width / 2 * data->height / 2; -+ uvoff = line / 2 * data->width / 2 + offset / 2; -+ udp = up + uvoff; -+ vdp = vp + uvoff; -+ p = payload; -+ -+ for (i = 0; i < length; i += 6) { -+ *yd1p++ = p[0]; -+ *yd1p++ = p[1]; -+ *yd2p++ = p[2]; -+ *yd2p++ = p[3]; -+ *udp++ = p[4]; -+ *vdp++ = p[5]; -+ p += 6; -+ } -+ } else { -+ /* prevent ill-formed packets to write after buffer's end */ -+ copy_offset = (line * data->width + offset) * data->pgroup / data->xinc; -+ if (copy_offset + length > data->frame_size || !data->frame) -+ return AVERROR_INVALIDDATA; -+ -+ dest = data->frame + copy_offset; -+ memcpy(dest, payload, length); -+ } - - payload += length; - payload_len -= length; diff --git a/deps/ffmpeg/8.1/README.md b/deps/ffmpeg/8.1/README.md deleted file mode 100644 index 4aaef1e0..00000000 --- a/deps/ffmpeg/8.1/README.md +++ /dev/null @@ -1,159 +0,0 @@ -# FFmpeg 8.1 compatibility series - -Base: upstream `n8.1`, commit `9047fa1b084f76b1b4d065af2d743df1b40dfb56`. -The exact patched tree is recorded in `base.env`. - -This is the 8.1 adaptation of the seven features in `../7.1.5`, not a new -pixel-format or mixer pipeline design. - -## Adaptations - -1. **AArch64 ARGB conversion:** use `SwsInternal`, `opts` fields and the updated - unscaled callback signature. Retain the existing conversion algorithms. -2. **CUDA composition:** use `FFFilter` registrations for `convert_cuda`, - `crop_cuda`, `overlay_many_cuda`, `pad_cuda` and `transition_cuda`. - The custom `pad_cuda` implementation intentionally replaces the upstream - implementation in this series; switching padding semantics is out of scope. - Keep upstream 8.1 `scale_cuda`, including its expanded pixel formats, and - carry the existing scaling-edge and overlay-context fixes. Retain YUVA444P - CUDA frame support. Use upstream's existing compute-75 compiler fallback. -3. **NPP CUDA 13 compatibility:** carry the stream-context helper and filter - changes; probe the stream-context API in configure, since NPP 13 removes - the legacy API checked by upstream. The mixer demo build keeps NPP disabled. -4. **NVDEC intra-only handling:** move the existing initialization hunk to its - corresponding 8.1 location. -5. **RFC 4175:** carry the existing frame handling patch. -6. **V4L2:** carry the existing source-timestamp patch. -7. **NDI v5 registration:** rebase configuration and documentation. As in the - 7.1.5 series, this patch only registers the optional integration; it does not - supply the NDI device implementation files or SDK. NDI remains disabled in - the demo build and is not covered by its compile check. - -FFmpeg 8 removed `AVFrame.pkt_pos`. The legacy `transition_cuda` expression -variable `pos` therefore evaluates to `NAN` (unavailable). `crop_cuda` already -guards that legacy variable by FFmpeg API version. Time/frame-based expressions -and the demo transition configuration do not use packet byte positions. - -FFmpeg 8 also removed the `C` command-support marker from `-filters` output. -The mixer Dockerfile checks the transition's runtime-capable `mode` option in -filter help instead; this check works with both series. - -The AVP filter node sets the buffer source's `hw_frames_ctx` before initializing -the filter. FFmpeg 8.1 validates CUDA input formats during initialization; -setting the context after `avfilter_graph_create_filter` is too late. -The metadata-driven CUDA crop node builds its own graph and follows the same -allocate -> attach frames context -> initialize sequence. Updating only the -generic filter node does not cover crop, portrait, or two-box output paths. -Filters that request a hardware device also receive it before initialization. -The AVP node uses FFmpeg's segmented graph parser to attach `hw_device_ctx` -between filter allocation and initialization; this is needed by `hwupload` -when preloading alpha wipes. These APIs are also available in FFmpeg 7.1.5, -so the same AVP filter source supports both versions without a version fork. -The binaries and Python modules must still be built separately for each -FFmpeg ABI. The 7.1.5 patch series and default Docker build version are unchanged. - -## Hardware acceleration gains and limits - -Compared with upstream n7.1.5, the n8.1 CUDA/NVIDIA path makes these features -available to applications that select the corresponding formats and codecs: - -| Capability | Gain | Requirement / current coverage | -| --- | --- | --- | -| H.264 10-bit NVDEC/NVENC | Hardware High10 decode and encode | Blackwell GPU, SDK 13 headers and compatible driver; not tested on Blackwell. | -| H.264 / HEVC 4:2:2 NVDEC/NVENC | Hardware paths for higher chroma resolution, including 10-bit 4:2:2 | Blackwell GPU, SDK 13 headers and compatible driver; not provided by a T4 or L4 upgrade to FFmpeg alone. | -| CUDA scaling formats | Adds planar 4:2:2, NV16, P210/P216 and planar 10-bit 4:2:0/4:2:2/4:4:4 to upstream `scale_cuda` | CUDA format/scaling support is distinct from hardware codec support. P010 10-bit 4:2:0 already existed in 7.1.5. | -| Existing HEVC Main10 | Remains available on supporting GPUs | Not a new 8.1 capability; this PR does not qualify an end-to-end Main10 graph. | -| Existing custom CUDA composition and CUDA 13 NPP | Keeps the seven-patch suite buildable and usable with the new FFmpeg API | Tested 8-bit paths; custom padding/overlays/inference do not become 10-bit or 4:2:2 automatically. | - -The recorder and mixer remain configured for 8-bit NV12/4:2:0. A 10-bit or 4:2:2 -end-to-end product pipeline still needs compatible decode, filter, composition, -inference and encode stages, plus matching frame metadata. This update does not -add HDR tone mapping or qualify HDR metadata preservation. T4 testing cannot -establish Blackwell codec support or throughput gains. - -SDK-dependent NVENC options are compiled conditionally. The mixer demo still -pins `NV_CODEC_HEADERS_TAG=n12.1.14.0`; a Blackwell build must select SDK 13-era -headers and a matching driver as well as `FFMPEG_TAG=n8.1`. - -Sources: [NVIDIA SDK 13 release notes](https://docs.nvidia.com/video-technologies/video-codec-sdk/13.0/read-me/index.html), -[FFmpeg n8.1 H.264 NVENC profiles](https://github.com/FFmpeg/FFmpeg/blob/n8.1/libavcodec/nvenc_h264.c), -[FFmpeg n8.1 CUDA scaler](https://github.com/FFmpeg/FFmpeg/blob/n8.1/libavfilter/vf_scale_cuda.c), -[FFmpeg n7.1.5 CUDA scaler](https://github.com/FFmpeg/FFmpeg/blob/n7.1.5/libavfilter/vf_scale_cuda.c). - -## Apply and verify - -```bash -git clone --branch n8.1 --depth 1 https://github.com/FFmpeg/FFmpeg -git -C am /deps/ffmpeg/8.1/*.patch -deps/ffmpeg/8.1/verify.sh -``` - -Compilation/linking and filter registration are the acceptance criteria for this -port. They do not establish runtime correctness, performance, or support for -uncompiled optional NPP, NDI or AArch64 paths. Do not deploy over an existing -demo until separate runtime validation is completed. - -Validated on 2026-09-15 in an isolated x86-64 CUDA development container: - -- Both ordered patch series reproduce their pinned trees. -- FFmpeg 8.1 builds; all seven composition/scaling filters are registered. -- Pinned avcpp `31de3f4f937ed3bb30d083275e5e76192dfc9cb3` builds unchanged. -- avplumber binary and Python module build with CUDA/NVCC/DRM/GL enabled, - FRUC/neural/TensorRT disabled. -- No GPU media graph or running demo was changed by this compile check. - -Subsequent GPU runtime checks covered the 8-bit mixer at 30 and 60 fps, 41 prewarmed -scenes, video and DMA-BUF browser inputs, all five wipe-cache loads, cut/fade/wipe -commands, and NVENC output received by a WebRTC browser. The same filter source -also compiled against FFmpeg 7.1.5 and passed CUDA scaling and alpha-upload tests. - -When reusing a build tree with different GL feature flags, rebuild -`deps/cuda_loader/cuda_drvapi_dynlink.o` with the new flags. A loader compiled -without GL lacks the EGL function-pointer variables. Linking `libcuda` directly -to satisfy those missing symbols is incorrect: it supplies functions where AVP -expects variables and crashes during DMA-BUF import. The validated mixer module -uses the GL-enabled dynamic loader, without direct `libcuda` linkage. - -The current avcpp pin additionally backports custom-IO allocation/cleanup fixes -and CMake link-list handling. These retain the existing wrapper API; they are -maintenance fixes, not requirements for FFmpeg 8.1 compilation or a v3 migration. - -## Full reframer and composition checks - -The 2026-09-15 T4 check with the reframer's eight-patch FFmpeg 8.1 runtime, -CUDA 13/NPP, TensorRT and legacy float TrackNet covered native 1080p25 input, -H=20 camera-pan planning, Player 360p, salient detection, frame classification, -15 Hz scoreboard OCR, DMA-BUF browser overlays, portrait/square crops and eight -HLS video renditions. All 2,502 measured frames reached every pre-NVENC branch; -the latency collector reported no incomplete frames or dropped packets. All -eight finalized renditions were 25 fps and 100.2 seconds long. - -This validates functionality, not steady low-latency performance: processing -had catch-up bursts, with post-NVDEC-to-pre-NVENC latency of 1.095 s median, -3.635 s p95 and 4.359 s maximum. GPU utilization was 54% median and peak device -memory was 2,558 MiB. Native 60 fps remains unqualified. - -The independent `demos/cuda-overlay` pixel-reference matrix passed all 45 cases -on FFmpeg 8.1: 1-15 overlays in 420/420, 420/444 and 444/444 combinations, -including a 641-pixel-wide canvas. Every compared YUV sample matched. - -`tests/cuda/smoke_crop_filter_chain.py` exercises the AVP crop node together -with CUDA padding, scaling, format conversion and NVENC. -Use `--scaler scale_npp` to cover NPP and `--band-blur` when the reframer's -optional `band_blur_cuda` patch is installed. CPU decoding is only the final -encoded-output assertion, not a transfer inside the CUDA processing chain. - -## Live recorder EOF regression - -A live SRT disconnect can finish the input group and propagate EOF into the -permanent pre-sentinel format declaration. `ignore_eof=true` on -`fake_video_format` / `fake_audio_metadata` keeps those nodes accepting frames -across reconnection. This is opt-in; default finite-graph EOF still propagates. - -The FFmpeg 8.1 T4 recorder check survived two SRT disconnects with one recorder -generation. Its 1,600 consecutive 25 fps metadata records matched Kafka and GCS -JSONL; all primary HLS outputs contained 64 finalized one-second segments. -Native audio/video reconnect tests preserve their decoded frame timestamp -sequences, while default finite-EOF tests still finish. The unpatched image -fails the live-EOF regression. Full-recorder finite-VOD completion remains -separate work; the recorder still applies its live restart policy to file input. diff --git a/deps/ffmpeg/8.1/base.env b/deps/ffmpeg/8.1/base.env deleted file mode 100644 index 0ccaa766..00000000 --- a/deps/ffmpeg/8.1/base.env +++ /dev/null @@ -1,3 +0,0 @@ -base_commit=9047fa1b084f76b1b4d065af2d743df1b40dfb56 -expected_tree=3f4f8ba98c493fe9c0e0bfae5a61cd8b83fc30ca -expected_patch_count=7 diff --git a/deps/ffmpeg/8.1/verify.sh b/deps/ffmpeg/8.1/verify.sh deleted file mode 100755 index ab01ece5..00000000 --- a/deps/ffmpeg/8.1/verify.sh +++ /dev/null @@ -1,4 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail -script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) -exec bash "$script_dir/../verify.sh" 8.1 "$@" diff --git a/deps/ffmpeg/8.1/0001-swscale-aarch64-argb-yuva420p.patch b/deps/ffmpeg/8/0001-swscale-aarch64-argb-yuva420p.patch similarity index 100% rename from deps/ffmpeg/8.1/0001-swscale-aarch64-argb-yuva420p.patch rename to deps/ffmpeg/8/0001-swscale-aarch64-argb-yuva420p.patch diff --git a/deps/ffmpeg/8.1/0002-avfilter-cuda-composition-suite.patch b/deps/ffmpeg/8/0002-avfilter-cuda-composition-suite.patch similarity index 100% rename from deps/ffmpeg/8.1/0002-avfilter-cuda-composition-suite.patch rename to deps/ffmpeg/8/0002-avfilter-cuda-composition-suite.patch diff --git a/deps/ffmpeg/7.1.5/0003-avfilter-npp-cuda13-compat.patch b/deps/ffmpeg/8/0003-avfilter-npp-cuda13-compat.patch similarity index 72% rename from deps/ffmpeg/7.1.5/0003-avfilter-npp-cuda13-compat.patch rename to deps/ffmpeg/8/0003-avfilter-npp-cuda13-compat.patch index f745d185..d82d6839 100644 --- a/deps/ffmpeg/7.1.5/0003-avfilter-npp-cuda13-compat.patch +++ b/deps/ffmpeg/8/0003-avfilter-npp-cuda13-compat.patch @@ -1,11 +1,8 @@ -From e5ba36a6cfabc10f5c745fba9e8a91fef92cf4a7 Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Wed, 12 Aug 2026 15:54:13 +0000 -Subject: [PATCH 3/7] avfilter/npp: support CUDA 13 stream context APIs +From 585909cb4234d7ed8af27158a5fb53bf0c2a2c87 Mon Sep 17 00:00:00 2001 +From: x +Date: Wed, 16 Sep 2026 20:00:13 +0200 +Subject: [PATCH] avfilter/npp: CUDA 13 stream-context compatibility -Add an NPP compatibility layer and use explicit CUDA stream contexts in scale, sharpen, and transpose filters. - -Original-commit: 30b851f13d97bad9b0b28e7310122d914ac6fc1d --- libavfilter/cuda/npp_compat.h | 65 ++++++++++++++++++++++++++++++++++ libavfilter/vf_scale_npp.c | 56 ++++++++++++++++++++--------- @@ -86,20 +83,14 @@ index 0000000..bf5d73e + +#endif /* AVFILTER_CUDA_NPP_COMPAT_H */ diff --git a/libavfilter/vf_scale_npp.c b/libavfilter/vf_scale_npp.c -index 0c38987..f90b291 100644 +index 8e91135..de179ed 100644 --- a/libavfilter/vf_scale_npp.c +++ b/libavfilter/vf_scale_npp.c -@@ -36,6 +36,7 @@ - #include "libavutil/pixdesc.h" - +@@ -38,2 +38,3 @@ #include "avfilter.h" +#include "cuda/npp_compat.h" #include "filters.h" - #include "formats.h" - #include "scale_eval.h" -@@ -708,14 +709,22 @@ static int nppscale_deinterleave(AVFilterContext *ctx, NPPScaleStageContext *sta - AVFrame *out, AVFrame *in) - { +@@ -698,3 +699,10 @@ static int nppscale_deinterleave(AVFilterContext *ctx, NPPScaleStageContext *sta AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; + AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; + NppStreamContext npp_ctx; @@ -110,7 +101,7 @@ index 0c38987..f90b291 100644 + if (ret < 0) + return ret; - switch (in_frames_ctx->sw_format) { +@@ -702,6 +710,7 @@ static int nppscale_deinterleave(AVFilterContext *ctx, NPPScaleStageContext *sta case AV_PIX_FMT_NV12: - err = nppiYCbCr420_8u_P2P3R(in->data[0], in->linesize[0], - in->data[1], in->linesize[1], @@ -122,11 +113,7 @@ index 0c38987..f90b291 100644 + (NppiSize){ in->width, in->height }, + npp_ctx); break; - default: - return AVERROR_BUG; -@@ -731,9 +740,16 @@ static int nppscale_deinterleave(AVFilterContext *ctx, NPPScaleStageContext *sta - static int nppscale_resize(AVFilterContext *ctx, NPPScaleStageContext *stage, - AVFrame *out, AVFrame *in) +@@ -721,5 +730,12 @@ static int nppscale_resize(AVFilterContext *ctx, NPPScaleStageContext *stage, { + AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; + AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; @@ -140,11 +127,7 @@ index 0c38987..f90b291 100644 + if (ret < 0) + return ret; - for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { - int iw = stage->planes_in[i].width; -@@ -741,12 +757,12 @@ static int nppscale_resize(AVFilterContext *ctx, NPPScaleStageContext *stage, - int ow = stage->planes_out[i].width; - int oh = stage->planes_out[i].height; +@@ -731,8 +747,8 @@ static int nppscale_resize(AVFilterContext *ctx, NPPScaleStageContext *stage, - err = nppiResizeSqrPixel_8u_C1R(in->data[i], (NppiSize){ iw, ih }, - in->linesize[i], (NppiRect){ 0, 0, iw, ih }, @@ -159,11 +142,7 @@ index 0c38987..f90b291 100644 + (double)ow / iw, (double)oh / ih, + 0.0, 0.0, s->interp_algo, npp_ctx); if (err != NPP_SUCCESS) { - av_log(ctx, AV_LOG_ERROR, "NPP resize error: %d\n", err); - return AVERROR_UNKNOWN; -@@ -760,15 +776,23 @@ static int nppscale_interleave(AVFilterContext *ctx, NPPScaleStageContext *stage - AVFrame *out, AVFrame *in) - { +@@ -750,3 +766,10 @@ static int nppscale_interleave(AVFilterContext *ctx, NPPScaleStageContext *stage AVHWFramesContext *out_frames_ctx = (AVHWFramesContext*)out->hw_frames_ctx->data; + AVCUDADeviceContext *device_hwctx = out_frames_ctx->device_ctx->hwctx; + NppStreamContext npp_ctx; @@ -174,7 +153,7 @@ index 0c38987..f90b291 100644 + if (ret < 0) + return ret; - switch (out_frames_ctx->sw_format) { +@@ -754,7 +777,8 @@ static int nppscale_interleave(AVFilterContext *ctx, NPPScaleStageContext *stage case AV_PIX_FMT_NV12: - err = nppiYCbCr420_8u_P3P2R((const uint8_t**)in->data, - in->linesize, @@ -188,27 +167,19 @@ index 0c38987..f90b291 100644 + (NppiSize){ in->width, in->height }, + npp_ctx); break; - default: - return AVERROR_BUG; diff --git a/libavfilter/vf_sharpen_npp.c b/libavfilter/vf_sharpen_npp.c -index 4989126..da086fb 100644 +index 3ec74f8..e3f3782 100644 --- a/libavfilter/vf_sharpen_npp.c +++ b/libavfilter/vf_sharpen_npp.c -@@ -24,6 +24,7 @@ - #include - #include +@@ -26,2 +26,3 @@ +#include "cuda/npp_compat.h" #include "filters.h" - #include "libavutil/pixdesc.h" - #include "libavutil/cuda_check.h" -@@ -157,17 +158,24 @@ static int nppsharpen_sharpen(AVFilterContext* ctx, AVFrame* out, AVFrame* in) - { - FilterLink *inl = ff_filter_link(ctx->inputs[0]); +@@ -161,2 +162,3 @@ static int nppsharpen_sharpen(AVFilterContext* ctx, AVFrame* out, AVFrame* in) AVHWFramesContext* in_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; + AVCUDADeviceContext *device_hwctx = in_ctx->device_ctx->hwctx; NPPSharpenContext* s = ctx->priv; - +@@ -164,2 +166,8 @@ static int nppsharpen_sharpen(AVFilterContext* ctx, AVFrame* out, AVFrame* in) const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(in_ctx->sw_format); + NppStreamContext npp_ctx; + int ret; @@ -217,9 +188,7 @@ index 4989126..da086fb 100644 + if (ret < 0) + return ret; - for (int i = 0; i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { - int ow = AV_CEIL_RSHIFT(in->width, (i == 1 || i == 2) ? desc->log2_chroma_w : 0); - int oh = AV_CEIL_RSHIFT(in->height, (i == 1 || i == 2) ? desc->log2_chroma_h : 0); +@@ -169,5 +177,5 @@ static int nppsharpen_sharpen(AVFilterContext* ctx, AVFrame* out, AVFrame* in) - NppStatus err = nppiFilterSharpenBorder_8u_C1R( + NppStatus err = nppiFilterSharpenBorder_8u_C1R_Ctx( @@ -227,23 +196,15 @@ index 4989126..da086fb 100644 - out->data[i], out->linesize[i], (NppiSize){ow, oh}, s->border_type); + out->data[i], out->linesize[i], (NppiSize){ow, oh}, s->border_type, npp_ctx); if (err != NPP_SUCCESS) { - av_log(ctx, AV_LOG_ERROR, "NPP sharpen error: %d\n", err); - return AVERROR_EXTERNAL; diff --git a/libavfilter/vf_transpose_npp.c b/libavfilter/vf_transpose_npp.c -index 1706267..e24dc53 100644 +index 2315b10..ad43356 100644 --- a/libavfilter/vf_transpose_npp.c +++ b/libavfilter/vf_transpose_npp.c -@@ -29,6 +29,7 @@ - #include "libavutil/pixdesc.h" - +@@ -31,2 +31,3 @@ #include "avfilter.h" +#include "cuda/npp_compat.h" #include "filters.h" - #include "formats.h" - #include "video.h" -@@ -292,9 +293,16 @@ static int npptranspose_config_props(AVFilterLink *outlink) - static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *stage, - AVFrame *out, AVFrame *in) +@@ -296,5 +297,12 @@ static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *s { + AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; + AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; @@ -257,11 +218,7 @@ index 1706267..e24dc53 100644 + if (ret < 0) + return ret; - for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { - int iw = stage->planes_in[i].width; -@@ -309,11 +317,11 @@ static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *s - int shiftw = (s->dir == NPP_TRANSPOSE_CLOCK || s->dir == NPP_TRANSPOSE_CLOCK_FLIP) ? ow - 1 : 0; - int shifth = (s->dir == NPP_TRANSPOSE_CCLOCK || s->dir == NPP_TRANSPOSE_CLOCK_FLIP) ? oh - 1 : 0; +@@ -313,7 +321,7 @@ static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *s - err = nppiRotate_8u_C1R(in->data[i], (NppiSize){ iw, ih }, - in->linesize[i], (NppiRect){ 0, 0, iw, ih }, @@ -274,11 +231,7 @@ index 1706267..e24dc53 100644 + (NppiRect){ 0, 0, ow, oh }, + angle, shiftw, shifth, NPPI_INTER_NN, npp_ctx); if (err != NPP_SUCCESS) { - av_log(ctx, AV_LOG_ERROR, "NPP rotate error: %d\n", err); - return AVERROR_UNKNOWN; -@@ -326,16 +334,23 @@ static int npptranspose_rotate(AVFilterContext *ctx, NPPTransposeStageContext *s - static int npptranspose_transpose(AVFilterContext *ctx, NPPTransposeStageContext *stage, - AVFrame *out, AVFrame *in) +@@ -330,4 +338,11 @@ static int npptranspose_transpose(AVFilterContext *ctx, NPPTransposeStageContext { + AVHWFramesContext *in_frames_ctx = (AVHWFramesContext*)in->hw_frames_ctx->data; + AVCUDADeviceContext *device_hwctx = in_frames_ctx->device_ctx->hwctx; @@ -291,9 +244,7 @@ index 1706267..e24dc53 100644 + if (ret < 0) + return ret; - for (i = 0; i < FF_ARRAY_ELEMS(stage->planes_in) && i < FF_ARRAY_ELEMS(in->data) && in->data[i]; i++) { - int iw = stage->planes_in[i].width; - int ih = stage->planes_in[i].height; +@@ -337,5 +352,5 @@ static int npptranspose_transpose(AVFilterContext *ctx, NPPTransposeStageContext - err = nppiTranspose_8u_C1R(in->data[i], in->linesize[i], - out->data[i], out->linesize[i], @@ -302,5 +253,6 @@ index 1706267..e24dc53 100644 + out->data[i], out->linesize[i], + (NppiSize){ iw, ih }, npp_ctx); if (err != NPP_SUCCESS) { - av_log(ctx, AV_LOG_ERROR, "NPP transpose error: %d\n", err); - return AVERROR_UNKNOWN; +-- +2.55.0 + diff --git a/deps/ffmpeg/8.1/0004-avcodec-nvdec-intra.patch b/deps/ffmpeg/8/0004-avcodec-nvdec-intra.patch similarity index 100% rename from deps/ffmpeg/8.1/0004-avcodec-nvdec-intra.patch rename to deps/ffmpeg/8/0004-avcodec-nvdec-intra.patch diff --git a/deps/ffmpeg/7.1.5/0005-avformat-rtp-rfc4175.patch b/deps/ffmpeg/8/0005-avformat-rtp-rfc4175.patch similarity index 65% rename from deps/ffmpeg/7.1.5/0005-avformat-rtp-rfc4175.patch rename to deps/ffmpeg/8/0005-avformat-rtp-rfc4175.patch index 4380e8ac..8c094a34 100644 --- a/deps/ffmpeg/7.1.5/0005-avformat-rtp-rfc4175.patch +++ b/deps/ffmpeg/8/0005-avformat-rtp-rfc4175.patch @@ -1,11 +1,8 @@ -From 1a90f8bb489004be416498b9bd3c1f76e498a675 Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Wed, 12 Aug 2026 15:54:13 +0000 -Subject: [PATCH 5/7] avformat/rtp: extend RFC 4175 frame handling +From 01084d45b0b91ee8d14118b44eeaa84f042c6026 Mon Sep 17 00:00:00 2001 +From: x +Date: Wed, 16 Sep 2026 19:56:02 +0200 +Subject: [PATCH] 0005-avformat-rtp-rfc4175.patch -Support the legacy 4:2:0 payload layout and skip incomplete RFC 4175 frames without emitting corrupted output. - -Original-commits: e4b4dbdccaa82b007bf3f320d7fd761f1cbd9a14 364d9773b677feedc39ed4a3e30a15e95878b811 --- libavformat/rtpdec_rfc4175.c | 88 ++++++++++++++++++++++++++++++++---- 1 file changed, 80 insertions(+), 8 deletions(-) @@ -14,53 +11,31 @@ diff --git a/libavformat/rtpdec_rfc4175.c b/libavformat/rtpdec_rfc4175.c index b49fc55..2f0dff0 100644 --- a/libavformat/rtpdec_rfc4175.c +++ b/libavformat/rtpdec_rfc4175.c -@@ -43,6 +43,9 @@ struct PayloadContext { - unsigned int frame_size; - unsigned int pgroup; /* size of the pixel group in bytes */ +@@ -45,2 +45,5 @@ struct PayloadContext { unsigned int xinc; + int is_yuv420; + int next_offset; + int next_line; - uint32_t timestamp; - }; -@@ -53,6 +56,7 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) - int tag; - const AVPixFmtDescriptor *desc; +@@ -55,2 +58,3 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) + data->is_yuv420 = 0; if (!strncmp(data->sampling, "YCbCr-4:2:2", 11)) { - tag = MKTAG('U', 'Y', 'V', 'Y'); - data->xinc = 2; -@@ -71,6 +75,7 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) - } else if (!strncmp(data->sampling, "YCbCr-4:2:0", 11)) { - tag = MKTAG('I', '4', '2', '0'); +@@ -73,2 +77,3 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) data->xinc = 4; + data->is_yuv420 = 1; - if (data->depth == 8) { - data->pgroup = 6; -@@ -108,6 +113,8 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) - stream->codecpar->codec_tag = tag; - stream->codecpar->bits_per_coded_sample = av_get_bits_per_pixel(desc); +@@ -110,2 +115,4 @@ static int rfc4175_parse_format(AVStream *stream, PayloadContext *data) data->frame_size = data->width * data->height * data->pgroup / data->xinc; + data->next_offset = 0; + data->next_line = 0; - if (data->interlaced) - stream->codecpar->field_order = AV_FIELD_TT; -@@ -225,6 +232,8 @@ static int rfc4175_finalize_packet(PayloadContext *data, AVPacket *pkt, - av_freep(&data->frame); - } +@@ -227,2 +234,4 @@ static int rfc4175_finalize_packet(PayloadContext *data, AVPacket *pkt, data->frame = NULL; + data->next_offset = 0; + data->next_line = 0; } - - data->field = 0; -@@ -232,6 +241,14 @@ static int rfc4175_finalize_packet(PayloadContext *data, AVPacket *pkt, - return ret; - } +@@ -234,2 +243,10 @@ static int rfc4175_finalize_packet(PayloadContext *data, AVPacket *pkt, +static void rfc4175_discard_packet(PayloadContext *data) +{ @@ -71,11 +46,7 @@ index b49fc55..2f0dff0 100644 +} + static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - AVStream *st, AVPacket *pkt, uint32_t *timestamp, - const uint8_t * buf, int len, -@@ -255,8 +272,12 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - * previous frame (or pair of fields) anyway by filling the AVPacket. - */ +@@ -257,4 +274,8 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, av_log(ctx, AV_LOG_ERROR, "Missed previous RTP Marker\n"); - missed_last_packet = 1; - rfc4175_finalize_packet(data, pkt, st->index); @@ -86,11 +57,7 @@ index b49fc55..2f0dff0 100644 + rfc4175_finalize_packet(data, pkt, st->index); + } } - - if (!data->frame) -@@ -270,6 +291,11 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - } - } +@@ -272,2 +293,7 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, + if (!data->frame) { + /* buffer already freed by discard, or duplicate packet */ @@ -98,11 +65,7 @@ index b49fc55..2f0dff0 100644 + } + /* - * looks for the 'Continuation bit' in scan lines' headers - * to find where data start -@@ -310,13 +336,59 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - if (line >= data->height) - return AVERROR_INVALIDDATA; +@@ -312,9 +338,55 @@ static int rfc4175_handle_packet(AVFormatContext *ctx, PayloadContext *data, - /* prevent ill-formed packets to write after buffer's end */ - copy_offset = (line * data->width + offset) * data->pgroup / data->xinc; @@ -164,5 +127,6 @@ index b49fc55..2f0dff0 100644 + memcpy(dest, payload, length); + } - payload += length; - payload_len -= length; +-- +2.55.0 + diff --git a/deps/ffmpeg/8.1/0006-avdevice-v4l2-compat.patch b/deps/ffmpeg/8/0006-avdevice-v4l2-compat.patch similarity index 100% rename from deps/ffmpeg/8.1/0006-avdevice-v4l2-compat.patch rename to deps/ffmpeg/8/0006-avdevice-v4l2-compat.patch diff --git a/deps/ffmpeg/8.1/0007-avdevice-ndi-v5.patch b/deps/ffmpeg/8/0007-avdevice-ndi-v5.patch similarity index 65% rename from deps/ffmpeg/8.1/0007-avdevice-ndi-v5.patch rename to deps/ffmpeg/8/0007-avdevice-ndi-v5.patch index 9daeb08e..8a0191f9 100644 --- a/deps/ffmpeg/8.1/0007-avdevice-ndi-v5.patch +++ b/deps/ffmpeg/8/0007-avdevice-ndi-v5.patch @@ -1,7 +1,7 @@ -From 8b939f7ee621f63ac2ecc6175e0c9644eca7861d Mon Sep 17 00:00:00 2001 -From: Jan Pietek -Date: Tue, 15 Sep 2026 09:38:18 +0200 -Subject: [PATCH 7/7] avdevice/ndi: rebase optional NDI v5 registration on 8.1 +From 7fa87c315b2896edc23547c7dc867c3516524331 Mon Sep 17 00:00:00 2001 +From: x +Date: Wed, 16 Sep 2026 19:56:02 +0200 +Subject: [PATCH] 0007-avdevice-ndi-v5.patch --- configure | 7 +++++ @@ -12,51 +12,33 @@ Subject: [PATCH 7/7] avdevice/ndi: rebase optional NDI v5 registration on 8.1 5 files changed, 123 insertions(+) diff --git a/configure b/configure -index 584e1df313..f086594121 100755 +index 584e1df..f086594 100755 --- a/configure +++ b/configure -@@ -323,6 +323,7 @@ External library support: - --enable-lv2 enable LV2 audio filtering [no] - --disable-lzma disable lzma [autodetect] +@@ -325,2 +325,3 @@ External library support: --enable-decklink enable Blackmagic DeckLink I/O support [no] + --enable-libndi_newtek enable Newteck NDI I/O support [no] --enable-mbedtls enable mbedTLS, needed for https support - if openssl, gnutls or libtls is not used [no] - --enable-mediacodec enable Android MediaCodec support [no] -@@ -2003,6 +2004,7 @@ EXTERNAL_LIBRARY_GPL_LIST=" - - EXTERNAL_LIBRARY_NONFREE_LIST=" +@@ -2005,2 +2006,3 @@ EXTERNAL_LIBRARY_NONFREE_LIST=" decklink + libndi_newtek libfdk_aac - libmpeghdec - " -@@ -3983,6 +3985,10 @@ decklink_indev_suggest="libzvbi" - decklink_outdev_deps="decklink threads" - decklink_outdev_suggest="libklvanc" +@@ -3985,2 +3987,6 @@ decklink_outdev_suggest="libklvanc" decklink_outdev_extralibs="-lstdc++" +libndi_newtek_indev_deps="libndi_newtek" +libndi_newtek_indev_extralibs="-lndi" +libndi_newtek_outdev_deps="libndi_newtek" +libndi_newtek_outdev_extralibs="-lndi" dshow_indev_deps="IBaseFilter" - dshow_indev_extralibs="-lpsapi -lole32 -lstrmiids -luuid -loleaut32 -lshlwapi" - fbdev_indev_deps="linux_fb_h" -@@ -7233,6 +7239,7 @@ enabled chromaprint && { check_pkg_config chromaprint libchromaprint "chro - require chromaprint chromaprint.h chromaprint_get_version -lchromaprint; } - enabled decklink && { require_headers DeckLinkAPI.h && +@@ -7235,2 +7241,3 @@ enabled decklink && { require_headers DeckLinkAPI.h && { test_cpp_condition DeckLinkAPIVersion.h "BLACKMAGIC_DECKLINK_API_VERSION >= 0x0a0b0000" || die "ERROR: Decklink API version must be >= 10.11"; } } +enabled libndi_newtek && require libndi_newtek Processing.NDI.Lib.h NDIlib_initialize -lndi enabled frei0r && require_headers "frei0r.h" - enabled gmp && require gmp gmp.h mpz_export -lgmp - enabled gnutls && require_pkg_config gnutls gnutls gnutls/gnutls.h gnutls_global_init diff --git a/doc/indevs.texi b/doc/indevs.texi -index 8822e070fe..5ef5d4c208 100644 +index 8822e07..5ef5d4c 100644 --- a/doc/indevs.texi +++ b/doc/indevs.texi -@@ -1121,6 +1121,71 @@ Set the video size given as a string such as @code{640x480} or @code{hd720}. - Default is @code{qvga}. - @end table +@@ -1123,2 +1123,67 @@ Default is @code{qvga}. +@section libndi_newtek + @@ -124,15 +106,11 @@ index 8822e070fe..5ef5d4c208 100644 +@end itemize + @section openal - - The OpenAL input device provides audio capture on all systems with a diff --git a/doc/outdevs.texi b/doc/outdevs.texi -index 86c78f31b7..62d342f2c2 100644 +index 86c78f3..62d342f 100644 --- a/doc/outdevs.texi +++ b/doc/outdevs.texi -@@ -301,6 +301,51 @@ ffmpeg -re -i INPUT -c:v rawvideo -pix_fmt bgra -f fbdev /dev/fb0 - - See also @url{http://linux-fbdev.sourceforge.net/}, and fbset(1). +@@ -303,2 +303,47 @@ See also @url{http://linux-fbdev.sourceforge.net/}, and fbset(1). +@section libndi_newtek + @@ -180,40 +158,29 @@ index 86c78f31b7..62d342f2c2 100644 +@end itemize + @section oss - - OSS (Open Sound System) output device. diff --git a/libavdevice/Makefile b/libavdevice/Makefile -index a226368d16..272c80deed 100644 +index a226368..272c80d 100644 --- a/libavdevice/Makefile +++ b/libavdevice/Makefile -@@ -21,6 +21,8 @@ OBJS-$(CONFIG_AVFOUNDATION_INDEV) += avfoundation.o - OBJS-$(CONFIG_CACA_OUTDEV) += caca.o - OBJS-$(CONFIG_DECKLINK_OUTDEV) += decklink_enc.o decklink_enc_c.o decklink_common.o +@@ -23,2 +23,4 @@ OBJS-$(CONFIG_DECKLINK_OUTDEV) += decklink_enc.o decklink_enc_c.o deck OBJS-$(CONFIG_DECKLINK_INDEV) += decklink_dec.o decklink_dec_c.o decklink_common.o +OBJS-$(CONFIG_LIBNDI_NEWTEK_OUTDEV) += libndi_newtek_enc.o +OBJS-$(CONFIG_LIBNDI_NEWTEK_INDEV) += libndi_newtek_dec.o OBJS-$(CONFIG_DSHOW_INDEV) += dshow_crossbar.o dshow.o dshow_enummediatypes.o \ - dshow_enumpins.o dshow_filter.o \ - dshow_pin.o dshow_common.o -@@ -62,6 +64,8 @@ SHLIBOBJS-$(HAVE_GNU_WINDRES) += avdeviceres.o - SKIPHEADERS += decklink_common.h - SKIPHEADERS-$(CONFIG_DECKLINK) += decklink_enc.h decklink_dec.h \ +@@ -64,2 +66,4 @@ SKIPHEADERS-$(CONFIG_DECKLINK) += decklink_enc.h decklink_dec.h \ decklink_common_c.h +SKIPHEADERS-$(CONFIG_LIBNDI_NEWTEK_INDEV) += libndi_newtek_common.h +SKIPHEADERS-$(CONFIG_LIBNDI_NEWTEK_OUTDEV) += libndi_newtek_common.h SKIPHEADERS-$(CONFIG_DSHOW_INDEV) += dshow_capture.h - SKIPHEADERS-$(CONFIG_FBDEV_INDEV) += fbdev_common.h - SKIPHEADERS-$(CONFIG_FBDEV_OUTDEV) += fbdev_common.h diff --git a/libavdevice/alldevices.c b/libavdevice/alldevices.c -index 573595f416..3b516e6d3e 100644 +index 573595f..3b516e6 100644 --- a/libavdevice/alldevices.c +++ b/libavdevice/alldevices.c -@@ -35,6 +35,8 @@ extern const FFInputFormat ff_avfoundation_demuxer; - extern const FFOutputFormat ff_caca_muxer; - extern const FFInputFormat ff_decklink_demuxer; +@@ -37,2 +37,4 @@ extern const FFInputFormat ff_decklink_demuxer; extern const FFOutputFormat ff_decklink_muxer; +extern const FFInputFormat ff_libndi_newtek_demuxer; +extern const FFOutputFormat ff_libndi_newtek_muxer; extern const FFInputFormat ff_dshow_demuxer; - extern const FFInputFormat ff_fbdev_demuxer; - extern const FFOutputFormat ff_fbdev_muxer; +-- +2.55.0 + diff --git a/deps/ffmpeg/8/bases.env b/deps/ffmpeg/8/bases.env new file mode 100644 index 00000000..ea60d972 --- /dev/null +++ b/deps/ffmpeg/8/bases.env @@ -0,0 +1,5 @@ +n80_commit=140fd653aed8cad774f991ba083e2d01e86420c7 +n80_tree=6ff5e4f32a3acb868c6fd4a2ae9e16c25115a6f2 +n81_commit=9047fa1b084f76b1b4d065af2d743df1b40dfb56 +n81_tree=3f4f8ba98c493fe9c0e0bfae5a61cd8b83fc30ca +patch_count=7 diff --git a/deps/ffmpeg/README.md b/deps/ffmpeg/README.md index 33477759..9ec7429f 100644 --- a/deps/ffmpeg/README.md +++ b/deps/ffmpeg/README.md @@ -1,37 +1,67 @@ -# FFmpeg patch series +# FFmpeg patch series (FFmpeg 8.x) -Each version directory contains a complete series for one exact upstream tag. -Do not apply the 7.1.5 series and then the 8.1 series to the same checkout. - -| Directory | Upstream tag | Purpose | -| --- | --- | --- | -| `7.1.5/` | `n7.1.5` | Existing default; patch contents preserved unchanged. | -| `8.1/` | `n8.1` | Compatibility port with 8-bit CUDA mixer runtime checks; see its README for coverage. | - -The mixer, CUDA-overlay and DMA-BUF CUDA consumer Dockerfiles select the series -using their existing `FFMPEG_TAG` argument. Their default remains `n7.1.5`. -An isolated 8.1 mixer build can be requested from the repository root with: +One ordered series in `8/` applies to upstream **n8.0 and n8.1** from a single +copy; there are no per-version directories. `8/bases.env` pins each base +commit and the exact patched tree; `patch_count` guards against stray files. ```bash -docker build --build-arg FFMPEG_TAG=n8.1 \ - -f demos/mixer/Dockerfile -t avplumber-mixer:ffmpeg8.1 . +deps/ffmpeg/apply.sh # git am 8/*.patch + configure fix-up +deps/ffmpeg/verify.sh # isolated worktree, whole series, tree check ``` -Build on an NVIDIA development host, with the required submodules populated. -This command creates a separate image; it does not replace a running container. -avcpp and avplumber must be rebuilt against the selected FFmpeg libraries. -Keep the current avcpp revision unless a verified compatibility issue requires -a change. The mixer graph remains 8-bit NV12; this port does not add MXL or -10-bit composition. +The demo Dockerfiles (`demos/mixer`, `demos/cuda-overlay`, +`demos/dmabuf-browser/consumer`) call `apply.sh`; `FFMPEG_TAG` defaults to +`n8.1` and may be set to `n8.0`. avcpp and avplumber must be rebuilt against +the selected FFmpeg libraries. -Each version's `base.env` pins the upstream commit, patched tree and patch count. -Verify either series without changing the source checkout's branch: +## How one series serves both bases -```bash -deps/ffmpeg/7.1.5/verify.sh -deps/ffmpeg/8.1/verify.sh -``` +The composition-suite filters are new files, so they carry no base-tree +context; the remaining patches are generated with minimal (`-U1`) context so +they apply where 8.0 and 8.1 differ only cosmetically. The single genuine +8.0/8.1 difference is a libnpp `configure` check that 8.1 added and that fails +on CUDA 13 (the legacy `nppiYCbCr420_8u_P2P3R` symbol is gone). `apply.sh` +rewrites that check to probe the stream-context API; on 8.0, which has no such +check, the rewrite is a no-op. Everything else in the NPP patch (`npp_compat.h` +and the filter changes) is base-independent. + +## The series + +1. **swscale aarch64 ARGB→YUVA420P** — uses `SwsInternal`/`opts` and the + unscaled callback signature; algorithms unchanged. +2. **CUDA composition suite** — `convert_cuda`, `crop_cuda`, `overlay_many_cuda`, + `pad_cuda` (intentionally replaces upstream's), `transition_cuda`, plus the + scaling-edge/overlay-context fixes and YUVA444P CUDA frames. Upstream 8.x + `scale_cuda` (with its expanded pixel formats) is kept. +3. **NPP CUDA 13 compatibility** — stream-context helper and filter changes. +4. **NVDEC intra-only handling.** +5. **RFC 4175 RTP frame handling.** +6. **V4L2 source timestamps.** +7. **NDI v5 registration** — registers the optional integration only; no SDK + or device implementation is supplied, and NDI stays disabled in demo builds. + +## FFmpeg 8 notes + +- `AVFrame.pkt_pos` is gone: `transition_cuda`'s legacy `pos` expression + variable evaluates to `NAN`; `crop_cuda` guards it by API version. +- The `C` command-support marker left `-filters` output; the mixer Dockerfile + checks the transition's `mode` option in filter help instead. +- FFmpeg 8 validates CUDA input formats at filter init, so the AVP filter node + attaches `hw_frames_ctx` / `hw_device_ctx` between allocation and + initialisation (segmented graph parser); the metadata-driven crop node does + the same. +- When reusing a build tree with different GL flags, rebuild + `deps/cuda_loader/cuda_drvapi_dynlink.o`; a loader built without GL lacks the + EGL function-pointer variables, and linking `libcuda` directly to paper over + that crashes during DMA-BUF import. + +## Validation -The supplied checkout must contain the corresponding upstream commit. The -shared verifier uses an isolated worktree and checks the entire ordered series, -not just individual patches. +Compilation, linking and filter registration are the acceptance criteria for +the series itself; runtime behaviour is covered by the demo and `tests/cuda` +suites. Validated on a T4 with FFmpeg 8.1: the 8-bit mixer at 30/60 fps with +video and DMA-BUF browser inputs, wipe-cache loads, cut/fade/wipe, NVENC to a +WebRTC browser; the `demos/cuda-overlay` 45-case pixel-reference matrix; and +the live recorder surviving SRT disconnects (`ignore_eof` on the pre-sentinel +format nodes, opt-in). NPP, NDI and AArch64 paths are not compiled in the demo +images. diff --git a/deps/ffmpeg/apply.sh b/deps/ffmpeg/apply.sh new file mode 100755 index 00000000..0959298a --- /dev/null +++ b/deps/ffmpeg/apply.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# Apply the FFmpeg 8.x patch series to a checkout of upstream n8.0 or n8.1. +set -euo pipefail +[[ $# -eq 1 ]] || { echo "usage: $0 /path/to/FFmpeg" >&2; exit 2; } +dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) +src=$1 +git -C "$src" -c user.name="avplumber patches" -c user.email="patches@local" \ + am --whitespace=nowarn "$dir"/8/*.patch +# FFmpeg 8.1 added a libnpp configure check that fails on CUDA 13 (the legacy +# nppiYCbCr420_8u_P2P3R symbol is gone); probe the stream-context API instead. +# 8.0 has no such check, so this is a no-op there. +sed -i.bak \ + -e 's/check_func_headers "nppi.h" nppiYCbCr420_8u_P2P3R \$libnpp_extralibs/check_func_headers "nppi.h" nppiYCbCr420_8u_P2P3R_Ctx $libnpp_extralibs/' \ + -e 's/libnpp support is deprecated, version 13.0 and up are not supported/libnpp stream context APIs not found/' \ + "$src/configure" && rm -f "$src/configure.bak" +if ! git -C "$src" diff --quiet -- configure; then + git -C "$src" -c user.name="avplumber patches" -c user.email="patches@local" \ + commit -qam "configure: probe the libnpp stream-context API (CUDA 13)" +fi diff --git a/deps/ffmpeg/verify.sh b/deps/ffmpeg/verify.sh index 7d63c0b1..252b1d01 100755 --- a/deps/ffmpeg/verify.sh +++ b/deps/ffmpeg/verify.sh @@ -1,52 +1,19 @@ #!/usr/bin/env bash +# Verify the series reproduces the pinned tree for an upstream base (n8.0 or n8.1). set -euo pipefail - -if [[ $# -ne 2 ]]; then - echo "usage: $0 <7.1.5|8.1> /path/to/FFmpeg" >&2 - exit 2 -fi - -script_dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) -case "$1" in - 7.1.5|8.1) series_dir="$script_dir/$1" ;; - *) echo "unsupported FFmpeg series: $1" >&2; exit 2 ;; -esac -source "$series_dir/base.env" -source_repo=$(git -C "$2" rev-parse --show-toplevel) -if ! git -C "$source_repo" cat-file -e "${base_commit}^{commit}"; then - echo "FFmpeg checkout does not contain base commit ${base_commit}" >&2 - exit 2 -fi - -shopt -s nullglob -patches=("$series_dir"/*.patch) -if [[ ${#patches[@]} -ne $expected_patch_count ]]; then - echo "expected ${expected_patch_count} patches, found ${#patches[@]}" >&2 - exit 1 -fi - -audit_root=$(mktemp -d "${TMPDIR:-/tmp}/avplumber-ffmpeg-verify.XXXXXX") -audit_worktree="$audit_root/ffmpeg" - -cleanup() { - if [[ -d "$audit_worktree" ]]; then - git -C "$source_repo" worktree remove --force "$audit_worktree" \ - >/dev/null 2>&1 || true - fi - rmdir "$audit_root" >/dev/null 2>&1 || true -} -trap cleanup EXIT - -git -C "$source_repo" worktree add --detach "$audit_worktree" "$base_commit" \ - >/dev/null -git -C "$audit_worktree" -c user.name="avplumber patch verifier" \ - -c user.email="patch-verifier@local" am --whitespace=nowarn "${patches[@]}" >/dev/null - -actual_tree=$(git -C "$audit_worktree" rev-parse 'HEAD^{tree}') -if [[ "$actual_tree" != "$expected_tree" ]]; then - echo "unexpected patched tree: ${actual_tree}" >&2 - echo "expected patched tree: ${expected_tree}" >&2 - exit 1 -fi - -echo "FFmpeg patch stack verified: ${actual_tree}" +[[ $# -eq 2 ]] || { echo "usage: $0 /path/to/FFmpeg" >&2; exit 2; } +dir=$(CDPATH= cd -- "$(dirname -- "$0")" && pwd) +source "$dir/8/bases.env" +case "$1" in n8.0) k=n80 ;; n8.1) k=n81 ;; *) echo "unsupported base: $1" >&2; exit 2 ;; esac +v="${k}_commit"; base_commit=${!v}; v="${k}_tree"; expected_tree=${!v} +repo=$(git -C "$2" rev-parse --show-toplevel) +git -C "$repo" cat-file -e "${base_commit}^{commit}" || { echo "checkout lacks $1 ($base_commit)" >&2; exit 2; } +shopt -s nullglob; patches=("$dir"/8/*.patch) +[[ ${#patches[@]} -eq $patch_count ]] || { echo "expected $patch_count patches, found ${#patches[@]}" >&2; exit 1; } +root=$(mktemp -d "${TMPDIR:-/tmp}/avplumber-ffmpeg-verify.XXXXXX"); wt="$root/ffmpeg" +trap 'git -C "$repo" worktree remove --force "$wt" >/dev/null 2>&1 || true; rmdir "$root" 2>/dev/null || true' EXIT +git -C "$repo" worktree add --detach "$wt" "$base_commit" >/dev/null +"$dir/apply.sh" "$wt" >/dev/null +actual=$(git -C "$wt" rev-parse 'HEAD^{tree}') +[[ "$actual" == "$expected_tree" ]] || { echo "unexpected patched tree for $1: $actual (expected $expected_tree)" >&2; exit 1; } +echo "FFmpeg $1 patch series verified: $actual" From 5cd2d568632d522759af7f43eb688d3a23d2f529 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Wed, 16 Sep 2026 20:36:03 +0200 Subject: [PATCH 11/11] demos: finish migration to the shared FFmpeg 8.x series --- demos/cuda-overlay/compose.yaml | 2 +- demos/cuda-overlay/docs/guide.md | 3 ++- demos/cuda-overlay/tests/test_run_matrix.py | 20 ++++++++------------ demos/cuda-overlay/tools/run_matrix.py | 8 ++++---- demos/mixer/docs/guide.md | 2 +- 5 files changed, 16 insertions(+), 19 deletions(-) diff --git a/demos/cuda-overlay/compose.yaml b/demos/cuda-overlay/compose.yaml index 535a292b..518cb3dc 100644 --- a/demos/cuda-overlay/compose.yaml +++ b/demos/cuda-overlay/compose.yaml @@ -5,7 +5,7 @@ services: dockerfile: demos/cuda-overlay/Dockerfile args: AVPLUMBER_REVISION: ${AVPLUMBER_REVISION:-workspace} - FFMPEG_TAG: n7.1.5 + FFMPEG_TAG: ${FFMPEG_TAG:-n8.1} image: avplumber-cuda-overlay-demo:local gpus: all environment: diff --git a/demos/cuda-overlay/docs/guide.md b/demos/cuda-overlay/docs/guide.md index 8dde2203..e38c450d 100644 --- a/demos/cuda-overlay/docs/guide.md +++ b/demos/cuda-overlay/docs/guide.md @@ -5,7 +5,8 @@ One 1080p base and 15 transparent, labeled overlays composed on the GPU by `overlay_many_cuda`. -Build the repository's FFmpeg patch stack on public FFmpeg `n7.1.5`, run the +Build the repository's shared FFmpeg 8.x patch stack on public FFmpeg `n8.1` +(or select `n8.0` with `FFMPEG_TAG`), run the patched `overlay_many_cuda` through a purpose-built PyPlumber graph, and compare every output plane against an independent CPU reference. diff --git a/demos/cuda-overlay/tests/test_run_matrix.py b/demos/cuda-overlay/tests/test_run_matrix.py index f998ba99..a9f50b64 100644 --- a/demos/cuda-overlay/tests/test_run_matrix.py +++ b/demos/cuda-overlay/tests/test_run_matrix.py @@ -17,24 +17,20 @@ class RunMatrixTest(unittest.TestCase): def test_patch_identity_reads_the_repository_patch_series(self) -> None: expected_paths = sorted( - (REPOSITORY_DIR / "deps" / "ffmpeg" / "7.1.5").glob("*.patch") + (REPOSITORY_DIR / "deps" / "ffmpeg" / "8").glob("*.patch") ) self.assertEqual(REPO_DIR, REPOSITORY_DIR) - self.assertEqual(len(expected_paths), 7) + self.assertGreater(len(expected_paths), 0) self.assertEqual(list(_patch_identity()), [path.name for path in expected_paths]) - def test_patch_identity_selects_ffmpeg81(self) -> None: - expected_paths = sorted( - (REPOSITORY_DIR / "deps" / "ffmpeg" / "8.1").glob("*.patch") - ) - self.assertEqual(len(expected_paths), 7) - self.assertEqual(list(_patch_identity("n8.1")), [path.name for path in expected_paths]) - self.assertNotEqual(_patch_identity(), _patch_identity("n8.1")) + def test_patch_identity_shares_series_between_ffmpeg80_and_81(self) -> None: + self.assertEqual(_patch_identity(), _patch_identity("n8.1")) + self.assertEqual(_patch_identity("n8.0"), _patch_identity("n8.1")) def test_patch_identity_rejects_unknown_tags(self) -> None: with self.assertRaises(ValueError): - _patch_identity("n8.0") + _patch_identity("n7.1.5") def test_cuda_images_keep_the_default_and_select_matching_patches(self) -> None: for relative_path in ( @@ -44,9 +40,9 @@ def test_cuda_images_keep_the_default_and_select_matching_patches(self) -> None: ): with self.subTest(dockerfile=relative_path): dockerfile = (REPOSITORY_DIR / relative_path).read_text() - self.assertIn("ARG FFMPEG_TAG=n7.1.5", dockerfile) + self.assertIn("ARG FFMPEG_TAG=n8.1", dockerfile) self.assertIn("COPY deps/ffmpeg /build/deps/ffmpeg", dockerfile) - self.assertIn("/build/deps/ffmpeg/${FFMPEG_TAG#n}/*.patch", dockerfile) + self.assertIn("/build/deps/ffmpeg/apply.sh /tmp/ffmpeg", dockerfile) def test_overlay_image_reports_the_selected_version(self) -> None: dockerfile = (REPOSITORY_DIR / "demos/cuda-overlay/Dockerfile").read_text() diff --git a/demos/cuda-overlay/tools/run_matrix.py b/demos/cuda-overlay/tools/run_matrix.py index aac728d2..a4118bf3 100755 --- a/demos/cuda-overlay/tools/run_matrix.py +++ b/demos/cuda-overlay/tools/run_matrix.py @@ -33,11 +33,11 @@ def _command_text(command: list[str]) -> str: return output if output else f"exit {result.returncode}" -def _patch_identity(ffmpeg_tag: str = "n7.1.5") -> dict[str, str]: - if ffmpeg_tag not in ("n7.1.5", "n8.1"): +def _patch_identity(ffmpeg_tag: str = "n8.1") -> dict[str, str]: + if ffmpeg_tag not in ("n8.0", "n8.1"): raise ValueError(f"unsupported FFmpeg tag: {ffmpeg_tag}") identities: dict[str, str] = {} - for path in sorted((REPO_DIR / "deps" / "ffmpeg" / ffmpeg_tag[1:]).glob("*.patch")): + for path in sorted((REPO_DIR / "deps" / "ffmpeg" / "8").glob("*.patch")): identities[path.name] = hashlib.sha256(path.read_bytes()).hexdigest() return identities @@ -80,7 +80,7 @@ def main() -> int: parser.add_argument("--height", type=_positive_dimension, default=HEIGHT) args = parser.parse_args() - ffmpeg_tag = os.environ.get("FFMPEG_TAG", "n7.1.5") + ffmpeg_tag = os.environ.get("FFMPEG_TAG", "n8.1") started_at = dt.datetime.now(dt.timezone.utc) run_id = started_at.strftime("run-%Y%m%dT%H%M%SZ") diff --git a/demos/mixer/docs/guide.md b/demos/mixer/docs/guide.md index 55bdc5e8..ad89fa60 100644 --- a/demos/mixer/docs/guide.md +++ b/demos/mixer/docs/guide.md @@ -195,7 +195,7 @@ groups without reordering sources. Follow the [shared NVIDIA setup guide](../../README.md) first. It also provides a local Janus preview if you want WebRTC output. -The demo image defaults to FFmpeg 7.1.5 with `deps/ffmpeg/7.1.5`, verifies the +The demo image defaults to FFmpeg 8.1 with the shared `deps/ffmpeg/8` series, verifies the patched CUDA overlay and transition filters, and builds the CUDA-enabled AVPlumber Python module against that FFmpeg installation: