From 2e6223bace0bc4e074a514c26069ed41c909a322 Mon Sep 17 00:00:00 2001 From: Jan Pietek Date: Thu, 17 Sep 2026 13:55:51 +0200 Subject: [PATCH] tonemap_cuda: native 4:2:2 (NV16/P210) in and out, chroma resampled in the same pass The kernel handled one 2x2 luma block with a single shared chroma sample, i.e. 4:2:0 only, so every conversion touching a P210 canvas needed a scale_cuda pass to P010 before and one back to P210 after. It now takes NV12/P010/NV16/P210 and writes any of them: 4:2:2 input keeps one chroma sample per luma row, 4:2:0 output averages the block, 4:2:2 output keeps the rows. The default output keeps the input subsampling (8 bits for SDR, 10 for HDR); `format=` selects it explicitly. Passthrough requires matching subsampling as well. conversion_graph drops the P010 sandwich: converted sources on a P210 canvas stay 4:2:2, and a rendition off a P210 canvas subsamples once, inside the tone mapper, instead of scale, tonemap, scale. That is one fewer 156 us pass per converted source and per rendition on the T4. Planar CUDA storage is still re-laid out by scale_cuda first. Series re-pinned on n8.0 and n8.1. Co-Authored-By: Claude Fable 5.1 --- avpmixer/color.py | 17 ++- demos/mixer/docs/config.md | 2 +- demos/mixer/tests/test_graph.py | 2 +- .../ffmpeg/8/0009-avfilter-tonemap-cuda.patch | 123 +++++++++++------- deps/ffmpeg/8/bases.env | 4 +- deps/ffmpeg/README.md | 5 +- tests/test_mixer_color.py | 13 +- 7 files changed, 102 insertions(+), 64 deletions(-) diff --git a/avpmixer/color.py b/avpmixer/color.py index 6b7d2603..b2cbae00 100644 --- a/avpmixer/color.py +++ b/avpmixer/color.py @@ -12,7 +12,8 @@ OPERATORS = ("none", "linear", "gamma", "clip", "reinhard", "hable", "mobius") COLOR_KEYS = ("color_trc", "color_primaries", "colorspace", "color_range") TEN_BIT_FORMATS = ("p010le", "p210le", "yuv420p10le", "yuv422p10le", "yuv444p10le") -YUV_FORMATS = ("nv12", "yuv420p", "yuv422p", "yuv444p", *TEN_BIT_FORMATS) +SEMIPLANAR_FORMATS = ("nv12", "nv16", "p010le", "p210le") # what tonemap_cuda reads and writes +YUV_FORMATS = ("nv12", "nv16", "yuv420p", "yuv422p", "yuv444p", *TEN_BIT_FORMATS) @dataclass(frozen=True) @@ -36,7 +37,7 @@ def setparams(self): for k, v in self.tags.items()) def validate_format(self, pixel_format): - if pixel_format not in YUV_FORMATS: + if pixel_format not in SEMIPLANAR_FORMATS: raise ValueError(f"unsupported canvas/output pixel format {pixel_format!r}") if self.transfer != "sdr" and pixel_format not in TEN_BIT_FORMATS: raise ValueError("HLG/PQ canvas and outputs require a 10-bit pixel format") @@ -96,18 +97,16 @@ def conversion_graph(target, pixel_format, *, source=None, source_format=None, parts = [target.setparams] return ",".join(parts if source_format == pixel_format else parts + [f"scale_cuda=format={pixel_format}"]) parts = [Color.parse(source).setparams] if source is not None else [] - if source_format and source_format not in ("nv12", "p010le"): - # Preserve precision before a potential HDR conversion, regardless of target depth. - parts.append("scale_cuda=format=p010le") - intermediate = "p010le" if pixel_format in TEN_BIT_FORMATS else "nv12" + if source_format and source_format not in SEMIPLANAR_FORMATS: + # tonemap_cuda works on semiplanar storage; planar sources are re-laid out at 10 bits. + parts.append("scale_cuda=format=p210le" if "422" in source_format else "scale_cuda=format=p010le") + # tonemap_cuda converts colour and storage in one pass, 4:2:0 or 4:2:2 in and out. # param is the operator knee in reference-white units (mobius/reinhard; 0 keeps the # filter default 0.3). mobius at 0.9 keeps 0..90% of SDR white linear and folds # everything brighter into the top 10% of the SDR range; 1.0 would be a plain clip. - parts.append(f"tonemap_cuda=transfer_in=auto:transfer_out={target.transfer}:format={intermediate}" + parts.append(f"tonemap_cuda=transfer_in=auto:transfer_out={target.transfer}:format={pixel_format}" f":tonemap={tonemap}:sdr_white={sdr_white:g}:hdr_peak={hdr_peak:g}:desat={desat:g}" + (f":param={param:g}" if param else "")) - if pixel_format != intermediate: - parts.append(f"scale_cuda=format={pixel_format}") return ",".join(parts) diff --git a/demos/mixer/docs/config.md b/demos/mixer/docs/config.md index d2384332..18cb492d 100644 --- a/demos/mixer/docs/config.md +++ b/demos/mixer/docs/config.md @@ -41,7 +41,7 @@ supplied by that file; the runtime does not depend on the recorded demo's inputs | --- | --- | --- | | `width`, `height` | — | the program raster the compositor draws into | | `fps` | `30` | **how often the compositor renders**, and the clock the whole mixer runs on: inputs are re-timed to it and browser pages are asked to paint at it | -| `working_format` | `nv12` | compositor and transition pixel storage: `nv12` (8-bit 4:2:0), `p010le` (10-bit 4:2:0) or `p210le` (10-bit 4:2:2). 8-bit sources are promoted onto a 10-bit canvas; `p210le` keeps 4:2:2 content (`v210` sources) native, renditions subsample once for NVENC | +| `working_format` | `nv12` | compositor and transition pixel storage: `nv12` (8-bit 4:2:0), `p010le` (10-bit 4:2:0) or `p210le` (10-bit 4:2:2). 8-bit sources are promoted onto a 10-bit canvas; `p210le` keeps 4:2:2 through colour conversion and compositing, and renditions subsample to 4:2:0 once, inside the tone mapper, for NVENC | | `color` | `sdr` | canvas colour contract: `sdr` (BT.709), `hlg` or `pq` (BT.2020). HLG/PQ need a 10-bit `working_format`. Every source is converted to it on the GPU; renditions convert from it | The canvas rate is the single biggest load knob. Halving it from 60 to 30 on diff --git a/demos/mixer/tests/test_graph.py b/demos/mixer/tests/test_graph.py index f0691c8f..e1b7e71f 100644 --- a/demos/mixer/tests/test_graph.py +++ b/demos/mixer/tests/test_graph.py @@ -782,7 +782,7 @@ def test_v210_sources_keep_422_through_a_p210_canvas(tmp_path): # HDR out: one chroma subsample to P010 for NVENC, no tone-map pass. assert nodes["scale_hdr"]["graph"] == Color("hlg").setparams + ",scale_cuda=format=p010le" assert nodes["janus_format"]["real_pixel_format"] == "p010le" - assert "tonemap_cuda=transfer_in=auto:transfer_out=sdr:format=nv12" in nodes["scale_sdr"]["graph"] + assert nodes["scale_sdr"]["graph"].startswith(Color("hlg").setparams + ",tonemap_cuda=transfer_in=auto:transfer_out=sdr:format=nv12") def test_cli_inputs_declare_browser_rgb_and_optional_file_color(tmp_path): diff --git a/deps/ffmpeg/8/0009-avfilter-tonemap-cuda.patch b/deps/ffmpeg/8/0009-avfilter-tonemap-cuda.patch index 943c19a1..67e9584b 100644 --- a/deps/ffmpeg/8/0009-avfilter-tonemap-cuda.patch +++ b/deps/ffmpeg/8/0009-avfilter-tonemap-cuda.patch @@ -1,7 +1,7 @@ -From 550c76492522f58c3c7ea1decd17c9e513274449 Mon Sep 17 00:00:00 2001 +From 9a3dd6de0ed5dd3cb65b27aee4bf774ce0aa881c Mon Sep 17 00:00:00 2001 From: avplumber patches Date: Thu, 17 Sep 2026 13:16:00 +0200 -Subject: [PATCH 9/9] avfilter: add tonemap_cuda, SDR/HLG/PQ conversion on CUDA +Subject: [PATCH] avfilter: add tonemap_cuda, SDR/HLG/PQ conversion on CUDA NV12/P010 frames A CUDA filter converting limited-range BT.709 SDR and BT.2020 HLG/PQ in @@ -17,12 +17,12 @@ dependent side data is dropped on conversion. The math is ported from libavfilter/opencl/tonemap.cl and colorspace_common.cl. --- configure | 1 + - doc/filters.texi | 58 ++++ + doc/filters.texi | 61 ++++ libavfilter/Makefile | 2 + libavfilter/allfilters.c | 1 + - libavfilter/vf_tonemap_cuda.c | 529 +++++++++++++++++++++++++++++++++ - libavfilter/vf_tonemap_cuda.cu | 285 ++++++++++++++++++ - 6 files changed, 876 insertions(+) + libavfilter/vf_tonemap_cuda.c | 542 +++++++++++++++++++++++++++++++++ + libavfilter/vf_tonemap_cuda.cu | 304 ++++++++++++++++++ + 6 files changed, 911 insertions(+) create mode 100644 libavfilter/vf_tonemap_cuda.c create mode 100644 libavfilter/vf_tonemap_cuda.cu @@ -39,19 +39,22 @@ index f086594..82cdbd2 100755 overlay_cuda_filter_deps="ffnvcodec" overlay_cuda_filter_deps_any="cuda_nvcc cuda_llvm" diff --git a/doc/filters.texi b/doc/filters.texi -index 5ae2fc5..8e4e6b1 100644 +index 5ae2fc5..82baec9 100644 --- a/doc/filters.texi +++ b/doc/filters.texi -@@ -27409,6 +27409,64 @@ scale_cuda=passthrough=0 +@@ -27409,6 +27409,67 @@ scale_cuda=passthrough=0 @end example @end itemize +@subsection tonemap_cuda + -+Convert limited-range CUDA NV12/P010 frames between SDR BT.709, HLG BT.2020 -+and PQ BT.2020. Width and height must be even. HDR output uses P010; SDR -+output uses NV12. Equal input and output transfers forward the original -+frames and metadata without allocating or converting pixels. ++Convert limited-range CUDA semiplanar frames (NV12/P010 4:2:0, NV16/P210 ++4:2:2) between SDR BT.709, HLG BT.2020 and PQ BT.2020. Width and height must ++be even. By default the output keeps the input chroma subsampling, 8 bits for ++SDR and 10 bits for HDR; @option{format} selects @code{nv12}, @code{p010le}, ++@code{nv16} or @code{p210le} explicitly and resamples 4:2:2 and 4:2:0 in the ++same pass. Equal input and output transfers, depth and subsampling forward ++the original frames and metadata without allocating or converting pixels. + +@table @option +@item transfer_in @@ -134,10 +137,10 @@ index 01c75b2..64b5cd7 100644 extern const FFFilter ff_vf_tpad; diff --git a/libavfilter/vf_tonemap_cuda.c b/libavfilter/vf_tonemap_cuda.c new file mode 100644 -index 0000000..3591127 +index 0000000..2c270f8 --- /dev/null +++ b/libavfilter/vf_tonemap_cuda.c -@@ -0,0 +1,529 @@ +@@ -0,0 +1,542 @@ +/* + * This file is part of FFmpeg. + * @@ -158,7 +161,8 @@ index 0000000..3591127 + +/** + * @file -+ * SDR BT.709 / HDR BT.2020 HLG and PQ conversion on CUDA 4:2:0 frames. ++ * SDR BT.709 / HDR BT.2020 HLG and PQ conversion on CUDA semiplanar frames ++ * (NV12/P010 4:2:0, NV16/P210 4:2:2), with chroma resampling between them. + */ + +#include @@ -237,6 +241,8 @@ index 0000000..3591127 + int output_format; + int input_depth; + int output_depth; ++ int input_422; ++ int output_422; + double sdr_white; + double hdr_peak; + double param; @@ -296,7 +302,8 @@ index 0000000..3591127 + out_ctx = (AVHWFramesContext*)out_ref->data; + + out_ctx->format = AV_PIX_FMT_CUDA; -+ out_ctx->sw_format = s->output_depth == 10 ? AV_PIX_FMT_P010 : AV_PIX_FMT_NV12; ++ out_ctx->sw_format = s->output_depth == 10 ? (s->output_422 ? AV_PIX_FMT_P210 : AV_PIX_FMT_P010) ++ : (s->output_422 ? AV_PIX_FMT_NV16 : AV_PIX_FMT_NV12); + out_ctx->width = FFALIGN(width, 32); + out_ctx->height = FFALIGN(height, 32); + @@ -367,8 +374,9 @@ index 0000000..3591127 + + in_frames_ctx = (AVHWFramesContext*)inl->hw_frames_ctx->data; + -+ if (in_frames_ctx->sw_format != AV_PIX_FMT_P010 && in_frames_ctx->sw_format != AV_PIX_FMT_NV12) { -+ av_log(ctx, AV_LOG_ERROR, "Unsupported input format: %s (NV12 or P010 required)\n", ++ if (in_frames_ctx->sw_format != AV_PIX_FMT_P010 && in_frames_ctx->sw_format != AV_PIX_FMT_NV12 && ++ in_frames_ctx->sw_format != AV_PIX_FMT_P210 && in_frames_ctx->sw_format != AV_PIX_FMT_NV16) { ++ av_log(ctx, AV_LOG_ERROR, "Unsupported input format: %s (NV12, P010, NV16 or P210 required)\n", + av_get_pix_fmt_name(in_frames_ctx->sw_format)); + return AVERROR(ENOSYS); + } @@ -380,18 +388,23 @@ index 0000000..3591127 + av_log(ctx, AV_LOG_ERROR, "sdr_white must not exceed hdr_peak\n"); + return AVERROR(EINVAL); + } -+ s->input_depth = in_frames_ctx->sw_format == AV_PIX_FMT_P010 ? 10 : 8; ++ s->input_depth = in_frames_ctx->sw_format == AV_PIX_FMT_P010 || in_frames_ctx->sw_format == AV_PIX_FMT_P210 ? 10 : 8; ++ s->input_422 = in_frames_ctx->sw_format == AV_PIX_FMT_NV16 || in_frames_ctx->sw_format == AV_PIX_FMT_P210; ++ /* Default output: depth from the transfer, chroma subsampling from the input. */ + s->output_depth = s->transfer_out == TONEMAP_TRANSFER_SDR ? 8 : 10; ++ s->output_422 = s->input_422; + if (s->output_format != AV_PIX_FMT_NONE) { -+ if ((s->output_format != AV_PIX_FMT_NV12 && s->output_format != AV_PIX_FMT_P010) || -+ (s->transfer_out != TONEMAP_TRANSFER_SDR && s->output_format != AV_PIX_FMT_P010)) { -+ av_log(ctx, AV_LOG_ERROR, "format must be nv12 or p010le; HDR requires p010le\n"); ++ const int ten_bit = s->output_format == AV_PIX_FMT_P010 || s->output_format == AV_PIX_FMT_P210; ++ const int eight_bit = s->output_format == AV_PIX_FMT_NV12 || s->output_format == AV_PIX_FMT_NV16; ++ if (!(ten_bit || eight_bit) || (s->transfer_out != TONEMAP_TRANSFER_SDR && !ten_bit)) { ++ av_log(ctx, AV_LOG_ERROR, "format must be nv12, p010le, nv16 or p210le; HDR requires 10 bits\n"); + return AVERROR(EINVAL); + } -+ s->output_depth = s->output_format == AV_PIX_FMT_P010 ? 10 : 8; ++ s->output_depth = ten_bit ? 10 : 8; ++ s->output_422 = s->output_format == AV_PIX_FMT_NV16 || s->output_format == AV_PIX_FMT_P210; + } -+ s->passthrough = s->transfer_in == s->transfer_out && -+ (s->output_format == AV_PIX_FMT_NONE || s->input_depth == s->output_depth); ++ s->passthrough = s->transfer_in == s->transfer_out && s->input_depth == s->output_depth && ++ s->input_422 == s->output_422; + + /* Resolve the per-operator default tone mapping parameter. */ + if (isnan(s->param)) { @@ -452,6 +465,8 @@ index 0000000..3591127 + int transfer_out = s->transfer_out; + int input_depth = s->input_depth; + int output_depth = s->output_depth; ++ int input_422 = s->input_422; ++ int output_422 = s->output_422; + float sdr_white = s->sdr_white; + float hdr_peak = s->hdr_peak; + int op = tonemap_op_map[s->tonemap]; @@ -465,7 +480,7 @@ index 0000000..3591127 + &s->own_frame->data[1], &s->own_frame->linesize[1], + &width, &height, + &transfer, &op, ¶m, &desat, -+ &transfer_out, &input_depth, &output_depth, &sdr_white, &hdr_peak ++ &transfer_out, &input_depth, &output_depth, &input_422, &output_422, &sdr_white, &hdr_peak + }; + + ret = CHECK_CU(cu->cuLaunchKernel(s->cu_func, @@ -556,7 +571,8 @@ index 0000000..3591127 + if (ret < 0) + goto fail; + s->current_transfer = ret; -+ if (s->passthrough || (s->current_transfer == s->transfer_out && s->input_depth == s->output_depth)) { ++ if (s->passthrough || (s->current_transfer == s->transfer_out && s->input_depth == s->output_depth && ++ s->input_422 == s->output_422)) { + // Identity frames leave with the resolved contract stamped: an assumed + // (untagged) SDR input must not stay untagged for the encoder VUI or a + // compositor colour check downstream. @@ -629,7 +645,7 @@ index 0000000..3591127 + { "hlg", "ARIB STD-B67 (Hybrid Log-Gamma)", 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_TRANSFER_HLG }, 0, 0, FLAGS, .unit = "transfer" }, + { "pq", "SMPTE ST 2084 (Perceptual Quantizer)", 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_TRANSFER_PQ }, 0, 0, FLAGS, .unit = "transfer" }, + { "sdr", "BT.709 gamut with BT.1886 display response", 0, AV_OPT_TYPE_CONST, { .i64 = TONEMAP_TRANSFER_SDR }, 0, 0, FLAGS, .unit = "transfer" }, -+ { "format", "Explicit output storage (nv12 or p010le)", OFFSET(output_format), AV_OPT_TYPE_PIXEL_FMT, { .i64 = AV_PIX_FMT_NONE }, -1, INT_MAX, FLAGS }, ++ { "format", "Explicit output storage (nv12, p010le, nv16 or p210le)", OFFSET(output_format), AV_OPT_TYPE_PIXEL_FMT, { .i64 = AV_PIX_FMT_NONE }, -1, INT_MAX, FLAGS }, + { "sdr_white", "SDR reference white in nits", OFFSET(sdr_white), AV_OPT_TYPE_DOUBLE, { .dbl = 203.0 }, 1, 10000, FLAGS }, + { "hdr_peak", "HLG display peak / HDR tone mapping peak in nits", OFFSET(hdr_peak), AV_OPT_TYPE_DOUBLE, { .dbl = 1000.0 }, 100, 10000, FLAGS }, + { "param", "tonemap parameter", OFFSET(param), AV_OPT_TYPE_DOUBLE, { .dbl = NAN }, DBL_MIN, DBL_MAX, FLAGS }, @@ -669,10 +685,10 @@ index 0000000..3591127 +}; diff --git a/libavfilter/vf_tonemap_cuda.cu b/libavfilter/vf_tonemap_cuda.cu new file mode 100644 -index 0000000..59c2206 +index 0000000..ccfc54d --- /dev/null +++ b/libavfilter/vf_tonemap_cuda.cu -@@ -0,0 +1,285 @@ +@@ -0,0 +1,304 @@ +/* + * This file is part of FFmpeg. + * @@ -907,8 +923,9 @@ index 0000000..59c2206 + row[x] = (unsigned char)fminf(fmaxf(rintf(code), 0.0f), 255.0f); +} + -+// A thread handles one 2x2 luma block and its shared chroma sample. -+// NV12/P010 have identical chroma geometry; all pitches are in bytes. ++// A thread handles one 2x2 luma block. 4:2:0 input shares one chroma sample ++// across the block, 4:2:2 input has one per luma row; the output side likewise ++// writes one sample (average of the block) or one per row. All pitches are bytes. +extern "C" __global__ void Tonemap_Cuda( + unsigned char *src_y, int src_y_linesize, + unsigned char *src_uv, int src_uv_linesize, @@ -916,7 +933,7 @@ index 0000000..59c2206 + unsigned char *dst_uv, int dst_uv_linesize, + int width, int height, + int transfer, int op, float param, float desat, -+ int transfer_out, int input_depth, int output_depth, ++ int transfer_out, int input_depth, int output_depth, int input_422, int output_422, + float sdr_white, float hdr_peak) +{ + int xi = blockIdx.x * blockDim.x + threadIdx.x; @@ -930,8 +947,12 @@ index 0000000..59c2206 + int x = 2 * xi; + int y = 2 * yi; + -+ float cb = read_code(src_uv, src_uv_linesize, x, yi, input_depth); -+ float cr = read_code(src_uv, src_uv_linesize, x + 1, yi, input_depth); ++ float cb[2], cr[2]; ++ for (int r = 0; r < 2; ++r) { ++ int crow = input_422 ? y + r : yi; ++ cb[r] = read_code(src_uv, src_uv_linesize, x, crow, input_depth); ++ cr[r] = read_code(src_uv, src_uv_linesize, x + 1, crow, input_depth); ++ } + if (transfer == transfer_out) { + // Depth-only conversion preserves code values, including foot/headroom. + float scale = input_depth == 10 ? 0.25f : 1.0f; @@ -940,23 +961,37 @@ index 0000000..59c2206 + write_code(dst_y, dst_y_linesize, px, py, output_depth, + read_code(src_y, src_y_linesize, px, py, input_depth) * scale); + } -+ write_code(dst_uv, dst_uv_linesize, x, yi, output_depth, cb * scale); -+ write_code(dst_uv, dst_uv_linesize, x + 1, yi, output_depth, cr * scale); ++ if (output_422) { ++ for (int r = 0; r < 2; ++r) { ++ write_code(dst_uv, dst_uv_linesize, x, y + r, output_depth, cb[r] * scale); ++ write_code(dst_uv, dst_uv_linesize, x + 1, y + r, output_depth, cr[r] * scale); ++ } ++ } else { ++ write_code(dst_uv, dst_uv_linesize, x, yi, output_depth, 0.5f * (cb[0] + cb[1]) * scale); ++ write_code(dst_uv, dst_uv_linesize, x + 1, yi, output_depth, 0.5f * (cr[0] + cr[1]) * scale); ++ } + return; + } -+ float u = 0.0f, v = 0.0f; ++ float u[2] = {0.0f, 0.0f}, v[2] = {0.0f, 0.0f}; + for (int i = 0; i < 4; ++i) { -+ int px = x + (i & 1), py = y + (i >> 1); ++ int px = x + (i & 1), r = i >> 1, py = y + r; + float code = read_code(src_y, src_y_linesize, px, py, input_depth); -+ float3 c = rgb_to_yuv(convert_rgb(yuv_to_rgb(code, cb, cr, input_depth, transfer), ++ float3 c = rgb_to_yuv(convert_rgb(yuv_to_rgb(code, cb[r], cr[r], input_depth, transfer), + transfer, transfer_out, sdr_white, hdr_peak, op, param, desat), + transfer_out); + write_code(dst_y, dst_y_linesize, px, py, output_depth, 219.0f * c.x + 16.0f); -+ u += c.y; -+ v += c.z; ++ u[r] += c.y; ++ v[r] += c.z; ++ } ++ if (output_422) { ++ for (int r = 0; r < 2; ++r) { ++ write_code(dst_uv, dst_uv_linesize, x, y + r, output_depth, 224.0f * (u[r] * 0.5f) + 128.0f); ++ write_code(dst_uv, dst_uv_linesize, x + 1, y + r, output_depth, 224.0f * (v[r] * 0.5f) + 128.0f); ++ } ++ } else { ++ write_code(dst_uv, dst_uv_linesize, x, yi, output_depth, 224.0f * ((u[0] + u[1]) * 0.25f) + 128.0f); ++ write_code(dst_uv, dst_uv_linesize, x + 1, yi, output_depth, 224.0f * ((v[0] + v[1]) * 0.25f) + 128.0f); + } -+ write_code(dst_uv, dst_uv_linesize, x, yi, output_depth, 224.0f * (u * 0.25f) + 128.0f); -+ write_code(dst_uv, dst_uv_linesize, x + 1, yi, output_depth, 224.0f * (v * 0.25f) + 128.0f); +} -- 2.55.0 diff --git a/deps/ffmpeg/8/bases.env b/deps/ffmpeg/8/bases.env index 1004c3ca..0bd31c18 100644 --- a/deps/ffmpeg/8/bases.env +++ b/deps/ffmpeg/8/bases.env @@ -1,5 +1,5 @@ n80_commit=140fd653aed8cad774f991ba083e2d01e86420c7 -n80_tree=333490e1b1d3ec9af6aa890c5f58565b66ae61fd +n80_tree=7deb4ab6997451f44e4876137881b631fdbcfa51 n81_commit=9047fa1b084f76b1b4d065af2d743df1b40dfb56 -n81_tree=21d9c70f3d1d136f429e9a999de53ce721f1fb5a +n81_tree=90acd8bee46eee9eccedf37f148e434f261ef008 patch_count=9 diff --git a/deps/ffmpeg/README.md b/deps/ffmpeg/README.md index 4bf75c08..27e47a86 100644 --- a/deps/ffmpeg/README.md +++ b/deps/ffmpeg/README.md @@ -41,8 +41,9 @@ and the filter changes) is base-independent. or device implementation is supplied, and NDI stays disabled in demo builds. 8. **10-bit CUDA transitions** — YUV420P10/422P10/444P10, P010 and P210 (plus 8-bit 4:2:2/4:4:4) in a word-sample `transition_cuda` kernel. -9. **`tonemap_cuda`** — SDR BT.709 / HLG / PQ conversion on CUDA NV12/P010 - frames in both directions: display-light conversion with configurable SDR +9. **`tonemap_cuda`** — SDR BT.709 / HLG / PQ conversion on CUDA semiplanar + frames (NV12/P010 4:2:0, NV16/P210 4:2:2, resampled in the same pass) in + both directions: display-light conversion with configurable SDR white and HDR peak, HDR-to-SDR operators with a knee parameter, automatic per-frame contract resolution (untagged frames are BT.709 SDR), zero-copy identity frames and fixed NV12/P010 output storage. diff --git a/tests/test_mixer_color.py b/tests/test_mixer_color.py index 57cdfee8..789a4137 100644 --- a/tests/test_mixer_color.py +++ b/tests/test_mixer_color.py @@ -49,11 +49,14 @@ def test_unsupported_contracts_fail_instead_of_retagging(tags): Color.parse(tags) -def test_chroma_conversion_preserves_depth_before_tone_mapping(): - graph = conversion_graph("sdr", "nv12", source_format="p210le") - assert graph.startswith("scale_cuda=format=p010le,tonemap_cuda=") - assert "hwdownload" not in graph and "hwupload" not in graph - assert conversion_graph("hlg", "p210le").endswith(",scale_cuda=format=p210le") +def test_tonemap_converts_storage_in_the_same_pass(): + # 4:2:2 stays inside tonemap_cuda: P210 in, NV12/P210 out, no scale_cuda round trip. + assert conversion_graph("sdr", "nv12", source_format="p210le") == \ + "tonemap_cuda=transfer_in=auto:transfer_out=sdr:format=nv12:tonemap=clip:sdr_white=203:hdr_peak=1000:desat=0" + assert conversion_graph("hlg", "p210le").endswith(":format=p210le:tonemap=clip:sdr_white=203:hdr_peak=1000:desat=0") + assert "scale_cuda" not in conversion_graph("hlg", "p210le", source="sdr", source_format="nv12") + # Planar CUDA storage is re-laid out once before the tone mapper. + assert conversion_graph("sdr", "nv12", source_format="yuv422p10le").startswith("scale_cuda=format=p210le,tonemap_cuda=") with pytest.raises(ValueError, match="10-bit"): conversion_graph("hlg", "nv12") with pytest.raises(ValueError, match="source pixel format"):